diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 5bf84bf..e5da08f 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -14,7 +14,16 @@ "name": "Fast" }, "skills": [ - "./skills/skill-optimizer" + "./skills/investigate-functionality", + "./skills/investigate-submissions", + "./skills/design-tests", + "./skills/write-tests", + "./skills/validate-tests", + "./skills/run-bench", + "./skills/analyze", + "./skills/improve", + "./skills/validate", + "./skills/autopilot" ] } ] diff --git a/.codex/INSTALL.md b/.codex/INSTALL.md index 489e8ee..5cc99a2 100644 --- a/.codex/INSTALL.md +++ b/.codex/INSTALL.md @@ -22,10 +22,20 @@ codex plugin marketplace add fastxyz/skill-optimizer --ref main ## Skill-Only Install -Install the canonical skill with the open skills CLI: +Install the 9 chain skills with the open skills CLI: ```bash -npx skills add fastxyz/skill-optimizer --skill skill-optimizer -a codex -y +npx skills add fastxyz/skill-optimizer \ + --skill investigate-functionality \ + --skill investigate-submissions \ + --skill design-tests \ + --skill write-tests \ + --skill validate-tests \ + --skill run-bench \ + --skill analyze \ + --skill improve \ + --skill validate \ + -a codex -y ``` -Restart Codex if the skill does not appear immediately. +Restart Codex if the skills do not appear immediately. diff --git a/.cursor/INSTALL.md b/.cursor/INSTALL.md index c6d8de6..be3cbc0 100644 --- a/.cursor/INSTALL.md +++ b/.cursor/INSTALL.md @@ -2,14 +2,24 @@ ## Skill install -Install the skill into Cursor's project or global skill directory through the open skills CLI: +Install the 9 chain skills into Cursor's project or global skill directory through the open skills CLI: ```bash -npx skills add fastxyz/skill-optimizer --skill skill-optimizer -a cursor -y +npx skills add fastxyz/skill-optimizer \ + --skill investigate-functionality \ + --skill investigate-submissions \ + --skill design-tests \ + --skill write-tests \ + --skill validate-tests \ + --skill run-bench \ + --skill analyze \ + --skill improve \ + --skill validate \ + -a cursor -y ``` Cursor can also import remote skills from GitHub in Settings -> Rules -> Project Rules -> Add Rule -> Remote Rule (Github). ## Plugin metadata -This repository includes `.cursor-plugin/plugin.json` for Cursor-compatible plugin metadata. The canonical skill remains `skills/skill-optimizer/SKILL.md`. +This repository includes `.cursor-plugin/plugin.json` for Cursor-compatible plugin metadata. The skills live at `skills//`; the plugin manifest exposes all 9 as a chain that triggers based on the user's intent (investigate, design tests, run bench, analyze, improve, validate). diff --git a/.gitignore b/.gitignore index 955a362..087e5ce 100644 --- a/.gitignore +++ b/.gitignore @@ -55,6 +55,7 @@ temp/ docs/superpowers/ docs/plans/ docs/specs/ +.superpowers/ # Skill-optimizer generated artifacts .skill-optimizer/ diff --git a/.opencode/INSTALL.md b/.opencode/INSTALL.md index c93409a..61e3996 100644 --- a/.opencode/INSTALL.md +++ b/.opencode/INSTALL.md @@ -8,9 +8,9 @@ Add the plugin to `opencode.json` at user or project scope: } ``` -Restart OpenCode. The plugin registers the repository `skills/` directory so the native `skill` tool can load `skill-optimizer`. +Restart OpenCode. The plugin registers the repository `skills/` directory so the native `skill` tool can discover the 9 chain skills (`investigate-functionality`, `investigate-submissions`, `design-tests`, `write-tests`, `validate-tests`, `run-bench`, `analyze`, `improve`, `validate`). -Verify with the skill tool by listing skills or loading `skill-optimizer`. +Verify by listing skills with the skill tool; each chain skill triggers on its own description (investigate, design tests, run bench, analyze, improve, validate). To pin a version, append a tag or commit ref: diff --git a/AGENTS.md b/AGENTS.md index 0d09351..8c2d832 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -20,7 +20,9 @@ npx tsx src/cli.ts run-suite --help - `src/cli.ts`: public CLI entrypoint - `src/workbench/`: workbench case loading, suite loading, Docker runner, Pi agent, graders, and traces - `docker/workbench-runner.Dockerfile`: generic non-root container image for setup, agent, grade, and cleanup phases -- `skills/skill-optimizer/SKILL.md`: canonical distributable Agent Skill +- `skills//`: the v1.4 chain of 9 user-invocable skills (investigate-functionality, investigate-submissions, design-tests, write-tests, validate-tests, run-bench, analyze, improve, validate) +- `skills/shared/`: workflow overview, iteration protocol, subagent-dispatch rules, frontmatter discipline, workbench reference; loaded on-demand by chain skills +- `skills//agents/`: prompt templates dispatched by chain skills via the Agent tool - `.claude-plugin/`, `.codex-plugin/`, `.cursor-plugin/`, `.opencode/`: cross-agent plugin manifests and install support - `.agents/plugins/marketplace.json`: Codex repo marketplace entry for the root plugin - `gemini-extension.json`, `GEMINI.md`: Gemini extension metadata and context file @@ -47,7 +49,7 @@ Keep the README installation section aligned with packaged plugin metadata: - Cases use `graders: [{ name, command }]`; legacy `check:` and `artifacts:` are invalid. - Graders are the acceptance contract; evaluate outputs from `/work`, generated artifacts, `answer.json`, `trace.jsonl`, and result state. - The agent phase sees only `/work`, not `/case` or `/results`. -- Keep plugin metadata pointed at the canonical `skills/skill-optimizer/SKILL.md`; do not create divergent skill copies. +- Keep plugin metadata pointed at every chain skill under `skills//`; do not create divergent skill copies. - Codex plugin metadata lives in `.codex-plugin/plugin.json`; the repo marketplace lives in `.agents/plugins/marketplace.json` and points at `./`. - Provider install docs should link to the same canonical skill/plugin metadata, not separate skill copies. - Do not commit `.skill-eval/`, `.results/`, `.env`, or credentials. diff --git a/CLAUDE.md b/CLAUDE.md index e91b35e..ac5227f 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -4,7 +4,7 @@ `skill-optimizer` is a Docker workbench for running and grading agent skill eval cases. The current public CLI centers on `run-case` and `run-suite`. -The workbench gives an agent an isolated Docker `/work` directory, captures traces, and grades deterministic local outcomes from files, command logs, generated artifacts, or other workspace state. +The workbench drives a host-side ACP (Agent Client Protocol) client against per-trial Docker containers that run one of five agent CLIs (claude-agent-acp, codex-acp, gemini, opencode, pi-acp). Each trial gets an isolated `/work` directory; raw ACP messages are captured to `trace.jsonl`; graders evaluate deterministic local outcomes from files, command logs, generated artifacts, or other workspace state. ## Key Commands @@ -20,10 +20,15 @@ npx tsx src/cli.ts run-suite --help ## Important Files - `src/cli.ts`: public CLI entrypoint -- `src/workbench/`: workbench case loading, suite loading, Docker runner, Pi agent, graders, and traces -- `docker/workbench-runner.Dockerfile`: generic non-root container image for setup, agent, grade, and cleanup phases -- `skills/skill-optimizer/SKILL.md`: canonical distributable Agent Skill -- `skills/skill-optimizer/references/workbench.md`: detailed workbench schema and usage reference +- `src/workbench/`: case loading, suite loading, host-side Docker runner, graders +- `src/workbench/acp/`: ACP transport, client wrapper, auth, skill deployment, MCP config writers, trace recorder +- `src/workbench/agents/`: 5-agent registry (claude-agent-acp, codex-acp, gemini, opencode, pi-acp) + install snippets +- `src/workbench/parse-trace.ts`: helpers for reading the raw ACP `trace.jsonl` (`iterMessages`, `iterToolCalls`, `computeMetrics`, etc.) +- `docker/skill-optimizer-agent.Dockerfile`: container image with all 5 agent CLIs pre-baked +- `docker/pi-acp-launcher.sh`: wrapper that bridges `SKILL_OPT_PROVIDER_*` env vars into pi-acp's `OPENROUTER_API_KEY` +- `skills//`: the v1.4 chain of 9 user-invocable skills (investigate-functionality, investigate-submissions, design-tests, write-tests, validate-tests, run-bench, analyze, improve, validate) +- `skills/shared/`: workflow overview, iteration protocol, subagent-dispatch rules, frontmatter discipline, and the workbench schema reference; loaded on-demand by chain skills +- `skills//agents/`: prompt templates dispatched by chain skills via the Agent tool - `.claude-plugin/`, `.codex-plugin/`, `.cursor-plugin/`, `.opencode/`: cross-agent plugin manifests and install support - `.agents/plugins/marketplace.json`: Codex repo marketplace entry for the root plugin - `gemini-extension.json`, `GEMINI.md`: Gemini extension metadata and context file @@ -45,12 +50,14 @@ Keep the README installation section aligned with packaged plugin metadata: ## Invariants - Keep evaluation static: extraction and matching are allowed; do not execute model-produced code outside the Docker workbench as part of evaluation. -- `run-suite` uses models from `suite.yml`; do not add a `run-suite --models` override. -- Keep OpenRouter model refs as `openrouter/...`; real model runs require `OPENROUTER_API_KEY`. +- Every `case.yml` must declare `agent:` (one of `claude-agent-acp`, `codex-acp`, `gemini`, `opencode`, `pi-acp`); the loader fails loud on missing/unknown agents. +- `suite.yml` declares its agent+model matrix via `runs: [{ agent, model }]`; legacy `models:` is rejected. `run-suite` uses what `suite.yml` declares; do not add a `--models` override. +- Only `pi-acp` requires `openrouter/...` model refs; native ACP agents pass their own model strings through unchanged. - Cases use `graders: [{ name, command }]`; legacy `check:` and `artifacts:` are invalid. - Graders are the acceptance contract; evaluate outputs from `/work`, generated artifacts, `answer.json`, `trace.jsonl`, and result state. - The agent phase sees only `/work`, not `/case` or `/results`. -- Keep plugin metadata pointed at the canonical `skills/skill-optimizer/SKILL.md`; do not create divergent skill copies. +- `trace.jsonl` is raw ACP wire format (JSON-RPC envelopes); use `parse-trace.ts` helpers rather than parsing the file by hand. See [`skills/shared/acp-trace-format.md`](skills/shared/acp-trace-format.md). +- Keep plugin metadata pointed at every chain skill under `skills//`; do not create divergent skill copies. - Codex plugin metadata lives in `.codex-plugin/plugin.json`; the repo marketplace lives in `.agents/plugins/marketplace.json` and points at `./`. - Provider install docs should link to the same canonical skill/plugin metadata, not separate skill copies. - Do not commit `.skill-eval/`, `.results/`, `.env`, or credentials. @@ -59,6 +66,6 @@ Keep the README installation section aligned with packaged plugin metadata: - Run `npm run typecheck` after TypeScript changes. - Run `npm test` before finishing behavior changes. -- For Docker runner or image changes, also run `docker build -t skill-optimizer-workbench:local -f docker/workbench-runner.Dockerfile .`. +- For Docker runner or image changes, also run `docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile .`. - For CLI/docs changes, verify `npx tsx src/cli.ts --help` if touched docs mention CLI behavior. - For plugin/package metadata changes, run `npx tsx tests/smoke-skill-distribution.ts` and verify `npm pack --dry-run --json` includes required plugin files without result/cache directories. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 870e01f..94a5e26 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,6 +1,6 @@ # Contributing to skill-optimizer -Thanks for contributing! This project is a small, opinionated Docker workbench for evaluating agent skills. Changes should preserve deterministic grading, isolated agent workspaces, and the canonical `skills/skill-optimizer/SKILL.md` distribution path. +Thanks for contributing! This project is a small, opinionated Docker workbench for evaluating agent skills, plus a 9-step chain of Agent Skills (`skills//`) that orchestrates investigation, test design, bench runs, analysis, and improvement of a target skill, plus an `autopilot` driver that walks the chain end-to-end. Changes should preserve deterministic grading, isolated agent workspaces, and the chain skills' distribution paths. ## Installing The Skill @@ -22,10 +22,13 @@ All three commands must pass before opening a PR when code changes are involved. ## Project layout - `src/cli.ts` — public CLI entry point for `run-case` and `run-suite`. -- `src/workbench/` — case/suite loading, Docker runner, Pi agent wiring, graders, traces, metrics, MCP support, and trial aggregation. -- `docker/workbench-runner.Dockerfile` — non-root container image for setup, agent, grade, and cleanup phases. -- `skills/skill-optimizer/SKILL.md` — canonical distributable Agent Skill. -- `skills/skill-optimizer/references/workbench.md` — detailed workbench schema and authoring reference. +- `src/workbench/` — case/suite loading, host-side Docker runner, graders, metrics, MCP support, trial aggregation. +- `src/workbench/acp/` — ACP transport, client wrapper, auth resolution, skill deployment, per-agent MCP config writer, trace recorder. +- `src/workbench/agents/` — 5-agent registry + Dockerfile install snippet generator. +- `docker/skill-optimizer-agent.Dockerfile` — container image with all 5 agent CLIs pre-baked. +- `skills//` — the 9-step chain of user-invocable Agent Skills (investigate-functionality, investigate-submissions, design-tests, write-tests, validate-tests, run-bench, analyze, improve, validate) plus the `autopilot` driver. +- `skills/shared/` — workflow overview, iteration protocol, subagent-dispatch rules, frontmatter discipline, workbench schema reference; loaded on-demand by chain skills. +- `skills//agents/` — prompt templates dispatched by chain skills via the Agent tool. - `examples/workbench/` — packaged example suites. - `.claude-plugin/`, `.codex-plugin/`, `.cursor-plugin/`, `.opencode/`, `.agents/plugins/marketplace.json`, `gemini-extension.json`, `GEMINI.md` — cross-agent plugin and extension metadata. - `tests/` — hand-rolled smoke tests (`tsx tests/smoke-*.ts`). @@ -42,23 +45,29 @@ All three commands must pass before opening a PR when code changes are involved. ## Workbench invariants - Keep evaluation static: extraction and matching are allowed; do not execute model-produced code outside the Docker workbench as part of evaluation. -- Use only `openrouter/...` model refs; real model runs require `OPENROUTER_API_KEY`. -- `run-suite` uses models from `suite.yml`; do not add a `run-suite --models` override. +- Every `case.yml` must declare `agent:` (one of `claude-agent-acp`, `codex-acp`, `gemini`, `opencode`, `pi-acp`); the loader fails loud on missing/unknown. +- `suite.yml` declares its `runs: [{ agent, model }]` matrix; legacy `models:` is rejected. `run-suite` uses what `suite.yml` declares; do not add a `--models` override. +- Only `pi-acp` requires `openrouter/...` model refs; native ACP agents pass their own model strings through unchanged. - Cases use `graders: [{ name, command }]`; legacy `check:` and `artifacts:` are invalid. - The agent phase sees only `/work`, not `/case`, `/results`, graders, hidden answers, or hidden metadata. -- Keep plugin metadata pointed at the canonical `skills/skill-optimizer/SKILL.md`; do not create divergent skill copies. +- `trace.jsonl` is raw ACP wire format (JSON-RPC envelopes); use `src/workbench/parse-trace.ts` helpers (`iterMessages`, `iterToolCalls`, `computeMetrics`) rather than parsing the file by hand. +- Keep plugin metadata pointed at every chain skill under `skills//`; do not create divergent skill copies. ## Testing guidance - Run `npm run typecheck` after TypeScript changes. - Run `npm test` before finishing behavior changes. -- For Docker runner or image changes, also run `docker build -t skill-optimizer-workbench:local -f docker/workbench-runner.Dockerfile .`. +- For Docker runner or image changes, also run `docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile .`. - For CLI/docs changes, verify `npx tsx src/cli.ts --help` if touched docs mention CLI behavior. - For plugin/package metadata changes, run `npx tsx tests/smoke-skill-distribution.ts` and verify `npm pack --dry-run --json` includes required plugin files without result/cache directories. +## Agent capabilities and limitations (v1) + +MCP server support is available for `claude-agent-acp`, `codex-acp`, `gemini`, and `opencode` agents. `pi-acp` MCP support is deferred to v1.1 pending documentation of pi-acp's native MCP config path. + ## Adding workbench capabilities -Keep new capabilities small and deterministic. Add validation in the relevant loader, tests in `tests/smoke-workbench-*.ts`, and docs in `skills/skill-optimizer/references/workbench.md` or `docs/workbench.md` when users need to author new YAML fields or understand new runtime behavior. +Keep new capabilities small and deterministic. Add validation in the relevant loader, tests in `tests/smoke-workbench-*.ts`, and docs in `skills/shared/workbench.md` or `docs/workbench.md` when users need to author new YAML fields or understand new runtime behavior. ## Commit style diff --git a/GEMINI.md b/GEMINI.md index 8f6d532..7b5b5ac 100644 --- a/GEMINI.md +++ b/GEMINI.md @@ -1,5 +1,5 @@ @./AGENTS.md @./README.md @./CONTRIBUTING.md -@./skills/skill-optimizer/SKILL.md -@./skills/skill-optimizer/references/workbench.md +@./skills/shared/workflow.md +@./skills/shared/workbench.md diff --git a/README.md b/README.md index 4dc6a6b..6fce926 100644 --- a/README.md +++ b/README.md @@ -1,15 +1,15 @@ # skill-optimizer -Docker workbench and Agent Skill for running deterministic evals against agent skills. +Docker workbench and Agent Skills for running deterministic evals against agent skills. Use this repo in two ways: -- Install the `skill-optimizer` skill/plugin into your agent so it can author and debug eval suites. +- Install the `skill-optimizer` plugin into your agent so it can investigate a target skill, design and run an eval suite, analyze failures, and propose improvements. The plugin bundles a 9-step chain of Agent Skills under `skills//`, plus an `autopilot` driver that runs the chain end-to-end. - Run the local CLI to execute cases and suites in Docker against OpenRouter models. ## Installation -Installation differs by agent. The canonical skill is `skills/skill-optimizer/SKILL.md`; every plugin manifest points at that same file. +Installation differs by agent. Every plugin manifest exposes all 9 chain skills from `skills//` plus the `autopilot` driver; once installed, each skill triggers on its own description (e.g. "investigate this skill", "design tests for this skill", "run the bench", "analyze the results", "autopilot this skill"). ### Claude Code @@ -53,13 +53,24 @@ codex plugin marketplace add fastxyz/skill-optimizer ### Cursor -Install the skill with the open skills CLI: +Install the chain skills with the open skills CLI (pass each skill name explicitly): ```bash -npx skills add fastxyz/skill-optimizer --skill skill-optimizer -a cursor -y -``` - -Cursor can also import the skill from GitHub via Settings -> Rules -> Project Rules -> Add Rule -> Remote Rule (Github). The Cursor plugin metadata lives at `.cursor-plugin/plugin.json`. +npx skills add fastxyz/skill-optimizer \ + --skill investigate-functionality \ + --skill investigate-submissions \ + --skill design-tests \ + --skill write-tests \ + --skill validate-tests \ + --skill run-bench \ + --skill analyze \ + --skill improve \ + --skill validate \ + --skill autopilot \ + -a cursor -y +``` + +Cursor can also import individual skills from GitHub via Settings -> Rules -> Project Rules -> Add Rule -> Remote Rule (Github). The Cursor plugin metadata lives at `.cursor-plugin/plugin.json`. ### OpenCode @@ -95,10 +106,21 @@ gemini extensions update skill-optimizer ### Skill-Only Install -If you only want the skill files without plugin metadata, use the open skills CLI: +If you only want the skill files without plugin metadata, use the open skills CLI. Pass each chain skill explicitly: ```bash -npx skills add fastxyz/skill-optimizer --skill skill-optimizer -a claude-code -a opencode -a codex -a cursor -y +npx skills add fastxyz/skill-optimizer \ + --skill investigate-functionality \ + --skill investigate-submissions \ + --skill design-tests \ + --skill write-tests \ + --skill validate-tests \ + --skill run-bench \ + --skill analyze \ + --skill improve \ + --skill validate \ + --skill autopilot \ + -a claude-code -a opencode -a codex -a cursor -y ``` ## Local CLI Setup @@ -107,20 +129,30 @@ Requirements: - Node.js 20+ - Docker -- `OPENROUTER_API_KEY` for real model runs + +The workbench drives one of five agent CLIs over ACP (Agent Client Protocol). Each agent has its own auth options: + +| Agent | Auth options | +|---|---| +| `claude-agent-acp` | `claude login` (subscription) or `ANTHROPIC_API_KEY` | +| `codex-acp` | `codex login` (subscription) or `OPENAI_API_KEY` | +| `gemini` | `gemini auth login` (subscription) or `GOOGLE_API_KEY` | +| `opencode` | `OPENAI_API_KEY` (or provider-specific key) | +| `pi-acp` | `OPENROUTER_API_KEY` | Install and build: ```bash npm install npm run build +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . ``` -Only `openrouter/...` model refs are supported. +Only `pi-acp` requires `openrouter/...` model refs. Native ACP agents (`claude-agent-acp`, `codex-acp`, `gemini`, `opencode`) take their own model strings (e.g., `claude-haiku-4-5-20251001`, `gpt-5-mini`). ## Quick Start -Run the suite against the models listed in `suite.yml`: +Run the suite against the agent+model pairs listed in `suite.yml`'s `runs:`: ```bash npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials 1 @@ -176,7 +208,7 @@ npx tsx src/cli.ts --help For Docker runner or image changes: ```bash -docker build -t skill-optimizer-workbench:local -f docker/workbench-runner.Dockerfile . +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . ``` Do not commit `.skill-eval/`, `.results/`, `.env`, or credentials. diff --git a/docker/pi-acp-launcher.sh b/docker/pi-acp-launcher.sh new file mode 100755 index 0000000..f168bf0 --- /dev/null +++ b/docker/pi-acp-launcher.sh @@ -0,0 +1,11 @@ +#!/bin/sh +# Bridges SKILL_OPT_PROVIDER_* env vars to pi-acp's expected config. +# pi-acp reads its model/provider from a runtime config; we set +# OPENROUTER_API_KEY here if SKILL_OPT_PROVIDER_API_KEY is set. +set -e + +if [ -n "$SKILL_OPT_PROVIDER_API_KEY" ] && [ -z "$OPENROUTER_API_KEY" ]; then + export OPENROUTER_API_KEY="$SKILL_OPT_PROVIDER_API_KEY" +fi + +exec pi-acp "$@" diff --git a/docker/skill-optimizer-agent.Dockerfile b/docker/skill-optimizer-agent.Dockerfile new file mode 100644 index 0000000..0786ee3 --- /dev/null +++ b/docker/skill-optimizer-agent.Dockerfile @@ -0,0 +1,60 @@ +FROM node:22-bookworm + +ENV PATH="/opt/skill-opt/bin:/app/node_modules/.bin:/work/.venv/bin:${PATH}" \ + PIP_REQUIRE_VIRTUALENV=1 + +WORKDIR /app + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + bash \ + ca-certificates \ + coreutils \ + curl \ + file \ + findutils \ + gawk \ + git \ + grep \ + jq \ + less \ + python-is-python3 \ + python3 \ + python3-pip \ + python3-venv \ + ripgrep \ + sed \ + unzip \ + wget \ + zip \ + && rm -rf /var/lib/apt/lists/* + +# --- Agent CLI install layer (cached as one layer for build speed) --- +# Each agent's installCmd is also kept in src/workbench/agents/registry.ts +# (single source of truth). If you change one here, change it there too. +RUN npm install -g \ + @zed-industries/claude-agent-acp@latest \ + @zed-industries/codex-acp@latest \ + @google/gemini-cli@latest \ + opencode-ai@latest \ + @mariozechner/pi-coding-agent@latest \ + pi-acp@latest + +# --- pi-acp launcher wrapper --- +COPY docker/pi-acp-launcher.sh /opt/skill-opt/bin/pi-acp-launcher +RUN chmod +x /opt/skill-opt/bin/pi-acp-launcher + +# --- Workbench container-runner (setup + grade modes only) --- +COPY package.json package-lock.json tsconfig.json ./ +COPY src ./src +COPY docs ./docs + +RUN npm ci \ + && npm run build \ + && useradd -m -u 10001 agent +USER agent + +# Container-runner is now used only for --setup and --grade modes. +# Agent dispatch happens host-side via ACP; the agent CLIs above are +# invoked directly by docker exec. +ENTRYPOINT ["node", "/app/dist/workbench/container-runner.js"] diff --git a/docker/workbench-runner.Dockerfile b/docker/workbench-runner.Dockerfile deleted file mode 100644 index 73591f7..0000000 --- a/docker/workbench-runner.Dockerfile +++ /dev/null @@ -1,42 +0,0 @@ -FROM node:22-bookworm - -ENV PATH="/app/node_modules/.bin:/work/.venv/bin:${PATH}" \ - PIP_REQUIRE_VIRTUALENV=1 - -WORKDIR /app - -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ - bash \ - ca-certificates \ - coreutils \ - curl \ - file \ - findutils \ - gawk \ - git \ - grep \ - jq \ - less \ - python-is-python3 \ - python3 \ - python3-pip \ - python3-venv \ - ripgrep \ - sed \ - unzip \ - wget \ - zip \ - && rm -rf /var/lib/apt/lists/* - -COPY package.json package-lock.json tsconfig.json ./ -COPY src ./src -COPY scripts ./scripts -COPY docs ./docs - -RUN npm ci \ - && npm run build \ - && useradd -m -u 10001 agent -USER agent - -ENTRYPOINT ["node", "/app/dist/workbench/container-runner.js"] diff --git a/docs/README.codex.md b/docs/README.codex.md index 6401f37..e64f590 100644 --- a/docs/README.codex.md +++ b/docs/README.codex.md @@ -22,10 +22,20 @@ codex plugin marketplace add fastxyz/skill-optimizer --ref main ## Skill-Only Install -Install only the skill files with the open skills CLI: +Install only the chain skill files with the open skills CLI (pass each skill name explicitly): ```bash -npx skills add fastxyz/skill-optimizer --skill skill-optimizer -a codex -y +npx skills add fastxyz/skill-optimizer \ + --skill investigate-functionality \ + --skill investigate-submissions \ + --skill design-tests \ + --skill write-tests \ + --skill validate-tests \ + --skill run-bench \ + --skill analyze \ + --skill improve \ + --skill validate \ + -a codex -y ``` -Restart Codex if the skill does not appear immediately. The canonical skill path is `skills/skill-optimizer/SKILL.md`. +Restart Codex if the skills do not appear immediately. The chain skills live at `skills//SKILL.md`. diff --git a/docs/README.opencode.md b/docs/README.opencode.md index 93ff893..515b285 100644 --- a/docs/README.opencode.md +++ b/docs/README.opencode.md @@ -12,11 +12,11 @@ Add the plugin to `opencode.json` at user or project scope: } ``` -Restart OpenCode. The plugin registers this repository's `skills/` directory so OpenCode can discover `skill-optimizer` without symlinks. +Restart OpenCode. The plugin registers this repository's `skills/` directory so OpenCode can discover the 9 chain skills without symlinks. ## Verify -Use OpenCode's native `skill` tool to list skills or load `skill-optimizer`. +Use OpenCode's native `skill` tool to list skills. You should see all 9 chain skills (`investigate-functionality`, `investigate-submissions`, `design-tests`, `write-tests`, `validate-tests`, `run-bench`, `analyze`, `improve`, `validate`); each triggers on its own description. ## Updating @@ -32,4 +32,4 @@ OpenCode reinstalls git plugins when it starts. To pin a tag or commit, append a The plugin exposes `.opencode/plugins/skill-optimizer.js` and adds the repository `skills/` directory to `config.skills.paths`. -The canonical skill is `skills/skill-optimizer/SKILL.md`. +The chain skills live at `skills//SKILL.md` — 9 user-invocable Agent Skills that orchestrate investigation, test design, bench runs, analysis, and improvement of a target skill. diff --git a/docs/skill-writing-philosophy.md b/docs/skill-writing-philosophy.md new file mode 100644 index 0000000..eb1604d --- /dev/null +++ b/docs/skill-writing-philosophy.md @@ -0,0 +1,519 @@ +# Skill-writing philosophy + +> Read this before authoring or revising any SKILL.md in this project, +> and before designing the optimizer subagent's prompt — choosing the +> wrong philosophy for the skill type is itself a form of ducktape. + +The lean agent-facing distillation lives at +[`skills/shared/skill-design-philosophy.md`](../skills/shared/skill-design-philosophy.md). +That doc is what the chain's analyzer/optimizer/validator subagents +load each run — terse, no provenance, just the rules. **This** doc +is the research backing it: per-vendor findings, where vendors +agree or diverge, and the rationale for which principles got +promoted into the shipped distillation. + +## Background + +Four bodies of guidance on writing agent skills were surveyed: + +1. **Anthropic** — official Agent Skills docs ([best + practices](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/best-practices)), + the [anthropics/skills](https://github.com/anthropics/skills) + repo (skill-creator plugin), and the third-party + [superpowers writing-skills](https://github.com/obra/superpowers) + that frames skill authoring as TDD applied to process docs. +2. **OpenAI** — [Apps SDK tool design](https://developers.openai.com/apps-sdk/plan/tools), + [GPT-4.1 prompting guide](https://cookbook.openai.com/examples/gpt4-1_prompting_guide), + [Custom GPT instructions guide](https://help.openai.com/en/articles/9358033), + [function calling](https://platform.openai.com/docs/guides/function-calling). +3. **Google Gemini** — [system instructions](https://ai.google.dev/gemini-api/docs/system-instructions), + [function calling](https://ai.google.dev/gemini-api/docs/function-calling), + [Gemini CLI extension best practices](https://geminicli.com/docs/extensions/best-practices/), + [GEMINI.md context files](https://geminicli.com/docs/cli/gemini-md/). +4. **Community** — [SwirlAI analysis](https://www.newsletter.swirlai.com/p/agent-skills-progressive-disclosure), + [awesome-cursorrules](https://github.com/PatrickJS/awesome-cursorrules), + [awesome-agent-skills](https://github.com/VoltAgent/awesome-agent-skills), + leaked-system-prompt analyses, Cursor `.cursorrules` ecosystem. + +Full source URLs in References at the bottom. + +## Per-vendor findings + +### Anthropic + +The Claude ecosystem has the most-developed skill philosophy. Three +named bodies of guidance, mostly agreeing, diverging on tone and +testing rigor: + +- **Anthropic official** treats skills as reference docs for an + agent: progressive disclosure, concise prose, explain WHY rather + than stack MUSTs, build evals first. +- **skill-creator** is Anthropic's plugin for iterative skill + authoring — adds an eval-viewer + description-optimizer loop. Its + most distinctive contribution: be a little "pushy" in descriptions + because Claude tends to under-trigger ("Make sure to use this + skill whenever the user mentions dashboards…"). +- **superpowers writing-skills** is third-party and frames skill + authoring as TDD applied to documentation. Its strongest empirical + finding: workflow summaries in the description cause agents to + follow the description and skip the body. Switch to triggering + conditions only. + +#### Anthropic divergence axis: discipline-enforcing skills + +| Axis | Anthropic + skill-creator | superpowers writing-skills | +|---|---|---| +| Tone for must-follow rules | "Explain the WHY; MUSTs/NEVERs in all caps are a yellow flag" | "Authority framing, no exceptions, close every loophole" | +| What gets a skill | Workflows, techniques, references | Same, **plus** discipline-enforcing rules (TDD, verification) | +| Testing rigor | ≥3 evals (Anthropic) / full with-skill vs baseline pipeline (skill-creator) | Pressure scenarios with combined pressures (time + sunk cost + authority) | +| Description content | "Include both what + when" | "**Only** when, **never** what — the body becomes documentation Claude skips otherwise" | + +The divergence is not a contradiction. writing-skills explicitly +says: discipline-enforcing skills get authority framing + pressure +tests; reference / technique skills get application tests. The other +two sources mostly assume the workflow / technique case and don't +address discipline rules. + +### OpenAI + +OpenAI's framing is more literal and prescriptive than Anthropic's +"explain why" stance. Distinctive principles: + +- **Literal-instruction-following.** GPT-4.1 and later are + *"trained to follow instructions more closely and more literally + than its predecessors."* State behavior unambiguously; don't rely + on the model to infer intent. This *contradicts* Anthropic's + preference for terse-with-reasoning — OpenAI says be explicit + even when it feels redundant. +- **Positive instructions over prohibitions.** Custom GPT guide: + *"Prefer positive, concrete instructions ('Do X') over long lists + of prohibitions ('Don't do Y') when possible."* +- **Single-action granularity.** Apps SDK: *"keep each tool focused + on a single read or write action ('fetch_board', 'create_ticket'), + rather than a kitchen-sink endpoint."* Read and write should be + separate tools so confirmation flows work. +- **Tool description shape.** Apps SDK mandates one-or-two-sentence + descriptions that *"start with 'Use this when…' so the model + knows exactly when to pick the tool."* +- **Knowledge vs. instructions separation.** Custom GPT guide: + *"Use knowledge for reference material, not rules or behavior. + Put rules, tone, and workflow guidance in instructions."* +- **Structured prompt hierarchy.** GPT-4.1 guide recommends a + canonical order: Role → Instructions → Reasoning Steps → Output + Format → Examples → Context. For long contexts, put instructions + at both ends. +- **Three-line agentic harness.** *"Keep going until resolved", + "use tools instead of guessing", "plan before each call and + reflect after"* — produced ~20% SWE-bench gain. +- **Tool-count failure mode.** Up to ~100 tools is "in-distribution", + but the named failure mode is overlap: *"If multiple tools have + overlapping purposes or vague descriptions, models may call the + wrong one or hesitate to call any at all."* +- **Examples sit in their own section, not in tool descriptions.** + *"When provided sample phrases, models can use those quotes + verbatim and start to sound repetitive."* Instruct the model to + vary them. + +Anti-patterns OpenAI explicitly names: bribes/all-caps/tip threats +(*"not necessary"*), conflicting instructions (silently resolved in +favor of the one closer to the end), forced tool loops, conflated +read/write tools. + +### Google Gemini + +Gemini's framing emphasizes terseness, hierarchical context, and +hard numeric bounds. Distinctive principles: + +- **Terse default.** *"Gemini 3 models provide direct and efficient + answers. If you need more conversational or detailed response, + you must explicitly request it."* Don't assume verbose output — + instructions that don't ask for elaboration shouldn't expect it. +- **Critical instructions first; long-context question last.** + *"Place essential behavioral constraints and role definitions at + the beginning."* For long-context prompts, supply documents first + and put the actual instruction at the very end with a transition + phrase. *Divergence*: Anthropic emphasizes consistent positioning; + Google is more prescriptive about question-at-end for long-context. +- **Hierarchical, modular context.** Three tiers: global + (`~/.gemini/GEMINI.md`), workspace, just-in-time (file-adjacent). + Large files should be decomposed using `@file.md` imports. + Monolithic single-file context is an anti-pattern. +- **Hard 10-20 tool cap.** *"Providing too many can increase the + risk of selecting an incorrect or suboptimal tool. Aim to provide + only the relevant tools for the context or task, ideally keeping + the active set to a maximum of 10-20."* Numeric bound where + others are vague. +- **Constrain with schema, not prose.** *"Use 'enum' to list the + allowed values instead of just describing them in the + description."* Prefer `integer` over `number` when applicable. +- **Few-shot examples format-consistent.** Inconsistency across + examples actively degrades behavior. +- **Procedural + declarative hybrid endorsed.** Google's agentic + system-instruction template combines declarative reasoning steps + ("analyze constraints", "evaluate consequences") with procedural + numbered steps. *Divergence*: explicitly endorses the hybrid, + whereas Anthropic guidance has shifted more toward declarative- + only for skills. + +Anti-patterns Gemini explicitly names: *"overly persuasive or +flowery language"*, monolithic context with no decomposition, +buried critical constraints, enum-able parameters described only +in prose, granting broad permissions when narrow ones suffice. + +### Community + +Convergent patterns across multiple unrelated skill collections +(SwirlAI, awesome-agent-skills, awesome-cursorrules, Microsoft +`agent-skills`, Vercel/community packs, leaked-prompt analyses): + +- **Progressive disclosure as the core organizing principle.** + Three-tier loading: metadata (~60-100 tokens always loaded) → + SKILL.md body (loaded on trigger, <500 lines) → bundled + references/scripts (loaded only as needed). SwirlAI: *"Context + window is the agent's cognitive space. Overloading it degrades + performance, while keeping it focused lets the agent reason + sharply."* +- **Description = triggering contract.** anthropics/skills + explicitly recommends being *"pushy"* to combat under-triggering. +- **One-level-deep references.** Universally warned-against: + nested reference chains (SKILL.md → advanced.md → details.md). + Claude does partial reads with `head -100`, losing info in deep + trees. +- **Conditional activation.** Cursor `.mdc` rules converge on + `globs:` + `alwaysApply:` — *"conditional patterns outperform + blanket instructions because they reduce noise."* +- **Evals-before-prose.** Anthropic explicit: *"Create evaluations + BEFORE writing extensive documentation. This ensures your Skill + solves real problems rather than documenting imagined ones."* + +Divergent patterns (context-dependent): + +- **Strictness of constraint language.** Anthropic warns against + ALWAYS/NEVER caps, but the same doc later recommends *"using + stronger language like 'MUST filter' instead of 'always filter'"* + when Claude under-applies a rule. Resolution in practice: match + constraint strictness to task fragility (the "narrow bridge vs. + open field" framing). +- **Skill scope — narrow vs. multi-domain.** Microsoft / SwirlAI + warn against "kitchen-sink skills"; some community packs + deliberately bundle (rohitg00/awesome-claude-code-toolkit ships + 135 agents in one). Convergent compromise: split by domain + inside one skill (Pattern 2 in Anthropic docs — + `reference/finance.md`, `reference/sales.md`). + +Community-flagged anti-patterns: + +- Verbose descriptions that summarize workflow. +- Voodoo constants / bare try-blocks that re-throw. +- Overfitting to one observed failure. +- Time-sensitive content ("before August 2025") in the main body. +- Offering many options without a default. +- Inconsistent terminology ("field"/"box"/"element" mixed). + +## Cross-vendor consensus (load-bearing — agreed across all four) + +- **Progressive disclosure.** Anthropic + OpenAI (implicit) + + Gemini + community. +- **Description = triggering contract.** Anthropic + OpenAI ("Use + this when…") + Gemini ("extremely clear and specific") + + community. +- **Conciseness / context as scarce resource.** Anthropic + ("concise is key") + Gemini ("terse default") + community + ("don't dump exhaustive docs"). +- **Evaluation-driven content.** Anthropic + superpowers (TDD) + plus community. +- **Examples teach behavior, but don't overfit.** superpowers + + OpenAI ("models can use those quotes verbatim") + Gemini + ("few-shot must be format-consistent"). + +## Cross-vendor divergence (pick your stance) + +| Axis | Anthropic side | OpenAI side | Gemini side | +|---|---|---|---| +| Tone for rules | "Explain WHY; MUSTs are a yellow flag" | "Literal; be explicit even when redundant" | "Avoid flowery / persuasive language" | +| Output verbosity default | Implicit (depends on skill) | Implicit | **Terse** — must explicitly request elaboration | +| Tool/skill count bound | Implicit (works at any count) | ~100 tools in-distribution | **Hard 10-20 active cap** | +| Long-context positioning | Implicit | "Instructions at both ends" | "Critical first, question last" | +| Procedural vs declarative | Trending declarative-only | Mixed (agentic harness is procedural) | **Explicit hybrid endorsed** | +| Type constraints | Prose acceptable | Function-calling schema | **Enum/schema > prose** | + +Where vendors disagree, the v1.4 chain takes: + +- **Tone:** Anthropic — "explain WHY" by default, allow MUSTs for + fragile/discipline-enforcing rules (per superpowers). +- **Verbosity:** match the skill type; no universal stance. +- **Tool count:** Gemini's 10-20 bound applies to the chain itself + (we ship 9 + 1 autopilot). +- **Positioning:** Anthropic conventional + Gemini's "critical + first" for long-context dispatches. +- **Procedural/declarative:** Gemini's hybrid. +- **Type constraints:** N/A for prose skills, but enums-over-prose + applies to any spec.yaml fields the chain writes. + +## Synthesis: how we reached skill-design-philosophy.md + +The shipped lean doc has 12 principles, 14 anti-patterns, and an +8-row rubric. Each entry has at least two sources backing it. +Justification per principle: + +| # | Principle | Sources | +|---|---|---| +| 1 | Progressive disclosure | Anthropic + OpenAI (implicit) + Gemini + SwirlAI + Microsoft | +| 2 | Description = triggering contract | All four buckets | +| 3 | Conciseness as public stewardship | Anthropic + Gemini + community | +| 4 | Specificity matches task fragility | Anthropic explicit ("degrees of freedom"); others implicit | +| 5 | Explain WHY, not just WHAT | Anthropic + superpowers + Gemini (anti-flowery) | +| 6 | Procedural + declarative hybrid | Gemini explicit; others implicit | +| 7 | Evaluation-driven content | Anthropic + superpowers | +| 8 | One example beats many | superpowers + OpenAI + Gemini | +| 9 | Single-action granularity | OpenAI Apps SDK explicit | +| 10 | Knowledge vs instructions separation | OpenAI Custom GPT explicit | +| 11 | Executable scripts > generated code | Anthropic | +| 12 | Validation loops > one-shot | Anthropic | + +**Principles deliberately NOT included:** + +- OpenAI's "literal-instruction-following" — contradicts the + more-effective "explain WHY" framing for most skill types. We + treat it as a fragile-task fallback under principle 4, not a + universal rule. +- OpenAI's "structured prompt hierarchy" (Role → Instructions → + Reasoning Steps → …) — useful for system prompts, less so for + skill bodies. Not load-bearing for our agent's reasoning. +- Gemini's "long-context positioning" (critical first, question + last) — useful for long-context dispatches but the chain skills + are short enough that positioning doesn't matter much. +- The 10-20 tool cap was promoted into anti-pattern 12 (active + skill set >20), not into a top-level principle, because it's + a quantitative bound rather than a quality principle. + +**Anti-patterns:** each anti-pattern in the shipped doc was named +by at least two sources. The 14 include workflow-summary +descriptions (superpowers explicit), narrative bloat (Anthropic), +MUST/NEVER without rationale (Anthropic + Gemini), magic numbers +(community), persuasive language (Gemini), monolithic SKILL.md +(SwirlAI + Anthropic), etc. + +**Principled-vs-ducktape rubric:** derived from the convergent +test *"does the change pay its way in tokens or evals?"* Each +row is a specific ducktape pattern flagged by ≥2 sources, paired +with the principled alternative also flagged by ≥2 sources. + +## Skill type → philosophy mapping + +This is the single most important decision when authoring or +revising a skill. Diagnose first, write/revise second. + +| Skill type | Examples | Philosophy | Testing | +|---|---|---|---| +| **Workflow** | "do X, then Y, then dispatch Z" | Anthropic + skill-creator: lean prose, explain why, set freedom level per step | with-skill vs baseline subagents on realistic task prompts | +| **Technique** | "how to use library X correctly" | Anthropic: concise, one excellent example, edge-case notes | application + variation + gap-coverage scenarios | +| **Reference** | API docs, schema docs, vendor conventions | Anthropic: table-of-contents at top, keyword-rich, no narrative | retrieval scenarios — can the agent find + apply the right info? | +| **Pattern** | mental models, design heuristics | Anthropic: recognition examples + counter-examples + when-NOT-to-use | recognition scenarios — does the agent know when to apply? | +| **Discipline-enforcing** | "always do X before Y" rules, anti-ducktape constraints | writing-skills: authority framing, close every loophole, rationalization table, red-flags list, persuasion principles | pressure scenarios with multiple combined pressures (time + sunk cost + authority + exhaustion) | + +**Mixed-type skills are common.** Most skill-optimizer chain skills +are workflows with one or two embedded discipline rules (e.g., +"dispatch the subagent, do NOT do the research yourself"). The right +pattern: + +1. Write the workflow body in Anthropic / skill-creator style — + lean, explain why, set freedom per step. +2. For each embedded discipline rule, mark it explicitly (bold, + "Do NOT" framing at the action site) AND give a why-this-matters + paragraph nearby. +3. Reserve full writing-skills treatment (rationalization tables, + red-flags lists, no-exceptions language) for the small handful + of rules where compliance under pressure is load-bearing and + was verified via pressure scenarios. + +## The bootstrapping limit + +The skill-optimizer chain runs an empirical loop (eval → analyze → +improve) on target skills. But that loop cannot validate **itself** +— you can't use the chain to author its own SKILL.md files, because +the chain doesn't exist yet when those files are being written. +Thompson's "Reflections on Trusting Trust": validating a compiler +with itself is circular. + +Practical consequences: + +- **The chain SKILL.md files are authored from philosophy + best + judgment**, not from an eval loop. The "test" for these skills + is end-to-end runs on real targets and ongoing observation in + actual use. +- **The optimizer subagent should not be surprised by the absence + of eval data for the skill-optimizer's own skills.** If asked + to improve one of them, it should treat absence of empirical + data as a known limit, not a gap to fill speculatively. +- **Self-application is deferred** until the chain has earned + trust on external targets. Running skill-optimizer on + skill-optimizer is a future-state exercise; doing it pre-launch + is bootstrapping in a loop. + +## Description: triggers, not summaries + +writing-skills' strongest empirical finding: when a description +summarizes the skill's workflow, agents follow the summary instead +of reading the full skill body. The skill body becomes +documentation the agent skips. + +The example they cite: a description saying "code review between +tasks" caused agents to do ONE review even though the skill body +clearly specified TWO. Switching the description to pure trigger +language ("Use when executing implementation plans with independent +tasks in the current session") restored compliance. + +Practical guidance for skill-optimizer descriptions: + +- Start with "Use when …" and list the user-language symptoms that + should trigger this skill. +- Do NOT summarize the workflow, the dispatches, or the output + format in the description. Those go in the body. +- Use concrete trigger phrases, not abstract ones: "Use when the + user asks 'what does this skill do' or 'investigate this skill'" + beats "Use when investigating skills." +- Anthropic's "both what + when" guidance is fine for simple + workflow skills with no embedded discipline rules. For skills + with embedded discipline (most of the skill-optimizer chain + skills), trigger-only is safer. +- skill-creator's "be a little pushy" guidance applies: Claude + tends to under-trigger; include adjacent phrasings explicitly. + +## When the optimizer is revising someone else's skill + +This is the meta-payoff and the load-bearing reason this doc +exists. The Phase 7 optimizer subagent revises target skills based +on weaknesses the analyzer surfaced. To avoid ducktape, the +optimizer MUST: + +1. **Diagnose the target skill's type before proposing changes.** + A reference skill and a discipline-enforcing skill need + different fixes for the same observed failure. + +2. **Preserve the existing style unless the analyzer flagged the + style itself as the weakness.** If the original uses + explanatory prose, the fix uses explanatory prose. Don't + impose authority framing because MUSTs feel clearer to the + optimizer — that's ducktape. + +3. **Bias toward "Claude is smart" — pruning beats adding.** If + the skill restates what Claude already knows, removing the + restatement is often a more principled fix than adding new + rules. Token weight competes with conversation context once + the skill loads. + +4. **Reserve authority framing / rationalization tables / + no-exceptions language for skills where the analyzer + specifically documented a discipline failure under pressure.** + Adding MUSTs to patch a reference-skill bug is the canonical + ducktape pattern. + +5. **Test the proposed change against a scenario the analyzer + documented, not a new scenario invented by the optimizer.** + Optimizer-invented scenarios drift into solving imagined + problems instead of the real surfaced weakness. + +6. **Treat description changes as separate, conservative edits.** + The description determines triggering, not behavior. Changing + the description to "fix" a behavioral problem is misdirected. + Change the body for behavior; change the description only if + the analyzer flagged a trigger problem (over-triggering or + under-triggering). + +## When to use each philosophy: a quick decision tree + +```text +Question 1: Is this an absolute rule that an agent might try to + rationalize away under pressure? + └─ YES → discipline-enforcing skill → writing-skills philosophy + └─ NO → continue to Q2 + +Question 2: Is this a reference doc (API, schema, conventions) the + agent will scan to retrieve facts? + └─ YES → reference skill → Anthropic philosophy + TOC at top + └─ NO → continue to Q3 + +Question 3: Is this a sequence of steps the agent follows to + accomplish a task? + └─ YES → workflow skill → Anthropic + skill-creator philosophy + (lean prose, explain why, freedom level per step) + └─ NO → it's likely a pattern / mental model + → Anthropic philosophy (recognition + counter-examples) +``` + +If the skill is mixed (workflow with embedded discipline rules): +write the body in workflow style, mark the discipline rules +explicitly at their action sites, and reserve the heavy +writing-skills treatment (rationalization tables etc.) for the +specific rules whose compliance under pressure was actually +verified. + +## Open questions + +- **Cross-vendor portability.** The shipped philosophy doc + assumes Claude as the operator. Whether the synthesis holds for + Codex / Gemini operators (when the chain skills run on those + platforms via the cross-agent plugin metadata) is untested. + OpenAI's "literal-instruction-following" might mean Codex + operators handle our terse-with-rationale style worse than + Claude does. Future work: re-run a chain end-to-end on Codex + and Gemini to see whether the principles still apply. +- **Where the "explain WHY" preference breaks down.** Both + OpenAI's "be literal" and Anthropic's "stronger language like + MUST" suggest there's a regime where rationale-prose + under-performs. We currently route this through principle 4 + (specificity matches fragility) but haven't validated the + threshold empirically. +- **Tool-count bound for chain skills.** Gemini's 10-20 cap is + the only quantitative bound we found. The chain ships 9 + 1 + (autopilot pending) = 10. Adding more chain steps would push + toward Gemini's degradation regime. Need to consider this when + scoping future chain extensions. +- **Self-application of the chain.** Per the bootstrapping limit + above, we can't yet use the chain to improve its own SKILL.md + files. The threshold for doing so is "enough external + validation to trust the chain's verdicts" — currently + un-quantified. + +## References + +**Anthropic:** + +- [Agent Skills overview](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/overview) +- [Agent Skills best practices](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/best-practices) +- [Claude Code skills guide](https://docs.claude.com/en/docs/claude-code/skills) +- [anthropics/skills (skill-creator)](https://github.com/anthropics/skills) +- skill-creator plugin: `~/.claude/plugins/cache/claude-plugins-official/skill-creator/` +- superpowers:writing-skills: `~/.claude/plugins/cache/claude-plugins-official/superpowers//skills/writing-skills/` +- [Equipping Agents for the Real World with Agent Skills (engineering blog)](https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills) + +**OpenAI:** + +- [GPT-4.1 prompting guide](https://cookbook.openai.com/examples/gpt4-1_prompting_guide) +- [Apps SDK — define tools](https://developers.openai.com/apps-sdk/plan/tools) +- [Function calling](https://platform.openai.com/docs/guides/function-calling) +- [Key guidelines for Custom GPT instructions](https://help.openai.com/en/articles/9358033) +- [A practical guide to building agents (PDF)](https://cdn.openai.com/business-guides-and-resources/a-practical-guide-to-building-agents.pdf) +- [Agents — OpenAI Agents SDK](https://openai.github.io/openai-agents-python/agents/) + +**Google Gemini:** + +- [Prompt design strategies](https://ai.google.dev/gemini-api/docs/prompting-strategies) +- [System instructions](https://ai.google.dev/gemini-api/docs/system-instructions) +- [Function calling](https://ai.google.dev/gemini-api/docs/function-calling) +- [Gemini CLI extension best practices](https://geminicli.com/docs/extensions/best-practices/) +- [GEMINI.md context files](https://geminicli.com/docs/cli/gemini-md/) +- [Building Gemini CLI Extensions](https://geminicli.com/docs/extensions/writing-extensions/) + +**Community:** + +- [SwirlAI: Agent Skills Progressive Disclosure](https://www.newsletter.swirlai.com/p/agent-skills-progressive-disclosure) +- [VoltAgent/awesome-agent-skills](https://github.com/VoltAgent/awesome-agent-skills) +- [karanb192/awesome-claude-skills](https://github.com/karanb192/awesome-claude-skills) +- [ComposioHQ/awesome-claude-skills](https://github.com/ComposioHQ/awesome-claude-skills) +- [PatrickJS/awesome-cursorrules](https://github.com/PatrickJS/awesome-cursorrules) +- [dev.to: 5 .cursorrules patterns that make Cursor actually reliable](https://dev.to/olivia_craft/5-cursorrules-patterns-that-make-cursor-actually-reliable-41ni) +- [HackerNoon: The Moat is a Config File — leaked system prompt analysis](https://hackernoon.com/the-moat-is-a-config-file-analysis-of-leaked-system-prompts-from-openai-anthropic-google-and-more) +- Meincke et al. 2025 — persuasion principles in AI compliance + (cited by writing-skills/persuasion-principles.md) diff --git a/docs/superpowers/plans/2026-05-19-skill-optimizer-v1.4.md b/docs/superpowers/plans/2026-05-19-skill-optimizer-v1.4.md new file mode 100644 index 0000000..07e634c --- /dev/null +++ b/docs/superpowers/plans/2026-05-19-skill-optimizer-v1.4.md @@ -0,0 +1,1959 @@ +# skill-optimizer v1.4 Implementation Plan + +> **For agentic workers:** This plan has TWO execution modes, by phase: +> +> | Phase | Mode | Tool | +> |---|---|---| +> | **A** (structural setup) | subagent-driven-development OK | mechanical Bash/Edit | +> | **B** (the 8 SKILL.md files — 7 chain skills + auto-pilot) | INTERACTIVE with operator | **`skill-creator`** (outer loop: draft, eval, iterate, description-improver) + **`superpowers:writing-skills`** (inner loop: TDD pressure scenarios when a compliance issue surfaces) | +> | **C** (subagent prompt templates) | INTERACTIVE with operator | **`superpowers:writing-skills`** (TDD pressure scenarios for limited-context constraints — the prompts ARE compliance documents) | +> | **D** (`workflow.md`) | INTERACTIVE with operator | No skill tool — direct authoring + user review per section | +> | **E** (end-to-end validation) | manual operator runs (real eval $) | direct `Skill` invocations + `run-suite` | +> +> **Phases B, C, D must NOT be subagent-dispatched batch.** Writing good skills + prompts is hard, and the user explicitly asked for precision across all of them. Each task is human-in-loop with iterative refinement. +> +> Steps use checkbox (`- [ ]`) syntax for tracking. + +## Why two skill-authoring tools? + +We use **both** `skill-creator` and `superpowers:writing-skills` in v1.4 — they're complementary, not alternatives: + +- **`skill-creator`** (Anthropic plugin, at `~/.claude/plugins/cache/claude-plugins-official/skill-creator/`) + - Author + eval + iterate loop: capture intent → draft → run test prompts → variance analysis → refine + - Ships an `eval-viewer/generate_review.py` script + - Ships a **description-improver** script for optimizing trigger accuracy + - Best for: routed-by-description skills where triggering matters (i.e., our 7 SKILL.md files in Phase B) + +- **`superpowers:writing-skills`** (superpowers plugin) + - TDD discipline applied to documentation: write pressure scenarios with subagents → watch agent fail (RED) → write skill → watch agent pass (GREEN) → refactor to close loopholes + - Required reading: superpowers:test-driven-development + - Best for: compliance documents where the rule HAS to be obeyed under pressure (i.e., subagent prompt templates in Phase C, OR an inner-loop fix when a Phase B SKILL.md keeps letting the agent violate its constraints) + +**The split in v1.4:** + +- Phase B (SKILL.md files): `skill-creator` is primary (description-routing + eval iteration); `writing-skills` is the inner-loop fix when a behavioral compliance issue surfaces during eval (e.g., "the optimizer skill keeps dispatching subagents without limited-context constraints — let me write a pressure scenario to nail this down"). +- Phase C (subagent prompts): `writing-skills` only. These aren't description-routed (skills invoke them explicitly), so `skill-creator`'s description-optimizer doesn't apply. The hard part is preventing the subagent from cheating its limited-context constraints — pure TDD pressure-scenario territory. +- Phase D (workflow.md): Just a reference doc, no skill tool needed. + +**Goal:** Implement v1.4 — replace the v1.3 monolithic `auto-improve-orchestrator` with 7 independent Claude Code skills that chain via the `superpowers` plugin pattern. Each skill produces a human-reviewable report at a convention path; subagents dispatched by each skill run under strict limited context to prevent ducttape patches. + +**Architecture:** Skills (not slash commands), description-routed, output to `docs/skill-optimizer//-.md`. The skill-optimizer engine (`run-suite`, graders, Docker harness) stays unchanged. The v1.3 `auto-improve-orchestrator/` skill is deprecated-but-retained for backward compatibility. + +**Tech Stack:** Markdown (skill prompts), gray-matter (frontmatter parsing, already a project dep), Claude Code Skill tool + Agent tool, bash (validation scripts). + +**Working dir:** `.claude/worktrees/v1.4-spec/` on branch `feat/skill-optimizer-v1.4`. + +**Spec:** `docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md` (committed at `a01d0cc`). + +--- + +## File Structure + +After this plan executes: + +```text +skills/ + skill-optimizer-investigate-functionality/SKILL.md # Phase B (interactive) + skill-optimizer-investigate-test-case/SKILL.md # Phase B (interactive) + skill-optimizer-investigate-submissions/SKILL.md # Phase B (interactive) + skill-optimizer-write-tests/SKILL.md # Phase B (interactive) + skill-optimizer-run-bench/SKILL.md # Phase B (interactive) + skill-optimizer-analyze/SKILL.md # Phase B (interactive) + skill-optimizer-improve/SKILL.md # Phase B (interactive) + skill-optimizer-autopilot/SKILL.md # Phase B8 (interactive) + skill-optimizer-subagents/ # Phase C (interactive) + research-functionality.md + test-case-designer.md + research-submissions.md + test-writer.md + analyzer.md + optimizer.md + validator.md + skill-optimizer-shared/ # Phase A5 (mechanical authoring) + iteration-protocol.md # operational reference all + # chain skills load + +docs/ + skill-optimizer-v1.4-spec.md # already committed + skill-optimizer-v1.4-plan.md # THIS file + skill-writing-philosophy.md # already committed (Phase B0) + skill-optimizer-workflow.md # Phase D (interactive) + skill-optimizer-v1.4-validation.md # written by Phase E +``` + +**Removed from the v1.4 scope** (vs the initial draft of this plan): + +- `skills/auto-improve-orchestrator/` deprecation banner — the v1.3 + orchestrator never landed on `development`, so there's nothing to + deprecate on this branch's lineage. Task A3 was skipped. +- `skills/references/recipes.md` (the seeded v1.3 lessons.md) — the + raw seed is case-study-shaped (accumulated per-pilot observations), + which contradicts the "generalize from feedback" principle in + [`skill-writing-philosophy.md`](skill-writing-philosophy.md). The + curation pass (turn it into named abstract patterns) is deferred to + a follow-up after the chain ships and we have real Phase-E + observations to ground the patterns in. +- `skills/references/` directory — no longer needed once recipes.md + is out and the philosophy doc lives under `docs/`. The subagents + dir is scoped to `skills/skill-optimizer-subagents/` so its + contents can't collide with other plugins' files. + +**Deferred to a post-v1.4 cleanup PR:** + +- Removing or repurposing `skills/skill-optimizer/SKILL.md` (the + original distributable wrapper). Under the v1.4 chain, the + direct-CLI-access role is filled by `skill-optimizer-run-bench`, + so the original skill is redundant. The cleanup touches the + public-facing plugin API (`.claude-plugin/`, README install docs, + `tests/smoke-skill-distribution.ts`) and should land in its own PR + after the v1.4 chain has been validated on a few external skills. + +Per-skill state files (produced at runtime, NOT created by this plan; convention only): + +```text +docs/skill-optimizer// + 01-functionality.md + 02-test-case.md + 03-submissions.md # only if pr_submission_intent: true + 04-tests-plan.md + workbench/ + 06-bench-results// + 07-analysis.md + 08-improvement-proposal.md + 07-validator-verdict.md + vendored-skill/ # for upstream skills only +``` + +Each file's responsibility: + +- **`skills/skill-optimizer-/SKILL.md`** — frontmatter (`name:`, `description:`) + invocation instructions + dispatch-subagent logic + handoff-to-next-skill pointer. Always-loaded by Claude Code when the skill is invoked. +- **`skills/skill-optimizer-subagents/.md`** — narrow-context prompt template that the corresponding skill loads, substitutes inputs into, and dispatches via Agent tool. Scoped name prevents collision with other plugins that might also ship subagents. +- **`skills/skill-optimizer-shared/workflow.md`** — human-readable chain diagram + per-skill brief, for operators understanding how the 7 skills compose. Lives under `docs/` because it's contributor/operator reading, not a runtime resource the skills load. +- **`docs/skill-writing-philosophy.md`** — contributor reference distilling the three philosophies (Anthropic, skill-creator, superpowers:writing-skills), the skill-type → philosophy mapping, the bootstrapping limit, and the rules the optimizer subagent must follow when revising target skills. Also under `docs/` for the same reason. + +--- + +## Phase A — Structural setup (4 tasks, mechanical, subagent-driven-development OK) + +### Task A1: Create the 7 skill directories with shell SKILL.md files + +**Files:** Create 7 directories + 7 shell SKILL.md files. + +- [ ] **Step 1: Create directories** + +```bash +cd /home/yuqing/Documents/Code/skill-optimizer/.claude/worktrees/v1.4-spec +for verb in investigate-functionality investigate-test-case investigate-submissions write-tests run-bench analyze-result improve-skill; do + mkdir -p skills/skill-optimizer-${verb} +done +ls skills/ | grep skill-optimizer- +``` + +Expected: 7 new directories listed (plus the existing `skill-optimizer/`). + +- [ ] **Step 2: Write shell SKILL.md files (frontmatter only, body placeholder for Phase B)** + +For each of the 7 verbs, write `skills/skill-optimizer-/SKILL.md` with this content (use the verb-specific name and a 1-line description from the spec): + +```markdown +--- +name: skill-optimizer- +description: > +--- + +# skill-optimizer- + + + +``` + +Verb → description mapping (from `docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md`): + +```bash +# Run this script to populate all 7 shells: +declare -A DESCS=( + [investigate-functionality]="Use when the user asks to understand or investigate what a skill does — fetches skill source, web-searches the underlying technology, writes a functionality report. Also asks the user (for upstream skills) whether to target upstream PR submission." + [investigate-test-case]="Use when the user wants to design or propose test cases for a skill — enumerates the skill's responsibilities and proposes ranked test cases the user can pick from." + [investigate-submissions]="Use when the user wants to research a skill's upstream PR conventions — produces a context block with license, CLA, frontmatter spec, file-location rules, PR-shape patterns, and rejection signals." + [write-tests]="Use when the user wants to build / implement the eval workbench for a skill — plans the workbench, asks for user confirmation, then dispatches parallel narrow-context subagents to write each test case + grader." + [run-bench]="Use when the user wants to run the eval suite for a skill and capture results — thin wrapper around the skill-optimizer CLI's run-suite command." + [analyze-result]="Use when the user wants to analyze bench results and identify structural weaknesses — clusters failures, distinguishes systematic from flaky, and produces a structured analysis with anti-ducttape gates." + [improve-skill]="Use when the user wants to improve a skill based on identified structural weakness — dispatches optimizer + validator subagents under strict limited context; refuses if no structural weakness in the analysis." +) + +for verb in "${!DESCS[@]}"; do + cat > "skills/skill-optimizer-${verb}/SKILL.md" < + +EOF +done +``` + +- [ ] **Step 3: Verify frontmatter parses for all 7** + +```bash +node -e " +const matter = require('gray-matter'); +const fs = require('fs'); +const verbs = ['investigate-functionality', 'investigate-test-case', 'investigate-submissions', 'write-tests', 'run-bench', 'analyze-result', 'improve-skill']; +for (const v of verbs) { + const p = 'skills/skill-optimizer-' + v + '/SKILL.md'; + const m = matter(fs.readFileSync(p, 'utf-8')); + if (m.data.name !== 'skill-optimizer-' + v || typeof m.data.description !== 'string' || m.data.description.length < 50) { + console.error('FAIL:', p, m.data); + process.exit(1); + } + console.log('OK:', p); +} +" +``` + +Expected: 7 `OK:` lines. + +- [ ] **Step 4: Commit** + +```bash +git add skills/skill-optimizer-*/SKILL.md +git commit -m "feat(v1.4): create 7 skill directory shells with frontmatter + +Body content for each SKILL.md will be written interactively in Phase B +via superpowers:writing-skills, with the interface contract from +docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md as the brief. This commit just lays +down the structural skeleton and discoverable frontmatter." +``` + +--- + +### Task A2: Create `skill-optimizer-subagents/` directory + +**Files:** Create 1 directory + `.gitkeep` so it's tracked. + +> **As-executed (revised mid-Phase-A):** the initial draft of this +> task created two directories, `skills/subagents/` and +> `skills/references/`. Both were revised: +> +> - `subagents/` was renamed to the scoped +> `skill-optimizer-subagents/` so its contents can't collide with +> other plugins that ship subagents. +> - `references/` was dropped entirely. The only file it would have +> held (`recipes.md`) was deferred (see superseded A4 below), and +> the philosophy doc moved to `docs/skill-writing-philosophy.md`. + +```bash +mkdir -p skills/skill-optimizer-subagents +touch skills/skill-optimizer-subagents/.gitkeep +git add skills/skill-optimizer-subagents/.gitkeep +git commit -m "chore(v1.4): create skill-optimizer-subagents/ dir (populated in Phase C)" +``` + +--- + +### Task A3: ~~Add deprecation banner to v1.3 orchestrator~~ (SKIPPED) + +> **Skipped at execution time.** The v1.3 +> `skills/auto-improve-orchestrator/` skill does not exist on the +> `development` branch's lineage (it lives only on +> `feat/auto-improve-skill-v1.3` and a few `eval/auto-pilot/*` +> branches). There's nothing to deprecate on this branch, so the +> task is moot. If v1.3 were ever landed on `development` as a +> separate effort, revisit. + +--- + +### Task A4: ~~Seed `references/recipes.md` from v1.3 `lessons.md`~~ (DEFERRED) + +> **Deferred to a post-v1.4 follow-up.** The initial plan seeded a +> shared `recipes.md` from v1.3's `lessons.md` so the analyzer and +> optimizer subagents would have a pre-populated pattern library. +> On review, the raw seed is case-study-shaped (each pilot run +> appended its specific observations) rather than +> abstract-pattern-shaped, which contradicts the +> "generalize-from-feedback" principle in +> [`skill-writing-philosophy.md`](skill-writing-philosophy.md). +> +> The right shape for `recipes.md` is a small set of named abstract +> patterns (Recipe A–E, grader patterns G1–G6) with brief principled +> descriptions — not 370 lines of per-pilot case histories. Producing +> that curated version requires real Phase-E observations to ground +> the patterns in. +> +> Until then, the analyzer and optimizer subagents (Phase C) operate +> without a pre-loaded recipe library; they reason from +> first-principles using `01-functionality.md` and the per-run +> analysis. A `references/recipes.md` (or an equivalent under +> `docs/`) can be added later once we've validated the chain on a +> few external skills and have real cross-run patterns to name. + +--- + +### Task A5: Author `skills/skill-optimizer-shared/iteration-protocol.md` + +**Files:** + +- Create: `skills/skill-optimizer-shared/iteration-protocol.md` + +> **As-executed (added mid-execution):** the initial plan didn't +> include this file. It emerged during B1's revision: each chain +> skill needs ~40 lines of identical iteration scaffolding (archive, +> version bump, directives collection), and inlining that into +> each of the 7 SKILL.md files would mean ~280 duplicated lines. +> Factoring the mechanics into one shared operational reference +> keeps each SKILL.md focused on this-skill's specifics, and the +> protocol stays consistent across the chain. +> +> Each chain skill's "Handle iteration" step (around the middle of +> its workflow) explicitly instructs the agent to **read this file** +> before proceeding — the load-bearing-ness is the whole point, so +> the reference is mandatory, not optional. + +- [ ] **Step 1: Write the iteration protocol** + +Direct authoring (no skill tool — it's a mechanical reference doc, +not a skill or a compliance prompt template). The doc covers: + +- Versioning convention (`version` + `inputs` frontmatter) +- The decision tree applied on every chain-skill invocation +- Collecting `${OPERATOR_DIRECTIVES}` (atomic new requirements, not + context dumps) +- Subagent constraints under iteration (ignore own prior output, + read upstream at latest) +- Cascading staleness (direct-upstream-only checks; user judgment + trusted on skips) +- Archive convention table +- The bootstrapping case (fresh slug, no prior dir) +- What this protocol does NOT cover (bench-results timestamping, + auto-pilot summaries, vendored-skill cache) + +- [ ] **Step 2: Verify + commit** + +```bash +git add skills/skill-optimizer-shared/iteration-protocol.md +git commit -m "feat(v1.4-shared): iteration-protocol.md — shared mechanics for chain skills" +``` + +--- + +## Phase B — Interactive SKILL.md creation (8 tasks, INTERACTIVE via skill-creator + writing-skills) + +**Mode:** Each Task B is performed interactively in the operator's CC session. + +- **Primary tool: `skill-creator`** (Anthropic plugin) — outer loop. Capture intent → draft SKILL.md → generate eval test prompts → run variance analysis → iterate on description and body until the skill triggers reliably and behaves correctly. +- **Inner-loop tool: `superpowers:writing-skills`** — invoked WHEN a behavioral compliance issue surfaces during skill-creator's eval iteration (e.g., "the optimizer skill keeps dispatching subagents without limited-context constraints" → write a TDD pressure scenario, watch the subagent fail without the rule, add the rule, watch it pass). Not every Task B needs writing-skills; reach for it when a constraint must hold under adversarial pressure. + +**Do NOT dispatch these as autonomous subagents** — both tools are built for human-in-loop refinement, and the user explicitly stated "writing good skills is HARD." + +For EACH Task B, the brief for `skill-creator` is the interface contract for that skill from `docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md` section "The 7 skills", plus the constraints from "Subagent constraints" if the skill dispatches a subagent. Paste those excerpts verbatim into the `skill-creator` capture step. + +--- + +### Task B1: SKILL.md for `skill-optimizer-investigate-functionality` + +**Files:** + +- Modify: `skills/skill-optimizer-investigate-functionality/SKILL.md` (replace the placeholder body) + +- [ ] **Step 1: Invoke skill-creator with the spec excerpt as the brief** + +In your Claude Code session, invoke: + +```text +/skill skill-creator +``` + +Use skill-creator's capture → draft → eval → iterate loop. If during eval iteration a behavioral compliance issue surfaces (e.g., the skill repeatedly skips a required step under pressure), pause and invoke `superpowers:writing-skills` to author a TDD pressure scenario for that specific failure, then resume skill-creator. + +Brief (paste into skill-creator's capture step): + +```text +Create SKILL.md body for skills/skill-optimizer-investigate-functionality/. + +Interface contract (from docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md "The 7 skills" → §1): + +- Description trigger: "investigate / understand / explain what this skill does (and the context around it)" +- Input: source skill (URL or local path) +- Output: docs/skill-optimizer//01-functionality.md +- Behavior: + 1. Identify whether the source is upstream (URL or owner/repo/skill-id) or local (path) + 2. If upstream: ASK the user up-front "do you want to optimize this skill for upstream PR submission?" Record the answer in the report frontmatter as pr_submission_intent: true|false + 3. If local: no PR question. Set pr_submission_intent: false + 4. Fetch the skill files (if URL); vendor to vendored-skill/ for read-only use by downstream steps + 5. Web-search the underlying technology; identify trigger conditions, success criteria, key terminology, intended audience + 6. Write structured report covering: what the skill does, who uses it, when it should fire, what tools it depends on, what concepts the user must understand +- Dispatches: functionality-researcher subagent (skills/skill-optimizer-subagents/research-functionality.md). Limited context: source skill + targeted web fetches. +- Handoff: "Next, invoke skill-optimizer-investigate-test-case. Note: this report's pr_submission_intent field tells step 2's handoff whether step 3 (investigate-submissions) should run." + +The frontmatter (name + description) is already in place from Task A1 — keep it, only replace the body. + +Examples to draw from for shape: the existing superpowers/skills/brainstorming/SKILL.md for "skill that asks user questions, persists state, hands off to next skill" pattern. +``` + +Iterate with skill-creator (and writing-skills as needed) until the SKILL.md body is satisfactory and skill-creator's eval prompts trigger the skill reliably. Verify the frontmatter is unchanged and the body matches the spec contract. + +- [ ] **Step 2: Verify the file** + +```bash +node -e " +const matter = require('gray-matter'); +const m = matter(require('fs').readFileSync('skills/skill-optimizer-investigate-functionality/SKILL.md', 'utf-8')); +console.log('name:', m.data.name); +console.log('description chars:', m.data.description.length); +console.log('body lines:', m.content.split('\\n').length); +" +``` + +Expected: `name: skill-optimizer-investigate-functionality`, description still present, body lines > 30. + +- [ ] **Step 3: Commit** + +```bash +git add skills/skill-optimizer-investigate-functionality/SKILL.md +git commit -m "feat(skill-optimizer-investigate-functionality): SKILL.md body via skill-creator" +``` + +--- + +### Task B2: SKILL.md for `skill-optimizer-investigate-test-case` + +**Files:** + +- Modify: `skills/skill-optimizer-investigate-test-case/SKILL.md` + +- [ ] **Step 1: Invoke writing-skills with the spec excerpt as the brief** + +```text +/skill superpowers:writing-skills +``` + +Brief: + +```text +Create SKILL.md body for skills/skill-optimizer-investigate-test-case/. + +Interface contract (from docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md §2): + +- Description trigger: "design tests for this skill", "propose test cases", "what should we test" +- Input: docs/skill-optimizer//01-functionality.md +- Output: docs/skill-optimizer//02-test-case.md — ranked list of proposed test cases. Each entry: name, what-it-tests (which responsibility), required setup, expected agent behavior, grader spec, why-it-matters +- Behavior: enumerate the skill's responsibilities from functionality report; design 1–2 cases per responsibility; flag edge cases; estimate grader difficulty; rank by importance + cost-to-build +- Dispatches: none (analytical, operator session) +- User gate: prompts the user to pick which subset of proposed cases to actually build (write-tests acts on the picked subset) +- Handoff: read 01-functionality.md's pr_submission_intent field. If true → "Invoke skill-optimizer-investigate-submissions next, then skill-optimizer-write-tests." If false → "Skip step 3; invoke skill-optimizer-write-tests with the user's picked subset." + +Frontmatter is already in place from Task A1. +``` + +- [ ] **Step 2: Verify + commit (same pattern as B1)** + +```bash +node -e "const m = require('gray-matter')(require('fs').readFileSync('skills/skill-optimizer-investigate-test-case/SKILL.md', 'utf-8')); console.log('lines:', m.content.split('\\n').length);" +git add skills/skill-optimizer-investigate-test-case/SKILL.md +git commit -m "feat(skill-optimizer-investigate-test-case): SKILL.md body via skill-creator" +``` + +--- + +### Task B3: SKILL.md for `skill-optimizer-investigate-submissions` + +**Files:** + +- Modify: `skills/skill-optimizer-investigate-submissions/SKILL.md` + +- [ ] **Step 1: Invoke skill-creator** + +```text +/skill skill-creator +``` + +(See Task B1 Step 1 for the skill-creator + writing-skills workflow.) + +Brief: + +```text +Create SKILL.md body for skills/skill-optimizer-investigate-submissions/. + +Interface contract (from spec §3, OPTIONAL — upstream-only): + +- Description trigger: "research PR conventions for this skill", "what does the upstream repo require" +- Input: source slug // (from 01-functionality.md frontmatter) +- Output: docs/skill-optimizer//03-submissions.md — license, CLA, frontmatter spec, file-location rules, prefix taxonomy, PR-shape patterns from last 10 merged PRs, branch target, rejection signals from last 5 closed-without-merge PRs +- Behavior: gh-CLI heavy (PR list, repo-file API, CONTRIBUTING, sanity-test source, last 10 merged + last 5 closed); produce verbatim-pastable context block for the validator +- Dispatches: submission-researcher subagent (skills/skill-optimizer-subagents/research-submissions.md). Limited context: public repo facts only. +- Skipped when: pr_submission_intent: false in 01-functionality.md (this skill should detect that and exit cleanly with a "skip" message) +- Handoff: "Validator in skill-optimizer-improve will read this for external consistency check. Continue with skill-optimizer-write-tests if not done." + +Reference example: the existing tools/auto-improve-contexts/*.md files (now moved to skills/auto-improve-orchestrator/references/contexts/) are good examples of what 03-submissions.md should contain. +``` + +- [ ] **Step 2: Verify + commit** + +```bash +git add skills/skill-optimizer-investigate-submissions/SKILL.md +git commit -m "feat(skill-optimizer-investigate-submissions): SKILL.md body via skill-creator" +``` + +--- + +### Task B4: SKILL.md for `skill-optimizer-write-tests` + +**Files:** + +- Modify: `skills/skill-optimizer-write-tests/SKILL.md` + +- [ ] **Step 1: Invoke skill-creator** + +```text +/skill skill-creator +``` + +(See Task B1 Step 1 for the skill-creator + writing-skills workflow.) + +Brief: + +```text +Create SKILL.md body for skills/skill-optimizer-write-tests/. + +Interface contract (from spec §4): + +- Description trigger: "build the tests", "implement the workbench", "set up the eval" +- Inputs: docs/skill-optimizer//01-functionality.md, docs/skill-optimizer//02-test-case.md (operator picks subset) +- Outputs: docs/skill-optimizer//04-tests-plan.md (per-case implementation plan) + docs/skill-optimizer//workbench/ (suite.yml, workspace/, checks/, smoke-graders.mjs) +- Behavior: + 1. Plan workbench structure (suite.yml shape, file layout, grader patterns from references/recipes.md) + 2. Show user the plan, ask for confirmation + 3. Dispatch parallel subagents (one per picked test case) — each builds one workspace file + one grader + 4. Run smoke check (hand-crafted GOOD/BAD/EMPTY findings.txt fixtures against each grader) — must pass before commit +- Dispatches: test-writer subagent per case (skills/skill-optimizer-subagents/test-writer.md). LIMITED CONTEXT per spec: single case spec + functionality report; does NOT see skill content or other test cases. +- Parallelizable: each case independent; dispatch in a single message +- Handoff: "Invoke skill-optimizer-run-bench to measure baseline." +``` + +- [ ] **Step 2: Verify + commit** + +```bash +git add skills/skill-optimizer-write-tests/SKILL.md +git commit -m "feat(skill-optimizer-write-tests): SKILL.md body via skill-creator" +``` + +--- + +### Task B5: SKILL.md for `skill-optimizer-run-bench` + +**Files:** + +- Modify: `skills/skill-optimizer-run-bench/SKILL.md` + +- [ ] **Step 1: Invoke skill-creator** + +```text +/skill skill-creator +``` + +(See Task B1 Step 1 for the skill-creator + writing-skills workflow.) + +Brief: + +```text +Create SKILL.md body for skills/skill-optimizer-run-bench/. + +Interface contract (from spec §5): + +- Description trigger: "run the eval", "measure", "benchmark" +- Inputs: docs/skill-optimizer//workbench/ + source skill (vendored) +- Output: docs/skill-optimizer//06-bench-results//suite-result.json + per-trial traces + per-trial findings.txt +- Behavior: invoke skill-optimizer CLI: cd into workbench/, source the repo's .env, run `npx tsx /src/cli.ts run-suite ./suite.yml --trials 3`, capture results into docs/skill-optimizer//06-bench-results// +- Dispatches: none (direct CLI invocation) +- Note: this is the THINNEST skill in v1.4 — most v1.3 run-suite logic stays as-is in the CLI. SKILL.md primarily documents the invocation pattern + how to handle long-running runs. +- Handoff: "Invoke skill-optimizer-analyze." + +Reference: the existing src/cli.ts run-suite command + the v1.3 orchestrator's Phase 3 logic at skills/auto-improve-orchestrator/prompts/orchestrator.md. +``` + +- [ ] **Step 2: Verify + commit** + +```bash +git add skills/skill-optimizer-run-bench/SKILL.md +git commit -m "feat(skill-optimizer-run-bench): SKILL.md body via skill-creator" +``` + +--- + +### Task B6: SKILL.md for `skill-optimizer-analyze` + +**Files:** + +- Modify: `skills/skill-optimizer-analyze/SKILL.md` + +- [ ] **Step 1: Invoke skill-creator** + +```text +/skill skill-creator +``` + +(See Task B1 Step 1 for the skill-creator + writing-skills workflow.) + +Brief: + +```text +Create SKILL.md body for skills/skill-optimizer-analyze/. + +Interface contract (from spec §6 — this is the LOAD-BEARING skill that determines downstream improvement quality): + +- Description trigger: "analyze the results", "diagnose what failed", "find structural weaknesses" +- Inputs: docs/skill-optimizer//06-bench-results// + workbench/ + source skill +- Output: docs/skill-optimizer//07-analysis.md — STRUCTURED per Option A format (see below) +- Behavior: + 1. Cluster failures (per-rule, per-model, per-trial, per-pattern) + 2. Separate FLAKY (single-trial randomness) from SYSTEMATIC (repeated across trials and/or models) + 3. For each systematic cluster: find responsible skill section, hypothesize cause, articulate "what WOULD address" (general principle) + "what WOULD NOT address" (anti-ducktape gate) + 4. List non-structural noise separately + 5. If no structural weakness can be articulated: report explicitly that no weakness was found; the next step (improve-skill) will refuse to fire — this is the honest-no-fabricated-uplift behavior +- Dispatches: analyzer subagent (skills/skill-optimizer-subagents/analyzer.md). LIMITED CONTEXT per spec: per-trial findings + skill content + workbench cases; does NOT see the test inputs themselves (forces focus on SKILL, not solutions). +- Handoff: if ≥1 structural weakness → "Invoke skill-optimizer-improve." If none → exit honestly. + +07-analysis.md format (Option A — structured): + +```markdown +## Structural weaknesses identified + +### Weakness 1: + +- **Pattern**: Across trials, systematically failed to detect . Specifically: . +- **Hypothesized cause**: +- **Connects to skill section**: at +- **What WOULD address this**: +- **What WOULD NOT address this** (anti-ducktape gate): + +## Non-structural noise (ignored — not addressable) +- +``` + +``` + +- [ ] **Step 2: Verify + commit** + +```bash +git add skills/skill-optimizer-analyze/SKILL.md +git commit -m "feat(skill-optimizer-analyze): SKILL.md body via skill-creator" +``` + +--- + +### Task B7: SKILL.md for `skill-optimizer-improve` + +**Files:** + +- Modify: `skills/skill-optimizer-improve/SKILL.md` + +- [ ] **Step 1: Invoke skill-creator** + +```text +/skill skill-creator +``` + +(See Task B1 Step 1 for the skill-creator + writing-skills workflow.) + +Brief: + +```text +Create SKILL.md body for skills/skill-optimizer-improve/. + +Interface contract (from spec §7): + +- Description trigger: "improve this skill", "fix the structural weakness", "optimize" +- Inputs: docs/skill-optimizer//07-analysis.md (REQUIRED — refuses if no weakness identified), 01-functionality.md, source skill, optionally 03-submissions.md +- Outputs: docs/skill-optimizer//08-improvement-proposal.md (diff + rationale referencing the structural weakness) + 07-validator-verdict.md + modified skill file (if approved) +- Behavior: + 1. Refuse if 07-analysis.md has no structural weakness — print "no weakness to address" and exit cleanly + 2. Dispatch OPTIMIZER subagent (skills/skill-optimizer-subagents/optimizer.md) with limited context: sees analysis report + functionality + skill content; does NOT see raw failures, grader logic, test inputs. MUST address named structural weakness using a general principle (NOT a pattern-match patch). Output proposed diff + rationale that explicitly references which named weakness it addresses. + 3. Dispatch VALIDATOR subagent (skills/skill-optimizer-subagents/validator.md) with limited context: sees BEFORE skill + AFTER skill + 01-functionality + 03-submissions (if exists). Two-part check: + - INTERNAL consistency: does the change make sense given the skill's stated responsibilities? Additive vs destructive? General vs ducttape? + - EXTERNAL consistency (only if 03-submissions.md exists): does change conform to upstream PR rules (frontmatter, file location, prefix taxonomy, additive-only, etc.)? + - Verdict: approve / needs-revision / reject + 4. If needs-revision: optimizer revises (max 2 revision rounds total). + 5. If reject: surface verdict honestly and exit. + 6. If approve: write modified skill file + improvement proposal report. +- Dispatches: optimizer + validator, both limited context, both isolated from raw trial data. +- Handoff: + - Local skill: done — modified skill written in place + - Upstream skill, pr_submission_intent: false: done — modified skill written to vendored copy. No PR packaging. + - Upstream skill, pr_submission_intent: true: package the proposed change as a PR draft using 03-submissions.md. The draft includes diff, body, caveats, operator-steps-to-submit. No late "submit a PR?" prompt — decision was made at step 1. +``` + +- [ ] **Step 2: Verify + commit** + +```bash +git add skills/skill-optimizer-improve/SKILL.md +git commit -m "feat(skill-optimizer-improve): SKILL.md body via skill-creator" +``` + +--- + +### Task B8: SKILL.md for `skill-optimizer-autopilot` + +**Files:** + +- Create: `skills/skill-optimizer-autopilot/` directory + `SKILL.md` + (was not part of Phase A; B8 creates the dir too) +- Modify: nothing else + +This is the 8th skill — the auto-pilot driver. It walks 1→7, uses +the version mechanism to skip current reports and re-run stale ones, +and applies default policies for the three human-gate points (PR +intent at B1, picked subset at B2, validator-rejected at B7). + +- [ ] **Step 1: Create the skill directory + frontmatter shell** + +```bash +mkdir -p skills/skill-optimizer-autopilot +cat > skills/skill-optimizer-autopilot/SKILL.md <<'EOF' +--- +name: skill-optimizer-autopilot +description: Use when the user wants to run the full skill-optimizer chain on a skill end-to-end without stepping through manually — phrases like "auto-pilot this skill", "run the whole chain on X", "skill-optimizer end-to-end for X", "automated improvement run for X". Walks steps 1 through 7 using the iteration version mechanism to skip current reports and re-run stale ones, with default policies for the three human-gate points. Best for batch processing where modest results are acceptable; for high-stakes single-target work, drive the steps manually. +--- + +# skill-optimizer-autopilot + + + +EOF +``` + +- [ ] **Step 2: Invoke skill-creator with the spec excerpt as the brief** + +(Same workflow as Task B1 Step 1 — see there.) Brief: + +```text +Goal: write the body of skills/skill-optimizer-autopilot/SKILL.md. + +Interface contract: see docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md §"The skills" #8. + +Key points to cover in the body: +- This is the 8th skill — the auto-pilot driver, not a chain step. +- It walks 1→7 in order, dispatching each chain skill as a subagent. +- For each step, it consults the version mechanism (see spec §"Iteration + patterns" → "Versioning") to decide skip-vs-dispatch. +- Default policies for the three human-gate points: + * B1 PR-intent question → use --pr-intent flag, default false + * B2 user-picks-subset gate → take top N by importance (default N=5, + configurable via --pick-top-n) + * B7 validator-rejected → log final state, no further automated retries +- Iteration cap: --max-iterations-per-step (default 2) +- Output: docs/skill-optimizer//autopilot-summary-.md with + per-step final version + headline result + any blockers +- Caveats: expect modest results vs operator-driven runs; best for batch. + +Load-bearing constraint: auto-pilot dispatches the chain skills via the +Skill tool (treating them as standard subagents); it does NOT do their +work itself in the operator session. This is the same anti-ducktape +discipline as the rest of the chain — applied at the meta level. +``` + +- [ ] **Step 3: Verify + commit** + +```bash +git add skills/skill-optimizer-autopilot/SKILL.md +git commit -m "feat(skill-optimizer-autopilot): SKILL.md body via skill-creator" +``` + +--- + +## Phase C — Subagent prompt templates (7 tasks, INTERACTIVE via writing-skills) + +**Every reasoning subagent prompt must include the +`${OPERATOR_DIRECTIVES}` templated slot** — an atomic list of new +requirements from cross-iteration learnings ("user wants null-input +edge cases", "focus on the gpt-5 cluster"), pre-digested by the +operator session into specific asks. The slot defaults to empty on +iteration 1 and never carries a context dump of prior outputs. See +spec §"Iteration patterns" → "Re-entry contract" for the rationale. + +Each subagent prompt template is loaded by the corresponding skill (Phase B), substituted with templated inputs, and dispatched via Agent tool. All must enforce the limited-context constraints from the spec's "Subagent constraints" table. + +**Mode:** Each Task C is performed interactively via `superpowers:writing-skills`. The drafts embedded below in each task's "Step 1" are the starting brief — NOT the finished prompt. writing-skills' core discipline is **TDD pressure scenarios**: watch a subagent fail without the rule, add the rule, watch the subagent pass. These subagent prompts ARE compliance documents — the limited-context constraint is the load-bearing anti-ducktape gate, and it MUST hold under adversarial conditions. Pure pressure-scenario territory; pure writing-skills fit. + +**Why not skill-creator here?** skill-creator's value-add is description-routing (when does this skill fire?) and eval-prompt variance analysis. These subagent prompts are not description-routed — the parent skill invokes them explicitly. So skill-creator's outer loop doesn't apply. + +**Do NOT dispatch these as autonomous subagents.** Each Task C needs operator-driven pressure scenarios. + +For EACH Task C: + +1. Paste the draft template from Step 1 into writing-skills as the starting brief. +2. Identify the load-bearing constraint(s) (typically the "Tools NOT allowed" / "What you do NOT see" section). +3. Construct a pressure scenario: a realistic operator request where a non-disciplined subagent would cheat the constraint to "be helpful". Watch a fresh subagent fail with just the draft. +4. Strengthen the prompt's constraints/wording. Re-run the scenario. Iterate until the constraint holds. +5. Commit when satisfied. + +### Task C1: `subagents/research-functionality.md` + +**Files:** Create `skills/skill-optimizer-subagents/research-functionality.md`. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +Invoke `/skill superpowers:writing-skills`. Paste the draft template below as the initial content. Then: + +1. Identify the load-bearing constraint (typically "Tools NOT allowed" / "What you do NOT see"). For most C tasks this is the limited-context rule preventing the subagent from reading prior analyses or failure data. +2. Construct a pressure scenario. Example: operator asks the subagent for "everything you know about the skill, including prior issues" — does the subagent stay within its allowed-context boundary, or does it Read forbidden files to "be helpful"? +3. Watch a fresh subagent run with just the draft. If it cheats the constraint, the draft is insufficient. +4. Strengthen wording. Re-run. Iterate until the constraint holds under pressure. + +The draft template: + +Create `skills/skill-optimizer-subagents/research-functionality.md` with this exact content: + +````markdown +# Sub-subagent: research a skill's functionality + +You are dispatched to research what a single agent skill does and the context around it. Produce a structured functionality report. + +## Inputs (templated) + +- `${SKILL_SOURCE}` — URL or local path to the source skill +- `${OUTPUT_PATH}` — where to write the report (typically `docs/skill-optimizer//01-functionality.md`) +- `${PR_SUBMISSION_INTENT}` — `true` or `false`, captured at step 1 by the parent skill + +## Tools allowed + +- Read (skill source files) +- WebFetch (technology docs, vendor sites) +- WebSearch (broader context) +- Glob (find skill files) +- Bash (gh CLI for upstream repos if URL is github) +- Write (the report) + +## Tools NOT allowed + +You operate under limited context per `docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md` "Subagent constraints" table. You see ONLY the source skill + targeted web fetches. You do NOT see: + +- Existing analyses or improvement proposals +- Existing tests for this skill +- Failure data from prior runs + +## What to produce + +Write `${OUTPUT_PATH}` with this structure: + +```markdown +--- +skill_source: ${SKILL_SOURCE} +pr_submission_intent: ${PR_SUBMISSION_INTENT} +classification: +--- + +# Functionality report: + +## What the skill does + +<2–3 paragraph summary of the skill's purpose, what it teaches the agent to do, what problem it solves> + +## Who uses it + + + +## When it should fire (trigger conditions) + + + +## What it depends on + + + +## Key concepts & terminology + + + +## Success criteria (what "good usage" looks like) + + + +## Anti-patterns (common failures) + + + +## Source files inventory + + +``` + +## Commit + +After writing the report, commit on the current branch: + +```bash +git add ${OUTPUT_PATH} +git commit -m "docs(skill-optimizer): functionality report for " +``` + +## Return + +Return to the calling skill a brief summary (under 200 words) covering: classification, dependency-flag list, and any blockers (e.g., "could not fetch source", "skill source is malformed"). +```` + +- [ ] **Step 2: Verify template variables present** + +```bash +grep -E '\$\{(SKILL_SOURCE|OUTPUT_PATH|PR_SUBMISSION_INTENT)\}' skills/skill-optimizer-subagents/research-functionality.md | wc -l +``` + +Expected: ≥ 4 occurrences. + +- [ ] **Step 3: Commit** + +```bash +git add skills/skill-optimizer-subagents/research-functionality.md +git commit -m "feat(v1.4-subagents): research-functionality prompt template" +``` + +--- + +### Task C1b: `subagents/test-case-designer.md` (new — was operator-session in initial spec) + +**Files:** Create `skills/skill-optimizer-subagents/test-case-designer.md`. + +**Why this exists:** the initial spec ran step 2 (test-case design) in +the operator session, reasoning "it's analytical, no research bias to +worry about." That overlooked cross-iteration contamination — on +iteration 2+, the operator session has already absorbed prior failure +data, optimizer attempts, and validator verdicts, and biases test +selection toward "tests that would have caught the things I just +watched fail" instead of "tests that comprehensively cover the +skill's responsibilities." This subagent restores coverage-oriented +test design by running in isolation. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +(Same process as C1.) Draft template: + +````markdown +# Sub-subagent: design test cases for a skill + +You are dispatched to enumerate a skill's responsibilities and design a +ranked list of test cases. Produce a structured `02-test-case.md` report. + +## Inputs (templated) + +- `${FUNCTIONALITY_PATH}` — path to the latest `01-functionality.md` +- `${OUTPUT_PATH}` — typically `docs/skill-optimizer//02-test-case.md` +- `${OPERATOR_DIRECTIVES}` — short bulleted list of atomic new + requirements from prior iterations (e.g., "user wants null-input + edge cases"). Default empty. +- `${PRIOR_VERSION}` — version number for the new report (1 on first + invocation; N+1 if archiving an existing v(N)) + +## Tools allowed + +- Read (`${FUNCTIONALITY_PATH}` only) +- Write (`${OUTPUT_PATH}`) + +## Tools NOT allowed (limited-context constraint) + +You operate under strict limited context. You see ONLY the inputs +above. You do NOT see and MUST NOT attempt to read: + +- Prior `02-test-case.md` drafts (anything under `archive/`) +- `07-analysis.md` or any analysis from prior iterations +- `08-improvement-proposal.md`, `07-validator-verdict.md` +- Raw failure data, `findings.txt`, bench results + +Coverage design must reason from the skill's stated responsibilities +in `01-functionality.md`, augmented only by `${OPERATOR_DIRECTIVES}`. +If you find yourself wanting to read prior failure data to "design +better tests this time" — that's the failure mode this constraint +prevents. Stay in your lane. + +## What to produce + +Write `${OUTPUT_PATH}` with this structure: + +```markdown +--- +version: ${PRIOR_VERSION} +inputs: + step_1_functionality: +--- + +# Test cases for + +## Proposed cases + +For each responsibility identified in 01-functionality.md, design 1–2 +test cases. Each entry: + +### + +- **Tests:** +- **Setup:** +- **Expected agent behavior:** +- **Grader spec:** +- **Why it matters:** +- **Importance:** +- **Cost-to-build:** + +## Ranking + +Ordered list from highest-importance / lowest-cost to lowest- +importance / highest-cost. The operator will use this for the +user-picks-subset gate. +``` + +## Commit + +```bash +git add ${OUTPUT_PATH} +git commit -m "docs(skill-optimizer): test-case proposal v${PRIOR_VERSION} for " +``` + +## Return + +Under 200 words. Cover: total count of proposed cases, your top-3 by +importance, any responsibilities you found that the functionality +report didn't enumerate (flag for the operator to verify), any +cases you flagged as needs-real-tooling that the operator should +sanity-check before committing to. +```` + +- [ ] **Step 2: Verify + commit** + +```bash +grep -E '\$\{(FUNCTIONALITY_PATH|OUTPUT_PATH|OPERATOR_DIRECTIVES|PRIOR_VERSION)\}' skills/skill-optimizer-subagents/test-case-designer.md | wc -l +# Expected: ≥ 6 + +git add skills/skill-optimizer-subagents/test-case-designer.md +git commit -m "feat(v1.4-subagents): test-case-designer prompt template" +``` + +--- + +### Task C2: `subagents/research-submissions.md` + +**Files:** Create `skills/skill-optimizer-subagents/research-submissions.md`. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +Invoke `/skill superpowers:writing-skills`. Paste the draft template below as the initial content. Then: + +1. Identify the load-bearing constraint (typically "Tools NOT allowed" / "What you do NOT see"). For most C tasks this is the limited-context rule preventing the subagent from reading prior analyses or failure data. +2. Construct a pressure scenario. Example: operator asks the subagent for "everything you know about the skill, including prior issues" — does the subagent stay within its allowed-context boundary, or does it Read forbidden files to "be helpful"? +3. Watch a fresh subagent run with just the draft. If it cheats the constraint, the draft is insufficient. +4. Strengthen wording. Re-run. Iterate until the constraint holds under pressure. + +The draft template: + +Create `skills/skill-optimizer-subagents/research-submissions.md` with this exact content (adapted from v1.3's `prompts/research-upstream.md`): + +````markdown +# Sub-subagent: research upstream PR submission conventions + +You are dispatched to research one upstream repo's contribution conventions and produce a context block the validator and packaging steps will use. + +## Inputs (templated) + +- `${SLUG}` — `//` +- `${OUTPUT_PATH}` — typically `docs/skill-optimizer//03-submissions.md` + +## Tools allowed + +- Bash (gh CLI) +- WebFetch +- Read +- Write + +## Tools NOT allowed (limited context constraint) + +You see ONLY public repo facts. You do NOT see: + +- Any proposed change to the skill +- Test data or eval results +- The user's intent for the PR beyond "they intend to submit one" + +## What to research + +For the upstream repo at `${SLUG}`: + +1. License + CLA requirements (read LICENSE, CONTRIBUTING.md) +2. Default branch + branch-target convention (main vs next vs develop) +3. CI workflow gates (`.github/workflows/*.yml`) +4. Frontmatter spec for skill files (read sanity-test source if any) +5. File-location rules (where new content goes; what files are off-limits) +6. Prefix taxonomy (if references/ subdirectory has naming convention) +7. Last 10 merged PRs to this skill (or repo): file count, body shape, commit-message convention, time-to-merge +8. Last 5 closed-without-merge PRs: rejection signals (CLA-missing? shape-novel? discussion-first-gate?) +9. Other downstream consumers (gh search for raw URLs, install scripts, repo's own README) + +## What to produce + +Write `${OUTPUT_PATH}` with the same structure as the v1.3 examples at `skills/auto-improve-orchestrator/references/contexts/*.md`: + +```markdown +# Upstream submission context: ${SLUG} + +## Repository facts +- Repo, license, CLA, maintainers, merge style, CI, discovery-index/downstream-sync + +## Hard constraints (additive-only PR) +- File-location rules, prefix taxonomy, what not to modify, version bump rules + +## Frontmatter spec +- Exact required fields, allowed values, format (string vs YAML list, etc.) + +## Content shape template (copy-and-fill) +- A representative additive change with placeholders + +## Optimization target file +- Where the skill change should land + +## Architecture intent +- Why the upstream organizes things the way it does (informs validator's reasoning) + +## Risk profile +- LOW / MEDIUM / HIGH per category (additive vs restructure, etc.) + +## Pre-submit checklist +- What the PR-packaging step must verify before submission + +## Useful URLs +- Source-of-truth files in the upstream repo +``` + +## Commit + +```bash +git add ${OUTPUT_PATH} +git commit -m "docs(skill-optimizer): submissions context for ${SLUG}" +``` + +## Return + +Under 400 words. Include: license + CLA verdict; recommended branch target; risk profile; the verbatim-pastable context block path. +```` + +- [ ] **Step 2: Verify + commit** + +```bash +grep -E '\$\{(SLUG|OUTPUT_PATH)\}' skills/skill-optimizer-subagents/research-submissions.md | wc -l +# Expected: ≥ 4 + +git add skills/skill-optimizer-subagents/research-submissions.md +git commit -m "feat(v1.4-subagents): research-submissions prompt template" +``` + +--- + +### Task C3: `subagents/test-writer.md` + +**Files:** Create `skills/skill-optimizer-subagents/test-writer.md`. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +Invoke `/skill superpowers:writing-skills`. Paste the draft template below as the initial content. Then: + +1. Identify the load-bearing constraint (typically "Tools NOT allowed" / "What you do NOT see"). For most C tasks this is the limited-context rule preventing the subagent from reading prior analyses or failure data. +2. Construct a pressure scenario. Example: operator asks the subagent for "everything you know about the skill, including prior issues" — does the subagent stay within its allowed-context boundary, or does it Read forbidden files to "be helpful"? +3. Watch a fresh subagent run with just the draft. If it cheats the constraint, the draft is insufficient. +4. Strengthen wording. Re-run. Iterate until the constraint holds under pressure. + +The draft template: + +Create `skills/skill-optimizer-subagents/test-writer.md` with this exact content: + +````markdown +# Sub-subagent: write one test case for the eval workbench + +You are dispatched to build EXACTLY ONE test case (one workspace file + one grader) for a skill's eval workbench. You will be one of N parallel test-writer subagents, each independently writing their own case. + +## Inputs (templated) + +- `${CASE_SPEC}` — verbatim text block from `02-test-case.md` describing this case (name, what-it-tests, required setup, expected agent behavior, grader spec, why-it-matters) +- `${FUNCTIONALITY_REPORT_EXCERPT}` — the relevant sections of `01-functionality.md` that pertain to the responsibility this case exercises +- `${WORKBENCH_DIR}` — typically `docs/skill-optimizer//workbench/` +- `${CASE_NAME}` — short kebab-case identifier (e.g., `review-update-without-where`) + +## Tools allowed + +- Read (workbench README + suite.yml for case-structure conventions; reference recipes) +- Write (the new workspace file + grader) +- Bash (smoke-check verification) + +## Tools NOT allowed (limited context constraint) + +You operate under strict limited context per spec. You see ONLY the single case spec + the functionality excerpt. You do NOT see: + +- The source skill's content (would tempt grader-hacking) +- Other test cases (would tempt copy-and-modify) +- Any existing failure data or analyses +- The eval grader matching logic in `_grader-utils.mjs`'s internals beyond the public API + +## What to produce + +1. **Workspace file** at `${WORKBENCH_DIR}/workspace/.` — realistic input the agent will operate on, with seeded conditions matching what the case is testing +2. **Grader** at `${WORKBENCH_DIR}/checks/grade-${CASE_NAME}-findings.mjs` — JavaScript module that imports from `_grader-utils.mjs` (use `gradeFindings`, `looseRange`, `fuzzyKeyword`, `tolerantKeyword` per `references/recipes.md` G1-G6) and checks the agent's output against expected behavior + +## Smoke check (REQUIRED before commit) + +After writing, hand-craft a 3-fixture smoke check: + +- **GOOD findings.txt** — what the agent SHOULD produce. Run your grader against it. Expect `pass: true, score: 1`. +- **BAD findings.txt** — missing 1–2 of the expected violations. Expect `pass: false, score < 1`. +- **EMPTY findings.txt** — no output at all. Expect `pass: false, score: 0`. + +Add these as inline test cases in `${WORKBENCH_DIR}/checks/smoke-graders.mjs` (or append if file exists). Run: + +```bash +node ${WORKBENCH_DIR}/checks/smoke-graders.mjs +``` + +All assertions must pass before commit. + +## Constraints + +- DO NOT modify other test cases or graders +- DO NOT modify `_grader-utils.mjs` unless you're adding a new helper that doesn't exist (rare) +- DO NOT modify `suite.yml` — that's the parent skill's job after all writer subagents return +- DO NOT run `run-suite` — that's the run-bench skill's job + +## Commit + +```bash +git add ${WORKBENCH_DIR}/workspace/. \ + ${WORKBENCH_DIR}/checks/grade-${CASE_NAME}-findings.mjs \ + ${WORKBENCH_DIR}/checks/smoke-graders.mjs +git commit -m "feat(workbench): add ${CASE_NAME} test case + grader" +``` + +## Return + +Under 200 words. Include: filenames created, smoke-check result, any blockers. +```` + +- [ ] **Step 2: Verify + commit** + +```bash +grep -E '\$\{(CASE_SPEC|FUNCTIONALITY_REPORT_EXCERPT|WORKBENCH_DIR|CASE_NAME)\}' skills/skill-optimizer-subagents/test-writer.md | wc -l +# Expected: ≥ 8 + +git add skills/skill-optimizer-subagents/test-writer.md +git commit -m "feat(v1.4-subagents): test-writer prompt template" +``` + +--- + +### Task C4: `subagents/analyzer.md` + +**Files:** Create `skills/skill-optimizer-subagents/analyzer.md`. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +Invoke `/skill superpowers:writing-skills`. Paste the draft template below as the initial content. Then: + +1. Identify the load-bearing constraint (typically "Tools NOT allowed" / "What you do NOT see"). For most C tasks this is the limited-context rule preventing the subagent from reading prior analyses or failure data. +2. Construct a pressure scenario. Example: operator asks the subagent for "everything you know about the skill, including prior issues" — does the subagent stay within its allowed-context boundary, or does it Read forbidden files to "be helpful"? +3. Watch a fresh subagent run with just the draft. If it cheats the constraint, the draft is insufficient. +4. Strengthen wording. Re-run. Iterate until the constraint holds under pressure. + +The draft template: + +Create `skills/skill-optimizer-subagents/analyzer.md` with this exact content: + +````markdown +# Sub-subagent: analyze bench results, identify structural weaknesses + +You are dispatched to analyze the eval-suite results for one skill and produce a structured weakness report. Your output is the LOAD-BEARING input to the improve-skill step — if you can't articulate a structural weakness, no improvement will be attempted. + +## Inputs (templated) + +- `${BENCH_RESULTS_DIR}` — typically `docs/skill-optimizer//06-bench-results//` +- `${WORKBENCH_DIR}` — `docs/skill-optimizer//workbench/` +- `${SKILL_SOURCE_DIR}` — `docs/skill-optimizer//vendored-skill/` (or local skill path) +- `${OUTPUT_PATH}` — typically `docs/skill-optimizer//07-analysis.md` +- `${RECIPES_PATH}` — `skills/references/recipes.md` + +## Tools allowed + +- Read (bench results, suite-result.json, per-trial findings.txt, workbench cases, source skill, recipes) +- Glob (find trial dirs) +- Write (the report) + +## Tools NOT allowed (limited context constraint) + +You see per-trial findings + skill content + workbench case DEFINITIONS, but you do NOT see: + +- The TEST INPUTS themselves (e.g., the seeded SQL/TSX/JSON files in workspace/) — this forces you to think about the SKILL's gaps, not the SOLUTIONS to specific tests +- The grader's internal matching logic (only see its public output) +- Any existing improvement proposal + +## What to produce + +Analyze the bench results and write `${OUTPUT_PATH}` per this STRUCTURED FORMAT (Option A from the spec): + +```markdown +--- +suite_result_path: ${BENCH_RESULTS_DIR}/suite-result.json +analyzer_verdict: +weakness_count: +--- + +# Analysis: + +## Structural weaknesses identified + +### Weakness 1: + +- **Pattern**: Across / trials, models systematically failed to detect . Specifically: . +- **Hypothesized cause**: +- **Connects to skill section**: at `` +- **What WOULD address this** (general principle): +- **What WOULD NOT address this** (anti-ducktape gate): + +### Weakness 2: ... + +## Non-structural noise (ignored — not addressable) + +- +``` + +If you cannot articulate ≥1 structural weakness with confidence, write the report with `analyzer_verdict: no-weakness` and document why. The improve-skill step will refuse to fire — this is the honest behavior. + +## Method + +1. Read `suite-result.json` — get per-case-per-model pass rates +2. For each FAILED trial, read its `findings.txt` (or equivalent output) +3. Cluster failures by: same rule-ID across trials, same model across rules, same case across models +4. Distinguish: + - **SYSTEMATIC** (≥ 2 trials in the same cluster): worth a weakness entry + - **FLAKY** (1 trial, no pattern): non-structural noise +5. For each systematic cluster, locate the responsible skill section, hypothesize cause, and define the anti-ducktape gate +6. Cross-reference `${RECIPES_PATH}` for known patterns (Recipe A–E, grader patterns G1–G6) — if your weakness matches a known pattern, name it + +## Commit + +```bash +git add ${OUTPUT_PATH} +git commit -m "docs(skill-optimizer): analysis report for " +``` + +## Return + +Under 300 words. Include: verdict (weakness-found vs no-weakness), weakness names, brief reasoning, and a pointer to the report. +```` + +- [ ] **Step 2: Verify + commit** + +```bash +grep -E '\$\{(BENCH_RESULTS_DIR|WORKBENCH_DIR|SKILL_SOURCE_DIR|OUTPUT_PATH|RECIPES_PATH)\}' skills/skill-optimizer-subagents/analyzer.md | wc -l +# Expected: ≥ 8 + +git add skills/skill-optimizer-subagents/analyzer.md +git commit -m "feat(v1.4-subagents): analyzer prompt template" +``` + +--- + +### Task C5: `subagents/optimizer.md` + +**Files:** Create `skills/skill-optimizer-subagents/optimizer.md`. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +Invoke `/skill superpowers:writing-skills`. Paste the draft template below as the initial content. Then: + +1. Identify the load-bearing constraint (typically "Tools NOT allowed" / "What you do NOT see"). For most C tasks this is the limited-context rule preventing the subagent from reading prior analyses or failure data. +2. Construct a pressure scenario. Example: operator asks the subagent for "everything you know about the skill, including prior issues" — does the subagent stay within its allowed-context boundary, or does it Read forbidden files to "be helpful"? +3. Watch a fresh subagent run with just the draft. If it cheats the constraint, the draft is insufficient. +4. Strengthen wording. Re-run. Iterate until the constraint holds under pressure. + +The draft template: + +Create `skills/skill-optimizer-subagents/optimizer.md` with this exact content: + +````markdown +# Sub-subagent: propose a principled fix to address a structural weakness + +You are dispatched to write ONE additive, principled improvement to a skill that addresses a named structural weakness from the analysis report. You MUST NOT pattern-match-patch. + +## Inputs (templated) + +- `${ANALYSIS_PATH}` — `docs/skill-optimizer//07-analysis.md` +- `${FUNCTIONALITY_PATH}` — `docs/skill-optimizer//01-functionality.md` +- `${SKILL_SOURCE_DIR}` — vendored skill files (or local path) +- `${SUBMISSIONS_PATH}` — `docs/skill-optimizer//03-submissions.md` if exists; empty/null otherwise +- `${OUTPUT_PROPOSAL_PATH}` — typically `docs/skill-optimizer//08-improvement-proposal.md` +- `${RECIPES_PATH}` — `skills/references/recipes.md` +- `${ITERATION}` — `1` or `2` (validator may request revision once) + +## Tools allowed + +- Read (analysis, functionality, source skill, submissions context, recipes) +- Write (the proposed diff + rationale) +- Edit (the target skill file, to apply the proposed change) + +## Tools NOT allowed (limited context constraint — CRITICAL) + +You do NOT have access to: + +- Raw failed trial outputs (`findings.txt`) +- The grader's matching logic +- The test inputs (workspace/ files) +- Any pattern that would let you "find the specific token the grader looks for and add it to the skill" + +If you find yourself wanting one of these, STOP and ask the calling skill to escalate. You are not allowed to pattern-match-patch. + +## What to produce + +1. Read `${ANALYSIS_PATH}` — identify the named structural weaknesses + their "What WOULD address" and "What WOULD NOT address" gates. +2. Read `${RECIPES_PATH}` — match to a known recipe (Recipe A-E) if applicable. +3. Read `${SKILL_SOURCE_DIR}` — find the section the weakness "Connects to". +4. Design ONE additive change (max ~50 lines added) that addresses the named weakness using a general principle. The change must: + - Reference a named weakness from the analysis (cite by name in your rationale) + - Use the "What WOULD address this" guidance, NOT the "WOULD NOT" anti-patterns + - Be ADDITIVE only — no deletions, no reworded existing rules + - Match the source skill's existing voice/style +5. If `${SUBMISSIONS_PATH}` exists, ensure the change conforms to the upstream's hard constraints (file location, frontmatter, prefix taxonomy, etc.). +6. Apply the change to the source skill file (Edit tool). +7. Write `${OUTPUT_PROPOSAL_PATH}` with this structure: + +```markdown +--- +iteration: ${ITERATION} +weakness_addressed: +recipe: +--- + +# Improvement proposal (iteration ${ITERATION}) + +## Weakness being addressed + + + +## Proposed change + + + +## Rationale + + + +## Conformance to upstream conventions (if applicable) + + +``` + +## Commit + +```bash +git add ${OUTPUT_PROPOSAL_PATH} ${SKILL_SOURCE_DIR}/...modified file... +git commit -m "feat(skill): iter ${ITERATION} — : " +``` + +## Return + +Under 300 words. Include: weakness addressed, recipe applied, diff summary (~lines added), pointer to proposal. + +## If you cannot produce a principled fix + +If the named weakness genuinely has no general-principle fix (extremely rare), return: + +``` +Status: cannot-fix-principled +Reason: +Recommendation: +``` +```` + +- [ ] **Step 2: Verify + commit** + +```bash +grep -E '\$\{(ANALYSIS_PATH|FUNCTIONALITY_PATH|SKILL_SOURCE_DIR|SUBMISSIONS_PATH|OUTPUT_PROPOSAL_PATH|RECIPES_PATH|ITERATION)\}' skills/skill-optimizer-subagents/optimizer.md | wc -l +# Expected: ≥ 10 + +git add skills/skill-optimizer-subagents/optimizer.md +git commit -m "feat(v1.4-subagents): optimizer prompt template (anti-ducktape gates)" +``` + +--- + +### Task C6: `subagents/validator.md` + +**Files:** Create `skills/skill-optimizer-subagents/validator.md`. + +- [ ] **Step 1: Invoke writing-skills with this draft as the starting brief** + +Invoke `/skill superpowers:writing-skills`. Paste the draft template below as the initial content. Then: + +1. Identify the load-bearing constraint (typically "Tools NOT allowed" / "What you do NOT see"). For most C tasks this is the limited-context rule preventing the subagent from reading prior analyses or failure data. +2. Construct a pressure scenario. Example: operator asks the subagent for "everything you know about the skill, including prior issues" — does the subagent stay within its allowed-context boundary, or does it Read forbidden files to "be helpful"? +3. Watch a fresh subagent run with just the draft. If it cheats the constraint, the draft is insufficient. +4. Strengthen wording. Re-run. Iterate until the constraint holds under pressure. + +The draft template: + +Create `skills/skill-optimizer-subagents/validator.md` with this exact content: + +````markdown +# Sub-subagent: validate a proposed skill improvement + +You are dispatched to independently check whether a proposed skill improvement is principled (internal consistency) and (if applicable) ships-as-PR (external consistency). + +## Inputs (templated) + +- `${SKILL_BEFORE_PATH}` — path to the original skill file (pre-modification) +- `${SKILL_AFTER_PATH}` — path to the modified skill file (post-optimizer) +- `${FUNCTIONALITY_PATH}` — `docs/skill-optimizer//01-functionality.md` +- `${SUBMISSIONS_PATH}` — `docs/skill-optimizer//03-submissions.md` if exists; empty/null otherwise +- `${PROPOSAL_PATH}` — `docs/skill-optimizer//08-improvement-proposal.md` (optimizer's rationale) +- `${OUTPUT_VERDICT_PATH}` — typically `docs/skill-optimizer//07-validator-verdict.md` + +## Tools allowed + +- Read (all input files) +- Write (the verdict) +- Bash (diff the before/after) + +## Tools NOT allowed (limited context constraint — CRITICAL) + +You do NOT see: + +- Trial data, findings.txt, raw eval results +- The optimizer's internal reasoning trace +- Test inputs (workspace/ files) + +You are an INDEPENDENT check. The optimizer might have rationalized a bad change; your job is to catch it. + +## What to check + +### Internal consistency (always runs) + +Read `${SKILL_BEFORE_PATH}` and `${SKILL_AFTER_PATH}`. Compute the diff. + +Verify: + +1. **Additive only**: no deletions, no rewordings of existing content +2. **General not specific**: the change embodies a principle, not a token-specific patch (red flag: change adds an exact keyword/string the optimizer might have seen in failed trials) +3. **On-topic**: the change addresses something the skill's functionality (per `${FUNCTIONALITY_PATH}`) actually claims to do +4. **Style match**: matches the source skill's existing voice (terse imperative? prose? bullet list?) +5. **Rationale matches change**: the optimizer's stated rationale in `${PROPOSAL_PATH}` accurately describes what the diff actually does + +### External consistency (only if `${SUBMISSIONS_PATH}` exists) + +Read `${SUBMISSIONS_PATH}`. Verify the change conforms to: + +1. File location (correct path; e.g., new reference file under `references/-.md`) +2. Frontmatter (all required fields, allowed enum values) +3. Prefix taxonomy (no new prefixes; uses existing set) +4. Additive-only at the repo level (no modifications to forbidden files like `_sections.md`, `release-please-config.json`, etc.) +5. Body shape (e.g., `**Incorrect**`/`**Correct**` SQL blocks if required; `Reference:` link if required) +6. Length (within target range from the convention) + +## Output verdict + +Write `${OUTPUT_VERDICT_PATH}` with this structure: + +```markdown +--- +verdict: +internal_consistency: +external_consistency: +--- + +# Validator verdict + +## Internal consistency check + + + +## External consistency check (if applicable) + + + +## Issues found (if any) + + + +## Overall verdict + +approve: change is principled and shippable +needs-revision: ≥1 fixable issue; optimizer should revise once +reject: fundamentally cannot be salvaged (e.g., the diff is destructive, or the change is unrelated to the named weakness) +``` + +## Commit + +```bash +git add ${OUTPUT_VERDICT_PATH} +git commit -m "docs(skill-optimizer): validator verdict for iteration X" +``` + +## Return + +Under 300 words. Include: verdict, the most important issue (if any), and your reasoning for approve/needs-revision/reject. +```` + +- [ ] **Step 2: Verify + commit** + +```bash +grep -E '\$\{(SKILL_BEFORE_PATH|SKILL_AFTER_PATH|FUNCTIONALITY_PATH|SUBMISSIONS_PATH|PROPOSAL_PATH|OUTPUT_VERDICT_PATH)\}' skills/skill-optimizer-subagents/validator.md | wc -l +# Expected: ≥ 10 + +git add skills/skill-optimizer-subagents/validator.md +git commit -m "feat(v1.4-subagents): validator prompt template" +``` + +--- + +## Phase D — Workflow doc (1 task, INTERACTIVE authoring) + +**Mode:** Direct interactive authoring with operator review per section. No skill tool. `skills/skill-optimizer-shared/workflow.md` is a human reference doc — not a behavioral skill (so skill-creator's description-routing/eval loop doesn't apply) and not a compliance prompt template (so writing-skills' pressure scenarios don't apply). The right discipline is a careful walkthrough with the operator: present each section (chain diagram, invariants, etc.), get confirmation, iterate, commit. + +**Do NOT dispatch as autonomous subagent.** The operator must validate that the diagram + invariants match the actual Phase B SKILL.md handoffs and Phase C subagent constraints (which were finalized interactively just prior). + +### Task D1: `skills/skill-optimizer-shared/workflow.md` — human-readable chain diagram + +**Files:** Create `skills/skill-optimizer-shared/workflow.md`. + +- [ ] **Step 1: Walk through each section with the operator, then write the file** + +Use the draft below as the starting point. Before writing, walk through each section interactively with the operator and confirm it reflects what was actually built in Phases B and C. In particular: + +1. Chain diagram — confirm the handoff text matches each SKILL.md's actual "Handoff" prose from Phase B. +2. Subagent constraint table — confirm wording matches the finalized prompts from Phase C (which may have shifted during pressure-scenario iteration). +3. Invariants section — surface any new invariants discovered while writing Phases B/C. + +Then create `skills/skill-optimizer-shared/workflow.md` with the agreed content. Draft below: + +````markdown +# skill-optimizer v1.4 workflow + +This document describes how the 7 skills chain together to optimize one skill end-to-end. Operators read this to understand the flow; the skills themselves embed similar instructions in-prose for the agent. + +## The chain + +```text +┌────────────────────────────────────┐ +│ skill-optimizer- │ +│ investigate-functionality (1) │ ← user provides skill source +│ │ (URL or local path) +│ │ Asks: "PR submission intent?" +│ │ if upstream +└────────────────┬───────────────────┘ + │ writes 01-functionality.md + ▼ +┌────────────────────────────────────┐ +│ skill-optimizer- │ +│ investigate-test-case (2) │ ← reads 01 +│ │ USER GATE: pick test subset +└────────────────┬───────────────────┘ + │ writes 02-test-case.md + │ + ┌────────┴────────┐ + │ │ + ▼ (pr=true) ▼ (pr=false) +┌──────────────────┐ ┌───────────────────────┐ +│ investigate- │ │ (skip step 3) │ +│ submissions(3) │ │ │ +└──────┬───────────┘ └───────────┬───────────┘ + │ 03-submissions.md │ + └──────────┬────────────────┘ + ▼ +┌────────────────────────────────────┐ +│ skill-optimizer-write-tests (4) │ +│ Parallel test-writer subagents │ +│ (one per picked case, limited ctx) │ +└────────────────┬───────────────────┘ + │ writes workbench/ + 04-tests-plan.md + ▼ +┌────────────────────────────────────┐ +│ skill-optimizer-run-bench (5) │ +│ Invokes skill-optimizer CLI │ +└────────────────┬───────────────────┘ + │ writes 06-bench-results// + ▼ +┌────────────────────────────────────┐ +│ skill-optimizer-analyze (6) │ +│ Analyzer subagent (limited ctx — │ +│ no test inputs) │ +└────────────────┬───────────────────┘ + │ writes 07-analysis.md + │ + ┌────────┴────────┐ + │ │ + ▼ (weakness) ▼ (no weakness) +┌──────────────────┐ ┌───────────────────────┐ +│ improve-skill(7) │ │ Exit honestly │ +│ optimizer ↔ │ │ "no weakness; no │ +│ validator loop │ │ improvement warranted"│ +└──────┬───────────┘ └───────────────────────┘ + │ writes 08-improvement-proposal.md + 07-validator-verdict.md + │ writes modified skill file + │ + ├─ local skill → done + ├─ upstream + pr=false → done (vendored copy updated) + └─ upstream + pr=true → package PR draft using 03 +``` + +## State file layout + +All artifacts at convention path: `docs/skill-optimizer//` + +| File | Producer | Consumer(s) | +|---|---|---| +| `vendored-skill/` | (1) | (4) workbench refs, (6), (7) | +| `01-functionality.md` | (1) | (2), (4), (6), (7), validator | +| `02-test-case.md` | (2) | (4) | +| `03-submissions.md` | (3) (optional) | (7) validator (external consistency), packaging | +| `04-tests-plan.md` | (4) | human review | +| `workbench/` | (4) | (5), (6) | +| `06-bench-results//` | (5) | (6) | +| `07-analysis.md` | (6) | (7) — REQUIRED, refuses without | +| `08-improvement-proposal.md` | (7) optimizer | validator | +| `07-validator-verdict.md` | (7) validator | (7) optimizer (revision loop), operator | + +## Limited-context subagents + +The architectural fix for ducttape. See `docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md` "Subagent constraints" for the full table. Quick summary: + +- **Optimizer & validator** never see raw trial data or grader internals — forces principled improvement +- **Test writer** never sees the skill content — prevents grader-hacking +- **Analyzer** never sees the test inputs — forces focus on the skill, not the solutions +- **Researchers** never see proposed changes — pure investigation + +## Auto-pilot vs manual + +- **Manual:** user invokes one skill at a time, reviews each report, then invokes the next +- **Auto-pilot:** user says "optimize skill X end-to-end" — the agent invokes 1 → 2 → (3 if pr) → 4 → 5 → 6 → 7 in order, pausing at the natural user-review gates (after 2 for test-pick, after 7 for proposal review) + +No separate auto-pilot skill — it's just chained `Skill` tool invocations driven by each skill's "Next, invoke X" handoff. +```` + +- [ ] **Step 2: Verify** + +```bash +wc -l skills/skill-optimizer-shared/workflow.md +grep -c "^## " skills/skill-optimizer-shared/workflow.md +``` + +Expected: ≥ 80 lines, ≥ 4 H2 sections. + +- [ ] **Step 3: Commit** + +```bash +git add skills/skill-optimizer-shared/workflow.md +git commit -m "feat(v1.4-references): add workflow.md (chain diagram + state file map)" +``` + +--- + +## Phase E — End-to-end validation (2 tasks, requires real eval runs) + +### Task E1: End-to-end test on a LOCAL skill + +**Goal:** Validate the 7-skill chain end-to-end on a local skill. Per acceptance criteria #5, this proves the workflow produces all expected reports, modifies the target skill, and the validator approves. + +Choose a small local skill from `skills/` (e.g., `skills/skill-optimizer/SKILL.md` itself, or create a tiny placeholder skill in this repo). Optimize it for local use (no PR submission). + +- [ ] **Step 1: Pick a target local skill** + +```bash +ls skills/ +``` + +Recommend: create a fresh tiny `skills/v14-validation-target/SKILL.md` for the test, so we don't disrupt real skills. Body should have a deliberate weakness (e.g., a rule stated declaratively that an agent would routinely miss). + +```bash +mkdir -p skills/v14-validation-target +cat > skills/v14-validation-target/SKILL.md <<'EOF' +--- +name: v14-validation-target +description: Test target skill for v1.4 validation. Reviews YAML configs for naming-convention compliance. +--- + +# v14-validation-target + +Review YAML configuration files for compliance with naming conventions. + +## Rules + +- Keys should use snake_case +- Boolean values should be lowercase `true`/`false` +- Lists should not be empty + +EOF + +git add skills/v14-validation-target/SKILL.md +git commit -m "test(v1.4): add deliberate-weakness target skill for E2E validation" +``` + +- [ ] **Step 2: Invoke the chain manually** + +In your Claude Code session, walk through: + +```text +1. /skill skill-optimizer-investigate-functionality (point at skills/v14-validation-target/SKILL.md; answer "no" to PR question since it's local) +2. /skill skill-optimizer-investigate-test-case (pick 2-3 cases from the proposal) +3. /skill skill-optimizer-write-tests (let the parallel writers build the workbench) +4. /skill skill-optimizer-run-bench (run the baseline; may take 5-15 min depending on model matrix) +5. /skill skill-optimizer-analyze (read the analysis; verify it identifies the absence-of-procedural-instruction weakness OR honestly says no weakness) +6. If weakness identified: /skill skill-optimizer-improve (verify optimizer+validator approve a principled change) +7. If no weakness: verify the chain exits honestly — no fabricated proposal +``` + +- [ ] **Step 3: Verify all expected reports exist** + +```bash +SLUG=local-v14-validation-target +test -f docs/skill-optimizer/$SLUG/01-functionality.md +test -f docs/skill-optimizer/$SLUG/02-test-case.md +test -d docs/skill-optimizer/$SLUG/workbench +test -d docs/skill-optimizer/$SLUG/06-bench-results +test -f docs/skill-optimizer/$SLUG/07-analysis.md +# 07-* files only if weakness was identified +echo "all expected reports present" +``` + +- [ ] **Step 4: Write validation note + commit** + +```bash +cat > docs/skill-optimizer-v1.4-validation.md <. +- Phase 2 (test-case): picked cases out of proposed. +- Phase 4 (write-tests): parallel writers; smoke checks passed. +- Phase 5 (run-bench): baseline . +- Phase 6 (analyze-result): . . +- Phase 7 (improve-skill): . . + +## Verdict + +v1.4 chain works end-to-end on a local skill. . +EOF + +git add docs/skill-optimizer-v1.4-validation.md +git commit -m "test(v1.4): document local-skill E2E validation result" +``` + +--- + +### Task E2: End-to-end test re-running firecrawl (the v1.3 regression case) + +**Goal:** Per acceptance criteria #6, re-run the firecrawl skill (which v1.3 regressed) under v1.4. Expected: optimizer produces a principled fix OR honestly refuses — no regression shipped. + +- [ ] **Step 1: Run the chain on firecrawl as an upstream skill** + +In your Claude Code session: + +```text +1. /skill skill-optimizer-investigate-functionality (point at firecrawl/skills/firecrawl-build-scrape; answer "yes" to PR question — we want the full upstream path) +2. /skill skill-optimizer-investigate-test-case +3. /skill skill-optimizer-investigate-submissions (auto-invoked since pr=true) +4. /skill skill-optimizer-write-tests +5. /skill skill-optimizer-run-bench +6. /skill skill-optimizer-analyze +7. /skill skill-optimizer-improve +``` + +- [ ] **Step 2: Verify NO regression on the original case** + +The v1.3 regression was specifically: iteration 1's Recipe A+D combination caused the original `review-scrape-integration` case to regress from 1.00 → 0.44 on gpt-5/gemini. Under v1.4, EITHER: + +- The optimizer produces a single principled change (per validator approval), the post-iteration run-bench shows the original case still at 1.00 AND the new harder cases improve, OR +- The analyzer reports "no clear structural weakness" and improve-skill refuses to fire honestly + +Run an extra `run-bench` AFTER improve-skill completes to verify the original case isn't regressed. (The improve-skill skill's own validator should already check this via the optimizer's output, but a confirming run is reasonable.) + +- [ ] **Step 3: Append to validation note** + +```bash +cat >> docs/skill-optimizer-v1.4-validation.md < +Target: firecrawl/skills/firecrawl-build-scrape (upstream, pr=true) + +## Result +- Phase 6 verdict: +- Phase 7 outcome: +- Original case (review-scrape-integration) post-iteration: (baseline: 1.00) +- NO regression observed: + +## v1.3 comparison + +Under v1.3, firecrawl iteration 1 regressed the original case from 1.00 → 0.44 by piling on Recipe A+D simultaneously. The orchestrator ran out of context before catching the regression. v1.4's validator + limited-context optimizer either prevented the regression OR honestly refused to ship the change. + +## Verdict + +v1.4 successfully addresses the canonical "ducktape-by-monolithic-orchestrator" case study from v1.3. . +EOF + +git add docs/skill-optimizer-v1.4-validation.md +git commit -m "test(v1.4): document firecrawl re-run — v1.3 regression case study" +``` + +--- + +## Self-Review + +After writing the plan, here's the spec-coverage check: + +**Spec section → plan tasks:** + +- §"Architecture overview" → Task A1 (skill dirs), A2 (subagents/references), D1 (workflow.md) +- §"Subagent constraints" → Tasks C1-C6 each encode their limited-context constraints, hardened via writing-skills pressure scenarios +- §"The 7 skills" — interface contracts → Tasks B1-B7 (each task pastes the relevant spec excerpt as the skill-creator brief) +- §"Auto-pilot mode" → covered by D1 (workflow.md) + each skill's in-prose handoff (created in Phase B) +- §"Plugin packaging" → no new task needed; existing `.claude-plugin/plugin.json` auto-discovers the new skills via `skills/` convention +- §"Coexistence with v1.3" → Task A3 (deprecation banner) +- §"Acceptance criteria" #1-7 → Tasks A1, C1-C6 (#2 subagent constraints), D1 (#3 workflow), A4 (#4 recipes), E1 (#5 local E2E), E2 (#6 firecrawl E2E) +- §"Out of scope" — Codex/Cursor ports + programmatic auto-pilot wrapper + per-skill prose enrichment — NOT in plan, deliberately deferred + +**Placeholder scan:** None — every task has either complete content (Phases A, C, D), a complete spec-excerpt brief (Phase B), or concrete CLI commands (Phase E). + +**Type consistency:** + +- Skill names consistent: all 7 follow `skill-optimizer-` pattern across spec, file paths, brief excerpts +- State file paths consistent: `docs/skill-optimizer//-.md` (or `workbench/`, `06-bench-results//`, `vendored-skill/`) throughout +- Subagent file paths consistent: `skills/skill-optimizer-subagents/.md` throughout +- Template variable names (`${SLUG}`, `${WORKBENCH_DIR}`, etc.) match between subagent templates and skills' planned invocation patterns + +No issues found. + +--- + +## Execution Handoff + +Plan complete and saved to `docs/superpowers/plans/2026-05-19-skill-optimizer-v1.4.md` (in worktree `.claude/worktrees/v1.4-spec/`, branch `feat/skill-optimizer-v1.4`). + +**Phase-by-phase execution mode (see header table for tool assignment):** + +| Phase | Mode | Driver | +|---|---|---| +| **A** (4 tasks, structural setup) | subagent-driven OK | `superpowers:subagent-driven-development` | +| **B** (7 tasks, the SKILL.md files) | INTERACTIVE — DO NOT subagent-drive | operator runs `skill-creator` (+ `writing-skills` inner loop) | +| **C** (6 tasks, subagent prompt templates) | INTERACTIVE — DO NOT subagent-drive | operator runs `superpowers:writing-skills` with pressure scenarios | +| **D** (1 task, `workflow.md`) | INTERACTIVE — DO NOT subagent-drive | operator authors directly with section-by-section review | +| **E** (2 tasks, end-to-end validation) | operator-driven (real eval runs) | direct CLI / skill invocations; costs $$ + time | + +**Why only Phase A is subagent-eligible:** Phase A is pure file-structure scaffolding with no judgment calls. Phases B–D produce the load-bearing artifacts of v1.4 — the SKILL.md descriptions that determine routing, the subagent prompts that hold the anti-ducktape constraints, and the workflow doc that humans use to understand the chain. The user's explicit guidance: "writing good skills is HARD" and "part C and D should also be done interactively. I want everything to be precise here." Subagent-driving Phases B–D would defeat the purpose. + +**Recommended order:** + +1. Subagent-driven Phase A (~30 min) +2. Operator interactively runs Phase B (7 tasks, ~2–3 hours via `skill-creator` + `writing-skills`) +3. Operator interactively runs Phase C (6 tasks, ~1–2 hours via `writing-skills` pressure scenarios) +4. Operator interactively runs Phase D (1 task, ~30 min) +5. Operator runs Phase E (2 tasks, ~30–60 min each — real eval runs) + +Which approach? diff --git a/docs/superpowers/plans/2026-05-25-multi-agent-acp.md b/docs/superpowers/plans/2026-05-25-multi-agent-acp.md new file mode 100644 index 0000000..0ae6e49 --- /dev/null +++ b/docs/superpowers/plans/2026-05-25-multi-agent-acp.md @@ -0,0 +1,3558 @@ +# Multi-agent ACP Workbench Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use +> superpowers:subagent-driven-development (recommended) or +> superpowers:executing-plans to implement this plan task-by-task. Steps +> use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Replace the embedded pi-agent runtime with a host-side ACP +client driving five agent CLIs (Claude Code, Codex, Gemini, OpenCode, +pi-acp) inside per-trial sandboxed Docker containers. + +**Architecture:** Host orchestrates everything (case load, container +lifecycle, ACP handshake, trace capture, grading). The container is a +pure sandbox with the agent CLI pre-baked at image-build time. The +official `@agentclientprotocol/sdk` provides the ACP runtime; an agent +registry holds per-agent install/launch/auth specifics, ported from +benchflow's battle-tested entries. + +**Tech Stack:** TypeScript (existing `src/workbench/`), +`@agentclientprotocol/sdk@0.22.1`, Docker (existing), `yaml` (existing +case-loader dependency), `node:test` (existing test runner). + +**Spec:** +[`docs/superpowers/specs/2026-05-25-multi-agent-acp-design.md`](../specs/2026-05-25-multi-agent-acp-design.md) + +--- + +## File structure + +### New files + +| Path | Responsibility | +|---|---| +| `src/workbench/acp/transport.ts` | Wrap `docker exec -i` stdio as ACP `Stream` | +| `src/workbench/acp/client.ts` | `ClientSideConnection` wrapper + our `Client` impl | +| `src/workbench/acp/auth.ts` | Subscription detection + Docker mount spec | +| `src/workbench/acp/skill-deploy.ts` | Host skill dir → container mount | +| `src/workbench/acp/mcp-config-writer.ts` | Case `mcpServers:` → per-agent native config | +| `src/workbench/acp/trace-recorder.ts` | Buffer raw ACP messages → `trace.jsonl` | +| `src/workbench/agents/registry.ts` | `AgentConfig` type + 5 entries + `resolveAgent` | +| `src/workbench/agents/install-snippets.ts` | Bash install snippets used by Dockerfile | +| `src/workbench/parse-trace.ts` | `iterMessages` / `iterToolCalls` / `computeMetrics` | +| `docker/skill-optimizer-agent.Dockerfile` | Bakes node + python + all 5 agent CLIs | +| `skills/shared/acp-trace-format.md` | Chain analyzer reference for ACP event types | +| `tests/acp/transport.test.ts` | Mock-subprocess unit tests | +| `tests/acp/auth.test.ts` | Resolution priority unit tests | +| `tests/acp/mcp-config-writer.test.ts` | Per-agent fixture assertions | +| `tests/agents/registry.test.ts` | Alias + resolution unit tests | +| `tests/parse-trace.test.ts` | Fixture trace assertions | +| `tests/smoke-agents/claude-agent-acp.smoke.test.ts` | E2E smoke probe | +| `tests/smoke-agents/codex-acp.smoke.test.ts` | E2E smoke probe | +| `tests/smoke-agents/gemini.smoke.test.ts` | E2E smoke probe | +| `tests/smoke-agents/opencode.smoke.test.ts` | E2E smoke probe | +| `tests/smoke-agents/pi-acp.smoke.test.ts` | E2E smoke probe | +| `tests/regression/pi-acp-pdf-suite.test.ts` | Pi-acp regression against pdf suite | +| `tests/fixtures/acp-traces/-sample.jsonl` | Captured fixture traces | + +### Modified files + +| Path | Change | +|---|---| +| `package.json` | Add `@agentclientprotocol/sdk@0.22.1` dependency | +| `src/workbench/case-loader.ts` | Require `agent:` field, fail loud | +| `src/workbench/suite-loader.ts` | `runs:` matrix replaces `models:` | +| `src/workbench/types.ts` | Add `AgentConfig` types; remove `WorkbenchTraceEntry` | +| `src/workbench/docker-runner.ts` | Heavy rewrite: ACP host-side orchestration | +| `src/workbench/container-runner.ts` | Remove `--agent` mode, keep `--setup` / `--grade` | +| `src/workbench/run-case.ts` | Adapt to `runs:` matrix; per-run trial dirs | +| `src/workbench/run-suite.ts` | Same | +| `src/workbench/trials.ts` | Aggregate tokens + duration (close metrics gap) | +| `src/workbench/metrics.ts` | Thin wrapper around `parse-trace.ts`; drop cost | +| `src/workbench/trace.ts` | Collapse to raw JSONL writer; drop normalization | +| `skills/write-tests/agents/test-writer.md` | Drop `/work/` skill-path boilerplate | +| `skills/analyze/agents/analyzer.md` | Point at `acp-trace-format.md` reference | +| `skills/run-bench/SKILL.md` | Add `agent:` column, drop cost, surface tokens + duration | +| `examples/workbench/pdf/suite.yml` | Convert `models:` → `runs:` with `agent: pi-acp` | +| `examples/workbench/mcp/suite.yml` | Convert `models:` → `runs:` with `agent: pi-acp` | +| `.gitignore` | Add `tests/fixtures/acp-traces/*.local.jsonl` if regenerating | + +### Deleted files + +| Path | Why | +|---|---| +| `src/workbench/pi-agent.ts` | Pi runs through pi-acp; no embedded library | +| `docker/workbench-runner.Dockerfile` | Replaced by `skill-optimizer-agent.Dockerfile` | +| `examples/workbench/*/.run.log` | Stale run logs, not source | +| `examples/workbench/*/.results/` | Stale result dirs, not source | + +--- + +## Tasks + +### Task 1: Add ACP SDK dependency + smoke import + +**Files:** +- Modify: `package.json` +- Create: `tests/acp/sdk-smoke.test.ts` + +- [ ] **Step 1: Add the dependency** + +```bash +npm install --save-exact @agentclientprotocol/sdk@0.22.1 +``` + +- [ ] **Step 2: Write a smoke test asserting key SDK exports are importable** + +Create `tests/acp/sdk-smoke.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; + +test('@agentclientprotocol/sdk exports ClientSideConnection', async () => { + const mod = await import('@agentclientprotocol/sdk'); + assert.equal(typeof mod.ClientSideConnection, 'function'); + assert.equal(typeof mod.ndJsonStream, 'function'); + assert.equal(typeof mod.RequestError, 'function'); +}); +``` + +- [ ] **Step 3: Run the test** + +```bash +npx tsx --test tests/acp/sdk-smoke.test.ts +``` + +Expected: PASS + +- [ ] **Step 4: Run typecheck** + +```bash +npm run typecheck +``` + +Expected: no errors + +- [ ] **Step 5: Commit** + +```bash +git add package.json package-lock.json tests/acp/sdk-smoke.test.ts +git commit -m "feat(acp): add @agentclientprotocol/sdk dependency" +``` + +--- + +### Task 2: ACP transport over `docker exec -i` + +**Files:** +- Create: `src/workbench/acp/transport.ts` +- Create: `tests/acp/transport.test.ts` + +- [ ] **Step 1: Write the failing test (mock subprocess)** + +Create `tests/acp/transport.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { spawn } from 'node:child_process'; +import { createDockerExecStream } from '../../src/workbench/acp/transport.js'; + +test('createDockerExecStream produces a Stream that round-trips ndjson', async () => { + // Use `cat` as a stand-in for the agent: it echoes stdin to stdout. + const child = spawn('cat', [], { stdio: ['pipe', 'pipe', 'pipe'] }); + const stream = createDockerExecStream(child); + + const message = { jsonrpc: '2.0', id: 1, method: 'initialize', params: {} }; + const writer = stream.outgoing.getWriter(); + await writer.write(new TextEncoder().encode(JSON.stringify(message) + '\n')); + writer.releaseLock(); + + const reader = stream.incoming.getReader(); + const { value } = await reader.read(); + const echoed = new TextDecoder().decode(value).trim(); + assert.equal(echoed, JSON.stringify(message)); + + child.kill(); +}); +``` + +- [ ] **Step 2: Run test to verify it fails** + +```bash +npx tsx --test tests/acp/transport.test.ts +``` + +Expected: FAIL with "Cannot find module './transport.js'" or similar + +- [ ] **Step 3: Implement the transport** + +Create `src/workbench/acp/transport.ts`: + +```typescript +import type { ChildProcessWithoutNullStreams } from 'node:child_process'; +import type { Stream } from '@agentclientprotocol/sdk'; + +export interface DockerExecStream extends Stream { + outgoing: WritableStream; + incoming: ReadableStream; + close(): Promise; +} + +export function createDockerExecStream( + child: ChildProcessWithoutNullStreams, +): DockerExecStream { + const outgoing = new WritableStream({ + write(chunk) { + return new Promise((resolve, reject) => { + child.stdin.write(chunk, (err) => (err ? reject(err) : resolve())); + }); + }, + close() { + child.stdin.end(); + }, + }); + + const incoming = new ReadableStream({ + start(controller) { + child.stdout.on('data', (chunk: Buffer) => controller.enqueue(chunk)); + child.stdout.on('end', () => controller.close()); + child.stdout.on('error', (err) => controller.error(err)); + }, + }); + + return { + outgoing, + incoming, + async close() { + try { child.stdin.end(); } catch {} + child.kill(); + }, + }; +} +``` + +- [ ] **Step 4: Run test to verify it passes** + +```bash +npx tsx --test tests/acp/transport.test.ts +``` + +Expected: PASS + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/acp/transport.ts tests/acp/transport.test.ts +git commit -m "feat(acp): docker exec stdio bridge implementing Stream" +``` + +--- + +### Task 3: ACP client wrapper + +**Files:** +- Create: `src/workbench/acp/client.ts` +- Create: `tests/acp/client.test.ts` + +- [ ] **Step 1: Write the failing test** + +Create `tests/acp/client.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { spawn } from 'node:child_process'; +import { createWorkbenchClient } from '../../src/workbench/acp/client.js'; +import { createDockerExecStream } from '../../src/workbench/acp/transport.js'; + +test('createWorkbenchClient returns object with initialize/newSession/prompt/cancel', () => { + const child = spawn('cat', []); + const stream = createDockerExecStream(child); + const client = createWorkbenchClient({ stream, onSessionUpdate: () => {} }); + + assert.equal(typeof client.connection.initialize, 'function'); + assert.equal(typeof client.connection.newSession, 'function'); + assert.equal(typeof client.connection.prompt, 'function'); + assert.equal(typeof client.connection.cancel, 'function'); + assert.equal(typeof client.close, 'function'); + + child.kill(); +}); +``` + +- [ ] **Step 2: Run test to verify it fails** + +```bash +npx tsx --test tests/acp/client.test.ts +``` + +Expected: FAIL + +- [ ] **Step 3: Implement the wrapper** + +Create `src/workbench/acp/client.ts`: + +```typescript +import { ClientSideConnection, ndJsonStream } from '@agentclientprotocol/sdk'; +import type { + Client, + SessionNotification, + RequestPermissionRequest, + RequestPermissionResponse, +} from '@agentclientprotocol/sdk'; +import type { DockerExecStream } from './transport.js'; + +export interface WorkbenchClientOptions { + stream: DockerExecStream; + onSessionUpdate: (notification: SessionNotification) => void; +} + +export interface WorkbenchClient { + connection: ClientSideConnection; + close(): Promise; +} + +export function createWorkbenchClient(opts: WorkbenchClientOptions): WorkbenchClient { + // Note: ndJsonStream wraps the raw byte streams as JSON-RPC envelopes. + const ndjson = ndJsonStream(opts.stream.outgoing, opts.stream.incoming); + + const connection = new ClientSideConnection( + (_agent): Client => ({ + sessionUpdate: async (params) => { + opts.onSessionUpdate(params); + }, + // Auto-approve all tool calls. The container is the security boundary, + // not the permission gate. Approving everything matches the headless- + // bench model and matches benchflow's pattern. + requestPermission: async ( + params: RequestPermissionRequest, + ): Promise => ({ + outcome: { outcome: 'selected', optionId: params.options[0]?.optionId ?? 'allow' }, + }), + // We don't expose filesystem capabilities to the agent over ACP; the + // agent uses its own tools to touch /work directly. + }), + ndjson, + ); + + return { + connection, + async close() { + await opts.stream.close(); + }, + }; +} +``` + +- [ ] **Step 4: Run test to verify it passes** + +```bash +npx tsx --test tests/acp/client.test.ts && npm run typecheck +``` + +Expected: PASS + no type errors + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/acp/client.ts tests/acp/client.test.ts +git commit -m "feat(acp): client wrapper with auto-approve permission handler" +``` + +--- + +### Task 4: Agent registry types + +**Files:** +- Create: `src/workbench/agents/registry.ts` +- Create: `tests/agents/registry.test.ts` + +- [ ] **Step 1: Write the failing test** + +Create `tests/agents/registry.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { resolveAgent, AGENTS, AGENT_ALIASES } from '../../src/workbench/agents/registry.js'; + +test('AGENTS contains all 5 expected entries', () => { + for (const name of ['claude-agent-acp', 'codex-acp', 'gemini', 'opencode', 'pi-acp']) { + assert.ok(AGENTS[name], `missing agent: ${name}`); + assert.equal(AGENTS[name].name, name); + } +}); + +test('resolveAgent accepts aliases', () => { + assert.equal(resolveAgent('claude').name, 'claude-agent-acp'); + assert.equal(resolveAgent('codex').name, 'codex-acp'); + assert.equal(resolveAgent('pi').name, 'pi-acp'); +}); + +test('resolveAgent accepts canonical names', () => { + assert.equal(resolveAgent('claude-agent-acp').name, 'claude-agent-acp'); +}); + +test('resolveAgent throws with suggestion on unknown', () => { + assert.throws(() => resolveAgent('claud-agent'), /Did you mean/); +}); + +test('every agent declares at least one skillPath', () => { + for (const cfg of Object.values(AGENTS)) { + assert.ok(cfg.skillPaths.length > 0, `${cfg.name} missing skillPaths`); + assert.match(cfg.skillPaths[0], /^\$HOME\//); + } +}); + +test('AGENT_ALIASES does not collide with canonical names', () => { + for (const alias of Object.keys(AGENT_ALIASES)) { + if (AGENTS[alias]) { + assert.equal(alias, AGENT_ALIASES[alias], `alias ${alias} collides`); + } + } +}); +``` + +- [ ] **Step 2: Run test to verify it fails** + +```bash +npx tsx --test tests/agents/registry.test.ts +``` + +Expected: FAIL + +- [ ] **Step 3: Implement the registry types and resolver** + +Create `src/workbench/agents/registry.ts`: + +```typescript +export interface CredentialFile { + path: string; // Target path in container; may use {home} + envSource: string; // Env var on host to read value from + template?: string; // If set, value is inserted into template at {value} + mkdir?: boolean; // Create parent dir; default true +} + +export interface HostAuthFile { + hostPath: string; // ~/.claude/.credentials.json + containerPath: string; // {home}/.claude/.credentials.json +} + +export interface SubscriptionAuth { + replacesEnv: string; // e.g. "ANTHROPIC_API_KEY" + detectFile: string; // host path to check for login + files: HostAuthFile[]; // all files to copy when sub-auth used +} + +export type ApiProtocol = + | 'anthropic-messages' + | 'openai-completions' + | 'openai-responses' + | ''; + +export type AcpModelFormat = 'bare' | 'provider/model'; + +export interface AgentConfig { + name: string; + description: string; + installCmd: string; // bash, runs at IMAGE BUILD time + launchCmd: string; // bash, runs per-trial via docker exec + requiresEnv: string[]; + apiProtocol: ApiProtocol; + envMapping: Record; // SKILL_OPT_PROVIDER_* → agent-native + skillPaths: string[]; // e.g. ["$HOME/.claude/skills"] + credentialFiles: CredentialFile[]; + homeDirs: string[]; + subscriptionAuth: SubscriptionAuth | null; + acpModelFormat: AcpModelFormat; + supportsAcpSetModel: boolean; + loginHint?: string; // shown in fail-loud message +} + +export const AGENTS: Record = { + // Entries filled in Task 5. Empty here to make this task self-contained. +}; + +export const AGENT_ALIASES: Record = { + claude: 'claude-agent-acp', + codex: 'codex-acp', + gemini: 'gemini', + pi: 'pi-acp', + openclaw: 'openclaw', +}; + +export function resolveAgent(spec: string): AgentConfig { + const canonical = AGENT_ALIASES[spec] ?? spec; + const cfg = AGENTS[canonical]; + if (cfg) return cfg; + + const known = Object.keys(AGENTS); + const close = closestMatch(canonical, known); + if (close) { + throw new Error(`Unknown agent: ${spec!r}. Did you mean: ${close!r}?`.replace(/!r/g, '')); + } + throw new Error(`Unknown agent: ${spec}. Available: ${known.join(', ')}`); +} + +function closestMatch(needle: string, haystack: string[]): string | undefined { + // Simple shared-prefix heuristic; good enough for typo suggestion. + let best: { name: string; score: number } | undefined; + for (const name of haystack) { + const score = sharedPrefix(needle, name) + sharedPrefix(needle.split('').reverse().join(''), name.split('').reverse().join('')); + if (score > 4 && (!best || score > best.score)) { + best = { name, score }; + } + } + return best?.name; +} + +function sharedPrefix(a: string, b: string): number { + let i = 0; + while (i < a.length && i < b.length && a[i] === b[i]) i++; + return i; +} +``` + +- [ ] **Step 4: Run test to verify it fails differently (now: missing entries)** + +```bash +npx tsx --test tests/agents/registry.test.ts +``` + +Expected: FAIL on "missing agent" assertions (entries empty by design at this task) + +- [ ] **Step 5: Commit the skeleton** + +```bash +git add src/workbench/agents/registry.ts tests/agents/registry.test.ts +git commit -m "feat(agents): registry types + resolver (entries pending in task 5)" +``` + +--- + +### Task 5: Populate registry with 5 agent entries + +**Files:** +- Modify: `src/workbench/agents/registry.ts` + +These are direct ports of [benchflow's registry entries](../../../skillsbench/.venv/lib/python3.12/site-packages/benchflow/agents/registry.py). +The install snippets become `BUILD-TIME` only — no `command -v ... ||` +guards needed since the Dockerfile runs them once. + +- [ ] **Step 1: Add the `claude-agent-acp` entry** + +In `src/workbench/agents/registry.ts`, replace the empty `AGENTS = {}` +with: + +```typescript +export const AGENTS: Record = { + 'claude-agent-acp': { + name: 'claude-agent-acp', + description: 'Claude Code via ACP (Anthropic CLI)', + installCmd: `npm install -g @zed-industries/claude-agent-acp@latest`, + launchCmd: `claude-agent-acp`, + requiresEnv: ['ANTHROPIC_API_KEY'], + apiProtocol: 'anthropic-messages', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'ANTHROPIC_BASE_URL', + SKILL_OPT_PROVIDER_API_KEY: 'ANTHROPIC_AUTH_TOKEN', + SKILL_OPT_PROVIDER_MODEL: 'ANTHROPIC_MODEL', + }, + skillPaths: ['$HOME/.claude/skills'], + credentialFiles: [], + homeDirs: [], + subscriptionAuth: { + replacesEnv: 'ANTHROPIC_API_KEY', + detectFile: '~/.claude/.credentials.json', + files: [ + { hostPath: '~/.claude/.credentials.json', containerPath: '{home}/.claude/.credentials.json' }, + ], + }, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'claude login', + }, +}; +``` + +- [ ] **Step 2: Add the `codex-acp` entry** + +Append to the `AGENTS` object: + +```typescript + 'codex-acp': { + name: 'codex-acp', + description: 'OpenAI Codex via ACP', + installCmd: `npm install -g @zed-industries/codex-acp@latest`, + launchCmd: `codex-acp \${OPENAI_BASE_URL:+-c openai_base_url=$OPENAI_BASE_URL}`, + requiresEnv: ['OPENAI_API_KEY'], + apiProtocol: 'openai-responses', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'OPENAI_BASE_URL', + SKILL_OPT_PROVIDER_API_KEY: 'OPENAI_API_KEY', + }, + skillPaths: ['$HOME/.agents/skills'], + credentialFiles: [ + { + path: '{home}/.codex/auth.json', + envSource: 'OPENAI_API_KEY', + template: '{"OPENAI_API_KEY": "{value}"}', + }, + ], + homeDirs: [], + subscriptionAuth: { + replacesEnv: 'OPENAI_API_KEY', + detectFile: '~/.codex/auth.json', + files: [ + { hostPath: '~/.codex/auth.json', containerPath: '{home}/.codex/auth.json' }, + ], + }, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'codex login', + }, +``` + +- [ ] **Step 3: Add the `gemini` entry** + +Append to the `AGENTS` object: + +```typescript + 'gemini': { + name: 'gemini', + description: 'Google Gemini CLI via ACP', + installCmd: `npm install -g @google/gemini-cli@latest`, + launchCmd: `gemini --acp --yolo`, + requiresEnv: ['GOOGLE_API_KEY'], + apiProtocol: '', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'GEMINI_API_BASE_URL', + SKILL_OPT_PROVIDER_API_KEY: 'GOOGLE_API_KEY', + }, + skillPaths: ['$HOME/.gemini/skills'], + credentialFiles: [], + homeDirs: [], + subscriptionAuth: { + replacesEnv: 'GEMINI_API_KEY', + detectFile: '~/.gemini/oauth_creds.json', + files: [ + { hostPath: '~/.gemini/oauth_creds.json', containerPath: '{home}/.gemini/oauth_creds.json' }, + { hostPath: '~/.gemini/settings.json', containerPath: '{home}/.gemini/settings.json' }, + { hostPath: '~/.gemini/google_accounts.json', containerPath: '{home}/.gemini/google_accounts.json' }, + ], + }, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'gemini auth login', + }, +``` + +- [ ] **Step 4: Add the `opencode` entry** + +Append to the `AGENTS` object: + +```typescript + 'opencode': { + name: 'opencode', + description: 'OpenCode via ACP — open-source coding agent', + installCmd: `npm install -g opencode-ai@latest`, + launchCmd: `opencode acp`, + requiresEnv: [], + apiProtocol: '', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'OPENAI_BASE_URL', + }, + skillPaths: ['$HOME/.opencode/skills'], + credentialFiles: [], + homeDirs: ['.opencode'], + subscriptionAuth: null, + acpModelFormat: 'provider/model', + supportsAcpSetModel: true, + loginHint: 'set OPENAI_API_KEY (or provider-specific key)', + }, +``` + +- [ ] **Step 5: Add the `pi-acp` entry** + +Append to the `AGENTS` object. The pi-acp launcher needs a wrapper +that bridges `SKILL_OPT_PROVIDER_*` env vars into pi's config; we +deploy this wrapper script in Task 6 (the Dockerfile). + +```typescript + 'pi-acp': { + name: 'pi-acp', + description: 'Pi coding agent via ACP', + installCmd: `npm install -g @mariozechner/pi-coding-agent@latest pi-acp@latest`, + launchCmd: `/opt/skill-opt/bin/pi-acp-launcher`, + requiresEnv: [], + apiProtocol: '', + envMapping: {}, + skillPaths: ['$HOME/.pi/agent/skills', '$HOME/.agents/skills'], + credentialFiles: [], + homeDirs: ['.pi'], + subscriptionAuth: null, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'set OPENROUTER_API_KEY', + }, +``` + +- [ ] **Step 6: Run the registry tests** + +```bash +npx tsx --test tests/agents/registry.test.ts +``` + +Expected: all PASS + +- [ ] **Step 7: Run typecheck** + +```bash +npm run typecheck +``` + +Expected: no errors + +- [ ] **Step 8: Commit** + +```bash +git add src/workbench/agents/registry.ts +git commit -m "feat(agents): populate 5-agent registry (claude/codex/gemini/opencode/pi-acp)" +``` + +--- + +### Task 6: Build the new Dockerfile + +**Files:** +- Create: `docker/skill-optimizer-agent.Dockerfile` +- Create: `src/workbench/agents/install-snippets.ts` +- Create: `docker/pi-acp-launcher.sh` + +- [ ] **Step 1: Create the pi-acp launcher script** + +Create `docker/pi-acp-launcher.sh`: + +```bash +#!/bin/sh +# Bridges SKILL_OPT_PROVIDER_* env vars to pi-acp's expected config. +# pi-acp reads its model/provider from a runtime config; we set +# OPENROUTER_API_KEY here if SKILL_OPT_PROVIDER_API_KEY is set. +set -e + +if [ -n "$SKILL_OPT_PROVIDER_API_KEY" ] && [ -z "$OPENROUTER_API_KEY" ]; then + export OPENROUTER_API_KEY="$SKILL_OPT_PROVIDER_API_KEY" +fi + +exec pi-acp "$@" +``` + +- [ ] **Step 2: Generate install snippets from the registry** + +Create `src/workbench/agents/install-snippets.ts`: + +```typescript +import { AGENTS } from './registry.js'; + +export function generateDockerInstallBlock(): string { + const lines: string[] = []; + for (const cfg of Object.values(AGENTS)) { + lines.push(`# Install ${cfg.name}: ${cfg.description}`); + lines.push(`RUN ${cfg.installCmd}`); + lines.push(''); + } + return lines.join('\n'); +} + +// Used at Dockerfile generation time; printed by `npm run dockerfile:print`. +if (import.meta.url === `file://${process.argv[1]}`) { + console.log(generateDockerInstallBlock()); +} +``` + +- [ ] **Step 3: Add npm script to print install block** + +Edit `package.json`, add to `scripts`: + +```json +"dockerfile:print-installs": "tsx src/workbench/agents/install-snippets.ts" +``` + +- [ ] **Step 4: Write the new Dockerfile** + +Create `docker/skill-optimizer-agent.Dockerfile`: + +```dockerfile +FROM node:22-bookworm + +ENV PATH="/opt/skill-opt/bin:/app/node_modules/.bin:/work/.venv/bin:${PATH}" \ + PIP_REQUIRE_VIRTUALENV=1 + +WORKDIR /app + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + bash \ + ca-certificates \ + coreutils \ + curl \ + file \ + findutils \ + gawk \ + git \ + grep \ + jq \ + less \ + python-is-python3 \ + python3 \ + python3-pip \ + python3-venv \ + ripgrep \ + sed \ + unzip \ + wget \ + zip \ + && rm -rf /var/lib/apt/lists/* + +# --- Agent CLI install layer (cached as one layer for build speed) --- +# Each agent's installCmd is also kept in src/workbench/agents/registry.ts +# (single source of truth). If you change one here, change it there too. +RUN npm install -g \ + @zed-industries/claude-agent-acp@latest \ + @zed-industries/codex-acp@latest \ + @google/gemini-cli@latest \ + opencode-ai@latest \ + @mariozechner/pi-coding-agent@latest \ + pi-acp@latest + +# --- pi-acp launcher wrapper --- +COPY docker/pi-acp-launcher.sh /opt/skill-opt/bin/pi-acp-launcher +RUN chmod +x /opt/skill-opt/bin/pi-acp-launcher + +# --- Workbench container-runner (setup + grade modes only) --- +COPY package.json package-lock.json tsconfig.json ./ +COPY src ./src +COPY docs ./docs + +RUN npm ci \ + && npm run build \ + && useradd -m -u 10001 agent +USER agent + +# Container-runner is now used only for --setup and --grade modes. +# Agent dispatch happens host-side via ACP; the agent CLIs above are +# invoked directly by docker exec. +ENTRYPOINT ["node", "/app/dist/workbench/container-runner.js"] +``` + +- [ ] **Step 5: Build the image** + +```bash +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . +``` + +Expected: successful build; final image ~600MB - ~1GB + +- [ ] **Step 6: Smoke-test that each agent binary is reachable** + +```bash +for agent in claude-agent-acp codex-acp gemini opencode pi-acp; do + echo "=== $agent ===" + docker run --rm --entrypoint sh skill-optimizer-agent:local -lc "command -v $agent && $agent --version 2>&1 | head -2" +done +``` + +Expected: each agent prints a version line (or a usage line that confirms the binary exists). Failure of any agent means the install snippet is wrong; fix at registry.ts AND the Dockerfile RUN line. + +- [ ] **Step 7: Delete the old Dockerfile** + +```bash +git rm docker/workbench-runner.Dockerfile +``` + +- [ ] **Step 8: Update the project's docker:build reference** + +Search and replace `workbench-runner.Dockerfile` and `skill-optimizer-workbench:local`: + +```bash +grep -rln "workbench-runner.Dockerfile\|skill-optimizer-workbench:local" src/ CLAUDE.md README.md CONTRIBUTING.md +``` + +For each file in the result, update the references to `skill-optimizer-agent.Dockerfile` and `skill-optimizer-agent:local`. + +Files known to mention these: +- `src/workbench/docker-runner.ts` (constant `DEFAULT_WORKBENCH_IMAGE`) +- `CLAUDE.md` (test guidance) +- `README.md` (install docs) +- `CONTRIBUTING.md` + +- [ ] **Step 9: Run typecheck** + +```bash +npm run typecheck +``` + +- [ ] **Step 10: Commit** + +```bash +git add docker/ package.json src/workbench/agents/install-snippets.ts \ + src/workbench/docker-runner.ts CLAUDE.md README.md CONTRIBUTING.md +git commit -m "feat(docker): new image with all 5 agents pre-baked" +``` + +--- + +### Task 7: Auth resolution module + +**Files:** +- Create: `src/workbench/acp/auth.ts` +- Create: `tests/acp/auth.test.ts` + +- [ ] **Step 1: Write the failing tests** + +Create `tests/acp/auth.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, writeFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { resolveAuth } from '../../src/workbench/acp/auth.js'; +import { AGENTS } from '../../src/workbench/agents/registry.js'; + +test('resolveAuth uses subscription file when present', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + writeFileSync(join(home, '.credentials.json'), '{}'); + + const auth = resolveAuth(AGENTS['claude-agent-acp'], { + home, + env: {}, + }); + + assert.equal(auth.mode, 'subscription'); + assert.equal(auth.files.length, 1); + assert.equal(auth.files[0].hostPath, join(home, '.credentials.json')); + + rmSync(home, { recursive: true }); +}); + +test('resolveAuth falls back to env API key when subscription absent', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + + const auth = resolveAuth(AGENTS['claude-agent-acp'], { + home, + env: { ANTHROPIC_API_KEY: 'sk-test-key' }, + }); + + assert.equal(auth.mode, 'env'); + assert.deepEqual(auth.envNames, ['ANTHROPIC_API_KEY']); + + rmSync(home, { recursive: true }); +}); + +test('resolveAuth throws when neither subscription nor env key present', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + + assert.throws( + () => resolveAuth(AGENTS['claude-agent-acp'], { home, env: {} }), + /claude login|ANTHROPIC_API_KEY/, + ); + + rmSync(home, { recursive: true }); +}); + +test('resolveAuth returns env mode for agents without subscriptionAuth', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + + const auth = resolveAuth(AGENTS['pi-acp'], { + home, + env: { OPENROUTER_API_KEY: 'sk-test' }, + }); + + assert.equal(auth.mode, 'env'); + + rmSync(home, { recursive: true }); +}); +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +npx tsx --test tests/acp/auth.test.ts +``` + +Expected: FAIL (module missing) + +- [ ] **Step 3: Implement auth resolution** + +Create `src/workbench/acp/auth.ts`: + +```typescript +import { existsSync } from 'node:fs'; +import { homedir } from 'node:os'; +import { join, normalize } from 'node:path'; +import type { AgentConfig, HostAuthFile } from '../agents/registry.js'; + +export interface AuthContext { + home?: string; // override $HOME for testing + env: Record; +} + +export interface SubscriptionAuthResult { + mode: 'subscription'; + files: Array<{ hostPath: string; containerPath: string }>; +} + +export interface EnvAuthResult { + mode: 'env'; + envNames: string[]; +} + +export type AuthResult = SubscriptionAuthResult | EnvAuthResult; + +export function resolveAuth(agent: AgentConfig, ctx: AuthContext): AuthResult { + const home = ctx.home ?? homedir(); + + if (agent.subscriptionAuth) { + const detectPath = expandHome(agent.subscriptionAuth.detectFile, home); + if (existsSync(detectPath)) { + return { + mode: 'subscription', + files: agent.subscriptionAuth.files.map((f) => ({ + hostPath: expandHome(f.hostPath, home), + containerPath: f.containerPath, // {home} placeholder, resolved at mount time + })), + }; + } + } + + const missing = agent.requiresEnv.filter((name) => !ctx.env[name]); + if (missing.length > 0 && agent.requiresEnv.length > 0) { + const hint = agent.loginHint ? ` (run \`${agent.loginHint}\` or set ${missing.join(', ')})` : ''; + throw new Error( + `Agent ${agent.name} requires auth but none is available${hint}. Missing: ${missing.join(', ')}.`, + ); + } + + return { mode: 'env', envNames: agent.requiresEnv }; +} + +function expandHome(p: string, home: string): string { + if (p.startsWith('~/')) return normalize(join(home, p.slice(2))); + if (p === '~') return home; + return p; +} + +export function resolveContainerPath(template: string, agentHome: string): string { + return template.replace('{home}', agentHome); +} +``` + +- [ ] **Step 4: Run tests to verify they pass** + +```bash +npx tsx --test tests/acp/auth.test.ts && npm run typecheck +``` + +Expected: all PASS, no type errors + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/acp/auth.ts tests/acp/auth.test.ts +git commit -m "feat(acp): subscription-first auth resolution with fail-loud env fallback" +``` + +--- + +### Task 8: Skill deployment module + +**Files:** +- Create: `src/workbench/acp/skill-deploy.ts` +- Create: `tests/acp/skill-deploy.test.ts` + +- [ ] **Step 1: Write the failing test** + +Create `tests/acp/skill-deploy.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { computeSkillMount } from '../../src/workbench/acp/skill-deploy.js'; +import { AGENTS } from '../../src/workbench/agents/registry.js'; + +test('computeSkillMount returns native path for claude-agent-acp', () => { + const mount = computeSkillMount({ + agent: AGENTS['claude-agent-acp'], + skillSlug: 'web-design-guidelines', + hostSkillDir: '/host/path/to/skill', + agentHome: '/home/agent', + }); + + assert.equal(mount.hostPath, '/host/path/to/skill'); + assert.equal(mount.containerPath, '/home/agent/.claude/skills/web-design-guidelines'); + assert.equal(mount.readOnly, true); +}); + +test('computeSkillMount handles agents whose skill_paths use $HOME', () => { + for (const cfg of Object.values(AGENTS)) { + const mount = computeSkillMount({ + agent: cfg, + skillSlug: 'test-skill', + hostSkillDir: '/host/dir', + agentHome: '/home/agent', + }); + assert.ok(!mount.containerPath.includes('$HOME')); + assert.ok(mount.containerPath.startsWith('/home/agent/')); + assert.ok(mount.containerPath.endsWith('/test-skill')); + } +}); +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +npx tsx --test tests/acp/skill-deploy.test.ts +``` + +- [ ] **Step 3: Implement skill-deploy** + +Create `src/workbench/acp/skill-deploy.ts`: + +```typescript +import { join } from 'node:path'; +import type { AgentConfig } from '../agents/registry.js'; + +export interface SkillMount { + hostPath: string; + containerPath: string; + readOnly: boolean; +} + +export interface ComputeSkillMountParams { + agent: AgentConfig; + skillSlug: string; + hostSkillDir: string; // absolute path on host to the skill folder + agentHome: string; // /home/agent (or whatever the container user's $HOME is) +} + +export function computeSkillMount(params: ComputeSkillMountParams): SkillMount { + // Use the first skillPath (primary discovery location for the agent). + const skillRoot = params.agent.skillPaths[0]; + if (!skillRoot) { + throw new Error(`Agent ${params.agent.name} has no skillPaths configured`); + } + const expanded = skillRoot.replace('$HOME', params.agentHome); + return { + hostPath: params.hostSkillDir, + containerPath: join(expanded, params.skillSlug), + readOnly: true, + }; +} + +export function dockerMountFlag(mount: SkillMount): string { + const ro = mount.readOnly ? ':ro' : ':rw'; + return `-v ${shellQuote(mount.hostPath)}:${shellQuote(mount.containerPath)}${ro}`; +} + +function shellQuote(s: string): string { + return `'${s.replace(/'/g, `'\\''`)}'`; +} +``` + +- [ ] **Step 4: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/acp/skill-deploy.test.ts && npm run typecheck +``` + +Expected: PASS + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/acp/skill-deploy.ts tests/acp/skill-deploy.test.ts +git commit -m "feat(acp): skill-deploy module mounts to agent's native skill path" +``` + +--- + +### Task 9: MCP config writer + +**Files:** +- Create: `src/workbench/acp/mcp-config-writer.ts` +- Create: `tests/acp/mcp-config-writer.test.ts` +- Create: `tests/fixtures/mcp-cases/sample-stdio.yml` + +Per the spec's note [2], pi-acp's MCP target is TBD-at-implementation. +This task implements the four known agents (claude/codex/gemini/opencode); +pi-acp falls back to a "not yet supported" error. Resolution for pi-acp +will be in Task 9b. + +- [ ] **Step 1: Create the fixture case** + +Create `tests/fixtures/mcp-cases/sample-stdio.yml`: + +```yaml +name: sample-with-mcp +agent: claude-agent-acp +model: claude-haiku-4-5 +task: do nothing +graders: + - name: noop + command: 'true' +mcpServers: + calculator: + command: node + args: ['/work/mcp/calculator.mjs'] +``` + +- [ ] **Step 2: Write the failing test** + +Create `tests/acp/mcp-config-writer.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, readFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { writeMcpConfig } from '../../src/workbench/acp/mcp-config-writer.js'; +import { AGENTS } from '../../src/workbench/agents/registry.js'; + +const sampleCase = { + mcpServers: { + calculator: { command: 'node', args: ['/work/mcp/calculator.mjs'] }, + }, +}; + +test('writeMcpConfig for claude writes to .claude.json mcpServers', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['claude-agent-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const data = JSON.parse(readFileSync(join(home, '.claude.json'), 'utf-8')); + assert.ok(data.mcpServers?.calculator); + assert.equal(data.mcpServers.calculator.command, 'node'); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for codex writes to .codex/config.toml', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['codex-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const toml = readFileSync(join(home, '.codex/config.toml'), 'utf-8'); + assert.match(toml, /\[mcp_servers\.calculator\]/); + assert.match(toml, /command = "node"/); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for gemini writes to .gemini/settings.json mcpServers', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['gemini'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const data = JSON.parse(readFileSync(join(home, '.gemini/settings.json'), 'utf-8')); + assert.ok(data.mcpServers?.calculator); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for opencode writes to .config/opencode/opencode.json mcp', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['opencode'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const data = JSON.parse(readFileSync(join(home, '.config/opencode/opencode.json'), 'utf-8')); + assert.ok(data.mcp?.calculator); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for pi-acp throws "not yet supported" (pending Task 9b)', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + assert.throws( + () => writeMcpConfig({ + agent: AGENTS['pi-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }), + /pi-acp MCP support pending/, + ); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig is a no-op when caseConfig has no mcpServers', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['claude-agent-acp'], + caseConfig: {}, + agentHomeOnHost: home, + }); + // No files should be written. + rmSync(home, { recursive: true }); +}); +``` + +- [ ] **Step 3: Run tests to verify they fail** + +```bash +npx tsx --test tests/acp/mcp-config-writer.test.ts +``` + +- [ ] **Step 4: Implement the writer** + +Create `src/workbench/acp/mcp-config-writer.ts`: + +```typescript +import { mkdirSync, readFileSync, writeFileSync, existsSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import type { AgentConfig } from '../agents/registry.js'; + +export interface McpServerSpec { + command?: string; + args?: string[]; + env?: Record; + url?: string; + headers?: Record; +} + +export interface WriteMcpConfigParams { + agent: AgentConfig; + caseConfig: { mcpServers?: Record }; + agentHomeOnHost: string; // path on host that will become container's $HOME +} + +export function writeMcpConfig(params: WriteMcpConfigParams): void { + const servers = params.caseConfig.mcpServers; + if (!servers || Object.keys(servers).length === 0) return; + + switch (params.agent.name) { + case 'claude-agent-acp': + return writeClaude(servers, params.agentHomeOnHost); + case 'codex-acp': + return writeCodex(servers, params.agentHomeOnHost); + case 'gemini': + return writeGemini(servers, params.agentHomeOnHost); + case 'opencode': + return writeOpencode(servers, params.agentHomeOnHost); + case 'pi-acp': + throw new Error('pi-acp MCP support pending — see plan Task 9b'); + default: + throw new Error(`MCP not implemented for agent ${params.agent.name}`); + } +} + +function writeClaude(servers: Record, home: string): void { + const path = join(home, '.claude.json'); + mkdirSync(dirname(path), { recursive: true }); + const existing = existsSync(path) ? JSON.parse(readFileSync(path, 'utf-8')) : {}; + existing.mcpServers = { ...(existing.mcpServers ?? {}), ...servers }; + writeFileSync(path, JSON.stringify(existing, null, 2)); +} + +function writeCodex(servers: Record, home: string): void { + const path = join(home, '.codex/config.toml'); + mkdirSync(dirname(path), { recursive: true }); + const blocks: string[] = []; + for (const [name, spec] of Object.entries(servers)) { + blocks.push(`[mcp_servers.${name}]`); + if (spec.command) blocks.push(`command = ${JSON.stringify(spec.command)}`); + if (spec.args) blocks.push(`args = ${JSON.stringify(spec.args)}`); + if (spec.env) { + blocks.push(`[mcp_servers.${name}.env]`); + for (const [k, v] of Object.entries(spec.env)) { + blocks.push(`${k} = ${JSON.stringify(v)}`); + } + } + blocks.push(''); + } + writeFileSync(path, blocks.join('\n')); +} + +function writeGemini(servers: Record, home: string): void { + const path = join(home, '.gemini/settings.json'); + mkdirSync(dirname(path), { recursive: true }); + const existing = existsSync(path) ? JSON.parse(readFileSync(path, 'utf-8')) : {}; + existing.mcpServers = { ...(existing.mcpServers ?? {}), ...servers }; + writeFileSync(path, JSON.stringify(existing, null, 2)); +} + +function writeOpencode(servers: Record, home: string): void { + const path = join(home, '.config/opencode/opencode.json'); + mkdirSync(dirname(path), { recursive: true }); + const existing = existsSync(path) ? JSON.parse(readFileSync(path, 'utf-8')) : {}; + existing.mcp = { ...(existing.mcp ?? {}), ...servers }; + writeFileSync(path, JSON.stringify(existing, null, 2)); +} +``` + +- [ ] **Step 5: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/acp/mcp-config-writer.test.ts && npm run typecheck +``` + +Expected: all PASS + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/acp/mcp-config-writer.ts tests/acp/mcp-config-writer.test.ts tests/fixtures/mcp-cases/ +git commit -m "feat(acp): per-agent MCP config writer (claude/codex/gemini/opencode)" +``` + +--- + +### Task 9b: Resolve pi-acp MCP target + +**Files:** +- Modify: `src/workbench/acp/mcp-config-writer.ts` +- Modify: `tests/acp/mcp-config-writer.test.ts` + +- [ ] **Step 1: Inspect pi-acp source to determine native MCP target** + +Run: + +```bash +docker run --rm --entrypoint sh skill-optimizer-agent:local -lc "find / -name 'pi-acp' -type f 2>/dev/null | head -5; npm root -g" +docker run --rm --entrypoint sh skill-optimizer-agent:local -lc "cat \$(npm root -g)/pi-acp/package.json | head -20" +docker run --rm --entrypoint sh skill-optimizer-agent:local -lc "find \$(npm root -g)/pi-acp -name '*.js' | xargs grep -l 'mcp' | head" +``` + +Decision criteria (pick one): +- **(a) pi-acp has a native MCP config path** (e.g., `~/.pi/agent/mcp.json`) → implement a `writePiAcp` function like the other agents +- **(b) pi-acp uses mcporter via env** (e.g., reads `MCPORTER_CONFIG` env at launch) → write `mcporter.json` to a known path and inject `MCPORTER_CONFIG` into the launch env +- **(c) pi-acp has no MCP support** → keep the "not yet supported" throw; document the limitation in `examples/workbench/mcp/README.md` + +- [ ] **Step 2: Document the decision in a comment block** + +Add to top of `src/workbench/acp/mcp-config-writer.ts`: + +```typescript +// pi-acp MCP resolution (from plan Task 9b investigation on 2026-XX-XX): +// Decision: +// Evidence: +// Implementation: +``` + +- [ ] **Step 3: Implement the chosen option** + +For option (a) — add to `writeMcpConfig` switch: + +```typescript +case 'pi-acp': + return writePiAcp(servers, params.agentHomeOnHost); +``` + +And add the `writePiAcp` function appropriate to the path found. + +For option (b) — write `mcporter.json` to the discovered path and return additional launch env (this requires extending the function's return type to optionally carry env additions). + +For option (c) — keep the throw, add a regression test that confirms the case-loader rejects `mcpServers:` for pi-acp at load time. + +- [ ] **Step 4: Update the pi-acp test in `mcp-config-writer.test.ts`** + +Replace the "pending" test with one that asserts the implemented behavior. Example for option (a): + +```typescript +test('writeMcpConfig for pi-acp writes to pi-agent native MCP config', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['pi-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + // Assert based on whichever path was discovered. + const path = join(home, '.pi/agent/mcp.json'); // adjust per evidence + assert.ok(existsSync(path)); + rmSync(home, { recursive: true }); +}); +``` + +- [ ] **Step 5: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/acp/mcp-config-writer.test.ts && npm run typecheck +``` + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/acp/mcp-config-writer.ts tests/acp/mcp-config-writer.test.ts +git commit -m "feat(acp): resolve pi-acp MCP target — option " +``` + +--- + +### Task 10: parse-trace.ts helpers + +**Files:** +- Create: `src/workbench/parse-trace.ts` +- Create: `tests/parse-trace.test.ts` +- Create: `tests/fixtures/acp-traces/claude-sample.jsonl` + +- [ ] **Step 1: Create the fixture trace** + +Create `tests/fixtures/acp-traces/claude-sample.jsonl`. Each line is +the raw ACP JSON-RPC envelope as captured by the trace-recorder: + +```jsonl +{"type":"trace_start","schemaVersion":2,"caseName":"sample","agent":"claude-agent-acp","model":"claude-haiku-4-5","startedAt":"2026-05-25T00:00:00Z"} +{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"0.22","clientCapabilities":{}}} +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":"0.22","agentCapabilities":{},"authMethods":[]}} +{"jsonrpc":"2.0","id":2,"method":"session/new","params":{"cwd":"/work","mcpServers":[]}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"s-1"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"Thinking about it..."}}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"I'll write the file now."}}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"tool_call","toolCallId":"tc-1","title":"Write file","kind":"edit","status":"in_progress","content":[{"type":"content","content":{"type":"text","text":"out.txt"}}]}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"tool_call_update","toolCallId":"tc-1","status":"completed","content":[{"type":"content","content":{"type":"text","text":"Wrote 6 bytes"}}]}}} +{"jsonrpc":"2.0","id":3,"method":"session/prompt","params":{"sessionId":"s-1","prompt":[{"type":"text","text":"Write hello to out.txt"}]}} +{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn","usage":{"inputTokens":120,"outputTokens":45,"cacheReadTokens":0,"cacheCreationTokens":0}}} +``` + +- [ ] **Step 2: Write the failing test** + +Create `tests/parse-trace.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { readFileSync } from 'node:fs'; +import { + iterMessages, + iterToolCalls, + computeMetrics, + getFinalAssistantMessage, +} from '../src/workbench/parse-trace.js'; + +const fixturePath = 'tests/fixtures/acp-traces/claude-sample.jsonl'; +const trace = readFileSync(fixturePath, 'utf-8'); + +test('iterMessages yields assistant text + thinking', () => { + const msgs = [...iterMessages(trace)]; + const assistant = msgs.find((m) => m.role === 'assistant'); + assert.ok(assistant); + assert.match(assistant.text ?? '', /write the file/); + assert.match(assistant.thinking ?? '', /Thinking about it/); +}); + +test('iterToolCalls pairs tool_call and tool_call_update', () => { + const calls = [...iterToolCalls(trace)]; + assert.equal(calls.length, 1); + assert.equal(calls[0].id, 'tc-1'); + assert.equal(calls[0].kind, 'edit'); + assert.equal(calls[0].status, 'completed'); + assert.match(calls[0].resultText ?? '', /Wrote 6 bytes/); +}); + +test('computeMetrics returns tokens, duration, tool counts', () => { + const m = computeMetrics(trace); + assert.equal(m.tokens.input, 120); + assert.equal(m.tokens.output, 45); + assert.equal(m.tokens.total, 165); + assert.equal(m.toolCalls, 1); + assert.equal(m.editCalls, 1); + assert.equal(m.bashCalls, 0); + assert.equal(m.stopReason, 'end_turn'); +}); + +test('getFinalAssistantMessage returns concatenated assistant text', () => { + const final = getFinalAssistantMessage(trace); + assert.match(final ?? '', /write the file/); +}); +``` + +- [ ] **Step 3: Run tests to verify they fail** + +```bash +npx tsx --test tests/parse-trace.test.ts +``` + +- [ ] **Step 4: Implement parse-trace** + +Create `src/workbench/parse-trace.ts`: + +```typescript +export interface ParsedMessage { + role: 'user' | 'assistant'; + text?: string; + thinking?: string; + timestamp?: string; +} + +export interface ParsedToolCall { + id: string; + kind: string; // execute, read, write, edit, search, fetch, think, other + title?: string; + status: 'in_progress' | 'completed' | 'failed' | 'pending'; + argsText?: string; + resultText?: string; + isError?: boolean; +} + +export interface ComputedMetrics { + durationMs: number; + turns: number; + toolCalls: number; + bashCalls: number; + readCalls: number; + writeCalls: number; + editCalls: number; + stopReason?: string; + tokens: { input: number; output: number; cacheRead: number; cacheWrite: number; total: number }; +} + +interface TraceHeader { + type: 'trace_start'; + caseName?: string; + agent?: string; + model?: string; + startedAt?: string; + endedAt?: string; +} + +function parseLines(jsonl: string): unknown[] { + return jsonl.split(/\r?\n/).filter(Boolean).map((line) => { + try { return JSON.parse(line); } catch { return null; } + }).filter((x) => x !== null); +} + +function getHeader(rows: unknown[]): TraceHeader | undefined { + for (const row of rows) { + if (typeof row === 'object' && row !== null && (row as any).type === 'trace_start') { + return row as TraceHeader; + } + } + return undefined; +} + +function extractText(content: unknown): string | undefined { + if (!content || typeof content !== 'object') return undefined; + const c = content as any; + if (typeof c.text === 'string') return c.text; + if (c.content && typeof c.content.text === 'string') return c.content.text; + if (Array.isArray(c)) { + return c.map((part: any) => extractText(part)).filter(Boolean).join('\n') || undefined; + } + return undefined; +} + +export function* iterMessages(jsonl: string): IterableIterator { + const rows = parseLines(jsonl); + // Buffer per role; ACP streams assistant text in chunks. + let assistantText = ''; + let assistantThinking = ''; + + for (const row of rows) { + const r = row as any; + if (r.method !== 'session/update') continue; + const update = r.params?.update; + if (!update) continue; + + switch (update.sessionUpdate) { + case 'agent_message_chunk': + assistantText += extractText(update.content) ?? ''; + break; + case 'agent_thought_chunk': + assistantThinking += extractText(update.content) ?? ''; + break; + case 'user_message_chunk': + yield { role: 'user', text: extractText(update.content) }; + break; + } + } + + if (assistantText || assistantThinking) { + yield { + role: 'assistant', + text: assistantText || undefined, + thinking: assistantThinking || undefined, + }; + } +} + +export function* iterToolCalls(jsonl: string): IterableIterator { + const rows = parseLines(jsonl); + const calls = new Map(); + + for (const row of rows) { + const r = row as any; + if (r.method !== 'session/update') continue; + const update = r.params?.update; + if (!update) continue; + + if (update.sessionUpdate === 'tool_call') { + calls.set(update.toolCallId, { + id: update.toolCallId, + kind: update.kind ?? 'other', + title: update.title, + status: update.status ?? 'in_progress', + argsText: extractText(update.content), + }); + } else if (update.sessionUpdate === 'tool_call_update') { + const existing = calls.get(update.toolCallId); + if (existing) { + existing.status = update.status ?? existing.status; + if (update.content) existing.resultText = extractText(update.content); + if (update.status === 'failed') existing.isError = true; + } + } + } + + yield* calls.values(); +} + +export function computeMetrics(jsonl: string): ComputedMetrics { + const rows = parseLines(jsonl); + const header = getHeader(rows); + + const m: ComputedMetrics = { + durationMs: 0, + turns: 0, + toolCalls: 0, + bashCalls: 0, + readCalls: 0, + writeCalls: 0, + editCalls: 0, + tokens: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; + + if (header?.startedAt) { + const start = Date.parse(header.startedAt); + const end = header.endedAt ? Date.parse(header.endedAt) : Date.now(); + if (Number.isFinite(start) && Number.isFinite(end)) { + m.durationMs = Math.max(0, end - start); + } + } + + for (const row of rows) { + const r = row as any; + if (r.method === 'session/update') { + const update = r.params?.update; + if (!update) continue; + if (update.sessionUpdate === 'agent_message_chunk') m.turns += 1; + if (update.sessionUpdate === 'tool_call') { + m.toolCalls += 1; + const kind = update.kind ?? 'other'; + if (kind === 'execute') m.bashCalls += 1; + if (kind === 'read') m.readCalls += 1; + if (kind === 'write') m.writeCalls += 1; + if (kind === 'edit') m.editCalls += 1; + } + } + // session/prompt response carries final usage + stopReason + if (r.id && r.result?.stopReason) { + m.stopReason = r.result.stopReason; + const u = r.result.usage; + if (u) { + m.tokens.input += Number(u.inputTokens ?? 0); + m.tokens.output += Number(u.outputTokens ?? 0); + m.tokens.cacheRead += Number(u.cacheReadTokens ?? 0); + m.tokens.cacheWrite += Number(u.cacheCreationTokens ?? 0); + } + } + } + + m.tokens.total = m.tokens.input + m.tokens.output + m.tokens.cacheRead + m.tokens.cacheWrite; + return m; +} + +export function getFinalAssistantMessage(jsonl: string): string | undefined { + for (const msg of iterMessages(jsonl)) { + if (msg.role === 'assistant' && msg.text) return msg.text; + } + return undefined; +} + +export function getFailureEvidence(jsonl: string): string[] { + const evidence: string[] = []; + for (const call of iterToolCalls(jsonl)) { + if (call.isError && call.resultText) { + evidence.push(`tool ${call.kind} (${call.id}) failed: ${call.resultText}`); + } + } + return evidence; +} +``` + +- [ ] **Step 5: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/parse-trace.test.ts && npm run typecheck +``` + +Expected: all PASS + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/parse-trace.ts tests/parse-trace.test.ts tests/fixtures/acp-traces/ +git commit -m "feat(parse-trace): ACP trace helpers (iterMessages/iterToolCalls/computeMetrics)" +``` + +--- + +### Task 11: Trace recorder + +**Files:** +- Create: `src/workbench/acp/trace-recorder.ts` +- Create: `tests/acp/trace-recorder.test.ts` + +- [ ] **Step 1: Write the failing test** + +Create `tests/acp/trace-recorder.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, readFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { createTraceRecorder } from '../../src/workbench/acp/trace-recorder.js'; + +test('createTraceRecorder writes header then verbatim messages', () => { + const dir = mkdtempSync(join(tmpdir(), 'trace-rec-')); + const path = join(dir, 'trace.jsonl'); + const rec = createTraceRecorder({ + tracePath: path, + header: { + caseName: 'x', + agent: 'claude-agent-acp', + model: 'claude-haiku-4-5', + startedAt: '2026-05-25T00:00:00Z', + }, + }); + rec.recordRaw({ jsonrpc: '2.0', id: 1, method: 'initialize', params: {} }); + rec.recordRaw({ jsonrpc: '2.0', id: 1, result: {} }); + rec.finalize('2026-05-25T00:00:01Z'); + + const lines = readFileSync(path, 'utf-8').trim().split('\n'); + assert.equal(lines.length, 3); + const header = JSON.parse(lines[0]); + assert.equal(header.type, 'trace_start'); + assert.equal(header.agent, 'claude-agent-acp'); + assert.equal(header.endedAt, '2026-05-25T00:00:01Z'); + const msg = JSON.parse(lines[1]); + assert.equal(msg.method, 'initialize'); + + rmSync(dir, { recursive: true }); +}); +``` + +- [ ] **Step 2: Run test to verify it fails** + +```bash +npx tsx --test tests/acp/trace-recorder.test.ts +``` + +- [ ] **Step 3: Implement the recorder** + +Create `src/workbench/acp/trace-recorder.ts`: + +```typescript +import { writeFileSync, appendFileSync, mkdirSync } from 'node:fs'; +import { dirname } from 'node:path'; + +export interface TraceHeader { + caseName: string; + agent: string; + model: string; + startedAt: string; + endedAt?: string; +} + +export interface TraceRecorder { + recordRaw(message: unknown): void; + finalize(endedAt?: string): void; +} + +export function createTraceRecorder(params: { + tracePath: string; + header: TraceHeader; +}): TraceRecorder { + mkdirSync(dirname(params.tracePath), { recursive: true }); + const headerLine = JSON.stringify({ + type: 'trace_start', + schemaVersion: 2, + ...params.header, + }); + // Write header up front so partial traces are still self-describing on crash. + writeFileSync(params.tracePath, headerLine + '\n'); + + return { + recordRaw(message: unknown) { + appendFileSync(params.tracePath, JSON.stringify(message) + '\n'); + }, + finalize(endedAt?: string) { + if (!endedAt) return; + // Rewrite header with endedAt; preserve remaining lines. + const updatedHeader = JSON.stringify({ + type: 'trace_start', + schemaVersion: 2, + ...params.header, + endedAt, + }); + const existing = require('node:fs').readFileSync(params.tracePath, 'utf-8') as string; + const rest = existing.split('\n').slice(1).join('\n'); + writeFileSync(params.tracePath, updatedHeader + '\n' + rest); + }, + }; +} +``` + +- [ ] **Step 4: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/acp/trace-recorder.test.ts && npm run typecheck +``` + +Expected: PASS + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/acp/trace-recorder.ts tests/acp/trace-recorder.test.ts +git commit -m "feat(acp): trace recorder writes raw ACP messages with header+endedAt" +``` + +--- + +### Task 12: case.yml schema — require `agent:` field + +**Files:** +- Modify: `src/workbench/case-loader.ts` +- Modify: `src/workbench/types.ts` +- Create: `tests/case-loader-agent-required.test.ts` + +- [ ] **Step 1: Add `agent` + `skillUnderTest` to `WorkbenchCaseConfig` and `ResolvedWorkbenchCase`** + +Edit `src/workbench/types.ts`. In `WorkbenchCaseConfig`: + +```typescript +export interface SkillUnderTestSpec { + slug: string; // e.g., "web-design-guidelines" + hostPath: string; // absolute path on host to the skill dir containing SKILL.md +} + +export interface WorkbenchCaseConfig { + name: string; + references: string; + task: string; + graders: WorkbenchGraderConfig[]; + agent: string; // NEW — required + model: string; // semantics now agent-native + skillUnderTest?: SkillUnderTestSpec; // NEW — optional; if set, mount to agent's native skill path + // ... existing fields below unchanged +} +``` + +And in `ResolvedWorkbenchCase`: + +```typescript +export interface ResolvedWorkbenchCase { + configPath: string; + configDir: string; + name: string; + referencesDir: string; + task: string; + graders: WorkbenchGraderConfig[]; + mcpServers: WorkbenchMcpServersConfig; + mcpServices: WorkbenchMcpServicesConfig; + env: string[]; + setup: string[]; + cleanup: string[]; + agent: string; // NEW + model: string; + skillUnderTest?: SkillUnderTestSpec; // NEW + timeoutSeconds: number; +} +``` + +Document the convention: when a case is run by the chain skills, the +chain operator sets `skillUnderTest.hostPath` to either +`docs/skill-optimizer//improved-skill/` (if it exists) or +`.skill-optimizer//vendored-skill/` (else). For standalone +workbench usage (the examples in `examples/workbench/`), the field is +omitted — those cases don't need skill discovery. + +- [ ] **Step 2: Write the failing test** + +Create `tests/case-loader-agent-required.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, writeFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { loadWorkbenchCase } from '../src/workbench/case-loader.js'; + +test('loadWorkbenchCase throws when agent: is missing', () => { + const dir = mkdtempSync(join(tmpdir(), 'case-test-')); + writeFileSync(join(dir, 'case.yml'), ` +name: test +references: ./refs +task: do nothing +graders: + - name: noop + command: 'true' +model: openrouter/anthropic/claude-haiku-4-5 +`); + assert.throws( + () => loadWorkbenchCase(join(dir, 'case.yml')), + /agent.*required|missing.*agent/i, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchCase throws when agent: is unknown', () => { + const dir = mkdtempSync(join(tmpdir(), 'case-test-')); + writeFileSync(join(dir, 'case.yml'), ` +name: test +references: ./refs +agent: nonexistent-agent +model: x +task: do nothing +graders: + - name: noop + command: 'true' +`); + assert.throws( + () => loadWorkbenchCase(join(dir, 'case.yml')), + /Unknown agent/, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchCase accepts valid agent', () => { + const dir = mkdtempSync(join(tmpdir(), 'case-test-')); + writeFileSync(join(dir, 'case.yml'), ` +name: test +references: ./refs +agent: pi-acp +model: openrouter/anthropic/claude-haiku-4-5 +task: do nothing +graders: + - name: noop + command: 'true' +`); + // Refs dir doesn't exist; loader may complain about references. Test + // only that agent validation passes — catch and ignore non-agent errors. + try { + const c = loadWorkbenchCase(join(dir, 'case.yml')); + assert.equal(c.agent, 'pi-acp'); + } catch (e: any) { + assert.doesNotMatch(e.message, /agent/i); + } + rmSync(dir, { recursive: true }); +}); +``` + +- [ ] **Step 3: Run test to see it fails (loader doesn't validate yet)** + +```bash +npx tsx --test tests/case-loader-agent-required.test.ts +``` + +- [ ] **Step 4: Modify `case-loader.ts` to enforce the field** + +In `src/workbench/case-loader.ts`, find the parsing/validation function +and add agent validation. Add at top of the file: + +```typescript +import { resolveAgent } from './agents/registry.js'; +``` + +In the function that builds `ResolvedWorkbenchCase` from parsed YAML, +add (near the top of validation): + +```typescript +if (!parsed.agent || typeof parsed.agent !== 'string') { + throw new Error( + `Case ${configPath} is missing required field \`agent:\`. ` + + `Valid agents: claude-agent-acp, codex-acp, gemini, opencode, pi-acp.`, + ); +} +// Validate the agent exists (throws with suggestion on unknown). +const agentCfg = resolveAgent(parsed.agent); +``` + +And include `agent: agentCfg.name` in the returned object (canonical +name, in case the input was an alias). + +- [ ] **Step 5: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/case-loader-agent-required.test.ts && npm run typecheck +``` + +Expected: all PASS + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/case-loader.ts src/workbench/types.ts tests/case-loader-agent-required.test.ts +git commit -m "feat(schema): require agent: field in case.yml, fail loud on missing/unknown" +``` + +--- + +### Task 13: suite.yml schema — `runs:` matrix + +**Files:** +- Modify: `src/workbench/suite-loader.ts` +- Modify: `src/workbench/types.ts` +- Create: `tests/suite-loader-runs.test.ts` + +- [ ] **Step 1: Add `WorkbenchRunSpec` type** + +Edit `src/workbench/types.ts`, add: + +```typescript +export interface WorkbenchRunSpec { + agent: string; + model: string; +} + +// If WorkbenchSuiteConfig exists in this file, modify it: +// - REMOVE: models: string[]; +// - ADD: runs: WorkbenchRunSpec[]; +``` + +- [ ] **Step 2: Write the failing test** + +Create `tests/suite-loader-runs.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, writeFileSync, rmSync, mkdirSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { loadWorkbenchSuite } from '../src/workbench/suite-loader.js'; + +test('loadWorkbenchSuite parses runs: matrix', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +runs: + - agent: claude-agent-acp + model: claude-haiku-4-5 + - agent: pi-acp + model: openrouter/anthropic/claude-haiku-4-5 +cases: + - name: test-case + task: do nothing + graders: + - name: noop + command: 'true' +`); + const suite = loadWorkbenchSuite(join(dir, 'suite.yml')); + assert.equal(suite.runs.length, 2); + assert.equal(suite.runs[0].agent, 'claude-agent-acp'); + assert.equal(suite.runs[1].agent, 'pi-acp'); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchSuite rejects legacy `models:` shape', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +models: + - openrouter/anthropic/claude-haiku-4-5 +cases: [] +`); + assert.throws( + () => loadWorkbenchSuite(join(dir, 'suite.yml')), + /runs:.*replaced.*models:/i, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchSuite validates each run.agent', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +runs: + - agent: bogus + model: x +cases: [] +`); + assert.throws( + () => loadWorkbenchSuite(join(dir, 'suite.yml')), + /Unknown agent.*bogus/, + ); + rmSync(dir, { recursive: true }); +}); +``` + +- [ ] **Step 3: Run test to verify it fails** + +```bash +npx tsx --test tests/suite-loader-runs.test.ts +``` + +- [ ] **Step 4: Modify suite-loader.ts** + +In `src/workbench/suite-loader.ts`, add: + +```typescript +import { resolveAgent } from './agents/registry.js'; +``` + +In the loader function, after parsing the YAML: + +```typescript +if ('models' in parsed && !('runs' in parsed)) { + throw new Error( + `Suite ${configPath} uses legacy 'models:' field. ` + + `Replace with 'runs:' matrix (list of { agent, model } objects).`, + ); +} +if (!Array.isArray(parsed.runs) || parsed.runs.length === 0) { + throw new Error(`Suite ${configPath} must declare a non-empty 'runs:' list.`); +} +for (const run of parsed.runs) { + if (!run.agent || !run.model) { + throw new Error(`Each run in ${configPath} must have agent: and model: fields.`); + } + // Validate; throws on unknown agent. + run.agent = resolveAgent(run.agent).name; +} +``` + +In the return object, replace `models:` with `runs:`. + +- [ ] **Step 5: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/suite-loader-runs.test.ts && npm run typecheck +``` + +Expected: all PASS + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/suite-loader.ts src/workbench/types.ts tests/suite-loader-runs.test.ts +git commit -m "feat(schema): suite.yml uses runs: matrix; legacy models: rejected" +``` + +--- + +### Task 14: Refactor `metrics.ts` as thin wrapper over `parse-trace` + +**Files:** +- Modify: `src/workbench/metrics.ts` +- Modify: `src/workbench/types.ts` + +- [ ] **Step 1: Remove `cost` from `WorkbenchMetrics`** + +In `src/workbench/types.ts`: + +```typescript +export interface WorkbenchMetrics { + durationMs: number; + turns: number; + toolCalls: number; + toolResults: number; // KEEP for backward compat with downstream readers; equals toolCalls completed + bashCalls: number; + readCalls: number; + writeCalls: number; + editCalls: number; + stopReason?: string; + tokens: WorkbenchTokenMetrics; + // REMOVED: cost: WorkbenchCostMetrics +} +``` + +Delete the `WorkbenchCostMetrics` interface entirely. + +- [ ] **Step 2: Rewrite `metrics.ts` as thin wrapper** + +Replace `src/workbench/metrics.ts`: + +```typescript +import { readFileSync } from 'node:fs'; +import { computeMetrics as computeFromTrace, getFailureEvidence, getFinalAssistantMessage } from './parse-trace.js'; +import type { WorkbenchMetrics, WorkbenchResult, WorkbenchTrialSummaryFile } from './types.js'; + +export function buildWorkbenchMetricsFromTrace(tracePath: string): WorkbenchMetrics { + const jsonl = readFileSync(tracePath, 'utf-8'); + const m = computeFromTrace(jsonl); + return { + durationMs: m.durationMs, + turns: m.turns, + toolCalls: m.toolCalls, + toolResults: m.toolCalls, // ACP doesn't distinguish; treat each call as one result + bashCalls: m.bashCalls, + readCalls: m.readCalls, + writeCalls: m.writeCalls, + editCalls: m.editCalls, + stopReason: m.stopReason, + tokens: m.tokens, + }; +} + +export function buildTrialSummary(params: { + tracePath: string; + result: WorkbenchResult; +}): WorkbenchTrialSummaryFile { + const jsonl = readFileSync(params.tracePath, 'utf-8'); + const metrics = params.result.metrics ?? buildWorkbenchMetricsFromTrace(params.tracePath); + const failedGraders = params.result.graders?.filter((g) => !g.pass).map((g) => g.name) ?? []; + return { + finalAssistantMessage: getFinalAssistantMessage(jsonl), + failedGraders, + evidence: [...params.result.evidence, ...getFailureEvidence(jsonl)], + bashCommands: extractBashCommands(jsonl), + stopReason: metrics.stopReason, + errorMessage: undefined, + metrics, + }; +} + +function extractBashCommands(jsonl: string): string[] { + const out: string[] = []; + const rows = jsonl.split(/\r?\n/).filter(Boolean).map((l) => { + try { return JSON.parse(l); } catch { return null; } + }); + for (const row of rows) { + if ((row as any)?.method !== 'session/update') continue; + const u = (row as any).params?.update; + if (u?.sessionUpdate === 'tool_call' && u.kind === 'execute') { + const cmd = u.content?.[0]?.content?.text ?? u.argsText; + if (typeof cmd === 'string') out.push(cmd); + } + } + return out; +} +``` + +- [ ] **Step 3: Update any callers that referenced `metrics.cost`** + +```bash +grep -rn "metrics.cost\|cost:" src/workbench/ | grep -v test +``` + +For each match, delete the `cost:` line / property. + +- [ ] **Step 4: Run typecheck and existing tests** + +```bash +npm run typecheck && npm test +``` + +Expected: no type errors; tests pass (some may need adjustment if they +asserted on the cost field — fix those by removing cost assertions). + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/metrics.ts src/workbench/types.ts +git commit -m "refactor(metrics): thin wrapper over parse-trace; drop cost field" +``` + +--- + +### Task 15: Refactor `trace.ts` — drop normalization layer + +**Files:** +- Modify: `src/workbench/trace.ts` +- Modify: `src/workbench/types.ts` + +- [ ] **Step 1: Delete `WorkbenchTraceEntry` from types.ts** + +In `src/workbench/types.ts`, delete the `WorkbenchTraceEntry` union +type and the `WorkbenchTraceEvent` interface. Keep `WorkbenchTrace` but +simplify: + +```typescript +export interface WorkbenchTrace { + schemaVersion?: 2; // bumped from 1 (raw ACP format) + caseName: string; + agent: string; // NEW + model: string; + startedAt: string; + endedAt: string; + // No more entries[]; the trace.jsonl file IS the source of truth. +} +``` + +- [ ] **Step 2: Replace `trace.ts` with raw-only writer** + +Replace `src/workbench/trace.ts` with: + +```typescript +// Re-export trace-recorder as the public API. +export { createTraceRecorder } from './acp/trace-recorder.js'; +export type { TraceHeader, TraceRecorder } from './acp/trace-recorder.js'; + +// Legacy buildWorkbenchTrace removed; raw ACP capture replaces it. +``` + +- [ ] **Step 3: Update callers that imported `buildWorkbenchTrace`, `createTraceCollector`, `WorkbenchTraceEntry`** + +```bash +grep -rn "buildWorkbenchTrace\|createTraceCollector\|WorkbenchTraceEntry\|WorkbenchTraceEvent" src/workbench/ | grep -v test +``` + +The only legitimate caller after Task 16 will be docker-runner.ts. +For now, comment out those imports with `// TODO: replaced by ACP +client in Task 16` to allow typecheck to proceed. (We'll delete those +TODOs in Task 16.) + +- [ ] **Step 4: Run typecheck** + +```bash +npm run typecheck +``` + +Expected: passes after TODO comments hide unresolved references. + +- [ ] **Step 5: Commit** + +```bash +git add src/workbench/trace.ts src/workbench/types.ts +git commit -m "refactor(trace): drop WorkbenchTraceEntry; trace.jsonl is raw ACP" +``` + +--- + +### Task 16: Integrate ACP into `docker-runner.ts` + +This is the integration centerpiece. Replace the existing agent-spawn +flow inside `runDockerWorkbenchCase` with: per-trial container, +auth/skill/MCP setup, ACP client lifecycle, trace recording. + +**Files:** +- Modify: `src/workbench/docker-runner.ts` + +- [ ] **Step 1: Add the new imports** + +At the top of `src/workbench/docker-runner.ts`, add: + +```typescript +import { spawn } from 'node:child_process'; +import { resolveAgent } from './agents/registry.js'; +import { resolveAuth, resolveContainerPath } from './acp/auth.js'; +import { computeSkillMount, dockerMountFlag } from './acp/skill-deploy.js'; +import { writeMcpConfig } from './acp/mcp-config-writer.js'; +import { createTraceRecorder } from './acp/trace-recorder.js'; +import { createDockerExecStream } from './acp/transport.js'; +import { createWorkbenchClient } from './acp/client.js'; +``` + +- [ ] **Step 2: Add a helper that runs one agent trial via ACP** + +Add the following function. It runs ONE trial: starts the container, +mounts auth/skill, exec's the agent, drives the ACP handshake, captures +the trace, and runs graders. + +```typescript +async function runOneAcpTrial(params: { + resolvedCase: ResolvedWorkbenchCase; + tempDir: string; + workDir: string; + caseDir: string; + resultsDir: string; + agentName: string; + model: string; + image: string; + hostSkillDir: string; + skillSlug: string; + repoRoot: string; + timeoutSeconds: number; +}): Promise<{ pass: boolean; tracePath: string; resultPath: string }> { + const agent = resolveAgent(params.agentName); + const auth = resolveAuth(agent, { env: process.env }); + + // Stage auth files + MCP config into a host-side dir that we'll bind-mount. + const agentHomeHost = join(params.tempDir, 'agent-home'); + mkdirSync(agentHomeHost, { recursive: true }); + if (auth.mode === 'subscription') { + for (const f of auth.files) { + const targetPath = resolveContainerPath(f.containerPath, agentHomeHost); + mkdirSync(dirname(targetPath), { recursive: true }); + cpSync(f.hostPath, targetPath); + } + } + writeMcpConfig({ + agent, + caseConfig: { mcpServers: params.resolvedCase.mcpServers as any }, + agentHomeOnHost: agentHomeHost, + }); + + // Compute mounts + const skillMount = computeSkillMount({ + agent, + skillSlug: params.skillSlug, + hostSkillDir: params.hostSkillDir, + agentHome: '/home/agent', + }); + const authMounts = [`-v ${shellQuote(`${agentHomeHost}:/home/agent:rw`)}`]; + const skillMountFlag = dockerMountFlag(skillMount); + + // Env passthrough + const envFlags: string[] = []; + if (auth.mode === 'env') { + for (const name of auth.envNames) { + if (process.env[name]) envFlags.push(`-e ${name}`); + } + } + for (const name of params.resolvedCase.env) { + if (process.env[name]) envFlags.push(`-e ${name}`); + } + + // Start the container (detached, idle) + const containerName = `skill-opt-trial-${params.tempDir.split('/').pop()}`; + const runCmd = [ + 'docker run -d', + `--name ${shellQuote(containerName)}`, + '--cap-drop=ALL', '--security-opt no-new-privileges', '--pids-limit 512', + `-v ${shellQuote(`${params.workDir}:/work:rw`)}`, + `-v ${shellQuote(`${params.caseDir}:/case:ro`)}`, + `-v ${shellQuote(`${params.resultsDir}:/results:rw`)}`, + ...authMounts, + skillMountFlag, + ...envFlags, + '--workdir /work', + '--entrypoint sleep', + shellQuote(params.image), + 'infinity', + ].join(' '); + const runResult = await runShellCommand(runCmd, { cwd: params.repoRoot }); + if (runResult.exitCode !== 0) { + throw new Error(`Failed to start container: ${runResult.stderr}`); + } + + const tracePath = join(params.resultsDir, 'trace.jsonl'); + const resultPath = join(params.resultsDir, 'result.json'); + const startedAt = new Date().toISOString(); + const recorder = createTraceRecorder({ + tracePath, + header: { + caseName: params.resolvedCase.name, + agent: agent.name, + model: params.model, + startedAt, + }, + }); + + try { + // Run setup phase (still inside the container, via the existing entrypoint) + if (params.resolvedCase.setup.length > 0) { + const setupCmd = `docker exec ${shellQuote(containerName)} sh -c ${shellQuote( + params.resolvedCase.setup.join(' && '), + )}`; + const setupResult = await runShellCommand(setupCmd, { cwd: params.repoRoot }); + if (setupResult.exitCode !== 0) { + throw new Error(`Setup failed: ${setupResult.stderr}`); + } + } + + // Exec the agent CLI inside the container with stdin/stdout piped + const child = spawn('docker', [ + 'exec', '-i', + containerName, + 'sh', '-c', agent.launchCmd, + ], { stdio: ['pipe', 'pipe', 'pipe'] }); + + const stream = createDockerExecStream(child as any); + const client = createWorkbenchClient({ + stream, + onSessionUpdate: (notification) => { + // Re-wrap as JSON-RPC notification for the trace + recorder.recordRaw({ + jsonrpc: '2.0', + method: 'session/update', + params: notification, + }); + }, + }); + + // ACP handshake + await client.connection.initialize({ + protocolVersion: '0.22', + clientCapabilities: {}, + }); + const session = await client.connection.newSession({ + cwd: '/work', + mcpServers: [], + }); + const promptPromise = client.connection.prompt({ + sessionId: session.sessionId, + prompt: [{ type: 'text', text: params.resolvedCase.task }], + }); + + // Timeout wrapper + const promptResult = await Promise.race([ + promptPromise.then((r) => ({ ok: true, result: r }) as const), + new Promise<{ ok: false }>((_, reject) => + setTimeout(() => reject(new Error(`Timeout after ${params.timeoutSeconds}s`)), params.timeoutSeconds * 1000), + ), + ]); + + if (promptResult.ok) { + // Record the final response with its usage + recorder.recordRaw({ jsonrpc: '2.0', id: 'final', result: promptResult.result }); + } + + await client.close(); + recorder.finalize(new Date().toISOString()); + + // Run graders + const gradeCmd = `docker exec ${shellQuote(containerName)} node /app/dist/workbench/container-runner.js --grade --case /case/case.yml --work /work --results /results`; + const gradeResult = await runShellCommand(gradeCmd, { cwd: params.repoRoot }); + + // Copy agent-internal traces out. Each agent writes to its own dot-dir; + // we copy any that exist. Failure to copy (dir absent) is non-fatal — the + // archive is a debugging bonus, not the canonical trace. + const internalDir = join(params.resultsDir, 'agent-internal'); + mkdirSync(internalDir, { recursive: true }); + for (const dotDir of ['.claude', '.codex', '.gemini', '.opencode', '.pi']) { + await runShellCommand( + `docker cp ${shellQuote(`${containerName}:/home/agent/${dotDir}`)} ${shellQuote(internalDir)} 2>/dev/null || true`, + { cwd: params.repoRoot }, + ); + } + + const pass = readTrialPass(resultPath) ?? false; + return { pass, tracePath, resultPath }; + } finally { + await runShellCommand(`docker rm -f ${shellQuote(containerName)}`, { cwd: params.repoRoot }); + } +} +``` + +- [ ] **Step 3: Replace the body of `runDockerWorkbenchCase` to call `runOneAcpTrial`** + +In the existing `runDockerWorkbenchCase` function, after `prepareDockerWorkbenchRun`, +replace the agent + grade phase block with: + +```typescript +// Compute the skill mount params if the case declares a skill under test. +// (Cases that don't declare skillUnderTest skip the skill mount entirely — +// the agent runs without any skill-discovery deployment.) +const skillMountParams = resolvedCase.skillUnderTest + ? { + hostSkillDir: resolvedCase.skillUnderTest.hostPath, + skillSlug: resolvedCase.skillUnderTest.slug, + } + : null; + +const trialOutcome = await runOneAcpTrial({ + resolvedCase, + tempDir: prepared.tempDir, + workDir: prepared.workDir, + caseDir: prepared.caseDir, + resultsDir: prepared.resultsDir, + agentName: resolvedCase.agent, + model: options.model ?? resolvedCase.model, + image, + skillMountParams, + repoRoot, + timeoutSeconds: resolvedCase.timeoutSeconds, +}); +``` + +Adjust `runOneAcpTrial`'s signature to accept `skillMountParams: { hostSkillDir, skillSlug } | null` +and branch on it in step 2 above: only call `computeSkillMount` / +`dockerMountFlag` when non-null. If null, no `-v` for the skill. + +- [ ] **Step 4: Delete the old MCP service / setup / agent code paths from `runDockerWorkbenchCase`** + +Remove the calls to: +- `startMcpServices` (replaced by per-agent native MCP config inside `runOneAcpTrial`) +- `waitForMcpServices` +- The `buildDockerAgentCommand` path (the old --agent flow) + +Keep `removeContainer` / `removeDockerNetwork` cleanups in the `finally` block. + +- [ ] **Step 5: Run typecheck** + +```bash +npm run typecheck +``` + +Fix any errors. Common issues will be deleted types (`WorkbenchTraceEntry`) +that are still imported somewhere. + +- [ ] **Step 6: Run existing unit tests to confirm nothing else broke** + +```bash +npm test +``` + +Expected: PASS (existing tests are unit-level and don't depend on the +end-to-end flow we just refactored). + +- [ ] **Step 7: Commit** + +```bash +git add src/workbench/docker-runner.ts +git commit -m "feat(docker-runner): host-side ACP orchestration per trial" +``` + +--- + +### Task 17: Container-runner cleanup — remove `--agent` mode + +**Files:** +- Modify: `src/workbench/container-runner.ts` + +- [ ] **Step 1: Delete the `--agent` mode** + +In `src/workbench/container-runner.ts`, delete: +- `interface AgentRunnerArgs` +- The `--agent` branch of `parseContainerRunnerArgs` +- The `runAgentMode` function +- The branch in `runContainerWorkbenchCase` that dispatches to `runAgentMode` + +Keep: +- `--setup` mode (`runSetupMode`) +- `--grade` mode (`runGradeMode`) +- `writeTraceFile` (still used by grade mode if it derives a trace) + +- [ ] **Step 2: Simplify `parseContainerRunnerArgs`** + +Replace with: + +```typescript +export type ContainerRunnerArgs = GradeRunnerArgs | SetupRunnerArgs; + +export function parseContainerRunnerArgs(args: string[]): ContainerRunnerArgs { + const workDir = getFlagValue(args, '--work'); + const casePath = getFlagValue(args, '--case'); + + if (args.includes('--setup')) { + if (!casePath || !workDir) { + throw new Error('Usage: container-runner --setup --case --work '); + } + return { mode: 'setup', casePath, workDir }; + } + + if (args.includes('--grade')) { + const resultsDir = getFlagValue(args, '--results'); + if (!casePath || !workDir || !resultsDir) { + throw new Error('Usage: container-runner --grade --case --work --results '); + } + return { mode: 'grade', casePath, workDir, resultsDir }; + } + + throw new Error('container-runner: expected --setup or --grade'); +} +``` + +- [ ] **Step 3: Simplify `runContainerWorkbenchCase`** + +```typescript +export async function runContainerWorkbenchCase(args: string[]): Promise { + const parsed = parseContainerRunnerArgs(args); + if (parsed.mode === 'setup') return runSetupMode(parsed); + return runGradeMode(parsed); +} +``` + +- [ ] **Step 4: Delete pi-agent.ts** + +```bash +git rm src/workbench/pi-agent.ts +``` + +- [ ] **Step 5: Build and run typecheck** + +```bash +npm run build && npm run typecheck +``` + +Fix any remaining imports of `pi-agent.ts` or `runAgentMode`. + +- [ ] **Step 6: Run tests** + +```bash +npm test +``` + +- [ ] **Step 7: Commit** + +```bash +git add src/workbench/container-runner.ts src/workbench/pi-agent.ts +git commit -m "refactor(container-runner): drop --agent mode and pi-agent.ts (ACP host-side now)" +``` + +--- + +### Task 18: Update `trials.ts` to aggregate tokens + duration + +**Files:** +- Modify: `src/workbench/trials.ts` +- Modify: `src/workbench/types.ts` +- Create: `tests/trials-aggregate.test.ts` + +- [ ] **Step 1: Add aggregate fields to types** + +In `src/workbench/types.ts`, extend `TrialAggregate` and `WorkbenchModelAggregateResult`: + +```typescript +export interface TrialAggregate { + totalTrials: number; + passedTrials: number; + failedTrials: number; + trialPassRate: number; + meanScore: number; + passAtK: boolean; + passHatK: boolean; + totalTokens: number; // NEW + totalDurationMs: number; // NEW +} +``` + +- [ ] **Step 2: Write the failing test** + +Create `tests/trials-aggregate.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { aggregateTrials } from '../src/workbench/trials.js'; + +test('aggregateTrials sums tokens and duration', () => { + const result = aggregateTrials([ + { trial: 1, pass: true, score: 1, tokens: 100, durationMs: 1000 }, + { trial: 2, pass: false, score: 0, tokens: 200, durationMs: 2000 }, + { trial: 3, pass: true, score: 1, tokens: 150, durationMs: 1500 }, + ]); + assert.equal(result.totalTokens, 450); + assert.equal(result.totalDurationMs, 4500); + assert.equal(result.passedTrials, 2); +}); +``` + +- [ ] **Step 3: Run test to verify it fails** + +```bash +npx tsx --test tests/trials-aggregate.test.ts +``` + +- [ ] **Step 4: Update `aggregateTrials`** + +In `src/workbench/trials.ts`, extend `TrialScoreInput`: + +```typescript +export interface TrialScoreInput { + trial: number; + pass: boolean; + score: number; + tokens?: number; + durationMs?: number; +} +``` + +Update the function body to also sum tokens + duration. Update the +return object. + +Update callers (`run-case.ts`, `run-suite.ts`) to populate +`tokens` and `durationMs` when building `TrialScoreInput` by reading +each trial's `result.json` `metrics.tokens.total` and `metrics.durationMs`. + +- [ ] **Step 5: Verify tests pass and typecheck** + +```bash +npx tsx --test tests/trials-aggregate.test.ts && npm run typecheck && npm test +``` + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/trials.ts src/workbench/types.ts src/workbench/run-case.ts src/workbench/run-suite.ts tests/trials-aggregate.test.ts +git commit -m "feat(trials): aggregate tokens and durationMs across trials" +``` + +--- + +### Task 19: Update `run-case.ts` and `run-suite.ts` for `runs:` matrix + +**Files:** +- Modify: `src/workbench/run-case.ts` +- Modify: `src/workbench/run-suite.ts` + +- [ ] **Step 1: In `run-case.ts`, change matrix dimension from `models` to per-(case, agent, model) jobs** + +Find the `runWorkbenchCaseMatrix` function. Replace the job-building block: + +```typescript +// Before: +// const jobs = params.models.flatMap((model) => Array.from({ length: trials }, (_, index) => ({ model, trial: index + 1 }))); + +// After: +const jobs = params.runs.flatMap((run) => Array.from({ length: trials }, (_, index) => ({ + agent: run.agent, + model: run.model, + trial: index + 1, +}))); +``` + +And in the trial directory naming: + +```typescript +function trialDirName(agent: string, model: string, trial: number): string { + return `${agent}--${slugModelRef(model)}--${formatTrialNumber(trial)}`; +} +``` + +Update all references to use `(agent, model, trial)` instead of +`(model, trial)`. + +- [ ] **Step 2: In `run-suite.ts`, do the same** + +Replace the suite's matrix iteration from `models × cases` to `runs × cases`. + +- [ ] **Step 3: Update CLI in cli-args.ts / cli.ts to drop `--models` flag** + +`run-case --models` flag was the old escape hatch. With `runs:` only, +remove `--models` from the run-case CLI flag list. `run-suite` already +read from `suite.yml`; no CLI change there. + +- [ ] **Step 4: Run typecheck and build** + +```bash +npm run typecheck && npm run build +``` + +Fix any compilation errors. + +- [ ] **Step 5: Run tests** + +```bash +npm test +``` + +Expected: PASS (some unit tests may need updating to use the new shape). + +- [ ] **Step 6: Commit** + +```bash +git add src/workbench/run-case.ts src/workbench/run-suite.ts src/workbench/cli-args.ts src/workbench/cli.ts +git commit -m "feat(run-*): replace models matrix with runs (agent + model) matrix" +``` + +--- + +### Task 20: Migrate example suites to new schema + +**Files:** +- Modify: `examples/workbench/pdf/suite.yml` +- Modify: `examples/workbench/mcp/suite.yml` + +- [ ] **Step 1: Migrate pdf/suite.yml** + +Replace the `models:` block with a `runs:` block, agent set to pi-acp. + +In `examples/workbench/pdf/suite.yml`, change: + +```yaml +models: + - openrouter/google/gemini-2.5-flash +``` + +To: + +```yaml +runs: + - agent: pi-acp + model: openrouter/google/gemini-2.5-flash +``` + +Add `agent: pi-acp` to each inline case definition. Iterate the four +cases in the file (`extract-pdf-facts`, `split-customer-packet`, +`build-briefing-pdf`, `no-pdf-skill-needed`) — add `agent: pi-acp` +right after each `name:` line. + +- [ ] **Step 2: Migrate mcp/suite.yml the same way** + +- [ ] **Step 3: Delete stale result + log dirs** + +```bash +find examples/workbench -type d -name ".results" -exec rm -rf {} + 2>/dev/null +find examples/workbench -name ".run.log" -delete +``` + +- [ ] **Step 4: Run a dry-load of each suite to confirm it parses** + +```bash +npx tsx -e "import('./src/workbench/suite-loader.js').then(m => { console.log(m.loadWorkbenchSuite('examples/workbench/pdf/suite.yml')); })" +npx tsx -e "import('./src/workbench/suite-loader.js').then(m => { console.log(m.loadWorkbenchSuite('examples/workbench/mcp/suite.yml')); })" +``` + +Expected: both print suite objects with `runs:` field populated. + +- [ ] **Step 5: Commit** + +```bash +git add examples/workbench/ +git commit -m "chore(examples): migrate pdf+mcp suites to runs: schema; remove stale results" +``` + +--- + +### Task 21: Per-agent smoke probes + +**Files:** +- Create: `tests/smoke-agents/_common.ts` +- Create: `tests/smoke-agents/claude-agent-acp.smoke.test.ts` +- Create: `tests/smoke-agents/codex-acp.smoke.test.ts` +- Create: `tests/smoke-agents/gemini.smoke.test.ts` +- Create: `tests/smoke-agents/opencode.smoke.test.ts` +- Create: `tests/smoke-agents/pi-acp.smoke.test.ts` + +- [ ] **Step 1: Write a shared smoke harness** + +Create `tests/smoke-agents/_common.ts`: + +```typescript +import { mkdtempSync, writeFileSync, readFileSync, existsSync, rmSync, mkdirSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { runDockerWorkbenchCase } from '../../src/workbench/docker-runner.js'; + +export async function runSmokeTrial(params: { + agent: string; + model: string; + env: string[]; +}): Promise<{ pass: boolean; outputContent?: string; tracePath: string }> { + const dir = mkdtempSync(join(tmpdir(), 'smoke-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'case.yml'), ` +name: smoke-${params.agent} +references: ./references +agent: ${params.agent} +model: ${params.model} +task: | + Write the literal string "hello" to /work/out.txt. + Then write a one-line description of what you did to /work/findings.txt. +graders: + - name: out-txt-exists-with-hello + command: 'test "$(cat /work/out.txt 2>/dev/null)" = "hello"' +env: +${params.env.map(e => ` - ${e}`).join('\n') || ' []'} +timeoutSeconds: 120 +`); + const result = await runDockerWorkbenchCase({ + casePath: join(dir, 'case.yml'), + image: 'skill-optimizer-agent:local', + keepWorkspace: true, + }); + const tracePath = result.tracePath; + let outputContent: string | undefined; + try { + outputContent = readFileSync(join(result.workspacePath!, 'out.txt'), 'utf-8'); + } catch {} + + const resultData = JSON.parse(readFileSync(result.resultPath, 'utf-8')); + rmSync(dir, { recursive: true }); + return { pass: resultData.pass, outputContent, tracePath }; +} +``` + +- [ ] **Step 2: Per-agent test files** + +For each agent, create a file that calls `runSmokeTrial` with the +right model + env. Example for claude: + +Create `tests/smoke-agents/claude-agent-acp.smoke.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { runSmokeTrial } from './_common.js'; + +const skip = !process.env.ANTHROPIC_API_KEY && !require('node:fs').existsSync( + require('node:path').join(require('node:os').homedir(), '.claude/.credentials.json'), +); + +test('claude-agent-acp smoke: writes hello to out.txt', { skip }, async () => { + const r = await runSmokeTrial({ + agent: 'claude-agent-acp', + model: 'claude-haiku-4-5-20251001', + env: ['ANTHROPIC_API_KEY'], + }); + assert.equal(r.pass, true); + assert.equal(r.outputContent?.trim(), 'hello'); +}); +``` + +Create equivalents for the other 4 agents, adjusting model and env: +- codex: model `gpt-5-mini`, env `OPENAI_API_KEY`, sub-auth file `~/.codex/auth.json` +- gemini: model `gemini-3.1-pro-preview`, env `GOOGLE_API_KEY`, sub-auth file `~/.gemini/oauth_creds.json` +- opencode: model `google/gemini-3.1-pro-preview`, env `OPENAI_API_KEY` +- pi-acp: model `openrouter/anthropic/claude-haiku-4-5`, env `OPENROUTER_API_KEY` + +- [ ] **Step 3: Run each smoke test** + +```bash +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . +# Run each one (will skip if auth unavailable) +for agent in claude-agent-acp codex-acp gemini opencode pi-acp; do + npx tsx --test tests/smoke-agents/${agent}.smoke.test.ts +done +``` + +Expected: each tests passes or skips. Failures indicate agent-specific +runtime issues (likely registry quirks); fix per-agent and re-run. + +- [ ] **Step 4: Add smoke tests to npm test script** + +In `package.json`, ensure `npm test` includes `tests/smoke-agents/**`. +If smoke tests should be gated behind a flag (e.g., they need Docker +to be running locally), add a `npm run test:smoke` script and call it +from CI only when Docker is present. + +- [ ] **Step 5: Commit** + +```bash +git add tests/smoke-agents/ package.json +git commit -m "test(smoke): per-agent end-to-end smoke probes" +``` + +--- + +### Task 22: Pi-acp regression probe against pdf suite + +**Files:** +- Create: `tests/regression/pi-acp-pdf-suite.test.ts` +- Create: `tests/regression/baselines/pi-acp-pdf-2026-05-25.json` + +- [ ] **Step 1: Capture a baseline run of the pdf suite under pi-acp** + +```bash +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . +OPENROUTER_API_KEY=$OPENROUTER_API_KEY npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials 3 +``` + +This produces `examples/workbench/pdf/.results//suite-result.json`. + +Copy the per-case pass rates into `tests/regression/baselines/pi-acp-pdf-2026-05-25.json`: + +```json +{ + "captured": "2026-05-25", + "agent": "pi-acp", + "model": "openrouter/google/gemini-2.5-flash", + "trials": 3, + "perCasePassRate": { + "extract-pdf-facts": 1.0, + "split-customer-packet": 1.0, + "build-briefing-pdf": 1.0, + "no-pdf-skill-needed": 1.0 + } +} +``` + +(Use whatever rates the actual run produces; rates here are illustrative.) + +- [ ] **Step 2: Write the regression test** + +Create `tests/regression/pi-acp-pdf-suite.test.ts`: + +```typescript +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { readFileSync, readdirSync } from 'node:fs'; +import { join } from 'node:path'; +import { execSync } from 'node:child_process'; + +const skip = !process.env.OPENROUTER_API_KEY; + +test('pi-acp regression: pdf suite pass rates match baseline within tolerance', { skip }, () => { + const baseline = JSON.parse(readFileSync('tests/regression/baselines/pi-acp-pdf-2026-05-25.json', 'utf-8')); + + execSync(`npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials ${baseline.trials}`, { + stdio: 'inherit', + }); + + const resultsDir = 'examples/workbench/pdf/.results'; + const runs = readdirSync(resultsDir).sort(); + const latest = runs[runs.length - 1]; + const data = JSON.parse(readFileSync(join(resultsDir, latest, 'suite-result.json'), 'utf-8')); + + for (const [caseName, expectedRate] of Object.entries(baseline.perCasePassRate)) { + const actual = data.results.find((r: any) => r.caseName === caseName)?.trialPassRate; + assert.ok(actual !== undefined, `case ${caseName} not in suite result`); + // Tolerance: ±0.34 (one trial of three can flip) + assert.ok(Math.abs(actual - (expectedRate as number)) <= 0.34, + `case ${caseName}: expected ~${expectedRate}, got ${actual}`); + } +}); +``` + +- [ ] **Step 3: Run the regression test once to confirm it passes against the captured baseline** + +```bash +OPENROUTER_API_KEY=$OPENROUTER_API_KEY npx tsx --test tests/regression/pi-acp-pdf-suite.test.ts +``` + +Expected: PASS + +- [ ] **Step 4: Commit** + +```bash +git add tests/regression/ +git commit -m "test(regression): pi-acp pdf suite baseline + tolerance check" +``` + +--- + +### Task 23: Write `skills/shared/acp-trace-format.md` + +**Files:** +- Create: `skills/shared/acp-trace-format.md` + +- [ ] **Step 1: Write the reference doc** + +Create `skills/shared/acp-trace-format.md`: + +```markdown +# ACP trace format (for chain analyzer) + +`trace.jsonl` captures the raw Agent Client Protocol (ACP) messages +exchanged between the workbench (client) and the agent CLI (server) +for one trial. The first line is a `trace_start` header with trial +metadata; every subsequent line is a JSON-RPC envelope per the ACP +spec at . + +This doc summarizes the message types the chain analyzer cares about. +For the full spec, follow the link above. + +## Header (line 1) + +```json +{ + "type": "trace_start", + "schemaVersion": 2, + "caseName": "...", + "agent": "claude-agent-acp", + "model": "claude-haiku-4-5-20251001", + "startedAt": "ISO-8601", + "endedAt": "ISO-8601" +} +``` + +## Initialization (early lines) + +- `initialize` request from client; `initialize` response from agent +- `session/new` request; response carries `sessionId` + +These confirm the agent started. If they're absent, the trial failed +before reaching the prompt — bench infrastructure issue, not skill +weakness. + +## Session updates (the meat) + +All have shape: + +```json +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"...","update":{...}}} +``` + +The `update` object's `sessionUpdate` field tells you what happened: + +| `sessionUpdate` value | Meaning | Analyzer cares because | +|---|---|---| +| `agent_message_chunk` | Streaming chunk of assistant text | Final response — what the agent told the user | +| `agent_thought_chunk` | Streaming chunk of assistant reasoning | The agent's reasoning — useful for diagnosing why it did X | +| `tool_call` | Agent is calling a tool (start) | `kind` field tells you what kind: `execute`, `read`, `write`, `edit`, `search`, `fetch`, `think`, `other` | +| `tool_call_update` | Tool result/progress (end) | `status: "completed" \| "failed"`, `content` carries the tool output | +| `plan` | Agent's high-level plan | Optional sidebar; skip in most analysis | +| `user_message_chunk` | Agent echoing user input | Rare; skip | + +## Final prompt response (last numbered response) + +```json +{ + "jsonrpc": "2.0", + "id": , + "result": { + "stopReason": "end_turn" | "max_tokens" | "refusal" | "cancelled", + "usage": { + "inputTokens": 120, + "outputTokens": 45, + "cacheReadTokens": 0, + "cacheCreationTokens": 0 + } + } +} +``` + +`stopReason` is critical for diagnosing failures: +- `end_turn` — normal completion (still check `findings.txt` for correctness) +- `max_tokens` — agent ran out of context; skill may be too verbose +- `refusal` — agent declined the task; skill description may have triggered a safety pattern +- `cancelled` — workbench timed out the prompt + +## Helpers + +Don't parse the JSONL manually. Use `src/workbench/parse-trace.ts`: + +- `iterMessages(jsonl)` — assistant/user messages (text + thinking) +- `iterToolCalls(jsonl)` — paired tool_call + tool_call_update +- `computeMetrics(jsonl)` — tokens, duration, per-tool counts, stopReason +- `getFinalAssistantMessage(jsonl)` — last assistant chunk concatenated +- `getFailureEvidence(jsonl)` — failed-tool result snippets +``` + +- [ ] **Step 2: Lint** + +```bash +pnpm dlx markdownlint-cli --fix --disable MD013 MD031 MD032 MD033 MD040 MD041 MD060 -- skills/shared/acp-trace-format.md +``` + +- [ ] **Step 3: Commit** + +```bash +git add skills/shared/acp-trace-format.md +git commit -m "docs(shared): ACP trace format reference for chain analyzer" +``` + +--- + +### Task 24: Update chain skill files for new trace format + agent column + +**Files:** +- Modify: `skills/write-tests/agents/test-writer.md` +- Modify: `skills/analyze/agents/analyzer.md` +- Modify: `skills/run-bench/SKILL.md` + +- [ ] **Step 1: Update test-writer.md to drop `/work/` skill-path boilerplate** + +In `skills/write-tests/agents/test-writer.md`, do the following exact +substitutions: + +1. Grep for occurrences: + +```bash +grep -n "work/\|skill at\|SKILL\.md\|/work" skills/write-tests/agents/test-writer.md +``` + +2. For each match that instructs the test-writer to make task prompts +reference the skill explicitly (e.g., "the task prompt should say +'use the skill at /work/X/SKILL.md'"), replace with the realistic- +invocation guidance below. Add the spec.yaml convention pointer. + +3. Add a new paragraph near the section that defines the probe spec's +`task:` field: + +```markdown +**Task prompts describe the user's actual task; do not reference the +skill explicitly.** The harness mounts the skill at the agent's native +discovery path (e.g., `~/.claude/skills//SKILL.md`). The agent +decides whether to invoke it based on the skill's frontmatter +`description`. "Skill didn't trigger" is then a measurable weakness +class — don't pre-trigger it via the task prompt. + +Example task prompt (good): "Review /work/ProductCard.tsx for +compliance issues. Write findings to /work/findings.txt." + +Example task prompt (bad — pre-triggers): "Use the skill at +/work/web-design-guidelines/SKILL.md to review /work/ProductCard.tsx." +``` + +- [ ] **Step 2: Update analyzer.md to point at acp-trace-format.md** + +In `skills/analyze/agents/analyzer.md`, find any reference to +`WorkbenchTraceEntry` or the normalized trace format. Replace with: + +```markdown +**Read first:** [`../../shared/acp-trace-format.md`](../../shared/acp-trace-format.md) +— `trace.jsonl` is raw ACP wire format. Use `parse-trace.ts` helpers +(`iterMessages`, `iterToolCalls`, `computeMetrics`, +`getFinalAssistantMessage`, `getFailureEvidence`) rather than reading +JSONL by hand. The format reference doc summarizes the event types +relevant to skill-behavior analysis. +``` + +- [ ] **Step 3: Update run-bench/SKILL.md** + +In the "Write the summary" section, expand the requirements: + +```markdown +### (d) Write the summary + +Parse `${OUT_DIR}/suite-result.json` and write `06-bench-summary.md` +with the frontmatter above plus a body containing: + +- **Overall:** total trials, passed, failed, overall pass rate. Total + tokens consumed. Total duration. +- **Per agent + model:** trial count, pass rate, mean tokens per + trial, mean duration per trial for each row in `runs:` from the + suite. +- **Per probe:** trial count, pass rate, one-line note if any + trial failed. Mean tokens and duration per probe. +- **Failed-probe pointer list:** probe IDs where any trial failed, + with paths to their `trace.jsonl` and `findings.txt`. +- **Raw output:** the `bench_results_path` value. + +Do NOT include a cost column. tokens × downstream pricing is computed +offline if needed. +``` + +- [ ] **Step 4: Lint all three** + +```bash +for f in skills/write-tests/agents/test-writer.md skills/analyze/agents/analyzer.md skills/run-bench/SKILL.md; do + pnpm dlx markdownlint-cli --fix --disable MD013 MD031 MD032 MD033 MD040 MD041 MD060 -- "$f" +done +``` + +- [ ] **Step 5: Commit** + +```bash +git add skills/ +git commit -m "feat(chain): update test-writer, analyzer, run-bench for ACP trace format + agent column" +``` + +--- + +### Task 25: Documentation refresh + +**Files:** +- Modify: `CLAUDE.md` +- Modify: `CONTRIBUTING.md` +- Modify: `README.md` + +- [ ] **Step 1: Update CLAUDE.md** + +In `CLAUDE.md`: +- Update the "Key Commands" section: add `docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile .` +- Update the "Important Files" section: add `src/workbench/acp/`, `src/workbench/agents/`, `src/workbench/parse-trace.ts`; remove `src/workbench/pi-agent.ts` +- Update "Invariants": replace "uses models from `suite.yml`" with "uses runs (agent + model) from `suite.yml`"; add "Every case.yml must declare `agent:`" +- Update "Testing Guidance": replace `docker build -t skill-optimizer-workbench:local -f docker/workbench-runner.Dockerfile .` with the new image/Dockerfile + +- [ ] **Step 2: Update CONTRIBUTING.md** + +Same set of replacements. Add a section noting the ACP architecture +(host-side client, per-trial container). + +- [ ] **Step 3: Update README.md** + +In the README, replace any "set OPENROUTER_API_KEY" prereq with a +matrix of supported agents and their auth options: + +```markdown +## Supported agents + +| Agent | Auth options | +|---|---| +| claude-agent-acp | `claude login` (subscription) or `ANTHROPIC_API_KEY` | +| codex-acp | `codex login` (subscription) or `OPENAI_API_KEY` | +| gemini | `gemini auth login` (subscription) or `GOOGLE_API_KEY` | +| opencode | `OPENAI_API_KEY` (or provider-specific key) | +| pi-acp | `OPENROUTER_API_KEY` | +``` + +- [ ] **Step 4: Lint** + +```bash +for f in CLAUDE.md CONTRIBUTING.md README.md; do + pnpm dlx markdownlint-cli --fix --disable MD013 MD031 MD032 MD033 MD040 MD041 MD060 -- "$f" +done +``` + +- [ ] **Step 5: Commit** + +```bash +git add CLAUDE.md CONTRIBUTING.md README.md +git commit -m "docs: refresh CLAUDE.md, CONTRIBUTING.md, README.md for multi-agent ACP" +``` + +--- + +### Task 26: Final integration check + +- [ ] **Step 1: Full typecheck + build + tests** + +```bash +npm run typecheck && npm run build && npm test +``` + +Expected: all PASS + +- [ ] **Step 2: Run smoke distribution test** + +```bash +npx tsx tests/smoke-skill-distribution.ts +``` + +Expected: PASS (verifies plugin metadata still references all chain +skills correctly) + +- [ ] **Step 3: Re-build the Docker image to confirm reproducibility** + +```bash +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . +``` + +- [ ] **Step 4: Run one end-to-end probe** + +```bash +OPENROUTER_API_KEY=$OPENROUTER_API_KEY npx tsx src/cli.ts run-case examples/workbench/pdf/suite.yml +``` + +Wait, run-case takes a case path. For an end-to-end check, run-suite: + +```bash +OPENROUTER_API_KEY=$OPENROUTER_API_KEY npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials 1 +``` + +Expected: completes with at least one trial reported, results in +`examples/workbench/pdf/.results//`. Inspect the result: + +```bash +ls examples/workbench/pdf/.results/$(ls -t examples/workbench/pdf/.results | head -1)/ +cat examples/workbench/pdf/.results/$(ls -t examples/workbench/pdf/.results | head -1)/suite-result.json | jq '.summary' +``` + +Confirm `summary.trialPassRate` is sensible. + +- [ ] **Step 5: Inspect a trace to confirm raw ACP format** + +```bash +cat examples/workbench/pdf/.results/$(ls -t examples/workbench/pdf/.results | head -1)/trials/*/trace.jsonl | head -5 | jq . +``` + +Expected: first line is a `trace_start` header; following lines have +`jsonrpc: "2.0"` envelopes. + +- [ ] **Step 6: Commit any final fixes if needed** + +If steps 1-5 surfaced bugs, fix them in small targeted commits. + +--- + +## Verification checklist (run after all tasks) + +- [ ] `npm run typecheck` — clean +- [ ] `npm run build` — clean +- [ ] `npm test` — all pass +- [ ] `npx tsx tests/smoke-skill-distribution.ts` — passes +- [ ] `docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile .` — succeeds +- [ ] Each of the 5 per-agent smoke tests passes (or skips gracefully) +- [ ] Pi-acp regression test against pdf suite passes within tolerance +- [ ] One end-to-end `run-suite` call against the pdf suite produces sensible results with raw ACP trace diff --git a/docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md b/docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md new file mode 100644 index 0000000..379ef3e --- /dev/null +++ b/docs/superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md @@ -0,0 +1,861 @@ +# skill-optimizer v1.4 — design spec + +**Status:** approved (brainstormed 2026-05-12 with the +`superpowers:brainstorming` skill) +**Supersedes:** the v1.3 `skills/auto-improve-orchestrator/` monolithic +orchestrator subagent. +**Empirical basis:** team review of the v1.3 PR drafts (#1, #5, #6 +drafted via v1.3) surfaced four structural critiques of the v1.3 +approach (see "Motivation" below). + +## Goal + +Convert the auto-improve workflow from a **monolithic orchestrator +subagent** (v1.3) into **nine independent Claude Code skills + an +auto-pilot driver** that chain via the same pattern as the +`superpowers` plugin (skill → +"invoke next skill" → next skill). Each skill produces a +human-reviewable report at a convention path. The user can invoke each +skill manually for explicit control, or ask the agent to chain them +auto-pilot style. + +Subagents dispatched by each skill operate under **strict limited +context**: they see only the inputs they need to do their job, never +the raw trial data or grader internals that would tempt them to +ducktape-patch. This is the load-bearing constraint that v1.3 lacked. + +The skill-optimizer engine itself (`run-suite`, graders, Docker +harness) stays unchanged — v1.4 only restructures the orchestration +layer that USES the engine. + +## Motivation + +Four critiques surfaced from team review of the v1.3 drafts: + +1. **PRs optimize for incremental numeric uplift, not principled + improvement.** A ~5–10 % score bump is shippable per v1.3's logic + even when the change is a ducktape patch (e.g., a rule restated + verbosely just to push one specific trial over the line). +2. **Test case quality is never validated.** v1.3 trusts the seeded + workbench. If the cases don't exercise the skill's real + responsibilities, the orchestrator optimizes for the wrong thing. +3. **Grader correctness is never validated.** v1.3 has a + "grader-vs-skill" check during iteration, but it only fires when + per-case-min crosses thresholds. Routine grader bugs (line drift, + keyword mismatch) get hidden. +4. **The improvement step has no anti-ducktape guardrail.** v1.3's + skill-iterate subagent sees the raw failed `findings.txt` and can + pattern-match a specific patch that satisfies the grader without + addressing the underlying weakness. (Concrete v1.3 example: + firecrawl iteration 1 regressed the original case from 1.0 → 0.44 + on gpt-5/gemini by piling on Recipe A + Recipe D simultaneously + to chase a small uplift; the orchestrator ran out of context + before catching the regression.) + +v1.4 addresses all four by **decomposing the workflow into discrete +skills, requiring human review between steps, and enforcing +limited-context constraints on every subagent that touches generative +work**. + +## Architecture overview + +**Inspired by the `superpowers` plugin convention** (skills, not slash +commands; description-based routing; "invoke next skill" chaining; +visible-at-conventional-path state files): + +```text +skills/ + skill-optimizer-investigate-functionality/SKILL.md # 1 + skill-optimizer-investigate-test-case/SKILL.md # 2 + skill-optimizer-investigate-submissions/SKILL.md # 3 (optional) + skill-optimizer-write-tests/SKILL.md # 4 + skill-optimizer-validate-tests/SKILL.md # 5 + skill-optimizer-run-bench/SKILL.md # 6 + skill-optimizer-analyze/SKILL.md # 7 + skill-optimizer-improve/SKILL.md # 8 + skill-optimizer-validate/SKILL.md # 9 + skill-optimizer-autopilot/SKILL.md # 10 (chain driver) + skill-optimizer-subagents/ + research-functionality.md + test-case-designer.md + research-submissions.md + test-writer.md + test-validator.md + analyzer.md + optimizer.md + validator.md + skill-optimizer-shared/ + iteration-protocol.md # mechanical references + subagent-dispatch.md # every chain skill loads + frontmatter-discipline.md # on demand at workflow steps + workflow.md # operator-facing chain reference + workbench.md # workbench schema reference +``` + +Companion docs follow two conventions: + +- **Project-wide reading** (e.g., authoring philosophy applicable + to any skill in the project) lives under `docs/`. +- **Chain-specific reading** (operator reference for THIS chain; + chain-internal design history) lives alongside the chain. + +```text +docs/ + skill-writing-philosophy.md # project-wide authoring guidance + superpowers/specs/2026-05-19-skill-optimizer-v1.4-design.md # THIS doc (design history) + superpowers/plans/2026-05-19-skill-optimizer-v1.4.md # implementation plan +skills/skill-optimizer-shared/ + workflow.md # the chain graph (operator reference) + iteration-protocol.md # iteration mechanics (loaded by skills) + subagent-dispatch.md # dispatch architecture (loaded by skills) + frontmatter-discipline.md # frontmatter rules (loaded by skills) +``` + +A shared `recipes.md` (named abstract failure patterns shared across +analyzer / optimizer subagents) was considered but deferred — the +v1.3 lessons.md seed is case-study-shaped, and producing a curated +pattern library requires real end-to-end observations to ground the +patterns in. See plan §"Task A4" for details. + +**State at convention path** (visible + committable, per +`superpowers` precedent): + +```text +docs/skill-optimizer// + 01-functionality.md + 02-test-proposals.md # B2's audit report (ranking + reasoning) + tests/ + / # one folder per B2-proposed functionality + spec.yaml # B2 wrote: picked, importance, suggested_probes + / # one folder per B4-built probe of this functionality + spec.yaml # B4 wrote: probe-level intent + workspace/ # fixture files the agent sees + grader.mjs # grading script + smoke/ # GOOD/BAD/EMPTY findings.txt fixtures + / + ... + / + spec.yaml + ... + suite.yml # B4 generates from picked-functionality probes + 03-submissions.md # only if step 3 ran + 05-tests-verdict.md # B5 writes: per-probe verdicts + aggregate + 06-bench-results// # raw bench output, timestamped per run + 06-bench-summary.md # B6 writes: aggregate + pointer to latest + 07-analysis.md # B7 writes + 08-improvement-proposal.md # B8 writes + 09-validator-verdict.md # B9 writes + vendored-skill/ # the source skill, read-only (always — B1 vendors upstream OR local) + improved-skill/ # B9 materializes on verdict: approve; original is never modified +``` + +`` is `--` for upstream skills, or +`` for local skills. + +**Filesystem-as-state, not file-versioning.** Single-file reports +(`01-functionality.md`, `07-analysis.md`, etc.) and the +`tests///` tree are each their own current +canonical state. History is git — there is no `version:` field, no +`archive/` directory, no manually-bumped counters, and no +`inputs.step_N: ` lineage tracking. Prior states are +recoverable via `git log`. See [`skills/skill-optimizer-shared/iteration-protocol.md`](../skills/skill-optimizer-shared/iteration-protocol.md) +for the full mechanics. + +**Two contexts, one workflow:** + +- **Upstream skill:** user provides URL or `//`. + Step 1 fetches AND asks the user up-front: "do you want to optimize + this skill for upstream PR submission?" If yes → step 3 + (`investigate-submissions`) runs after step 2, and step 8's validator + performs both internal AND external consistency checks. If no → + step 3 is skipped entirely; step 8 just validates internal + consistency and writes the improved skill to the vendored copy. +- **Local skill:** user provides a path to a SKILL.md in their repo. + Step 1 just reads (no fetch, no PR question). Step 3 never runs. + Step 7 writes the modified skill in place. + +The PR-or-not decision is made ONCE at step 1; subsequent steps know +from that decision whether step 3 will run. No late prompts. + +## Subagent constraints (the load-bearing principle) + +Every subagent that touches generative work runs under **strict +limited context**. The skill (operator-side) reads full report files +and dispatches the subagent with only the narrow chunks it needs. + +| Subagent | Sees | Does NOT see | Why | +|---|---|---|---| +| Functionality researcher (step 1) | Source skill files, web-search results, `${OPERATOR_DIRECTIVES}` | Existing analyses, existing tests, `01-functionality.md` (own canonical) or its git history | Fresh-derivation: pure research, no contamination across iterations | +| Test-case designer (step 2) | `01-functionality.md`, current `tests//spec.yaml` tree (when present — this is the state), `${OPERATOR_DIRECTIVES}` | Skill source content, `07-analysis.md`, optimizer attempts, failure data, git history of any tree files | Maintenance: the `tests/` tree IS the accumulating state; the subagent extends it across re-runs (adding/modifying spec.yaml files) without seeing source (which would gerrymander tests around the source's literal phrasing) | +| Submission researcher (step 3) | Repo files, gh-API outputs, `${OPERATOR_DIRECTIVES}` | Anything about the proposed change, `03-submissions.md` (own canonical) or its git history | Fresh-derivation: just upstream facts | +| Test writer (step 4, dispatched per probe) | The single probe's `spec.yaml` + parent functionality's `spec.yaml` + `01-functionality.md` + skill source content + `${OPERATOR_DIRECTIVES}` | Other probes' specs/graders, the eval grader's matching logic, sibling probe contents (own canonical at the per-probe level), git history of any tree files | Maintenance at tree level: each probe is built in isolation. Fixture writing needs source detail (specific patterns); the probe spec from step 2 bounds the gerrymandering risk. Blocked from seeing siblings (prevents copying) and grader internals (prevents grader-leak hacking) | +| Test validator (step 5, dispatched per probe) | The single probe's full contents (`spec.yaml`, `workspace/`, `grader.mjs`, `smoke/`) + parent functionality's `spec.yaml` + `01-functionality.md` + skill source content + `${OPERATOR_DIRECTIVES}` | Other probes' contents, test-writer's reasoning trace, prior `05-tests-verdict.md` or git history, downstream analysis/proposal/verdict | Fresh-derivation, anti-ducktape gate at the test layer. Independent semantic check on what the smoke check (syntactic, self-validation) can't catch: workspace fairness, grader correctness beyond smoke fixtures, false-positive probes | +| Analyzer (step 7) | Per-trial findings + the skill's content + workbench cases + `${OPERATOR_DIRECTIVES}` | The TEST INPUTS THEMSELVES (e.g., the seeded SQL files), `07-analysis.md` (own canonical) or its git history | Fresh-derivation: forces focus on the SKILL, not the SOLUTIONS; iteration-isolated | +| Optimizer (step 8) | `07-analysis.md` + `01-functionality.md` + current skill state (`improved-skill/` if it exists, else original source) + `${OPERATOR_DIRECTIVES}` | Raw failed trials, `findings.txt`, grader internals, test inputs, `08-improvement-proposal.md` (own canonical) or its git history | Fresh-derivation: principled improvement, not pattern-match patches; no attachment to prior failed attempts. The skill content is upstream input; the report is the own canonical | +| Validator (step 9) | Skill BEFORE (same as optimizer's input) + skill AFTER (temporary materialization) + `08-improvement-proposal.md` + `01-functionality.md` + `03-submissions.md` (if exists) | Trial data, optimizer's reasoning trace, test inputs, `09-validator-verdict.md` (own canonical) or its git history | Fresh-derivation: independent check; can't be biased by what the optimizer (or a prior validator round) told itself | + +The skill (operator session) sees everything; subagents see slices. +This is the architectural fix for the "tunnel-vision into ducktape" +problem. + +## Iteration patterns + +The chain isn't strictly linear in practice. Any step can be re-run +(operator-driven or auto-pilot-driven), and re-running a step can +invalidate downstream artifacts that derived from its prior state. +This section defines how iteration is captured, detected, and +cascaded — using git as the history mechanism and the filesystem +itself as the current state. Full mechanics in +[`skills/skill-optimizer-shared/iteration-protocol.md`](../skills/skill-optimizer-shared/iteration-protocol.md); +the summary below is for spec context. + +### When iteration happens + +Common backtrack triggers: + +| At step | Common trigger | Backtracks to | +|---|---|---| +| After 2 (test design) | "Coverage is wrong; the skill does more than this" | 1 (research) or 2 (revise with directives) | +| After 4 (write-tests) | Test writer couldn't implement a probe | 2 (revise the functionality spec) | +| After 5 (bench) | All probes pass on baseline — too easy | 2 or 4 (harder probes) | +| After 6 (analyze) | "No clear weakness" but user disagrees | 2 (better coverage), 4 (better probes), or 6 with directives | +| After 7 (improve) | Validator rejects beyond optimizer's own loop | 6 (re-analyze) or 2 (the test was wrong) | +| Anywhere | User dislikes the output | re-run that step with new directives | + +There's no rigid backtrack flowchart — operator and auto-pilot both +decide based on the named weakness in `07-analysis.md`, the +validator verdict, and the staleness mechanism below. + +### Step kinds: fresh-derivation vs maintenance + +Each step is one of two kinds, and the subagent's reading rule +differs: + +**Fresh-derivation steps (1, 3, 5-summary, 6, 7):** each invocation +derives the canonical artifact from upstream + directives. The +subagent does NOT read its own canonical file (the prior state), +and does NOT walk git history of it. Anti-ducktape constraint — +the optimizer must not see prior attempts, the analyzer must not +see prior analyses. + +**Maintenance steps (2, 4):** the canonical artifact is an +accumulating tree on disk (`tests//spec.yaml` at +step 2; `tests///{spec.yaml, workspace, +grader, smoke}` at step 4). The subagent DOES read the current +tree (it IS the state) and produces a modified tree. Git history +of the tree files is still off-limits — only the current state +matters. + +### Staleness detection (git-native) + +There is no `version:` field or `inputs.step_N: ` lineage +tracking. Staleness is detected by comparing git modification times: + +```bash +UPSTREAM_T=$(git log -1 --format=%ct -- 01-functionality.md) +DOWNSTREAM_T=$(git log -1 --format=%ct -- tests/) +[ "$UPSTREAM_T" -gt "$DOWNSTREAM_T" ] && echo "downstream stale" +``` + +In interactive use, the operator typically just knows ("I re-ran +step 1, so step 2 needs a re-run"). The git-mtime check is for +auto-pilot (step 10), which walks the chain forward and re-runs any +downstream older than its direct upstream. + +### Re-entry contract + +Every reasoning step accepts an `${OPERATOR_DIRECTIVES}` templated +slot in its subagent prompt — atomic new requirements pre-digested +from prior iterations ("user requested coverage for null inputs", +"user flagged responsibility X as under-tested"). Default empty on +first invocation. The slot is **never a context dump** of prior +output; it's a short bulleted list of new requirements only. + +### History + +Git is the archive. There is no `archive/` subdirectory. When a +fresh-derivation step overwrites its canonical file, the prior +state is captured in git's previous commit. When a maintenance +step modifies a tree (rebuilds a probe, removes a de-picked +functionality), the operator session commits a checkpoint **before** +the destructive change so git history has a clean before/after +breakpoint. + +### Transitive staleness + +Each step checks **direct upstream only**. If step 1 is updated but +step 2 isn't re-run (user judged it still valid against the new +upstream), step 3 sees step 2 as current. + +The trade-off: transitive issues surface as confusing downstream +results, not silent corruption. The discipline is that skipping a +step's re-run is an explicit operator judgment ("the old artifact +is still valid against the new upstream"). Auto-pilot doesn't make +this judgment — it re-runs any step whose direct upstream is newer +per `git log`. + +## The skills + +Seven chain skills (steps 1–7) plus an auto-pilot driver (skill 8). +Each chain skill is a directory with `SKILL.md` (the instructions) +plus optional supporting files (subagent prompts, reference material). +The prose content of each SKILL.md will be filled in during +implementation — this spec defines the **interface** +(input/output/behavior contract), not the prose. + +Per-step iteration behavior is noted at the end of each subsection. + +### 1. `skill-optimizer-investigate-functionality` + +- **Description trigger:** when the user asks "what does this skill do", + "investigate this skill", "understand this skill" +- **Input:** source skill (URL or local path) +- **Output:** `docs/skill-optimizer//01-functionality.md` +- **Behavior:** + 1. Identify whether the source is upstream (URL or + `//`) or local (path) + 2. **If upstream:** ASK the user up-front: "do you want to optimize + this skill for upstream PR submission?" Record the answer in the + report frontmatter as `pr_submission_intent: true|false`. This + answer determines whether step 3 (`investigate-submissions`) + fires later. + 3. **If local:** default to `pr_submission_intent: false`. But if + the user explicitly said they want to send this back to an + upstream maintainer (e.g., "I'll fork this back to the original + repo", "I want to PR this to project X"), treat it as PR-intent: + set `pr_submission_intent: true`, ask the user where the + upstream contribution guidelines live (URL, `CONTRIBUTING.md` + path, Slack channel), and record their answer in the report + body under a "PR submission notes" subsection — step 3 uses this + as the starting point. + 4. Fetch the skill files (if URL); vendor to + `vendored-skill/` for read-only use by downstream steps. + 5. Web-search the underlying technology; identify trigger + conditions, success criteria, key terminology, intended audience + 6. Write structured report covering: what the skill does, who uses + it, when it should fire, what tools it depends on, what concepts + the user must understand. The `classification` frontmatter field + takes one of the canonical types (`tool-use`, `code-patterns`, + `document`, `prose-guidance`, `meta`, `interactive`) when one + fits, or a short descriptive label of the subagent's choosing — + a specific label is preferred over a vague `other`. +- **Dispatches:** functionality-researcher subagent (limited context: + source skill + targeted web fetches) +- **Handoff:** "Next, invoke `skill-optimizer-investigate-test-case`. + Note: this report's `pr_submission_intent` field tells step 2's + handoff whether step 3 (`investigate-submissions`) should run." +- **Iteration behavior:** re-run when source skill changes or when + the operator wants fresh research with new directives. + Fresh-derivation step — the subagent doesn't read its own + canonical or its git history. Re-running overwrites + `01-functionality.md`; prior state is in git history. Downstream + steps detect the change via `git log` mtime comparison. + +### 2. `skill-optimizer-investigate-test-case` + +- **Description trigger:** "design tests for this skill", "propose + test cases", "what should we test" +- **Input:** `01-functionality.md` +- **Output (two artifacts):** + - **`02-test-proposals.md`** — one-time audit report from the + subagent: ranked list of proposed functionalities, with + importance + suggested_probes + why-it-matters for each. This + file captures the design reasoning; it is not consulted by + downstream steps. + - **`tests//spec.yaml`** — one folder per + proposed functionality, each containing a single `spec.yaml` + with fields: `name`, `description`, `picked: true|false` + (initially set per the subagent's recommendation; user edits to + finalize), `importance`, `suggested_probes: [list]`, + `why_test`. The filesystem IS the state — there is no + `picked: []` array anywhere. +- **Behavior:** enumerate the skill's responsibilities from + functionality report; design 1–N functionalities to test; flag + edge cases; suggest probes per functionality; rank by importance. +- **Dispatches:** test-case-designer subagent (limited context: + `01-functionality.md` + the current `tests//spec.yaml` + tree if present + `${OPERATOR_DIRECTIVES}`; does NOT see skill + source content, `07-analysis.md`, optimizer attempts, failure + data, or git history of the tree). Maintenance step — the + subagent extends the tree (adds new functionality folders; + modifies existing spec.yaml files per directives) rather than + replacing it. +- **User gate:** after the subagent returns, the operator session + shows the user the proposed functionalities + ranking, and the + user edits `picked: true|false` in each `spec.yaml`. Step 4 will + build probes for `picked: true` functionalities only. +- **Handoff:** read `01-functionality.md`'s `pr_submission_intent` + field. If `true` → "Invoke `skill-optimizer-investigate-submissions` + next, then `skill-optimizer-write-tests`." If `false` → "Skip step + 3; invoke `skill-optimizer-write-tests`." +- **Iteration behavior:** re-run when the user wants different + coverage, when step 4/5/6 surface a coverage gap, or when step 1 + changed. Maintenance step — existing functionality folders the + user has invested in (especially `picked: true` ones) are + preserved across re-runs unless a directive explicitly asks for + revision. Destructive edits (removing a functionality) checkpoint + via git commit before the change. + +### 3. `skill-optimizer-investigate-submissions` (OPTIONAL — upstream only) + +- **Description trigger:** "research PR conventions for this skill", + "what does the upstream repo require" +- **Input:** source slug (`//`) +- **Output:** `03-submissions.md` — license, CLA, frontmatter spec, + file-location rules, prefix taxonomy, PR-shape patterns from recent + merged PRs, branch target, rejection signals from closed-without- + merge PRs. (Same shape as v1.3's research-upstream subagent output.) +- **Behavior:** gh-CLI heavy (PR list — both merged and + closed-without-merge, repo-file API, CONTRIBUTING, sanity-test + source); produce verbatim-pastable context block for the + validator +- **Dispatches:** submission-researcher subagent (limited context: + public repo facts only) +- **Skipped when:** `pr_submission_intent: false` in + `01-functionality.md` (i.e., local skill OR user said no to the PR + question at step 1) +- **Handoff:** "Validator in `improve-skill` will read this for + external consistency check. Continue with `write-tests` if not done." +- **Iteration behavior:** rarely needs re-running — upstream PR + conventions change slowly. Fresh-derivation step. Re-run when the + upstream repo's CONTRIBUTING/CLA changes or when + `01-functionality.md` changed (the skill source URL itself may + have changed). Overwrites `03-submissions.md`; prior state in git + history. + +### 4. `skill-optimizer-write-tests` + +- **Description trigger:** "build the tests", "implement the + workbench", "set up the eval" +- **Input:** `01-functionality.md`, `tests//spec.yaml` + tree (from step 2, picked subset determined by each spec's + `picked: true|false`) +- **Output:** for each picked functionality, one or more probe + folders at + `tests///{spec.yaml, workspace/, + grader.mjs, smoke/}` + a generated `tests/suite.yml` that lists + the probes step 6 will run. +- **Behavior:** + 1. Walk `tests/` for `picked: true` functionalities. + 2. For each, decide the probe set (informed by + `suggested_probes` in the functionality spec.yaml + any + `${OPERATOR_DIRECTIVES}` for that functionality). + 3. Show the user the planned probe set per functionality; user + confirms or revises. + 4. Dispatch parallel test-writer subagents (one per probe); + each builds its probe's workspace + grader + smoke fixtures. + 5. Run the smoke check on each probe's grader. + 6. Regenerate `tests/suite.yml` from the current picked + functionalities' probes. +- **Dispatches:** test-writer subagent per probe (limited context: + single probe's `spec.yaml` from step 2 + parent functionality's + `spec.yaml` + `01-functionality.md` + skill source content; does + NOT see other probes' specs/graders/workspaces or the eval grader's + matching logic — prevents copying across probes and grader-leak + hacking. Source access is granted because concrete fixture + writing needs specific violation patterns; the probe spec from + step 2 bounds the gerrymandering risk.) +- **Parallelizable:** each probe independent; dispatch in a single + message. +- **Destructive-edit safety:** before rebuilding an existing probe + (e.g., per a directive), the operator session commits the current + state so git history has a clean before/after breakpoint. +- **Handoff:** "Invoke `skill-optimizer-validate-tests` to check + probe quality before benching." +- **Iteration behavior:** re-run when step 2's tree changed (new + `picked: true` functionalities; revised functionality specs that + warrant probe rebuilds), when a test-writer flagged unimplementable + probes, when step 5's validator flagged probes for revision, or + when bench results suggest probes are systematically too easy or + too narrow. Maintenance step — existing probe folders are + preserved across re-runs unless a directive explicitly asks to + rebuild a specific probe. + +### 5. `skill-optimizer-validate-tests` + +- **Description trigger:** "validate the tests", "check the + graders", "are these probes fair", "review the test suite" +- **Input:** `tests/` tree (probes from step 4) + + `01-functionality.md` + source skill +- **Output:** `05-tests-verdict.md` — per-probe verdicts + (approve/needs-revision/reject) + aggregate frontmatter + (`all_probes_approved`, counts). Step 6 refuses to fire if + `all_probes_approved: false`. +- **Behavior:** dispatch test-validator subagents in parallel + (one per probe). Each judges: does the workspace exercise the + parent functionality? is the grader correct and fair? do the + smoke fixtures truly distinguish (vs. coincidentally match)? + Aggregate verdicts. +- **Dispatches:** test-validator subagent per probe (limited + context: this probe's full contents, parent functionality spec, + `01-functionality.md`, skill source; does NOT see other + probes, prior verdicts, downstream analysis) +- **Why this exists:** the smoke check at step 4 only verifies + syntactic consistency (grader correctly classifies the GOOD/BAD/ + EMPTY fixtures the test-writer also wrote). Semantic issues — + workspace doesn't exercise the responsibility, grader unfair, + false-positive probes — slip through. Grader bugs propagate to + misleading bench results and ducktape-shaped improvements. An + independent validator parallel to step 9 closes this gate. +- **Handoff (two branches):** if `all_probes_approved: true` → + "Invoke `skill-optimizer-run-bench`." If `false` → distill + per-probe verdicts into directives, re-run step 4 for affected + probes, re-run this step. Don't proceed to bench against bad + probes. +- **Iteration behavior:** re-run when step 4 produced new or + revised probes, or when user wants a fresh test-validation pass + with directives. Fresh-derivation — validator doesn't read its + own canonical or git history of it. + +### 6. `skill-optimizer-run-bench` + +- **Description trigger:** "run the eval", "measure", "benchmark" +- **Input:** `tests/suite.yml` (generated by step 4) + source skill + (vendored) +- **Output (two artifacts):** + - **`06-bench-results//`** — raw CLI output: + `suite-result.json` + per-trial traces + per-trial findings.txt. + Timestamped per run; old runs preserved naturally. + - **`06-bench-summary.md`** — single canonical aggregate: + overall pass rate, per-model pass rate, per-probe pass rate, + failed-probe pointer list, pointer to the latest + `06-bench-results//`. Fresh-derivation per run; prior + summary in git history. +- **Behavior:** invoke skill-optimizer CLI + (`npx tsx /src/cli.ts run-suite ./tests/suite.yml + --trials 3 --out 06-bench-results//`); parse the suite result; + write the summary. +- **Dispatches:** none — direct CLI invocation +- **Note:** this is intentionally thin; the entire v1.3 run-suite + logic stays as-is in the CLI +- **Handoff:** "Invoke `skill-optimizer-analyze`." +- **Iteration behavior:** re-run when `tests/` changed (step 4 added + or revised probes) or when the operator wants fresh trial data. + Each run produces a new timestamped raw directory; the summary is + overwritten and git tracks the prior version. The raw + bench-results dirs are outside the iteration protocol (they're + naturally accumulating snapshots, not versioned artifacts). + +### 7. `skill-optimizer-analyze` + +- **Description trigger:** "analyze the results", "diagnose what + failed", "find structural weaknesses" +- **Input:** `06-bench-results//` + `workbench/` + source skill +- **Output:** `07-analysis.md` — structured per the format below +- **Behavior:** cluster failures (per-rule, per-model, per-trial, + per-pattern); separate flaky (single-trial randomness) from + systematic (repeated across trials and/or models); for each + systematic cluster, find the responsible skill section, hypothesize + cause, articulate the general principle that WOULD address it AND + the anti-patterns that would NOT (anti-ducktape gate); list + non-structural noise separately; **if no structural weakness can be + articulated, the report says so explicitly and the next step refuses + to fire** +- **Dispatches:** analyzer subagent (limited context: per-trial + findings + skill content + workbench cases; does NOT see the test + inputs themselves — forces focus on the SKILL, not the test data) +- **Handoff:** if at least one structural weakness identified → "Invoke + `skill-optimizer-improve`." Otherwise → exit honestly ("no + structural weakness; no improvement warranted"). +- **Iteration behavior:** re-run when bench results change or when + the operator/user wants a fresh look with new framing. + Fresh-derivation — the subagent doesn't read its own canonical + `07-analysis.md` or its git history. Re-running overwrites the + canonical; prior state in git. The `${OPERATOR_DIRECTIVES}` slot + carries hints like "focus on the gpt-5 cluster" or "the user + thinks weakness X is actually two separate issues" — additional + analytical lenses, never a context dump of prior conclusions. + +**`07-analysis.md` format (Option A — structured):** + +```markdown +## Structural weaknesses identified + +### Weakness 1: + +- **Pattern**: Across trials, systematically failed to + detect . Specifically: . +- **Hypothesized cause**: +- **Connects to skill section**: at +- **What WOULD address this**: +- **What WOULD NOT address this** (anti-ducktape gate): + +### Weakness 2: ... + +## Non-structural noise (ignored — not addressable) + +- 1 gemini transient API error on multi-redirect (not a pattern) +- 1 gpt-5 timeout (infrastructure, not skill) +``` + +### 8. `skill-optimizer-improve` + +- **Description trigger:** "improve this skill", "fix the structural + weakness", "optimize" +- **Input:** `07-analysis.md` (REQUIRED — refuses if no weakness + identified), `01-functionality.md`, current skill state + (`improved-skill/` if it exists, else original source), optionally + `03-submissions.md` +- **Output:** `08-improvement-proposal.md` (the optimizer's proposed + change + rationale referencing the structural weakness). Step 7 + does NOT materialize `improved-skill/` — that's step 9's job + after validator approval. +- **Behavior:** + 1. Refuse if `07-analysis.md` has `has_structural_weakness: false` + — print "no weakness to address" and exit + 2. Dispatch **optimizer subagent** with limited context (see + "Subagent constraints" table). Required: address the named + structural weakness using a general principle from the + analysis, NOT a pattern-match patch. Output the proposed + change + rationale that explicitly references which named + weakness it addresses, plus a self-check against the "What + WOULD NOT address this" anti-pattern list. +- **Dispatches:** optimizer subagent, limited context, isolated + from raw trial data +- **Out of scope:** validating the proposal (that's step 9) and + packaging the change as a PR draft (PR composition is a separate + downstream concern; the auto-pilot at step 10 or a dedicated + composer can handle it if `pr_submission_intent: true`). +- **Single-shot per invocation:** no in-step revision loop. If + step 9 returns `needs-revision`, the operator (or auto-pilot) + distills the validator's rationale into a directive and + re-invokes step 8. +- **Handoff:** "Next, invoke `skill-optimizer-validate` + to check the proposal independently." +- **Iteration behavior:** re-run when `07-analysis.md` changed + (new analysis = potentially different weakness), when step 9 + returned `needs-revision`/`reject` with a distillable rationale, + or when the operator wants a fresh optimization attempt. + Fresh-derivation — the optimizer doesn't read its own canonical + (`08-improvement-proposal.md`) or git history of it, and doesn't + read step 9's prior verdicts. `${OPERATOR_DIRECTIVES}` carries + the distilled lessons. + +### 9. `skill-optimizer-validate` + +- **Description trigger:** "validate the improvement", "check the + proposal", "is the optimizer's change sound", "review the fix" +- **Input:** `08-improvement-proposal.md` (REQUIRED), + `01-functionality.md`, current skill state (`improved-skill/` + if it exists, else original source), optionally + `03-submissions.md` +- **Output:** `09-validator-verdict.md` (the verdict + + rationale). On `verdict: approve`, ALSO materializes + `improved-skill/` by applying the optimizer's diff to a copy of + the current state. **The original source is never modified** — + vendored copies stay frozen, local source files stay untouched. + `improved-skill/` accumulates across approved iterations; git + tracks its history. +- **Behavior:** + 1. Dispatch **validator subagent** with limited context. + - **Internal consistency check:** does the proposed change + make sense given the skill's stated responsibilities in + `01-functionality.md`? Additive vs. destructive? General + vs. ducktape? + - **External consistency check (only if `03-submissions.md` + exists):** does the change conform to upstream PR rules + (frontmatter, file location, prefix taxonomy, + additive-only, etc.)? Forward-looking — verifies the + improved skill COULD be turned into a valid PR, even + though neither step 8 nor step 9 produces one. + - Verdict: `approve` / `needs-revision` / `reject`. + 2. On `verdict: approve`: materialize `improved-skill/` by + applying the diff to a temp copy of the current state, then + atomically place at the canonical path. Commit. + 3. On `verdict: needs-revision` or `reject`: do NOT + materialize. Surface honestly per the handoff branch. +- **Dispatches:** validator subagent, limited context, isolated + from raw trial data + the optimizer's reasoning trace +- **Single-shot per invocation:** no in-step revision loop. Each + invocation produces one verdict; the 7→8→7 cycle on + `needs-revision` is operator-driven (or auto-pilot-driven at + step 10). +- **Handoff (three branches by verdict):** + - **approve:** "Validation complete; improvement applied at + `improved-skill/`. The chain has reached its natural endpoint + for this iteration. Review, copy locally, hand to a PR + composer, or re-bench." + - **needs-revision:** "Validator says needs-revision. Distill + the rationale into a directive and re-invoke step 8, then + re-invoke this step." + - **reject:** "Validator rejects the proposal outright. Two + paths: (1) re-invoke step 7 with a reframed weakness; (2) + accept that this weakness isn't addressable and exit + honestly." +- **Iteration behavior:** re-run when `08-improvement-proposal.md` + changed (step 8 produced a new proposal). Fresh-derivation — + the validator doesn't read its own canonical + (`09-validator-verdict.md`) or git history of it. + Independence-from-self is load-bearing: if the validator saw + its prior verdict it would gravitate toward + consistency-with-itself across re-validation cycles, defeating + the point of re-running. + +### 10. `skill-optimizer-autopilot` + +- **Description trigger:** "auto-pilot this skill", "run the whole + chain on X", "skill-optimizer end-to-end for X", "automated + improvement run for X" +- **Input:** same as step 1 (source skill — URL or local path), plus + optional flags: `pr_intent`, `max_iterations_per_step`, + `pick_top_n` (for B2's user-picks gate) +- **Output:** an end-of-run summary report at + `docs/skill-optimizer//autopilot-summary-.md` listing + each step's final version, headline result, and any blockers +- **Behavior:** + 1. Walks 1→9 in order, dispatching each chain skill. + 2. For each step: check whether the existing artifact is current + via `git log` mtime comparison against direct upstream. If + current, skip; if missing or stale, dispatch. + 3. Handles the three human-gate points with default policies: + - **B1 PR-intent question** (upstream skills) → uses the + `pr_intent` flag value (default `false`) + - **B2 user-picks gate** → sets `picked: true` on top-N + functionalities by importance (default `pick_top_n: 5`) + - **B8 validator-rejected verdict** → on `needs-revision`, + distills the rationale into a directive and re-invokes + step 8 then step 9 (up to `max_iterations_per_step` rounds). + On `reject`, logs the final state and surfaces blocker in + the summary; does not retry automatically. + 4. Bounds iterations: at most `max_iterations_per_step` re-runs + per step (default 2) to cap runaway loops. This is also what + caps the 7→8→7 revision loop on `needs-revision` verdicts. + 5. Surfaces blockers (subagent BLOCKED status, validator + unresolvable, missing inputs) in the summary rather than + halting the whole run. +- **Dispatches:** the eight chain skills as ordinary subagents + (each chain skill internally dispatches its own narrow-context + subagents). +- **Caveats** (baked into SKILL.md): expect modest results compared + to operator-driven runs. The nine steps are hard even with human + judgment; auto-pilot is best for batch processing where some + failures are acceptable, not for high-stakes single-target + optimization. +- **Iteration behavior:** auto-pilot is itself iterable. Re-running + picks up at whatever step is stale per the git-mtime comparison; + steps that are current are skipped. The summary report is + timestamped per run rather than treated as a versioned artifact — + each auto-pilot run produces a fresh summary so the audit trail + is preserved. + +## Plugin packaging + +The 7 skills + subagent prompts + reference material ship as part of +the existing `skill-optimizer` Claude Code plugin +(`.claude-plugin/plugin.json`). They sit alongside the existing +`skills/skill-optimizer/SKILL.md`. + +**Codex / Cursor ports:** v1.4 implementation targets Claude Code +first. Codex and Cursor have similar skill concepts; porting is a +separate workstream (out of scope for v1.4). + +## Coexistence with v1.3 + +The v1.3 `skills/auto-improve-orchestrator/` skill is **deprecated but +retained** for backward compatibility during the v1.4 rollout: + +- The v1.3 `auto-improve-orchestrator/` skill never landed on + `development` (it lives on `feat/auto-improve-skill-v1.3` and a few + experimental branches), so there's no on-branch artifact to mark + deprecated. The intended deprecation banner (initial plan task A3) + was skipped during execution. Existing v1.3 context files under + experimental branches remain useful as draft inputs to v1.4's step + 3, but they're not in v1.4's lineage. +- v1.3's `lessons.md` was originally planned as the seed for a + `references/recipes.md` shared pattern library. That was deferred + (see "Architecture overview" / Plan §"Task A4") — the raw seed is + case-study-shaped and a curated pattern version needs real + end-to-end observations. + +After v1.4 is validated on a few skills, the original +`skills/skill-optimizer/SKILL.md` (the direct workbench-CLI wrapper) +can be removed or repurposed in a separate cleanup PR — under the +v1.4 chain its role is filled by `skill-optimizer-run-bench`. + +## Acceptance criteria + +For v1.4 to be considered done: + +1. All 7 chain SKILL.md files + the 8th auto-pilot SKILL.md exist + with the interface described in this spec, plus appropriate + "Next, invoke X" handoff instructions on the chain skills +2. Subagent prompt templates exist in + `skills/skill-optimizer-subagents/` with the limited-context + constraints from the "Subagent constraints" table, and each + reasoning subagent accepts an `${OPERATOR_DIRECTIVES}` slot +2b. `skills/skill-optimizer-shared/iteration-protocol.md` exists and + is referenced explicitly (with a "Read this now" instruction) + from each chain skill's "Handle iteration" step +3. `skills/skill-optimizer-shared/workflow.md` documents the chain visually +4. ~~`references/recipes.md` is seeded from v1.3's `lessons.md`~~ — + deferred; the analyzer / optimizer subagents operate without a + pre-loaded recipe library until real end-to-end observations + surface curated patterns +5. **Iteration mechanism works end-to-end** — re-running a step on + an existing slug correctly overwrites its canonical artifact + (prior state in git history); the next downstream step on next + invocation detects the upstream change via `git log` mtime + comparison and re-runs itself +6. **End-to-end test on a local skill** (e.g., one of the existing + `skills/skill-optimizer/SKILL.md` or a small new skill in this + repo) — walks 1→2→4→5→6→7→8→9, produces all expected reports, modifies + the target skill, validator approves +7. **End-to-end test on an upstream skill** — re-run firecrawl (the + v1.3 regression case) under v1.4; expected: optimizer either + produces a principled fix OR honestly refuses (no regression + shipped) +8. **Auto-pilot smoke test** — running the auto-pilot on a local + skill end-to-end produces a `autopilot-summary-.md` covering + all nine steps with their final versions and any blockers +9. Plugin manifest unchanged in shape (still + `.claude-plugin/plugin.json`); skills auto-discoverable via the + existing plugin loading mechanism + +## Out of scope (deferred to v1.5 or later) + +- **Codex / Cursor plugin ports** — v1.4 ships Claude Code first; + multi-IDE comes later +- **Programmatic auto-pilot wrapper** — the natural-language "walk + through all 7" pattern is sufficient for v1.4 +- **Per-recipe SKILL.md prose enrichment** — the implementation phase + fills in the prose; later iterations add more pattern guidance to + each SKILL.md based on accumulated usage. The spec defines the + interface, not the full prose. +- **Cost-tracking** — v1.4 uses the operator's Claude Code plan + (subagents free per plan) and the OpenRouter cost is tracked by + the existing CLI. No new cost-budget logic in the skills. +- **Cross-skill recipe sharing** — recipes from one skill's iteration + could inform another's analyzer. Possible future feature; not + designed in v1.4. + +## Open questions (tracked but deferred) + +1. **When does a SKILL.md grow recipes vs reference an external + `recipes.md`?** Initial answer: shared recipes would go in a + curated `recipes.md` (location TBD when the file is actually + produced — see Plan §"Task A4" for the deferral); per-skill + nuances go inline. Will revisit based on usage. +2. **What if the optimizer subagent's revision loop exceeds the max + rounds (2)?** Initial answer: surface as `validator-rejected` and + require human intervention. Could add a "human-help-required" + status. Will revisit. +3. **How does step 7's analyzer subagent distinguish "no pattern" from + "pattern but I missed it"?** Initial answer: if the analyzer can't + articulate a weakness, the next step refuses to fire — we don't + force improvement. The user can re-invoke step 7 with hints if they + disagree with the analyzer's verdict. + +## Provenance + +- v1.3 design: `docs/auto-improve-skill-v1.3-spec.md` +- v1.3 implementation: branch `feat/auto-improve-skill-v1.3` + (PR #50) +- v1.3 PR drafts that surfaced the critiques: PR #51 (drafts #1, #5, + #6); the firecrawl regression (uncommitted, in the + `agent-abe33dd2c200c608a` worktree) is the canonical + "ducktape-by-monolithic-orchestrator" case study +- Brainstorming session: 2026-05-12 (this spec is the output) diff --git a/docs/superpowers/specs/2026-05-25-multi-agent-acp-design.md b/docs/superpowers/specs/2026-05-25-multi-agent-acp-design.md new file mode 100644 index 0000000..dd54eb2 --- /dev/null +++ b/docs/superpowers/specs/2026-05-25-multi-agent-acp-design.md @@ -0,0 +1,490 @@ +# Multi-agent workbench via Agent Client Protocol + +**Date:** 2026-05-25 +**Status:** design (brainstormed with `superpowers:brainstorming`) +**Supersedes:** the embedded-pi-agent runtime in +`src/workbench/pi-agent.ts` and the `--agent` mode of +`src/workbench/container-runner.ts`. + +## Goal + +Run skill-optimizer's bench against five real agent CLIs instead of one +embedded pi-agent library: Claude Code, OpenAI Codex, Google Gemini, +OpenCode, and pi (rehosted via pi-acp). Adopt the Agent Client Protocol +(ACP) as the uniform runtime interface, with the host driving each agent +inside a per-trial sandboxed container. + +## Why + +skill-optimizer today calls `@mariozechner/pi-coding-agent` as an +in-process library inside the Docker container. That gives one runtime +flavor — bench results reflect how a skill behaves under pi-agent as a +proxy, not under the agents real users actually load it in. The chain's +analyzer/optimizer/validator make decisions based on this proxy data. + +ACP is Zed's standardized JSON-RPC-over-stdio protocol for driving +coding agents. The official TypeScript SDK +(`@agentclientprotocol/sdk@0.22.1`) ships a stable client. +benchflow/skillsbench has already validated the pattern in production +across the same five agents. Adopting ACP gives skill-optimizer: + +- Realistic measurement: a skill's behavior under Claude Code is what + matters when shipping a skill targeting Claude Code users. +- Apples-to-apples cross-agent comparison: same trace schema across all + five agents. +- One uniform code path: no more "pi via library + others somehow." +- Future-proof: new ACP-supporting agents become registry entries, not + runtime work. + +## Architecture + +### Host / container split + +Host owns all bench orchestration logic. The container is a pure +sandbox: it holds the agent CLI binary, the read-only auth files, the +skill under test, and `/work`. It has no bench logic of its own. + +```text +HOST + docker-runner.ts + - resolve agent from registry + - detect subscription-auth files on host + - prepare workspace + - docker run -d agent-sandbox container (per-trial fresh) + - docker exec -i -> stdio pipes + - ACP client (from @agentclientprotocol/sdk): + initialize -> session/new -> session/prompt + <- session/update events streamed live + - write raw ACP messages verbatim to trace.jsonl + - on prompt completion: docker exec graders + - write result.json + summary.json + - docker rm -f container + +CONTAINER (per-trial, fresh from cached image) + - /opt/skill-opt/agents/ (all 5 CLIs pre-installed at image build) + - /home/agent/.claude/ (auth files + skill under test, ro) + - /work (rw, agent's cwd) + - /case (ro, bundled case) + - /results (rw, graders write here) + - agent process, spawned by docker exec, speaks ACP to host +``` + +### Per-trial container lifecycle + +Each trial gets a fresh container. The Docker **image** is built once +and cached (all five agent CLIs baked in at image build time); the +**container** is per-trial, ensuring full state isolation: + +| State | Mechanism | Isolated? | +|---|---|---| +| Agent conversation memory | ACP `session/new` per trial | yes | +| Filesystem (`/work`) | Container fresh, per-trial tmp mount | yes | +| Agent local state | Container fresh | yes | +| Auth files | Bind-mounted read-only | yes | +| Skill under test | Bind-mounted read-only per trial | yes | +| MCP servers | Sidecar containers per-trial on private network | yes | +| Agent CLI binaries | Image-baked, shared across trials | shared by design | + +`install_cmd` from each agent's registry entry runs at **image build +time** — not per trial. Per-trial `docker run` is ~1s startup, not 30s +of `npm install`. + +`launch_cmd` runs at trial time via `docker exec -i`, stays alive as +the ACP server until the trial completes. + +## Agent registry + +One file: `src/workbench/agents/registry.ts`. TypeScript port of +[benchflow's +`registry.py`](../../../skillsbench/.venv/lib/python3.12/site-packages/benchflow/agents/registry.py). +Per-agent quirks (e.g. `ANTHROPIC_AUTH_TOKEN` vs `_API_KEY`, opencode's +`provider/model` ID format, codex's base-url shell expansion) are +direct ports — those are battle-tested. + +```typescript +export interface AgentConfig { + name: string; + description: string; + installCmd: string; // bash, runs at IMAGE BUILD time + launchCmd: string; // bash, runs per-trial via docker exec + requiresEnv: string[]; // API key env vars + apiProtocol: 'anthropic-messages' | 'openai-completions' + | 'openai-responses' | ''; + envMapping: Record; + skillPaths: string[]; // e.g. ["$HOME/.claude/skills"] + credentialFiles: CredentialFile[]; + homeDirs: string[]; + subscriptionAuth: SubscriptionAuth | null; + acpModelFormat: 'bare' | 'provider/model'; + supportsAcpSetModel: boolean; +} + +export const AGENTS: Record = { + 'claude-agent-acp': { ... }, + 'codex-acp': { ... }, + 'gemini': { ... }, + 'opencode': { ... }, + 'pi-acp': { ... }, +}; + +export const AGENT_ALIASES: Record = { + 'claude': 'claude-agent-acp', + 'codex': 'codex-acp', + 'pi': 'pi-acp', +}; +``` + +Resolution: `resolveAgent(name)` looks up via alias map, then registry, +throws with fuzzy-match suggestion on unknown. + +## Case + suite schema + +### case.yml + +`agent:` is a **required** field. The case loader fails loudly if +missing. No backwards-compatible default. + +```yaml +name: review-product-card +agent: claude-agent-acp # REQUIRED +model: claude-haiku-4-5-20251001 # agent-native model ID +task: | + Review /work/ProductCard.tsx ... + +graders: [...] +setup: [...] +cleanup: [...] +env: [...] +mcpServers: {...} # per-agent native config writing (see MCP section) +mcpServices: {...} +``` + +`model:` field semantics: **agent-native** model ID, not a normalized +form. Each agent expects its own ID format: + +- `claude-agent-acp` → `claude-haiku-4-5-20251001` +- `codex-acp` → `gpt-5-mini` +- `gemini` → `gemini-3.1-pro-preview` +- `opencode` → `google/gemini-3.1-pro-preview` (provider/model) +- `pi-acp` → `openrouter/anthropic/claude-haiku-4-5` (unchanged) + +### suite.yml + +Matrix shape: `runs:` is the **only** matrix dimension. No `models:` +sugar. trials × runs.length = total trials. + +```yaml +name: my-suite +runs: + - agent: claude-agent-acp + model: claude-haiku-4-5-20251001 + - agent: claude-agent-acp + model: claude-sonnet-4-6-20250929 + - agent: codex-acp + model: gpt-5-mini + - agent: gemini + model: gemini-3.1-pro-preview + +cases: + - name: extract-pdf-facts + task: ... + graders: [...] +``` + +### Trial directory naming + +`------/` (gains the `` segment). + +## Skill deployment + +The skill under test is mounted **only at the agent's native skill +path** (Option A from the brainstorming). No `/work//` mounting. + +| Agent | Mount point inside container | +|---|---| +| claude-agent-acp | `/home/agent/.claude/skills//` | +| codex-acp | `/home/agent/.agents/skills//` | +| gemini | `/home/agent/.gemini/skills//` | +| opencode | `/home/agent/.opencode/skills//` | +| pi-acp | `/home/agent/.pi/agent/skills//` | + +Mounted read-only via `-v ::ro`. + +Task prompts describe the **user's actual task** ("Review +/work/ProductCard.tsx"), not "use the skill at X to do Y." This is the +realistic invocation pattern — the agent decides whether to trigger the +skill based on frontmatter `description`. "Skill didn't trigger" then +becomes a measurable weakness class, not a harness convenience. + +Chain implication: `skills/write-tests/agents/test-writer.md` updates +to drop "use the skill at /work/X" boilerplate from generated task +prompts. + +## Authentication + +Resolution priority per agent at trial start: + +1. **Subscription auth (host CLI login files)** — if the agent's + `subscriptionAuth.detectFile` exists on host (e.g., + `~/.claude/.credentials.json`), copy it into the container's + `$HOME` read-only. Done. +2. **API key from env** — if subscription file absent, require each + var in `requiresEnv` to be set on host. Pass into container via + `-e ENV_NAME` and applied per `envMapping`. +3. **Fail loudly** — if neither present, abort the trial with a + clear message indicating how to authenticate (login command or env + var name). + +Per-agent file paths: + +| Agent | Subscription detect path | +|---|---| +| claude-agent-acp | `~/.claude/.credentials.json` | +| codex-acp | `~/.codex/auth.json` | +| gemini | `~/.gemini/oauth_creds.json` (with siblings, see note) [1] | +| opencode | (no subscription auth; provider API keys only) | +| pi-acp | (no subscription auth; provider API keys only) | + +Subscription files are mounted **read-only**. The skill-optimizer +process on host never reads file contents — just checks existence and +hands the path to Docker. + +[1] Gemini's OAuth login writes three files that must all be copied +together: `~/.gemini/oauth_creds.json`, `~/.gemini/settings.json`, +`~/.gemini/google_accounts.json`. The detect path is the first; all +three are listed in the agent's `subscriptionAuth.files`. + +## Trace pipeline + +### Raw ACP capture + +`trace.jsonl` is the **raw ACP wire format**. One JSON-RPC envelope per +line, plus a `trace_start` header with trial metadata. + +```jsonl +{"type":"trace_start","caseName":"...","agent":"claude-agent-acp","model":"...","startedAt":"..."} +{"jsonrpc":"2.0","id":1,"method":"initialize","params":{...}} +{"jsonrpc":"2.0","id":1,"result":{...}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"...","update":{"sessionUpdate":"agent_message_chunk","content":{...}}}} +{"jsonrpc":"2.0","method":"session/update","params":{...,"update":{"sessionUpdate":"tool_call","toolCallId":"...","kind":"execute","content":{...}}}} +... +``` + +No normalization layer. The `WorkbenchTraceEntry` type is deleted. + +### Parse helpers + +`src/workbench/parse-trace.ts` exposes typed views over the raw trace: + +```typescript +iterMessages(trace): IterableIterator<{role, text?, thinking?, stopReason?, usage?}> +iterToolCalls(trace): IterableIterator<{id, kind, args, result?, isError?}> +computeMetrics(trace): WorkbenchMetrics +getFinalAssistantMessage(trace): string | undefined +getFailureEvidence(trace): string[] +``` + +`computeMetrics` returns the existing `WorkbenchMetrics` shape minus +the `cost` field (dropped permanently — tokens × downstream pricing +table if anyone needs cost). + +### Agent-internal trace archive + +In addition to `trace.jsonl`, we `docker cp` the agent's home dot-dir +at trial end into `/agent-internal/`. This captures whatever +the agent's own log machinery wrote (Claude Code's session jsonl, +Codex's session log, etc.) for human debugging. Bonus archive — chain +skills do not read this. + +```text +/ +├── trace.jsonl # canonical, raw ACP wire format +├── agent-internal/ # bonus archive, agent-specific format +│ └── (claude/projects/... | codex/sessions/... | ...) +├── result.json # graded outcome +├── summary.json # derived from trace.jsonl +└── workspace/ # filesystem state (on fail or --keep) +``` + +If disk pressure shows up later, gate `agent-internal/` behind +`--keep-internal-trace` (default off). For v1 it is always on. + +## MCP handling + +Per-agent native config writing. case.yml `mcpServers:` / +`mcpServices:` declarations stay the same; the harness translates them +into each agent's native MCP config file before launch: + +| Agent | Native MCP config target | +|---|---| +| claude-agent-acp | `~/.claude.json` `mcpServers` field | +| codex-acp | `~/.codex/config.toml` `[mcp_servers.*]` blocks | +| gemini | `~/.gemini/settings.json` `mcpServers` | +| opencode | `~/.config/opencode/opencode.json` `mcp` field | +| pi-acp | TBD at implementation (see note) [2] | + +MCP sidecar containers (the `mcpServices:` declaration in case.yml) +keep the existing pattern: spun up per-trial on a private Docker +network, agents reach them over HTTP/SSE. + +Implementation file: `src/workbench/acp/mcp-config-writer.ts`. One +small writer function per agent. Each writer is mechanical (case +config → agent-native shape) and unit-testable against fixture cases. + +[2] pi-acp's native MCP config target needs verification at +implementation time. benchflow's pi-acp registry entry doesn't +document an MCP config path explicitly. Three resolution options for +the implementer to pick from after inspecting pi-acp source: +(a) reuse the existing mcporter pattern (write `mcporter.json`, +inject `MCPORTER_CONFIG` env into the launch shim), (b) write to +pi-agent's native config location if pi-acp exposes one, (c) skip +MCP for pi-acp in v1 (fail loudly if a case using pi-acp declares +`mcpServers:`) and defer. Whichever path is chosen, surface in the +plan as an explicit task with the inspection step preceding the +implementation. + +## Chain integration + +Affected chain skills: + +- **`skills/write-tests/agents/test-writer.md`** — drop "use the skill + at /work/X" boilerplate from generated task prompts. Task prompts + describe the user's task only. + +- **`skills/analyze/agents/analyzer.md`** — analyzer reads ACP-format + `trace.jsonl` instead of normalized `WorkbenchTraceEntry`. Add + pointer to new `skills/shared/acp-trace-format.md` reference. + +- **`skills/run-bench/SKILL.md`** — `06-bench-summary.md` gains + `agent:` column alongside `model:`. Drops cost. Adds tokens + + duration + tool-call counts per probe and overall (closing the + pre-existing metrics gap). + +- **`skills/shared/acp-trace-format.md`** — NEW reference doc (~50 + lines) summarizing ACP `session/update` event types relevant to + skill-behavior analysis, plus pointers at the official ACP spec. + +## Code surface + +### New files + +- `src/workbench/agents/registry.ts` — `AgentConfig` type + 5 entries + aliases +- `src/workbench/agents/install-snippets.ts` — per-agent bash for the Dockerfile +- `src/workbench/acp/client.ts` — wrapper around `@agentclientprotocol/sdk` +- `src/workbench/acp/transport.ts` — `docker exec -i` stdio bridge +- `src/workbench/acp/auth.ts` — subscription detection + mount resolution +- `src/workbench/acp/mcp-config-writer.ts` — case → per-agent MCP config +- `src/workbench/acp/skill-deploy.ts` — mount skill to agent's `skillPaths[0]` +- `src/workbench/parse-trace.ts` — ACP trace helpers +- `docker/skill-optimizer-agent.Dockerfile` — replaces `workbench-runner.Dockerfile` +- `skills/shared/acp-trace-format.md` — chain analyzer reference +- `tests/smoke-agents/.spec.ts` — one smoke probe per agent + +### Deleted + +- `src/workbench/pi-agent.ts` +- `--agent` mode of `src/workbench/container-runner.ts` +- `WorkbenchTraceEntry` type + normalization code in `trace.ts` +- System-prompt logging +- mcporter injection into pi-agent (replaced by `mcp-config-writer.ts`) + +### Kept (modified as needed) + +- `src/workbench/container-runner.ts` — `--setup` and `--grade` modes only +- `src/workbench/docker-runner.ts` — heavily updated to orchestrate ACP host-side +- `src/workbench/trials.ts` — aggregation surfaces tokens + duration +- Workspace prep, MCP sidecar networking, cleanup orchestration + +## Migration + +### Existing tracked cases + +- `examples/workbench/pdf/suite.yml`, `examples/workbench/mcp/suite.yml` + — convert `models:` to `runs:`. Add `agent: pi-acp` per run + (these use OpenRouter refs that pi-acp accepts unchanged). +- Other `examples/workbench/*/.run.log` and stale result directories — + remove (past run artifacts, not source). +- Any `skill-evals//...` directories — past run artifacts, can + be removed entirely. Future probes generated by the updated + test-writer subagent are correct from creation. + +### Existing chain-skill docs + +Updated in the chain integration section above. Mechanical edits, no +schema migration. + +## Testing strategy + +### Per-agent smoke probes (5) + +Each of the 5 agents gets one minimal probe in `tests/smoke-agents/`: + +- Task: "write 'hello' to /work/out.txt; report what you did to + /work/findings.txt" +- Grader: assert `out.txt` exists with content `hello`, assert + `findings.txt` mentions writing it +- Verifies the full lifecycle: image-time install, container launch, + ACP handshake, prompt, tool call (write), termination, trace + well-formedness + +Runs in CI on every PR. Failure of any smoke probe blocks merge. + +### Pi-acp regression probe + +Run the existing `examples/workbench/pdf/suite.yml` through pi-acp +end-to-end. Compare pass rates against a captured baseline from the +last known-good direct-pi-agent run. Catches behavioral drift from the +runtime swap. + +### Unit tests + +- Registry resolution (alias → config, fail-loud on unknown) +- Auth resolution (subscription detected → uses it; missing → falls + back to env; both missing → throws) +- `parse-trace.ts` helpers against fixture ACP traces from each agent +- MCP config translation (one fixture case per agent → assertion on + emitted config file) + +## Out of scope (v1) + +- Cost computation (dropped permanently; tokens are the measurement) +- Web-tools disabling per agent (registry fields exist, no CLI flag + yet — adds when needed) +- Non-Docker sandbox backends (Daytona, e2b, firecracker, k8s) +- Runtime agent registration (hardcoded registry is enough for now) +- Custom agents beyond the 5 + +## Risks + +1. **Per-agent quirks surface only at runtime.** Budget ~1 day per + agent for smoke-test debugging. Known unknowns from benchflow's + experience: codex base-url shell expansion, opencode + `provider/model` ID format, gemini's multi-file subscription auth, + pi-acp's launcher shim for OpenRouter routing. + +2. **ACP version skew.** Zed's SDK (`@agentclientprotocol/sdk@0.22.1`) + and each agent CLI must agree on `session/update` semantics. We + pin SDK version; if a specific agent CLI version breaks, document + the compatible range in the agent's registry entry. + +3. **Subscription auth file paths vary across agent versions** (Claude + Code CLI vs Desktop login, gemini OAuth flow changes). Stick with + benchflow's tested paths; fall back to API key if subscription + detection fails. + +4. **Docker image size grows.** Five agent CLIs ≈ 200-300MB on top of + the base. Mitigation: multi-stage Dockerfile, agent install layer + cached. Acceptable for a controlled workbench. + +5. **Chain re-validation.** Existing `06-bench-summary.md` consumers + (analyzer subagent) need updated mental model. Pi-acp regression + probe catches the major drift; chain skill prompts get updated in + the same plan. + +## Open questions + +None at design time — all design questions resolved during +brainstorming. Implementation will surface tactical questions +(specific ACP message shapes per agent, exact docker exec semantics +for long-lived processes) which the writing-plans skill should +sequence as discrete tasks. diff --git a/docs/superpowers/specs/2026-05-27-autopilot-design.md b/docs/superpowers/specs/2026-05-27-autopilot-design.md new file mode 100644 index 0000000..8ce5d40 --- /dev/null +++ b/docs/superpowers/specs/2026-05-27-autopilot-design.md @@ -0,0 +1,600 @@ +# skill-optimizer autopilot — design spec + +**Status:** approved (brainstormed 2026-05-27 with the +`superpowers:brainstorming` skill) +**Fulfills:** v1.4 acceptance criterion #8 (auto-pilot smoke test) — +the only criterion from the v1.4 design that was never built. The +chain was deliberately structured to slot a driver in cleanly; this +spec defines what that driver does. +**Supersedes:** v1.4 design §10 (`skill-optimizer-autopilot`). That +section described a single-cycle retry loop with halt-on-honest-exit +defaults. This spec replaces it with a strategy-menu model that +defaults to grinding for an improvement until either it lands or +caps are exhausted. + +## Goal + +A single user-invocable skill at `skills/autopilot/SKILL.md` that +runs the full 9-step skill-optimizer chain end-to-end on one skill, +fully automated by default, with optional per-gate breakpoints that +the operator can opt into at startup. The pilot follows the same +re-run and directive-distillation mechanics the chain already uses +for manual operation — autopilot is the chain's tenth skill, not a +parallel pipeline. + +## Motivation + +The chain works step-by-step today, but driving it manually across +nine steps is the bottleneck for batch-processing skills. The v1.4 +design anticipated this and reserved step 10 for a driver, but +deferred the build. Two prior attempts inform this design: + +1. **v1.3 `tools/auto-improve-skill.mjs`** — a wrapper script that + spawned `claude -p` with a 5-phase prompt. Worked for batch runs + (the May 9 batch-2 hit 8 of 10 targets) but predates the v1.4 + chain decomposition. Retiring with this spec. +2. **v1.4 §10** — sketched a SKILL.md interface with `pr_intent`, + `pick_top_n`, `max_iterations_per_step` flags. The retry policy + was single-cycle ("on `needs-revision`, distill and retry up to + N rounds"). Subsequent team review surfaced that recovery isn't + single-cycle: when the skill is too easy or no weakness is + found, the right move depends on what's available (more picks? + harder probes? reframe?). This spec replaces the single-cycle + model with a strategy menu per gate. + +Both legacies inform what to keep: a SKILL.md (not a wrapper +script), per-gate retry caps, distillation of validator rationales +into directives. What changes: each retry-eligible gate has a menu +of recovery strategies the pilot picks from based on state, not a +fixed cycle. The default disposition is **grind for an improvement +until something lands or the cap is exhausted**, not "honest-exit +on first no-finding." + +## Audience + +The autopilot's target user is an operator who wants the +skill-optimizer applied to one skill without understanding the +internals of the chain. The startup interview is written for that +audience — plain language, no skill-optimizer jargon, breakpoints +described as natural questions ("if the skill passes every test, do +I push for harder probes or accept it's strong?"). Skills internals +people can still use it, but the UX is tuned for the operator who +just wants results. + +Autopilot is for **single-skill optimization**. Batch processing +multiple skills is a separate orchestrator and out of scope here. + +## Architecture overview + +```text +skills/ + investigate-functionality/ # step 1 + investigate-submissions/ # step 2 (optional) + design-tests/ # step 3 + write-tests/ # step 4 + validate-tests/ # step 5 + run-bench/ # step 6 + analyze/ # step 7 + improve/ # step 8 + validate/ # step 9 + autopilot/ # step 10 — NEW (this spec) + shared/ +``` + +Autopilot dispatches the nine chain skills sequentially via the +**Skill tool** (not the Agent tool). Each chain skill, when +invoked, internally dispatches its own narrow-context subagents via +the Agent tool — autopilot doesn't touch that machinery. The +operator session running autopilot is the single place where +backward-trigger rationales get distilled into +`${OPERATOR_DIRECTIVES}` for the next dispatch. + +**Why Skill-sequential, not Agent-nested.** If autopilot dispatched +the chain skills as agents-within-an-agent, the outer pilot +couldn't read each step's canonical output to distill directives +for the next step. Skill-sequential keeps the operator session as +the mediator, matching how the chain already works during manual +operation. + +State writes go to the same paths the chain already uses, plus one +new pair specific to autopilot: + +```text +docs/skill-optimizer// + 01-functionality.md # existing chain output + ... (steps 2-9 outputs) + autopilot-summary-.md # NEW — tracked, per-run summary + +.skill-optimizer// + vendored-skill/ # existing chain output + bench-results// # existing chain output + autopilot-/ # NEW — gitignored, per-run scratch + journal.md # appended-to during the run +``` + +The journal is gitignored (ephemeral scratch); the summary is +tracked (the audit record). Promotion from journal to summary +happens at end-of-run. + +## The startup interview + +The interview captures these values into the run journal's +frontmatter: + +| Captured | Type | Default | Asked when | +|---|---|---|---| +| `pr_intent` | bool | `false` | source is upstream URL | +| `pick_top_n` | int | `5` | always | +| `max_retries_per_gate` | int | `2` | always | +| `global_retry_cap` | int | `8` | always | +| `surfaced_gates` | set of gate IDs | `{G6-CLI-FAIL}` | always (operator may add more) | +| `default_overrides` | map | `{}` | always (operator may override any gate's default behavior) | + +When the operator invokes autopilot, the pilot: + +1. **Classifies the source** as upstream URL or local filesystem + path. (Reuses step 1's classification logic — see + `skills/investigate-functionality/SKILL.md`.) +2. **Presents the interview prompt** (below) as a single message. +3. **Records the operator's responses** into the run journal's + frontmatter. Anything not changed keeps its default per the + table above. +4. **Initializes the run journal** at + `.skill-optimizer//autopilot-/journal.md`. +5. **Fires the forward walk.** + +Interview prompt: + +```text +I'm going to optimize end-to-end. By default I'm fully +automated — I run all 9 chain steps and retry up to 2 times at each +recovery point. I'll only pause and ask if the bench environment +breaks (docker/agent crash — that's not something to fix without +you). + +Optional breakpoints — pick any to be in the loop on: + +- Reviewing picked test cases before bench runs. Default: top 5 by + importance are picked automatically. +- A probe gets blocked or rejected during test-writing. Default: + drop the affected probe and continue; the others still run. +- Validate-tests says one or more probes need revision. Default: + distill the rationale into a directive and retry the probe up to + the cap. +- The skill passes every test or no structural weakness gets found. + Default: try recovery strategies in order — first reframe the + analysis, then more bench trials, then enable more picks, then + propose new probes — until something surfaces or the cap is hit. +- The optimizer gets stuck on a weakness it can't translate to a + concrete fix. Default: reframe the weakness and retry up to the + cap. +- The validator says the proposed fix needs revision. Default: + distill the rationale and retry the proposal up to the cap. +- The validator outright rejects the proposed fix. Default: try + recovery strategies in order — first a different fix for the same + weakness, then reframe the weakness, then add more probe coverage + — until something lands or the cap is hit. +- A recovery point has multiple strategies available (meta-breakpoint + covering the three above). Default: I pick the cheapest unattempted + strategy. Surfacing this means I'll show you the menu and let you + choose. + +A few defaults you can also override: + +- PR intent: no (only matters for upstream skills) +- Top-N test cases picked: 5 +- Max retries per recovery point: 2 + +Say 'all auto, go' to just run, or tell me what to change. +``` + +The interview is one turn. After the operator responds, no further +mid-run prompts fire except for surfaced gates (always-surfaces +G6-CLI-FAIL or anything the operator opted into). + +## Control flow + +```text +Phase 1: setup + - Classify source + - Run startup interview + - Initialize run journal + +Phase 2: forward walk (steps 1 → 9) + For each step: + - Check staleness vs direct upstream (git-mtime) + - If artifact is current AND no operator directive forces rerun → skip + - Else: dispatch via Skill, wait for completion + - After completion: check for gate firings (see below) + +Phase 3: gate handling (on any gate firing) + - Read gate's menu from SKILL.md + - Read journal for prior attempts at this gate + - If gate is in surfaced_gates: prompt operator, follow their answer + - Else: pick cheapest unattempted strategy + - Append journal entry + - Increment per-gate counter; check caps + - If cap exhausted at this gate OR global cap exhausted: + honest-exit (do NOT fail; this is principled) + - Else: execute strategy (re-dispatch the relevant chain step(s) + with ${OPERATOR_DIRECTIVES} distilled from the gate's rationale) + - Resume forward walk + +Phase 4: termination + - Reach step 9 verdict: approve → exit_status: improved + - Cap exhaustion at a gate → exit_status: unchanged-honest-exit + - Infra failure (G6-CLI-FAIL) the operator didn't recover → exit_status: blocked-on-infra + - Write autopilot-summary-.md (promote from journal) +``` + +### Re-entrancy + +Re-invoking autopilot on the same slug picks up where it left off +via git-mtime staleness: artifacts ≤ direct upstream are current +and skipped; everything else re-runs. The new invocation gets its +own `autopilot-/` journal directory and its own +`autopilot-summary-.md`. **Caps reset per run**, not per slug — +two separate invocations each get `max_retries_per_gate = 2`. + +The prior run's outputs are preserved naturally by their +timestamps. This matches the chain's general "filesystem IS the +state; history is git" discipline. + +## Gates and recovery menus + +A **gate** is a decision point in the chain where autopilot might +need to take action beyond just dispatching the next step. + +### Gate definitions + +| ID | Fires when | Retry policy | +|---|---|---| +| G3 | Step 3 has emitted N proposals; picks need to be set | Single action — mark top-`pick_top_n` by importance; not a retry gate | +| G4-BLOCKED | Step 4's test-writer subagent returns BLOCKED on a probe | No retry — skip, log, continue | +| G5-REVISION | Step 5 verdict is `needs-revision` on ≥1 probe | Menu (1 option), retry-eligible | +| G5-REJECT | Step 5 verdict is `reject` on ≥1 probe | No retry — drop the probe, continue | +| G6-ALLPASS | Step 6 bench is 100% pass across every model+probe | Not a halt gate; continue to step 7 (G7-NW handles actual recovery) | +| G6-CLI-FAIL | Step 6 CLI/infra failure (docker, agent crash) | No retry — always surfaces | +| G7-NW | Step 7 reports `has_structural_weakness: false` | Menu (4 options), retry-eligible | +| G8-BLOCKED | Step 8 optimizer subagent returns BLOCKED | Menu (2 options), retry-eligible | +| G9-REVISION | Step 9 verdict is `needs-revision` | Menu (1 option), retry-eligible | +| G9-REJECT | Step 9 verdict is `reject` | Menu (3 options), retry-eligible | + +### Recovery menus + +Each retry-eligible gate has a menu of strategies, ordered +cheap → expensive. The pilot picks the cheapest **available + +unattempted** option at each gate firing. + +#### G5-REVISION + +1. Distill validator rationale → re-run step 4 for affected probes + only → re-run step 5 + +#### G7-NW (cheap → expensive) + +1. **Reframe analysis angle.** Re-run step 7 with a directive like + "look at marginal failures, model-specific drift, latent issues + even if pass-rate is high." Cheap: 1 step re-runs. +2. **More bench trials.** Re-run step 6 with extra trials per probe + (e.g., 3× instead of default 1×) to surface variance, then + re-run step 7. Medium: a full bench cycle. +3. **Enable more picks.** Flip next-highest-importance unpicked + functionalities to `picked: true` in step 3's spec.yaml, then + re-run steps 4 → 5 → 6 → 7 for the new ones (existing probes + stay). Medium-high: incremental probe build + full bench. +4. **Add new probes.** Re-run step 3 with a directive "propose new + functionalities the existing pass missed; focus on edge cases the + existing probes don't cover," then 4 → 5 → 6 → 7. Expensive: + fresh design work + full bench. + +#### G8-BLOCKED (cheap → expensive) + +1. **Reframe weakness.** Re-run step 7 with a directive "the + optimizer was blocked on weakness W; please reformulate more + concretely or at a different abstraction," then re-run step 8. +2. **Sharper directive.** If the weakness reads concrete enough but + the optimizer struggled with the general principle, re-run step + 8 alone with a sharpened directive (e.g., "make the principle + more specific to how the skill is structured today"). + +#### G9-REVISION + +1. Distill validator rationale into a directive → re-run step 8 → + re-run step 9 + +#### G9-REJECT (cheap → expensive) + +1. **Different fix, same weakness.** Re-run step 8 with the rejected + proposal noted as anti-pattern ("proposal P rejected because Y; + try a different general principle for the same weakness"). +2. **Reframe weakness.** Re-run step 7 with the validator's rejection + rationale "weakness X rejected because Y; find a different angle + on what's failing," then 8 → 9. +3. **Add coverage.** If the validator rejected on grounds that the + weakness wasn't real in the trace data, treat as G7-NW menu + (more trials, more picks, more probes). + +### Reasoning protocol at each gate firing + +```text +1. Read the gate's menu from this SKILL.md. +2. Read journal.md for prior attempts at this gate (strategies + already tried this run). +3. If gate ID is in `surfaced_gates`: + - Surface in plain language to operator + - Show the menu options + reasoning for each + - Wait for operator decision (pick strategy / supply directive + / accept honest-exit) +4. Else: pick cheapest unattempted strategy from the menu. +5. Append journal entry: timestamp, gate ID, strategy chosen, + reason, retry counter (e.g., "3/4 strategies remaining"). +6. Distill the gate's rationale into ${OPERATOR_DIRECTIVES} (atomic + list, no context dump — per shared/subagent-dispatch.md). +7. Re-dispatch the relevant chain step(s) with that directive. +8. On the next gate firing (which may be the same gate again): + start from step 1 of this protocol. +``` + +### Cap accounting + +- **Per-gate cap** (`max_retries_per_gate`, default 2). Counts + total retry events for that gate type, regardless of which menu + strategy was picked. Two G7-NW retries means strategies 1+2 were + attempted; strategies 3+4 aren't reached. +- **Global cap** (`global_retry_cap`, default 8). Total retry + events across all gates. Safety net against thrashing. +- **Cap exhaustion is an honest exit**, not a failure. The journal + records what was tried; the summary explains the principled + outcome. + +## State tracking — the run journal + +Single markdown file at +`.skill-optimizer//autopilot-/journal.md`, gitignored. +Appended-to throughout the run. The pilot reads it before each gate +decision; the journal IS the pilot's working memory across the run. + +```markdown +--- +run_started_at: 2026-05-27T12:34:56Z +slug: anthropics-skills-pdf +source: https://github.com/anthropics/skills/tree/main/pdf +source_kind: upstream | local +mode_settings: + pr_intent: false + pick_top_n: 5 + max_retries_per_gate: 2 + global_retry_cap: 8 + surfaced_gates: [G6-CLI-FAIL] + default_overrides: {} +--- + +## Run journal + +- 12:35:01Z step 1 dispatched (investigate-functionality) +- 12:36:14Z step 1 complete; 01-functionality.md written +- 12:36:14Z step 3 dispatched (design-tests) +- 12:37:42Z step 3 complete; G3 fired +- 12:37:42Z G3 strategy 1 (top-5 by importance) — auto-applied +- ... +- 13:02:15Z step 7 complete; G7-NW fired (has_structural_weakness: false) +- 13:02:15Z G7-NW strategy 1 (reframe) | reason: cheapest unattempted | retry: 1/2 +- 13:02:16Z step 7 re-dispatched with directive "...look at marginal failures..." +- 13:03:48Z step 7 complete; G7-NW fired again (still false) +- 13:03:48Z G7-NW strategy 2 (more trials) | reason: 1 attempted | retry: 2/2 +- ... +- 13:18:02Z honest-exit: G7-NW cap exhausted across 2 strategies (1+2) +- 13:18:02Z run ended + +## Retry counters + +- G7-NW: 2/2 (strategies tried: 1=reframe, 2=more-trials) +- G9-REVISION: 0/2 +- ... + +## Global retry counter + +- Total retries: 2/8 +``` + +Format is markdown with frontmatter for the run config + appended +log lines + counters block. Human-readable; the pilot parses the +counters block on each gate firing. + +## End-of-run summary report + +Written to `docs/skill-optimizer//autopilot-summary-.md` +(tracked). Promoted from the journal at end-of-run. Concise audit +record, not a dump of every event. + +```markdown +--- +run_started_at: 2026-05-27T12:34:56Z +run_ended_at: 2026-05-27T13:18:02Z +slug: anthropics-skills-pdf +source: https://github.com/anthropics/skills/tree/main/pdf +exit_status: improved | unchanged-honest-exit | blocked-on-infra +total_retries: 2 +journal_path: .skill-optimizer//autopilot-/journal.md +--- + +# Autopilot run for `anthropics-skills-pdf` + +## Headline + +Skill-improvement run ended in honest-exit. G7-NW cap exhausted after +2 recovery strategies (reframe + more trials). Analyzer found no +structural weakness; nothing to ship. + +## Per-step outcomes + +| Step | Outcome | Artifact | +|---|---|---| +| 1 investigate-functionality | written | docs/.../01-functionality.md | +| 3 design-tests | written, 5 picked | docs/.../03-test-proposals.md | +| 4 write-tests | 5 probes built | skill-evals/.../ | +| 5 validate-tests | all approved | docs/.../05-tests-verdict.md | +| 6 run-bench | 84% pass | docs/.../06-bench-summary.md | +| 7 analyze | no structural weakness (after 2 retries) | docs/.../07-analysis.md | +| 8 improve | skipped (no weakness) | — | +| 9 validate | skipped | — | + +## Recovery summary + +- G7-NW retries: 2/2 + - strategy 1 (reframe analysis) — analyzer still reported no weakness + - strategy 2 (more bench trials, ran 3× per probe) — same outcome +- All other gates: not fired + +## Caveats + +[anything the operator should know — flaky bench, suspicious trace + patterns, etc. Auto-pilot's interpretation, not a substitute for + operator review.] + +## What changed in the repo + +[file-list of new/modified canonicals so the operator can scan diffs.] +``` + +### exit_status enum + +- `improved` — step 9 approved; `improved-skill/` materialized at + `docs/skill-optimizer//improved-skill/` +- `unchanged-honest-exit` — cap exhausted at any gate, OR no + weakness with all G7-NW strategies exhausted, OR validator-reject + with all G9-REJECT strategies exhausted +- `blocked-on-infra` — G6-CLI-FAIL or another infra surface the + operator didn't recover from + +## SKILL.md shape + +```text +--- +name: autopilot +description: +--- + +# autopilot + +[purpose statement + audience expectations] + +## Before you start + +[upfront interview contract, with the plain-language prompt block + from this spec verbatim] + +## Workflow + +### (a) Classify source + run startup interview +### (b) Initialize the run journal +### (c) Forward walk through the chain (steps 1 → 9) +### (d) Handle gate firings +### (e) Honest-exit and write summary + +## Gate menus + +[the full menu table from this spec — load-bearing reference] + +## Reasoning protocol at each gate firing + +[the 8-step protocol from this spec] + +## Cap accounting + +[per-gate + global, definitions of "retry event"] + +## Surfacing a gate (interactive) + +[for surfaced gates: how to format the plain-language prompt to the + operator and parse the response] +``` + +### Description routing + +```text +description: Use when the user wants to run the entire skill-optimizer chain +end-to-end on a single skill without driving each step manually — phrases like +"autopilot this skill", "run the full chain on X", "skill-optimizer end-to-end +for X", "automated improvement run for X". Walks steps 1 → 9 with the iteration +mechanism, applies per-gate recovery strategies up to a configurable cap, and +honest-exits when no improvement is doable. Use even when the user doesn't +explicitly say "autopilot" — any phrasing about running the full chain +end-to-end on one skill should trigger this. +``` + +## Acceptance criteria + +1. `skills/autopilot/SKILL.md` exists with the structure above. The + plain-language interview block is verbatim from this spec. +2. Plugin metadata lists the new skill across all five provider + manifests (`.claude-plugin/`, `.codex-plugin/`, + `.cursor-plugin/`, `.opencode/`, `gemini-extension.json`). +3. `skills/shared/workflow.md` row 10 description matches the + as-built behavior (cap names, exit-status enum, journal + location). +4. `skills/shared/subagent-dispatch.md` step 2↔3 classification + swap is fixed (existing bug surfaced during this brainstorm). +5. **End-to-end smoke test** on a local skill (e.g., one of this + repo's own chain skills, or a tiny purpose-built skill): + autopilot walks 1 → 9, writes the summary, exits cleanly. + Acceptance: a real `autopilot-summary-.md` exists, journal + is coherent, retries (if any) match the per-gate and global + caps. +6. **Surfaced-gate smoke test**: invoke autopilot with one gate + explicitly surfaced (e.g., G7-NW); verify the operator-prompt + fires and the run resumes after operator response. +7. **Cap exhaustion smoke test**: rig a probe deliberately too-easy + so G7-NW fires; verify autopilot tries strategies 1 and 2, then + honest-exits with `exit_status: unchanged-honest-exit`. + +## Out of scope (deferred) + +- **Batch processing.** Autopilot is single-skill. A batch + orchestrator that loops autopilot over a list of slugs is a + separate workstream. +- **Cost tracking.** Autopilot inherits the chain's cost + characteristics (subagents are free per Claude Code plan; + OpenRouter usage tracked by the run-bench CLI). No new budget + logic at the autopilot layer. +- **Cross-run learning.** Each autopilot run is independent. The + `default_overrides` recorded in a run's journal don't propagate + to subsequent runs. Possible future feature. +- **Surfacing partial bench results.** During a long bench run, + autopilot doesn't surface intermediate trial results to the + operator. The bench is treated atomically (start → finish → + evaluate G6-* gates). +- **Multi-skill dependencies.** If a skill being optimized depends + on another skill (chain of plugins), autopilot doesn't follow + those dependencies. One slug per run. + +## Open questions + +1. **Should `default_overrides` be persisted to a per-user config?** + Right now they're one-shot per invocation. A user who always + wants `pick_top_n: 10` re-supplies that every run. A persistent + per-user config (e.g., `~/.skill-optimizer/autopilot.yml`) could + eliminate the repetition but adds another piece of state. Defer + until real usage shows the friction. +2. **How should autopilot interact with the v1.3 + `tools/auto-improve-skill.mjs` wrapper?** The v1.3 wrapper still + exists in the repo on some branches. Once autopilot lands and is + validated, the v1.3 wrapper can be removed in a separate + cleanup PR. No coexistence needed during rollout — they're for + different chain generations. +3. **Strategy menu evolution.** The menus in this spec reflect the + patterns visible today. As operators use autopilot across more + skills, new strategies may emerge (e.g., "lower a probe's + acceptance threshold by N% before re-bench" as a cheap G6-ALLPASS + move). The SKILL.md is the source of truth for the menus; new + strategies get added by edit, no architectural change. + +## Side cleanup flagged by this brainstorm + +`skills/shared/subagent-dispatch.md:37` says fresh-derivation steps +are `(1, 3, 5, 6-summary, 7, 8, 9)` and `:66` says maintenance steps +are `(2, 4)`. Reading each chain skill's own iteration section +confirms: step 2 is fresh-derivation, step 3 is maintenance. The +two doc lines have the step numbers swapped. Fix during the +autopilot implementation, not a blocker. diff --git a/examples/workbench/mcp/suite.yml b/examples/workbench/mcp/suite.yml index 5ac689a..6be750a 100644 --- a/examples/workbench/mcp/suite.yml +++ b/examples/workbench/mcp/suite.yml @@ -1,7 +1,8 @@ name: mcp-calculator-example references: ./references -models: - - openrouter/google/gemini-2.5-flash +runs: + - agent: pi-acp + model: openrouter/google/gemini-2.5-flash env: - OPENROUTER_API_KEY timeoutSeconds: 600 @@ -15,6 +16,7 @@ mcpServices: - calculator-server.mjs cases: - name: use-calculator-mcp + agent: pi-acp task: | Compute this expression: diff --git a/examples/workbench/pdf/suite.yml b/examples/workbench/pdf/suite.yml index 0dc13de..d6388d3 100644 --- a/examples/workbench/pdf/suite.yml +++ b/examples/workbench/pdf/suite.yml @@ -2,8 +2,9 @@ name: pdf-workbench-example references: ./references appendSystemPrompt: | Keep task outputs at the top level of /work unless the user asks for a different path. -models: - - openrouter/google/gemini-2.5-flash +runs: + - agent: pi-acp + model: openrouter/google/gemini-2.5-flash env: - OPENROUTER_API_KEY timeoutSeconds: 600 @@ -15,6 +16,7 @@ setup: - cp input/briefing-source.pdf briefing-source.pdf cases: - name: extract-pdf-facts + agent: pi-acp task: | Extract the key facts from statement.pdf and write answer.json with this exact schema: { @@ -30,6 +32,7 @@ cases: command: node $CASE/checks/extract-pdf-facts.mjs - name: split-customer-packet + agent: pi-acp task: | Create customer-copy.pdf from customer-packet.pdf. Include only the pages marked CUSTOMER COPY, in their original order. Exclude the page marked INTERNAL NOTES. The output must be a PDF, not a text file. graders: @@ -37,6 +40,7 @@ cases: command: node $CASE/checks/split-customer-packet.mjs - name: build-briefing-pdf + agent: pi-acp task: | Create briefing.pdf as a one-page PDF briefing based on briefing-source.pdf. It must include these exact lines: PDF Skill Briefing @@ -49,6 +53,7 @@ cases: command: node $CASE/checks/build-briefing-pdf.mjs - name: no-pdf-skill-needed + agent: pi-acp task: | Write note.txt with exactly this text: done diff --git a/package-lock.json b/package-lock.json index d985d28..91fb901 100644 --- a/package-lock.json +++ b/package-lock.json @@ -8,8 +8,8 @@ "name": "skill-optimizer", "version": "2.0.0", "license": "MIT", - "main": ".opencode/plugins/skill-optimizer.js", "dependencies": { + "@agentclientprotocol/sdk": "0.22.1", "@mariozechner/pi-agent-core": "^0.66.1", "@mariozechner/pi-ai": "^0.66.1", "@mariozechner/pi-coding-agent": "^0.66.1", @@ -29,6 +29,15 @@ "node": ">=20" } }, + "node_modules/@agentclientprotocol/sdk": { + "version": "0.22.1", + "resolved": "https://registry.npmjs.org/@agentclientprotocol/sdk/-/sdk-0.22.1.tgz", + "integrity": "sha512-DfqXtl/8gO9NImq094MTaCXEU2vkhh6v7q/kT+9UjZxUqj8hYaya2OjLVIqn16MzNHcXEpShTR2RIauLSYeDQQ==", + "license": "Apache-2.0", + "peerDependencies": { + "zod": "^3.25.0 || ^4.0.0" + } + }, "node_modules/@anthropic-ai/sdk": { "version": "0.73.0", "resolved": "https://registry.npmjs.org/@anthropic-ai/sdk/-/sdk-0.73.0.tgz", diff --git a/package.json b/package.json index 1f1be4c..961f1ed 100644 --- a/package.json +++ b/package.json @@ -79,9 +79,12 @@ "build": "tsc && chmod +x dist/cli.js", "typecheck": "tsc --noEmit", "lint": "tsc --noUnusedLocals --noEmit", - "test": "tsx tests/smoke-workbench-case.ts && tsx tests/smoke-workbench-checks.ts && tsx tests/smoke-workbench-trace.ts && tsx tests/smoke-workbench-container.ts && tsx tests/smoke-workbench-docker-runner.ts && tsx tests/smoke-workbench-pi-agent.ts && tsx tests/smoke-workbench-run-case.ts && tsx tests/smoke-workbench-models.ts && tsx tests/smoke-workbench-suite.ts && tsx tests/smoke-workbench-trials.ts && tsx tests/smoke-workbench-metrics.ts && tsx tests/smoke-skill-distribution.ts" + "dockerfile:print-installs": "tsx src/workbench/agents/install-snippets.ts", + "test": "tsx tests/smoke-workbench-case.ts && tsx tests/smoke-workbench-checks.ts && tsx tests/smoke-workbench-docker-runner.ts && tsx tests/smoke-workbench-run-case.ts && tsx tests/smoke-workbench-models.ts && tsx tests/smoke-workbench-suite.ts && tsx tests/smoke-workbench-trials.ts && tsx tests/smoke-workbench-metrics.ts && tsx tests/smoke-skill-distribution.ts && tsx --test tests/acp/sdk-smoke.test.ts && tsx --test tests/acp/transport.test.ts && tsx --test tests/acp/client.test.ts && tsx --test tests/agents/registry.test.ts && tsx --test tests/acp/auth.test.ts && tsx --test tests/acp/skill-deploy.test.ts && tsx --test tests/acp/mcp-config-writer.test.ts && tsx --test tests/parse-trace.test.ts && tsx --test tests/acp/trace-recorder.test.ts && tsx --test tests/case-loader-agent-required.test.ts && tsx --test tests/suite-loader-runs.test.ts && tsx --test tests/trials-aggregate.test.ts", + "test:smoke-agents": "tsx --test tests/smoke-agents/claude-agent-acp.smoke.test.ts && tsx --test tests/smoke-agents/codex-acp.smoke.test.ts && tsx --test tests/smoke-agents/gemini.smoke.test.ts && tsx --test tests/smoke-agents/opencode.smoke.test.ts && tsx --test tests/smoke-agents/pi-acp.smoke.test.ts" }, "dependencies": { + "@agentclientprotocol/sdk": "0.22.1", "@mariozechner/pi-agent-core": "^0.66.1", "@mariozechner/pi-ai": "^0.66.1", "@mariozechner/pi-coding-agent": "^0.66.1", diff --git a/skills/analyze/SKILL.md b/skills/analyze/SKILL.md new file mode 100644 index 0000000..991f342 --- /dev/null +++ b/skills/analyze/SKILL.md @@ -0,0 +1,147 @@ +--- +name: analyze +description: Use when the user wants to diagnose why a bench run produced failures — phrases like "analyze the results", "diagnose what failed", "find structural weaknesses", "why did the skill miss X". Triggers mid-way through skill-optimizer chain work, after `skill-optimizer:run-bench` has produced a `06-bench-summary.md` with at least one failed trial. Use even when the user doesn't explicitly say "analyze" — any phrasing about understanding bench failures should trigger this. +--- + +# analyze + +Step 7 of the skill-optimizer chain. **Fresh-derivation step.** Takes +the bench summary + raw trial output from step 6, dispatches an +analyzer subagent to cluster failures into named **structural +weaknesses** of the skill (or explicitly say there are none), and +writes `docs/skill-optimizer//07-analysis.md`. This is the +chain's **anti-ducktape gate**: step 8 refuses to fire unless this +report names at least one structural weakness, with the general +principle that WOULD address it and the anti-patterns that would NOT. + +## What you produce + +A single report at `docs/skill-optimizer//07-analysis.md`. + +Frontmatter (runtime-relevant facts only, per +[`frontmatter-discipline.md`](../shared/frontmatter-discipline.md)): + +```yaml +--- +has_structural_weakness: true | false +weakness_count: +bench_results_path: .skill-optimizer//bench-results// +--- +``` + +**`has_structural_weakness`** is load-bearing for step 8 — `false` +gates step 8 from firing ("no weakness to address"). Forcing +`true` when the analyzer found nothing is the ducktape failure +this step exists to prevent. + +Body has per-weakness sections + a non-structural-noise section + +an optional honest-refusal note. **Each per-weakness entry must +include five required parts**: Pattern, Hypothesized cause, +Connects to skill section, What WOULD address this, What WOULD NOT +address this. Step (d) verifies these five parts are present. + +The "What WOULD NOT address this" anti-pattern list is +**load-bearing architecture**. Without it, step 8's optimizer can +pattern-match a patch that fits the symptom without addressing the +cause; the validator (step 9) then has no explicit "this would be +a ducktape" signal to check against. Losing the anti-pattern list +breaks the gate. + +Full body template and reasoning protocol in +[`agents/analyzer.md`](./agents/analyzer.md). + +## Workflow + +### (a) Confirm prerequisites + +`06-bench-summary.md` must exist with `bench_results_path` and +`overall_pass_rate`. If not, tell the user to run +`skill-optimizer:run-bench` first. + +If `overall_pass_rate == 1.0`: there's nothing to analyze. Surface +honestly — either accept that probes don't expose a weakness, or +re-run step 3 with a "make probes harder" directive. Don't run +the analyzer; there are no failures to cluster. + +If the bench results dir is missing or `suite-result.json` is +malformed, surface as a step-6 problem (incomplete or corrupted +bench run) and tell the user to re-run step 6. + +`skill-evals//` and the source skill must also be available. + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Each invocation overwrites the canonical from +current bench data plus directives. Collect +`${OPERATOR_DIRECTIVES}` per the protocol — examples: "focus on +the gpt-5 cluster", "the user thinks weakness X is actually two +separate issues". + +If operator directives contradict each other, surface the +contradiction before re-dispatching. + +### (c) Dispatch the analyzer subagent + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT cluster failures or name weaknesses +yourself in this session** — dispatch the subagent via the `Agent` +tool. Render the prompt template at +[`agents/analyzer.md`](./agents/analyzer.md) +inline by substituting `${BENCH_RESULTS_PATH}`, `${SUMMARY_PATH}`, +`${TESTS_TREE_PATH}`, `${SKILL_SOURCE_PATH}`, `${OUTPUT_PATH}`, +and `${OPERATOR_DIRECTIVES}`, then pass the rendered text as the +Agent tool's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +The subagent sees: `06-bench-summary.md` (entry point with +failed-probe pointer list); `.skill-optimizer//bench-results//` with +per-trial `trace.jsonl` and `findings.txt`; each probe's +`spec.yaml` from the `skill-evals//` tree (intent only); the skill source +content; `${OPERATOR_DIRECTIVES}`. + +The subagent does NOT see: **the test inputs themselves** +(`skill-evals////workspace/` files) — this is load-bearing; the +analyzer must think about the SKILL, not the SOLUTIONS; its own +prior `07-analysis.md` or git history; `08-improvement-proposal.md`, +`09-validator-verdict.md`, or any prior optimizer attempts. + +**Why this matters:** two specific bias risks. **Solution-thinking** +— seeing the raw fixtures would lead the analyzer to recommend a +patch for that specific input shape (ducktape); blocking the +inputs forces it to reason about why the skill failed to instruct +the agent properly. **Prior-analysis bias** — the operator session +has read prior analyses and prior optimizer attempts; an +in-session analyzer would gravitate toward "what we said last +time" or defensively pivot away from it. Neither is the job. + +### (d) Confirm subagent output + +Verify the report parses, `weakness_count` matches the number of +`### Weakness :` sections, and each weakness has all five +required parts (especially "What WOULD NOT address this" — the +anti-ducktape signal step 8 needs). If anything's inconsistent or +missing, surface to the user; don't fill it in yourself. + +If the anti-pattern lists are empty/vague (anti-ducktape gate +compromised), re-dispatch with a directive: "each weakness needs +a concrete anti-pattern list — what specific moves should the +optimizer avoid?" + +### (e) Hand off + +Two messages depending on `has_structural_weakness`: + +- **`true`:** "Analysis complete. `` structural weakness(es) + identified. Next, invoke `skill-optimizer:improve`." +- **`false`:** "No structural weakness identified — failures + consistent with noise rather than a fixable defect. Step 8 will + refuse to fire. Either accept the conclusion, or re-invoke step + 7 with a directive ('the analyzer missed the X cluster')." + +If the subagent returned `false` but the user disagrees: surface +the disagreement, but do NOT pressure the subagent to manufacture +a weakness. Honest path is re-invoking step 7 with a directive +pointing at what the user thinks was missed. Don't auto-invoke +step 7 — per the no-auto-invocation rule, the user decides. diff --git a/skills/analyze/agents/analyzer.md b/skills/analyze/agents/analyzer.md new file mode 100644 index 0000000..a3676ca --- /dev/null +++ b/skills/analyze/agents/analyzer.md @@ -0,0 +1,200 @@ +# Analyzer subagent + +You are dispatched by `skill-optimizer:analyze` to read the bench +results, cluster failures into named **structural weaknesses** of +the skill (or explicitly say there are none), and write +`07-analysis.md`. This report is the chain's anti-ducktape gate: +step 8 (improve) refuses to fire unless this report names at least +one structural weakness with the general principle that WOULD +address it AND the anti-patterns that would NOT. + +**Read first:** +[`../../shared/skill-design-philosophy.md`](../../shared/skill-design-philosophy.md) +— the cross-vendor synthesis of what makes a skill good and what +makes an improvement principled vs ducktape. Your "What WOULD +address this" principle and "What WOULD NOT address this" +anti-pattern list should map to specific entries in that doc. If +your reasoning doesn't connect to a named principle or anti-pattern +there, your weakness framing is probably under-grounded. + +**Read first:** [`../../shared/acp-trace-format.md`](../../shared/acp-trace-format.md) +— `trace.jsonl` is raw ACP wire format. Use `parse-trace.ts` helpers +(`iterMessages`, `iterToolCalls`, `computeMetrics`, +`getFinalAssistantMessage`, `getFailureEvidence`) rather than reading +JSONL by hand. The format reference doc summarizes the event types +relevant to skill-behavior analysis. + +## Inputs (templated by the operator session) + +- `${SUMMARY_PATH}` — `06-bench-summary.md` (entry point; + failed-probe pointer list) +- `${BENCH_RESULTS_PATH}` — `.skill-optimizer//bench-results//` with + per-trial `trace.jsonl` and per-trial `findings.txt` +- `${TESTS_TREE_PATH}` — `skill-evals//` tree. **Read ONLY each probe's + `spec.yaml`** — what each probe was probing at the level of + INTENT. Do NOT read `workspace/` contents (raw input fixtures). +- `${SKILL_SOURCE_PATH}` — the skill's content at `.skill-optimizer//vendored-skill/` + (step 1 vendored the source regardless of upstream/local) +- `${OUTPUT_PATH}` — where to write `07-analysis.md` +- `${OPERATOR_DIRECTIVES}` — atomic new requirements (empty unless + this is a re-run with sharper guidance) + +## What you see + +- The bench summary + raw per-trial findings.txt + per-trial + trace.jsonl +- Each probe's `spec.yaml` (intent only — what it was probing) +- The skill source (so you can quote the responsible skill section + in each weakness entry) +- Operator directives + +## What you do NOT see + +- **The test inputs themselves** (`skill-evals////workspace/` + files). This is load-bearing — you must think about the SKILL + (what it instructs the agent to do), not the SOLUTIONS (what + the agent should have done in this specific input shape). + Reading raw inputs would lead you to recommend patches + tailored to specific inputs — the textbook ducktape mode. +- Your own prior `07-analysis.md` or git history of it. Each + invocation is a fresh derivation; consistency-with-self bias + would defeat re-analysis after operator directives sharpen the + framing. +- `08-improvement-proposal.md`, `09-validator-verdict.md`, or any + prior optimizer attempts (would bias your analysis toward + "weaknesses the prior optimizer tried to address"). +- The eval grader's matching internals (you reason from + pass/fail + findings + trace, not from grader source code). + +## Output: `${OUTPUT_PATH}` — `07-analysis.md` + +Frontmatter (runtime-relevant only): + +```yaml +--- +has_structural_weakness: true | false +weakness_count: +bench_results_path: .skill-optimizer//bench-results// +--- +``` + +**`has_structural_weakness`** is load-bearing. Step 8 refuses to +fire if `false`. Forcing `true` when you actually found nothing +is the ducktape mode this step exists to prevent. + +Body sections: + +### Structural weaknesses identified + +For each weakness, a section with **all five required parts** +(operator session verifies their presence): + +```markdown +### Weakness N: + +- **Pattern**: Across trials, systematically failed to + detect . Specifically: ///>. +- **Hypothesized cause**: . +- **Connects to skill section**: at + . +- **What WOULD address this**: . +- **What WOULD NOT address this** (anti-ducktape gate): . +``` + +### Non-structural noise (ignored — not addressable) + +Failures consistent with model nondeterminism, infrastructure +flakiness, transient API errors. List them so they're +acknowledged but exclude them from the weakness count. + +### Honest refusal (if applicable) + +If you cannot articulate at least one structural weakness: + +> No structural weakness identified. Failures observed are +> consistent with model nondeterminism / infrastructure noise / +> single-trial flakiness, not a fixable defect in the skill itself. +> Step 8 should not fire. + +Set `has_structural_weakness: false` in the frontmatter to match. + +## The "What WOULD NOT address this" list is load-bearing architecture + +Without it, step 8's optimizer can pattern-match a patch that +fits the symptom without addressing the cause; the validator (step +9) then has no explicit "this would be a ducktape" signal to check +against. Your job is to name BOTH the principle that should be +applied AND the ducktape moves the optimizer must avoid. + +If you can't name concrete anti-patterns for a weakness, the +weakness isn't sharp enough to warrant the optimizer firing. +Either sharpen it or drop it. + +## Reasoning protocol + +1. **Start at `${SUMMARY_PATH}`.** Read the failed-probe pointer + list. These are the probes with at least one trial failing. +2. **For each failed probe, walk its trials.** Read the + per-trial `findings.txt` and `trace.jsonl`. Distinguish: + - **Systematic** — same failure repeats across trials and/or + models (a real pattern) + - **Flaky** — one trial failed in a way other trials didn't + (noise, not pattern) +3. **Cluster systematic failures.** Group probes whose failures + share a common root cause. A cluster = a candidate weakness. +4. **For each cluster, locate the skill section** that SHOULD + have prevented it. Quote the section + path:line. If the + skill HAS the relevant section but it's worded in a way the + agent doesn't operationalize (declarative vs procedural), + that's the gap — name it. +5. **Articulate the general principle** that addresses the + weakness. Not a specific patch ("add a step to do X for these + specific inputs") — a principle ("teach the agent to enumerate + Xs in general, then check each one"). The optimizer at step 8 + applies the principle. +6. **Articulate the anti-patterns.** What ducktape moves would + appear to fix the symptom but miss the cause? Be specific: + "adding a regex pattern for these specific tokens"; "wrapping + the rule in a MUST/NEVER"; "restating the rule more + emphatically". These are what the optimizer is instructed to + avoid. +7. **Distinguish noise** explicitly. Don't bury infrastructure + errors or single-trial flakes in the weakness count. +8. **Honest refusal** if you can't articulate a weakness. Forcing + weaknesses to justify firing step 8 is the failure mode this + gate exists to prevent. + +## Operator directives + +Examples that count as atomic requirements: + +- "focus on the gpt-5 cluster — gemini and claude passed" +- "the user thinks weakness X from a prior run is actually two + separate issues — look for both" +- "include the trace from trial 3 of probe Y specifically" + +Examples that don't (context dumps — reject): + +- The full prior analysis pasted in for you to "reconcile" +- The prior optimizer's proposal pasted in + +## Return summary + +After writing `${OUTPUT_PATH}`, return: + +- `has_structural_weakness` (true/false) +- Weakness count +- Top weakness by trial-coverage (one-line name + cluster size) +- Honest-refusal flag (set if you declined to name weaknesses + because the failures were noise) + +Keep it under 200 words. The operator session uses this for the +handoff decision (proceed to step 8 vs. exit honestly). diff --git a/skills/autopilot/SKILL.md b/skills/autopilot/SKILL.md new file mode 100644 index 0000000..5f90490 --- /dev/null +++ b/skills/autopilot/SKILL.md @@ -0,0 +1,403 @@ +--- +name: autopilot +description: Use when the user wants to run the entire skill-optimizer chain end-to-end on a single skill without driving each step manually — phrases like "autopilot this skill", "run the full chain on X", "skill-optimizer end-to-end for X", "automated improvement run for X". Walks steps 1 through 9 with the chain's iteration mechanism, applies per-gate recovery strategies up to a configurable cap, and honest-exits when no improvement is doable. Use even when the user doesn't explicitly say "autopilot" — any phrasing about running the full chain end-to-end on one skill should trigger this. +--- + +# autopilot + +Step 10 of the skill-optimizer chain. **Chain driver, not a step.** +Takes a skill (URL or local path), walks the 9-step chain +sequentially via the `Skill` tool, handles each step's gate firings +with a strategy-menu, and writes a per-run summary at +`docs/skill-optimizer//autopilot-summary-.md`. + +Fully automated by default. Operator picks optional breakpoints at +startup; everything else auto-recovers up to a per-gate cap. The +only always-on breakpoint is bench infrastructure failure (docker +or agent crash) — autopilot doesn't recover from that. + +## What you produce + +Two outputs per run: + +1. **`.skill-optimizer//autopilot-/journal.md`** — the + per-run journal. Appended to throughout the run; this IS your + working memory across the chain. Gitignored. +2. **`docs/skill-optimizer//autopilot-summary-.md`** — + the per-run audit summary. Promoted from the journal at + end-of-run. Tracked. + +Multiple runs accumulate naturally with their own ``. Caps +reset per run; the new invocation gets fresh +`max_retries_per_gate`. + +## Workflow + +### (a) Confirm prerequisites and classify source + +The user must have provided a skill source (URL or local path). If +they invoked autopilot without a source, ask which skill to +optimize. + +Classify the source the same way step 1 does: + +- **Upstream:** URL or `//`. Triggers the + PR-intent question in the interview below. +- **Local:** filesystem path to a SKILL.md (or a directory + containing one). Skips the PR question. + +**Auto-handle the branch in full-auto mode.** Step 1 has its own +explicit branching prompt for manual invocation, but in autopilot +the operator already said "all auto" — interrupting with a branch +question contradicts that. Autopilot handles it up-front: + +1. Run `git rev-parse --abbrev-ref HEAD` to read the current branch. +2. If the current branch already matches `eval/` or `feat/`, + reuse it. +3. Otherwise, create `eval/` and switch to it: + `git checkout -b eval/`. +4. Log the decision to the journal: either + `auto-created branch eval/` or + `using existing branch `. +5. When dispatching step 1 (in workflow (d)), include in + `${OPERATOR_DIRECTIVES}` the line + `branching: already-resolved ()` so step 1 skips + its branching prompt. + +**Auto-resolve the optimization-target picker.** Step 1's +researcher subagent may identify multiple candidate optimization +targets (typically the source SKILL.md plus any underlying file +that the source fetches at runtime, for wrapper skills). In +manual invocation, step 1 surfaces a picker; in autopilot +full-auto mode, that's another contradiction. When dispatching +step 1, also include in `${OPERATOR_DIRECTIVES}` the line +`optimization-target: auto-pick-recommended` so step 1 honors +the researcher's recommendation without surfacing. + +If the operator picked the interactive surface for branching or +optimization-target (exceptions — autopilot's startup interview +doesn't surface either by default), prompt instead per +[`agents/surfacing-prompt.md`](./agents/surfacing-prompt.md) and +follow their answer. + +Create a TodoWrite list with all 10 chain steps so progress is +visible across the long run. Mark step 10 (this skill) as +`in_progress`; subsequent chain skills update their own entry as +they fire. + +### (b) Run the startup interview + +Load +[`agents/interview-prompt.md`](./agents/interview-prompt.md), +substitute `${SKILL_NAME}`, and present it to the user as a single +message. Wait for the response. + +Capture the response into the run journal frontmatter (see (c) for +the shape). Defaults if the user said "all auto, go": + +| Field | Default | Source | +|---|---|---| +| `pr_intent` | `false` | upstream source only | +| `pick_top_n` | `5` | always | +| `max_retries_per_gate` | `2` | always | +| `global_retry_cap` | `8` | always | +| `surfaced_gates` | `{G6-CLI-FAIL}` | operator may add more | +| `default_overrides` | `{}` | operator may override any gate default | + +The response-parsing rules (including the user-phrasing → gate-ID +mapping) live in `agents/interview-prompt.md`. + +### (c) Initialize the run journal + +```bash +TS=$(date -u +%Y%m%dT%H%M%SZ) +mkdir -p ".skill-optimizer//autopilot-${TS}" +``` + +Write `journal.md` with YAML frontmatter holding the `mode_settings` +fields from (b), plus three empty H2 sections to be appended to as +the run proceeds: `## Run journal` (event log), `## Retry counters` +(per-gate `N/max` lines), `## Global retry counter`. The journal is +your working memory across the run — append every dispatch, every +gate firing, and every retry decision. Read it before each gate +decision to know what's been tried. + +### (d) Forward walk through the chain (steps 1 → 9) + +For each step in order: + +1. **Staleness check.** Read the canonical artifact's git mtime + against direct upstream (per + [`iteration-protocol.md`](../shared/iteration-protocol.md)). If + current and no operator directive forces a re-run, skip and log + `step N skipped (current)` in the journal. +2. **Dispatch.** Call the chain skill via the `Skill` tool: + + ```text + Skill skill-optimizer:investigate-functionality + Skill skill-optimizer:investigate-submissions # only if pr_intent + Skill skill-optimizer:design-tests + Skill skill-optimizer:write-tests + Skill skill-optimizer:validate-tests + Skill skill-optimizer:run-bench + Skill skill-optimizer:analyze + Skill skill-optimizer:improve + Skill skill-optimizer:validate + ``` + + Each chain skill does its own work (subagent dispatch, file + writes); you wait for it to complete. +3. **Read the canonical.** After the chain skill returns, read its + canonical output (frontmatter is the load-bearing signal — see + gate triggers below). +4. **Check for gate firings.** Match the canonical against the gate + triggers in [Gate menus](#gate-menus). If any gate fires, go to + workflow step (e). Otherwise, continue to the next chain step. + +Step 2 (`investigate-submissions`) runs only if +`pr_intent: true` (set in the interview or carried from a prior +step 1 canonical). Skip otherwise. + +### (e) Handle gate firings + +Follow the [Reasoning protocol at each gate firing](#reasoning-protocol-at-each-gate-firing). + +The protocol resolves to one of: + +- **Execute a recovery strategy.** Re-dispatch one or more chain + skills with a distilled `${OPERATOR_DIRECTIVES}` derived from the + gate's rationale. Then resume the forward walk from the earliest + re-dispatched step. +- **Skip and continue.** For `No retry` gates (G4-BLOCKED, + G5-REJECT) — drop the affected probe(s), log, and resume the + forward walk where you left off. +- **Honest-exit.** Cap exhausted (per-gate or global) — go to + workflow step (f). +- **Surface.** Gate is in `surfaced_gates` — prompt the operator + per [Surfacing a gate](#surfacing-a-gate), then follow their + answer. + +Every gate firing gets one journal entry. Every retry decision +increments the gate's counter and the global counter. + +### (e.5) Empirical verification re-bench + +Fires only when step 9 returns `verdict: approve` AND +`docs/skill-optimizer//improved-skill/` exists. This phase +empirically demonstrates that the change actually helps — pass-rate +goes up, with no regressions. **Without this verification an +"improved" claim is unbacked.** + +1. **Snapshot baseline.** Read `06-bench-summary.md` frontmatter + for `overall_pass_rate` and `bench_results_path`; capture both + into the journal as `baseline_pass_rate` and + `baseline_bench_results_path`. The raw per-case results live at + `/suite-result.json` — preserve + that path; it's what we compare per-case against. +2. **Temporarily patch suite.yml.** Back up + `skill-evals//suite.yml` to + `.skill-optimizer//autopilot-/suite.yml.backup`. Then + add a top-level `skillUnderTest` field pointing at the improved + skill (the suite loader propagates this to all inline cases — + see `skills/shared/workbench.md` Suite Schema): + + ```yaml + skillUnderTest: + slug: + hostPath: //docs/skill-optimizer//improved-skill + ``` + +3. **Re-dispatch step 6** via `Skill skill-optimizer:run-bench`. + Output writes to a fresh + `.skill-optimizer//bench-results/-after-improvement/` + directory; `06-bench-summary.md` overwrites with the + after-improvement state (this is fine — it's now the latest + bench for the improved skill). +4. **Restore suite.yml** from the backup. The skill-evals tree + stays clean; future re-runs of the chain start from the original + suite. +5. **Compare.** Read the new `06-bench-summary.md` for + `overall_pass_rate` (after) and the new `bench_results_path`. + Per-case comparison: parse both `suite-result.json` files; for + each `(probe × run)` pair, note `baseline=pass/fail` vs + `after=pass/fail`. A `pass → fail` flip is a **regression**. +6. **Classify.** Decide which exit_status applies in (f) based on: + - `after_pass_rate > baseline_pass_rate` AND zero regressions → + `improved` + - `after_pass_rate > baseline_pass_rate` AND ≥1 regression → + `improved-with-regression` (surface clearly; the optimizer + bought one case at the cost of another) + - `after_pass_rate ≤ baseline_pass_rate` (or `==` with no + winning case flip) → **G10-NO-EMPIRICAL-GAIN fires** (see + gate menu) + +Record all four numbers into the journal: +`baseline_pass_rate`, `after_pass_rate`, +`pass_to_fail_regressions: []`, +`fail_to_pass_flips: []`. + +### (f) Honest-exit and write the summary + +Set `exit_status` per the run's final state: + +- **`improved`** — step 9 approve AND the empirical re-bench + showed strictly increased pass-rate with zero regressions. +- **`improved-with-regression`** — pass-rate up but ≥1 case + regressed. Distinct from clean `improved` because it's worth a + human eye before merging. +- **`unchanged-empirical`** — step 9 approved but the empirical + re-bench showed pass-rate ≤ baseline (or zero net gain), and + all G10 recovery strategies were exhausted. Validator believed + the proposal helps; the bench says it doesn't. +- **`unchanged-honest-exit`** — cap exhausted at some earlier + gate (G7-NW, G8-BLOCKED, G9-REVISION, G9-REJECT) before reaching + step 9 approval. The chain never produced an improved-skill. +- **`blocked-on-infra`** — G6-CLI-FAIL surfaced and the operator + did not recover the environment. + +Promote the journal into +`docs/skill-optimizer//autopilot-summary-.md` using +[`agents/summary-template.md`](./agents/summary-template.md). The +summary is an aggregate; trim narrative, keep retry trail + exit +reason + repo changes + the empirical-verification numbers +(baseline → after pass-rate, regressions list, flips list). + +Update the chain TodoWrite list to mark step 10 `completed` and +print the path to the summary report. + +## Gate menus + +A **gate** is a decision point in the chain where autopilot might +need to take action beyond just dispatching the next step. Gates +are detected from the chain skill's canonical frontmatter and body. + +### Gate definitions + +| ID | Fires when (frontmatter / body signal) | Retry policy | +|---|---|---| +| G3 | Step 3's `03-test-proposals.md` lists N functionalities; picks need to be set | Single action — mark top-`pick_top_n` by importance; not retry-eligible | +| G4-BLOCKED | Step 4 test-writer subagent returned `status: blocked` on a probe | No retry — skip, log, continue | +| G5-REVISION | `05-tests-verdict.md` has `needs-revision` on ≥1 probe | Menu (1 option) | +| G5-REJECT | `05-tests-verdict.md` has `reject` on ≥1 probe | No retry — drop those probes, continue | +| G6-ALLPASS | `06-bench-summary.md` shows `overall_pass_rate: 1.0` | Not a halt gate; continue to step 7 (G7-NW handles real recovery) | +| G6-CLI-FAIL | Step 6 CLI exited non-zero or no `suite-result.json` | No retry — always surfaces (infra) | +| G7-NW | `07-analysis.md` has `has_structural_weakness: false` | Menu (4 options) | +| G8-BLOCKED | Step 8 optimizer subagent returned `status: blocked` | Menu (2 options) | +| G9-REVISION | `09-validator-verdict.md` has `verdict: needs-revision` | Menu (1 option) | +| G9-REJECT | `09-validator-verdict.md` has `verdict: reject` | Menu (3 options) | +| G10-NO-EMPIRICAL-GAIN | (e.5) re-bench `overall_pass_rate` ≤ baseline OR ≥1 case regressed | Menu (3 options) | + +### Recovery menus + +Each retry-eligible gate has a menu of strategies, ordered +cheap → expensive. Full menu reference (loaded at each gate +firing): [`agents/recovery-menus.md`](./agents/recovery-menus.md). + +Read it before deciding, and pick the cheapest **available + +unattempted** option each time the gate fires this run. + +## Reasoning protocol at each gate firing + +When a gate fires: + +1. Read the gate's menu in the section above. +2. Read the journal's retry counters and the appended event log + to see which strategies have already been attempted this run + for this gate. +3. If the gate ID is in `surfaced_gates` (from interview): surface + per [Surfacing a gate](#surfacing-a-gate). Wait for the + operator's response and follow it. +4. Else: pick the cheapest unattempted strategy from the menu. +5. Append a journal entry. Format: + + ```text + strategy () | + reason: | + retry: / + ``` + +6. Distill the gate's rationale into `${OPERATOR_DIRECTIVES}` — an + atomic list of new requirements for the chain skill being + re-dispatched. Never a context dump. See + [`subagent-dispatch.md`](../shared/subagent-dispatch.md) for + the directive contract. +7. Re-dispatch the relevant chain step(s) via the `Skill` tool + with that directive in scope. +8. After the re-dispatch returns, resume the forward walk. If the + same gate fires again, repeat from step 1 of this protocol. + +## Cap accounting + +- **Per-gate cap** (`max_retries_per_gate`, default 2). Counts + total retry events for that gate type, regardless of which menu + strategy was picked. Two G7-NW retries means strategies 1 and 2 + have been attempted; strategies 3 and 4 are unreachable this + run. +- **Global cap** (`global_retry_cap`, default 8). Counts retry + events across all gates. Safety net for thrashing. +- A retry event is one re-dispatch of a chain step driven by a + gate's recovery menu. Mechanical re-dispatches (e.g., step 4 + re-running per-probe inside step 5's revision loop) count once + at the gate that triggered them, not per chain skill. +- **Cap exhaustion is honest-exit**, not failure. The journal + records what was tried; the summary explains the principled + outcome. + +## Surfacing a gate + +When a gate ID is in `surfaced_gates` (or the meta-breakpoint for +multi-strategy menus is enabled), pause the run and present the +template at +[`agents/surfacing-prompt.md`](./agents/surfacing-prompt.md) with +the gate's context substituted in. Wait for the operator's +response; the parsing rules (strategy number, custom directive, +"honest exit") live in that template. + +## Working-directory discipline + +A long chain run with many `Bash` calls is sensitive to CWD drift. +The `Bash` tool preserves CWD between calls — a single bare `cd` +into a subdirectory persists for the rest of the run, and later +relative paths (e.g., `mkdir -p docs/skill-optimizer//...`) +silently resolve in the wrong place. + +**Two rules:** + +1. **Never use bare `cd subdir && cmd`** that would persist. Either: + - Use absolute paths: `curl -o /home/.../vendored-skill/SKILL.md ...` + - Or scope with a subshell: `(cd vendored-skill && curl -o SKILL.md ...)` +2. **Verify with absolute paths** when checking that a downstream + step's output landed: `ls -la /home/.../docs/skill-optimizer//` + rather than `ls -la docs/...`. The orchestrator and the + dispatched subagents may have different effective CWDs; absolute + paths remove that ambiguity. + +If a step's canonical "appears missing" after a successful +dispatch, the first thing to check is the orchestrator's current +working directory — `pwd` it, then re-look at the canonical with +an absolute path. The subagent was likely correct; the relative +path was resolving to a sibling tree. + +## Edge cases + +- **Re-entrancy on the same slug.** Re-invoking autopilot on the + same slug picks up where it left off via the staleness check + in (d). The new invocation gets its own + `autopilot-/journal.md` and its own summary. Caps reset. + Prior run's outputs are preserved naturally by timestamp. +- **Step 1 wrapper detection.** If step 1 flags + `likely_wrapper: true`, the chain skill itself surfaces the + decision to the operator. Autopilot does not auto-resolve + wrapper redirection — treat this as a structural surface, not a + gate. The operator's response determines whether to re-vendor + and continue. +- **No suite.yml yet on a fresh slug.** First-time invocation has + no canonicals; staleness check treats everything as stale and + dispatches step 1 normally. Same for downstream steps. +- **G6-CLI-FAIL during a retry.** Infra failure during a + recovery step is still G6-CLI-FAIL. Surface, do not retry, + `exit_status: blocked-on-infra` if operator does not recover. +- **All-pass after an enable-more-picks attempt.** If G7-NW + strategy 3 succeeds (picks are added, bench runs, weakness is + found), the forward walk resumes from the new step 7. No + special handling. diff --git a/skills/autopilot/agents/interview-prompt.md b/skills/autopilot/agents/interview-prompt.md new file mode 100644 index 0000000..715fab8 --- /dev/null +++ b/skills/autopilot/agents/interview-prompt.md @@ -0,0 +1,83 @@ +# Startup interview prompt + +Operator-facing text presented by `skill-optimizer:autopilot` at +workflow step (b). Render with the inputs below substituted, then +send as a single message to the user. Wait for the user's response +before continuing. + +## Inputs (templated by the operator session) + +- `${SKILL_NAME}` — display name of the skill being optimized + (the slug or the user's original phrasing) + +## Prompt body + +```text +I'm going to optimize ${SKILL_NAME} end-to-end. By default I'm +fully automated — I run all 9 chain steps and retry up to 2 times +at each recovery point. I'll only pause and ask if the bench +environment breaks (docker/agent crash — that's not something to +fix without you). + +Optional breakpoints — pick any to be in the loop on: + +- Reviewing picked test cases before bench runs. Default: top 5 by + importance are picked automatically. +- A probe gets blocked or rejected during test-writing. Default: + drop the affected probe and continue; the others still run. +- Validate-tests says one or more probes need revision. Default: + distill the rationale into a directive and retry the probe up to + the cap. +- The skill passes every test or no structural weakness gets found. + Default: try recovery strategies in order — first reframe the + analysis, then more bench trials, then enable more picks, then + propose new probes — until something surfaces or the cap is hit. +- The optimizer gets stuck on a weakness it can't translate to a + concrete fix. Default: reframe the weakness and retry up to the + cap. +- The validator says the proposed fix needs revision. Default: + distill the rationale and retry the proposal up to the cap. +- The validator outright rejects the proposed fix. Default: try + recovery strategies in order — first a different fix for the same + weakness, then reframe the weakness, then add more probe coverage + — until something lands or the cap is hit. +- A recovery point has multiple strategies available (meta-breakpoint + covering the three above). Default: I pick the cheapest unattempted + strategy. Surfacing this means I'll show you the menu and let you + choose. + +A few defaults you can also override: + +- PR intent: no (only matters for upstream skills) +- Top-N test cases picked: 5 +- Max retries per recovery point: 2 + +Say 'all auto, go' to just run, or tell me what to change. +``` + +## Parsing the response + +The operator's response may be: + +- **"all auto, go"** (or equivalent) — proceed with all defaults. +- **A list of breakpoints to surface** — add the listed gate IDs + to `surfaced_gates`. The mapping from the user-facing phrasing + to gate IDs: + + | User phrasing | Gate ID(s) | + |---|---| + | reviewing picked test cases | G3 | + | probe blocked / rejected during test-writing | G4-BLOCKED, G5-REJECT | + | validate-tests probe needs-revision | G5-REVISION | + | skill passes every test / no structural weakness | G6-ALLPASS, G7-NW | + | optimizer stuck | G8-BLOCKED | + | validator needs-revision | G9-REVISION | + | validator rejects | G9-REJECT | + | meta-breakpoint (multi-strategy menus) | mark G7-NW, G8-BLOCKED, G9-REJECT as surfaced | + +- **Override values** for `pr_intent`, `pick_top_n`, or + `max_retries_per_gate` — capture into the matching field. +- **Per-gate default overrides** — capture into + `default_overrides` as `{: }`. +- **Ambiguous response** — ask one clarifying question; do not + guess. diff --git a/skills/autopilot/agents/recovery-menus.md b/skills/autopilot/agents/recovery-menus.md new file mode 100644 index 0000000..8b3d042 --- /dev/null +++ b/skills/autopilot/agents/recovery-menus.md @@ -0,0 +1,91 @@ +# Recovery menus + +Reference loaded by `skill-optimizer:autopilot` at each gate firing +(per the Reasoning protocol in SKILL.md). For each retry-eligible +gate, lists strategies ordered cheap → expensive. The pilot picks +the cheapest **available + unattempted** option, consulting the +journal's retry log for what's been tried this run. + +## G5-REVISION + +1. Distill the validator's per-probe rationale into directives. + Re-run step 4 for the affected probes only. Re-run step 5. + +## G7-NW (cheap → expensive) + +1. **Reframe analysis angle.** Re-run step 7 with a directive + pointing at marginal failures, model-specific drift, latent + issues even when pass-rate is high. +2. **More bench trials.** Re-run step 6 with an elevated trial + count (e.g., 3× the normal trials per probe) to surface + variance, then re-run step 7. +3. **Enable more picks.** Flip the next-highest-importance + `picked: false` functionalities in step 3's spec to + `picked: true`. Re-run steps 4 → 5 → 6 → 7 for the new probes + only (existing probes stay). +4. **Add new probes.** Re-run step 3 with a directive to propose + new functionalities the existing pass missed (focus on edge + cases the existing probes don't cover). Then 4 → 5 → 6 → 7. + +## G8-BLOCKED (cheap → expensive) + +1. **Reframe weakness.** Re-run step 7 with a directive that the + optimizer was blocked on weakness W; ask for a reformulation + that is more concrete or at a different abstraction. Then + re-run step 8. +2. **Sharper directive to optimizer.** If the weakness reads + concrete enough but the optimizer struggled with the general + principle, re-run step 8 alone with a sharpened directive + tying the principle to the skill's structure. + +## G9-REVISION + +1. Distill the validator's rationale into a directive. Re-run + step 8. Re-run step 9. + +## G9-REJECT (cheap → expensive) + +1. **Different fix, same weakness.** Re-run step 8 with the + rejected proposal noted as anti-pattern (proposal P rejected + because Y; try a different general principle for the same + weakness). +2. **Reframe weakness.** Re-run step 7 with the validator's + rejection rationale (weakness X rejected because Y; find a + different angle on what's failing). Then 8 → 9. +3. **Add coverage.** If the validator rejected on grounds that + the weakness wasn't real in the trace data, treat as G7-NW + menu (more trials, more picks, more probes). + +## G10-NO-EMPIRICAL-GAIN (cheap → expensive) + +Validator approved (step 9) but the empirical re-bench (phase +(e.5)) showed pass-rate ≤ baseline or introduced a regression. +The proposal sounded right but didn't move the needle. + +1. **Try a different fix, same weakness.** Re-run step 8 with the + rejected proposal noted as "approved by validator but added zero + bench gain — likely too abstract or addressed a non-load-bearing + detail." Then 9 → re-bench. Cheap (one optimization cycle). +2. **Reframe weakness.** Re-run step 7 with directive "prior + weakness X passed validator but didn't improve the bench; + reconsider whether the named pattern is the real failure mode." + Then 8 → 9 → re-bench. Medium. +3. **Increase test coverage.** If the after-rebench is identical + to baseline (e.g., both at 100%, no fail cases for the + improvement to flip), the probes can't *show* an improvement + that exists. Treat as G7-NW menu (more trials, enable more + picks, add new probes). Then continue from step 6 forward. + Expensive but the right move when the probes themselves are + the bottleneck. + +After cap exhaustion: exit_status `unchanged-empirical`. The +validator approved but the empirical evidence doesn't support +shipping the proposal as an improvement. + +## Gates that don't appear here + +- **G3** (picks) — single action, not retry-eligible. +- **G4-BLOCKED, G5-REJECT** — no retry: skip/drop, log, continue. +- **G6-ALLPASS** — not a halt gate; continue to step 7 (G7-NW + handles the actual recovery). +- **G6-CLI-FAIL** — no retry: always surfaces; infra needs human. diff --git a/skills/autopilot/agents/summary-template.md b/skills/autopilot/agents/summary-template.md new file mode 100644 index 0000000..1ca0cbb --- /dev/null +++ b/skills/autopilot/agents/summary-template.md @@ -0,0 +1,130 @@ +# End-of-run summary template + +Operator-facing audit record written by `skill-optimizer:autopilot` +at workflow step (f), promoted from the run journal. Tracked at +`docs/skill-optimizer//autopilot-summary-.md`. + +## Inputs (templated by the operator session) + +- `${RUN_STARTED_AT}` / `${RUN_ENDED_AT}` — ISO 8601 UTC +- `${SLUG}` — chain slug +- `${SOURCE}` — URL or local path the operator provided +- `${EXIT_STATUS}` — one of `improved`, `improved-with-regression`, + `unchanged-empirical`, `unchanged-honest-exit`, `blocked-on-infra` +- `${EMPIRICAL_BLOCK}` — empirical-verification block (see format + below); omit if the run never reached step 9 approve +- `${TOTAL_RETRIES}` — integer total across all gates +- `${JOURNAL_PATH}` — gitignored journal path +- `${HEADLINE}` — one paragraph: what happened end-to-end and why + it ended this way (operator writes from journal) +- `${PER_STEP_ROWS}` — table rows for the per-step outcomes +- `${RECOVERY_SUMMARY}` — bullet list of fired gates with retry + trail; bullet list of gates not fired +- `${CAVEATS}` — operator's interpretation of anything off (flaky + bench, suspicious trace patterns); not a substitute for operator + review +- `${REPO_CHANGES}` — file list of new/modified canonicals + +## Template body + +```markdown +--- +run_started_at: ${RUN_STARTED_AT} +run_ended_at: ${RUN_ENDED_AT} +slug: ${SLUG} +source: ${SOURCE} +exit_status: ${EXIT_STATUS} +total_retries: ${TOTAL_RETRIES} +journal_path: ${JOURNAL_PATH} +--- + +# Autopilot run for `${SLUG}` + +## Headline + +${HEADLINE} + +## Per-step outcomes + +| Step | Outcome | Artifact | +|---|---|---| +${PER_STEP_ROWS} + +## Recovery summary + +${RECOVERY_SUMMARY} + +## Empirical verification + +${EMPIRICAL_BLOCK} + +## Caveats + +${CAVEATS} + +## What changed in the repo + +${REPO_CHANGES} +``` + +## Empirical verification block format + +Always include this section when the run reached step 9 approve +(even if the empirical re-bench failed). Omit only if the chain +never produced an improved-skill (e.g., honest-exit at G7-NW with +no weakness found). + +```markdown +- **Baseline pass rate:** / (%) + — `.skill-optimizer//bench-results//` +- **After-improvement pass rate:** / (%) + — `.skill-optimizer//bench-results/-after-improvement/` +- **Net gain:** + cases passed (or 0, or negative) +- **Regressions** (pass → fail after improvement): +- **New passes** (fail → pass after improvement): +- **Verdict:** +``` + +When `exit_status: unchanged-empirical`, the verdict line should +spell out why this is honest rather than failed — e.g., "validator +approved but bench can't demonstrate the improvement (probes too +easy, weakness too subtle, or the fix is structural without +behavioral impact in the current test set)". + +## Per-step outcome row format + +```markdown +| | | | +``` + +Examples: + +- `| 1 investigate-functionality | written | docs/.../01-functionality.md |` +- `| 7 analyze | written (re-run 2x) | docs/.../07-analysis.md |` +- `| 8 improve | skipped (no weakness) | — |` + +## Recovery summary format + +For each gate that fired: + +```markdown +- retries: / + - strategy () — + - strategy () — +``` + +Then one line listing gates that did not fire: + +```markdown +- Gates not fired: +``` + +## Discipline + +- The summary is an aggregate, not a journal dump. Trim narrative; + keep retry trail + exit reason + repo changes. +- `exit_status` enum is the contract — do not invent new values. +- Do NOT embed trace excerpts. Pointers to canonical files are + enough; the operator opens those for detail. diff --git a/skills/autopilot/agents/surfacing-prompt.md b/skills/autopilot/agents/surfacing-prompt.md new file mode 100644 index 0000000..643da52 --- /dev/null +++ b/skills/autopilot/agents/surfacing-prompt.md @@ -0,0 +1,73 @@ +# Gate-surfacing prompt + +Operator-facing text presented by `skill-optimizer:autopilot` when +a gate in `surfaced_gates` fires (or when the meta-breakpoint for +multi-strategy menus is enabled). Render with the inputs below +substituted, then send as a single message to the user. Wait for +the user's response before continuing. + +## Inputs (templated by the operator session) + +- `${GATE_ID}` — e.g. `G7-NW` +- `${STEP_NUMBER}` — chain step that just produced the canonical + triggering the gate +- `${WHAT_HAPPENED}` — plain-language description of the canonical + finding (one or two sentences; no jargon if avoidable) +- `${MENU_OPTIONS}` — numbered list of strategies from this gate's + recovery menu, with one-sentence descriptions and an attempted + marker (yes/no) per option +- `${CURRENT_RETRY}` / `${MAX_RETRIES}` — current and max per-gate + retry counter +- `${GLOBAL_CURRENT}` / `${GLOBAL_CAP}` — current and max global + retry counter + +## Prompt body + +```text +Gate ${GATE_ID} just fired at step ${STEP_NUMBER}. + +What happened: +${WHAT_HAPPENED} + +Recovery options (cheap → expensive): +${MENU_OPTIONS} + +You can: +- Pick a strategy number to try it +- Supply your own directive ("look at the X cluster", "treat Y as a + marginal failure") +- Say "honest exit" to stop here + +Currently ${CURRENT_RETRY}/${MAX_RETRIES} retries used on this +gate; global ${GLOBAL_CURRENT}/${GLOBAL_CAP}. +``` + +## Menu option format + +Each line of `${MENU_OPTIONS}`: + +```text +. — [attempted: yes/no] +``` + +Example: + +```text +1. Reframe analysis angle — re-run step 7 with a directive pointing at + marginal failures and model-specific drift [attempted: yes] +2. More bench trials — re-run step 6 with 3x trials, then step 7 + [attempted: no] +``` + +## Parsing the response + +- **Strategy number** — execute that option (as if autopilot had + picked it itself). +- **Custom directive** — treat as a one-off operator directive; + pick the most-fitting menu option (based on which strategy the + directive aligns with) and prepend the operator's text to the + distilled `${OPERATOR_DIRECTIVES}` for that dispatch. +- **"Honest exit"** — go straight to workflow step (f), + `exit_status: unchanged-honest-exit`. +- **Ambiguous response** — ask one clarifying question; do not + guess which strategy to pick. diff --git a/skills/design-tests/SKILL.md b/skills/design-tests/SKILL.md new file mode 100644 index 0000000..6a1eeaa --- /dev/null +++ b/skills/design-tests/SKILL.md @@ -0,0 +1,171 @@ +--- +name: design-tests +description: Use when the user wants to design or propose test cases for a skill — phrases like "design tests for this skill", "propose test cases", "what should we test", "plan test coverage for this skill". Also triggers mid-way through skill-optimizer chain work, once a functionality report exists (and submissions research, if PR-bound) and the next thing is figuring out what to test. Use even when the user doesn't explicitly say "design" — any phrasing about figuring out what tests to build for a skill should trigger this. +--- + +# design-tests + +Step 3 of the skill-optimizer chain. **Maintenance step.** Takes the +functionality report from step 1, dispatches a designer subagent to +enumerate the skill's responsibilities and propose a ranked set of +**functionalities** to test (each functionality = one responsibility +the skill must fulfill), then asks the user to pick which ones to +actually build probes for at step 4. + +The filesystem IS the state: this step writes a +`skill-evals///spec.yaml` per proposed functionality +plus a one-time audit report at `03-test-proposals.md`. No +`picked: []` array anywhere; each spec.yaml has its own +`picked: true|false`. + +## What you produce + +Two artifacts at `docs/skill-optimizer//`: + +1. **`03-test-proposals.md`** — one-time audit report with the + ranked list of proposed functionalities + full reasoning. For + human review of the design reasoning; downstream steps do NOT + read it. The subagent rewrites it on re-runs. + +2. **`skill-evals///spec.yaml`** — one folder per + proposed functionality, with frontmatter per + [`frontmatter-discipline.md`](../shared/frontmatter-discipline.md): + + ```yaml + name: refuses-malformed-input + description: The skill refuses input that violates its expected schema. + picked: false # user flips to true after the user gate + importance: high # high | medium | low + suggested_probes: + - malformed-json + - missing-required-field + - type-mismatch + why_test: > + Without this guard, broken upstream data silently corrupts + the skill's downstream logic. + ``` + + Step 4 builds probes only for functionalities whose `spec.yaml` + has `picked: true`. + +For the body template of `03-test-proposals.md` and the exact +`spec.yaml` field list, see +[`agents/test-designer.md`](./agents/test-designer.md). + +## Workflow + +### (a) Confirm prerequisites + +`01-functionality.md` must exist. If not, tell the user to run +`skill-optimizer:investigate-functionality` first and stop here. + +If `skill-evals//` doesn't exist yet, create the empty +directory. + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply the maintenance-step flow. Re-runs read the current +`skill-evals//` tree as state and extend or modify it. Collect +`${OPERATOR_DIRECTIVES}` per the protocol. If a directive will be +destructive (e.g., "remove the X functionality"), commit a +checkpoint before dispatching per the protocol's safe-destructive- +edits section. + +If the user's directives contradict each other (e.g., "focus on X" +combined with "ignore X"), surface the contradiction before +re-dispatching; don't try to resolve it yourself. + +### (c) Dispatch the test-designer subagent + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT enumerate responsibilities or design +proposals yourself in this session** — dispatch the subagent via +the `Agent` tool. Render the prompt template at +[`agents/test-designer.md`](./agents/test-designer.md) +inline by substituting `${FUNCTIONALITY_PATH}`, +`${TESTS_TREE_PATH}`, `${PROPOSALS_PATH}`, and +`${OPERATOR_DIRECTIVES}`, then pass the rendered text as the Agent +tool's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +The subagent sees: `01-functionality.md`, the current +`skill-evals//` tree (load-bearing state per the +maintenance rule), +`${OPERATOR_DIRECTIVES}`, output paths. + +The subagent does NOT see: the skill's source content (would +gerrymander tests around the source's literal phrasing — +design from STATED responsibilities); `07-analysis.md`, +`08-improvement-proposal.md`, `09-validator-verdict.md`; raw +failure data, bench results; git history of any tree file. + +**Why this matters:** the operator session has absorbed prior +failure data, prior optimizer attempts, validator verdicts — that +context biases test design toward "what just failed" rather than +"what comprehensively covers responsibilities". The subagent walled +off from that produces coverage-oriented proposals, not +regression-defensive ones. + +The subagent writes: + +- `03-test-proposals.md` (the audit report) +- One `skill-evals///spec.yaml` per proposed + functionality. Existing spec.yaml files the user has already + edited (e.g., `picked: true` set) are preserved verbatim unless + a directive explicitly targets that functionality. + +Returns a brief summary: top-3 functionalities by importance, +preserved-unchanged list, newly-added list. + +### (d) Confirm subagent output + +Verify `03-test-proposals.md` parses and each new +`skill-evals///spec.yaml` parses with required fields. +If any spec.yaml fails to parse, surface to the user; don't repair +the subagent's output yourself. + +If the proposal has fewer functionalities than expected, the +functionality report is likely thin. Surface to the user with two +options: re-run step 1 with directives, or accept if the skill is +genuinely small. Per the no-auto-invocation rule, don't re-invoke +step 1 yourself. + +### (e) User gate: present proposals, collect picks + +Show the user the ranked list from `03-test-proposals.md` and tell +them how to indicate picks: edit `picked: true|false` in each +`skill-evals///spec.yaml`. They can also edit +`suggested_probes` or ask for a revised proposal. + +Three realistic responses: + +1. **User edits `picked` directly (or asks you to flip specific + ones).** Read each spec.yaml to confirm the picked set. If the + user explicitly says "flip these to true", do it and confirm + what you set. The rule is "no proactive flipping without user + direction", not "user must edit every file by hand". +2. **User wants additions or revisions** (e.g., "split X into + two", "add a functionality for Y"). Treat as directives. + Return to (b) and re-dispatch the subagent at (c). Don't write + the spec.yaml manually yourself — the subagent is the only + writer of test-design content. +3. **User picks zero.** Ask whether they want a revised proposal + (case 2) or are abandoning the optimization for this skill + (exit honestly). + +Don't auto-flip `picked` on the user's behalf. Even if all +functionalities look important, the user owns the decision (they're +paying for probe-building in step 4 and the bench run in step 6). + +### (f) Hand off + +> Next, invoke `skill-optimizer:write-tests`. + +## Edge cases + +- **User removes a functionality folder manually** — fine, + filesystem-as-state working as intended. On next re-run, the + subagent reads the current tree and doesn't propose the removed + one back unless directives ask. diff --git a/skills/design-tests/agents/test-designer.md b/skills/design-tests/agents/test-designer.md new file mode 100644 index 0000000..d7b1498 --- /dev/null +++ b/skills/design-tests/agents/test-designer.md @@ -0,0 +1,160 @@ +# Test-designer subagent + +You are dispatched by `skill-optimizer:design-tests` to +enumerate the skill's responsibilities (from `01-functionality.md`) +and propose a ranked set of **functionalities** to test, then write +a per-functionality `skill-evals///spec.yaml` +for each plus a one-time audit report `03-test-proposals.md`. + +A "functionality" here means **one distinct responsibility the +skill must fulfill**. It's what step 4 will build probes for (each +functionality typically gets 1-3 probes that test it from different +angles). + +## Inputs (templated by the operator session) + +- `${FUNCTIONALITY_PATH}` — current `01-functionality.md` +- `${TESTS_TREE_PATH}` — current `skill-evals//` directory (may be empty + on first invocation; on re-runs it has existing + `skill-evals///spec.yaml` files) +- `${PROPOSALS_PATH}` — where to write the audit report (typically + `docs/skill-optimizer//03-test-proposals.md`) +- `${OPERATOR_DIRECTIVES}` — bulleted list of atomic new + requirements (may be empty) + +## What you see + +- `${FUNCTIONALITY_PATH}` — the responsibilities, classification, + triggers, terminology +- `${TESTS_TREE_PATH}` — every existing + `skill-evals///spec.yaml` (load-bearing state per + the maintenance rule; you extend, not replace) +- The operator's directives + +## What you do NOT see + +- The skill's source content. Designing from the source's literal + phrasing biases tests toward "what the source says" rather than + "what the skill is supposed to do". Coverage design happens at + the responsibility level; concrete fixture writing (step 4) + handles source content. +- Any `07-analysis.md`, `08-improvement-proposal.md`, + `09-validator-verdict.md` — these are downstream and would bias + proposals toward "what failed last time" instead of + comprehensive coverage. +- Raw failure data, bench results +- Git history of any tree file or of your own audit report + +## Output + +**Two artifacts:** + +### `${PROPOSALS_PATH}` — `03-test-proposals.md` + +The audit report. Operator + user read this for design reasoning; +downstream steps do NOT read it. Rewrite top-to-bottom on each +re-run, anchored by the current tree state. + +Body: + +1. **Summary** — one paragraph: how many functionalities you're + proposing, what coverage they aim for, anything notable. +2. **Ranked proposals** — for each functionality, in importance + order: + + ### Functionality N: `` + + - **What it tests:** one-sentence statement of the + responsibility. + - **Why it matters:** the failure mode that would slip through + if this isn't tested. + - **Suggested probes:** 1-3 distinct probe scenarios for step 4 + to build (named slugs). + - **Importance:** high / medium / low (with brief justification). + - **Status:** new / existing-preserved / revised-per-directive. + +3. **Coverage gaps acknowledged** — responsibilities from + `01-functionality.md` you DIDN'T propose tests for, with a + one-line reason (e.g., "trivially tested by N1's probe set", + "not testable in a static workbench"). + +### `skill-evals///spec.yaml` — one per proposed functionality + +```yaml +name: +description: +picked: false # user flips to true after the user gate +importance: high|medium|low +suggested_probes: + - + - +why_test: > + +``` + +**`picked: false` is the default for new functionalities.** The +user gate at step 3 lets the user flip to `true` for the ones they +want built. Don't pre-pick. + +**Existing spec.yaml files are preserved verbatim** unless a +directive explicitly targets that functionality (e.g., "split +refuses-malformed-input into two", "add an empty-array probe to +X"). Don't overwrite the user's `picked: true` edits. + +## Reasoning protocol + +1. **Read `${FUNCTIONALITY_PATH}` first** — enumerate the + responsibilities. Each becomes a candidate functionality. +2. **Read the current `skill-evals//` tree.** Existing functionalities + are constraints: you don't re-propose them (unless directives + say otherwise); you extend or modify the set. +3. **Group adjacent responsibilities** if they're not meaningfully + distinct test targets. E.g., "validates inputs" and "rejects + bad inputs" might be one functionality. Don't fragment. +4. **Split monolithic responsibilities** if they actually represent + multiple distinct test targets. E.g., "handles inputs" is too + broad — split into "accepts valid inputs", "refuses malformed + inputs", "handles empty inputs" as separate functionalities. +5. **Suggest probes per functionality, most-representative first.** + For "refuses malformed input", probes might be: + `malformed-json`, `missing-required-field`, `type-mismatch`. + Each probe is one concrete scenario step 4's test-writer will + build. **Order matters:** step 4's minimal-coverage default + picks the first probe of each functionality, so put the probe + most likely to surface failures up front and trailing probes as + additional angles for full-coverage runs. +6. **Rank by importance.** High = a regression here would break + the skill's core promise. Medium = degrades but doesn't break. + Low = nice-to-have edge case. +7. **Honor `${OPERATOR_DIRECTIVES}` atomically.** Each directive + is a discrete change (add functionality X, split Y into two, + revise Z's probes). Apply the changes; preserve everything + else. + +## Edge cases + +- **`01-functionality.md` is thin** — propose what you can; note + in the "Coverage gaps acknowledged" section that the + functionality report likely missed responsibilities. Operator + may re-run step 1. +- **`${OPERATOR_DIRECTIVES}` contradict each other** — surface in + the audit report's summary section; don't try to resolve. The + operator session will handle. +- **Directive asks to remove a functionality the user has picked** + — apply (delete the spec.yaml file) and note in the audit + report under "Removed per directive". The operator's + destructive-edit checkpoint protects the prior state. + +## Return summary + +After writing the audit report and per-functionality spec.yaml +files, return a brief summary: + +- Top-3 functionalities by importance (with one-line rationale + each) +- Preserved-unchanged: list of functionality slugs (if any) +- Newly-added: list of functionality slugs (if any) +- Revised-per-directive: list of functionality slugs (if any) + +Keep it under 250 words. diff --git a/skills/improve/SKILL.md b/skills/improve/SKILL.md new file mode 100644 index 0000000..87cf9cb --- /dev/null +++ b/skills/improve/SKILL.md @@ -0,0 +1,172 @@ +--- +name: improve +description: Use when the user wants to improve a skill based on identified structural weaknesses — phrases like "improve this skill", "fix the structural weakness", "optimize the skill", "apply the analysis". Triggers after `skill-optimizer:analyze` has produced `07-analysis.md` with `has_structural_weakness: true`. Refuses to fire if no weakness was identified (anti-ducktape gate). Use even when the user doesn't explicitly say "improve" — any phrasing about acting on the analysis or modifying the skill should trigger this. +--- + +# improve + +Step 8 of the skill-optimizer chain. **Fresh-derivation step.** Takes +the named structural weaknesses from step 7, dispatches an optimizer +subagent to draft a principled fix, and writes +`08-improvement-proposal.md`. The proposal is then validated +independently by step 9, which materializes the improved skill on +approve. **This step does not produce the improved skill itself** — +it produces the proposal that step 9 acts on. + +**Refuses to fire** if `07-analysis.md` has +`has_structural_weakness: false` — there's nothing to optimize, and +forcing a fix is the ducktape failure mode the chain is built to +prevent. + +**Single-shot per invocation.** No in-step revision loop. If step +9's validator returns `needs-revision`, the operator (or +auto-pilot at step 10) distills the validator's rationale into a +directive and re-invokes this step. + +**Out of scope:** validating the proposal (step 9) and packaging +the change as a PR draft (PR composition is a separate downstream +concern; auto-pilot or a dedicated composer can handle it if +`pr_submission_intent: true`). + +## What you produce + +One artifact at `docs/skill-optimizer//`: + +**`08-improvement-proposal.md`** — the optimizer's proposed change +plus rationale. Frontmatter (runtime-relevant facts only, per +[`frontmatter-discipline.md`](../shared/frontmatter-discipline.md)): + +```yaml +--- +addresses_weaknesses: + - + - ... +--- +``` + +Body has: the proposed change, rationale referencing the named +weakness and the "What WOULD address this" principle being +applied, and an explicit self-check against the "What WOULD NOT +address this" anti-pattern list (the optimizer states why its +proposal is NOT one of the ducktape moves the analyzer flagged). +Subagent prompt template at +[`agents/optimizer.md`](./agents/optimizer.md) +specifies the section shape. + +## Workflow + +### (a) Confirm prerequisites + refuse-if-no-weakness gate + +Three checks: + +1. `07-analysis.md` must exist with valid frontmatter. If not, + tell the user to run `skill-optimizer:analyze` first. + +2. **Anti-ducktape gate:** `has_structural_weakness: true` must be + set. If `false`, REFUSE — print: "Step 7 found no structural + weakness. Step 8 won't fire — nothing principled to optimize. + If you disagree, re-invoke step 7 with a directive; if you + agree, exit honestly." Do NOT proceed. + +3. `01-functionality.md` must exist with both + `optimization_target` and `optimization_target_source` + frontmatter fields set (step 1 records these — see step 1's + "What you produce" section). Skill content must be readable at + `docs/skill-optimizer//improved-skill/` if it exists + (accumulated state from prior step-9 approvals), else + `.skill-optimizer//vendored-skill/`. The + **`optimization_target` field identifies the specific file in + that directory that the optimizer modifies** — everything else + in the dir is context only. + +If `pr_submission_intent: true`, `02-submissions.md` should exist +so the optimizer can shape the diff to upstream conventions from +the start. If missing, ask whether to run step 2 first — step 9's +external check still runs if it appears later. + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Each invocation overwrites the canonical from +`07-analysis.md` plus current skill plus directives. Collect +`${OPERATOR_DIRECTIVES}` — examples: "prefer additive changes", +"don't touch the description field — validator rejected that last +round". The operator reads prior proposals/verdicts and distills; +the subagent never sees the raw prior content. + +If a directive references a specific section of a prior proposal +or verdict, that's a context-dump masquerading as a directive. +Translate to an atomic new requirement before passing. + +### (c) Dispatch the optimizer subagent + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT propose the diff yourself in this +session** — dispatch the subagent via the `Agent` tool. Render the +prompt template at +[`agents/optimizer.md`](./agents/optimizer.md) +inline by substituting `${ANALYSIS_PATH}`, `${FUNCTIONALITY_PATH}`, +`${SKILL_CURRENT_PATH}`, `${SUBMISSIONS_PATH}` (if PR-bound), +`${PROPOSAL_OUTPUT_PATH}`, and `${OPERATOR_DIRECTIVES}`, then pass +the rendered text as the Agent tool's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +The optimizer sees: `07-analysis.md`; `01-functionality.md` +(including the `optimization_target` and +`optimization_target_source` frontmatter fields); +`${SKILL_CURRENT_PATH}` — `docs/skill-optimizer//improved-skill/` +if it exists, else `.skill-optimizer//vendored-skill/`; +`${OPTIMIZATION_TARGET}` — the relative path within +`${SKILL_CURRENT_PATH}` of the file to modify, copied from +`01-functionality.md`'s frontmatter; `02-submissions.md` if +PR-bound; `${OPERATOR_DIRECTIVES}`. + +The optimizer modifies **only the file at `${OPTIMIZATION_TARGET}`**. +The proposal's diff scope is that single file. Other files in +`${SKILL_CURRENT_PATH}` are reference context, not edit targets. + +The optimizer does NOT see: raw failed trials, `findings.txt`, +`trace.jsonl`; grader internals +(`skill-evals////grader.mjs`); test +inputs (`skill-evals////workspace/`); +its own prior +`08-improvement-proposal.md` or git history; step 9's prior +`09-validator-verdict.md` (would bias toward defending or pivoting +away from the prior attempt). + +**Why this matters:** the chain's anti-ducktape architecture +hinges on this. If the optimizer read failed trials, it would +pattern-match a patch making those specific trials pass — the +textbook ducktape mode. The analyzer's job (step 7) is to +translate raw failures into general principles + anti-patterns; +the optimizer's job is to apply the principle. The translation +through `07-analysis.md` is what enforces principled improvement. + +The optimizer writes `08-improvement-proposal.md` and returns: +weakness(es) addressed, principle applied, lines/sections modified. + +If the optimizer reports BLOCKED (the analyzer's named weakness +is too abstract to derive a concrete diff from), surface to the +user. Fix is typically a step 7 re-run with a reframing directive; +per the no-auto-invocation rule, the user invokes step 7. + +### (d) Confirm subagent output + +Verify the report parses and the body has: a clear proposed +change, a rationale referencing at least one named weakness and +the "What WOULD address this" principle, and an explicit +self-check against the anti-pattern list. If the self-check is +missing or vague, re-dispatch with a directive making the +requirement explicit — don't fill it in yourself. + +### (e) Hand off + +> Improvement proposal complete at +> `08-improvement-proposal.md`. Next, invoke +> `skill-optimizer:validate` to check the proposal +> independently. If the validator returns `needs-revision` or +> `reject`, you'll come back here with a directive distilling the +> validator's concerns. + +Don't auto-invoke step 9. diff --git a/skills/improve/agents/optimizer.md b/skills/improve/agents/optimizer.md new file mode 100644 index 0000000..352864c --- /dev/null +++ b/skills/improve/agents/optimizer.md @@ -0,0 +1,195 @@ +# Optimizer subagent + +You are dispatched by `skill-optimizer:improve` to draft a +principled fix for a named structural weakness from +`07-analysis.md` and write `08-improvement-proposal.md`. The +proposal is the SOLE output; you do NOT materialize the improved +skill (that's step 9's job after the validator approves). + +**Read first:** +[`../../shared/skill-design-philosophy.md`](../../shared/skill-design-philosophy.md) +— the cross-vendor synthesis of what makes a skill good and what +makes an improvement principled vs ducktape. Your proposed change +should map to one or more named principles. Your self-check +against the anti-pattern list (see "Output" below) must walk both +the analyzer's named anti-patterns AND the universal ducktape +moves listed in the philosophy doc's "Principled vs ducktape" +rubric. + +## Inputs (templated by the operator session) + +- `${ANALYSIS_PATH}` — `07-analysis.md` (named weaknesses + what + WOULD/WOULDN'T address each) +- `${FUNCTIONALITY_PATH}` — `01-functionality.md` (the skill's + stated responsibilities — your change must not contradict them) +- `${SKILL_CURRENT_PATH}` — the **current state of the skill**: + `docs/skill-optimizer//improved-skill/` if it exists (accumulated state from prior + step-9 approvals), else `.skill-optimizer//vendored-skill/` (step 1 vendored the + source regardless of upstream/local). You propose a new + improvement on top of whatever current state you read. +- `${OPTIMIZATION_TARGET}` — the **relative path within + `${SKILL_CURRENT_PATH}` of the file you modify**. From step 1's + `01-functionality.md` frontmatter. Default is `SKILL.md`; for + wrapper skills it points at the underlying rule file (e.g., + `command.md`). Your proposed diff is scoped to this single file — + other files in `${SKILL_CURRENT_PATH}` are context, not edit + targets. +- `${SUBMISSIONS_PATH}` — `02-submissions.md` if PR-bound (lets + you shape the diff to match upstream conventions from the start, + reducing step-9 round-trips) +- `${PROPOSAL_OUTPUT_PATH}` — where to write + `08-improvement-proposal.md` +- `${OPERATOR_DIRECTIVES}` — atomic new requirements (empty + unless this is a re-run, often distilled from step 9's prior + verdict) + +## What you see + +- `07-analysis.md` (named weaknesses + principles + anti-patterns) +- `01-functionality.md` +- The current skill content at `${SKILL_CURRENT_PATH}` +- `02-submissions.md` if PR-bound +- Operator directives + +## What you do NOT see + +- **Raw failed trials** (`findings.txt`, `trace.jsonl` from + `.skill-optimizer//bench-results/`). The analyzer translated raw failures into + general principles + anti-patterns at step 7; your job is to + apply the principle. Seeing the raw failures would pull you + toward pattern-matching a patch for THOSE specific trials — + the textbook ducktape mode. +- Grader internals (`skill-evals////grader.mjs` source). You + don't optimize against the grader; you optimize against the + named weakness. +- Test inputs (`skill-evals////workspace/`). Same reason — you + reason about the SKILL, not about specific inputs. +- Your own prior `08-improvement-proposal.md` or git history. No + consistency-with-prior-attempts bias. +- Step 9's prior `09-validator-verdict.md`. Seeing the prior + verdict would lead you to defend the prior attempt or + defensively pivot away from it. Lessons from the prior verdict + come through `${OPERATOR_DIRECTIVES}` distilled by the + operator session. + +## Output: `${PROPOSAL_OUTPUT_PATH}` — `08-improvement-proposal.md` + +Frontmatter (runtime-relevant only): + +```yaml +--- +addresses_weaknesses: + - + - ... +--- +``` + +Body has three required sections (the validator at step 9 +verifies all three): + +### 1. Proposed change + +A unified diff (or equivalent precise specification) showing +exactly what changes in the skill files. Be concrete enough that +the operator session can apply it mechanically; don't sketch. + +### 2. Rationale + +For each weakness in `addresses_weaknesses`: + +- Quote the weakness's "What WOULD address this" principle from + `07-analysis.md` +- Explain how your proposed change applies that principle +- Cite the specific skill section your change targets (path:line) + +### 3. Self-check against the anti-pattern list + +For each weakness, walk through its "What WOULD NOT address this" +list from `07-analysis.md` and state explicitly why your proposal +is NOT one of those ducktape moves. This section is load-bearing — +without it, the validator at step 9 can't see your reasoning +about the anti-patterns and the gate is compromised. + +If your proposal touches any of the listed anti-patterns +incidentally, justify why it's NOT functioning as that anti-pattern +(e.g., "I added a MUST clause, but it's preceded by procedural +guidance that operationalizes it — it's not a bare-MUST +restatement of the rule"). + +## Reasoning protocol + +1. **Read `07-analysis.md` start-to-finish.** Each weakness has + five required parts (Pattern, Hypothesized cause, Connects to + skill section, What WOULD address this, What WOULD NOT + address this). The principle and anti-patterns are your work + constraints. +2. **Read `01-functionality.md`** so your change doesn't + contradict the skill's stated responsibilities. +3. **Read the current skill** at `${SKILL_CURRENT_PATH}`. Locate + the section(s) named in each weakness's "Connects to skill + section" field. That's where the change goes. +4. **If `02-submissions.md` exists** (PR-bound), read the + frontmatter spec, file-location conventions, and prefix + taxonomy. Shape the diff to fit. The validator at step 9 will + check this; getting it right now saves a round-trip. +5. **Bias toward additive changes** unless the analyzer + specifically flagged the existing content as the weakness. + Additive (adding procedural guidance, adding an example, + adding a check) preserves what works; destructive (removing, + replacing, rewriting) risks regressions. +6. **Prune over add when prose is restating what Claude already + knows.** If the skill has a 3-paragraph explanation of an + obvious point and the analyzer flagged the skill as too + verbose to navigate, the principled fix may be trimming, not + adding more. +7. **Apply the principle, not the analyzer's words.** The + analyzer named what WOULD address the weakness; your job is + to translate that principle into concrete skill content. Don't + paste the analyzer's principle verbatim into the skill. +8. **Self-check before reporting done.** Walk each weakness's + anti-pattern list. For each anti-pattern, is your proposal + doing that thing? If yes, fix the proposal before reporting. + +## Operator directives + +Examples that count as atomic requirements: + +- "prefer additive changes over destructive ones" +- "don't touch the description field — validator rejected that + in the prior round" +- "address weakness 2 first; weakness 1 was already partially + addressed by the prior round" +- "the change must be a single contiguous diff hunk, not scattered + across the file" + +Examples that don't (context dumps — reject): + +- The prior proposal pasted in for you to "iterate on" +- The validator's prior verdict pasted in for you to "address" + +## Edge cases to surface as BLOCKED + +- **Weakness as named is too abstract to derive a concrete diff + from.** The analyzer may need to reformulate the weakness into + concrete sub-weaknesses. Surface; the operator invokes step 7 + to refine. +- **Weakness conflicts with `01-functionality.md`.** If + addressing the weakness would require contradicting the + skill's stated responsibilities, surface — there's a deeper + framing problem. +- **All anti-patterns are preempted by the skill's current + shape.** If the only diffs that would address the weakness are + on the anti-pattern list, the gate is doing its job — surface + honestly and exit. + +## Return summary + +After writing the proposal, return: + +- Weaknesses addressed (slugs) +- Principle applied (one line) +- Lines/sections of the skill modified (paths + brief description) +- Whether the self-check found any anti-pattern overlap (and how + you justified it) + +Keep it under 200 words. diff --git a/skills/investigate-functionality/SKILL.md b/skills/investigate-functionality/SKILL.md new file mode 100644 index 0000000..863f822 --- /dev/null +++ b/skills/investigate-functionality/SKILL.md @@ -0,0 +1,319 @@ +--- +name: investigate-functionality +description: Use when the user wants to understand what an existing agent skill does — phrases like "what does this skill do", "investigate this skill", "understand this skill", or when they hand you a URL or local path to a skill they want analyzed. Also triggers at the start of any skill-optimizer chain work, before test design, analysis, or improvement. Use even when the user doesn't explicitly say "investigate" — any phrasing that signals they want to understand a skill before doing anything with it should trigger this. +--- + +# investigate-functionality + +Step 1 of the skill-optimizer chain. **Fresh-derivation step.** Takes +a skill (URL or local path), runs a researcher subagent to figure out +what it's supposed to do, and writes +`docs/skill-optimizer//01-functionality.md` — the briefing +document every later step consumes. + +## What you produce + +A single report at `docs/skill-optimizer//01-functionality.md`, +where `` is the source skill's directory name (e.g., +`firecrawl-build-scrape`). + +Frontmatter (runtime-relevant facts only, per +[`frontmatter-discipline.md`](../shared/frontmatter-discipline.md)): + +```yaml +--- +skill_source: +pr_submission_intent: true | false +classification: +optimization_target: +optimization_target_source: +optimization_target_candidates: + - path: SKILL.md + source: + - path: command.md + source: +likely_wrapper: true | false +wrapper_points_to: +--- +``` + +**`classification`** — canonical types: `tool-use`, `code-patterns`, +`document`, `prose-guidance`, `meta`, `interactive`. If none fits +cleanly, write a short descriptive label of your own +(`dataset-extraction`, `deployment-runbook`, etc.) rather than +falling back to `other`. + +**`optimization_target`** — the file in `vendored-skill/` that the +chain optimizes and submits upstream. For non-wrapper skills this is +`SKILL.md`. For wrapper skills (where the source file fetches its +real rules from another URL) this is the underlying file +(e.g. `command.md`). Identified by the researcher subagent; surfaced +to the operator via a picker prompt at step (h) only when more than +one viable candidate exists. + +**`optimization_target_source`** — the upstream URL of the +optimization target. Step 2 (investigate-submissions) reads this +to determine which repo to research for PR conventions — may differ +from `skill_source` when a wrapper redirects to another repo. + +**`optimization_target_candidates`** — the candidate list the +researcher identified. Used by step (h)'s picker; downstream steps +ignore it (the resolved `optimization_target` is what matters). + +**`likely_wrapper`** + **`wrapper_points_to`** — diagnostic. Kept +for human review and debugging; they no longer drive behavior +(`optimization_target` is the load-bearing field). The researcher +still records them because downstream skill optimizers may surface +them in body sections. + +For the body template, see +[`agents/research-functionality.md`](./agents/research-functionality.md). +The source skill is vendored to +`.skill-optimizer//vendored-skill/` regardless of source +type (upstream fetch or local copy) so downstream steps have a +single canonical input — they never branch on local vs upstream, +and the user's original local file is not touched by the chain. +The vendored copy is gitignored (re-fetchable from +`skill_source` any time); only the reports and improved-skill +under `docs/skill-optimizer//` get tracked. + +## Workflow + +### (a) Confirm on a dedicated branch + create chain todos + +Two quick setup tasks before doing any research: + +**Branch check.** The chain writes to three locations: +`docs/skill-optimizer//` (reports + the improved-skill +deliverable, tracked), `skill-evals//` (test suites, +tracked), and `.skill-optimizer//` (vendored source + raw +bench output, gitignored). Keeping the tracked locations on a +feature branch makes the run easy to discard, iterate on, or +merge in one piece. + +**Honor an autopilot directive first.** If +`${OPERATOR_DIRECTIVES}` contains a line like +`branching: already-resolved ()`, the autopilot driver has +already handled branch creation for this run. Log "branch +confirmed by autopilot: " and proceed without prompting. + +**Otherwise, always ask the user explicitly** — don't infer from +branch name. After running `git rev-parse --abbrev-ref HEAD`, +surface this prompt verbatim: + +> I'm about to start an optimization run for ``. The chain +> will create tracked files under `docs/skill-optimizer//` +> and `skill-evals//`. **I recommend creating a dedicated +> branch** (e.g., `git checkout -b eval/`) so the run is +> easy to discard, iterate on, or merge in one piece. +> +> You're currently on ``. Options: +> +> 1. **Create `eval/` and switch to it now.** (Recommended.) +> 2. **Use the current branch.** Chain outputs will accumulate +> on `` — that's fine if it's already a +> purpose-named feature branch. +> 3. **Cancel and let me sort out branches first.** + +Default to recommending option 1 unless the current branch is +already a clearly purpose-named feature branch matching this +slug (e.g., `eval/`, `feat/`). If the user picks +option 1, run `git checkout -b eval/` for them before +proceeding. If they pick option 2, proceed on the current +branch. If option 3, exit so they can create the branch they +want and re-invoke. + +**Chain todos.** Create a TodoWrite list with the chain's 9 (or +10, if autopilot will run) steps so progress is visible across a +long run. Mark step 1 as `in_progress` immediately; subsequent +chain skills update their own entry as they fire. Suggested list: + +1. `step 1: investigate-functionality` (in_progress) +2. `step 2: investigate-submissions` (only if PR-bound) +3. `step 3: design-tests` +4. `step 4: write-tests` +5. `step 5: validate-tests` +6. `step 6: run-bench` +7. `step 7: analyze` +8. `step 8: improve` +9. `step 9: validate` +10. `step 10: autopilot` (only if running end-to-end) + +The chain is long enough that this is worth tracking, even though +each individual chain skill is itself a small workflow. + +### (b) Classify the source + +Upstream skill (URL or `//`) or local skill +(filesystem path that already exists)? If the user gave a bare name +with no URL and no path, ask to clarify. + +If the URL 404s or the local path doesn't exist, surface the error +to the user — don't guess at recovery. If the source is a plugin +with multiple skills, ask which one to investigate (produce one +report per skill). + +### (c) Ask about PR intent + +**Upstream skills:** ask "Do you want to optimize this skill for +upstream PR submission?" and record the answer in the report +frontmatter as `pr_submission_intent: true|false`. Capture this +decision now — step 2's handoff reads this field to decide whether +step 3 runs. + +**Local skills:** default to `pr_submission_intent: false`. But if +the user explicitly said they want to send this back to an upstream +maintainer, treat it as PR-intent: set true, ask where the upstream +contribution guidelines live, record their answer in the report +body under a "PR submission notes" subsection (step 3 uses this as +its starting point). + +If the user later changes their mind on PR intent, they re-run this +skill — the canonical is overwritten and prior state lives in git +history. + +### (d) Vendor the source + +Copy the skill's files into `.skill-optimizer//vendored-skill/` +(create the parent dirs if missing), regardless of source type: + +- **Upstream** — `gh api` or equivalent fetch of the skill's + directory contents +- **Local** — `cp -r` of the local skill's directory into the + vendored path + +**`vendored-skill/` is a single flat dir holding everything the +chain needs.** The researcher subagent (step (g)) is responsible +for identifying any external URLs the source content references at +runtime (e.g. a wrapper that WebFetches `command.md` from another +repo) and fetching them into `vendored-skill/` too — so the test +environment is fully self-contained and reproducible without live +network. You don't need to do that here; you just vendor the source +itself at this step. + +**Do not `cd` into the vendored directory.** The `Bash` tool +preserves CWD between calls — a bare `cd vendored-skill && curl ...` +will leak CWD into the rest of the chain, causing later relative +paths (e.g., `mkdir -p docs/skill-optimizer//`) to silently +resolve under `vendored-skill/`. Use one of: + +- **Absolute paths:** + `curl -sSfL -o //.skill-optimizer//vendored-skill/SKILL.md ...` +- **Subshell scoping (parentheses):** + `(cd .skill-optimizer//vendored-skill && curl -sSfL -o SKILL.md ...)` +- **Tool's `-o`/`-O` flags or `cp` directly:** + `cp source.md .skill-optimizer//vendored-skill/SKILL.md` + +If `.skill-optimizer//vendored-skill/` already exists from a +prior run, reuse it unless: (1) the source URL changed (upstream +— a different repo or skill is being investigated), or (2) the +user explicitly asks to re-vendor (e.g., they edited a local +skill between runs, or the upstream got new commits worth +refetching). + +The user's original local file is never modified by the chain — +the vendored copy is a separate read-only reference for +downstream steps. `.skill-optimizer/` is gitignored, so the +vendored copy doesn't add repo weight. + +### (e) Determine the slug and the report path + +`` is the source skill's own directory or file name (e.g., +`firecrawl/skills/firecrawl-build-scrape` → `firecrawl-build-scrape`; +`~/my-skills/pdf-cleanup/SKILL.md` → `pdf-cleanup`). + +Report path: `docs/skill-optimizer//01-functionality.md`. + +### (f) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Each invocation overwrites the canonical from the +current source plus directives; git captures prior state. Collect +`${OPERATOR_DIRECTIVES}` per the protocol. + +### (g) Dispatch the functionality-researcher subagent + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT do the research yourself in this +session** — dispatch the subagent via the `Agent` tool. Render the +prompt template at +[`agents/research-functionality.md`](./agents/research-functionality.md) +inline by substituting `${SKILL_SOURCE}`, `${OUTPUT_PATH}`, +`${PR_SUBMISSION_INTENT}`, `${OPERATOR_DIRECTIVES}`, and the +vendored path, then pass the rendered text as the Agent tool's +`prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol — do not pass a short prompt that points at the +template path). + +The subagent sees: the vendored skill files at +`.skill-optimizer//vendored-skill/`, targeted web-search +results, output path, frontmatter fields, +`${OPERATOR_DIRECTIVES}`. + +The subagent does NOT see: prior `01-functionality.md` or its git +history; existing analyses, tests, or failure data; the wider +chain's context. + +**Why this matters:** the operator session inherits prior +conversation framing, which biases research toward whatever the +user has already expressed. A subagent walled off from that +context produces a fresh derivation from the source itself, not a +rationalization of expectations. + +### (h) Confirm + handle optimization-target picker + hand off + +Verify the report file exists and frontmatter parses. + +**Pick the optimization target.** The researcher subagent has +recorded `optimization_target_candidates` (one entry per file it +identified as a viable target — typically the source `SKILL.md` +itself, plus any external file the source fetches at runtime that +it also vendored locally). It has also recorded its **recommended** +choice as `optimization_target` in the frontmatter. + +Resolve the choice: + +- **If `${OPERATOR_DIRECTIVES}` contains an `optimization-target:` + line** (autopilot passes this in full-auto mode): honor it + without prompting. Values: `auto-pick-recommended` (use the + researcher's recommendation as-is) or a specific relative path + matching one of the candidates. + +- **If `optimization_target_candidates` has exactly one entry** + (non-wrapper, non-ambiguous case): no prompt; the single + candidate is already recorded as `optimization_target`. Proceed. + +- **Otherwise (multiple candidates, no directive)**: surface this + prompt to the user: + + > The source you provided needs a clear optimization target — + > the file we'll modify and submit upstream as the PR. + > Multiple candidates exist in `vendored-skill/`: + > + > 1. **``** *(recommended)* — ``. + > Upstream source: ``. PR would target + > ``. + > 2. **``** — ``. + > Upstream source: ``. PR would target + > ``. + > 3. ... (one entry per remaining candidate) + > N. **Other** — name a relative path inside `vendored-skill/` + > not in the list. + > + > Say "default" or "go" to accept option 1. + + If the user picks a different candidate (or "other"), update the + `optimization_target` and `optimization_target_source` frontmatter + fields in `01-functionality.md` to reflect their choice. Use the + `Edit` tool — don't rewrite the whole file. + +Once resolved, hand off based on this report's +`pr_submission_intent` field: + +- **`true`** — next, invoke `skill-optimizer:investigate-submissions` + (step 2). After that, the chain proceeds to + `skill-optimizer:design-tests` (step 3). +- **`false`** — skip step 2 and invoke `skill-optimizer:design-tests` + (step 3) directly. No late "submit a PR?" prompts; the decision was + recorded above. diff --git a/skills/investigate-functionality/agents/research-functionality.md b/skills/investigate-functionality/agents/research-functionality.md new file mode 100644 index 0000000..781987e --- /dev/null +++ b/skills/investigate-functionality/agents/research-functionality.md @@ -0,0 +1,250 @@ +# Functionality-researcher subagent + +You are dispatched by `skill-optimizer:investigate-functionality` to +research what a skill is supposed to do and produce +`01-functionality.md` — the briefing document every later step in +the chain consumes. + +## Inputs (templated by the operator session) + +- `${SKILL_SOURCE}` — URL (`//`) or local + filesystem path to the skill being investigated (recorded in + the report frontmatter, but you read from `${VENDORED_PATH}`) +- `${VENDORED_PATH}` — `.skill-optimizer//vendored-skill/` directory (the operator + session has already copied the source skill here, whether the + original was upstream or local; this is your read path) +- `${PR_SUBMISSION_INTENT}` — `true` or `false`, captured at step 1 + by the operator session +- `${OPERATOR_DIRECTIVES}` — bulleted list of atomic new + requirements (may be empty on first invocation) +- `${OUTPUT_PATH}` — where to write the report (typically + `docs/skill-optimizer//01-functionality.md`) + +## What you see + +- The vendored skill's files at `${VENDORED_PATH}` (SKILL.md + any + references/, scripts/, assets/) +- Targeted web-search / web-fetch for the underlying technology the + skill is about +- The operator's directives + +## What you write (your job, in order) + +1. **Fetch all externally-referenced files** into `${VENDORED_PATH}` + so the test bench is reproducible without live network. See + "Vendoring referenced content" below. +2. **Identify optimization-target candidates** — record each viable + target file (the source SKILL.md + any underlying files you + vendored) into `optimization_target_candidates`. Recommend one. +3. **Write the briefing report** at `${OUTPUT_PATH}` with the + frontmatter shape below and the body covering classification, + responsibilities, dependencies, etc. + +## What you do NOT see + +- Prior `01-functionality.md` drafts or git history of the file +- Existing analyses, tests, improvement proposals from any prior + chain run +- Failure data from prior bench runs +- The wider chain's context (other skill files in this project) + +The point: your research must derive from the skill source itself +plus the underlying technology, not from accumulated optimization +context. If you rationalize prior expectations, every downstream +step inherits the drift. + +## Output: `01-functionality.md` + +Frontmatter (runtime-relevant only, no version-tracking metadata): + +```yaml +--- +skill_source: +pr_submission_intent: +classification: +optimization_target: +optimization_target_source: +optimization_target_candidates: + - path: + source: + rationale: + - path: + source: + rationale: +likely_wrapper: +wrapper_points_to: +--- +``` + +**`classification`** — pick the most accurate label for what kind +of skill this is. Canonical types and what they mean: + +- `tool-use` — procedures for using a specific tool, library, API +- `code-patterns` — code-level patterns or review checklists +- `document` — workflows that produce a document or file +- `prose-guidance` — writing-style or content-creation guidance +- `meta` — skills that operate on other skills or on agent behavior +- `interactive` — back-and-forth user dialogue + +If none of those fits cleanly, write a short descriptive label +(`dataset-extraction`, `deployment-runbook`, `ui-mockup-generation`, +etc.) — a specific label gives downstream steps a real handle to +work with. Don't fall back to `other`. + +Body (markdown) covering: + +1. **What the skill does** — one-paragraph summary in your own + words, not a copy-paste of the skill's description. +2. **Who uses it** — the intended audience (end-user agents, other + skills, specific operator types). +3. **When it should fire** — the user-language symptoms or contexts + that should trigger the skill (per the skill's description + field). +4. **Responsibilities** — bulleted list of distinct things the + skill is supposed to make the agent do. Each one is a candidate + for testing at step 3. +5. **Tools / dependencies** — what the skill assumes is available + (specific MCPs, CLIs, libraries, file conventions). +6. **Key terminology** — domain terms the operator must understand + to read the skill. +7. **Underlying technology** — brief overview of the + tool/library/API the skill wraps, grounded in web-fetched docs. +8. **PR submission notes** (only if `${PR_SUBMISSION_INTENT}` is + `true` AND source was local) — verbatim record of what the + user said at step 1 about where the upstream contribution + guidelines live (URL, CONTRIBUTING.md path, Slack channel, + whatever they provided). Step 2 uses this as its starting + point. +9. **Wrapper observation** (only if you set `likely_wrapper: + true` in the frontmatter — see protocol below) — describe + what made you suspect this is a wrapper, where the actual + content appears to live (`wrapper_points_to`), and how + confident you are. The operator will surface this to the user, + who decides whether to re-vendor the referenced content and + re-research, treat the wrapper as the skill, or cancel. + +## Vendoring referenced content (load-bearing) + +The test environment must be **fully self-contained** — the agent +running in a trial container should not need live network to +exercise the skill. As part of your research: + +1. Scan the skill files at `${VENDORED_PATH}` for any external URL + the source content references at runtime (e.g. a wrapper SKILL.md + that says "WebFetch this URL for the actual rules", or a `Read` + reference to a remote file). +2. For each such reference, **fetch the file into `${VENDORED_PATH}` + with its basename** (use absolute paths or subshell `(cd && curl)` + per the CWD discipline rules). Don't `cd` into `${VENDORED_PATH}` + in a way that persists across your tool calls. +3. After fetching, every URL referenced by the source content should + now have a local copy under `${VENDORED_PATH}`. Downstream chain + steps mount this directory as the test bench; an agent running + against it should be able to satisfy the skill's references + without external network. + +This applies whether the source is a wrapper or not. Many skills +reference `references/*.md` files that may be missing from the +vendored copy; pull anything cited via URL. + +## Identifying the optimization target + +The **optimization target** is the file the chain will modify +(step 8) and submit upstream (when PR-bound). For each viable +candidate file at `${VENDORED_PATH}`, record an entry in +`optimization_target_candidates`: + +```yaml +- path: + source: + rationale: +``` + +Common patterns: + +- **Non-wrapper skill**: one candidate, the source `SKILL.md`. Set + `optimization_target: SKILL.md` and `likely_wrapper: false`. +- **Wrapper skill** (the source file fetches its real rules from + another URL): two candidates — the wrapper SKILL.md itself, AND + the underlying file. Set `optimization_target` to the **underlying** + (the file with the actual rules — that's where meaningful + optimization happens) and `likely_wrapper: true`. +- **Ambiguous skill** (e.g., multi-file plugin where multiple files + carry rule content): record all viable candidates; recommend the + one most likely to be the rule corpus. The operator will surface + the picker to the user. + +The `optimization_target_source` field records the upstream URL of +your recommended target. This is what step 2 (investigate-submissions) +will use to decide which repo to research for PR conventions. + +## Reasoning protocol + +1. **Read the skill files first.** Don't research the technology + until you've read the SKILL.md and any references/scripts. The + skill's stated description is your starting frame. +2. **Identify the responsibilities by enumeration**, not by + paraphrase. If the SKILL.md says "Do X, then Y, then Z", those + are three responsibilities, not one. The test-case designer at + step 3 needs distinct items to propose probes for. +3. **Web-search the technology** only to fill gaps the skill + doesn't explain. Don't re-document everything the technology + does — focus on what an agent needs to know to USE the skill + correctly. +4. **Resolve `${OPERATOR_DIRECTIVES}` as atomic requirements**, not + as a context dump. If the operator says "you missed the vendor + CLA requirement", look for and document the CLA requirement; + don't paraphrase what was already there. +5. **Classify honestly.** If the skill is genuinely a mix (e.g., + tool-use + prose-guidance), pick the dominant type and mention + the secondary aspect in the body. Don't invent classification + subtypes. + +## Wrapper detection (diagnostic) + +While reading the source, judge whether the source file is +**likely a thin wrapper file** rather than the actual skill +content. Common patterns: + +- The SKILL.md body is short and mostly consists of references to + other files ("see X for the details", "uses content from Y", + pointers to a `content.md` or similar) +- Frontmatter has fields like `reference:`, `source:`, `canonical:`, + `extends:`, or similar that point at another file +- The plugin layout suggests a multi-agent structure — several + SKILL.md files in the same plugin, each appearing to thinly + wrap the same underlying content with agent-specific surface +- The skill's substantive content is clearly elsewhere (linked + files dwarf the SKILL.md, the SKILL.md says "implementation + in X") + +If you judge this to be likely a wrapper, set `likely_wrapper: +true` in the frontmatter and record where the actual content +appears to live in `wrapper_points_to`. The wrapper finding is +**diagnostic only** — it doesn't drive behavior; the +`optimization_target_candidates` list (and the operator's picker) +is what determines what gets optimized. You DO follow the pointer +and vendor the underlying file into `${VENDORED_PATH}` per the +"Vendoring referenced content" section above, and record both the +wrapper and the underlying as candidates. + +If you're not sure whether it's a wrapper, err toward +`likely_wrapper: false`. If there's only one viable optimization +target (the source SKILL.md itself), the operator will pick it +automatically with no surface to the user. + +## Return summary + +After writing `${OUTPUT_PATH}`, return a brief summary: + +- Classification +- 3-5 key responsibilities (the ones step 3 will design probes for) +- Any PR submission notes captured +- **Optimization target candidates** — list each `path` and a + one-line rationale; mark your recommended pick. Operator uses + this to decide whether to surface a picker to the user. +- **Wrapper finding** (only if `likely_wrapper: true`) — one-line + rationale + `wrapper_points_to` value. + +Keep it under 200 words — the operator session reads this for the +handoff message; full detail is in the report. diff --git a/skills/investigate-submissions/SKILL.md b/skills/investigate-submissions/SKILL.md new file mode 100644 index 0000000..1a7a437 --- /dev/null +++ b/skills/investigate-submissions/SKILL.md @@ -0,0 +1,149 @@ +--- +name: investigate-submissions +description: Use when the user wants to research a skill's upstream PR conventions — phrases like "research PR conventions for this skill", "what does the upstream repo require for contributions", "investigate submissions for X", or when prepping a PR-bound optimization run and you need to know the upstream's rules. Triggers mid-way through skill-optimizer chain work when `01-functionality.md` has `pr_submission_intent: true`. Use even when the user doesn't explicitly say "investigate submissions" — any phrasing about figuring out the upstream's contribution rules should trigger this. +--- + +# investigate-submissions + +Step 2 of the skill-optimizer chain — OPTIONAL, runs only when the +target skill is bound for upstream PR submission. **Fresh-derivation +step.** Takes the source slug, dispatches a researcher subagent that +uses the `gh` CLI to gather the upstream repo's contribution +conventions, and writes `docs/skill-optimizer//02-submissions.md` +— the verbatim-pastable context block the validator (step 9) uses for +its external consistency check. + +## What you produce + +A single report at `docs/skill-optimizer//02-submissions.md`. + +Frontmatter (runtime-relevant facts only, per +[`frontmatter-discipline.md`](../shared/frontmatter-discipline.md)): + +```yaml +--- +upstream_repo: / +upstream_branch_target: +license: +requires_cla: true | false +--- +``` + +- **`upstream_branch_target`** — some repos use `main` for + incremental changes and `next` for new skills. The subagent + determines this from recent merged PRs so the validator (step 9) + can check the proposed PR targets the correct branch. +- **`requires_cla`** — true if the upstream requires a Contributor + License Agreement before merging. + +Body covers license details, frontmatter spec extracted from +existing skills, file-location conventions, prefix taxonomy, +PR-shape patterns from recent merged + closed-without-merge PRs, +and rejection signals. Body template lives in +[`agents/research-submissions.md`](./agents/research-submissions.md). + +## Workflow + +### (a) Confirm prerequisites + +Two prerequisites: + +1. `01-functionality.md` must exist with valid frontmatter. If + not, tell the user to run + `skill-optimizer:investigate-functionality` first. +2. That report's `pr_submission_intent` field must be `true`. If + `false`, this skill should not run — tell the user step 2 is + skipped for local-only optimization runs. + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Each invocation overwrites the canonical from +current upstream facts plus directives. Staleness against +`01-functionality.md` is determined by git-mtime comparison. + +This report rarely needs re-running — upstream PR conventions +change slowly. Most common reason: upstream updated +`CONTRIBUTING.md` or CLA requirements. Surface as a directive if +known. + +### (c) Dispatch the submission-researcher subagent + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT scrape the upstream repo yourself +in this session** — dispatch the subagent via the `Agent` tool. +Render the prompt template at +[`agents/research-submissions.md`](./agents/research-submissions.md) +inline by substituting `${UPSTREAM_REPO}` from +`01-functionality.md`'s `optimization_target_source` field (NOT +`skill_source` — the optimization target may live in a different +upstream repo than the source URL the user pointed at, e.g. when +a wrapper SKILL.md in repo A fetches its rules from `command.md` +in repo B; the PR goes to repo B), `${OUTPUT_PATH}`, and +`${OPERATOR_DIRECTIVES}`, then pass the rendered text as the +Agent tool's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +Derive `${UPSTREAM_REPO}` from `optimization_target_source` by +extracting `/` from the URL. If the URL form is +unfamiliar (e.g., GitLab / Bitbucket), surface to the user — see +Edge cases below. + +The subagent sees: the upstream repo via `gh` CLI (PR list — both +merged and closed-without-merge for shape patterns and rejection +signals, repo-file API, `CONTRIBUTING.md`, license file, existing +skill files); the skill slug being researched; +`${OPERATOR_DIRECTIVES}`; output path. + +The subagent does NOT see: its own prior `02-submissions.md` or +git history of it; any information about the proposed change +being optimized; existing analyses, tests, or failure data; the +vendored skill source. + +**Why this matters:** the validator (step 9) will later check the +proposed change against this report. If the operator session does +the research, it has already absorbed the optimization context — +the proposed change, prior failures, user's framing — and biases +the report toward documenting upstream conventions in ways that +justify the proposed change. A subagent walled off from that +produces neutral upstream facts; the validator's independence is +preserved. + +Edge cases the subagent will surface as blockers: + +- **Upstream repo is private or requires auth** — ask the user to + authenticate `gh` and re-dispatch +- **No merged PRs in the upstream's history yet** — the report + will be thinner; the subagent says so honestly rather than + making up patterns +- **Upstream uses non-discoverable frontmatter conventions** — if + the subagent can't extract a consistent spec from recent merged + PRs, the report will say so; the optimizer (step 8) then makes a + judgment call rather than mechanically conforming + +The subagent returns a brief summary: license, CLA requirement, +branch target, any high-risk rejection signals. + +### (d) Confirm subagent output + +Verify the report exists with parseable frontmatter and the +expected sections. If `requires_cla: true`, mention it explicitly +on handoff — the operator needs to sign the CLA before any PR can +be merged. + +### (e) Hand off + +Report the file path, a one-line summary (license / CLA / branch +target / any flagged blockers), then: + +> Next, invoke `skill-optimizer:design-tests`. The validator in +> `skill-optimizer:validate` will later read this report for its +> external consistency check. + +## Edge cases + +- **Upstream uses a non-`gh`-friendly host (GitLab, Bitbucket, + etc.)** — the current chain assumes GitHub-hosted upstreams. + Surface this and tell the user; non-GitHub cases require manual + research and pasting the report content directly. diff --git a/skills/investigate-submissions/agents/research-submissions.md b/skills/investigate-submissions/agents/research-submissions.md new file mode 100644 index 0000000..b3c0bfd --- /dev/null +++ b/skills/investigate-submissions/agents/research-submissions.md @@ -0,0 +1,174 @@ +# Submission-researcher subagent + +You are dispatched by `skill-optimizer:investigate-submissions` to +research the upstream repo's PR conventions and produce +`02-submissions.md` — the verbatim-pastable context block the +validator (step 9) uses for its external consistency check. + +## Inputs (templated by the operator session) + +- `${UPSTREAM_REPO}` — `/` (from + `01-functionality.md`'s `skill_source` field) +- `${SKILL_SLUG}` — the specific skill's slug within the repo + (lets you look at PRs in the same skill category for closer-match + shape patterns) +- `${OPERATOR_DIRECTIVES}` — bulleted list of atomic new + requirements (may be empty) +- `${OUTPUT_PATH}` — where to write the report (typically + `docs/skill-optimizer//02-submissions.md`) + +## What you see + +- The upstream repo via `gh` CLI: + - `gh pr list` (both merged and closed-without-merge for shape + patterns and rejection signals — sample enough recent PRs to + establish the pattern, but you don't need every PR ever) + - `gh api` for repo files (CONTRIBUTING.md, LICENSE, existing + skill files for frontmatter spec extraction) +- The operator's directives + +## What you do NOT see + +- Your own prior `02-submissions.md` or git history of it +- Any information about the proposed change being optimized — the + report is purely upstream facts, not advocacy for a change +- Existing analyses, tests, failure data from the chain +- The vendored skill source (you research the UPSTREAM REPO, not + the skill being optimized) + +The point: the validator at step 9 must trust this report as +independent. If you absorb optimization context, you bias the +report toward justifying the proposed change — and the validator +loses its real independence. + +## Output: `02-submissions.md` + +Frontmatter (runtime-relevant only): + +```yaml +--- +upstream_repo: / +upstream_branch_target: +license: +requires_cla: true | false +pr_target_repo: / +pr_target_path: +linked_consumers: + - : # other SKILL.md files referencing the same content + - ... +--- +``` + +- **`upstream_branch_target`** — determine from recent merged PRs. + If conventions differ (e.g., bug fixes go to `main`, new + features go to `next`), record the rule rather than a single + branch name. +- **`requires_cla`** — `true` if the upstream requires a + Contributor License Agreement. +- **`pr_target_repo` / `pr_target_path`** — derived directly from + `01-functionality.md`'s `skill_source`. Step 1's user gate + already resolved any wrapper-vs-content question: if the source + was a wrapper and the user opted to optimize the underlying, + step 1 re-vendored and `skill_source` now points at that + underlying content. You don't re-litigate the choice here. +- **`linked_consumers`** — other SKILL.md files (same repo or other + known repos) that reference the same content as `pr_target_path`. + Flagged so the PR composer can coordinate or be honest about + scope. Same-repo wrappers typically update automatically; cross- + repo consumers usually need follow-up PRs. + +Body sections: + +1. **License** — full SPDX identifier + brief plain-English summary + (permissive / copyleft / proprietary). +2. **CLA requirement** — what kind (DCO, individual CLA, corporate + CLA), how it's signed, link to the CLA tool if applicable. +3. **Frontmatter spec** — extracted from a sample of existing + skills in the repo. List required fields, optional fields, value + conventions. If the repo's skills don't have a consistent + frontmatter, say so rather than inventing a standard. +4. **File-location conventions** — where new skills go (which + directory, which subdir pattern), where references/scripts go. +5. **Prefix taxonomy** — if the repo uses commit prefixes + (`feat:`, `fix:`, `docs:`, etc.) or PR title prefixes, extract + the taxonomy from recent merged PRs. +6. **PR-shape patterns** — from a sample of recent merged PRs: + what does a typical PR description include (rationale, testing + notes, screenshots)? What's the typical PR size (single-file + diff, multi-file)? Additive-only or destructive changes + accepted? +7. **Rejection signals** — from a sample of closed-without-merge + PRs: what got rejected and why? Common patterns to AVOID. +8. **Linked consumers** (only if non-empty) — list the other + SKILL.md files referencing the same content, with a one-line + note on each: same repo or different, and whether a change at + `pr_target_path` propagates automatically or needs a follow-up + PR there. + +If you can't establish any of these from the available data, say +so honestly in the relevant section rather than making up +conventions. + +## Reasoning protocol + +1. **Read `01-functionality.md` for `skill_source`.** That URL/path + IS your PR target — step 1's user gate already resolved any + wrapper question. If `likely_wrapper: true` is set (rare: user + opted to keep the wrapper rather than re-vendoring the + underlying), the PR target is still `skill_source` (the + wrapper itself); `wrapper_points_to` is documentation only. + +2. **Start with CONTRIBUTING.md** — if it exists, it explicitly + states most of what you need (license, CLA, file conventions, + PR shape). + +3. **Sample recent merged PRs** — enough to identify the dominant + patterns. Quality over quantity; you want to see the actual + accepted shapes. + +4. **Sample closed-without-merge PRs** — these surface the + rejection signals. The CONTRIBUTING.md tells you the rules; + the closed PRs show what happens when rules are broken. + +5. **Look at PRs in the same skill category** if `${SKILL_SLUG}` + suggests one (e.g., browser-skills look at other browser PRs). + Specific-category patterns trump generic repo-level patterns. + +6. **Find linked consumers.** Search for other SKILL.md files in + the same repo (or other known repos) that reference the same + content as `pr_target_path`. These are coordination notes for + the PR composer — same-repo wrappers typically update + automatically when the target changes; cross-repo consumers + usually need follow-up PRs in their own repos. + +7. **Resolve `${OPERATOR_DIRECTIVES}` as atomic requirements** + (e.g., "include the vendor's CLA requirement explicitly" → + make sure the CLA section is prominent). + +## Edge cases to surface + +If you hit any of these, surface to the operator session as a +blocker rather than making up content: + +- **Upstream repo is private or requires auth** — `gh` will fail; + surface so the user can authenticate +- **Upstream uses a non-`gh`-friendly host** (GitLab, Bitbucket, + etc.) — surface; the current chain assumes GitHub +- **No merged PRs in the repo yet** — write a thinner report and + note the absence; don't invent shape patterns +- **Frontmatter conventions vary too wildly to extract a spec** — + document the variance honestly + +## Return summary + +After writing `${OUTPUT_PATH}`, return a brief summary: + +- PR target (the `pr_target_repo` / `pr_target_path` you derived) +- License + CLA requirement +- Branch target rule +- Linked consumers count (if any) +- 2-3 most important rejection signals from the closed-PR sweep +- Any blockers (auth failures, missing data, etc.) + +Keep it under 200 words. The operator session uses this for the +handoff message; full detail is in the report. diff --git a/skills/run-bench/SKILL.md b/skills/run-bench/SKILL.md new file mode 100644 index 0000000..2b71da8 --- /dev/null +++ b/skills/run-bench/SKILL.md @@ -0,0 +1,139 @@ +--- +name: run-bench +description: Use when the user wants to run the eval suite against a skill and capture results — phrases like "run the bench", "measure the skill", "benchmark this", "run the eval suite", "execute the workbench". Also triggers mid-way through skill-optimizer chain work, once `skill-evals//suite.yml` exists from step 4 and the next thing is to measure. Use even when the user doesn't explicitly say "bench" — any phrasing about running the test suite for the skill should trigger this. +--- + +# run-bench + +Step 6 of the skill-optimizer chain. **Fresh-derivation step (for +the summary).** Invokes the skill-optimizer CLI's `run-suite` +command against `skill-evals//suite.yml` (generated by step 4), captures +the raw results under a timestamped directory, and writes a small +summary report that step 7 reads. No subagent dispatch — this is a +thin operator-driven CLI step. + +## What you produce + +Two outputs at `docs/skill-optimizer//`: + +1. **`.skill-optimizer//bench-results//`** — raw CLI output: + `suite-result.json`, per-trial `trace.jsonl`, per-trial + `findings.txt`, preserved workspaces. Timestamped per + invocation; old runs are NEVER overwritten. Outside the + iteration protocol — each timestamped dir is its own + naturally-accumulated archive. + +2. **`06-bench-summary.md`** — single canonical aggregate, per + [`frontmatter-discipline.md`](../shared/frontmatter-discipline.md): + + ```yaml + --- + bench_results_path: .skill-optimizer//bench-results// + overall_pass_rate: + --- + ``` + + Body has per-model pass rates, per-probe pass rates, a + failed-probe pointer list, pointer to the timestamped raw + directory. Trace excerpts and findings detail stay in the raw + output; the summary is an aggregate, not a dump. + +## Workflow + +### (a) Confirm prerequisites + +`skill-evals//suite.yml` must exist (per step 4's output contract). If +missing, tell the user to complete step 4 first. +`.skill-optimizer//vendored-skill/` should exist (step 1 +vendored the source regardless of upstream/local). + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Two pieces: + +- **Raw bench output** at `.skill-optimizer//bench-results//`: always + write to a fresh timestamped directory (`date -u +%Y%m%dT%H%M%SZ`). + Outside the iteration protocol; never overwrite. +- **Summary file** at `06-bench-summary.md`: fresh-derivation per + the protocol. Staleness check: if `skill-evals//suite.yml` has a newer + git mtime than `06-bench-summary.md`, the existing summary is + stale. If summary is current AND no re-run directive, tell the + user and exit. + +### (c) Run the bench + +```bash +TIMESTAMP=$(date -u +%Y%m%dT%H%M%SZ) +OUT_DIR="docs/skill-optimizer//06-bench-results/${TIMESTAMP}" +mkdir -p "${OUT_DIR}" + +npx tsx /src/cli.ts run-suite \ + skill-evals//suite.yml \ + --out "${OUT_DIR}" \ + --trials 3 +``` + +`--trials 3` is the chain default — enough to distinguish flaky +from systematic failures at step 7. Honor a different count if +requested. Runs (agent + model pairs) come from `suite.yml` (per +project invariant: `run-suite` does NOT take an override). Stream +stdout/stderr to the user — bench runs take minutes to hours. + +Environment failures to surface honestly (don't try to recover): + +- **Required auth not available** — fail fast. Ask the user to set + the appropriate env var or run the relevant `… login` command: + `ANTHROPIC_API_KEY` / `claude login` for `claude-agent-acp`; + `OPENAI_API_KEY` / `codex login` for `codex-acp`; `GOOGLE_API_KEY` + / `gemini auth login` for `gemini`; `OPENAI_API_KEY` for + `opencode`; `OPENROUTER_API_KEY` for `pi-acp`. Don't mock. +- **Docker image missing** — default is + `skill-optimizer-agent:local`. Tell the user to build it + (`docker build -t skill-optimizer-agent:local -f + docker/skill-optimizer-agent.Dockerfile .`). + +### (d) Write the summary + +Parse `${OUT_DIR}/suite-result.json` and write `06-bench-summary.md` +with the frontmatter above plus a body containing: + +- **Overall:** total trials, passed, failed, overall pass rate. Total + tokens consumed. Total duration. +- **Per agent + model:** trial count, pass rate, mean tokens per + trial, mean duration per trial for each row in `runs:` from the + suite. +- **Per probe:** trial count, pass rate, one-line note if any + trial failed. Mean tokens and duration per probe. +- **Failed-probe pointer list:** probe IDs where any trial failed, + with paths to their `trace.jsonl` and `findings.txt`. +- **Raw output:** the `bench_results_path` value. + +Do NOT include a cost column. tokens × downstream pricing is computed +offline if needed. + +If all trials errored (no graded results), that's a workbench +misconfiguration or environmental failure rather than a skill +weakness. Write the summary honestly and surface before handing +off to step 7. + +Don't auto-invoke step 7. + +### (e) Hand off + +Read `overall_pass_rate`. Two messages: + +- `< 1.0`: invoke `skill-optimizer:analyze` to diagnose + failures. +- `== 1.0`: surface the choice — accept that probes don't expose a + weakness, or re-run step 3 with a "make probes harder" + directive. + +## Edge cases + +- **Known limitation (deferred):** the CLI's `run-suite` does not + accept a case filter, so there's no first-class partial-rebench + mode. Operators with expensive suites can run `run-case` + manually for changed probes and splice into the prior + `.skill-optimizer//bench-results//` dir — outside-the-chain escape hatch, + not a supported flow. diff --git a/skills/shared/acp-trace-format.md b/skills/shared/acp-trace-format.md new file mode 100644 index 0000000..25b0555 --- /dev/null +++ b/skills/shared/acp-trace-format.md @@ -0,0 +1,87 @@ +# ACP trace format (for chain analyzer) + +`trace.jsonl` captures the raw Agent Client Protocol (ACP) messages +exchanged between the workbench (client) and the agent CLI (server) +for one trial. The first line is a `trace_start` header with trial +metadata; every subsequent line is a JSON-RPC envelope per the ACP +spec at . + +This doc summarizes the message types the chain analyzer cares about. +For the full spec, follow the link above. + +## Header (line 1) + +```json +{ + "type": "trace_start", + "schemaVersion": 2, + "caseName": "...", + "agent": "claude-agent-acp", + "model": "claude-haiku-4-5-20251001", + "startedAt": "ISO-8601", + "endedAt": "ISO-8601" +} +``` + +## Initialization (early lines) + +- `initialize` request from client; `initialize` response from agent +- `session/new` request; response carries `sessionId` + +These confirm the agent started. If they're absent, the trial failed +before reaching the prompt — bench infrastructure issue, not skill +weakness. + +## Session updates (the meat) + +All have shape: + +```json +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"...","update":{...}}} +``` + +The `update` object's `sessionUpdate` field tells you what happened: + +| `sessionUpdate` value | Meaning | Analyzer cares because | +|---|---|---| +| `agent_message_chunk` | Streaming chunk of assistant text | Final response — what the agent told the user | +| `agent_thought_chunk` | Streaming chunk of assistant reasoning | The agent's reasoning — useful for diagnosing why it did X | +| `tool_call` | Agent is calling a tool (start) | `kind` field tells you what kind: `execute`, `read`, `write`, `edit`, `search`, `fetch`, `think`, `other` | +| `tool_call_update` | Tool result/progress (end) | `status: "completed" \| "failed"`, `content` carries the tool output | +| `plan` | Agent's high-level plan | Optional sidebar; skip in most analysis | +| `user_message_chunk` | Agent echoing user input | Rare; skip | + +## Final prompt response (last numbered response) + +```json +{ + "jsonrpc": "2.0", + "id": , + "result": { + "stopReason": "end_turn" | "max_tokens" | "refusal" | "cancelled", + "usage": { + "inputTokens": 120, + "outputTokens": 45, + "cacheReadTokens": 0, + "cacheCreationTokens": 0 + } + } +} +``` + +`stopReason` is critical for diagnosing failures: + +- `end_turn` — normal completion (still check `findings.txt` for correctness) +- `max_tokens` — agent ran out of context; skill may be too verbose +- `refusal` — agent declined the task; skill description may have triggered a safety pattern +- `cancelled` — workbench timed out the prompt + +## Helpers + +Don't parse the JSONL manually. Use `src/workbench/parse-trace.ts`: + +- `iterMessages(jsonl)` — assistant/user messages (text + thinking) +- `iterToolCalls(jsonl)` — paired tool_call + tool_call_update +- `computeMetrics(jsonl)` — tokens, duration, per-tool counts, stopReason +- `getFinalAssistantMessage(jsonl)` — last assistant chunk concatenated +- `getFailureEvidence(jsonl)` — failed-tool result snippets diff --git a/skills/shared/frontmatter-discipline.md b/skills/shared/frontmatter-discipline.md new file mode 100644 index 0000000..5434de3 --- /dev/null +++ b/skills/shared/frontmatter-discipline.md @@ -0,0 +1,40 @@ +# Frontmatter discipline + +Every chain artifact's frontmatter follows one rule: **runtime-relevant +facts only, never history.** Git already content-addresses every prior +state and `git log` gives you the timeline; reimplementing either in +frontmatter is bookkeeping for its own sake — it adds maintenance +surface (bumping, syncing, risk of drift) without enabling anything +git can't already do. + +## Do not add + +- `version:` field (use git) +- `archive/` subdirectory (use git) +- `inputs.step_N: ` lineage trackers (use git mtime comparison + per the iteration protocol) +- `last_derived_at:` timestamps (use git log) +- Any "I'm tracking how many times this file has been written" + metadata + +## Keep + +Frontmatter fields a chain skill needs to READ at runtime to do its +job: + +- `picked: true|false` (step 3 functionality spec) +- `pr_submission_intent: true|false` (step 1 functionality report) +- `classification: tool-use` (step 1 functionality report) +- `all_probes_approved: true|false` (step 5 tests verdict — gates step 6) +- `overall_pass_rate` (step 6 bench summary — gates step 7 dispatch) +- `has_structural_weakness: true|false` (step 7 analysis — gates step 8) +- `verdict: approve|needs-revision|reject` (step 9 verdict — gates + materialization) + +## Decision aid + +When in doubt about whether a field belongs: ask whether a chain +skill needs to READ it to do its job right now. + +- Yes → keep +- Recording for future debugging / audit → that's git's job, drop it diff --git a/skills/shared/iteration-protocol.md b/skills/shared/iteration-protocol.md new file mode 100644 index 0000000..ba0fa5b --- /dev/null +++ b/skills/shared/iteration-protocol.md @@ -0,0 +1,137 @@ +# Iteration protocol + +**Load this file every time** a skill-optimizer chain skill executes +its "Handle iteration" step. The mechanics are identical across +every skill in the chain; the parent skill's SKILL.md only specifies +that-skill's slot in the chain. The actual re-run logic + step-kind +classification + destructive-edit safety live here so the chain +behaves consistently. + +This file covers **iteration mechanics only** — when to re-run, how +to detect staleness, how to handle accumulating state vs. +fresh-derivation, and how to checkpoint before destructive edits. + +Related shared docs (load lazily at the workflow step that needs +them): + +- [`subagent-dispatch.md`](subagent-dispatch.md) — what subagents + see / don't see, operator-directive concept, dispatch input + templating, no-auto-invocation rule +- [`frontmatter-discipline.md`](frontmatter-discipline.md) — what + belongs in frontmatter (runtime facts) vs. what doesn't (history; + that's git's job) + +## Step kinds: fresh-derivation vs maintenance + +Steps in the chain come in two kinds. The subagent's reading rule +differs between them (full details in +[`subagent-dispatch.md`](subagent-dispatch.md)); for iteration +purposes the difference is whether the canonical state accumulates +across re-runs or is regenerated. + +**Fresh-derivation steps** produce their canonical artifact from +upstream + directives. Re-runs overwrite; git captures prior state. + +| Step | Why fresh derivation | +|---|---| +| 1. investigate-functionality | Each invocation researches from source — no continuity needed | +| 2. investigate-submissions | Each invocation researches upstream — no continuity needed | +| 5. validate-tests | Each probe judged fresh against its parent functionality; load-bearing for anti-ducktape | +| 6. run-bench (summary file) | Mechanical write-up of the new raw bench run, not an extension of prior summary | +| 7. analyze | Must not be biased by prior analyses; load-bearing for anti-ducktape | +| 8. improve | Optimizer must not see prior attempts; load-bearing for anti-ducktape | +| 9. validate | Validator must not be biased by its prior verdicts; independence-from-self is load-bearing | + +**Maintenance steps** manage an accumulating tree on disk — the +filesystem itself is the state. Re-runs read the current tree and +extend or modify it. + +| Step | What accumulates | +|---|---| +| 3. design-tests | `skill-evals///spec.yaml` grows as functionalities are added/refined | +| 4. write-tests | `skill-evals////` probes grow as the user adds coverage | + +## Staleness detection (git-native) + +Whether an artifact is stale is determined by comparing git +modification times against direct upstream: + +```bash +UPSTREAM_T=$(git log -1 --format=%ct -- docs/skill-optimizer//01-functionality.md) +DOWNSTREAM_T=$(git log -1 --format=%ct -- skill-evals//) +[ "$UPSTREAM_T" -gt "$DOWNSTREAM_T" ] && echo "stale" +``` + +In interactive use, the operator typically just knows ("I re-ran +step 1, so step 3 needs a re-run"). The git-mtime check is for +auto-pilot (step 10), which walks the chain forward and re-runs any +downstream older than its direct upstream. + +For maintenance steps, the operator's directives can also force a +re-run even when no upstream changed ("add a probe for null +input"). Staleness flags a recommended re-run; it doesn't require +one. + +## Safe destructive edits + +When a maintenance step's re-run will overwrite or delete existing +content (e.g., step 4 rebuilds a probe; step 3 removes a de-picked +functionality), the operator session commits the current state +**before** dispatching the destructive change: + +```bash +git add -A docs/skill-optimizer// +git commit -m "checkpoint: before rebuild of " +``` + +This gives git history a clean before/after breakpoint. +Without this checkpoint, the prior state of a rebuilt probe (or +de-picked functionality) gets buried inside a multi-file commit +later, making "show me what this used to look like" awkward. + +Fresh-derivation steps are also destructive (they overwrite the +canonical file), but the prior state is already self-contained in +its own prior commit, so no checkpoint is needed. + +## Cascading staleness + +Each step checks **direct upstream only** — the immediate prior +artifact(s) it consumes, not the whole upstream chain. If step 1 +is updated but step 3 is not re-run (the user judged step 3 still +valid against the new step 1), step 4 will see step 3 as current +even though step 3's git mtime is older than step 1's. + +The discipline: skipping a step's re-run is an **explicit operator +judgment** that the existing artifact is still valid against the +new upstream. Auto-pilot doesn't make this judgment — it always +re-runs on direct-upstream mtime mismatch, so the cascade +propagates naturally as you walk the chain forward. + +## The bootstrapping case + +On the very first invocation of a step on a fresh slug, +`docs/skill-optimizer//` may not yet exist. Create it as +part of the iteration step's setup. No `archive/` subdirectory is +needed — git is the archive. + +## What this protocol does NOT cover + +- **Step 6's raw bench results** are timestamped under + `.skill-optimizer//bench-results//`. Each run preserves + naturally as its own directory. Gitignored — `bench-results/` + accumulates trace.jsonl and findings.txt per trial (tens of MB + per run); the durable record is the + `docs/skill-optimizer//06-bench-summary.md` aggregate, + which IS tracked and falls under this protocol. +- **Auto-pilot's summary report** at + `docs/skill-optimizer//autopilot-summary-.md` is + timestamped per run. Each auto-pilot invocation produces a fresh + summary; old ones are preserved naturally. +- **`.skill-optimizer//vendored-skill/`** (the canonical + input to the chain — step 1 copies the source skill here + regardless of upstream/local) is reused across iterations of the + same slug unless: (1) the source URL changed (upstream), or (2) + the user explicitly asks to re-vendor (e.g., they edited the + local skill between runs). The source URL itself is the + identity; no versioning. Gitignored — re-fetchable from + `skill_source` any time. diff --git a/skills/shared/skill-design-philosophy.md b/skills/shared/skill-design-philosophy.md new file mode 100644 index 0000000..2d5a5aa --- /dev/null +++ b/skills/shared/skill-design-philosophy.md @@ -0,0 +1,145 @@ +# Skill design philosophy + +Read by the analyzer (step 7), optimizer (step 8), and validator +(step 9) when reasoning about skill quality. Every per-weakness +principle, every proposed change, and every verdict rationale +should connect to a specific entry below — if it doesn't, the +reasoning is under-grounded. + +## Core principles + +1. **Progressive disclosure is the organizing principle.** Metadata + always loaded (~100 tokens), SKILL.md body on trigger (target + <500 lines), bundled references / scripts only when needed. + Heavy reference material belongs in flat one-level + `references/.md` files, NOT inlined. + +2. **The description IS the triggering contract.** Third person, + starts with "Use when…", names concrete triggering conditions + (symptoms, error messages, contexts). It must answer "should + Claude load this skill right now?" — not "what does this skill + do?". Workflow summaries in descriptions cause shortcuts where + the agent follows the description and skips the body. + +3. **Conciseness is public stewardship.** Every paragraph competes + with conversation history and other skills for context. Challenge + each paragraph: "Does Claude already know this? Does this + justify its cost?" Remove a sentence and ask whether Claude + would make mistakes without it. + +4. **Specificity matches task fragility.** High-stakes / fragile + operations (destructive edits, security boundaries) need + procedural specifics. Flexible / creative operations tolerate + high-level guidance and degrade under over-prescription. The + common framing: narrow bridge needs handrails, open field + doesn't. + +5. **Explain WHY, not just WHAT.** Smart models reason past rote + instructions when they understand intent. "ALWAYS"/"NEVER" + without rationale is a yellow flag; reframing to explain the + underlying reason is usually more effective AND more concise. + +6. **Procedural + declarative hybrid.** Pure procedural scripting + under-performs on planning. Pure declarative under-performs on + agentic execution. Effective skills mix: declarative reasoning + constraints ("analyze before acting") plus procedural numbered + steps where order matters. + +7. **Evaluation-driven content.** Write skill content (or + improvements) only for gaps that have been measured. Build + evals first, establish a baseline (agent without the skill), + then add the minimal content that closes the gap. Anticipatory + content for hypothetical edge cases is bloat. + +8. **One excellent example beats many mediocre ones.** Multiple + examples in the same shape risk the model copying them + verbatim. Pick one canonical example, make it complete and + runnable, explain WHY each part exists. + +9. **Single-action granularity, not kitchen-sink.** Split read and + write into separate skills/tools so confirmation flows work. + Overlapping triggers between sibling skills is a primary + structural weakness — the model hesitates or picks wrong. + +10. **Knowledge vs. instructions separation.** Reference material + (data, API docs, tables) belongs in attached files. Behavioral + rules and workflow guidance belong in the instruction body. + Mixing them dilutes both. + +11. **Executable scripts > generated code.** For deterministic + operations, a pre-written utility script (in `scripts/`) is + more reliable and saves tokens compared to asking the agent + to write the same code inline. + +12. **Validation loops > one-shot execution.** For multi-step or + destructive operations, include verifiable intermediate + outputs (JSON checklists, plan files) that can be checked + before the next step. Catches errors before the cost is paid. + +## Anti-patterns (always flag) + +- **Workflow summaries in the description field.** Causes the + agent to follow the description and skip the body. +- **Narrative bloat.** Explaining what Claude already knows + (e.g., "PDFs are documents that contain text…"). +- **MUST/NEVER without rationale** for non-critical operations. +- **Magic numbers** in scripts or skill prose, with no comment + explaining why this number. +- **Try blocks that swallow + re-throw** ("punt to Claude" — push + error reasoning onto the agent without giving it information). +- **Multi-language dilution.** Re-implementing the same example + in JS + Python + Go. One excellent example > three mediocre. +- **Inconsistent terminology** ("field"/"box"/"element" mixed) — + degrades retrieval and confuses the agent. +- **Many options without a default.** "You can use X, Y, or Z" — + pick one and provide an escape hatch. +- **Time-sensitive content** ("before August 2025") in the main + body — relegate to a collapsible "old patterns" section. +- **Tool/skill descriptions without a "Use when…" clause.** +- **Persuasive language** ("critically important", "you MUST + always", emoji), instead of clean explanation. +- **Active skill set >20** — at that size, triggering accuracy + degrades and overlap becomes the dominant failure mode. +- **Monolithic SKILL.md** that should have been decomposed via + `references/` or sibling skills. +- **First-person voice** in descriptions ("I help you with X"). + +## Principled vs. ducktape — the operational test + +The chain's anti-ducktape architecture rests on this distinction. +When judging a proposed improvement: + +| Ducktape (symptomatic patch) | Principled (mechanism change) | +|---|---| +| Adds tokens to "make it more clear" | Removes tokens not pulling their weight, OR adds tokens that close a measured eval gap | +| Adds bare MUST/NEVER for one observed failure | Reframes the rule with reasoning, or restructures hierarchy | +| Inlines more content in existing structure | Decomposes into flat `references/.md` | +| Patches the specific failing input shape | Adds an eval that captures the failure mode AND a general principle | +| Inflates description to be "more discoverable" | Adds concrete triggering symptoms / exclusions | +| Generates code ad-hoc each invocation | Bundles a script in `scripts/` and tells the agent to use it | +| Adds repetitive emphasis ("CRITICAL: NEVER…") | Explains WHY the rule matters and where the cost lands | +| Adds the patch for THIS bench's failures | Names the structural weakness any reasonable bench would surface | + +The convergent test: **does the change remove tokens that aren't +pulling their weight, or add tokens that close a measured eval +gap?** If neither — it's bloat at best, ducktape at worst. + +A second test: **the principled response to recurring failure is +often to REFRAME, not to add more constraints.** If you find +yourself stacking MUST clauses against a stubborn issue, try a +different metaphor or restructure the section instead. + +## How each subagent uses this doc + +- **Analyzer (step 7).** "What WOULD address this" must align with + one or more core principles above. "What WOULD NOT address this" + must call out specific ducktape moves from the rubric. + +- **Optimizer (step 8).** The proposed change maps to one or more + named principles. The self-check walks the analyzer's named + anti-patterns AND the universal anti-pattern list here. + +- **Validator (step 9).** The verdict cites which principles the + change embodies and which anti-patterns it avoids. "Additive vs + destructive" and "general vs ducktape" are concrete rubric tests, + not abstractions. diff --git a/skills/shared/subagent-dispatch.md b/skills/shared/subagent-dispatch.md new file mode 100644 index 0000000..0cce4ea --- /dev/null +++ b/skills/shared/subagent-dispatch.md @@ -0,0 +1,149 @@ +# Subagent dispatch architecture + +The skill-optimizer chain runs most of its generative work in +**limited-context subagents** rather than the operator session. This +file documents the dispatch architecture every chain skill follows: +what the subagent sees, what it doesn't, how the operator session +parameterizes the dispatch, and the no-auto-invocation rule that +keeps chain skills composable. + +If you're a chain skill's SKILL.md, the dispatch step in your +workflow should reference this file rather than re-explaining the +constraints — load it lazily at the dispatch step. + +## Why dispatch at all + +If the operator session does the generative work itself, it inherits +everything that's been in the conversation: the user's framing of +the problem, prior outputs the user has reviewed, the operator's +own thinking-out-loud. That context biases the work toward "what we +just talked about" — the textbook ducktape pattern: a patch that +addresses the symptom the user just described, not the underlying +weakness. + +A subagent dispatched with only the inputs it needs has none of that +contamination. It produces a fresh derivation from the load-bearing +inputs + atomic directives. The operator session then orchestrates +the result, gates against the user, and writes the final artifact. + +## Subagent constraints + +Three rules every reasoning subagent must follow: + +1. **Read upstream artifacts only at their current state.** The + filesystem is the source of truth; do not walk `git log` looking + for prior versions of upstream files. + +2. **For fresh-derivation steps (1, 2, 5, 6-summary, 7, 8, 9): do NOT + read your own canonical file, and do NOT read git history of + it.** Each invocation derives a new artifact from upstream + the + operator's directives, without direct access to prior + derivations of this same artifact. This is the load-bearing + anti-ducktape constraint. + + The directives ARE the channel for lessons learned from prior + iterations. The operator session CAN read the prior output — + that's part of its job between iterations — and distills any + lessons into atomic new requirements. The subagent satisfies + those distilled requirements as fresh constraints, without + seeing the raw prior content. This separation keeps the new + derivation from rationalizing the prior one while still letting + the chain converge. + + **For step 8 and step 9 specifically:** the SKILL CONTENT (the + target being improved) is upstream input, not your own + canonical. Both the optimizer (step 8) and the validator (step + 9) read the **current state of the skill** — which is + `docs/skill-optimizer//improved-skill/` if it exists (the + accumulated state from prior step-9 approvals), else + `.skill-optimizer//vendored-skill/` (step 1 vendored the + source regardless of upstream/local). Neither step modifies the + original. The "own canonical" off-limits to each subagent is + its report (`08-improvement-proposal.md` for the optimizer; + `09-validator-verdict.md` for the validator), not the skill + content itself. + +3. **For maintenance steps (3, 4): DO read your own canonical + tree** (when it exists). Your job on a re-run is to extend or + modify the current state per `${OPERATOR_DIRECTIVES}`, + preserving entries the user has invested in unless a directive + explicitly says to revise. Do NOT walk git history of the tree + — the current state is what matters; prior states are inert. + +## Collect operator directives + +`${OPERATOR_DIRECTIVES}` is a short bulleted list of **atomic new +requirements** that surfaced from prior iterations or the user's +current request. **Never** a context dump of prior artifact content +— a list of specific asks that the new derivation must satisfy on +top of its normal inputs. + +Examples that count as atomic requirements: + +- "user requested coverage for null inputs" +- "focus on the gpt-5 cluster — gemini and claude both passed" + +Examples that do NOT count (these are context dumps — reject the +temptation): + +- The full text of a prior canonical file pasted in for the + subagent to "see what we already had" +- The validator's prior verdict pasted in for the subagent to read + +If the user supplied no new requirements and you're re-running +because an upstream artifact changed, leave directives empty. + +## Dispatch protocol + +The chain uses one dispatch shape, consistently. **You — the +operator session — render the prompt inline, then pass the +rendered text as the Agent tool's `prompt` parameter.** The +subagent should receive a fully self-contained instruction set +that needs no file I/O to begin work. + +Concretely, for every step's dispatch: + +1. **Read the subagent's prompt template** at + `skills//agents/.md`. The chain skill's + workflow tells you which template. +2. **Substitute every `${VAR}` placeholder** with the concrete + value the chain skill's workflow specifies. `${OPERATOR_DIRECTIVES}` + may be empty; everything else is a path or value you already + have in this session. +3. **Pass the fully rendered text** as the Agent tool's `prompt` + parameter. Do NOT pass a short prompt that points at the + template path (e.g., "Your instructions live at X — read that + first"). Path-pointer dispatch costs an extra Read tool call, + leaves `${VAR}` placeholders for the subagent to mentally + substitute (error-prone), and obscures the actual prompt in + transcripts. + +The rendered prompt is what shows up in the conversation log and +what the subagent acts on; if it isn't right, you can see why +immediately. That visibility is why we inline. + +Templated inputs every dispatch carries: + +- `${OPERATOR_DIRECTIVES}` — the bulleted list (may be empty) +- Step-specific inputs (paths to upstream artifacts, output path, + etc.) per that step's SKILL.md + +## Re-run authorization + +**A chain skill never invokes another chain skill on its own.** +When a step finds that its upstream input is unsatisfactory — a +thin functionality report, a too-easy bench, a missing-but-needed +probe, a wrong-target PR-conventions report — it surfaces the +finding to the user and stops. The user (or the auto-pilot driver +applying its default policies) decides whether to re-run an +upstream step, accept the situation, or abandon the run. + +This applies to backward triggers specifically. Forward handoffs +(step N hands off to step N+1) are part of the chain's normal flow +and the handing-off skill emits the handoff message; the user or +auto-pilot acts on it. + +Why this rule exists: chain skills running on user request must +give the user control over what they're paying for. Re-running an +upstream step takes time and tokens; the user authorizes that +explicitly, not the agent. diff --git a/skills/skill-optimizer/references/workbench.md b/skills/shared/workbench.md similarity index 80% rename from skills/skill-optimizer/references/workbench.md rename to skills/shared/workbench.md index e0c5a0f..78c415c 100644 --- a/skills/skill-optimizer/references/workbench.md +++ b/skills/shared/workbench.md @@ -18,8 +18,7 @@ Avoid evals that require running model-produced arbitrary production code outsid ```bash npx tsx src/cli.ts run-case -npx tsx src/cli.ts run-case --model openrouter/google/gemini-2.5-flash -npx tsx src/cli.ts run-case --models openrouter/google/gemini-2.5-flash,openrouter/openai/gpt-5.4 --trials 3 --concurrency 2 +npx tsx src/cli.ts run-case --trials 3 --concurrency 2 npx tsx src/cli.ts run-suite --trials 3 --concurrency 2 ``` @@ -28,16 +27,20 @@ Options: | Command | Option | Meaning | |---------|--------|---------| | `run-case` | `--out ` | Results root, default `/.results` | -| `run-case` | `--model ` | Single OpenRouter model override | -| `run-case` | `--models ` | Comma-separated OpenRouter model refs | -| `run-case` | `--trials ` | Independent trials per model | +| `run-case` | `--trials ` | Independent trials per case | | `run-suite` | `--out ` | Results root, default `/.results` | -| `run-suite` | `--trials ` | Independent trials per case/model | +| `run-suite` | `--trials ` | Independent trials per case × run | | both | `--concurrency ` | Maximum concurrent trial containers | -| both | `--image ` | Docker image, default `skill-optimizer-workbench:local` | +| both | `--image ` | Docker image, default `skill-optimizer-agent:local` | | both | `--keep-workspace` | Preserve successful workspaces too; failures are always preserved | -Only `openrouter/...` model refs are accepted. `run-suite` uses the `models:` array in the suite file. +There is **no `--model` / `--models` override**. The case file's `model:` (and the suite's `runs: [{ agent, model }]` matrix) is the only source of truth — this is intentional, so a suite's matrix is reproducible from its file. + +The image must contain the agent CLIs. Build it once with: + +```bash +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . +``` ## Case Schema @@ -45,6 +48,8 @@ Case files may be `.yml`, `.yaml`, or `.json`. ```yaml name: extract-pdf-facts +agent: pi-acp +model: openrouter/google/gemini-2.5-flash references: ./references task: | Read statement.pdf and write answer.json with the account, quarter, approval code, and risk flags. @@ -64,7 +69,6 @@ mcpServices: command: node args: - calculator-server.mjs -model: openrouter/google/gemini-2.5-flash timeoutSeconds: 600 ``` @@ -73,35 +77,40 @@ Required fields: | Field | Type | Meaning | |-------|------|---------| | `name` | string | Human-readable case name; suite inline cases slug this for result dirs | +| `agent` | string | One of `claude-agent-acp`, `codex-acp`, `gemini`, `opencode`, `pi-acp`. Fails loud on missing/unknown | | `references` | string | Directory copied into `/work` before the agent starts | | `task` | string | User-like task sent to the agent | -| `graders` | array | Non-empty list of `{ name, command }` grader commands | +| `graders` | array | Non-empty list of `{ name, command }` grader commands. **Legacy `check:` and `artifacts:` are rejected.** | Optional fields: | Field | Type | Meaning | |-------|------|---------| +| `model` | string | Per-agent model identifier (see [Per-Agent Model Strings](#per-agent-model-strings)). Defaults to `openrouter/google/gemini-2.5-flash`. Only `pi-acp` enforces the `openrouter/...` prefix; native ACP agents (`claude-agent-acp`, `codex-acp`, `gemini`, `opencode`) pass the model string through to the agent CLI unchanged | | `setup` | string[] | Commands run in `/work` before the agent phase | | `cleanup` | string[] | Commands run after grading | | `env` | string[] | Host environment variable names forwarded into setup, agent, grading, and cleanup containers | | `mcpServers` | object | MCP servers exposed through the agent `mcp` tool | | `mcpServices` | object | Hidden local MCP services started as separate Docker containers | -| `model` | string | Default model for `run-case`; defaults to `openrouter/google/gemini-2.5-flash` | | `timeoutSeconds` | number | Agent timeout; defaults to `600` | All relative paths resolve from the case file directory. ## Suite Schema -Suites may contain inline case objects or paths to external case files. +Suites declare their agent + model matrix explicitly via `runs:`. Inline cases inherit the matrix; external case files declare their own `agent:`. ```yaml name: pdf-workbench-example references: ./references -models: - - openrouter/google/gemini-2.5-flash +runs: + - agent: pi-acp + model: openrouter/google/gemini-2.5-flash + - agent: claude-agent-acp + model: opus[1m] env: - OPENROUTER_API_KEY + - ANTHROPIC_API_KEY timeoutSeconds: 600 setup: - node $CASE/checks/_pdf.mjs write-inputs input @@ -109,6 +118,7 @@ appendSystemPrompt: | Keep task outputs at the top level of /work unless the user asks otherwise. cases: - name: extract-pdf-facts + agent: pi-acp task: | Read statement.pdf and write answer.json with the account, quarter, approval code, and risk flags. graders: @@ -122,8 +132,8 @@ Suite fields: | Field | Required | Meaning | |-------|----------|---------| | `name` | yes | Suite name in aggregate output | -| `models` | yes | OpenRouter model refs for the case/model matrix | -| `cases` | yes | Inline case objects or paths to case files | +| `runs` | yes | Non-empty list of `{ agent, model }` rows. Each row is one (agent, model) combination the suite executes. **Legacy `models:` is rejected.** | +| `cases` | yes | Inline case objects or paths to case files. Inline cases must still declare `agent:` (an inline case's `agent:` field is what its trial uses; the suite's `runs:` matrix is what's iterated) | | `references` | no | Default references dir for inline cases; defaults to `./references` | | `env` | no | Default env allowlist for inline cases | | `setup` | no | Default setup commands for inline cases | @@ -132,11 +142,36 @@ Suite fields: | `mcpServices` | no | Default hidden MCP service containers for inline cases, merged by service name | | `timeoutSeconds` | no | Default agent timeout for inline cases | | `appendSystemPrompt` | no | Extra suite-wide system prompt appended after the workbench prompt | +| `skillUnderTest` | no | `{ slug, hostPath }` — overrides which skill directory is mounted into the trial container. Inline cases inherit this; an inline case may override it with its own `skillUnderTest`. Used by autopilot's empirical re-bench to point the same suite at `improved-skill/` instead of the original vendored source | Inline case fields override suite defaults. External case files are loaded from their own file directory and do not inherit suite defaults. Environment variables listed in `env` are forwarded unchanged. This intentionally supports live integration evals such as authenticated CLI calls, but it also means the agent can read or print those values through shell tools. Use dedicated test accounts, least-privilege credentials, and cleanup routines for live systems. Treat `trace.jsonl`, `result.json`, grader evidence, stdout/stderr, and preserved `workspace/` directories as potentially sensitive if an agent or grader prints or writes secret values. +## Per-Agent Model Strings + +Each agent CLI has its own model-naming convention. The workbench passes the case's `model:` string through to the agent's CLI unchanged (except for `pi-acp`, which enforces an `openrouter/...` prefix). If the model string isn't valid for that agent, the agent CLI fails the trial — the workbench will NOT catch this at suite-load time. **Use the table below when writing `suite.yml` to avoid runtime rejections.** + +| Agent | Format | Valid examples | Notes | +|-------|--------|----------------|-------| +| `claude-agent-acp` | Claude CLI enum | `opus`, `opus[1m]`, `sonnet`, `sonnet[1m]`, `haiku`, `default` | Bracket suffix `[1m]` selects the 1M-context window. Full model IDs like `claude-opus-4-7` are **not** accepted by the CLI. Set via `unstable_setSessionModel` after `newSession` | +| `codex-acp` | OpenAI model IDs | `gpt-5`, `gpt-5-mini`, `gpt-4o`, `o3-mini` | Whatever Codex's `--model` flag accepts | +| `gemini` | Gemini model IDs | `gemini-2.5-pro`, `gemini-2.5-flash`, `gemini-2.0-flash` | Whatever `gemini --model` accepts | +| `opencode` | `provider/model` form | `anthropic/claude-sonnet-4`, `openai/gpt-5`, `google/gemini-2.5-pro` | Must include the provider prefix | +| `pi-acp` | `openrouter/...` only | `openrouter/google/gemini-2.5-flash`, `openrouter/anthropic/claude-sonnet-4`, `openrouter/openai/gpt-5` | Enforced by the case loader; non-`openrouter/` prefixes are rejected at load time | + +**Default if `model:` is omitted:** `openrouter/google/gemini-2.5-flash`. This default only makes sense for `pi-acp`; specify `model:` explicitly when using any other agent. + +**Required env per agent** (declare these in `env:` so they propagate into the trial container): + +| Agent | Env var | Alternative | +|-------|---------|-------------| +| `claude-agent-acp` | `ANTHROPIC_API_KEY` | `~/.claude/.credentials.json` (subscription) | +| `codex-acp` | `OPENAI_API_KEY` | `~/.codex/auth.json` (subscription) | +| `gemini` | `GOOGLE_API_KEY` | `~/.gemini/oauth_creds.json` (subscription) | +| `opencode` | `OPENAI_API_KEY` (or provider-specific key) | — | +| `pi-acp` | `OPENROUTER_API_KEY` | — | + ## MCP Servers `mcpServers` uses mcporter-compatible server entries. During each Docker trial, the workbench writes `/work/mcporter.json` with `imports: []` and exposes an `mcp` command on `PATH`. @@ -507,7 +542,7 @@ Files to inspect: | File | Purpose | |------|---------| | `examples/workbench/README.md` | Top-level example command walkthrough | -| `examples/workbench/pdf/suite.yml` | Inline suite using models, setup, graders, and append prompt | +| `examples/workbench/pdf/suite.yml` | Inline suite using `runs:`, setup, graders, and append prompt | | `examples/workbench/pdf/references/pdf-skill/SKILL.md` | Skill under test copied into `/work` | | `examples/workbench/pdf/checks/*.mjs` | Deterministic graders and setup helpers | | `examples/workbench/pdf/README.md` | Demo walkthrough | @@ -526,8 +561,9 @@ npx tsx src/cli.ts --help node dist/cli.js --help ``` -For runner/Docker changes, rebuild the image: +For runner/Docker changes, rebuild the image (this is also what +`run-case` / `run-suite` expect as the default image): ```bash -docker build -t skill-optimizer-workbench:local -f docker/workbench-runner.Dockerfile . +docker build -t skill-optimizer-agent:local -f docker/skill-optimizer-agent.Dockerfile . ``` diff --git a/skills/shared/workflow.md b/skills/shared/workflow.md new file mode 100644 index 0000000..bf09476 --- /dev/null +++ b/skills/shared/workflow.md @@ -0,0 +1,173 @@ +# skill-optimizer workflow + +Operator-facing reference for the 10-step skill-optimizer chain. +Each step's SKILL.md is self-contained for its own work; this doc +covers cross-skill concerns: the high-level flow, what triggers +re-runs, and the relationship between steps. + +## The chain + +| # | Skill | Kind | Input | Output | +|---|---|---|---|---| +| 1 | `investigate-functionality` | fresh-derivation | source skill (URL or local) | `docs/.../01-functionality.md`, `.skill-optimizer/.../vendored-skill/` | +| 2 | `investigate-submissions` (optional) | fresh-derivation | step 1 (PR-bound) | `docs/.../02-submissions.md` | +| 3 | `design-tests` | maintenance | step 1 | `docs/.../03-test-proposals.md`, `skill-evals/...//spec.yaml` | +| 4 | `write-tests` | maintenance | steps 1+3 | `skill-evals/...///`, `skill-evals/.../suite.yml` | +| 5 | `validate-tests` | fresh-derivation | step 4 + step 1 + skill | `docs/.../05-tests-verdict.md` | +| 6 | `run-bench` | fresh-derivation (summary) | step 5 (must approve) + skill | `.skill-optimizer/.../bench-results//`, `docs/.../06-bench-summary.md` | +| 7 | `analyze` | fresh-derivation | step 6 + skill | `docs/.../07-analysis.md` | +| 8 | `improve` | fresh-derivation | step 7 + skill (+ step 2 if PR-bound) | `docs/.../08-improvement-proposal.md` | +| 9 | `validate` | fresh-derivation | step 8 + skill (+ step 2 if PR-bound) | `docs/.../09-validator-verdict.md`, `docs/.../improved-skill/` (on approve) | +| 10 | `autopilot` | chain driver | source skill + startup-interview flags | `.skill-optimizer/.../autopilot-/journal.md`, `docs/.../autopilot-summary-.md` | + +(Paths abbreviated; full state layout below.) + +The PR-or-not decision is made ONCE at step 1; subsequent steps +know from `01-functionality.md`'s `pr_submission_intent` field +whether step 2 will run. No late prompts. + +Two independent validators in the chain: step 5 (validate-tests) +checks that probes fairly test their functionality before +measurement; step 9 (validate) checks that improvements address +the named weakness in a principled way. Both are anti-ducktape +gates — step 6 refuses to fire if step 5 didn't approve all +probes, and step 9 won't materialize `improved-skill/` unless it +approves the proposal. + +## Re-run triggers (when each step should be re-invoked) + +A chain skill never auto-invokes another chain skill (per +[`subagent-dispatch.md`](./subagent-dispatch.md) re-run +authorization). The operator (or auto-pilot at step 10) decides +when to re-run each step. Common triggers: + +| Step | Re-run when... | +|---|---| +| 1 | source URL changed; PR-intent changed; user wants fresh research with new directives | +| 2 | upstream updated `CONTRIBUTING.md`/license/CLA; PR conventions visibly shifted; step 1 changed | +| 3 | user wants different coverage; step 4/5/6/7 surfaced a coverage gap; step 1 changed | +| 4 | step 3's `skill-evals//` tree changed; a probe's smoke check failed; step 5 flagged probes for revision; step 7 showed probes systematically too easy/hard | +| 5 | step 4 produced new or revised probes; user disagrees with prior test verdict | +| 6 | step 4's `skill-evals//` changed (and step 5 re-approved); user wants fresh trial data (flakiness, model list changed); step 7 wants more trials | +| 7 | step 6 produced new results; user disagrees with prior analysis; step 8 was unable to address a named weakness | +| 8 | step 7 produced a new analysis; step 9 returned `needs-revision`/`reject` with a distillable rationale; user wants a different approach | +| 9 | step 8 produced a new proposal; user disagrees with prior verdict; step 2 was updated and prior external check is stale | +| 10 | self-iterable; picks up at whatever step is stale per git-mtime | + +Each skill checks **direct upstream only** for staleness — if step +1 went stale but step 3 wasn't re-run (user judged it still +valid), step 4 sees step 3 as current. Skipping a step's re-run +is an explicit operator judgment. + +## Backward triggers + +When a step surfaces a problem with an earlier step. These do NOT +auto-fire; they're surfaced to the user. + +| At step | Surfaced finding | Resolution path | +|---|---|---| +| 3 | proposal is thin (subagent found few responsibilities) | re-run step 1 with directives, OR accept | +| 4 | test-writer reports BLOCKED on a probe | re-run step 3 to refine the functionality spec | +| 5 | any probe is `needs-revision`/`reject` | distill verdict into directives, re-run step 4 for affected probes, re-run step 5 | +| 6 | bench all-pass (probes too easy) | re-run step 3 with "make probes harder" directive | +| 7 | analyzer's `has_structural_weakness: false`, user disagrees | re-run step 7 with directive pointing at missed cluster | +| 8 | optimizer reports BLOCKED (weakness too abstract) | re-run step 7 to reformulate weakness | +| 9 | `needs-revision` | distill validator's rationale, re-run step 8, re-run step 9 | +| 9 | `reject` | re-run step 7 with reframed weakness, OR accept that weakness isn't addressable | + +## State layout + +Chain output splits across three top-level locations, each with +its own tracking discipline: + +```text +docs/skill-optimizer// # TRACKED — reports + deliverable (PR-reviewable) + 01-functionality.md # step 1 + 02-submissions.md # step 2 — only if PR-bound + 03-test-proposals.md # step 3 audit report + 05-tests-verdict.md # step 5 verdict (per-probe + aggregate) + 06-bench-summary.md # step 6 aggregate summary + 07-analysis.md # step 7 + 08-improvement-proposal.md # step 8 + 09-validator-verdict.md # step 9 + improved-skill/ # step 9 materializes on approve — the deliverable + autopilot-summary-.md # step 10 per run + +skill-evals// # TRACKED — eval suites + probes (reusable across runs) + / + spec.yaml # step 3 writes + / # step 4 writes one per probe + spec.yaml + workspace/ + grader.mjs + smoke/{good,bad,empty}/ + checks/smoke.mjs + suite.yml # step 4 generates + +.skill-optimizer// # GITIGNORED — heavy ephemeral artifacts + vendored-skill/ # step 1 vendors source + all externally-referenced files (single flat dir) + bench-results// # step 6 raw output (suite-result.json, trace.jsonl, findings.txt) + autopilot-/ # step 10 per-run scratch + journal.md # appended-to during the run +``` + +`vendored-skill/` is a **single self-contained dir**: step 1's +researcher vendors the source plus any external URLs the source +references at runtime, so the test bench reproduces the install +behavior without live network. The specific file the chain +**optimizes and submits upstream** is named explicitly by +`01-functionality.md`'s `optimization_target` frontmatter field — +either the source `SKILL.md` (non-wrapper) or an underlying file +the source points at (wrapper). Step 2 reads +`optimization_target_source` (the upstream URL of that file) to +decide which repo to research for PR conventions; this may differ +from `skill_source`. + +`` is `--` for upstream skills, or +`` for local skills. The same `` is reused +across all three locations so a slug's full state can be located +by name. + +**Why the split:** + +- **`docs/skill-optimizer//`** holds what a human reviews + during PR: the markdown reports, plus the improved-skill the + PR is about. Tracked so reviewers see history. +- **`skill-evals//`** holds reusable test machinery + (workbench probes). Tracked because tests are valuable on + their own — they run against future skill versions, get reused + across runs, get reviewed for fairness. +- **`.skill-optimizer//`** holds heavy or trivially + re-derivable artifacts. Gitignored: + - `vendored-skill/` can be re-fetched from `skill_source` any + time; tracking it doubles repo size for no review value. + - `bench-results//` accumulates `trace.jsonl` and + `findings.txt` per trial — tens of MB per run. The summary + at `docs/skill-optimizer//06-bench-summary.md` is the + durable record. + +Filesystem IS the state; history is git (no `version:` fields or +`archive/` directories per +[`frontmatter-discipline.md`](./frontmatter-discipline.md)). +`.skill-optimizer//` is the one location where state is +intentionally NOT versioned — re-runs overwrite freely. + +## Shared docs + +The chain skills load these on-demand at the workflow steps that +need them: + +- [`iteration-protocol.md`](./iteration-protocol.md) — + iteration mechanics (staleness, step kinds, + destructive-edit checkpoints, cascading, bootstrapping) +- [`subagent-dispatch.md`](./subagent-dispatch.md) — + subagent constraints, operator directives, templated dispatch + inputs, no-auto-invocation rule +- [`frontmatter-discipline.md`](./frontmatter-discipline.md) — + runtime facts vs. history rule +- [`workbench.md`](./workbench.md) — workbench schema (probe + layout, grader contract, smoke-check format); referenced by step 4 +- [`skill-design-philosophy.md`](./skill-design-philosophy.md) — + cross-vendor synthesis of skill-quality principles, anti-patterns, + and the principled-vs-ducktape rubric; referenced by analyzer + (step 7), optimizer (step 8), validator (step 9) diff --git a/skills/skill-optimizer/SKILL.md b/skills/skill-optimizer/SKILL.md deleted file mode 100644 index b376e89..0000000 --- a/skills/skill-optimizer/SKILL.md +++ /dev/null @@ -1,210 +0,0 @@ ---- -name: skill-optimizer -description: Use when creating, running, debugging, or documenting skill-optimizer workbench evals; working with agent skill cases, suites, graders, traces, Docker workspaces, OpenRouter model matrices, or the skill-optimizer SDK/CLI. ---- - -# skill-optimizer - -`skill-optimizer` is an eval workbench for agent skills. It runs a model in an isolated Docker `/work` directory, provides skills/references as normal workspace files, captures an agent trace, and grades deterministic local outcomes. - -Use this skill as the source of truth for authoring eval suites in this repo. Detailed schema and patterns are in `references/workbench.md`. - -## Core Model - -- A case is one user-like task plus one or more deterministic graders. -- A suite is a set of cases and OpenRouter models to run as a matrix. -- `references` are copied into `/work` before the agent starts; this is where eval skills live. -- The agent phase sees `/work` only. It cannot see `/case`, `/results`, graders, hidden answers, or hidden metadata. -- Cases can define `mcpServers`; these are exposed through a workbench `mcp` command during the agent phase. -- Graders run after the agent with `/case`, `/work`, and `/results` mounted. -- `trace.jsonl` is the debugging source for what the agent saw, said, and did. - -## Commands - -| Goal | Command | -|------|---------| -| Install deps | `npm install` | -| Build CLI | `npm run build` | -| Run one case | `npx tsx src/cli.ts run-case ` | -| Run one case across models | `npx tsx src/cli.ts run-case --models openrouter/google/gemini-2.5-flash,openrouter/openai/gpt-5.4` | -| Run a suite | `npx tsx src/cli.ts run-suite ` | -| CLI help | `npx tsx src/cli.ts --help` | - -Rules: - -- Use only `openrouter/...` model refs. -- `OPENROUTER_API_KEY` is required for real model runs. -- `run-suite` uses `models:` from `suite.yml`; it has no model override flag. -- `run-case` can use its case `model:` or `--model` / `--models`. -- Docker image default is `skill-optimizer-workbench:local`. - -## Install This Skill - -This repository ships one canonical skill at `skills/skill-optimizer/SKILL.md` plus plugin metadata for Claude Code, OpenCode, Codex, Cursor, and Gemini. - -Install the skill for common agents with: - -```bash -npx skills add fastxyz/skill-optimizer --skill skill-optimizer -a claude-code -a opencode -a codex -a cursor -``` - -Plugin entrypoints: - -- Claude Code: `.claude-plugin/plugin.json` and `.claude-plugin/marketplace.json` -- OpenCode: `.opencode/plugins/skill-optimizer.js` -- Codex: `.codex-plugin/plugin.json` -- Cursor: `.cursor-plugin/plugin.json` -- Gemini: `gemini-extension.json` and `GEMINI.md` - -## Authoring Workflow - -1. Create `suite.yml` with `models`, shared defaults, and inline cases or case paths. -2. Put the skill/reference material under `references/`; it will be copied into `/work`. -3. Write natural user tasks. Do not mention graders, hidden answers, `/case`, or eval internals. -4. Put setup helpers and grader helpers under `checks/`; put fake CLIs or command shims under `bin/` when the agent should call them. -5. Add one or more `graders` per case. Prefer small deterministic graders over one broad grader. -6. Run `run-suite --trials ` and inspect `suite-result.json`, failing `result.json`, `summary.json`, and `trace.jsonl`. - -Variables listed in `env` are forwarded unchanged into setup, agent, grading, and cleanup containers. For live integration evals, use dedicated test accounts and scoped credentials because the agent can access those values through shell tools. Treat `trace.jsonl`, `result.json`, grader evidence, stdout/stderr, and preserved `workspace/` directories as potentially sensitive if an agent or grader prints or writes secret values. - -Use `mcpServers` when the task should interact with MCP tools. For local servers whose source should stay hidden from the agent, put server files under the case `mcp/` support directory and define `mcpServices`; Docker starts those as separate service containers and the agent only sees their HTTP MCP URL. Direct stdio `mcpServers.command` entries run inside the agent container and are only appropriate when the server implementation is intentionally agent-visible. Remote HTTP/SSE servers must be reachable from Docker. The workbench generates `/work/mcporter.json` with `imports: []`, so host/user MCP configs are not imported. OAuth/browser auth is not supported; use env/header credentials listed in `env`. - -Prefer the real CLI/API/service when you do not know its internal behavior well enough to mock it faithfully. Mock only when you are sure the mock matches the real command surface, validation, outputs, and failure modes; otherwise the eval will measure the mock, not the skill. For command skills, include cases for the basic command, important flags/options, a no-tool-needed control, and unsafe-instruction resistance. - -## Minimal Suite - -```yaml -name: pdf-skill-eval -references: ./references -models: - - openrouter/google/gemini-2.5-flash -env: - - OPENROUTER_API_KEY -timeoutSeconds: 600 -setup: - - node $CASE/checks/create-inputs.mjs -appendSystemPrompt: | - Keep task outputs at the top level of /work unless the user asks otherwise. -cases: - - name: extract-pdf-facts - task: | - Read statement.pdf and write answer.json with the account, quarter, approval code, and risk flags. - graders: - - name: answer-json - command: node $CASE/checks/extract-pdf-facts.mjs -``` - -## Directory Layout - -```text -my-eval/ - suite.yml - references/ - my-skill/SKILL.md - checks/ - create-inputs.mjs - extract-pdf-facts.mjs - bin/ - fake-cli - workspace/ - starter-app/ -``` - -Support directories are optional. `checks/` is mounted read-only at `/case/checks` for setup/grading. `bin/` is copied into `/work/bin` for the agent and is also available as `/case/bin` during setup/grading. `workspace/` is copied into `/work` after `references/`. - -## Grader Contract - -Graders are shell commands. They run with: - -- `$CASE`: read-only case directory mounted at `/case` -- `$WORK`: mutable workspace the agent used -- `$RESULTS`: result directory containing `trace.jsonl` - -Preferred grader output: - -```json -{ "pass": true, "score": 1, "evidence": ["answer matched"] } -``` - -If no JSON object is printed, exit code `0` passes and non-zero fails. Keep graders deterministic and local; do not use an LLM judge unless the eval explicitly requires one. - -Graders are the acceptance contract. They should evaluate evidence in `/work`, generated artifacts, `answer.json`, `trace.jsonl`, and any relevant result-state files under `$RESULTS`. - -## Outputs - -```text -.results// - suite-result.json # run-suite aggregate - run-result.json # run-case matrix aggregate - trials/----001/ - trace.jsonl # agent messages and tool calls - result.json # pass, score, evidence, graders, metrics - summary.json # final text, failed graders, commands - workspace/ # failures or --keep-workspace -``` - -Use `trace.jsonl` to debug failures and to grade negative behavior, such as whether a task read an irrelevant skill file. - -## Optimization Loop - -After a run, inspect failing `result.json`, `summary.json`, `trace.jsonl`, and preserved `workspace/` evidence. Classify each failure before changing anything: unclear skill guidance, missing reference material, brittle grader, unrealistic input data, task ambiguity, or product/code bug. Update the target skill, references, inputs, graders, or code according to that diagnosis, then re-run the same case or suite to verify the change. Repeat until the grader evidence shows the intended behavior across the target models/trials. - -For live CLI/API evals, use scoped test credentials and avoid printing secrets. Grade durable evidence: command traces, arguments, generated files, response summaries, and safety behavior. Keep service-specific setup facts in the suite prompt or setup commands, not in the portable skill under test. - -## Programmatic SDK - -The package exports workbench APIs from `skill-optimizer` after build: - -```ts -import { - loadWorkbenchCase, - loadWorkbenchSuite, - runWorkbenchCase, - runWorkbenchSuite, - runGraderCommands, - parseModelList, -} from 'skill-optimizer'; -``` - -The CLI is the stable path for normal eval runs. Use SDK functions for tests, wrappers, and internal automation. - -## Examples - -Tracked demos live in `examples/` (the same repo path users may refer to as `@examples/`). Read these alongside the skill docs when building or debugging evals: - -| Path | Why It Matters | -|------|----------------| -| `examples/workbench/README.md` | Short command walkthrough for demos | -| `examples/workbench/pdf/README.md` | Explains the PDF demo cases and expected outputs | -| `examples/workbench/pdf/suite.yml` | Concrete suite using models, setup, env, graders, and append prompt | -| `examples/workbench/pdf/references/pdf-skill/SKILL.md` | Example skill copied into `/work` for the agent | -| `examples/workbench/pdf/checks/*.mjs` | Deterministic grader and setup helper patterns | -| `examples/workbench/mcp/suite.yml` | Hidden-service MCP calculator example | -| `examples/workbench/mcp/mcp/calculator-server.mjs` | Example MCP server with add/subtract/multiply/divide tools | - -```bash -npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials 1 -npx tsx src/cli.ts run-suite examples/workbench/mcp/suite.yml --trials 1 -``` - -The PDF demo covers setup, suite models, positive output grading, and trace-based negative grading. - -## Development Checks - -After code or docs that affect behavior: - -```bash -npm run typecheck -npm test -npm run build -npx tsx src/cli.ts --help -node dist/cli.js --help -``` - -After Dockerfile/container-runner changes: - -```bash -docker build -t skill-optimizer-workbench:local -f docker/workbench-runner.Dockerfile . -``` - -Do not commit `.skill-eval/`; it is local ignored eval data. diff --git a/skills/validate-tests/SKILL.md b/skills/validate-tests/SKILL.md new file mode 100644 index 0000000..65dc558 --- /dev/null +++ b/skills/validate-tests/SKILL.md @@ -0,0 +1,175 @@ +--- +name: validate-tests +description: Use when the user wants to validate the probes built by `skill-optimizer:write-tests` before running the bench — phrases like "validate the tests", "check the graders", "are these probes fair", "review the test suite". Triggers after `skill-optimizer:write-tests` has populated `skill-evals////` folders and before `skill-optimizer:run-bench`. Use even when the user doesn't explicitly say "validate" — any phrasing about checking that the probes actually probe what they claim should trigger this. +--- + +# validate-tests + +Step 5 of the skill-optimizer chain. **Fresh-derivation step.** +Takes the probes built by step 4, dispatches a test-validator +subagent per probe (in parallel) to independently check whether +each probe fairly tests its parent functionality and whether its +grader is sound, and writes +`docs/skill-optimizer//05-tests-verdict.md` — the aggregate +verdict step 6 (run-bench) checks before proceeding. + +This step exists because the smoke check at step 4 only verifies +**syntactic** consistency (the grader correctly classifies the +GOOD/BAD/EMPTY fixtures the test-writer also wrote). It can't catch +**semantic** problems: workspace doesn't actually exercise the +functionality, grader is too strict/loose beyond the smoke +fixtures, false-positive probes that the skill could pass without +doing the right thing. Grader bugs are load-bearing — they +propagate to misleading bench results, misleading analyses, and +ducktape-shaped improvements. An independent validator is the +chain's anti-ducktape gate for the test layer, parallel to step 9 +for the optimizer layer. + +**Single-shot per invocation.** No in-step revision loop. If any +probe is `needs-revision` or `reject`, the operator (or auto-pilot +at step 10) distills the validator's rationale into a directive +and re-invokes step 4 for the affected probes, then re-invokes +this step. + +## What you produce + +One artifact at `docs/skill-optimizer//`: + +**`05-tests-verdict.md`** — aggregate verdict across all probes +plus per-probe sections. Frontmatter (runtime-relevant facts only, +per [`frontmatter-discipline.md`](../shared/frontmatter-discipline.md)): + +```yaml +--- +all_probes_approved: true | false +probe_count: +needs_revision_count: +reject_count: +--- +``` + +**`all_probes_approved`** is load-bearing for step 6 — if `false`, +step 6 refuses to fire ("can't measure against bad probes"). Forcing +`true` when probes had real issues is the test-layer ducktape failure +mode this step exists to prevent. + +Body has per-probe sections grouped by functionality. Each probe +entry includes: verdict (approve / needs-revision / reject), +rationale (covering workspace fairness, grader correctness, smoke +fixture distinguishing power, fairness across reasonable agent +outputs), and — for needs-revision/reject — a concrete suggestion +the operator can distill into a directive for step 4. + +Body template and the validator's reasoning protocol live at +[`agents/test-validator.md`](./agents/test-validator.md). + +## Workflow + +### (a) Confirm prerequisites + +`skill-evals//` must exist with at least one +`skill-evals////` folder containing `spec.yaml`, +`workspace/`, `grader.mjs`, `smoke/`. If any picked functionality +has zero built probes, surface as a step-4 problem and tell the +user to run step 4 first. + +`01-functionality.md` must also exist (the validator reads it for +context on what the skill is supposed to do). + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Each invocation overwrites the canonical verdict +from current probe state plus directives. Collect +`${OPERATOR_DIRECTIVES}` per the protocol — examples: "be stricter +on grader fairness", "the prior verdict missed that probe X +requires a specific JSON shape the skill never asks for". + +### (c) Dispatch test-validator subagents (parallel, one per probe) + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT judge probes yourself in this +session** — dispatch test-validator subagents via the `Agent` tool, +in parallel (emit all dispatches in a single message). For each +probe, render the prompt template at +[`agents/test-validator.md`](./agents/test-validator.md) +inline by substituting that probe's `${PROBE_NAME}`, `${PROBE_DIR}`, +`${FUNCTIONALITY_SPEC_PATH}`, `${FUNCTIONALITY_PATH}`, +`${SKILL_SOURCE_PATH}`, `${VERDICT_OUTPUT_PATH}`, and +`${OPERATOR_DIRECTIVES}`, then pass the rendered text as that +Agent dispatch's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +Each test-validator sees: its single probe's full contents +(spec.yaml, workspace files, grader.mjs, smoke fixtures); the +parent functionality's spec.yaml; `01-functionality.md`; the skill +source content; `${OPERATOR_DIRECTIVES}` for its probe. + +Each test-validator does NOT see: other probes (independence — a +validator that sees siblings would gravitate toward +consistency-with-siblings rather than judging each probe on its +own); its own prior verdict or git history; the test-writer's +reasoning (only the probe artifacts, not how they were arrived +at); `07-analysis.md`, `08-improvement-proposal.md`, or any +downstream output. + +**Why this matters:** the test-writer at step 4 has built both the +fixture AND the grader AND the smoke fixtures — self-validation +only catches syntactic problems. An independent validator that +reads the same probe from scratch (against the functionality spec) +catches semantic problems: probes that pass without exercising the +responsibility, graders unfair to reasonable agent outputs, smoke +fixtures that coincidentally match rather than truly distinguish. +The validator must judge "would I, as an external reviewer, accept +this as a fair test of the parent functionality?". + +Each test-validator returns: probe name, verdict, one-line +rationale, and the suggestion-for-revision (if not approved). + +### (d) Aggregate verdicts + write the report + +Collect per-probe verdicts. Compute aggregate: + +- `all_probes_approved = true` if every probe is `approve` +- `needs_revision_count` and `reject_count` from per-probe verdicts + +Write `05-tests-verdict.md` with the frontmatter above and a body +grouped by functionality, with per-probe sections containing +verdict, rationale, and suggestion-for-revision. + +If `all_probes_approved: true` but the body shows any +`needs-revision`/`reject` (or vice versa), the verdict is +internally inconsistent — surface to the user; don't try to +reconcile yourself. + +### (e) Hand off + +Two messages depending on `all_probes_approved`: + +- **`true`:** "All probes approved. Next, invoke + `skill-optimizer:run-bench` to measure baseline performance." +- **`false`:** "`` probes need revision, `` rejected. The + validator's per-probe rationale is in `05-tests-verdict.md`. + Distill the concerns into directives and re-invoke + `skill-optimizer:write-tests` for the affected probes (per the + per-probe rebuild flow at step 4), then re-invoke this step. + Don't proceed to bench against bad probes." + +Don't auto-invoke step 4 or step 6. + +## Edge cases + +- **No probes built (`skill-evals//` empty)** — caught at (a). Tell the + user to run step 4 first. +- **Validator's verdict contradicts the smoke check** (e.g., + validator rejects a probe whose smoke check passed) — that's + the expected case! Smoke is syntactic; validator is semantic. + The validator wins; proceed with `needs-revision`/`reject`. +- **Validator approves a probe whose smoke check failed** — that + shouldn't happen (smoke check at step 4 surfaces failures + immediately), but if it does, treat as inconsistent and + re-invoke step 4 to fix the smoke check first. +- **Operator directives reference a specific section of the prior + verdict** — context-dump masquerading as a directive. Translate + to an atomic new requirement before passing. diff --git a/skills/validate-tests/agents/test-validator.md b/skills/validate-tests/agents/test-validator.md new file mode 100644 index 0000000..2684934 --- /dev/null +++ b/skills/validate-tests/agents/test-validator.md @@ -0,0 +1,187 @@ +# Test-validator subagent + +You are dispatched by `skill-optimizer:validate-tests` to +**independently judge** whether ONE probe (built by step 4) fairly +tests the responsibility its parent functionality claims, and +whether its grader is sound. Many test-validator subagents run in +parallel (one per probe); each is responsible for verdict on its +own probe. + +This is the chain's anti-ducktape gate for the test layer. The +smoke check at step 4 only verifies syntactic consistency (the +test-writer wrote both the fixture AND the smoke fixtures — so the +smoke check is self-validation). You catch what the smoke check +can't: workspaces that don't exercise the responsibility, graders +unfair to reasonable agent outputs, false-positive probes that +pass without doing the right thing, smoke fixtures that +coincidentally match rather than truly distinguish. + +## Inputs (templated by the operator session) + +- `${PROBE_NAME}` — slug of the probe under judgment +- `${PROBE_DIR}` — `skill-evals////` (full + contents: spec.yaml, workspace/, grader.mjs, smoke/) +- `${FUNCTIONALITY_SPEC_PATH}` — the parent functionality's + spec.yaml (the responsibility this probe is supposed to test) +- `${FUNCTIONALITY_PATH}` — `01-functionality.md` +- `${SKILL_SOURCE_PATH}` — `.skill-optimizer//vendored-skill/` (step 1 vendored the + source regardless of upstream/local) — needed to judge whether + the probe exercises what the skill actually instructs the agent + to do +- `${VERDICT_OUTPUT_PATH}` — where to write your verdict (typically + one per-probe verdict file the operator aggregates, OR a single + return value the operator collects across parallel dispatches) +- `${OPERATOR_DIRECTIVES}` — atomic new requirements (empty unless + this is a re-run with sharper guidance) + +## What you see + +- `${PROBE_DIR}` — full probe contents (spec.yaml, workspace/, + grader.mjs, smoke/{good,bad,empty}/findings.txt, + checks/smoke.mjs) +- `${FUNCTIONALITY_SPEC_PATH}` (parent functionality) +- `${FUNCTIONALITY_PATH}` (briefing document) +- The skill source content +- `${OPERATOR_DIRECTIVES}` for your probe + +## What you do NOT see + +- **Other probes** — independence per-probe. A validator seeing + siblings would gravitate toward consistency-with-siblings rather + than judging each probe on its own merits. +- The test-writer's reasoning trace from step 4 — only the + artifacts they produced. Independence means judging the probe + as if you were a fresh external reviewer. +- Your own prior verdict on this probe or git history of it (no + consistency-with-self bias across re-runs) +- `07-analysis.md`, `08-improvement-proposal.md`, + `09-validator-verdict.md`, or any downstream output + +## Output: a per-probe verdict + +Verdict shape (returned to the operator session for aggregation +into `05-tests-verdict.md`): + +```yaml +probe: +parent_functionality: +verdict: approve | needs-revision | reject +rationale: > + +suggestion: > + +``` + +## Four judgment dimensions + +For each probe, judge along these four axes. The verdict is the +weakest of the four — approve only if all four hold. + +### 1. Workspace fairness + +Does `workspace/` actually exercise the responsibility named in +the parent functionality's spec? If the functionality is "refuses +malformed input", the workspace should contain malformed input. A +common failure mode is a workspace that's adjacent to the +responsibility but doesn't actually invoke it — the agent could +pass without exercising the skill's claim. + +Reasonable cross-check: read the parent functionality's +`why_test` field. Does the workspace surface the failure mode +`why_test` warns about? If not, the probe doesn't test what it +claims to. + +### 2. Grader correctness + +Does `grader.mjs` correctly verify that the agent did the right +thing, NOT that the agent produced a specific output format? +Common failure modes: + +- **Too strict:** grader requires an exact string/JSON shape the + skill never asks for. A correctly-behaving agent might phrase + the right answer differently and fail unfairly. +- **Too loose:** grader passes on partial work that misses the + responsibility. False positives let the skill ship without the + responsibility actually working. +- **Wrong target:** grader checks something adjacent to the + responsibility instead of the responsibility itself. + +### 3. Smoke fixture distinguishing power + +The smoke fixtures (`good/`, `bad/`, `empty/`) should TRULY +distinguish — not coincidentally match. A common failure mode: +GOOD passes because of a token the grader matches on, but the +token is incidental to the responsibility (e.g., the grader +matches "passed" anywhere in findings.txt; GOOD says "test +passed", but a hallucinating agent could output "passed +verification" and also pass). + +For each smoke fixture, check: would a substantively different +output also pass/fail in the same way, or is this the only output +that produces this verdict? Strong smoke fixtures have only one +clean reason to pass/fail; weak ones have multiple coincidental +paths. + +### 4. Fairness across reasonable agent outputs + +Imagine 3-4 plausible ways a correctly-behaving agent might +respond to the workspace. Would the grader pass all of them, or +only one specific shape? If only one shape, the grader is +implicitly demanding format conformance the skill doesn't teach +the agent — that's unfair. + +## Reasoning protocol + +1. **Read the parent functionality spec.** What responsibility? + What failure mode does `why_test` warn about? +2. **Read the probe's spec.yaml.** What does the test-writer + claim this probe tests? +3. **Read the workspace files.** Do they realistically surface + the failure mode? +4. **Read the grader.** Trace through what triggers pass vs fail. + Check for the three grader failure modes (too strict, too + loose, wrong target). +5. **Read the smoke fixtures.** Walk each through the grader + mentally; verify pass/fail outcomes match the labels AND for + the right reasons. +6. **Imagine alternative agent outputs** that should pass / should + fail. Verify the grader handles them as expected. +7. **Cross-check against the skill source.** Does the + responsibility-as-the-skill-describes-it match what the probe + tests? If the skill says one thing and the probe tests + something subtly different, that's a misalignment. +8. **Choose verdict.** approve / needs-revision / reject — the + weakest of the four dimensions. For needs-revision/reject, + write a concrete suggestion the operator can turn into a + directive for step 4 (e.g., "loosen the grader to accept + either 'rejected' or 'refused' as the refusal verb", not + "the grader is bad"). + +## Verdict definitions + +- **approve** — all four dimensions hold; probe is sound for + measurement at step 6. +- **needs-revision** — one or more dimensions has a fixable issue + (specific grader strictness, missing edge case in workspace, + smoke fixture that coincidentally matches). The probe can be + salvaged by a targeted rewrite. +- **reject** — fundamental misalignment with the parent + functionality. The probe tests the wrong thing, or the + functionality as named can't be cleanly tested. Rejection + surfaces to the operator who decides whether to re-frame at + step 3 or accept the gap. + +## Return summary + +After judging, return a brief structured result: + +- Probe name +- Verdict (approve / needs-revision / reject) +- One-line rationale (the weakest dimension or the dominant + concern) +- Suggestion (for non-approve only) + +Keep it under 100 words. The operator session aggregates these +across all parallel test-validators into `05-tests-verdict.md`. diff --git a/skills/validate/SKILL.md b/skills/validate/SKILL.md new file mode 100644 index 0000000..30f69c7 --- /dev/null +++ b/skills/validate/SKILL.md @@ -0,0 +1,174 @@ +--- +name: validate +description: Use when the user wants to validate an improvement proposal from `skill-optimizer:improve` — phrases like "validate the improvement", "check the proposal", "is the optimizer's change sound", "review the fix". Triggers after `skill-optimizer:improve` has produced `08-improvement-proposal.md`. Use even when the user doesn't explicitly say "validate" — any phrasing about checking whether the proposed change is sound and conformant should trigger this. +--- + +# validate + +Step 9 of the skill-optimizer chain. **Fresh-derivation step.** Takes +the proposal from step 8, dispatches a validator subagent to +independently check whether the proposed change is sound (internal +consistency) and conformant (external PR conventions if PR-bound), +then — on `verdict: approve` — materializes the improved skill at +`docs/skill-optimizer//improved-skill/`. Writes `09-validator-verdict.md` regardless of +verdict. + +**Single-shot per invocation.** No in-step revision loop. If the +verdict is `needs-revision`, the operator (or auto-pilot at step 10) +re-invokes step 8 with the validator's rationale distilled into a +directive, then re-invokes this step. + +## What you produce + +One or two artifacts at `docs/skill-optimizer//`: + +1. **`09-validator-verdict.md`** — the validator's independent + judgment. Frontmatter (runtime-relevant facts only, per + [`frontmatter-discipline.md`](../shared/frontmatter-discipline.md)): + + ```yaml + --- + verdict: approve | needs-revision | reject + addresses_weaknesses: + - + - ... + --- + ``` + + Body covers the internal consistency check (does the change + make sense for the named weakness? additive vs. destructive? + general vs. ducktape?) and — if `02-submissions.md` exists — + the external consistency check (conformant to upstream PR + rules?). External check is forward-looking — verifies the + improved skill COULD be turned into a valid PR, even though + step 9 doesn't produce one. Body template at + [`agents/validator.md`](./agents/validator.md). + +2. **`docs/skill-optimizer//improved-skill/`** — improved skill content, only + materialized when `verdict: approve`. Mirrors the + `vendored-skill/` directory structure exactly, with the + optimizer's diff applied **only to the + `optimization_target` file** (from + `01-functionality.md`'s frontmatter). All other files in the + tree are copied verbatim from `vendored-skill/` (or the prior + `improved-skill/` if accumulated). **The original is never + modified**: `.skill-optimizer//vendored-skill/` stays + frozen, and the user's original local file (if any) is + untouched. Git tracks `docs/skill-optimizer//improved-skill/` + history across iterations. + +## Workflow + +### (a) Confirm prerequisites + +Three checks: + +1. `08-improvement-proposal.md` must exist with valid frontmatter. + If not, tell the user to run `skill-optimizer:improve` + first. +2. Current skill state must be readable: `docs/skill-optimizer//improved-skill/` if it + exists (prior accumulated state), else `.skill-optimizer//vendored-skill/`. The + validator needs this as "skill BEFORE". If `.skill-optimizer//vendored-skill/` + was re-vendored between step 8 and step 9 (e.g., the operator + re-ran step 1 mid-chain), surface to the user — the BEFORE must + match what the optimizer read. The fix is re-invoking step 8 + against the new state. +3. `01-functionality.md` must exist with `optimization_target` + set. The validator compares BEFORE and AFTER **at that + specific file path** — not the whole skill tree. + +If `pr_submission_intent: true`, `02-submissions.md` should +exist; if missing, the external check can't run. Ask whether to +run step 2 first or proceed with internal consistency only (the +verdict body will note the omission). + +### (b) Handle iteration + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply it. Each invocation overwrites the canonical verdict. +Collect `${OPERATOR_DIRECTIVES}` — examples: "be stricter on +additive-vs-destructive", "the external check missed the +frontmatter `version:` field — look for it specifically". + +If a directive references a specific section of the prior verdict, +that's a context-dump masquerading as a directive. Translate to +an atomic new requirement before passing. + +### (c) Dispatch the validator subagent + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT judge the proposal yourself in this +session** — dispatch the subagent via the `Agent` tool. Render the +prompt template at +[`agents/validator.md`](./agents/validator.md) +inline by substituting `${PROPOSAL_PATH}`, `${SKILL_BEFORE_PATH}` +(current state — `docs/skill-optimizer//improved-skill/` if it exists, else source), +`${SKILL_AFTER_PATH}` (proposed result — apply the diff to a temp +copy; do NOT touch the canonical `docs/skill-optimizer//improved-skill/` yet), +`${FUNCTIONALITY_PATH}`, `${SUBMISSIONS_PATH}` (if PR-bound), +`${VERDICT_OUTPUT_PATH}`, and `${OPERATOR_DIRECTIVES}`, then pass +the rendered text as the Agent tool's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +The validator sees: skill BEFORE; skill AFTER (temporary +materialization); `08-improvement-proposal.md`; +`01-functionality.md`; `02-submissions.md` if PR-bound; +`${OPERATOR_DIRECTIVES}`. + +The validator does NOT see: raw failed trials, `findings.txt`, +`trace.jsonl`; test inputs (`skill-evals////workspace/`); the +optimizer's reasoning trace (only the proposal artifact); its own +prior `09-validator-verdict.md` or git history; `07-analysis.md` +directly. + +**Why this matters:** independence is the validator's load-bearing +property. **Bias from optimizer's reasoning** — seeing the +optimizer's reasoning trace would lead the validator to accept +the optimizer's arguments rather than judging the artifact fresh. +**Consistency-with-self across runs** — seeing prior verdicts +would lead the validator to repeat its prior judgment instead of +judging the new proposal on its own merits. + +### (d) Handle the verdict + materialize on approve + +**`verdict: approve`:** materialize `docs/skill-optimizer//improved-skill/` by copying +the source's directory structure (from `.skill-optimizer//vendored-skill/` +or the existing `improved-skill/`) and applying the optimizer's +diff **only to the `optimization_target` file**. All other files +are copied verbatim. **Do NOT modify the source.** Commit: + +```bash +git add docs/skill-optimizer// +git commit -m "step 9: validate + apply improvement for " +``` + +The source skill file is NOT in the commit — git tracks +`docs/skill-optimizer//improved-skill/` alongside the verdict report, but never touches +the user's source or the vendored reference. + +**`verdict: needs-revision`** or **`reject`:** do NOT materialize. + +### (e) Hand off + +Three messages by verdict: + +- **approve:** "Validation complete; improvement applied at + `docs/skill-optimizer//improved-skill/`. The chain has reached its natural endpoint + for this iteration. Three realistic next steps: review and copy + locally; hand to a PR composer (auto-pilot can do this + end-to-end if running step 10); or re-bench against a workbench + that points at `docs/skill-optimizer//improved-skill/`." +- **needs-revision:** "Validator says needs-revision. Distill the + rationale into a directive and re-invoke step 8, then re-invoke + this step. Operator owns distillation; chain skills don't + auto-invoke each other." +- **reject:** "Validator rejects outright. Two paths: re-invoke + step 7 with a reframed weakness, or accept that this weakness + isn't addressable and exit honestly. If the rejection + contradicts the analyzer's framing (validator says 'this + doesn't address the real problem' but the analyzer thought it + did), the analyzer/validator are misaligned — step 7 reframe + is the cleaner fix." + +Don't auto-invoke step 7 or step 8. diff --git a/skills/validate/agents/validator.md b/skills/validate/agents/validator.md new file mode 100644 index 0000000..75af250 --- /dev/null +++ b/skills/validate/agents/validator.md @@ -0,0 +1,211 @@ +# Validator subagent + +You are dispatched by `skill-optimizer:validate` to **independently +judge** an improvement proposal (from step 8) and write +`09-validator-verdict.md`. On `verdict: approve`, the operator +session materializes `docs/skill-optimizer//improved-skill/` from the proposal. On +`needs-revision`/`reject`, the proposal is sent back to step 8. + +**Read first:** +[`../../shared/skill-design-philosophy.md`](../../shared/skill-design-philosophy.md) +— the cross-vendor synthesis of what makes a skill good and what +makes an improvement principled vs ducktape. Your verdict +reasoning should cite which principles the change embodies and +which anti-patterns it avoids. "Additive vs destructive" and +"general vs ducktape" are rows in the rubric — use them as +concrete tests, not abstractions. + +Your independence is the chain's load-bearing property at this +gate. The optimizer at step 8 produced the proposal with +incentive to have it accepted; you must judge as if seeing the +artifact fresh, without the optimizer's reasoning trace and +without your own prior verdicts. + +## Inputs (templated by the operator session) + +- `${PROPOSAL_PATH}` — `08-improvement-proposal.md` (the + optimizer's diff + rationale + self-check) +- `${SKILL_BEFORE_PATH}` — the current state of the skill (same + state the optimizer read at step 8: `docs/skill-optimizer//improved-skill/` if it + exists, else `.skill-optimizer//vendored-skill/`) +- `${SKILL_AFTER_PATH}` — temporary materialization of the + proposal applied to a copy of BEFORE (the operator session + prepares this; you read it but it's NOT the canonical + `docs/skill-optimizer//improved-skill/` yet) +- `${OPTIMIZATION_TARGET}` — relative path within both + `${SKILL_BEFORE_PATH}` and `${SKILL_AFTER_PATH}` of the file + whose diff you're judging. From `01-functionality.md`'s + frontmatter. Other files in the trees should be identical + between BEFORE and AFTER — if they differ, the optimizer + exceeded its scope; flag that. +- `${FUNCTIONALITY_PATH}` — `01-functionality.md` (for the + internal consistency check: does the change make sense given + the skill's stated responsibilities?) +- `${SUBMISSIONS_PATH}` — `02-submissions.md` if PR-bound (for + the external consistency check) +- `${VERDICT_OUTPUT_PATH}` — where to write + `09-validator-verdict.md` +- `${OPERATOR_DIRECTIVES}` — atomic new requirements (empty unless + this is a re-run with sharper guidance) + +## What you see + +- Skill BEFORE the change +- Skill AFTER the change (temporary materialization) +- The proposal artifact (`08-improvement-proposal.md`) — what the + optimizer claims to have done + their rationale + their self- + check against anti-patterns +- `01-functionality.md` (stated responsibilities) +- `02-submissions.md` if PR-bound (upstream conventions) +- Operator directives + +## What you do NOT see + +- Raw failed trials, `findings.txt`, `trace.jsonl` — your job is + to validate the PROPOSAL, not re-analyze the failures +- Test inputs (`skill-evals////workspace/`) — same reason +- The optimizer's internal reasoning trace from step 8 (only the + proposal artifact, not how the optimizer arrived at it). + Reading the optimizer's reasoning would lead you to accept + arguments they made for the change rather than judging the + artifact fresh. +- Your own prior `09-validator-verdict.md` or git history of it. + No consistency-with-self bias across re-runs — each invocation + judges the new proposal on its own merits. +- `07-analysis.md` directly. The analyzer's reasoning is mediated + through the optimizer's proposal (the optimizer's rationale + references which weakness the change addresses). Reading the + analysis would make you a second analyzer; your job is + independent validation of the proposal. + +## Output: `${VERDICT_OUTPUT_PATH}` — `09-validator-verdict.md` + +Frontmatter (runtime-relevant only): + +```yaml +--- +verdict: approve | needs-revision | reject +addresses_weaknesses: + - + - ... +--- +``` + +Body has two required sections (three if PR-bound): + +### 1. Internal consistency check + +Does the change make sense given the skill's stated +responsibilities in `01-functionality.md`? Walk through these +checks: + +- **Addresses the named weakness?** The proposal claims to + address weakness W. Compare the actual diff to the principle + the analyzer named for W. Does the diff apply that principle? +- **Additive vs. destructive?** Additive changes (adding + procedural guidance, examples, checks) preserve what works. + Destructive changes (removing, replacing, rewriting) need + justification — was the existing content actually the weakness, + or did the optimizer rewrite something that worked? +- **General vs. ducktape?** The optimizer's self-check section + walks through the anti-pattern list and explains why their + proposal is NOT each anti-pattern. Spot-check their reasoning: + is the self-check honest, or is the proposal incidentally doing + one of the anti-patterns and the self-check is hand-waving? +- **Contradicts stated responsibilities?** Does the change + contradict what `01-functionality.md` says the skill is + supposed to do? + +### 2. External consistency check (only if `${SUBMISSIONS_PATH}` exists) + +Forward-looking: would a PR carrying this change conform to the +upstream conventions in `02-submissions.md`? Even though step 9 +doesn't produce a PR, this check verifies the improved skill +COULD be turned into a valid PR by a downstream composer. + +- **Frontmatter spec** — does AFTER match the upstream's + frontmatter conventions? Required fields present, value formats + correct? +- **File-location conventions** — is the diff in the right place + per upstream conventions? +- **Prefix taxonomy** — if the upstream uses commit/PR prefixes, + is the diff's logical scope consistent with one of them? +- **Additive-only conventions** — some upstreams reject + destructive changes for non-bug-fix work; flag if the proposal + is destructive and the upstream's pattern is additive. + +### 3. Rationale + verdict reasoning + +State the verdict and explain it. For `needs-revision`/`reject`, +be concrete — the operator distills your rationale into a +directive for step 8. + +## Verdict definitions + +- **approve** — internal + external (if applicable) checks both + hold. Proposal is sound; step 9 materializes `docs/skill-optimizer//improved-skill/` + and the chain reaches its endpoint for this iteration. +- **needs-revision** — one or more checks has a fixable issue. + Examples: anti-pattern self-check is hand-waving in a way + that's correctable; frontmatter doesn't match upstream's + required field. Step 8 re-fires with the rationale distilled + into a directive. +- **reject** — fundamental problem the optimizer can't fix by + revising. Examples: the proposal addresses a different weakness + than it claims; the change contradicts `01-functionality.md`; + the named weakness as framed isn't addressable without + ducktape. Rejection surfaces the framing problem upward — the + operator may re-invoke step 7 to reformulate. + +## Reasoning protocol + +1. **Read the proposal first.** What does the optimizer claim to + have done? Which weaknesses do they claim to address? What's + their stated principle? +2. **Read BEFORE and AFTER.** Compute the actual diff (don't + trust the proposal's diff description blindly — verify by + comparing). +3. **Cross-check the diff against the claim.** Does the diff + actually implement the principle the optimizer cited? If + they claim "added procedural guidance for enumerating Xs" + but the diff just adds a MUST clause, that's a mismatch. +4. **Walk the anti-pattern self-check.** For each anti-pattern + the optimizer says they avoided, verify their justification. + The most common failure mode is incidental anti-pattern + overlap with hand-waving justification. +5. **Read `01-functionality.md`** — does the change preserve the + skill's stated responsibilities? +6. **If `02-submissions.md` exists**, walk the external + consistency check. +7. **Choose verdict.** The weakest of the checks. Be honest about + verdicts — false approvals let ducktape ship; false rejects + waste operator cycles. + +## Operator directives + +Examples that count as atomic requirements: + +- "be stricter on additive-vs-destructive — the prior verdict + approved a change I think was actually destructive" +- "the external check missed the frontmatter `version:` field + last round — look for it specifically" +- "weakness W's anti-pattern list is the load-bearing one; + scrutinize the self-check carefully" + +Examples that don't (context dumps — reject): + +- The prior verdict pasted in for you to "reconcile" +- The analyzer's section on weakness W pasted in for you to + "consider" + +## Return summary + +After writing the verdict, return: + +- Verdict (approve / needs-revision / reject) +- Weaknesses addressed (slugs from the proposal) +- Internal-check + external-check summary (one line each) +- For non-approve: the concrete suggestion the operator will + distill into a step-8 directive + +Keep it under 200 words. diff --git a/skills/write-tests/SKILL.md b/skills/write-tests/SKILL.md new file mode 100644 index 0000000..600b57a --- /dev/null +++ b/skills/write-tests/SKILL.md @@ -0,0 +1,195 @@ +--- +name: write-tests +description: Use when the user wants to build or implement the eval probes for a skill — phrases like "build the tests", "implement the probes", "set up the eval cases", "write test fixtures for X". Triggers after `skill-optimizer:design-tests` has produced `skill-evals///spec.yaml` files with at least one marked `picked: true`, and before any bench run. Use even when the user doesn't explicitly say "write tests" — any phrasing about turning the picked functionalities into concrete probes should trigger this. +--- + +# write-tests + +Step 4 of the skill-optimizer chain. **Maintenance step.** Reads +the picked functionalities from step 3 +(`skill-evals///spec.yaml` files where `picked: true`), +decides the probe set per functionality, dispatches one test-writer +subagent per probe (in parallel) to build the concrete workspace +files + grader scripts + smoke fixtures, then generates +`skill-evals//suite.yml` for the run-bench step. + +## What you produce + +Two kinds of output under `docs/skill-optimizer//`: + +1. **Probe folders** at `skill-evals////`. + For each picked functionality, one or more probe folders containing: + + ```text + skill-evals//// + spec.yaml # probe-level intent + workspace/ # the files the agent sees in /work + grader.mjs # the grading script (or .py) + smoke/ + good/findings.txt # fixture that should PASS the grader + bad/findings.txt # fixture that should FAIL the grader + empty/findings.txt # fixture that should FAIL the grader + checks/smoke.mjs # runs the grader against the three smoke fixtures + ``` + + Filesystem IS the state — a probe exists if its folder exists + with these contents; `grader.mjs` present + smoke-check passed + = built and verified. + +2. **`skill-evals//suite.yml`** — generated from the current + tree. Lists every probe under every `picked: true` functionality. + Regenerated by step (f) on every run. + +Probe-level `spec.yaml` format and the workbench schema for +`suite.yml` are defined in +[`agents/test-writer.md`](./agents/test-writer.md) +and +[`skills/shared/workbench.md`](../shared/workbench.md) +respectively. + +## Workflow + +### (a) Confirm prerequisites + +Three prerequisites: + +1. `01-functionality.md` exists. +2. `skill-evals//` exists with at least one + `/spec.yaml` having `picked: true`. +3. Every picked functionality's spec.yaml parses with required + fields. + +If any fails, surface to the user — don't derive missing fields or +guess. + +### (b) Handle iteration (maintenance step) + +Read [`iteration-protocol.md`](../shared/iteration-protocol.md) +and apply the maintenance-step flow: + +- Walk `skill-evals//`. For each picked functionality, determine which + probes already have built folders (`grader.mjs` exists). + Existing built probes are preserved by default; don't + re-dispatch unless `${OPERATOR_DIRECTIVES}` explicitly names them + for revision. +- Collect `${OPERATOR_DIRECTIVES}` per the protocol. +- If a directive will be destructive (rebuilding an existing + probe), commit a checkpoint BEFORE dispatching per the protocol's + safe-destructive-edits section. + +### (c) Plan probe set + user gate + +For each picked functionality, decide the full probe set: start with +`suggested_probes`, apply directives (adds/removes/revisions), +preserve existing built probes unless directives target them. + +Then present TWO options side by side, with the cost difference +made explicit (one Agent dispatch per new probe; expected wall time +roughly probe-count × test-writer time): + +1. **Minimal coverage** (default for first runs and dry-runs): + one probe per picked functionality — the first entry of each + `suggested_probes` list. Sized to surface obvious weaknesses + without paying for the full matrix. +2. **Full coverage:** every probe in every picked functionality's + `suggested_probes`. Sized for thorough measurement, used once + minimal coverage has confirmed the chain is working. + +Show the user the planned probe set per functionality under each +option, marking each probe as existing / new / rebuild. Recommend +minimal for a first run and full once minimal has converged. Ask +which to proceed with. + +Three realistic responses: + +1. **User picks minimal or full.** Proceed to (d) with that set. +2. **User adds revisions** (e.g., "give me the 2 most discriminating + per functionality", "drop probe X"). Treat as directives, mark + named probes for rebuild (apply destructive-edit checkpoint), + proceed to (d). +3. **User rejects the plan structure.** Surface and ask whether to + abandon (loop back to step 3 to revise functionality specs) or + retry with their feedback as directives. + +### (d) Dispatch test-writer subagents (parallel, one per probe) + +Read [`subagent-dispatch.md`](../shared/subagent-dispatch.md) +for the constraints. **Do NOT write workspace files or graders +yourself in this session** — dispatch test-writer subagents via +the `Agent` tool, in parallel (emit all dispatches in a single +message). For each probe, render the prompt template at +[`agents/test-writer.md`](./agents/test-writer.md) +inline by substituting that probe's `${PROBE_NAME}`, +`${FUNCTIONALITY_SPEC_PATH}`, `${PROBE_SPEC_PATH}`, +`${FUNCTIONALITY_PATH}`, `${SKILL_SOURCE_PATH}`, +`${OUTPUT_PROBE_DIR}`, and `${OPERATOR_DIRECTIVES}`, then pass the +rendered text as that Agent dispatch's `prompt` parameter (per +[`subagent-dispatch.md`](../shared/subagent-dispatch.md)'s +Dispatch protocol). + +Each test-writer sees: its single probe's intent (slug + parent +functionality spec); `01-functionality.md`; the skill source +content; `${OPERATOR_DIRECTIVES}` for its probe. + +Each test-writer does NOT see: other probes' specs, graders, or +workspaces; the eval grader's matching internals; `07-analysis.md`, +`08-improvement-proposal.md`, raw failure data; git history of +its own probe folder. + +**Why this matters:** two concerns. First, **fixture isolation** — +a subagent seeing all probes' fixtures would notice cross-probe +patterns and inadvertently homogenize them; per-probe isolation +keeps each fixture a representative instance. Second, **no +grader-leak hacking** — a subagent seeing another grader's matching +logic could write fixtures that incidentally satisfy that grader +too, making eval results look correlated when they're not. + +Source-content access is the one exception across the chain: probe +spec from step 3 fixes WHAT to test; source provides the HOW +(concrete violation patterns). Without source, the subagent would +invent generic patterns that may not trigger the skill's rules. + +Each test-writer writes the probe folder contents and returns: +probe name, what the fixture tests, grader logic in one line, +smoke-check result. + +If a test-writer reports BLOCKED (probe spec too abstract to +derive a fixture from, or the probe is genuinely unimplementable +in a static workbench), surface to the user. The fix is typically +a step 3 re-run with a more specific functionality spec; per the +no-auto-invocation rule, the user invokes step 3 explicitly. + +### (e) Run smoke check + +Each probe has its own `checks/smoke.mjs`. After test-writer +dispatches return, run each new or rebuilt probe's smoke runner: + +```bash +node skill-evals////checks/smoke.mjs +``` + +Expected per probe: GOOD passes, BAD fails, EMPTY fails. If any +fails, surface to the user. Two realistic responses: + +1. **Re-dispatch the test-writer** with the smoke failure as a + directive. Treat as a single-probe rebuild (back to (b) for + the checkpoint, then (d)). +2. **Remove the probe** from this functionality's set, or de-pick + the parent functionality (a backward trigger to step 3; per + the no-auto-invocation rule, surface the option and let the + user invoke step 3). + +### (f) Generate suite.yml + commit + hand off + +Walk `skill-evals//`. For each `/` where `spec.yaml` +has `picked: true`, list every probe folder and emit a suite.yml +entry per the workbench schema. Overwrite `skill-evals//suite.yml`. + +Commit `skill-evals//` and hand off to `skill-optimizer:run-bench`. + +## Edge cases + +- **User de-picks a functionality between step 4 runs** — its + probe folders stay on disk; step (f) excludes them from the + regenerated `skill-evals//suite.yml`. If re-picked later, + probes are already there. diff --git a/skills/write-tests/agents/test-writer.md b/skills/write-tests/agents/test-writer.md new file mode 100644 index 0000000..70465aa --- /dev/null +++ b/skills/write-tests/agents/test-writer.md @@ -0,0 +1,224 @@ +# Test-writer subagent + +You are dispatched by `skill-optimizer:write-tests` to build ONE +probe for ONE functionality — the concrete workspace files, the +grader script, and the smoke-check fixtures. Many test-writer +subagents run in parallel (one per probe); each is responsible for +its own probe folder and nothing else. + +## Inputs (templated by the operator session) + +- `${PROBE_NAME}` — slug of this probe (becomes the folder name + under `skill-evals///`) +- `${FUNCTIONALITY_SPEC_PATH}` — the parent functionality's + `spec.yaml` (gives you context on what responsibility this probe + is testing) +- `${PROBE_SPEC_PATH}` — where you write the probe's own + `spec.yaml` describing what this probe sets up + expects +- `${FUNCTIONALITY_PATH}` — `01-functionality.md` (skill's stated + responsibilities and classification) +- `${SKILL_SOURCE_PATH}` — `.skill-optimizer//vendored-skill/` (step 1 vendored the + source regardless of upstream/local) +- `${OUTPUT_PROBE_DIR}` — `skill-evals////` +- `${OPERATOR_DIRECTIVES}` — case-level revision hints for this + probe (empty unless this is a rebuild) + +## What you see + +- `${FUNCTIONALITY_SPEC_PATH}` (parent functionality's spec) +- `${FUNCTIONALITY_PATH}` (the briefing document) +- The skill source content via `${SKILL_SOURCE_PATH}` — this is + the one step in the chain where a generative subagent sees the + source; needed for concrete violation patterns and realistic + fixture content +- `${OPERATOR_DIRECTIVES}` for your probe + +## What you do NOT see + +- **Other probes** — their specs, workspaces, graders, smoke + fixtures. Per-probe isolation prevents cross-probe + homogenization (a single subagent seeing all probes would + notice patterns and write similar fixtures, defeating + independent coverage) and grader-leak hacking (writing fixtures + that incidentally satisfy a sibling probe's grader, making eval + results look correlated when they're not). +- The eval grader's matching internals (you write the grader; you + don't read how OTHER graders match). +- `07-analysis.md`, `08-improvement-proposal.md`, + `09-validator-verdict.md`, raw failure data +- Git history of your own probe folder (anti-ducktape — derive + fresh from the probe spec, not from a prior fixture's shape) + +## Output: `${OUTPUT_PROBE_DIR}/` + +Five artifacts inside the probe folder. The exact layout is: + +```text +${OUTPUT_PROBE_DIR}/ + spec.yaml # 1. probe-level intent + workspace/ # 2. files the agent sees in /work + + grader.mjs # 3. grading script — AT PROBE ROOT + smoke/ # 4. smoke fixtures + good/findings.txt + bad/findings.txt + empty/findings.txt + checks/ # 5. smoke check runner + smoke.mjs # references ../grader.mjs (relative to checks/) +``` + +**`grader.mjs` lives at the probe root, NOT under `checks/`.** The +smoke runner at `checks/smoke.mjs` references it as +`../grader.mjs`. This layout is load-bearing for the workbench +(`skill-evals//suite.yml` generated by the operator points at +`grader.mjs` at the probe root); a grader put under `checks/` +will not be found. + +The workbench schema (suite.yml, grader contract, smoke format) +lives in +[`../../shared/workbench.md`](../../shared/workbench.md). + +### 1. `${PROBE_SPEC_PATH}` — the probe's own spec.yaml + +```yaml +name: +description: +workspace_overview: > + +expected_agent_behavior: > + +grader_logic: > + +``` + +**Task prompts describe the user's actual task; do not reference the +skill explicitly.** The harness mounts the skill at the agent's native +discovery path (e.g., `~/.claude/skills//SKILL.md`). The agent +decides whether to invoke it based on the skill's frontmatter +`description`. "Skill didn't trigger" is then a measurable weakness +class — don't pre-trigger it via the task prompt. + +Example task prompt (good): "Review /work/ProductCard.tsx for +compliance issues. Write findings to /work/findings.txt." + +Example task prompt (bad — pre-triggers): "Use the skill at +/work/web-design-guidelines/SKILL.md to review /work/ProductCard.tsx." + +### 2. `workspace/` — files the agent sees in `/work` + +The fixture content. Whatever files the parent functionality's +probe needs the agent to see. Be concrete — the skill source +should guide the realistic violation patterns; the probe spec's +`expected_agent_behavior` defines what the agent should do with +them. + +### 3. `grader.mjs` (or `.py`) — the grading script + +Per the workbench schema. Reads `findings.txt` and the workspace +state; returns pass/fail. Must be deterministic — no LLM calls in +the grader. Avoid over-strict matching (don't require an exact +output format the skill never asks for); avoid over-loose matching +(don't pass on partial work that misses the responsibility). + +### 4. `smoke/{good,bad,empty}/findings.txt` — smoke fixtures + +Three hand-crafted findings.txt examples: + +- `smoke/good/findings.txt` — what a correctly-behaving agent + would output; grader should pass. +- `smoke/bad/findings.txt` — what an incorrectly-behaving agent + would output; grader should fail. +- `smoke/empty/findings.txt` — minimal/empty output (agent gave + up or hallucinated nothing); grader should fail. + +These are syntactic smoke checks — they verify your grader +correctly classifies the three cases you yourself wrote. They do +NOT verify that the probe semantically tests the responsibility +(that's step 5's job). + +### 5. `checks/smoke.mjs` — runs the grader against the smoke fixtures + +A small script that exercises `grader.mjs` against +`smoke/{good,bad,empty}/findings.txt` and reports +GOOD-pass/BAD-fail/EMPTY-fail outcomes. The operator session runs +this at step 4 (e) to verify the grader. + +**Import path matters.** Because `smoke.mjs` lives in +`checks/` and `grader.mjs` lives at the probe root, the import +is `../grader.mjs`: + +```js +// checks/smoke.mjs +import { grade } from '../grader.mjs'; +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const probeRoot = dirname(dirname(fileURLToPath(import.meta.url))); +for (const label of ['good', 'bad', 'empty']) { + const findings = readFileSync(join(probeRoot, 'smoke', label, 'findings.txt'), 'utf-8'); + const passed = await grade({ findings, workspaceDir: join(probeRoot, 'workspace') }); + console.log(`${label}: ${passed ? 'PASS' : 'FAIL'}`); +} +``` + +Adjust the `grade` call signature to match what your `grader.mjs` +exports. + +## Reasoning protocol + +1. **Read the parent functionality spec first** — understand what + responsibility this probe is testing. The probe must exercise + that responsibility, not something adjacent. +2. **Read the skill source for concrete violation patterns** — + what specifically does the skill instruct the agent to do or + avoid? The workspace should contain inputs that surface those + patterns. +3. **Read `01-functionality.md`** for the broader context (the + skill's audience, triggers, terminology). +4. **Design the workspace.** Concrete files, not abstract setup + instructions. The agent gets these files in `/work` and must + act on them. +5. **Write the grader.** Determine what `findings.txt` content + indicates the agent did the right thing. Keep matching robust + (don't require exact strings the skill never asks for) but + discriminating (don't pass on outputs that miss the + responsibility). +6. **Write the three smoke fixtures.** GOOD = what a correctly- + behaving agent would output for THIS workspace. BAD = a + plausibly-wrong output that misses the responsibility. EMPTY + = no findings or trivial findings. +7. **Self-check the smoke fixtures against your grader logic** — + does GOOD pass? BAD fail? EMPTY fail? If not, fix the fixture + or the grader before reporting done. + +## Edge cases to surface as BLOCKED + +If you hit any of these, return BLOCKED with reasoning rather +than building a misleading probe: + +- **Probe spec is too abstract to derive a concrete fixture** — + e.g., "tests good code style" without specifying what code or + what style. Fix is at step 3 (more specific probe spec). +- **Probe requires real-time API access** (live data, external + services the workbench can't mock) — not testable in the + static workbench. +- **Probe requires the skill to interact with a human** — the + workbench runs agents headlessly; no interactive probes. +- **The responsibility can't be cleanly graded deterministically** + — e.g., "produces good prose" requires LLM judgment. + Deterministic graders only. + +## Return summary + +After writing the probe folder contents, return a brief summary: + +- Probe name +- One-line description of what the fixture tests +- One-line grader logic +- Smoke-check result (your self-check) +- Any blockers (if you couldn't build the probe) + +Keep it under 150 words. The operator session aggregates these +across all parallel test-writers. diff --git a/src/workbench/acp/auth.ts b/src/workbench/acp/auth.ts new file mode 100644 index 0000000..e8b4645 --- /dev/null +++ b/src/workbench/acp/auth.ts @@ -0,0 +1,84 @@ +import { existsSync, readFileSync } from 'node:fs'; +import { homedir } from 'node:os'; +import { join, normalize } from 'node:path'; +import type { AgentConfig, SubscriptionEnvExtract } from '../agents/registry.js'; + +export interface AuthContext { + home?: string; // override $HOME for testing + env: Record; +} + +export interface SubscriptionAuthResult { + mode: 'subscription'; + files: Array<{ hostPath: string; containerPath: string }>; + envOverrides: Record; // env vars to set in container (extracted from credentials file) +} + +export interface EnvAuthResult { + mode: 'env'; + envNames: string[]; +} + +export type AuthResult = SubscriptionAuthResult | EnvAuthResult; + +export function resolveAuth(agent: AgentConfig, ctx: AuthContext): AuthResult { + const home = ctx.home ?? homedir(); + + if (agent.subscriptionAuth) { + const detectPath = expandHome(agent.subscriptionAuth.detectFile, home); + if (existsSync(detectPath)) { + const envOverrides: Record = {}; + for (const extract of agent.subscriptionAuth.extractEnv ?? []) { + const value = extractEnvFromFile(extract, home); + if (value !== undefined) { + envOverrides[extract.targetEnv] = value; + } + } + return { + mode: 'subscription', + files: agent.subscriptionAuth.files.map((f) => ({ + hostPath: expandHome(f.hostPath, home), + containerPath: f.containerPath, // {home} placeholder, resolved at mount time + })), + envOverrides, + }; + } + } + + const missing = agent.requiresEnv.filter((name) => !ctx.env[name]); + if (missing.length > 0 && agent.requiresEnv.length > 0) { + const hint = agent.loginHint ? ` (run \`${agent.loginHint}\` or set ${missing.join(', ')})` : ''; + throw new Error( + `Agent ${agent.name} requires auth but none is available${hint}. Missing: ${missing.join(', ')}.`, + ); + } + + return { mode: 'env', envNames: agent.requiresEnv }; +} + +function extractEnvFromFile(extract: SubscriptionEnvExtract, home: string): string | undefined { + const path = expandHome(extract.source, home); + if (!existsSync(path)) return undefined; + let parsed: unknown; + try { + parsed = JSON.parse(readFileSync(path, 'utf-8')); + } catch { + return undefined; + } + let cursor: unknown = parsed; + for (const segment of extract.jsonPath.split('.')) { + if (cursor === null || typeof cursor !== 'object') return undefined; + cursor = (cursor as Record)[segment]; + } + return typeof cursor === 'string' ? cursor : undefined; +} + +function expandHome(p: string, home: string): string { + if (p.startsWith('~/')) return normalize(join(home, p.slice(2))); + if (p === '~') return home; + return p; +} + +export function resolveContainerPath(template: string, agentHome: string): string { + return template.replace('{home}', agentHome); +} diff --git a/src/workbench/acp/client.ts b/src/workbench/acp/client.ts new file mode 100644 index 0000000..d807617 --- /dev/null +++ b/src/workbench/acp/client.ts @@ -0,0 +1,49 @@ +import { ClientSideConnection, ndJsonStream } from '@agentclientprotocol/sdk'; +import type { + Client, + SessionNotification, + RequestPermissionRequest, + RequestPermissionResponse, +} from '@agentclientprotocol/sdk'; +import type { DockerExecStream } from './transport.js'; + +export interface WorkbenchClientOptions { + stream: DockerExecStream; + onSessionUpdate: (notification: SessionNotification) => void; +} + +export interface WorkbenchClient { + connection: ClientSideConnection; + close(): Promise; +} + +export function createWorkbenchClient(opts: WorkbenchClientOptions): WorkbenchClient { + // Note: ndJsonStream wraps the raw byte streams as JSON-RPC envelopes. + const ndjson = ndJsonStream(opts.stream.outgoing, opts.stream.incoming); + + const connection = new ClientSideConnection( + (_agent): Client => ({ + sessionUpdate: async (params: SessionNotification) => { + opts.onSessionUpdate(params); + }, + // Auto-approve all tool calls. The container is the security boundary, + // not the permission gate. Approving everything matches the headless- + // bench model and matches benchflow's pattern. + requestPermission: async ( + params: RequestPermissionRequest, + ): Promise => ({ + outcome: { outcome: 'selected', optionId: params.options[0]?.optionId ?? 'allow' }, + }), + // We don't expose filesystem capabilities to the agent over ACP; the + // agent uses its own tools to touch /work directly. + }), + ndjson, + ); + + return { + connection, + async close() { + await opts.stream.close(); + }, + }; +} diff --git a/src/workbench/acp/mcp-config-writer.ts b/src/workbench/acp/mcp-config-writer.ts new file mode 100644 index 0000000..80fb4d6 --- /dev/null +++ b/src/workbench/acp/mcp-config-writer.ts @@ -0,0 +1,89 @@ +import { mkdirSync, readFileSync, writeFileSync, existsSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import type { AgentConfig } from '../agents/registry.js'; + +// pi-acp MCP resolution (plan Task 9b, decided 2026-05-25): +// Decision: option (c) — defer pi-acp MCP support to a follow-up. +// Reason: pi-acp's native MCP config path is not documented in the +// pi-coding-agent or pi-acp npm packages; investigating + implementing +// would block v1 on a tangent. For v1, any case declaring `mcpServers:` +// with `agent: pi-acp` gets a clear "pending" error from writeMcpConfig +// (see switch case below). Users requiring MCP should use one of the +// supported agents: claude-agent-acp, codex-acp, gemini, opencode. + +export interface McpServerSpec { + command?: string; + args?: string[]; + env?: Record; + url?: string; + headers?: Record; +} + +export interface WriteMcpConfigParams { + agent: AgentConfig; + caseConfig: { mcpServers?: Record }; + agentHomeOnHost: string; // path on host that will become container's $HOME +} + +export function writeMcpConfig(params: WriteMcpConfigParams): void { + const servers = params.caseConfig.mcpServers; + if (!servers || Object.keys(servers).length === 0) return; + + switch (params.agent.name) { + case 'claude-agent-acp': + return writeClaude(servers, params.agentHomeOnHost); + case 'codex-acp': + return writeCodex(servers, params.agentHomeOnHost); + case 'gemini': + return writeGemini(servers, params.agentHomeOnHost); + case 'opencode': + return writeOpencode(servers, params.agentHomeOnHost); + case 'pi-acp': + throw new Error('pi-acp MCP support pending — see plan Task 9b'); + default: + throw new Error(`MCP not implemented for agent ${params.agent.name}`); + } +} + +function writeClaude(servers: Record, home: string): void { + const path = join(home, '.claude.json'); + mkdirSync(dirname(path), { recursive: true }); + const existing = existsSync(path) ? JSON.parse(readFileSync(path, 'utf-8')) : {}; + existing.mcpServers = { ...(existing.mcpServers ?? {}), ...servers }; + writeFileSync(path, JSON.stringify(existing, null, 2)); +} + +function writeCodex(servers: Record, home: string): void { + const path = join(home, '.codex/config.toml'); + mkdirSync(dirname(path), { recursive: true }); + const blocks: string[] = []; + for (const [name, spec] of Object.entries(servers)) { + blocks.push(`[mcp_servers.${name}]`); + if (spec.command) blocks.push(`command = ${JSON.stringify(spec.command)}`); + if (spec.args) blocks.push(`args = ${JSON.stringify(spec.args)}`); + if (spec.env) { + blocks.push(`[mcp_servers.${name}.env]`); + for (const [k, v] of Object.entries(spec.env)) { + blocks.push(`${k} = ${JSON.stringify(v)}`); + } + } + blocks.push(''); + } + writeFileSync(path, blocks.join('\n')); +} + +function writeGemini(servers: Record, home: string): void { + const path = join(home, '.gemini/settings.json'); + mkdirSync(dirname(path), { recursive: true }); + const existing = existsSync(path) ? JSON.parse(readFileSync(path, 'utf-8')) : {}; + existing.mcpServers = { ...(existing.mcpServers ?? {}), ...servers }; + writeFileSync(path, JSON.stringify(existing, null, 2)); +} + +function writeOpencode(servers: Record, home: string): void { + const path = join(home, '.config/opencode/opencode.json'); + mkdirSync(dirname(path), { recursive: true }); + const existing = existsSync(path) ? JSON.parse(readFileSync(path, 'utf-8')) : {}; + existing.mcp = { ...(existing.mcp ?? {}), ...servers }; + writeFileSync(path, JSON.stringify(existing, null, 2)); +} diff --git a/src/workbench/acp/skill-deploy.ts b/src/workbench/acp/skill-deploy.ts new file mode 100644 index 0000000..0e7ad91 --- /dev/null +++ b/src/workbench/acp/skill-deploy.ts @@ -0,0 +1,37 @@ +import { join } from 'node:path'; +import type { AgentConfig } from '../agents/registry.js'; + +export interface SkillMount { + hostPath: string; + containerPath: string; + readOnly: boolean; +} + +export interface ComputeSkillMountParams { + agent: AgentConfig; + skillSlug: string; + hostSkillDir: string; // absolute path on host to the skill folder + agentHome: string; // /home/agent (or whatever the container user's $HOME is) +} + +export function computeSkillMount(params: ComputeSkillMountParams): SkillMount { + const skillRoot = params.agent.skillPaths[0]; + if (!skillRoot) { + throw new Error(`Agent ${params.agent.name} has no skillPaths configured`); + } + const expanded = skillRoot.replace('$HOME', params.agentHome); + return { + hostPath: params.hostSkillDir, + containerPath: join(expanded, params.skillSlug), + readOnly: true, + }; +} + +export function dockerMountFlag(mount: SkillMount): string { + const ro = mount.readOnly ? ':ro' : ':rw'; + return `-v ${shellQuote(mount.hostPath)}:${shellQuote(mount.containerPath)}${ro}`; +} + +function shellQuote(s: string): string { + return `'${s.replace(/'/g, `'\\''`)}'`; +} diff --git a/src/workbench/acp/trace-recorder.ts b/src/workbench/acp/trace-recorder.ts new file mode 100644 index 0000000..367e0b0 --- /dev/null +++ b/src/workbench/acp/trace-recorder.ts @@ -0,0 +1,48 @@ +import { writeFileSync, appendFileSync, mkdirSync, readFileSync } from 'node:fs'; +import { dirname } from 'node:path'; + +export interface TraceHeader { + caseName: string; + agent: string; + model: string; + startedAt: string; + endedAt?: string; +} + +export interface TraceRecorder { + recordRaw(message: unknown): void; + finalize(endedAt?: string): void; +} + +export function createTraceRecorder(params: { + tracePath: string; + header: TraceHeader; +}): TraceRecorder { + mkdirSync(dirname(params.tracePath), { recursive: true }); + const headerLine = JSON.stringify({ + type: 'trace_start', + schemaVersion: 2, + ...params.header, + }); + // Write header up front so partial traces are still self-describing on crash. + writeFileSync(params.tracePath, headerLine + '\n'); + + return { + recordRaw(message: unknown) { + appendFileSync(params.tracePath, JSON.stringify(message) + '\n'); + }, + finalize(endedAt?: string) { + if (!endedAt) return; + // Rewrite header with endedAt; preserve remaining lines. + const updatedHeader = JSON.stringify({ + type: 'trace_start', + schemaVersion: 2, + ...params.header, + endedAt, + }); + const existing = readFileSync(params.tracePath, 'utf-8'); + const rest = existing.split('\n').slice(1).join('\n'); + writeFileSync(params.tracePath, updatedHeader + '\n' + rest); + }, + }; +} diff --git a/src/workbench/acp/transport.ts b/src/workbench/acp/transport.ts new file mode 100644 index 0000000..f99e0ff --- /dev/null +++ b/src/workbench/acp/transport.ts @@ -0,0 +1,52 @@ +import type { ChildProcessWithoutNullStreams } from 'node:child_process'; + +/** + * Raw byte transport for ACP over docker exec stdio. + * + * Exposes `outgoing` (bytes to write to the subprocess stdin) and + * `incoming` (bytes read from subprocess stdout) so that the ACP client + * (Task 3) can wrap them with `ndJsonStream` from `@agentclientprotocol/sdk`. + */ +export interface DockerExecStream { + outgoing: WritableStream; + incoming: ReadableStream; + close(): Promise; +} + +export function createDockerExecStream( + child: ChildProcessWithoutNullStreams, +): DockerExecStream { + const outgoing = new WritableStream({ + write(chunk) { + return new Promise((resolve, reject) => { + child.stdin.write(chunk, (err) => (err ? reject(err) : resolve())); + }); + }, + close() { + return new Promise((resolve, reject) => { + child.stdin.end((err?: Error | null) => (err ? reject(err) : resolve())); + }); + }, + }); + + const incoming = new ReadableStream({ + start(controller) { + child.stdout.on('data', (chunk: Buffer) => controller.enqueue(chunk)); + child.stdout.on('end', () => controller.close()); + child.stdout.on('error', (err) => controller.error(err)); + }, + }); + + return { + outgoing, + incoming, + async close() { + try { + child.stdin.end(); + } catch { + // ignore + } + child.kill(); + }, + }; +} diff --git a/src/workbench/agents/install-snippets.ts b/src/workbench/agents/install-snippets.ts new file mode 100644 index 0000000..26264c9 --- /dev/null +++ b/src/workbench/agents/install-snippets.ts @@ -0,0 +1,15 @@ +import { AGENTS } from './registry.js'; + +export function generateDockerInstallBlock(): string { + const lines: string[] = []; + for (const cfg of Object.values(AGENTS)) { + lines.push(`# Install ${cfg.name}: ${cfg.description}`); + lines.push(`RUN ${cfg.installCmd}`); + lines.push(''); + } + return lines.join('\n'); +} + +if (import.meta.url === `file://${process.argv[1]}`) { + console.log(generateDockerInstallBlock()); +} diff --git a/src/workbench/agents/registry.ts b/src/workbench/agents/registry.ts new file mode 100644 index 0000000..955c027 --- /dev/null +++ b/src/workbench/agents/registry.ts @@ -0,0 +1,234 @@ +export interface CredentialFile { + path: string; // Target path in container; may use {home} + envSource: string; // Env var on host to read value from + template?: string; // If set, value is inserted into template at {value} + mkdir?: boolean; // Create parent dir; default true +} + +export interface HostAuthFile { + hostPath: string; // ~/.claude/.credentials.json + containerPath: string; // {home}/.claude/.credentials.json +} + +export interface SubscriptionEnvExtract { + // Extract a value out of a JSON credentials file and expose it as an env var + // inside the trial container. Lets agents that use OAuth-style subscription + // auth (Claude, Codex) get their bearer token without us mounting and parsing + // the file on every agent CLI invocation. + source: string; // host file path (with ~ expansion); usually same as detectFile + jsonPath: string; // dot-separated path into the JSON, e.g. "claudeAiOauth.accessToken" + targetEnv: string; // env var to set in container, e.g. "ANTHROPIC_AUTH_TOKEN" +} + +export interface SubscriptionAuth { + replacesEnv: string; // e.g. "ANTHROPIC_API_KEY" + detectFile: string; // host path to check for login + files: HostAuthFile[]; // all files to copy when sub-auth used + extractEnv?: SubscriptionEnvExtract[]; // optional: derive container env vars from the credentials file +} + +export type ApiProtocol = + | 'anthropic-messages' + | 'openai-completions' + | 'openai-responses' + | ''; + +export type AcpModelFormat = 'bare' | 'provider/model'; + +export interface AgentConfig { + name: string; + description: string; + installCmd: string; // bash, runs at IMAGE BUILD time + launchCmd: string; // bash, runs per-trial via docker exec + requiresEnv: string[]; + apiProtocol: ApiProtocol; + envMapping: Record; // SKILL_OPT_PROVIDER_* → agent-native + skillPaths: string[]; // e.g. ["$HOME/.claude/skills"] + credentialFiles: CredentialFile[]; + homeDirs: string[]; + subscriptionAuth: SubscriptionAuth | null; + acpModelFormat: AcpModelFormat; + supportsAcpSetModel: boolean; + loginHint?: string; // shown in fail-loud message +} + +export const AGENTS: Record = { + 'claude-agent-acp': { + name: 'claude-agent-acp', + description: 'Claude Code via ACP (Anthropic CLI)', + installCmd: `npm install -g @zed-industries/claude-agent-acp@latest`, + launchCmd: `claude-agent-acp`, + requiresEnv: ['ANTHROPIC_API_KEY'], + apiProtocol: 'anthropic-messages', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'ANTHROPIC_BASE_URL', + SKILL_OPT_PROVIDER_API_KEY: 'ANTHROPIC_AUTH_TOKEN', + SKILL_OPT_PROVIDER_MODEL: 'ANTHROPIC_MODEL', + }, + skillPaths: ['$HOME/.claude/skills'], + credentialFiles: [], + // The Anthropic SDK writes runtime debug logs to ~/.claude/debug/.txt; + // session/new fails on ENOENT if that dir doesn't exist in the staged home. + homeDirs: ['.claude/debug'], + subscriptionAuth: { + replacesEnv: 'ANTHROPIC_API_KEY', + detectFile: '~/.claude/.credentials.json', + files: [ + { hostPath: '~/.claude/.credentials.json', containerPath: '{home}/.claude/.credentials.json' }, + ], + // claude-agent-acp uses @anthropic-ai/claude-agent-sdk, which reads + // ANTHROPIC_API_KEY / ANTHROPIC_AUTH_TOKEN env vars — NOT the credentials + // file directly. So we mount the file (for defensive completeness) AND + // extract the OAuth bearer into ANTHROPIC_AUTH_TOKEN. + extractEnv: [ + { + source: '~/.claude/.credentials.json', + jsonPath: 'claudeAiOauth.accessToken', + targetEnv: 'ANTHROPIC_AUTH_TOKEN', + }, + ], + }, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'claude login', + }, + 'codex-acp': { + name: 'codex-acp', + description: 'OpenAI Codex via ACP', + installCmd: `npm install -g @zed-industries/codex-acp@latest`, + launchCmd: `codex-acp \${OPENAI_BASE_URL:+-c openai_base_url=$OPENAI_BASE_URL}`, + requiresEnv: ['OPENAI_API_KEY'], + apiProtocol: 'openai-responses', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'OPENAI_BASE_URL', + SKILL_OPT_PROVIDER_API_KEY: 'OPENAI_API_KEY', + }, + skillPaths: ['$HOME/.agents/skills'], + credentialFiles: [ + { + path: '{home}/.codex/auth.json', + envSource: 'OPENAI_API_KEY', + template: '{"OPENAI_API_KEY": "{value}"}', + }, + ], + homeDirs: [], + subscriptionAuth: { + replacesEnv: 'OPENAI_API_KEY', + detectFile: '~/.codex/auth.json', + files: [ + { hostPath: '~/.codex/auth.json', containerPath: '{home}/.codex/auth.json' }, + ], + // codex auth.json format is { "OPENAI_API_KEY": "..." }; extract for env passthrough. + extractEnv: [ + { + source: '~/.codex/auth.json', + jsonPath: 'OPENAI_API_KEY', + targetEnv: 'OPENAI_API_KEY', + }, + ], + }, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'codex login', + }, + 'gemini': { + name: 'gemini', + description: 'Google Gemini CLI via ACP', + installCmd: `npm install -g @google/gemini-cli@latest`, + launchCmd: `gemini --acp --yolo`, + requiresEnv: ['GOOGLE_API_KEY'], + apiProtocol: '', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'GEMINI_API_BASE_URL', + SKILL_OPT_PROVIDER_API_KEY: 'GOOGLE_API_KEY', + }, + skillPaths: ['$HOME/.gemini/skills'], + credentialFiles: [], + homeDirs: [], + subscriptionAuth: { + replacesEnv: 'GEMINI_API_KEY', + detectFile: '~/.gemini/oauth_creds.json', + files: [ + { hostPath: '~/.gemini/oauth_creds.json', containerPath: '{home}/.gemini/oauth_creds.json' }, + { hostPath: '~/.gemini/settings.json', containerPath: '{home}/.gemini/settings.json' }, + { hostPath: '~/.gemini/google_accounts.json', containerPath: '{home}/.gemini/google_accounts.json' }, + ], + }, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'gemini auth login', + }, + 'opencode': { + name: 'opencode', + description: 'OpenCode via ACP — open-source coding agent', + installCmd: `npm install -g opencode-ai@latest`, + launchCmd: `opencode acp`, + requiresEnv: [], + apiProtocol: '', + envMapping: { + SKILL_OPT_PROVIDER_BASE_URL: 'OPENAI_BASE_URL', + }, + skillPaths: ['$HOME/.opencode/skills'], + credentialFiles: [], + homeDirs: ['.opencode'], + subscriptionAuth: null, + acpModelFormat: 'provider/model', + supportsAcpSetModel: true, + loginHint: 'set OPENAI_API_KEY (or provider-specific key)', + }, + 'pi-acp': { + name: 'pi-acp', + description: 'Pi coding agent via ACP', + installCmd: `npm install -g @mariozechner/pi-coding-agent@latest pi-acp@latest`, + launchCmd: `/opt/skill-opt/bin/pi-acp-launcher`, + requiresEnv: ['OPENROUTER_API_KEY'], + apiProtocol: '', + envMapping: {}, + skillPaths: ['$HOME/.pi/agent/skills', '$HOME/.agents/skills'], + credentialFiles: [], + homeDirs: ['.pi'], + subscriptionAuth: null, + acpModelFormat: 'bare', + supportsAcpSetModel: true, + loginHint: 'set OPENROUTER_API_KEY', + }, +}; + +export const AGENT_ALIASES: Record = { + claude: 'claude-agent-acp', + codex: 'codex-acp', + pi: 'pi-acp', +}; + +export function resolveAgent(spec: string): AgentConfig { + const canonical = AGENT_ALIASES[spec] ?? spec; + const cfg = AGENTS[canonical]; + if (cfg) return cfg; + + const known = Object.keys(AGENTS); + const close = closestMatch(canonical, known); + if (close) { + throw new Error(`Unknown agent: ${spec}. Did you mean: ${close}?`); + } + throw new Error(`Unknown agent: ${spec}. Available: ${known.join(', ')}`); +} + +function closestMatch(needle: string, haystack: string[]): string | undefined { + let best: { name: string; score: number } | undefined; + for (const name of haystack) { + const score = sharedPrefix(needle, name) + sharedPrefix( + needle.split('').reverse().join(''), + name.split('').reverse().join(''), + ); + if (score > 4 && (!best || score > best.score)) { + best = { name, score }; + } + } + return best?.name; +} + +function sharedPrefix(a: string, b: string): number { + let i = 0; + while (i < a.length && i < b.length && a[i] === b[i]) i++; + return i; +} diff --git a/src/workbench/case-loader.ts b/src/workbench/case-loader.ts index 5f55653..8087279 100644 --- a/src/workbench/case-loader.ts +++ b/src/workbench/case-loader.ts @@ -3,6 +3,7 @@ import { dirname, extname, resolve } from 'node:path'; import { parse as parseYaml } from 'yaml'; +import { resolveAgent } from './agents/registry.js'; import { ensureOpenRouterModelRef } from './models.js'; import type { ResolvedWorkbenchCase, @@ -56,6 +57,14 @@ export function resolveWorkbenchCaseConfig( throw new Error(`Workbench case ${resolvedConfigPath}: field "artifacts" is invalid; inspect outputs in the workspace or use --keep-workspace`); } + if (!parsed.agent || typeof parsed.agent !== 'string') { + throw new Error( + `Case ${resolvedConfigPath} is missing required field \`agent:\`. ` + + `Valid agents: claude-agent-acp, codex-acp, gemini, opencode, pi-acp.`, + ); + } + const agentCfg = resolveAgent(parsed.agent); + const name = requireNonEmptyString(parsed, 'name', resolvedConfigPath); const references = requireNonEmptyString(parsed, 'references', resolvedConfigPath); const task = requireNonEmptyString(parsed, 'task', resolvedConfigPath); @@ -66,7 +75,8 @@ export function resolveWorkbenchCaseConfig( const env = readStringArray(parsed, 'env', resolvedConfigPath); const setup = readStringArray(parsed, 'setup', resolvedConfigPath); const cleanup = readStringArray(parsed, 'cleanup', resolvedConfigPath); - const model = ensureOpenRouterModelRef(readOptionalString(parsed, 'model', resolvedConfigPath) ?? DEFAULT_WORKBENCH_MODEL); + const rawModel = readOptionalString(parsed, 'model', resolvedConfigPath) ?? DEFAULT_WORKBENCH_MODEL; + const model = agentCfg.name === 'pi-acp' ? ensureOpenRouterModelRef(rawModel) : rawModel.trim(); const timeoutSeconds = readOptionalTimeoutSeconds(parsed, resolvedConfigPath) ?? DEFAULT_WORKBENCH_TIMEOUT_SECONDS; const referencesDir = resolve(resolvedConfigDir, references); @@ -93,7 +103,9 @@ export function resolveWorkbenchCaseConfig( env, setup, cleanup, + agent: agentCfg.name, model, + skillUnderTest: parsed.skillUnderTest as ResolvedWorkbenchCase['skillUnderTest'], timeoutSeconds, }; } diff --git a/src/workbench/container-runner.ts b/src/workbench/container-runner.ts index 742183b..cef8032 100644 --- a/src/workbench/container-runner.ts +++ b/src/workbench/container-runner.ts @@ -1,42 +1,15 @@ -import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { writeFileSync } from 'node:fs'; import { dirname, join } from 'node:path'; import { runGraderCommands } from './check-runner.js'; import { loadWorkbenchCase } from './case-loader.js'; -import { buildWorkbenchMetrics } from './metrics.js'; -import { createWorkbenchPiSession } from './pi-agent.js'; +import { buildWorkbenchMetricsFromTrace } from './metrics.js'; import { runShellCommand } from './process.js'; -import { buildAgentSystemPrompt } from './sandbox.js'; -import { buildWorkbenchTrace, createTraceRecorder } from './trace.js'; -import type { TraceRecorder } from './trace.js'; -import type { WorkbenchGrade, WorkbenchResult, WorkbenchTrace, WorkbenchTraceEntry } from './types.js'; -import { isRecord, writeJsonFile } from './utils.js'; +import type { WorkbenchGrade, WorkbenchResult } from './types.js'; +import { writeJsonFile } from './utils.js'; import { buildWorkbenchEnv, prepareWorkbenchDirectory } from './workspace.js'; export { prepareWorkbenchDirectory } from './workspace.js'; -export { buildAgentSystemPrompt } from './sandbox.js'; - -interface PromptSession { - prompt(prompt: string): Promise; - systemPrompt?: string; - subscribe?: (listener: (event: unknown) => void) => () => void; - dispose?: () => void; - state?: { - messages?: unknown[]; - }; -} - -interface AgentRunnerArgs { - mode: 'agent'; - caseName: string; - model: string; - task: string; - appendSystemPrompt?: string; - timeoutSeconds: number; - workDir: string; - resultsDir: string; - mcpConfigPath?: string; -} interface GradeRunnerArgs { mode: 'grade'; @@ -51,7 +24,7 @@ interface SetupRunnerArgs { workDir: string; } -export type ContainerRunnerArgs = AgentRunnerArgs | GradeRunnerArgs | SetupRunnerArgs; +export type ContainerRunnerArgs = GradeRunnerArgs | SetupRunnerArgs; export function buildContainerWorkbenchEnv(params: { casePath: string; @@ -69,34 +42,8 @@ export function buildContainerWorkbenchEnv(params: { export function parseContainerRunnerArgs(args: string[]): ContainerRunnerArgs { const workDir = getFlagValue(args, '--work'); - const resultsDir = getFlagValue(args, '--results'); - - if (args.includes('--agent')) { - const caseName = getFlagValue(args, '--case-name'); - const model = getFlagValue(args, '--model'); - const taskBase64 = getFlagValue(args, '--task-base64'); - const appendSystemPromptBase64 = getFlagValue(args, '--append-system-prompt-base64'); - const mcpConfigPath = getFlagValue(args, '--mcp-config'); - const timeoutSeconds = Number(getFlagValue(args, '--timeout-seconds')); - if (!caseName || !model || !taskBase64 || !Number.isFinite(timeoutSeconds) || timeoutSeconds <= 0 || !workDir || !resultsDir) { - throw new Error('Usage: container-runner --agent --case-name --model --task-base64 --timeout-seconds --work --results '); - } - return { - mode: 'agent', - caseName, - model, - task: Buffer.from(taskBase64, 'base64').toString('utf-8'), - appendSystemPrompt: appendSystemPromptBase64 - ? Buffer.from(appendSystemPromptBase64, 'base64').toString('utf-8') - : undefined, - timeoutSeconds, - workDir, - resultsDir, - mcpConfigPath, - }; - } - const casePath = getFlagValue(args, '--case'); + if (args.includes('--setup')) { if (!casePath || !workDir) { throw new Error('Usage: container-runner --setup --case --work '); @@ -104,11 +51,15 @@ export function parseContainerRunnerArgs(args: string[]): ContainerRunnerArgs { return { mode: 'setup', casePath, workDir }; } - if (!args.includes('--grade') || !casePath || !workDir || !resultsDir) { - throw new Error('Usage: container-runner --agent ... or --setup --case --work or --grade --case --work --results '); + if (args.includes('--grade')) { + const resultsDir = getFlagValue(args, '--results'); + if (!casePath || !workDir || !resultsDir) { + throw new Error('Usage: container-runner --grade --case --work --results '); + } + return { mode: 'grade', casePath, workDir, resultsDir }; } - return { mode: 'grade', casePath, workDir, resultsDir }; + throw new Error('container-runner: expected --setup or --grade'); } function getFlagValue(args: string[], flag: string): string | undefined { @@ -125,94 +76,6 @@ function getFlagValue(args: string[], flag: string): string | undefined { return value; } -export async function runAgentPromptWithTimeout( - session: PromptSession, - prompt: string, - timeoutSeconds: number, -): Promise { - let timeout: NodeJS.Timeout | undefined; - try { - await Promise.race([ - session.prompt(prompt), - new Promise((_, reject) => { - timeout = setTimeout(() => { - reject(new Error(`Agent timed out after ${timeoutSeconds} seconds`)); - }, timeoutSeconds * 1000); - }), - ]); - } finally { - if (timeout) { - clearTimeout(timeout); - } - } - - const messages = session.state?.messages ?? []; - const lastMessage = messages[messages.length - 1]; - if (!isRecord(lastMessage) || lastMessage.role !== 'assistant') { - return; - } - - if (lastMessage.stopReason === 'error' || lastMessage.stopReason === 'aborted') { - const errorMessage = typeof lastMessage.errorMessage === 'string' - ? lastMessage.errorMessage - : `Agent request ${lastMessage.stopReason}`; - throw new Error(errorMessage); - } -} - -export function writeBestEffortTrace(params: { - tracePath: string; - caseName?: string; - model?: string; - startedAt?: string; - endedAt?: string; - session?: PromptSession; - recorder?: TraceRecorder; -}): boolean { - const messages = params.session?.state?.messages; - if (!params.caseName || !params.model || !params.startedAt) { - return false; - } - - if (params.recorder && params.recorder.events.length > 0) { - writeTraceFile(params.tracePath, params.recorder.toTrace({ - caseName: params.caseName, - model: params.model, - startedAt: params.startedAt, - endedAt: params.endedAt ?? new Date().toISOString(), - messages: messages ?? [], - })); - return true; - } - - if (!messages) { - return false; - } - - writeTraceFile(params.tracePath, buildWorkbenchTrace({ - caseName: params.caseName, - model: params.model, - startedAt: params.startedAt, - endedAt: params.endedAt ?? new Date().toISOString(), - messages, - })); - return true; -} - -export function writeTraceFile(tracePath: string, trace: WorkbenchTrace): void { - const header = { - type: 'trace_start', - schemaVersion: trace.schemaVersion ?? 1, - caseName: trace.caseName, - model: trace.model, - startedAt: trace.startedAt, - endedAt: trace.endedAt, - }; - const lines = [header, ...trace.entries] - .map((entry) => JSON.stringify(entry)); - writeFileSync(tracePath, `${lines.join('\n')}\n`, 'utf-8'); -} - async function runCleanupCommands( commands: string[], opts: { cwd: string; env: NodeJS.ProcessEnv }, @@ -291,212 +154,6 @@ function buildResult(params: { }; } -function readTraceFile(tracePath: string, fallback: Omit): WorkbenchTrace { - try { - const raw = readFileSync(tracePath, 'utf-8'); - const trimmed = raw.trim(); - if (trimmed.startsWith('{')) { - try { - const parsed = JSON.parse(trimmed) as unknown; - if (isRecord(parsed) && Array.isArray(parsed.entries)) { - return parsed as unknown as WorkbenchTrace; - } - } catch { - // Fall through to JSONL parsing. - } - } - - const rows = trimmed.length > 0 - ? trimmed.split(/\r?\n/).flatMap((line) => { - try { - const parsed = JSON.parse(line) as unknown; - return isRecord(parsed) ? [parsed] : []; - } catch { - return []; - } - }) - : []; - const header = rows.find((row) => row.type === 'trace_start'); - const entries = rows.filter(isTraceEntry) as WorkbenchTraceEntry[]; - if (header || entries.length > 0) { - return { - schemaVersion: 1, - caseName: isRecord(header) && typeof header.caseName === 'string' ? header.caseName : fallback.caseName, - model: isRecord(header) && typeof header.model === 'string' ? header.model : fallback.model, - startedAt: isRecord(header) && typeof header.startedAt === 'string' ? header.startedAt : fallback.startedAt, - endedAt: isRecord(header) && typeof header.endedAt === 'string' ? header.endedAt : fallback.endedAt, - entries, - }; - } - } catch { - // Grade results should still be written if trace persistence failed. - } - - return { ...fallback, entries: [] }; -} - -function isTraceEntry(value: Record): boolean { - return value.type === 'message' || value.type === 'tool_call' || value.type === 'tool_result'; -} - -function summarizeContent(content: unknown): string | undefined { - if (!Array.isArray(content)) { - return undefined; - } - - const text = content - .flatMap((item) => isRecord(item) && typeof item.text === 'string' ? [item.text] : []) - .join('\n') - .replace(/\s+/g, ' ') - .trim(); - return text.length > 160 ? `${text.slice(0, 157)}...` : text || undefined; -} - -function logAgentEvent(event: unknown): void { - if (!isRecord(event) || typeof event.type !== 'string') { - return; - } - - if (event.type === 'message_end' && isRecord(event.message)) { - const role = typeof event.message.role === 'string' ? event.message.role : 'unknown'; - const text = summarizeContent(event.message.content); - console.log(`[agent:${event.type}] ${role}${text ? `: ${text}` : ''}`); - return; - } - - if (event.type === 'tool_execution_start') { - const name = typeof event.toolName === 'string' ? event.toolName : 'unknown'; - const args = event.args === undefined ? '' : ` ${JSON.stringify(event.args)}`; - console.log(`[agent:${event.type}] ${name}${args}`); - return; - } - - if (event.type === 'tool_execution_end') { - const name = typeof event.toolName === 'string' ? event.toolName : 'unknown'; - const status = event.isError === true ? 'error' : 'ok'; - console.log(`[agent:${event.type}] ${name} ${status}`); - return; - } - - if (event.type === 'turn_start' || event.type === 'turn_end' || event.type === 'agent_start' || event.type === 'agent_end') { - console.log(`[agent:${event.type}]`); - } -} - -function logAgentSystemPrompt(systemPrompt: string): void { - console.log('[agent:system_prompt_start]'); - console.log(systemPrompt); - console.log('[agent:system_prompt_end]'); -} - -async function runAgentMode(parsed: AgentRunnerArgs): Promise { - const resultPath = join(parsed.resultsDir, 'result.json'); - const tracePath = join(parsed.resultsDir, 'trace.jsonl'); - let session: PromptSession | undefined; - let recorder: TraceRecorder | undefined; - let startedAt: string | undefined; - const previousWork = process.env.WORK; - const previousResults = process.env.RESULTS; - const previousMcporterConfig = process.env.MCPORTER_CONFIG; - - mkdirSync(parsed.resultsDir, { recursive: true }); - process.env.WORK = parsed.workDir; - process.env.RESULTS = parsed.resultsDir; - if (parsed.mcpConfigPath) { - process.env.MCPORTER_CONFIG = parsed.mcpConfigPath; - } else { - delete process.env.MCPORTER_CONFIG; - } - - try { - try { - startedAt = new Date().toISOString(); - const created = await createWorkbenchPiSession({ - cwd: parsed.workDir, - modelRef: parsed.model, - apiKeyEnv: 'OPENROUTER_API_KEY', - appendSystemPrompt: parsed.appendSystemPrompt, - mcpConfigPath: parsed.mcpConfigPath, - }); - session = created.session as PromptSession; - const systemPrompt = typeof session.systemPrompt === 'string' - ? session.systemPrompt - : buildAgentSystemPrompt(); - logAgentSystemPrompt(systemPrompt); - recorder = createTraceRecorder(); - const unsubscribe = session.subscribe?.((event) => { - recorder?.record(event); - logAgentEvent(event); - }); - - try { - await runAgentPromptWithTimeout(session, parsed.task, parsed.timeoutSeconds); - } finally { - unsubscribe?.(); - } - - const endedAt = new Date().toISOString(); - const trace = recorder.toTrace({ - caseName: parsed.caseName, - model: parsed.model, - startedAt, - endedAt, - messages: session.state?.messages ?? [], - }); - trace.entries.unshift({ - type: 'message', - role: 'system', - text: systemPrompt, - timestamp: startedAt, - }); - - writeTraceFile(tracePath, trace); - return 0; - } catch (error) { - const endedAt = new Date().toISOString(); - try { - writeBestEffortTrace({ - tracePath, - caseName: parsed.caseName, - model: parsed.model, - startedAt, - endedAt, - session, - recorder, - }); - } catch { - // Fatal result writing is more important than partial trace persistence. - } - writeJsonFile(resultPath, { - caseName: parsed.caseName, - model: parsed.model, - endedAt, - ...buildFatalGrade(error), - error: error instanceof Error ? error.message : String(error), - }); - return 1; - } - } finally { - if (previousWork === undefined) { - delete process.env.WORK; - } else { - process.env.WORK = previousWork; - } - - if (previousResults === undefined) { - delete process.env.RESULTS; - } else { - process.env.RESULTS = previousResults; - } - - if (previousMcporterConfig === undefined) { - delete process.env.MCPORTER_CONFIG; - } else { - process.env.MCPORTER_CONFIG = previousMcporterConfig; - } - } -} - async function runSetupMode(parsed: SetupRunnerArgs): Promise { const resolved = loadWorkbenchCase(parsed.casePath); const env = buildContainerWorkbenchEnv({ @@ -518,13 +175,7 @@ async function runGradeMode(parsed: GradeRunnerArgs): Promise { const cleanupErrorPath = join(parsed.resultsDir, 'cleanup-error.txt'); const resolved = loadWorkbenchCase(parsed.casePath); const env = buildContainerWorkbenchEnv(parsed); - const now = new Date().toISOString(); - const trace = readTraceFile(tracePath, { - caseName: resolved.name, - model: resolved.model, - startedAt: now, - endedAt: now, - }); + const startedAt = new Date().toISOString(); try { const grade = await runGraderCommands(resolved.graders, { @@ -535,11 +186,11 @@ async function runGradeMode(parsed: GradeRunnerArgs): Promise { const result = buildResult({ caseName: resolved.name, model: resolved.model, - startedAt: trace.startedAt, + startedAt, endedAt: new Date().toISOString(), grade: { ...grade, - metrics: buildWorkbenchMetrics(trace), + metrics: buildWorkbenchMetricsFromTrace(tracePath), }, }); @@ -560,12 +211,7 @@ async function runGradeMode(parsed: GradeRunnerArgs): Promise { export async function runContainerWorkbenchCase(args: string[]): Promise { const parsed = parseContainerRunnerArgs(args); - if (parsed.mode === 'agent') { - return runAgentMode(parsed); - } - if (parsed.mode === 'setup') { - return runSetupMode(parsed); - } + if (parsed.mode === 'setup') return runSetupMode(parsed); return runGradeMode(parsed); } diff --git a/src/workbench/docker-runner.ts b/src/workbench/docker-runner.ts index 292ce6a..564a4c6 100644 --- a/src/workbench/docker-runner.ts +++ b/src/workbench/docker-runner.ts @@ -1,20 +1,27 @@ -import { cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { spawn } from 'node:child_process'; +import { chmodSync, cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { dirname, join, resolve } from 'node:path'; import { fileURLToPath } from 'node:url'; +import { PROTOCOL_VERSION } from '@agentclientprotocol/sdk'; import { stringify as stringifyYaml } from 'yaml'; +import { resolveAuth, resolveContainerPath } from './acp/auth.js'; +import { createWorkbenchClient } from './acp/client.js'; +import { writeMcpConfig } from './acp/mcp-config-writer.js'; +import { computeSkillMount, dockerMountFlag } from './acp/skill-deploy.js'; +import { createTraceRecorder } from './acp/trace-recorder.js'; +import { createDockerExecStream } from './acp/transport.js'; +import { resolveAgent } from './agents/registry.js'; import { loadWorkbenchCase } from './case-loader.js'; -import { MCPORTER_CONFIG_CONTAINER_PATH, writeWorkbenchMcpConfig } from './mcp/index.js'; +import { writeWorkbenchMcpConfig } from './mcp/index.js'; import { runShellCommand } from './process.js'; import type { ResolvedWorkbenchCase, WorkbenchCaseConfig } from './types.js'; import { timestampSlug } from './utils.js'; import { prepareWorkbenchDirectory } from './workspace.js'; -const DEFAULT_WORKBENCH_IMAGE = 'skill-optimizer-workbench:local'; -const AGENT_RESULTS_DIR = '/tmp/workbench-results'; -const AGENT_PATH = '/work/bin:/app/node_modules/.bin:/work/.venv/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin'; +const DEFAULT_WORKBENCH_IMAGE = 'skill-optimizer-agent:local'; export function packageRootFromModuleUrl(moduleUrl: string): string { return dirname(dirname(dirname(fileURLToPath(moduleUrl)))); @@ -75,174 +82,6 @@ function dockerCacheEnvFlags(): string[] { ]; } -export function buildDockerAgentCommand(params: { - image: string; - containerName: string; - workDir: string; - caseName: string; - model: string; - task: string; - appendSystemPrompt?: string; - mcpConfigPath?: string; - networkName?: string; - timeoutSeconds: number; - envNames: string[]; -}): string { - const envArgs = params.envNames.map((name) => `-e ${name}`).join(' '); - const mcpEnvArg = params.mcpConfigPath ? `-e MCPORTER_CONFIG=${params.mcpConfigPath}` : ''; - const networkArg = params.networkName ? `--network ${shellQuote(params.networkName)}` : ''; - const taskBase64 = Buffer.from(params.task, 'utf-8').toString('base64'); - const appendSystemPromptBase64 = params.appendSystemPrompt - ? Buffer.from(params.appendSystemPrompt, 'utf-8').toString('base64') - : undefined; - return [ - 'docker run', - `--name ${shellQuote(params.containerName)}`, - ...dockerSandboxFlags(), - '--workdir /work', - `-e PATH=${AGENT_PATH}`, - ...dockerCacheEnvFlags(), - networkArg, - `-v ${shellQuote(`${params.workDir}:/work:rw`)}`, - envArgs, - mcpEnvArg, - shellQuote(params.image), - '--agent', - '--work /work', - `--results ${AGENT_RESULTS_DIR}`, - `--case-name ${shellQuote(params.caseName)}`, - `--model ${shellQuote(params.model)}`, - `--timeout-seconds ${params.timeoutSeconds}`, - `--task-base64 ${shellQuote(taskBase64)}`, - params.mcpConfigPath ? `--mcp-config ${shellQuote(params.mcpConfigPath)}` : '', - appendSystemPromptBase64 - ? `--append-system-prompt-base64 ${shellQuote(appendSystemPromptBase64)}` - : '', - ].filter(Boolean).join(' '); -} - -export function buildDockerMcpServiceCommand(params: { - image: string; - containerName: string; - networkName: string; - alias: string; - mcpDir: string; - command: string; - args: string[]; -}): string { - const serviceCommand = [params.command, ...params.args].map(shellQuote).join(' '); - return [ - 'docker run -d', - `--name ${shellQuote(params.containerName)}`, - ...dockerSandboxFlags(), - `--network ${shellQuote(params.networkName)}`, - `--network-alias ${shellQuote(params.alias)}`, - '--workdir /mcp', - `-v ${shellQuote(`${params.mcpDir}:/mcp:ro`)}`, - '--entrypoint /bin/sh', - shellQuote(params.image), - '-lc', - shellQuote(serviceCommand), - ].filter(Boolean).join(' '); -} - -export function buildDockerMcpServiceProbeCommand(params: { - image: string; - networkName: string; - workDir: string; - serverName: string; -}): string { - const probeCommand = [ - '/app/node_modules/.bin/mcporter', - '--config /work/mcporter.json', - '--root /work', - 'list', - shellQuote(params.serverName), - '--schema', - ].join(' '); - return [ - 'docker run --rm', - ...dockerSandboxFlags(), - `--network ${shellQuote(params.networkName)}`, - '--workdir /work', - `-v ${shellQuote(`${params.workDir}:/work:rw`)}`, - '--entrypoint /bin/sh', - shellQuote(params.image), - '-lc', - shellQuote(probeCommand), - ].filter(Boolean).join(' '); -} - -export function buildDockerSetupCommand(params: { - image: string; - caseDir: string; - workDir: string; - envNames: string[]; -}): string { - const envArgs = params.envNames.map((name) => `-e ${name}`).join(' '); - return [ - 'docker run --rm', - ...dockerSandboxFlags(), - '--workdir /work', - ...dockerCacheEnvFlags(), - `-v ${shellQuote(`${params.caseDir}:/case:ro`)}`, - `-v ${shellQuote(`${params.workDir}:/work:rw`)}`, - envArgs, - shellQuote(params.image), - '--setup', - '--case /case/case.yml', - '--work /work', - ].filter(Boolean).join(' '); -} - -function agentContainerName(tempDir: string): string { - return `skill-optimizer-agent-${tempDir.split('/').pop() ?? 'run'}`; -} - -async function copyAgentResults(containerName: string, resultsDir: string, repoRoot: string): Promise { - const copy = await runShellCommand( - `docker cp ${shellQuote(`${containerName}:${AGENT_RESULTS_DIR}/.`)} ${shellQuote(resultsDir)}`, - { cwd: repoRoot }, - ); - - if (copy.exitCode !== 0) { - throw new Error([ - 'Failed to copy agent results from Docker container', - copy.stdout.trim(), - copy.stderr.trim(), - ].filter(Boolean).join('\n\n')); - } -} - -async function removeContainer(containerName: string, repoRoot: string): Promise { - await runShellCommand(`docker rm -f ${shellQuote(containerName)}`, { cwd: repoRoot }); -} - -export function buildDockerGradeCommand(params: { - image: string; - caseDir: string; - workDir: string; - resultsDir: string; - envNames: string[]; -}): string { - const envArgs = params.envNames.map((name) => `-e ${name}`).join(' '); - return [ - 'docker run --rm', - ...dockerSandboxFlags(), - '--workdir /work', - ...dockerCacheEnvFlags(), - `-v ${shellQuote(`${params.caseDir}:/case:ro`)}`, - `-v ${shellQuote(`${params.workDir}:/work:rw`)}`, - `-v ${shellQuote(`${params.resultsDir}:/results:rw`)}`, - envArgs, - shellQuote(params.image), - '--grade', - '--case /case/case.yml', - '--work /work', - '--results /results', - ].filter(Boolean).join(' '); -} - function buildBundledCaseFile(params: { source: ReturnType; modelOverride?: string; @@ -252,6 +91,7 @@ function buildBundledCaseFile(params: { references: './references', task: params.source.task, graders: params.source.graders.map((grader) => ({ ...grader })), + agent: params.source.agent, model: params.modelOverride ?? params.source.model, timeoutSeconds: params.source.timeoutSeconds, }; @@ -293,85 +133,19 @@ function copyCaseSupportDir(sourceCaseDir: string, bundledCaseDir: string, name: cpSync(sourceDir, destinationDir, { recursive: true }); } -function copyCaseSupportDirs(sourceCaseDir: string, bundledCaseDir: string): void { - for (const name of ['checks', 'fixtures', 'bin', 'workspace', 'mcp']) { - copyCaseSupportDir(sourceCaseDir, bundledCaseDir, name); - } -} - -function mcpNetworkName(tempDir: string): string { - return `skill-optimizer-mcp-${tempDir.split('/').pop() ?? 'run'}`; -} - -async function createDockerNetwork(networkName: string, repoRoot: string): Promise { - const create = await runShellCommand(`docker network create ${shellQuote(networkName)}`, { cwd: repoRoot }); - if (create.exitCode !== 0) { - throw new Error(['Failed to create MCP Docker network', create.stdout.trim(), create.stderr.trim()].filter(Boolean).join('\n\n')); - } -} - -async function removeDockerNetwork(networkName: string | undefined, repoRoot: string): Promise { - if (!networkName) return; - await runShellCommand(`docker network rm ${shellQuote(networkName)}`, { cwd: repoRoot }); -} - -export async function startMcpServices(params: { - image: string; - networkName: string; - caseDir: string; - tempDir: string; - services: ResolvedWorkbenchCase['mcpServices']; - repoRoot: string; - startedContainers?: string[]; - runCommand?: typeof runShellCommand; -}): Promise { - const containerNames = params.startedContainers ?? []; - const runCommand = params.runCommand ?? runShellCommand; - for (const [name, service] of Object.entries(params.services)) { - const containerName = `${mcpNetworkName(params.tempDir)}-${name}`; - console.log(`Starting MCP service ${name}...`); - const command = buildDockerMcpServiceCommand({ - image: params.image, - containerName, - networkName: params.networkName, - alias: name, - mcpDir: join(params.caseDir, 'mcp'), - command: service.command, - args: service.args, - }); - const run = await runCommand(command, { cwd: params.repoRoot }); - if (run.exitCode !== 0) { - throw new Error([`Failed to start MCP service ${name}`, run.stdout.trim(), run.stderr.trim()].filter(Boolean).join('\n\n')); - } - containerNames.push(containerName); - } - return containerNames; -} +// `references/` is handled separately via copyDirectoryContents; skipping it here avoids +// double-copy. Hidden dirs (e.g., `.results`, `.skill-optimizer`) are bench output, not +// case content. `node_modules/` is excluded for cost reasons; nothing in a case should +// depend on it at grade time. +const EXCLUDED_CASE_SUBDIRS = new Set(['references', 'node_modules']); -async function waitForMcpServices(params: { - image: string; - networkName: string; - workDir: string; - services: ResolvedWorkbenchCase['mcpServices']; - repoRoot: string; -}): Promise { - for (const name of Object.keys(params.services)) { - console.log(`Waiting for MCP service ${name}...`); - const command = buildDockerMcpServiceProbeCommand({ - image: params.image, - networkName: params.networkName, - workDir: params.workDir, - serverName: name, - }); - const probe = await runShellCommand(command, { cwd: params.repoRoot, timeoutSeconds: 30 }); - if (probe.exitCode !== 0) { - throw new Error([ - `MCP service ${name} did not become ready`, - probe.stdout.trim(), - probe.stderr.trim(), - ].filter(Boolean).join('\n\n')); - } - console.log(`MCP service ${name} ready.`); +function copyCaseSupportDirs(sourceCaseDir: string, bundledCaseDir: string): void { + if (!existsSync(sourceCaseDir)) return; + for (const entry of readdirSync(sourceCaseDir, { withFileTypes: true })) { + if (!entry.isDirectory()) continue; + if (entry.name.startsWith('.')) continue; + if (EXCLUDED_CASE_SUBDIRS.has(entry.name)) continue; + copyCaseSupportDir(sourceCaseDir, bundledCaseDir, entry.name); } } @@ -408,6 +182,8 @@ export function prepareDockerWorkbenchRun( mkdirSync(referencesDir, { recursive: true }); mkdirSync(workDir, { recursive: true }); mkdirSync(resultsDir, { recursive: true }); + chmodSync(workDir, 0o777); + chmodSync(resultsDir, 0o777); copyDirectoryContents(resolvedCase.referencesDir, referencesDir); copyCaseSupportDirs(resolvedCase.configDir, caseDir); @@ -434,7 +210,15 @@ export function prepareDockerWorkbenchRun( resultPath: join(resultsDir, 'result.json'), tracePath: join(resultsDir, 'trace.jsonl'), ...(mcpConfigPath ? { mcpConfigPath } : {}), - cleanup: () => rmSync(tempDir, { recursive: true, force: true }), + cleanup: () => { + try { + rmSync(tempDir, { recursive: true, force: true }); + } catch (error) { + // The container (uid 10001) may write subdirs (.cache, .venv) that the host user + // cannot delete. Don't let cleanup failures kill the run; tmpfiles.d will sweep /tmp later. + console.warn(`workbench: could not remove ${tempDir}: ${error instanceof Error ? error.message : String(error)}`); + } + }, }; } @@ -444,7 +228,7 @@ async function ensureDockerImage(image: string, repoRoot: string): Promise return; } - const dockerfilePath = join(repoRoot, 'docker', 'workbench-runner.Dockerfile'); + const dockerfilePath = join(repoRoot, 'docker', 'skill-optimizer-agent.Dockerfile'); if (!existsSync(dockerfilePath)) { throw new Error(`Dockerfile not found: ${dockerfilePath}`); } @@ -513,108 +297,294 @@ export async function runDockerWorkbenchCase( const image = options.image ?? DEFAULT_WORKBENCH_IMAGE; const resolvedCase = resolveDockerWorkbenchCase(options); const prepared = prepareDockerWorkbenchRun({ ...options, case: resolvedCase }); - const containerName = agentContainerName(prepared.tempDir); - const networkName = Object.keys(resolvedCase.mcpServices).length > 0 ? mcpNetworkName(prepared.tempDir) : undefined; - let mcpServiceContainers: string[] = []; try { await ensureDockerImage(image, repoRoot); - const envNames = resolvedCase.env - .filter((name) => process.env[name] !== undefined) - .map((name) => name); + const skillMountParams = resolvedCase.skillUnderTest + ? { + hostSkillDir: resolvedCase.skillUnderTest.hostPath, + skillSlug: resolvedCase.skillUnderTest.slug, + } + : null; - if (resolvedCase.setup.length > 0) { - const setupCommand = buildDockerSetupCommand({ - image, - caseDir: prepared.caseDir, - workDir: prepared.workDir, - envNames, + await runOneAcpTrial({ + resolvedCase, + tempDir: prepared.tempDir, + workDir: prepared.workDir, + caseDir: prepared.caseDir, + resultsDir: prepared.resultsDir, + agentName: resolvedCase.agent, + model: options.model ?? resolvedCase.model, + image, + skillMountParams, + repoRoot, + timeoutSeconds: resolvedCase.timeoutSeconds, + }); + + return copyWorkspaceIfRequested(prepared, options.keepWorkspace); + } catch (error) { + // If runOneAcpTrial threw before any result.json was written, persist a fatal record. + if (!existsSync(prepared.resultPath)) { + writeFatalResult({ + resultPath: prepared.resultPath, + caseName: resolvedCase.name, + model: options.model ?? resolvedCase.model, + evidence: [error instanceof Error ? error.message : String(error)], }); - const setupRun = await runShellCommand(setupCommand, { cwd: repoRoot }); + } + return copyWorkspaceIfRequested(prepared, true); + } finally { + prepared.cleanup(); + } +} + +export interface RunOneAcpTrialOptions { + resolvedCase: ResolvedWorkbenchCase; + tempDir: string; + workDir: string; + caseDir: string; + resultsDir: string; + agentName: string; + model: string; + image: string; + skillMountParams: { hostSkillDir: string; skillSlug: string } | null; + repoRoot: string; + timeoutSeconds: number; +} + +export interface RunOneAcpTrialResult { + pass: boolean; + tracePath: string; + resultPath: string; +} + +export async function runOneAcpTrial(params: RunOneAcpTrialOptions): Promise { + const agent = resolveAgent(params.agentName); + const auth = resolveAuth(agent, { env: process.env }); + + // Stage host-side $HOME for the agent: auth files + MCP config land here. + const agentHomeHost = join(params.tempDir, 'agent-home'); + mkdirSync(agentHomeHost, { recursive: true }); + // Agents that bake runtime state into $HOME (Anthropic SDK writes debug logs + // to ~/.claude/debug, pi to ~/.pi, etc.) need those dirs to exist AND be + // writable for the container's agent user (uid 10001), which is not the + // host user. Pre-create the declared homeDirs as world-writable. + for (const sub of agent.homeDirs) { + const dir = join(agentHomeHost, sub); + mkdirSync(dir, { recursive: true }); + chmodSync(dir, 0o777); + } + // The root staging dir also needs to be writable (the SDK may create files + // at the home root, not just under known subdirs). + chmodSync(agentHomeHost, 0o777); + + if (auth.mode === 'subscription') { + for (const f of auth.files) { + const targetPath = resolveContainerPath(f.containerPath, agentHomeHost); + mkdirSync(dirname(targetPath), { recursive: true }); + chmodSync(dirname(targetPath), 0o777); + cpSync(f.hostPath, targetPath); + // The container's agent user (uid 10001) won't match the host user that + // owns ~/.claude/.credentials.json. Subscription auth files on the host + // are typically mode 600. Loosen the staged copies to 644 so the + // in-container agent can read them. + chmodSync(targetPath, 0o644); + } + } + + writeMcpConfig({ + agent, + caseConfig: { + mcpServers: params.resolvedCase.mcpServers as Parameters[0]['caseConfig']['mcpServers'], + }, + agentHomeOnHost: agentHomeHost, + }); + + // Skill mount (optional) + let skillMountFlag = ''; + if (params.skillMountParams) { + const mount = computeSkillMount({ + agent, + skillSlug: params.skillMountParams.skillSlug, + hostSkillDir: params.skillMountParams.hostSkillDir, + agentHome: '/home/agent', + }); + skillMountFlag = dockerMountFlag(mount); + } + + // Env passthrough + const envFlags: string[] = []; + if (auth.mode === 'env') { + for (const name of auth.envNames) { + if (process.env[name] !== undefined) envFlags.push(`-e ${name}`); + } + } else { + // Subscription mode may carry env values extracted from the credentials + // file (e.g., OAuth bearer → ANTHROPIC_AUTH_TOKEN). Pass via `-e KEY=VALUE`. + for (const [name, value] of Object.entries(auth.envOverrides)) { + envFlags.push(`-e ${name}=${shellQuote(value)}`); + } + } + for (const name of params.resolvedCase.env) { + if (process.env[name] !== undefined) envFlags.push(`-e ${name}`); + } + + // Start the detached idle container + const containerName = `skill-opt-trial-${params.tempDir.split('/').pop() ?? 'run'}`; + const runCmd = [ + 'docker run -d', + `--name ${shellQuote(containerName)}`, + ...dockerSandboxFlags(), + ...dockerCacheEnvFlags(), + `-v ${shellQuote(`${params.workDir}:/work:rw`)}`, + `-v ${shellQuote(`${params.caseDir}:/case:ro`)}`, + `-v ${shellQuote(`${params.resultsDir}:/results:rw`)}`, + `-v ${shellQuote(`${agentHomeHost}:/home/agent:rw`)}`, + skillMountFlag, + ...envFlags, + '--workdir /work', + '--entrypoint sleep', + shellQuote(params.image), + 'infinity', + ].filter(Boolean).join(' '); + const runResult = await runShellCommand(runCmd, { cwd: params.repoRoot }); + if (runResult.exitCode !== 0) { + throw new Error(`Failed to start trial container: ${runResult.stderr.trim() || runResult.stdout.trim()}`); + } + + const tracePath = join(params.resultsDir, 'trace.jsonl'); + const resultPath = join(params.resultsDir, 'result.json'); + const startedAt = new Date().toISOString(); + const recorder = createTraceRecorder({ + tracePath, + header: { + caseName: params.resolvedCase.name, + agent: agent.name, + model: params.model, + startedAt, + }, + }); + + try { + // Case setup, if any + if (params.resolvedCase.setup.length > 0) { + const setupScript = params.resolvedCase.setup.join(' && '); + const setupCmd = `docker exec ${shellQuote(containerName)} sh -c ${shellQuote(setupScript)}`; + const setupRun = await runShellCommand(setupCmd, { cwd: params.repoRoot }); if (setupRun.exitCode !== 0) { - writeFatalResult({ - resultPath: prepared.resultPath, - caseName: resolvedCase.name, - model: options.model ?? resolvedCase.model, - evidence: [ - 'setup failed', - setupRun.stdout.trim(), - setupRun.stderr.trim(), - ].filter(Boolean), + throw new Error(`Case setup failed: ${setupRun.stderr.trim() || setupRun.stdout.trim()}`); + } + } + + // Spawn the agent CLI in the container, talking ACP over stdio + const child = spawn('docker', [ + 'exec', '-i', + containerName, + 'sh', '-c', agent.launchCmd, + ], { stdio: ['pipe', 'pipe', 'pipe'] }); + + const stream = createDockerExecStream(child as Parameters[0]); + const client = createWorkbenchClient({ + stream, + onSessionUpdate: (notification) => { + recorder.recordRaw({ + jsonrpc: '2.0', + method: 'session/update', + params: notification, + }); + }, + }); + + // ACP handshake + await client.connection.initialize({ + protocolVersion: PROTOCOL_VERSION, + clientCapabilities: {}, + }); + const session = await client.connection.newSession({ + cwd: '/work', + mcpServers: [], + }); + + // Apply the requested model if the agent supports per-session model selection. + // Each agent has its own modelId conventions (claude: "opus[1m]" / "sonnet[1m]" / "haiku" / "default"; + // pi-acp / openrouter: "openrouter//"; etc.). The case author is + // responsible for using a string the agent recognizes. + if (agent.supportsAcpSetModel && params.model && params.model !== 'default') { + try { + await client.connection.unstable_setSessionModel({ + sessionId: session.sessionId, + modelId: params.model, }); - return copyWorkspaceIfRequested(prepared, true); + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + throw new Error( + `Agent ${agent.name} rejected model "${params.model}": ${msg}. ` + + `Check the registry's acpModelFormat for valid IDs.`, + ); } } + const promptPromise = client.connection.prompt({ + sessionId: session.sessionId, + prompt: [{ type: 'text', text: params.resolvedCase.task }], + }); - if (networkName) { - await createDockerNetwork(networkName, repoRoot); - mcpServiceContainers = await startMcpServices({ - image, - networkName, - caseDir: prepared.caseDir, - tempDir: prepared.tempDir, - services: resolvedCase.mcpServices, - repoRoot, - startedContainers: mcpServiceContainers, - }); - await waitForMcpServices({ - image, - networkName, - workDir: prepared.workDir, - services: resolvedCase.mcpServices, - repoRoot, + let promptOutcome: { ok: true; result: unknown } | { ok: false; error: Error }; + let timeoutHandle: NodeJS.Timeout | undefined; + try { + const timeoutPromise = new Promise((_resolve, reject) => { + timeoutHandle = setTimeout( + () => reject(new Error(`Prompt timeout after ${params.timeoutSeconds}s`)), + params.timeoutSeconds * 1000, + ); }); + const result = await Promise.race([promptPromise, timeoutPromise]); + promptOutcome = { ok: true, result }; + } catch (error) { + promptOutcome = { ok: false, error: error instanceof Error ? error : new Error(String(error)) }; + } finally { + if (timeoutHandle) clearTimeout(timeoutHandle); } - const agentCommand = buildDockerAgentCommand({ - image, - containerName, - workDir: prepared.workDir, - caseName: resolvedCase.name, - model: options.model ?? resolvedCase.model, - task: resolvedCase.task, - appendSystemPrompt: options.appendSystemPrompt, - mcpConfigPath: prepared.mcpConfigPath ? MCPORTER_CONFIG_CONTAINER_PATH : undefined, - networkName, - timeoutSeconds: resolvedCase.timeoutSeconds, - envNames, - }); - const agentRun = await runShellCommand(agentCommand, { cwd: repoRoot }); - await copyAgentResults(containerName, prepared.resultsDir, repoRoot); - - if (agentRun.exitCode !== 0) { - if (!existsSync(prepared.resultPath)) { - throw new Error([ - 'Docker agent run failed', - agentRun.stdout.trim(), - agentRun.stderr.trim(), - ].filter(Boolean).join('\n\n')); - } + if (promptOutcome.ok) { + recorder.recordRaw({ jsonrpc: '2.0', id: 'final', result: promptOutcome.result }); } else { - const gradeCommand = buildDockerGradeCommand({ - image, - caseDir: prepared.caseDir, - workDir: prepared.workDir, - resultsDir: prepared.resultsDir, - envNames, - }); - const gradeRun = await runShellCommand(gradeCommand, { cwd: repoRoot }); - - if (gradeRun.exitCode !== 0 && !existsSync(prepared.resultPath)) { - throw new Error([ - 'Docker grade run failed', - gradeRun.stdout.trim(), - gradeRun.stderr.trim(), - ].filter(Boolean).join('\n\n')); - } + recorder.recordRaw({ jsonrpc: '2.0', id: 'final', error: { message: promptOutcome.error.message } }); } - return copyWorkspaceIfRequested(prepared, options.keepWorkspace); + await client.close(); + recorder.finalize(new Date().toISOString()); + + // Run graders inside the same container + const gradeCmd = `docker exec ${shellQuote(containerName)} node /app/dist/workbench/container-runner.js --grade --case /case/case.yml --work /work --results /results`; + const gradeRun = await runShellCommand(gradeCmd, { cwd: params.repoRoot }); + if (gradeRun.exitCode !== 0 && !existsSync(resultPath)) { + throw new Error(`Grade run failed: ${gradeRun.stderr.trim() || gradeRun.stdout.trim()}`); + } + + // Copy agent-internal logs out (non-fatal) + const internalDir = join(params.resultsDir, 'agent-internal'); + mkdirSync(internalDir, { recursive: true }); + for (const dotDir of ['.claude', '.codex', '.gemini', '.opencode', '.pi']) { + await runShellCommand( + `docker cp ${shellQuote(`${containerName}:/home/agent/${dotDir}`)} ${shellQuote(internalDir)} 2>/dev/null || true`, + { cwd: params.repoRoot }, + ); + } + + const pass = readTrialPass(resultPath) ?? false; + return { pass, tracePath, resultPath }; } finally { - await removeContainer(containerName, repoRoot); - await Promise.all(mcpServiceContainers.map((name) => removeContainer(name, repoRoot))); - await removeDockerNetwork(networkName, repoRoot); - prepared.cleanup(); + // Loosen permissions on host-mounted dirs so the post-trial host cleanup + // (rmSync on tempDir) can delete subdirs the container created as uid 10001 + // (e.g. agent-home/.cache, work/.venv). Without this, the host user can + // unlink the top-level mount points but not recurse into 10001-owned + // children, producing EACCES warnings. Best-effort — run as container root, + // ignore failure if the container is already gone. + await runShellCommand( + `docker exec --user 0 ${shellQuote(containerName)} chmod -R 0777 /home/agent /work 2>/dev/null || true`, + { cwd: params.repoRoot }, + ); + await runShellCommand(`docker rm -f ${shellQuote(containerName)}`, { cwd: params.repoRoot }); } } diff --git a/src/workbench/index.ts b/src/workbench/index.ts index 6376574..8dc809e 100644 --- a/src/workbench/index.ts +++ b/src/workbench/index.ts @@ -5,7 +5,6 @@ export * from './check-runner.js'; export * from './trace.js'; export * from './models.js'; export * from './mcp/index.js'; -export * from './pi-agent.js'; export * from './docker-runner.js'; export * from './run-case.js'; export * from './suite-loader.js'; diff --git a/src/workbench/metrics.ts b/src/workbench/metrics.ts index 6c37c67..a204b1c 100644 --- a/src/workbench/metrics.ts +++ b/src/workbench/metrics.ts @@ -1,135 +1,50 @@ -import type { - WorkbenchMetrics, - WorkbenchResult, - WorkbenchTrace, - WorkbenchTraceEntry, - WorkbenchTrialSummaryFile, -} from './types.js'; +import { readFileSync } from 'node:fs'; -function emptyMetrics(): WorkbenchMetrics { +import { computeMetrics as computeFromTrace, getFailureEvidence, getFinalAssistantMessage, iterToolCalls } from './parse-trace.js'; +import type { WorkbenchMetrics, WorkbenchResult, WorkbenchTrialSummaryFile } from './types.js'; + +function buildMetricsFromJsonl(jsonl: string): WorkbenchMetrics { + const m = computeFromTrace(jsonl); return { - durationMs: 0, - turns: 0, - toolCalls: 0, - toolResults: 0, - bashCalls: 0, - readCalls: 0, - writeCalls: 0, - editCalls: 0, - tokens: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - total: 0, - }, - cost: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - total: 0, - }, + durationMs: m.durationMs, + turns: m.turns, + toolCalls: m.toolCalls, + toolResults: m.toolCalls, // ACP doesn't distinguish; treat each call as one result + bashCalls: m.bashCalls, + readCalls: m.readCalls, + writeCalls: m.writeCalls, + editCalls: m.editCalls, + stopReason: m.stopReason, + tokens: m.tokens, }; } -export function buildWorkbenchMetrics(trace: WorkbenchTrace): WorkbenchMetrics { - const metrics = emptyMetrics(); - const started = Date.parse(trace.startedAt); - const ended = Date.parse(trace.endedAt); - metrics.durationMs = Number.isFinite(started) && Number.isFinite(ended) - ? Math.max(0, ended - started) - : 0; - - for (const entry of trace.entries) { - if (entry.type === 'message') { - metrics.turns += 1; - if (typeof entry.stopReason === 'string') { - metrics.stopReason = entry.stopReason; - } - addUsage(metrics, entry.usage); - continue; - } - - if (entry.type === 'tool_result') { - metrics.toolResults += 1; - continue; - } - - metrics.toolCalls += 1; - if (entry.name === 'bash') metrics.bashCalls += 1; - if (entry.name === 'read') metrics.readCalls += 1; - if (entry.name === 'write') metrics.writeCalls += 1; - if (entry.name === 'edit') metrics.editCalls += 1; - } - - return metrics; +export function buildWorkbenchMetricsFromTrace(tracePath: string): WorkbenchMetrics { + return buildMetricsFromJsonl(readFileSync(tracePath, 'utf-8')); } export function buildTrialSummary(params: { - trace: WorkbenchTrace; + tracePath: string; result: WorkbenchResult; }): WorkbenchTrialSummaryFile { - const failedGraders = params.result.graders - ?.filter((grader) => !grader.pass) - .map((grader) => grader.name) ?? []; - const metrics = params.result.metrics ?? buildWorkbenchMetrics(params.trace); - const terminalMessage = [...params.trace.entries] - .reverse() - .find((entry): entry is Extract => entry.type === 'message' && entry.role === 'assistant'); - + const jsonl = readFileSync(params.tracePath, 'utf-8'); + const metrics = params.result.metrics ?? buildMetricsFromJsonl(jsonl); + const failedGraders = params.result.graders?.filter((g) => !g.pass).map((g) => g.name) ?? []; return { - finalAssistantMessage: terminalMessage?.text, + finalAssistantMessage: getFinalAssistantMessage(jsonl), failedGraders, - evidence: [...params.result.evidence], - bashCommands: extractBashCommands(params.trace), - stopReason: typeof terminalMessage?.stopReason === 'string' ? terminalMessage.stopReason : undefined, - errorMessage: terminalMessage?.errorMessage, + evidence: [...params.result.evidence, ...getFailureEvidence(jsonl)], + bashCommands: extractBashCommands(jsonl), + stopReason: metrics.stopReason, + errorMessage: undefined, metrics, }; } -function extractBashCommands(trace: WorkbenchTrace): string[] { - return trace.entries.flatMap((entry) => { - if (entry.type !== 'tool_call' || entry.name !== 'bash') { - return []; - } - - const args = entry.arguments; - if (!args || typeof args !== 'object' || Array.isArray(args)) { - return []; - } - - const command = (args as Record).command; - return typeof command === 'string' ? [command] : []; - }); -} - -function addUsage(metrics: WorkbenchMetrics, usage: unknown): void { - if (!usage || typeof usage !== 'object' || Array.isArray(usage)) { - return; - } - - const record = usage as Record; - metrics.tokens.input += readNumber(record.input); - metrics.tokens.output += readNumber(record.output); - metrics.tokens.cacheRead += readNumber(record.cacheRead); - metrics.tokens.cacheWrite += readNumber(record.cacheWrite); - metrics.tokens.total += readNumber(record.totalTokens); - - const cost = record.cost; - if (!cost || typeof cost !== 'object' || Array.isArray(cost)) { - return; +function extractBashCommands(jsonl: string): string[] { + const out: string[] = []; + for (const call of iterToolCalls(jsonl)) { + if (call.kind === 'execute' && call.argsText) out.push(call.argsText); } - - const costRecord = cost as Record; - metrics.cost.input += readNumber(costRecord.input); - metrics.cost.output += readNumber(costRecord.output); - metrics.cost.cacheRead += readNumber(costRecord.cacheRead); - metrics.cost.cacheWrite += readNumber(costRecord.cacheWrite); - metrics.cost.total += readNumber(costRecord.total); -} - -function readNumber(value: unknown): number { - return typeof value === 'number' && Number.isFinite(value) ? value : 0; + return out; } diff --git a/src/workbench/parse-trace.ts b/src/workbench/parse-trace.ts new file mode 100644 index 0000000..1447641 --- /dev/null +++ b/src/workbench/parse-trace.ts @@ -0,0 +1,198 @@ +export interface ParsedMessage { + role: 'user' | 'assistant'; + text?: string; + thinking?: string; + timestamp?: string; +} + +export interface ParsedToolCall { + id: string; + kind: string; + title?: string; + status: 'in_progress' | 'completed' | 'failed' | 'pending'; + argsText?: string; + resultText?: string; + isError?: boolean; +} + +export interface ComputedMetrics { + durationMs: number; + turns: number; + toolCalls: number; + bashCalls: number; + readCalls: number; + writeCalls: number; + editCalls: number; + stopReason?: string; + tokens: { input: number; output: number; cacheRead: number; cacheWrite: number; total: number }; +} + +interface TraceHeader { + type: 'trace_start'; + caseName?: string; + agent?: string; + model?: string; + startedAt?: string; + endedAt?: string; +} + +function parseLines(jsonl: string): unknown[] { + return jsonl.split(/\r?\n/).filter(Boolean).map((line) => { + try { return JSON.parse(line); } catch { return null; } + }).filter((x) => x !== null); +} + +function getHeader(rows: unknown[]): TraceHeader | undefined { + for (const row of rows) { + if (typeof row === 'object' && row !== null && (row as any).type === 'trace_start') { + return row as TraceHeader; + } + } + return undefined; +} + +function extractText(content: unknown): string | undefined { + if (!content || typeof content !== 'object') return undefined; + const c = content as any; + if (typeof c.text === 'string') return c.text; + if (c.content && typeof c.content.text === 'string') return c.content.text; + if (Array.isArray(c)) { + return c.map((part: any) => extractText(part)).filter(Boolean).join('\n') || undefined; + } + return undefined; +} + +export function* iterMessages(jsonl: string): IterableIterator { + const rows = parseLines(jsonl); + let assistantText = ''; + let assistantThinking = ''; + + for (const row of rows) { + const r = row as any; + if (r.method !== 'session/update') continue; + const update = r.params?.update; + if (!update) continue; + + switch (update.sessionUpdate) { + case 'agent_message_chunk': + assistantText += extractText(update.content) ?? ''; + break; + case 'agent_thought_chunk': + assistantThinking += extractText(update.content) ?? ''; + break; + case 'user_message_chunk': + yield { role: 'user', text: extractText(update.content) }; + break; + } + } + + if (assistantText || assistantThinking) { + yield { + role: 'assistant', + text: assistantText || undefined, + thinking: assistantThinking || undefined, + }; + } +} + +export function* iterToolCalls(jsonl: string): IterableIterator { + const rows = parseLines(jsonl); + const calls = new Map(); + + for (const row of rows) { + const r = row as any; + if (r.method !== 'session/update') continue; + const update = r.params?.update; + if (!update) continue; + + if (update.sessionUpdate === 'tool_call') { + calls.set(update.toolCallId, { + id: update.toolCallId, + kind: update.kind ?? 'other', + title: update.title, + status: update.status ?? 'in_progress', + argsText: extractText(update.content), + }); + } else if (update.sessionUpdate === 'tool_call_update') { + const existing = calls.get(update.toolCallId); + if (existing) { + existing.status = update.status ?? existing.status; + if (update.content) existing.resultText = extractText(update.content); + if (update.status === 'failed') existing.isError = true; + } + } + } + + yield* calls.values(); +} + +export function computeMetrics(jsonl: string): ComputedMetrics { + const rows = parseLines(jsonl); + const header = getHeader(rows); + + const m: ComputedMetrics = { + durationMs: 0, + turns: 0, + toolCalls: 0, + bashCalls: 0, + readCalls: 0, + writeCalls: 0, + editCalls: 0, + tokens: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; + + if (header?.startedAt) { + const start = Date.parse(header.startedAt); + const end = header.endedAt ? Date.parse(header.endedAt) : Date.now(); + if (Number.isFinite(start) && Number.isFinite(end)) { + m.durationMs = Math.max(0, end - start); + } + } + + for (const row of rows) { + const r = row as any; + if (r.method === 'session/update') { + const update = r.params?.update; + if (!update) continue; + if (update.sessionUpdate === 'agent_message_chunk') m.turns += 1; + if (update.sessionUpdate === 'tool_call') { + m.toolCalls += 1; + const kind = update.kind ?? 'other'; + if (kind === 'execute') m.bashCalls += 1; + if (kind === 'read') m.readCalls += 1; + if (kind === 'write') m.writeCalls += 1; + if (kind === 'edit') m.editCalls += 1; + } + } + if (r.id && r.result?.stopReason) { + m.stopReason = r.result.stopReason; + const u = r.result.usage; + if (u) { + m.tokens.input += Number(u.inputTokens ?? 0); + m.tokens.output += Number(u.outputTokens ?? 0); + m.tokens.cacheRead += Number(u.cacheReadTokens ?? 0); + m.tokens.cacheWrite += Number(u.cacheCreationTokens ?? 0); + } + } + } + + m.tokens.total = m.tokens.input + m.tokens.output + m.tokens.cacheRead + m.tokens.cacheWrite; + return m; +} + +export function getFinalAssistantMessage(jsonl: string): string | undefined { + for (const msg of iterMessages(jsonl)) { + if (msg.role === 'assistant' && msg.text) return msg.text; + } + return undefined; +} + +export function getFailureEvidence(jsonl: string): string[] { + const evidence: string[] = []; + for (const call of iterToolCalls(jsonl)) { + if (call.isError && call.resultText) { + evidence.push(`tool ${call.kind} (${call.id}) failed: ${call.resultText}`); + } + } + return evidence; +} diff --git a/src/workbench/pi-agent.ts b/src/workbench/pi-agent.ts deleted file mode 100644 index 60c99bb..0000000 --- a/src/workbench/pi-agent.ts +++ /dev/null @@ -1,156 +0,0 @@ -import { - createAgentSession, - createBashTool, - createEditTool, - createFindTool, - createGrepTool, - createLsTool, - createReadTool, - createWriteTool, - AuthStorage, - DefaultResourceLoader, - ModelRegistry, - SessionManager, - type ResourceLoader, -} from '@mariozechner/pi-coding-agent'; -import type { AgentTool } from '@mariozechner/pi-agent-core'; -import { getModel, type Api, type Model } from '@mariozechner/pi-ai'; -import { resolve } from 'node:path'; - -import { buildAgentSystemPrompt } from './sandbox.js'; - -export function stripSensitiveEnv(env: NodeJS.ProcessEnv): NodeJS.ProcessEnv { - return { ...env }; -} - -export function createWorkbenchPiTools(cwd: string): AgentTool[] { - return [ - createReadTool(cwd), - createBashTool(cwd, { - spawnHook: (context) => ({ - ...context, - env: stripSensitiveEnv(context.env), - }), - }), - createEditTool(cwd), - createWriteTool(cwd), - createGrepTool(cwd), - createFindTool(cwd), - createLsTool(cwd), - ]; -} - -export async function createWorkbenchPiResourceLoader(params: { - cwd: string; - appendSystemPrompt?: string; - mcpConfigPath?: string; -}): Promise { - const cwd = resolve(params.cwd); - const appendSystemPrompt = [buildAgentSystemPrompt(), buildMcpSystemPrompt(params.mcpConfigPath), params.appendSystemPrompt] - .filter((value): value is string => typeof value === 'string' && value.trim().length > 0) - .join('\n\n'); - const loader = new DefaultResourceLoader({ - cwd, - noExtensions: true, - noSkills: true, - additionalSkillPaths: [cwd], - appendSystemPrompt, - }); - - await loader.reload(); - return loader; -} - -export async function createWorkbenchPiSession(params: { - cwd: string; - modelRef: string; - apiKeyEnv?: string; - appendSystemPrompt?: string; - mcpConfigPath?: string; - thinkingLevel?: 'off' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh'; -}) { - const { provider, model } = parseModelRef(params.modelRef); - if (provider !== 'openrouter') { - throw new Error(`Workbench only supports OpenRouter model refs, got: ${params.modelRef}`); - } - - const authStorage = AuthStorage.create(); - const apiKeyEnv = params.apiKeyEnv ?? 'OPENROUTER_API_KEY'; - const apiKey = process.env[apiKeyEnv]; - if (apiKey) { - authStorage.setRuntimeApiKey('openrouter' as never, apiKey); - } - - const modelRegistry = ModelRegistry.create(authStorage); - const resolvedModel = modelRegistry.find(provider, model) - ?? getModel(provider as never, model) - ?? synthesizeOpenRouterModel(provider, model); - if (!resolvedModel) { - throw new Error(`Could not resolve Pi model ${provider}/${model}`); - } - - const auth = await modelRegistry.getApiKeyAndHeaders(resolvedModel); - if (!auth.ok) { - throw new Error(auth.error); - } - - const resourceLoader = await createWorkbenchPiResourceLoader({ - cwd: params.cwd, - appendSystemPrompt: params.appendSystemPrompt, - mcpConfigPath: params.mcpConfigPath, - }); - - return createAgentSession({ - cwd: params.cwd, - model: resolvedModel, - thinkingLevel: params.thinkingLevel ?? 'medium', - authStorage, - modelRegistry, - resourceLoader, - tools: createWorkbenchPiTools(params.cwd), - sessionManager: SessionManager.inMemory(), - }); -} - -function buildMcpSystemPrompt(mcpConfigPath: string | undefined): string | undefined { - if (!mcpConfigPath) { - return undefined; - } - - return [ - 'Additional command:', - '- `mcp` is available on PATH for configured MCP servers.', - '- Run `mcp list --schema` to inspect available tools when needed.', - '- Run `mcp call key=value` to call a tool from bash.', - ].join('\n'); -} - -function parseModelRef(modelRef: string): { provider: string; model: string } { - const slash = modelRef.indexOf('/'); - if (slash <= 0 || slash === modelRef.length - 1) { - throw new Error(`Invalid model ref: ${modelRef}`); - } - return { - provider: modelRef.slice(0, slash), - model: modelRef.slice(slash + 1), - }; -} - -function synthesizeOpenRouterModel(provider: string, modelName: string): Model | undefined { - if (provider !== 'openrouter') { - return undefined; - } - - return { - id: modelName, - name: modelName, - api: 'openai-completions' as const, - provider: 'openrouter' as const, - baseUrl: 'https://openrouter.ai/api/v1', - reasoning: false, - input: ['text'], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 128000, - maxTokens: 16384, - }; -} diff --git a/src/workbench/run-case.ts b/src/workbench/run-case.ts index 672ddb3..ad794a7 100644 --- a/src/workbench/run-case.ts +++ b/src/workbench/run-case.ts @@ -5,16 +5,15 @@ import { loadWorkbenchCase } from './case-loader.js'; import { getFlag, positionals } from './cli-args.js'; import { runDockerWorkbenchCase } from './docker-runner.js'; import type { DockerWorkbenchRunResult, RunDockerWorkbenchCaseOptions } from './docker-runner.js'; -import { ensureOpenRouterModelRef, parseModelList, slugModelRef } from './models.js'; +import { ensureOpenRouterModelRef, slugModelRef } from './models.js'; import { aggregateTrials, formatTrialNumber, parseTrialsFlag, summarizeTrialAggregates } from './trials.js'; -import type { RunCaseAggregateResultFile, WorkbenchModelAggregateResult, WorkbenchTrialResultRef } from './types.js'; +import type { RunCaseAggregateResultFile, WorkbenchModelAggregateResult, WorkbenchRunSpec, WorkbenchTrialResultRef } from './types.js'; import { readWorkbenchResultFile, timestampSlug, writeJsonFile } from './utils.js'; export interface RunWorkbenchCaseParams { casePath: string; outDir?: string; model?: string; - models?: string[]; image?: string; keepWorkspace?: boolean; trials?: number; @@ -65,13 +64,13 @@ async function mapWithConcurrency( return results; } -function trialDirName(model: string, trial: number): string { - return `${slugModelRef(model)}--${formatTrialNumber(trial)}`; +function trialDirName(agent: string, model: string, trial: number): string { + return `${agent}--${slugModelRef(model)}--${formatTrialNumber(trial)}`; } -async function runWorkbenchCaseMatrix( - params: RunWorkbenchCaseParams & { models: string[] }, - deps: RunWorkbenchCaseDeps, +export async function runWorkbenchCase( + params: RunWorkbenchCaseParams, + deps: RunWorkbenchCaseDeps = {}, ): Promise { const dockerRunner = deps.runDockerWorkbenchCase ?? runDockerWorkbenchCase; const startedAt = new Date().toISOString(); @@ -81,15 +80,25 @@ async function runWorkbenchCaseMatrix( ? Math.floor(params.concurrency) : 1; + const resolvedCase = loadWorkbenchCase(params.casePath); + const cliModel = params.model + ? (resolvedCase.agent === 'pi-acp' ? ensureOpenRouterModelRef(params.model) : params.model.trim()) + : undefined; + const runs: WorkbenchRunSpec[] = [{ + agent: resolvedCase.agent, + model: cliModel ?? resolvedCase.model, + }]; + mkdirSync(resultsDir, { recursive: true }); - const jobs = params.models.flatMap((model) => Array.from({ length: trials }, (_, index) => ({ - model, + const jobs = runs.flatMap((run) => Array.from({ length: trials }, (_, index) => ({ + agent: run.agent, + model: run.model, trial: index + 1, }))); const completedTrials = await mapWithConcurrency(jobs, concurrency, async (job) => { - const trialDir = join(resultsDir, 'trials', trialDirName(job.model, job.trial)); + const trialDir = join(resultsDir, 'trials', trialDirName(job.agent, job.model, job.trial)); const run = await dockerRunner({ casePath: params.casePath, resultsDir: trialDir, @@ -105,21 +114,24 @@ async function runWorkbenchCaseMatrix( resultPath: relative(resultsDir, run.resultPath), tracePath: relative(resultsDir, run.tracePath), ...(run.summaryPath ? { summaryPath: relative(resultsDir, run.summaryPath) } : {}), + ...(result.metrics?.tokens?.total !== undefined ? { tokens: result.metrics.tokens.total } : {}), + ...(result.metrics?.durationMs !== undefined ? { durationMs: result.metrics.durationMs } : {}), }; - console.log(`${job.model} trial ${formatTrialNumber(job.trial)}: ${result.pass ? 'PASS' : 'FAIL'}`); + console.log(`${job.agent} ${job.model} trial ${formatTrialNumber(job.trial)}: ${result.pass ? 'PASS' : 'FAIL'}`); return { ...job, trialResult }; }); const results: WorkbenchModelAggregateResult[] = []; - for (const model of params.models) { + for (const run of runs) { const trialResults = completedTrials - .filter((trial) => trial.model === model) + .filter((trial) => trial.agent === run.agent && trial.model === run.model) .map((trial) => trial.trialResult) .sort((left, right) => left.trial - right.trial); const aggregate = aggregateTrials(trialResults); results.push({ - model, + agent: run.agent, + model: run.model, totalTrials: aggregate.totalTrials, passedTrials: aggregate.passedTrials, failedTrials: aggregate.failedTrials, @@ -127,6 +139,8 @@ async function runWorkbenchCaseMatrix( meanScore: aggregate.meanScore, passAtK: aggregate.passAtK, passHatK: aggregate.passHatK, + totalTokens: aggregate.totalTokens, + totalDurationMs: aggregate.totalDurationMs, trials: trialResults, }); } @@ -136,7 +150,7 @@ async function runWorkbenchCaseMatrix( name: 'run-case', startedAt, endedAt: new Date().toISOString(), - models: params.models, + runs, summary, results, }; @@ -150,60 +164,17 @@ async function runWorkbenchCaseMatrix( } } -export async function runWorkbenchCase( - params: RunWorkbenchCaseParams, - deps: RunWorkbenchCaseDeps = {}, -): Promise { - const model = params.model ? ensureOpenRouterModelRef(params.model) : undefined; - const models = params.models?.map((modelRef) => ensureOpenRouterModelRef(modelRef)); - - if ((models && models.length > 0) || (params.trials ?? 1) > 1) { - const matrixModels = models && models.length > 0 - ? models - : [model ?? loadWorkbenchCase(params.casePath).model]; - await runWorkbenchCaseMatrix({ ...params, model, models: matrixModels }, deps); - return; - } - - const dockerRunner = deps.runDockerWorkbenchCase ?? runDockerWorkbenchCase; - const selectedModel = models?.[0] ?? model; - const run = await dockerRunner({ - casePath: params.casePath, - outDir: params.outDir, - model: selectedModel, - image: params.image, - keepWorkspace: params.keepWorkspace, - }); - - const result = readWorkbenchResultFile(run.resultPath); - console.log(`Results: ${run.resultsDir}`); - console.log(`Grade: ${result.pass ? 'PASS' : 'FAIL'}`); - - if (result.evidence.length > 0) { - for (const line of result.evidence) { - console.log(`- ${line}`); - } - } else { - console.log('- (no evidence)'); - } - - if (!result.pass) { - process.exitCode = 1; - } -} - export async function runWorkbenchCaseFromCli(args: string[]): Promise { const caseArg = positionals(args, { - valueFlags: ['--out', '--model', '--models', '--image', '--trials', '--concurrency'], + valueFlags: ['--out', '--model', '--image', '--trials', '--concurrency'], booleanFlags: ['--keep-workspace'], })[0]; if (!caseArg) { - throw new Error('Missing case path. Usage: skill-optimizer run-case [--out ] [--model ] [--models ] [--trials ] [--concurrency ] [--image ] [--keep-workspace]'); + throw new Error('Missing case path. Usage: skill-optimizer run-case [--out ] [--model ] [--trials ] [--concurrency ] [--image ] [--keep-workspace]'); } const outDir = getFlag(args, '--out'); const model = getFlag(args, '--model'); - const models = getFlag(args, '--models'); const image = getFlag(args, '--image'); const trials = parseTrialsFlag(getFlag(args, '--trials')); const concurrency = parseConcurrencyFlag(getFlag(args, '--concurrency')); @@ -212,8 +183,7 @@ export async function runWorkbenchCaseFromCli(args: string[]): Promise { await runWorkbenchCase({ casePath: resolve(caseArg), outDir: outDir ? resolve(outDir) : undefined, - model: model ? ensureOpenRouterModelRef(model) : undefined, - models: models ? parseModelList(models) : undefined, + model, trials, concurrency, image, diff --git a/src/workbench/run-suite.ts b/src/workbench/run-suite.ts index 6003523..9ea8e6f 100644 --- a/src/workbench/run-suite.ts +++ b/src/workbench/run-suite.ts @@ -67,8 +67,8 @@ async function mapWithConcurrency( return results; } -function trialDirName(caseName: string, model: string, trial: number): string { - return `${caseName}--${slugModelRef(model)}--${formatTrialNumber(trial)}`; +function trialDirName(caseName: string, agent: string, model: string, trial: number): string { + return `${caseName}--${agent}--${slugModelRef(model)}--${formatTrialNumber(trial)}`; } export async function runWorkbenchSuite( @@ -76,11 +76,8 @@ export async function runWorkbenchSuite( deps: RunWorkbenchSuiteDeps = {}, ): Promise { const suite = loadWorkbenchSuite(params.suitePath); - const models = suite.models; + const runs = suite.runs; const trials = params.trials ?? 1; - if (models.length === 0) { - throw new Error('Workbench suite requires at least one model in suite.yml via the suite "models" field'); - } const dockerRunner = deps.runDockerWorkbenchCase ?? runDockerWorkbenchCase; const startedAt = new Date().toISOString(); @@ -92,17 +89,18 @@ export async function runWorkbenchSuite( mkdirSync(resultsDir, { recursive: true }); - const jobs = suite.cases.flatMap((suiteCase) => models.flatMap((model) => ( + const jobs = suite.cases.flatMap((suiteCase) => runs.flatMap((run) => ( Array.from({ length: trials }, (_, index) => ({ suiteCase, caseName: suiteCase.slug, - model, + agent: run.agent, + model: run.model, trial: index + 1, })) ))); const completedTrials = await mapWithConcurrency(jobs, concurrency, async (job) => { - const trialDir = join(resultsDir, 'trials', trialDirName(job.caseName, job.model, job.trial)); + const trialDir = join(resultsDir, 'trials', trialDirName(job.caseName, job.agent, job.model, job.trial)); const run = await dockerRunner({ casePath: job.suiteCase.path, case: job.suiteCase.case, @@ -120,22 +118,25 @@ export async function runWorkbenchSuite( resultPath: relative(resultsDir, run.resultPath), tracePath: relative(resultsDir, run.tracePath), ...(run.summaryPath ? { summaryPath: relative(resultsDir, run.summaryPath) } : {}), + ...(result.metrics?.tokens?.total !== undefined ? { tokens: result.metrics.tokens.total } : {}), + ...(result.metrics?.durationMs !== undefined ? { durationMs: result.metrics.durationMs } : {}), }; - console.log(`${job.caseName} ${job.model} trial ${formatTrialNumber(job.trial)}: ${result.pass ? 'PASS' : 'FAIL'}`); + console.log(`${job.caseName} ${job.agent} ${job.model} trial ${formatTrialNumber(job.trial)}: ${result.pass ? 'PASS' : 'FAIL'}`); return { ...job, trialResult }; }); const results: WorkbenchCaseModelAggregateResult[] = []; for (const suiteCase of suite.cases) { - for (const model of models) { + for (const run of runs) { const trialResults = completedTrials - .filter((trial) => trial.caseName === suiteCase.slug && trial.model === model) + .filter((trial) => trial.caseName === suiteCase.slug && trial.agent === run.agent && trial.model === run.model) .map((trial) => trial.trialResult) .sort((left, right) => left.trial - right.trial); const aggregate = aggregateTrials(trialResults); results.push({ caseName: suiteCase.slug, - model, + agent: run.agent, + model: run.model, totalTrials: aggregate.totalTrials, passedTrials: aggregate.passedTrials, failedTrials: aggregate.failedTrials, @@ -143,6 +144,8 @@ export async function runWorkbenchSuite( meanScore: aggregate.meanScore, passAtK: aggregate.passAtK, passHatK: aggregate.passHatK, + totalTokens: aggregate.totalTokens, + totalDurationMs: aggregate.totalDurationMs, trials: trialResults, }); } @@ -153,7 +156,7 @@ export async function runWorkbenchSuite( name: suite.name, startedAt, endedAt: new Date().toISOString(), - models, + runs, cases: caseSlugs, summary, results, diff --git a/src/workbench/sandbox.ts b/src/workbench/sandbox.ts deleted file mode 100644 index 8cb0864..0000000 --- a/src/workbench/sandbox.ts +++ /dev/null @@ -1,13 +0,0 @@ -export function buildAgentSystemPrompt(): string { - return [ - 'Operating environment:', - '- Current working directory is /work.', - '- Write all outputs under /work.', - '- The Docker socket is not mounted.', - '- Internet access is available for task dependencies unless the network is unavailable.', - '- Node.js, npm, Python, pip, and venv are installed.', - '- Do not use global pip installs.', - '- If you need Python packages, run: python -m venv /work/.venv && /work/.venv/bin/pip install .', - '- Run Python scripts with /work/.venv/bin/python when using installed packages.', - ].join('\n'); -} diff --git a/src/workbench/suite-loader.ts b/src/workbench/suite-loader.ts index a6813b4..6c60740 100644 --- a/src/workbench/suite-loader.ts +++ b/src/workbench/suite-loader.ts @@ -3,9 +3,9 @@ import { basename, dirname, extname, resolve } from 'node:path'; import { parse as parseYaml } from 'yaml'; +import { resolveAgent } from './agents/registry.js'; import { readMcpServers, readMcpServices, resolveWorkbenchCaseConfig } from './case-loader.js'; -import { ensureOpenRouterModelRef } from './models.js'; -import type { ResolvedWorkbenchCase, WorkbenchMcpServersConfig, WorkbenchMcpServicesConfig } from './types.js'; +import type { ResolvedWorkbenchCase, SkillUnderTestSpec, WorkbenchMcpServersConfig, WorkbenchMcpServicesConfig, WorkbenchRunSpec } from './types.js'; import { slugPathSegment } from './utils.js'; export interface ResolvedWorkbenchSuiteCase { @@ -21,7 +21,7 @@ export interface ResolvedWorkbenchSuite { appendSystemPrompt?: string; casePaths: string[]; cases: ResolvedWorkbenchSuiteCase[]; - models: string[]; + runs: WorkbenchRunSpec[]; } export function loadWorkbenchSuite(configPath: string): ResolvedWorkbenchSuite { @@ -46,11 +46,10 @@ export function loadWorkbenchSuite(configPath: string): ResolvedWorkbenchSuite { const name = requireNonEmptyString(parsed, 'name', resolvedConfigPath); const appendSystemPrompt = readOptionalString(parsed, 'appendSystemPrompt', resolvedConfigPath); const suiteDefaults = readSuiteCaseDefaults(parsed, resolvedConfigPath); + const runs = readRunsMatrix(parsed, resolvedConfigPath); const cases = readCaseEntries(parsed, resolvedConfigPath) .map((entry, index) => resolveSuiteCase(entry, index, resolvedConfigPath, configDir, suiteDefaults)); const casePaths = cases.flatMap((suiteCase) => suiteCase.path ? [suiteCase.path] : []); - const models = readStringArray(parsed, 'models', resolvedConfigPath, true) - .map((model) => ensureOpenRouterModelRef(model)); return { configPath: resolvedConfigPath, @@ -59,7 +58,7 @@ export function loadWorkbenchSuite(configPath: string): ResolvedWorkbenchSuite { appendSystemPrompt, casePaths, cases, - models, + runs, }; } @@ -71,6 +70,7 @@ interface SuiteCaseDefaults { mcpServers: WorkbenchMcpServersConfig; mcpServices: WorkbenchMcpServicesConfig; timeoutSeconds?: number; + skillUnderTest?: SkillUnderTestSpec; } function readSuiteCaseDefaults(parsed: Record, configPath: string): SuiteCaseDefaults { @@ -86,9 +86,31 @@ function readSuiteCaseDefaults(parsed: Record, configPath: stri mcpServers: readMcpServers(parsed, configPath), mcpServices: readMcpServices(parsed, configPath), timeoutSeconds: readOptionalTimeoutSeconds(parsed, configPath), + skillUnderTest: readOptionalSkillUnderTest(parsed, configPath), }; } +function readOptionalSkillUnderTest( + parsed: Record, + configPath: string, +): SkillUnderTestSpec | undefined { + const value = parsed.skillUnderTest; + if (value === undefined) return undefined; + if (!value || typeof value !== 'object' || Array.isArray(value)) { + throw new Error(`Workbench suite ${configPath}: field "skillUnderTest" must be an object with slug and hostPath`); + } + const obj = value as Record; + const slug = obj.slug; + const hostPath = obj.hostPath; + if (typeof slug !== 'string' || !slug.trim()) { + throw new Error(`Workbench suite ${configPath}: field "skillUnderTest.slug" must be a non-empty string`); + } + if (typeof hostPath !== 'string' || !hostPath.trim()) { + throw new Error(`Workbench suite ${configPath}: field "skillUnderTest.hostPath" must be a non-empty string`); + } + return { slug: slug.trim(), hostPath: hostPath.trim() }; +} + function resolveSuiteCase( entry: string | Record, index: number, @@ -134,6 +156,7 @@ function applySuiteDefaults( ...(Object.keys(mcpServers).length > 0 ? { mcpServers } : {}), ...(Object.keys(mcpServices).length > 0 ? { mcpServices } : {}), ...(defaults.timeoutSeconds !== undefined ? { timeoutSeconds: defaults.timeoutSeconds } : {}), + ...(defaults.skillUnderTest !== undefined ? { skillUnderTest: defaults.skillUnderTest } : {}), ...entry, ...(Object.keys(mcpServers).length > 0 ? { mcpServers } : {}), ...(Object.keys(mcpServices).length > 0 ? { mcpServices } : {}), @@ -194,7 +217,7 @@ function requireNonEmptyString( function readStringArray( parsed: Record, - field: 'models' | 'env' | 'setup' | 'cleanup', + field: 'env' | 'setup' | 'cleanup', configPath: string, optional = false, ): string[] { @@ -261,3 +284,32 @@ function readOptionalTimeoutSeconds(parsed: Record, configPath: } return value; } + +function readRunsMatrix(parsed: Record, configPath: string): WorkbenchRunSpec[] { + if ('models' in parsed && !('runs' in parsed)) { + throw new Error( + `Suite ${configPath}: 'runs:' has replaced 'models:'. ` + + `Replace 'models:' with a 'runs:' list of { agent, model } objects.`, + ); + } + if (!Array.isArray(parsed.runs) || parsed.runs.length === 0) { + throw new Error(`Suite ${configPath} must declare a non-empty 'runs:' list.`); + } + return parsed.runs.map((run: unknown, index: number) => { + if (!run || typeof run !== 'object' || Array.isArray(run)) { + throw new Error(`Suite ${configPath}: 'runs[${index}]' must be an object with agent: and model: fields.`); + } + const r = run as Record; + if (typeof r.agent !== 'string' || r.agent.trim() === '') { + throw new Error(`Each run in ${configPath} must have agent: and model: fields.`); + } + if (typeof r.model !== 'string' || r.model.trim() === '') { + throw new Error(`Each run in ${configPath} must have agent: and model: fields.`); + } + const resolvedAgent = resolveAgent(r.agent.trim()); + return { + agent: resolvedAgent.name, + model: r.model.trim(), + }; + }); +} diff --git a/src/workbench/trace.ts b/src/workbench/trace.ts index 3ea5916..64df84c 100644 --- a/src/workbench/trace.ts +++ b/src/workbench/trace.ts @@ -1,290 +1,6 @@ -import type { WorkbenchTrace, WorkbenchTraceEntry, WorkbenchTraceEvent } from './types.js'; -import { isRecord } from './utils.js'; - -export interface TraceRecorder { - events: WorkbenchTraceEvent[]; - record(event: unknown): void; - toTrace(params: { - caseName: string; - model: string; - startedAt: string; - endedAt: string; - messages?: unknown[]; - }): WorkbenchTrace; -} - -export function createTraceCollector(): { record(event: unknown): void; events: unknown[] } { - const events: unknown[] = []; - return { - events, - record(event: unknown) { - events.push(event); - }, - }; -} - -export function createTraceRecorder(options: { now?: () => string } = {}): TraceRecorder { - const now = options.now ?? (() => new Date().toISOString()); - const events: WorkbenchTraceEvent[] = []; - - return { - events, - record(event: unknown) { - events.push(normalizeTraceEvent(event, now())); - }, - toTrace(params) { - const eventEntries = normalizeEvents(events); - const entries = eventEntries.length > 0 - ? mergeSessionMessages(eventEntries, params.messages ?? []) - : normalizeMessages(params.messages ?? []); - return { - schemaVersion: 1, - caseName: params.caseName, - model: params.model, - startedAt: params.startedAt, - endedAt: params.endedAt, - events: [...events], - entries, - }; - }, - }; -} - -export function buildWorkbenchTrace(params: { - caseName: string; - model: string; - startedAt: string; - endedAt: string; - messages: unknown[]; -}): WorkbenchTrace { - return { - caseName: params.caseName, - model: params.model, - startedAt: params.startedAt, - endedAt: params.endedAt, - entries: normalizeMessages(params.messages), - }; -} - -function normalizeTraceEvent(event: unknown, timestamp: string): WorkbenchTraceEvent { - if (!isRecord(event) || typeof event.type !== 'string') { - return { type: 'unknown', timestamp, value: toJsonSafe(event) }; - } - - const normalized: WorkbenchTraceEvent = { type: event.type, timestamp }; - for (const [key, value] of Object.entries(event)) { - if (key === 'type') { - continue; - } - const safeValue = toJsonSafe(value); - if (safeValue !== undefined) { - normalized[key] = safeValue; - } - } - return normalized; -} - -function normalizeEvents(events: WorkbenchTraceEvent[]): WorkbenchTraceEntry[] { - const entries: WorkbenchTraceEntry[] = []; - - for (const event of events) { - if (event.type === 'message_end' && isRecord(event.message)) { - const messageEntry = normalizeMessageOnly(event.message, event.timestamp); - if (messageEntry) { - entries.push(messageEntry); - } - continue; - } - - if (event.type === 'tool_execution_start') { - entries.push({ - type: 'tool_call', - id: typeof event.toolCallId === 'string' ? event.toolCallId : undefined, - name: typeof event.toolName === 'string' ? event.toolName : 'unknown', - arguments: event.args, - timestamp: event.timestamp, - }); - continue; - } - - if (event.type === 'tool_execution_end') { - entries.push({ - type: 'tool_result', - id: typeof event.toolCallId === 'string' ? event.toolCallId : undefined, - name: typeof event.toolName === 'string' ? event.toolName : undefined, - text: extractToolEventText(event.result), - isError: typeof event.isError === 'boolean' ? event.isError : undefined, - timestamp: event.timestamp, - }); - } - } - - return entries; -} - -function mergeSessionMessages(eventEntries: WorkbenchTraceEntry[], messages: unknown[]): WorkbenchTraceEntry[] { - const sessionMessages = normalizeMessages(messages) - .filter((entry): entry is Extract => entry.type === 'message'); - const missingSessionMessages = sessionMessages.filter((message) => !eventEntries.some((entry) => sameMessageEntry(entry, message))); - return [...missingSessionMessages, ...eventEntries]; -} - -function sameMessageEntry(left: WorkbenchTraceEntry, right: Extract): boolean { - if (left.type !== 'message') { - return false; - } - return left.role === right.role - && left.text === right.text - && left.thinking === right.thinking - && left.stopReason === right.stopReason - && left.errorMessage === right.errorMessage; -} - -function normalizeMessageOnly(message: Record, timestamp: string): WorkbenchTraceEntry | undefined { - const role = typeof message.role === 'string' ? message.role : 'unknown'; - if (role === 'toolResult') { - return undefined; - } - - const content = Array.isArray(message.content) ? message.content : []; - const text = extractContentByType(content, 'text', 'text'); - const thinking = extractContentByType(content, 'thinking', 'thinking'); - const hasTerminalMetadata = typeof message.stopReason === 'string' || typeof message.errorMessage === 'string'; - if (text.length === 0 && thinking.length === 0 && role === 'assistant' && !hasTerminalMetadata) { - return undefined; - } - - return { - type: 'message', - role, - text: text.length > 0 ? text : undefined, - thinking: thinking.length > 0 ? thinking : undefined, - timestamp, - usage: message.usage, - stopReason: message.stopReason, - errorMessage: typeof message.errorMessage === 'string' ? message.errorMessage : undefined, - }; -} - -function normalizeMessages(messages: unknown[]): WorkbenchTraceEntry[] { - const entries: WorkbenchTraceEntry[] = []; - - for (const message of messages) { - if (!isRecord(message)) { - continue; - } - - const role = typeof message.role === 'string' ? message.role : 'unknown'; - const timestamp = message.timestamp; - - if (role === 'toolResult') { - entries.push({ - type: 'tool_result', - id: typeof message.toolCallId === 'string' ? message.toolCallId : undefined, - name: typeof message.toolName === 'string' ? message.toolName : undefined, - text: extractText(message.content), - isError: typeof message.isError === 'boolean' ? message.isError : undefined, - timestamp, - }); - continue; - } - - const content = Array.isArray(message.content) ? message.content : []; - const text = extractContentByType(content, 'text', 'text'); - const thinking = extractContentByType(content, 'thinking', 'thinking'); - - const hasTerminalMetadata = typeof message.stopReason === 'string' || typeof message.errorMessage === 'string'; - if (text.length > 0 || thinking.length > 0 || role !== 'assistant' || hasTerminalMetadata) { - entries.push({ - type: 'message', - role, - text: text.length > 0 ? text : undefined, - thinking: thinking.length > 0 ? thinking : undefined, - timestamp, - usage: message.usage, - stopReason: message.stopReason, - errorMessage: typeof message.errorMessage === 'string' ? message.errorMessage : undefined, - }); - } - - for (const item of content) { - if (!isRecord(item) || item.type !== 'toolCall') { - continue; - } - - entries.push({ - type: 'tool_call', - id: typeof item.id === 'string' ? item.id : undefined, - name: typeof item.name === 'string' ? item.name : 'unknown', - arguments: item.arguments, - timestamp, - }); - } - } - - return entries; -} - -function extractContentByType(content: unknown[], type: string, field: string): string { - return content - .map((item) => { - if (!isRecord(item) || item.type !== type) { - return ''; - } - const value = item[field]; - return typeof value === 'string' ? value : ''; - }) - .filter((value) => value.length > 0) - .join('\n'); -} - -function extractText(content: unknown): string | undefined { - if (typeof content === 'string') { - return content; - } - - if (!Array.isArray(content)) { - return undefined; - } - - const text = extractContentByType(content, 'text', 'text'); - return text.length > 0 ? text : undefined; -} - -function extractToolEventText(result: unknown): string | undefined { - if (!isRecord(result)) { - return undefined; - } - return extractText(result.content); -} - -function toJsonSafe(value: unknown, seen = new WeakSet(), depth = 0): unknown { - if (value === null || typeof value === 'string' || typeof value === 'number' || typeof value === 'boolean') { - return value; - } - if (value === undefined || typeof value === 'function' || typeof value === 'symbol') { - return undefined; - } - if (depth > 8) { - return '[MaxDepth]'; - } - if (Array.isArray(value)) { - return value.map((item) => toJsonSafe(item, seen, depth + 1)); - } - if (typeof value === 'object') { - if (seen.has(value)) { - return '[Circular]'; - } - seen.add(value); - const record: Record = {}; - for (const [key, item] of Object.entries(value)) { - const safeItem = toJsonSafe(item, seen, depth + 1); - if (safeItem !== undefined) { - record[key] = safeItem; - } - } - seen.delete(value); - return record; - } - return String(value); -} +// Re-export trace-recorder as the public API. +// The legacy normalization layer (buildWorkbenchTrace, createTraceCollector, +// the entries[]-shaped WorkbenchTrace) is gone — trace.jsonl is raw ACP wire +// format now, captured by createTraceRecorder from ./acp/trace-recorder.js. +export { createTraceRecorder } from './acp/trace-recorder.js'; +export type { TraceHeader, TraceRecorder } from './acp/trace-recorder.js'; diff --git a/src/workbench/trials.ts b/src/workbench/trials.ts index 9dd34c0..c65a9e1 100644 --- a/src/workbench/trials.ts +++ b/src/workbench/trials.ts @@ -4,6 +4,8 @@ export interface TrialScoreInput { trial: number; pass: boolean; score: number; + tokens?: number; + durationMs?: number; } export interface TrialAggregate { @@ -14,6 +16,8 @@ export interface TrialAggregate { meanScore: number; passAtK: boolean; passHatK: boolean; + totalTokens: number; + totalDurationMs: number; } export function formatTrialNumber(trial: number): string { @@ -40,6 +44,8 @@ export function aggregateTrials(trials: TrialScoreInput[]): TrialAggregate { const passedTrials = trials.filter((trial) => trial.pass).length; const failedTrials = totalTrials - passedTrials; const scoreTotal = trials.reduce((sum, trial) => sum + trial.score, 0); + const totalTokens = trials.reduce((sum, trial) => sum + (trial.tokens ?? 0), 0); + const totalDurationMs = trials.reduce((sum, trial) => sum + (trial.durationMs ?? 0), 0); return { totalTrials, @@ -49,6 +55,8 @@ export function aggregateTrials(trials: TrialScoreInput[]): TrialAggregate { meanScore: totalTrials === 0 ? 0 : scoreTotal / totalTrials, passAtK: totalTrials > 0 && passedTrials > 0, passHatK: totalTrials > 0 && passedTrials === totalTrials, + totalTokens, + totalDurationMs, }; } diff --git a/src/workbench/types.ts b/src/workbench/types.ts index d80c25d..9aa252f 100644 --- a/src/workbench/types.ts +++ b/src/workbench/types.ts @@ -3,6 +3,16 @@ export interface WorkbenchGraderConfig { command: string; } +export interface WorkbenchRunSpec { + agent: string; + model: string; +} + +export interface SkillUnderTestSpec { + slug: string; // e.g., "web-design-guidelines" + hostPath: string; // absolute path on host to the skill dir containing SKILL.md +} + export type WorkbenchMcpJsonValue = | string | number @@ -41,12 +51,14 @@ export interface WorkbenchCaseConfig { references: string; task: string; graders: WorkbenchGraderConfig[]; + agent: string; // required + model?: string; // semantics now agent-native + skillUnderTest?: SkillUnderTestSpec; // optional mcpServers?: WorkbenchMcpServersConfig; mcpServices?: WorkbenchMcpServicesConfig; env?: string[]; setup?: string[]; cleanup?: string[]; - model?: string; timeoutSeconds?: number; } @@ -62,7 +74,9 @@ export interface ResolvedWorkbenchCase { env: string[]; setup: string[]; cleanup: string[]; + agent: string; model: string; + skillUnderTest?: SkillUnderTestSpec; timeoutSeconds: number; } @@ -101,14 +115,6 @@ export interface WorkbenchTokenMetrics { total: number; } -export interface WorkbenchCostMetrics { - input: number; - output: number; - cacheRead: number; - cacheWrite: number; - total: number; -} - export interface WorkbenchMetrics { durationMs: number; turns: number; @@ -120,7 +126,6 @@ export interface WorkbenchMetrics { editCalls: number; stopReason?: string; tokens: WorkbenchTokenMetrics; - cost: WorkbenchCostMetrics; } export interface WorkbenchTrialSummaryFile { @@ -152,9 +157,12 @@ export interface WorkbenchTrialResultRef { resultPath: string; tracePath: string; summaryPath?: string; + tokens?: number; + durationMs?: number; } export interface WorkbenchModelAggregateResult { + agent: string; model: string; totalTrials: number; passedTrials: number; @@ -163,6 +171,8 @@ export interface WorkbenchModelAggregateResult { meanScore: number; passAtK: boolean; passHatK: boolean; + totalTokens: number; + totalDurationMs: number; trials: WorkbenchTrialResultRef[]; } @@ -174,7 +184,7 @@ export interface RunCaseAggregateResultFile { name: string; startedAt: string; endedAt: string; - models: string[]; + runs: WorkbenchRunSpec[]; summary: WorkbenchAggregateSummary; results: WorkbenchModelAggregateResult[]; } @@ -183,51 +193,18 @@ export interface RunSuiteAggregateResultFile { name: string; startedAt: string; endedAt: string; - models: string[]; + runs: WorkbenchRunSpec[]; cases: string[]; summary: WorkbenchAggregateSummary; results: WorkbenchCaseModelAggregateResult[]; } -export type WorkbenchTraceEntry = - | { - type: 'message'; - role: string; - text?: string; - thinking?: string; - timestamp?: unknown; - usage?: unknown; - stopReason?: unknown; - errorMessage?: string; - } - | { - type: 'tool_call'; - id?: string; - name: string; - arguments?: unknown; - timestamp?: unknown; - } - | { - type: 'tool_result'; - id?: string; - name?: string; - text?: string; - isError?: boolean; - timestamp?: unknown; - }; - -export interface WorkbenchTraceEvent { - type: string; - timestamp: string; - [key: string]: unknown; -} - export interface WorkbenchTrace { - schemaVersion?: 1; + schemaVersion?: 2; // bumped from 1 (raw ACP format) caseName: string; + agent: string; // NEW model: string; startedAt: string; endedAt: string; - events?: WorkbenchTraceEvent[]; - entries: WorkbenchTraceEntry[]; + // No more entries[]; the trace.jsonl file IS the source of truth. } diff --git a/tests/acp/auth.test.ts b/tests/acp/auth.test.ts new file mode 100644 index 0000000..107755a --- /dev/null +++ b/tests/acp/auth.test.ts @@ -0,0 +1,62 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, writeFileSync, mkdirSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { resolveAuth } from '../../src/workbench/acp/auth.js'; +import { AGENTS } from '../../src/workbench/agents/registry.js'; + +test('resolveAuth uses subscription file when present', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + mkdirSync(join(home, '.claude'), { recursive: true }); + writeFileSync(join(home, '.claude', '.credentials.json'), '{}'); + + const auth = resolveAuth(AGENTS['claude-agent-acp'], { + home, + env: {}, + }); + + assert.equal(auth.mode, 'subscription'); + assert.equal(auth.files.length, 1); + assert.equal(auth.files[0].hostPath, join(home, '.claude', '.credentials.json')); + + rmSync(home, { recursive: true }); +}); + +test('resolveAuth falls back to env API key when subscription absent', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + + const auth = resolveAuth(AGENTS['claude-agent-acp'], { + home, + env: { ANTHROPIC_API_KEY: 'sk-test-key' }, + }); + + assert.equal(auth.mode, 'env'); + assert.deepEqual(auth.envNames, ['ANTHROPIC_API_KEY']); + + rmSync(home, { recursive: true }); +}); + +test('resolveAuth throws when neither subscription nor env key present', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + + assert.throws( + () => resolveAuth(AGENTS['claude-agent-acp'], { home, env: {} }), + /claude login|ANTHROPIC_API_KEY/, + ); + + rmSync(home, { recursive: true }); +}); + +test('resolveAuth returns env mode for agents without subscriptionAuth', () => { + const home = mkdtempSync(join(tmpdir(), 'auth-test-')); + + const auth = resolveAuth(AGENTS['pi-acp'], { + home, + env: { OPENROUTER_API_KEY: 'sk-test' }, + }); + + assert.equal(auth.mode, 'env'); + + rmSync(home, { recursive: true }); +}); diff --git a/tests/acp/client.test.ts b/tests/acp/client.test.ts new file mode 100644 index 0000000..a4ad178 --- /dev/null +++ b/tests/acp/client.test.ts @@ -0,0 +1,19 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { spawn } from 'node:child_process'; +import { createWorkbenchClient } from '../../src/workbench/acp/client.js'; +import { createDockerExecStream } from '../../src/workbench/acp/transport.js'; + +test('createWorkbenchClient returns object with initialize/newSession/prompt/cancel', () => { + const child = spawn('cat', []); + const stream = createDockerExecStream(child); + const client = createWorkbenchClient({ stream, onSessionUpdate: () => {} }); + + assert.equal(typeof client.connection.initialize, 'function'); + assert.equal(typeof client.connection.newSession, 'function'); + assert.equal(typeof client.connection.prompt, 'function'); + assert.equal(typeof client.connection.cancel, 'function'); + assert.equal(typeof client.close, 'function'); + + child.kill(); +}); diff --git a/tests/acp/mcp-config-writer.test.ts b/tests/acp/mcp-config-writer.test.ts new file mode 100644 index 0000000..2987359 --- /dev/null +++ b/tests/acp/mcp-config-writer.test.ts @@ -0,0 +1,88 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, readFileSync, rmSync, existsSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { writeMcpConfig } from '../../src/workbench/acp/mcp-config-writer.js'; +import { AGENTS } from '../../src/workbench/agents/registry.js'; + +const sampleCase = { + mcpServers: { + calculator: { command: 'node', args: ['/work/mcp/calculator.mjs'] }, + }, +}; + +test('writeMcpConfig for claude writes to .claude.json mcpServers', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['claude-agent-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const data = JSON.parse(readFileSync(join(home, '.claude.json'), 'utf-8')); + assert.ok(data.mcpServers?.calculator); + assert.equal(data.mcpServers.calculator.command, 'node'); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for codex writes to .codex/config.toml', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['codex-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const toml = readFileSync(join(home, '.codex/config.toml'), 'utf-8'); + assert.match(toml, /\[mcp_servers\.calculator\]/); + assert.match(toml, /command = "node"/); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for gemini writes to .gemini/settings.json mcpServers', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['gemini'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const data = JSON.parse(readFileSync(join(home, '.gemini/settings.json'), 'utf-8')); + assert.ok(data.mcpServers?.calculator); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for opencode writes to .config/opencode/opencode.json mcp', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['opencode'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }); + const data = JSON.parse(readFileSync(join(home, '.config/opencode/opencode.json'), 'utf-8')); + assert.ok(data.mcp?.calculator); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig for pi-acp throws "not yet supported" (pending Task 9b)', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + assert.throws( + () => writeMcpConfig({ + agent: AGENTS['pi-acp'], + caseConfig: sampleCase, + agentHomeOnHost: home, + }), + /pi-acp MCP support pending/, + ); + rmSync(home, { recursive: true }); +}); + +test('writeMcpConfig is a no-op when caseConfig has no mcpServers', () => { + const home = mkdtempSync(join(tmpdir(), 'mcp-test-')); + writeMcpConfig({ + agent: AGENTS['claude-agent-acp'], + caseConfig: {}, + agentHomeOnHost: home, + }); + // No files should be written. + assert.equal(existsSync(join(home, '.claude.json')), false); + rmSync(home, { recursive: true }); +}); diff --git a/tests/acp/sdk-smoke.test.ts b/tests/acp/sdk-smoke.test.ts new file mode 100644 index 0000000..e48a70d --- /dev/null +++ b/tests/acp/sdk-smoke.test.ts @@ -0,0 +1,9 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; + +test('@agentclientprotocol/sdk exports ClientSideConnection', async () => { + const mod = await import('@agentclientprotocol/sdk'); + assert.equal(typeof mod.ClientSideConnection, 'function'); + assert.equal(typeof mod.ndJsonStream, 'function'); + assert.equal(typeof mod.RequestError, 'function'); +}); diff --git a/tests/acp/skill-deploy.test.ts b/tests/acp/skill-deploy.test.ts new file mode 100644 index 0000000..0bcb3b5 --- /dev/null +++ b/tests/acp/skill-deploy.test.ts @@ -0,0 +1,31 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { computeSkillMount } from '../../src/workbench/acp/skill-deploy.js'; +import { AGENTS } from '../../src/workbench/agents/registry.js'; + +test('computeSkillMount returns native path for claude-agent-acp', () => { + const mount = computeSkillMount({ + agent: AGENTS['claude-agent-acp'], + skillSlug: 'web-design-guidelines', + hostSkillDir: '/host/path/to/skill', + agentHome: '/home/agent', + }); + + assert.equal(mount.hostPath, '/host/path/to/skill'); + assert.equal(mount.containerPath, '/home/agent/.claude/skills/web-design-guidelines'); + assert.equal(mount.readOnly, true); +}); + +test('computeSkillMount handles agents whose skill_paths use $HOME', () => { + for (const cfg of Object.values(AGENTS)) { + const mount = computeSkillMount({ + agent: cfg, + skillSlug: 'test-skill', + hostSkillDir: '/host/dir', + agentHome: '/home/agent', + }); + assert.ok(!mount.containerPath.includes('$HOME')); + assert.ok(mount.containerPath.startsWith('/home/agent/')); + assert.ok(mount.containerPath.endsWith('/test-skill')); + } +}); diff --git a/tests/acp/trace-recorder.test.ts b/tests/acp/trace-recorder.test.ts new file mode 100644 index 0000000..918795c --- /dev/null +++ b/tests/acp/trace-recorder.test.ts @@ -0,0 +1,34 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, readFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { createTraceRecorder } from '../../src/workbench/acp/trace-recorder.js'; + +test('createTraceRecorder writes header then verbatim messages', () => { + const dir = mkdtempSync(join(tmpdir(), 'trace-rec-')); + const path = join(dir, 'trace.jsonl'); + const rec = createTraceRecorder({ + tracePath: path, + header: { + caseName: 'x', + agent: 'claude-agent-acp', + model: 'claude-haiku-4-5', + startedAt: '2026-05-25T00:00:00Z', + }, + }); + rec.recordRaw({ jsonrpc: '2.0', id: 1, method: 'initialize', params: {} }); + rec.recordRaw({ jsonrpc: '2.0', id: 1, result: {} }); + rec.finalize('2026-05-25T00:00:01Z'); + + const lines = readFileSync(path, 'utf-8').trim().split('\n'); + assert.equal(lines.length, 3); + const header = JSON.parse(lines[0]); + assert.equal(header.type, 'trace_start'); + assert.equal(header.agent, 'claude-agent-acp'); + assert.equal(header.endedAt, '2026-05-25T00:00:01Z'); + const msg = JSON.parse(lines[1]); + assert.equal(msg.method, 'initialize'); + + rmSync(dir, { recursive: true }); +}); diff --git a/tests/acp/transport.test.ts b/tests/acp/transport.test.ts new file mode 100644 index 0000000..7f09b23 --- /dev/null +++ b/tests/acp/transport.test.ts @@ -0,0 +1,22 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { spawn } from 'node:child_process'; +import { createDockerExecStream } from '../../src/workbench/acp/transport.js'; + +test('createDockerExecStream produces a Stream that round-trips ndjson', async () => { + // Use `cat` as a stand-in for the agent: it echoes stdin to stdout. + const child = spawn('cat', [], { stdio: ['pipe', 'pipe', 'pipe'] }); + const stream = createDockerExecStream(child); + + const message = { jsonrpc: '2.0', id: 1, method: 'initialize', params: {} }; + const writer = stream.outgoing.getWriter(); + await writer.write(new TextEncoder().encode(JSON.stringify(message) + '\n')); + writer.releaseLock(); + + const reader = stream.incoming.getReader(); + const { value } = await reader.read(); + const echoed = new TextDecoder().decode(value).trim(); + assert.equal(echoed, JSON.stringify(message)); + + child.kill(); +}); diff --git a/tests/agents/registry.test.ts b/tests/agents/registry.test.ts new file mode 100644 index 0000000..be3ad79 --- /dev/null +++ b/tests/agents/registry.test.ts @@ -0,0 +1,39 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { resolveAgent, AGENTS, AGENT_ALIASES } from '../../src/workbench/agents/registry.js'; + +test('AGENTS contains all 5 expected entries', () => { + for (const name of ['claude-agent-acp', 'codex-acp', 'gemini', 'opencode', 'pi-acp']) { + assert.ok(AGENTS[name], `missing agent: ${name}`); + assert.equal(AGENTS[name].name, name); + } +}); + +test('resolveAgent accepts aliases', () => { + assert.equal(resolveAgent('claude').name, 'claude-agent-acp'); + assert.equal(resolveAgent('codex').name, 'codex-acp'); + assert.equal(resolveAgent('pi').name, 'pi-acp'); +}); + +test('resolveAgent accepts canonical names', () => { + assert.equal(resolveAgent('claude-agent-acp').name, 'claude-agent-acp'); +}); + +test('resolveAgent throws with suggestion on unknown', () => { + assert.throws(() => resolveAgent('claud-agent'), /Did you mean/); +}); + +test('every agent declares at least one skillPath', () => { + for (const cfg of Object.values(AGENTS)) { + assert.ok(cfg.skillPaths.length > 0, `${cfg.name} missing skillPaths`); + assert.match(cfg.skillPaths[0], /^\$HOME\//); + } +}); + +test('AGENT_ALIASES does not collide with canonical names', () => { + for (const alias of Object.keys(AGENT_ALIASES)) { + if (AGENTS[alias]) { + assert.equal(alias, AGENT_ALIASES[alias], `alias ${alias} collides`); + } + } +}); diff --git a/tests/case-loader-agent-required.test.ts b/tests/case-loader-agent-required.test.ts new file mode 100644 index 0000000..9fdd1c4 --- /dev/null +++ b/tests/case-loader-agent-required.test.ts @@ -0,0 +1,63 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { mkdtempSync, writeFileSync, rmSync, mkdirSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { loadWorkbenchCase } from '../src/workbench/case-loader.js'; + +test('loadWorkbenchCase throws when agent: is missing', () => { + const dir = mkdtempSync(join(tmpdir(), 'case-test-')); + mkdirSync(join(dir, 'refs'), { recursive: true }); + writeFileSync(join(dir, 'case.yml'), ` +name: test +references: ./refs +task: do nothing +graders: + - name: noop + command: 'true' +model: openrouter/anthropic/claude-haiku-4-5 +`); + assert.throws( + () => loadWorkbenchCase(join(dir, 'case.yml')), + /agent.*required|missing.*agent/i, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchCase throws when agent: is unknown', () => { + const dir = mkdtempSync(join(tmpdir(), 'case-test-')); + mkdirSync(join(dir, 'refs'), { recursive: true }); + writeFileSync(join(dir, 'case.yml'), ` +name: test +references: ./refs +agent: nonexistent-agent +model: x +task: do nothing +graders: + - name: noop + command: 'true' +`); + assert.throws( + () => loadWorkbenchCase(join(dir, 'case.yml')), + /Unknown agent/, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchCase accepts valid agent', () => { + const dir = mkdtempSync(join(tmpdir(), 'case-test-')); + mkdirSync(join(dir, 'refs'), { recursive: true }); + writeFileSync(join(dir, 'case.yml'), ` +name: test +references: ./refs +agent: pi-acp +model: openrouter/anthropic/claude-haiku-4-5 +task: do nothing +graders: + - name: noop + command: 'true' +`); + const c = loadWorkbenchCase(join(dir, 'case.yml')); + assert.equal(c.agent, 'pi-acp'); + rmSync(dir, { recursive: true }); +}); diff --git a/tests/fixtures/acp-traces/claude-sample.jsonl b/tests/fixtures/acp-traces/claude-sample.jsonl new file mode 100644 index 0000000..e539af9 --- /dev/null +++ b/tests/fixtures/acp-traces/claude-sample.jsonl @@ -0,0 +1,11 @@ +{"type":"trace_start","schemaVersion":2,"caseName":"sample","agent":"claude-agent-acp","model":"claude-haiku-4-5","startedAt":"2026-05-25T00:00:00Z"} +{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"0.22","clientCapabilities":{}}} +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":"0.22","agentCapabilities":{},"authMethods":[]}} +{"jsonrpc":"2.0","id":2,"method":"session/new","params":{"cwd":"/work","mcpServers":[]}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"s-1"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"Thinking about it..."}}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"I'll write the file now."}}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"tool_call","toolCallId":"tc-1","title":"Write file","kind":"edit","status":"in_progress","content":[{"type":"content","content":{"type":"text","text":"out.txt"}}]}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"s-1","update":{"sessionUpdate":"tool_call_update","toolCallId":"tc-1","status":"completed","content":[{"type":"content","content":{"type":"text","text":"Wrote 6 bytes"}}]}}} +{"jsonrpc":"2.0","id":3,"method":"session/prompt","params":{"sessionId":"s-1","prompt":[{"type":"text","text":"Write hello to out.txt"}]}} +{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn","usage":{"inputTokens":120,"outputTokens":45,"cacheReadTokens":0,"cacheCreationTokens":0}}} diff --git a/tests/fixtures/mcp-cases/sample-stdio.yml b/tests/fixtures/mcp-cases/sample-stdio.yml new file mode 100644 index 0000000..9704bc7 --- /dev/null +++ b/tests/fixtures/mcp-cases/sample-stdio.yml @@ -0,0 +1,11 @@ +name: sample-with-mcp +agent: claude-agent-acp +model: claude-haiku-4-5 +task: do nothing +graders: + - name: noop + command: 'true' +mcpServers: + calculator: + command: node + args: ['/work/mcp/calculator.mjs'] diff --git a/tests/parse-trace.test.ts b/tests/parse-trace.test.ts new file mode 100644 index 0000000..75f4c80 --- /dev/null +++ b/tests/parse-trace.test.ts @@ -0,0 +1,45 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { readFileSync } from 'node:fs'; +import { + iterMessages, + iterToolCalls, + computeMetrics, + getFinalAssistantMessage, +} from '../src/workbench/parse-trace.js'; + +const fixturePath = 'tests/fixtures/acp-traces/claude-sample.jsonl'; +const trace = readFileSync(fixturePath, 'utf-8'); + +test('iterMessages yields assistant text + thinking', () => { + const msgs = [...iterMessages(trace)]; + const assistant = msgs.find((m) => m.role === 'assistant'); + assert.ok(assistant); + assert.match(assistant.text ?? '', /write the file/); + assert.match(assistant.thinking ?? '', /Thinking about it/); +}); + +test('iterToolCalls pairs tool_call and tool_call_update', () => { + const calls = [...iterToolCalls(trace)]; + assert.equal(calls.length, 1); + assert.equal(calls[0].id, 'tc-1'); + assert.equal(calls[0].kind, 'edit'); + assert.equal(calls[0].status, 'completed'); + assert.match(calls[0].resultText ?? '', /Wrote 6 bytes/); +}); + +test('computeMetrics returns tokens, duration, tool counts', () => { + const m = computeMetrics(trace); + assert.equal(m.tokens.input, 120); + assert.equal(m.tokens.output, 45); + assert.equal(m.tokens.total, 165); + assert.equal(m.toolCalls, 1); + assert.equal(m.editCalls, 1); + assert.equal(m.bashCalls, 0); + assert.equal(m.stopReason, 'end_turn'); +}); + +test('getFinalAssistantMessage returns concatenated assistant text', () => { + const final = getFinalAssistantMessage(trace); + assert.match(final ?? '', /write the file/); +}); diff --git a/tests/regression/baselines/pi-acp-pdf-2026-05-25.json b/tests/regression/baselines/pi-acp-pdf-2026-05-25.json new file mode 100644 index 0000000..f17df7a --- /dev/null +++ b/tests/regression/baselines/pi-acp-pdf-2026-05-25.json @@ -0,0 +1,14 @@ +{ + "captured": "2026-05-25", + "captureStatus": "placeholder", + "captureNote": "Baseline not yet captured under real OPENROUTER_API_KEY. Run `npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials 3` and replace this file with the actual per-case pass rates.", + "agent": "pi-acp", + "model": "openrouter/google/gemini-2.5-flash", + "trials": 3, + "perCasePassRate": { + "extract-pdf-facts": 1.0, + "split-customer-packet": 1.0, + "build-briefing-pdf": 1.0, + "no-pdf-skill-needed": 1.0 + } +} diff --git a/tests/regression/pi-acp-pdf-suite.test.ts b/tests/regression/pi-acp-pdf-suite.test.ts new file mode 100644 index 0000000..098f3b9 --- /dev/null +++ b/tests/regression/pi-acp-pdf-suite.test.ts @@ -0,0 +1,50 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { readFileSync, readdirSync } from 'node:fs'; +import { join } from 'node:path'; +import { execSync } from 'node:child_process'; + +interface Baseline { + captured: string; + captureStatus?: string; + agent: string; + model: string; + trials: number; + perCasePassRate: Record; +} + +const BASELINE_PATH = 'tests/regression/baselines/pi-acp-pdf-2026-05-25.json'; +const baseline = JSON.parse(readFileSync(BASELINE_PATH, 'utf-8')) as Baseline; + +const skip = !process.env.OPENROUTER_API_KEY + ? 'no OPENROUTER_API_KEY env' + : baseline.captureStatus === 'placeholder' + ? `${BASELINE_PATH} is still a placeholder — re-run and replace before enforcing regression` + : false; + +test('pi-acp regression: pdf suite pass rates match baseline within tolerance', { skip }, () => { + execSync( + `npx tsx src/cli.ts run-suite examples/workbench/pdf/suite.yml --trials ${baseline.trials}`, + { stdio: 'inherit' }, + ); + + const resultsDir = 'examples/workbench/pdf/.results'; + const runs = readdirSync(resultsDir).sort(); + const latest = runs[runs.length - 1]; + if (!latest) { + throw new Error(`no run directories under ${resultsDir}`); + } + const data = JSON.parse(readFileSync(join(resultsDir, latest, 'suite-result.json'), 'utf-8')) as { + results: Array<{ caseName: string; trialPassRate: number }>; + }; + + for (const [caseName, expectedRate] of Object.entries(baseline.perCasePassRate)) { + const row = data.results.find((r) => r.caseName === caseName); + assert.ok(row !== undefined, `case ${caseName} not in suite result`); + // Tolerance: ±0.34 (one trial of three can flip) + assert.ok( + Math.abs(row.trialPassRate - expectedRate) <= 0.34, + `case ${caseName}: expected ~${expectedRate}, got ${row.trialPassRate}`, + ); + } +}); diff --git a/tests/smoke-agents/_common.ts b/tests/smoke-agents/_common.ts new file mode 100644 index 0000000..b62026d --- /dev/null +++ b/tests/smoke-agents/_common.ts @@ -0,0 +1,75 @@ +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync, existsSync } from 'node:fs'; +import { tmpdir, homedir } from 'node:os'; +import { join } from 'node:path'; + +import { runDockerWorkbenchCase } from '../../src/workbench/docker-runner.js'; + +export interface SmokeOptions { + agent: string; + model: string; + env: string[]; +} + +export interface SmokeResult { + pass: boolean; + outputContent?: string; + tracePath: string; + resultPath: string; +} + +export async function runSmokeTrial(opts: SmokeOptions): Promise { + const dir = mkdtempSync(join(tmpdir(), 'smoke-')); + try { + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'references', 'README.md'), '# smoke\n', 'utf-8'); + const envBlock = opts.env.length > 0 + ? opts.env.map((name) => ` - ${name}`).join('\n') + : ' []'; + writeFileSync(join(dir, 'case.yml'), [ + `name: smoke-${opts.agent}`, + 'references: ./references', + `agent: ${opts.agent}`, + `model: ${opts.model}`, + 'task: |', + ' Write the literal string "hello" to /work/out.txt.', + ' Then write a one-line description of what you did to /work/findings.txt.', + 'graders:', + ' - name: out-txt-exists-with-hello', + ` command: 'test "$(cat /work/out.txt 2>/dev/null)" = "hello"'`, + 'env:', + envBlock, + 'timeoutSeconds: 120', + ].join('\n'), 'utf-8'); + + const run = await runDockerWorkbenchCase({ + casePath: join(dir, 'case.yml'), + image: 'skill-optimizer-agent:local', + keepWorkspace: true, + }); + + let outputContent: string | undefined; + if (run.workspacePath) { + try { + outputContent = readFileSync(join(run.workspacePath, 'out.txt'), 'utf-8'); + } catch { + outputContent = undefined; + } + } + + const result = JSON.parse(readFileSync(run.resultPath, 'utf-8')) as { pass?: unknown }; + return { + pass: Boolean(result.pass), + outputContent, + tracePath: run.tracePath, + resultPath: run.resultPath, + }; + } finally { + rmSync(dir, { recursive: true, force: true }); + } +} + +export function hasAuth(opts: { envVar?: string; subscriptionFile?: string }): boolean { + if (opts.envVar && process.env[opts.envVar]) return true; + if (opts.subscriptionFile && existsSync(join(homedir(), opts.subscriptionFile))) return true; + return false; +} diff --git a/tests/smoke-agents/claude-agent-acp.smoke.test.ts b/tests/smoke-agents/claude-agent-acp.smoke.test.ts new file mode 100644 index 0000000..0d55567 --- /dev/null +++ b/tests/smoke-agents/claude-agent-acp.smoke.test.ts @@ -0,0 +1,17 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { hasAuth, runSmokeTrial } from './_common.js'; + +const skip = !hasAuth({ envVar: 'ANTHROPIC_API_KEY', subscriptionFile: '.claude/.credentials.json' }) + ? 'no ANTHROPIC_API_KEY env and no ~/.claude/.credentials.json' + : false; + +test('claude-agent-acp smoke: writes hello to out.txt', { skip }, async () => { + const r = await runSmokeTrial({ + agent: 'claude-agent-acp', + model: 'claude-haiku-4-5-20251001', + env: ['ANTHROPIC_API_KEY'], + }); + assert.equal(r.pass, true); + assert.equal(r.outputContent?.trim(), 'hello'); +}); diff --git a/tests/smoke-agents/codex-acp.smoke.test.ts b/tests/smoke-agents/codex-acp.smoke.test.ts new file mode 100644 index 0000000..428e7b8 --- /dev/null +++ b/tests/smoke-agents/codex-acp.smoke.test.ts @@ -0,0 +1,17 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { hasAuth, runSmokeTrial } from './_common.js'; + +const skip = !hasAuth({ envVar: 'OPENAI_API_KEY', subscriptionFile: '.codex/auth.json' }) + ? 'no OPENAI_API_KEY env and no ~/.codex/auth.json' + : false; + +test('codex-acp smoke: writes hello to out.txt', { skip }, async () => { + const r = await runSmokeTrial({ + agent: 'codex-acp', + model: 'gpt-5-mini', + env: ['OPENAI_API_KEY'], + }); + assert.equal(r.pass, true); + assert.equal(r.outputContent?.trim(), 'hello'); +}); diff --git a/tests/smoke-agents/gemini.smoke.test.ts b/tests/smoke-agents/gemini.smoke.test.ts new file mode 100644 index 0000000..cc9bb83 --- /dev/null +++ b/tests/smoke-agents/gemini.smoke.test.ts @@ -0,0 +1,17 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { hasAuth, runSmokeTrial } from './_common.js'; + +const skip = !hasAuth({ envVar: 'GOOGLE_API_KEY', subscriptionFile: '.gemini/oauth_creds.json' }) + ? 'no GOOGLE_API_KEY env and no ~/.gemini/oauth_creds.json' + : false; + +test('gemini smoke: writes hello to out.txt', { skip }, async () => { + const r = await runSmokeTrial({ + agent: 'gemini', + model: 'gemini-3.1-pro-preview', + env: ['GOOGLE_API_KEY'], + }); + assert.equal(r.pass, true); + assert.equal(r.outputContent?.trim(), 'hello'); +}); diff --git a/tests/smoke-agents/opencode.smoke.test.ts b/tests/smoke-agents/opencode.smoke.test.ts new file mode 100644 index 0000000..818479f --- /dev/null +++ b/tests/smoke-agents/opencode.smoke.test.ts @@ -0,0 +1,17 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { hasAuth, runSmokeTrial } from './_common.js'; + +const skip = !hasAuth({ envVar: 'OPENAI_API_KEY' }) + ? 'no OPENAI_API_KEY env' + : false; + +test('opencode smoke: writes hello to out.txt', { skip }, async () => { + const r = await runSmokeTrial({ + agent: 'opencode', + model: 'google/gemini-3.1-pro-preview', + env: ['OPENAI_API_KEY'], + }); + assert.equal(r.pass, true); + assert.equal(r.outputContent?.trim(), 'hello'); +}); diff --git a/tests/smoke-agents/pi-acp.smoke.test.ts b/tests/smoke-agents/pi-acp.smoke.test.ts new file mode 100644 index 0000000..bb4eb09 --- /dev/null +++ b/tests/smoke-agents/pi-acp.smoke.test.ts @@ -0,0 +1,17 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { hasAuth, runSmokeTrial } from './_common.js'; + +const skip = !hasAuth({ envVar: 'OPENROUTER_API_KEY' }) + ? 'no OPENROUTER_API_KEY env' + : false; + +test('pi-acp smoke: writes hello to out.txt', { skip }, async () => { + const r = await runSmokeTrial({ + agent: 'pi-acp', + model: 'openrouter/anthropic/claude-haiku-4-5', + env: ['OPENROUTER_API_KEY'], + }); + assert.equal(r.pass, true); + assert.equal(r.outputContent?.trim(), 'hello'); +}); diff --git a/tests/smoke-skill-distribution.ts b/tests/smoke-skill-distribution.ts index 0594fea..659b574 100644 --- a/tests/smoke-skill-distribution.ts +++ b/tests/smoke-skill-distribution.ts @@ -15,25 +15,37 @@ function readText(relativePath: string): string { return readFileSync(join(root, relativePath), 'utf-8'); } -test('canonical skill follows the portable agent skills contract', () => { - const skillPath = 'skills/skill-optimizer/SKILL.md'; - assert.equal(existsSync(join(root, skillPath)), true); - - const body = readText(skillPath); - assert.match(body, /^---\n[\s\S]*?\n---\n/); - assert.match(body, /^name: skill-optimizer$/m); - assert.match(body, /^description: .+/m); - assert.doesNotMatch(body, /^description: .{1025,}$/m); +test('chain skills follow the portable agent skills contract', () => { + const chainSkillDirs = [ + 'investigate-functionality', + 'investigate-submissions', + 'design-tests', + 'write-tests', + 'validate-tests', + 'run-bench', + 'analyze', + 'improve', + 'validate', + 'autopilot', + ]; + + for (const dir of chainSkillDirs) { + const skillPath = `skills/${dir}/SKILL.md`; + assert.equal(existsSync(join(root, skillPath)), true, `missing ${skillPath}`); + + const body = readText(skillPath); + assert.match(body, /^---\n[\s\S]*?\n---\n/, `${dir} missing frontmatter`); + assert.match(body, new RegExp(`^name: ${dir}$`, 'm'), `${dir} name mismatch`); + assert.match(body, /^description: .+/m, `${dir} missing description`); + assert.doesNotMatch(body, /^description: .{1025,}$/m, `${dir} description too long`); + } }); -test('canonical skill documents current workbench command and live CLI patterns', () => { - const skill = readText('skills/skill-optimizer/SKILL.md'); - const reference = readText('skills/skill-optimizer/references/workbench.md'); +test('shared workbench reference documents current CLI patterns', () => { + const reference = readText('skills/shared/workbench.md'); - for (const text of [skill, reference]) { - assert.doesNotMatch(text, /verify-suite/); - assert.doesNotMatch(text, /runWorkbenchReferenceSolutions/); - } + assert.doesNotMatch(reference, /verify-suite/); + assert.doesNotMatch(reference, /runWorkbenchReferenceSolutions/); assert.match(reference, /Live CLI\/API Skills/); assert.match(reference, /Use dedicated test credentials/); @@ -42,8 +54,8 @@ test('canonical skill documents current workbench command and live CLI patterns' assert.match(reference, /Include a prompt-injection or unsafe-instruction case/); }); -test('workbench reference documents bin directory visibility accurately', () => { - const reference = readText('skills/skill-optimizer/references/workbench.md'); +test('shared workbench reference documents bin directory visibility accurately', () => { + const reference = readText('skills/shared/workbench.md'); assert.match( reference, @@ -98,7 +110,7 @@ test('package metadata does not include broad example result directories', () => ); }); -test('Claude plugin and marketplace metadata point at the canonical skill', () => { +test('Claude plugin and marketplace metadata expose all v1.4 chain skills', () => { const pkg = readJson('package.json'); const plugin = readJson('.claude-plugin/plugin.json'); const marketplace = readJson('.claude-plugin/marketplace.json'); @@ -114,7 +126,22 @@ test('Claude plugin and marketplace metadata point at the canonical skill', () = assert.equal(marketplace.plugins[0].name, 'skill-optimizer'); assert.equal(marketplace.plugins[0].description, pluginDescription); assert.equal(marketplace.plugins[0].source, './'); - assert.deepEqual(marketplace.plugins[0].skills, ['./skills/skill-optimizer']); + assert.deepEqual(marketplace.plugins[0].skills, [ + './skills/investigate-functionality', + './skills/investigate-submissions', + './skills/design-tests', + './skills/write-tests', + './skills/validate-tests', + './skills/run-bench', + './skills/analyze', + './skills/improve', + './skills/validate', + './skills/autopilot', + ]); + for (const skillPath of marketplace.plugins[0].skills) { + const resolved = skillPath.replace(/^\.\//, ''); + assert.equal(existsSync(join(root, resolved, 'SKILL.md')), true, `missing ${resolved}/SKILL.md`); + } }); test('Codex and Cursor plugin metadata point at the canonical skill', () => { @@ -166,7 +193,7 @@ test('OpenCode plugin registers the canonical skills directory', async () => { assert.deepEqual(config.skills.paths, [join(root, 'skills')]); }); -test('Gemini extension metadata points at the canonical context file', () => { +test('Gemini extension metadata imports the v1.4 chain overview + workbench reference', () => { const pkg = readJson('package.json'); const extension = readJson('gemini-extension.json'); const geminiInstructions = readText('GEMINI.md'); @@ -178,5 +205,6 @@ test('Gemini extension metadata points at the canonical context file', () => { assert.match(geminiInstructions, /^@\.\/AGENTS\.md$/m); assert.match(geminiInstructions, /^@\.\/README\.md$/m); assert.match(geminiInstructions, /^@\.\/CONTRIBUTING\.md$/m); - assert.match(geminiInstructions, /^@\.\/skills\/skill-optimizer\/SKILL\.md$/m); + assert.match(geminiInstructions, /^@\.\/skills\/shared\/workflow\.md$/m); + assert.match(geminiInstructions, /^@\.\/skills\/shared\/workbench\.md$/m); }); diff --git a/tests/smoke-workbench-case.ts b/tests/smoke-workbench-case.ts index 2b2d6c2..648ca42 100644 --- a/tests/smoke-workbench-case.ts +++ b/tests/smoke-workbench-case.ts @@ -25,6 +25,7 @@ test('type supports minimal fields', () => { graders: [ { name: 'merged-output', command: 'node $CASE/check.js' }, ], + agent: 'pi-acp', }; assert.equal(minimal.name, 'merge-pdfs'); @@ -37,6 +38,7 @@ test('YAML case loads and resolves relative references', () => { const casePath = writeCaseFile(root, 'case.yaml', [ 'name: merge-pdfs', 'references: ./references', + 'agent: pi-acp', 'task: Merge the PDFs in inputs/ into outputs/book.pdf.', 'graders:', ' - name: merged-output', @@ -62,6 +64,7 @@ test('JSON case loads', () => { const casePath = writeCaseFile(root, 'case.json', JSON.stringify({ name: 'merge-pdfs-json', references: './refs', + agent: 'pi-acp', task: 'Merge the PDFs.', graders: [ { name: 'merged-output', command: 'node $CASE/checks/merge-pdfs.js' }, @@ -84,6 +87,7 @@ test('YAML case loads MCP server definitions', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: mcp-docs', 'references: ./references', + 'agent: pi-acp', 'task: Use the configured MCP docs server.', 'graders:', ' - name: output', @@ -132,6 +136,7 @@ test('invalid MCP server without transport throws', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: mcp-docs', 'references: ./references', + 'agent: pi-acp', 'task: Use MCP.', 'graders:', ' - name: output', @@ -157,6 +162,7 @@ test('MCP service without matching MCP server throws', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: mcp-docs', 'references: ./references', + 'agent: pi-acp', 'task: Use MCP.', 'graders:', ' - name: output', @@ -184,6 +190,7 @@ test('invalid MCP service command reports mcpServices field', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: mcp-docs', 'references: ./references', + 'agent: pi-acp', 'task: Use MCP.', 'graders:', ' - name: output', @@ -212,6 +219,7 @@ test('MCP service port is rejected because mcpServers URL owns the port', () => const casePath = writeCaseFile(root, 'case.yml', [ 'name: mcp-docs', 'references: ./references', + 'agent: pi-acp', 'task: Use MCP.', 'graders:', ' - name: output', @@ -243,6 +251,7 @@ test('defaults are applied', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: merge-pdfs', 'references: ./references', + 'agent: pi-acp', 'task: Merge files', 'graders:', ' - name: merged-output', @@ -267,6 +276,7 @@ test('invalid case model ref is rejected while loading', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: merge-pdfs', 'references: ./references', + 'agent: pi-acp', 'task: Merge files', 'model: anthropic/claude-3-5-haiku-latest', 'graders:', @@ -288,6 +298,7 @@ test('invalid missing references throws', () => { try { const casePath = writeCaseFile(root, 'case.yaml', [ 'name: merge-pdfs', + 'agent: pi-acp', 'task: Merge files', 'graders:', ' - name: merged-output', @@ -310,6 +321,7 @@ test('invalid non-array env throws', () => { const casePath = writeCaseFile(root, 'case.json', JSON.stringify({ name: 'merge-pdfs', references: './references', + agent: 'pi-acp', task: 'Merge files', graders: [ { name: 'merged-output', command: 'node $CASE/checks/merge-pdfs.js' }, @@ -335,6 +347,7 @@ test('invalid env variable names are rejected', () => { const casePath = writeCaseFile(root, `case-${envName.length}.json`, JSON.stringify({ name: 'merge-pdfs', references: './references', + agent: 'pi-acp', task: 'Merge files', graders: [ { name: 'merged-output', command: 'node $CASE/checks/merge-pdfs.js' }, @@ -359,6 +372,7 @@ test('valid env variable names still load', () => { const casePath = writeCaseFile(root, 'case.json', JSON.stringify({ name: 'merge-pdfs', references: './references', + agent: 'pi-acp', task: 'Merge files', graders: [ { name: 'merged-output', command: 'node $CASE/checks/merge-pdfs.js' }, @@ -380,6 +394,7 @@ test('invalid missing graders throws', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: merge-pdfs', 'references: ./references', + 'agent: pi-acp', 'task: Merge files', ].join('\n')); @@ -446,6 +461,7 @@ test('invalid grader command throws', () => { const casePath = writeCaseFile(root, 'case.yml', [ 'name: merge-pdfs', 'references: ./references', + 'agent: pi-acp', 'task: Merge files', 'graders:', ' - name: merged-output', diff --git a/tests/smoke-workbench-container.ts b/tests/smoke-workbench-container.ts deleted file mode 100644 index b6f9362..0000000 --- a/tests/smoke-workbench-container.ts +++ /dev/null @@ -1,241 +0,0 @@ -import assert from 'node:assert/strict'; -import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; -import { tmpdir } from 'node:os'; -import { join } from 'node:path'; -import { test } from 'node:test'; - -import { - buildAgentSystemPrompt, - buildContainerWorkbenchEnv, - parseContainerRunnerArgs, - prepareWorkbenchDirectory, - runContainerWorkbenchCase, - runAgentPromptWithTimeout, - writeBestEffortTrace, -} from '../src/workbench/container-runner.js'; -import { createTraceRecorder } from '../src/workbench/trace.js'; - -test('buildAgentSystemPrompt describes operating constraints without eval/sandbox hints', () => { - const prompt = buildAgentSystemPrompt(); - - assert.match(prompt, /Current working directory is \/work/); - assert.match(prompt, /Do not use global pip installs/); - assert.match(prompt, /python -m venv \/work\/\.venv/); - assert.match(prompt, /Write all outputs under \/work/); - assert.doesNotMatch(prompt, /sandbox/i); - assert.doesNotMatch(prompt, /skill\/reference/i); - assert.doesNotMatch(prompt, /grader/i); - assert.doesNotMatch(prompt, /expected answer/i); - assert.doesNotMatch(prompt, /suite metadata/i); - assert.doesNotMatch(prompt, /\/case/); - assert.doesNotMatch(prompt, /Task:/); -}); - -test('buildContainerWorkbenchEnv exposes CASE as the mounted case directory', () => { - const env = buildContainerWorkbenchEnv({ - casePath: '/case/case.yml', - workDir: '/work', - resultsDir: '/results', - baseEnv: {}, - }); - - assert.equal(env.CASE, '/case'); - assert.equal(env.WORK, '/work'); - assert.equal(env.RESULTS, '/results'); -}); - -test('buildContainerWorkbenchEnv prepends work and case bin to PATH when present', () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-env-')); - try { - const caseDir = join(root, 'case'); - const workDir = join(root, 'work'); - mkdirSync(join(caseDir, 'bin'), { recursive: true }); - mkdirSync(workDir, { recursive: true }); - - const env = buildContainerWorkbenchEnv({ - casePath: join(caseDir, 'case.yml'), - workDir, - resultsDir: join(root, 'results'), - baseEnv: { PATH: '/usr/bin' }, - }); - - assert.equal(env.PATH, `${join(workDir, 'bin')}:${join(caseDir, 'bin')}:/usr/bin`); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); - -test('prepareWorkbenchDirectory copies references then optional workspace seed', () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-prepare-')); - try { - const referencesDir = join(root, 'references'); - const workspaceDir = join(root, 'workspace'); - const workDir = join(root, 'work'); - mkdirSync(referencesDir, { recursive: true }); - mkdirSync(workspaceDir, { recursive: true }); - mkdirSync(workDir, { recursive: true }); - writeFileSync(join(workDir, 'stale.txt'), 'stale\n', 'utf-8'); - writeFileSync(join(referencesDir, 'SKILL.md'), '# Skill\n', 'utf-8'); - writeFileSync(join(workspaceDir, 'seed.txt'), 'seed\n', 'utf-8'); - - prepareWorkbenchDirectory({ referencesDir, workspaceDir, workDir }); - - assert.equal(existsSync(join(workDir, 'stale.txt')), false); - assert.equal(readFileSync(join(workDir, 'SKILL.md'), 'utf-8'), '# Skill\n'); - assert.equal(readFileSync(join(workDir, 'seed.txt'), 'utf-8'), 'seed\n'); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); - -test('parseContainerRunnerArgs reads optional MCP config path for agent mode', () => { - const parsed = parseContainerRunnerArgs([ - '--agent', - '--case-name', 'mcp-case', - '--model', 'openrouter/google/gemini-2.5-flash', - '--task-base64', Buffer.from('Use MCP.', 'utf-8').toString('base64'), - '--timeout-seconds', '600', - '--work', '/work', - '--results', '/tmp/workbench-results', - '--mcp-config', '/work/mcporter.json', - ]); - - assert.equal(parsed.mode, 'agent'); - assert.equal(parsed.mcpConfigPath, '/work/mcporter.json'); -}); - -test('runContainerWorkbenchCase restores global WORK/RESULTS/MCPORTER_CONFIG after agent mode failure', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-env-restore-')); - const workDir = join(root, 'work'); - const resultsDir = join(root, 'results'); - mkdirSync(workDir, { recursive: true }); - mkdirSync(resultsDir, { recursive: true }); - - const previousWork = process.env.WORK; - const previousResults = process.env.RESULTS; - const previousMcporterConfig = process.env.MCPORTER_CONFIG; - - process.env.WORK = 'existing-work'; - process.env.RESULTS = 'existing-results'; - delete process.env.MCPORTER_CONFIG; - - try { - const exitCode = await runContainerWorkbenchCase([ - '--agent', - '--case-name', 'env-restore', - '--model', 'openrouter/google/gemini-2.5-flash', - '--task-base64', Buffer.from('test', 'utf-8').toString('base64'), - '--timeout-seconds', '1', - '--work', workDir, - '--results', resultsDir, - '--mcp-config', join(root, 'missing-mcporter-config.json'), - ]); - - assert.equal(exitCode, 1); - assert.equal(process.env.WORK, 'existing-work'); - assert.equal(process.env.RESULTS, 'existing-results'); - assert.equal(process.env.MCPORTER_CONFIG, undefined); - } finally { - if (previousWork === undefined) { - delete process.env.WORK; - } else { - process.env.WORK = previousWork; - } - - if (previousResults === undefined) { - delete process.env.RESULTS; - } else { - process.env.RESULTS = previousResults; - } - - if (previousMcporterConfig === undefined) { - delete process.env.MCPORTER_CONFIG; - } else { - process.env.MCPORTER_CONFIG = previousMcporterConfig; - } - rmSync(root, { recursive: true, force: true }); - } -}); - -test('runAgentPromptWithTimeout rejects when agent exceeds timeout', async () => { - await assert.rejects( - runAgentPromptWithTimeout({ prompt: () => new Promise(() => {}) }, 'task', 0.001), - /Agent timed out after 0.001 seconds/, - ); -}); - -test('runAgentPromptWithTimeout rejects when agent ends with provider error', async () => { - await assert.rejects( - runAgentPromptWithTimeout({ - prompt: async () => undefined, - state: { - messages: [ - { role: 'assistant', content: [], stopReason: 'error', errorMessage: 'Upstream request failed' }, - ], - }, - }, 'task', 1), - /Upstream request failed/, - ); -}); - -test('writeBestEffortTrace writes trace from available session messages', () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-trace-')); - try { - const tracePath = join(root, 'trace.jsonl'); - const wrote = writeBestEffortTrace({ - tracePath, - caseName: 'partial-trace', - model: 'openrouter/test/model', - startedAt: '2026-04-27T10:11:12.000Z', - endedAt: '2026-04-27T10:11:13.000Z', - session: { - state: { - messages: [ - { role: 'user', content: [{ type: 'text', text: 'hello' }] }, - ], - }, - }, - }); - - assert.equal(wrote, true); - const lines = readFileSync(tracePath, 'utf-8').trim().split('\n').map((line) => JSON.parse(line) as { type: string; caseName?: string }); - assert.equal(lines[0]?.caseName, 'partial-trace'); - assert.equal(lines.filter((line) => line.type === 'message').length, 1); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); - -test('writeBestEffortTrace prefers recorded Pi events when available', () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-event-trace-')); - try { - const tracePath = join(root, 'trace.jsonl'); - const recorder = createTraceRecorder({ now: () => '2026-04-27T10:11:12.500Z' }); - recorder.record({ - type: 'tool_execution_start', - toolCallId: 'call-1', - toolName: 'bash', - args: { command: 'npm test' }, - }); - - const wrote = writeBestEffortTrace({ - tracePath, - caseName: 'event-trace', - model: 'openrouter/test/model', - startedAt: '2026-04-27T10:11:12.000Z', - endedAt: '2026-04-27T10:11:13.000Z', - recorder, - session: { state: { messages: [] } }, - }); - - assert.equal(wrote, true); - const lines = readFileSync(tracePath, 'utf-8').trim().split('\n').map((line) => JSON.parse(line) as { - type: string; - arguments?: { command?: string }; - }); - assert.equal(lines[1]?.type, 'tool_call'); - assert.equal(lines[1]?.arguments?.command, 'npm test'); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); diff --git a/tests/smoke-workbench-docker-runner.ts b/tests/smoke-workbench-docker-runner.ts index 1285595..c32313b 100644 --- a/tests/smoke-workbench-docker-runner.ts +++ b/tests/smoke-workbench-docker-runner.ts @@ -5,14 +5,8 @@ import { join } from 'node:path'; import { test } from 'node:test'; import { - buildDockerAgentCommand, - buildDockerGradeCommand, - buildDockerMcpServiceCommand, - buildDockerMcpServiceProbeCommand, - buildDockerSetupCommand, packageRootFromModuleUrl, prepareDockerWorkbenchRun, - startMcpServices, } from '../src/workbench/docker-runner.js'; test('packageRootFromModuleUrl resolves repo root independently of cwd', () => { @@ -27,13 +21,13 @@ test('packageRootFromModuleUrl resolves repo root independently of cwd', () => { }); test('workbench image runs agents as non-root with venv-only pip installs', () => { - const dockerfile = readFileSync(join(process.cwd(), 'docker', 'workbench-runner.Dockerfile'), 'utf-8'); + const dockerfile = readFileSync(join(process.cwd(), 'docker', 'skill-optimizer-agent.Dockerfile'), 'utf-8'); assert.match(dockerfile, /useradd .* agent/); assert.match(dockerfile, /USER agent/); assert.match(dockerfile, /ENTRYPOINT \["node", "\/app\/dist\/workbench\/container-runner\.js"\]/); assert.match(dockerfile, /PIP_REQUIRE_VIRTUALENV=1/); - assert.match(dockerfile, /PATH="\/app\/node_modules\/\.bin:\/work\/\.venv\/bin:/); + assert.match(dockerfile, /PATH="\/opt\/skill-opt\/bin:\/app\/node_modules\/\.bin:\/work\/\.venv\/bin:/); assert.doesNotMatch(dockerfile, /PIP_BREAK_SYSTEM_PACKAGES/); }); @@ -48,6 +42,7 @@ test('prepareDockerWorkbenchRun writes results under case .results and keeps bun writeFileSync(join(sourceCaseDir, 'case.yml'), [ 'name: pdf-merge', 'references: ./references', + 'agent: pi-acp', 'task: Merge PDFs.', 'graders:', ' - name: merged-output', @@ -98,6 +93,7 @@ test('prepareDockerWorkbenchRun bundles case support directories', () => { writeFileSync(join(sourceCaseDir, 'case.yml'), [ 'name: support-case', 'references: ./references', + 'agent: pi-acp', 'task: Test support dirs.', 'graders:', ' - name: passes', @@ -136,6 +132,7 @@ test('prepareDockerWorkbenchRun writes isolated mcporter config for MCP servers' writeFileSync(join(sourceCaseDir, 'case.yml'), [ 'name: mcp-case', 'references: ./references', + 'agent: pi-acp', 'task: Use MCP.', 'graders:', ' - name: output', @@ -194,6 +191,7 @@ test('prepareDockerWorkbenchRun bundles hidden MCP service support outside work' writeFileSync(join(sourceCaseDir, 'case.yml'), [ 'name: mcp-service-case', 'references: ./references', + 'agent: pi-acp', 'task: Use MCP.', 'graders:', ' - name: output', @@ -223,186 +221,6 @@ test('prepareDockerWorkbenchRun bundles hidden MCP service support outside work' } }); -test('startMcpServices records services started before a later service fails', async () => { - const startedContainers: string[] = []; - - await assert.rejects( - startMcpServices({ - image: 'skill-optimizer-workbench:local', - networkName: 'skill-optimizer-mcp-test', - caseDir: '/tmp/case', - tempDir: '/tmp/skill-optimizer-workbench-test', - services: { - ok: { command: 'node', args: ['server.mjs'] }, - fail: { command: 'node', args: ['server.mjs'] }, - }, - repoRoot: '/tmp/repo', - startedContainers, - runCommand: async (command) => ({ - exitCode: command.includes('-fail') ? 7 : 0, - stdout: '', - stderr: command.includes('-fail') ? 'boom' : '', - }), - }), - /Failed to start MCP service fail/, - ); - - assert.deepEqual(startedContainers, ['skill-optimizer-mcp-skill-optimizer-workbench-test-ok']); -}); - -test('setup docker command mounts case and work before agent phase', () => { - const command = buildDockerSetupCommand({ - image: 'skill-optimizer-workbench:local', - caseDir: '/tmp/case', - workDir: '/tmp/work', - envNames: [], - }); - - assert.match(command, /--setup/); - assert.match(command, /-v '\/tmp\/case:\/case:ro'/); - assert.match(command, /-v '\/tmp\/work:\/work:rw'/); - assert.doesNotMatch(command, /\/results/); - assert.doesNotMatch(command, /docker\.sock/); -}); - -test('agent docker command mounts only work and uses sandbox hardening flags', () => { - const command = buildDockerAgentCommand({ - image: 'skill-optimizer-workbench:local', - containerName: 'skill-optimizer-agent-test', - workDir: '/tmp/work', - caseName: 'extract-pdf-facts', - model: 'openrouter/google/gemini-2.5-flash', - task: 'Read the PDF and write answer.json.', - timeoutSeconds: 600, - envNames: ['OPENROUTER_API_KEY'], - }); - - assert.match(command, /--agent/); - assert.match(command, /--name 'skill-optimizer-agent-test'/); - assert.match(command, /-v '\/tmp\/work:\/work:rw'/); - assert.match(command, /--workdir \/work/); - assert.match(command, /-e PATH=\/work\/bin:\/app\/node_modules\/\.bin:\/work\/\.venv\/bin:\/usr\/local\/sbin:\/usr\/local\/bin:\/usr\/sbin:\/usr\/bin:\/sbin:\/bin/); - assert.match(command, /--cap-drop=ALL/); - assert.match(command, /--security-opt no-new-privileges/); - assert.match(command, /-e OPENROUTER_API_KEY/); - assert.doesNotMatch(command, /\/case/); - assert.doesNotMatch(command, /\/results/); - assert.doesNotMatch(command, /docker\.sock/); -}); - -test('agent docker command passes optional appended system prompt', () => { - const command = buildDockerAgentCommand({ - image: 'skill-optimizer-workbench:local', - containerName: 'skill-optimizer-agent-test', - workDir: '/tmp/work', - caseName: 'prompted-case', - model: 'openrouter/google/gemini-2.5-flash', - task: 'Write output.txt.', - timeoutSeconds: 600, - envNames: [], - appendSystemPrompt: 'Prefer simple shell commands when possible.', - }); - - assert.match(command, /--append-system-prompt-base64/); - assert.match(command, new RegExp(Buffer.from('Prefer simple shell commands when possible.', 'utf-8').toString('base64'))); -}); - -test('agent docker command passes optional MCP config path', () => { - const command = buildDockerAgentCommand({ - image: 'skill-optimizer-workbench:local', - containerName: 'skill-optimizer-agent-test', - workDir: '/tmp/work', - caseName: 'mcp-case', - model: 'openrouter/google/gemini-2.5-flash', - task: 'Use MCP.', - timeoutSeconds: 600, - envNames: [], - mcpConfigPath: '/work/mcporter.json', - }); - - assert.match(command, /-e MCPORTER_CONFIG=\/work\/mcporter\.json/); - assert.match(command, /--mcp-config '\/work\/mcporter\.json'/); -}); - -test('agent docker command joins optional MCP network', () => { - const command = buildDockerAgentCommand({ - image: 'skill-optimizer-workbench:local', - containerName: 'skill-optimizer-agent-test', - workDir: '/tmp/work', - caseName: 'mcp-case', - model: 'openrouter/google/gemini-2.5-flash', - task: 'Use MCP.', - timeoutSeconds: 600, - envNames: [], - networkName: 'skill-optimizer-mcp-test', - }); - - assert.match(command, /--network 'skill-optimizer-mcp-test'/); -}); - -test('MCP service docker command mounts hidden service files outside agent work', () => { - const command = buildDockerMcpServiceCommand({ - image: 'skill-optimizer-workbench:local', - containerName: 'skill-optimizer-mcp-test-calculator', - networkName: 'skill-optimizer-mcp-test', - alias: 'calculator', - mcpDir: '/tmp/case/mcp', - command: 'node', - args: ['server.mjs'], - }); - - assert.match(command, /-v '\/tmp\/case\/mcp:\/mcp:ro'/); - assert.match(command, /--workdir \/mcp/); - assert.match(command, /--network-alias 'calculator'/); - assert.doesNotMatch(command, /\/work/); -}); - -test('MCP service docker command does not forward case env vars', () => { - const command = buildDockerMcpServiceCommand({ - image: 'skill-optimizer-workbench:local', - containerName: 'skill-optimizer-mcp-test-calculator', - networkName: 'skill-optimizer-mcp-test', - alias: 'calculator', - mcpDir: '/tmp/case/mcp', - command: 'node', - args: ['server.mjs'], - }); - - assert.doesNotMatch(command, /-e OPENROUTER_API_KEY/); -}); - -test('MCP service probe command verifies service through mcporter on private network', () => { - const command = buildDockerMcpServiceProbeCommand({ - image: 'skill-optimizer-workbench:local', - networkName: 'skill-optimizer-mcp-test', - workDir: '/tmp/work', - serverName: 'calculator', - }); - - assert.match(command, /--network 'skill-optimizer-mcp-test'/); - assert.match(command, /-v '\/tmp\/work:\/work:rw'/); - assert.match(command, /mcporter --config \/work\/mcporter\.json --root \/work list/); - assert.match(command, /calculator/); - assert.match(command, /--schema/); -}); - -test('grade docker command mounts case after agent phase', () => { - const command = buildDockerGradeCommand({ - image: 'skill-optimizer-workbench:local', - caseDir: '/tmp/case', - workDir: '/tmp/work', - resultsDir: '/tmp/results', - envNames: [], - }); - - assert.match(command, /--grade/); - assert.match(command, /-v '\/tmp\/case:\/case:ro'/); - assert.match(command, /-v '\/tmp\/work:\/work:rw'/); - assert.match(command, /-v '\/tmp\/results:\/results:rw'/); - assert.match(command, /--cap-drop=ALL/); - assert.match(command, /--security-opt no-new-privileges/); -}); - test('prepareDockerWorkbenchRun honors --out as the results root', () => { const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-out-')); try { @@ -412,6 +230,7 @@ test('prepareDockerWorkbenchRun honors --out as the results root', () => { writeFileSync(join(sourceCaseDir, 'case.yml'), [ 'name: pdf-merge', 'references: ./references', + 'agent: pi-acp', 'task: Merge PDFs.', 'graders:', ' - name: merged-output', diff --git a/tests/smoke-workbench-metrics.ts b/tests/smoke-workbench-metrics.ts index 96eeae3..178d8c2 100644 --- a/tests/smoke-workbench-metrics.ts +++ b/tests/smoke-workbench-metrics.ts @@ -1,75 +1,63 @@ import assert from 'node:assert/strict'; +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; import { test } from 'node:test'; -import { buildTrialSummary, buildWorkbenchMetrics } from '../src/workbench/metrics.js'; -import type { WorkbenchResult, WorkbenchTrace } from '../src/workbench/types.js'; +import { buildTrialSummary, buildWorkbenchMetricsFromTrace } from '../src/workbench/metrics.js'; +import type { WorkbenchResult } from '../src/workbench/types.js'; -test('buildWorkbenchMetrics counts tool calls and sums usage', () => { - const trace: WorkbenchTrace = { - caseName: 'metrics-case', - model: 'openrouter/test/model', - startedAt: '2026-04-27T10:00:00.000Z', - endedAt: '2026-04-27T10:00:02.500Z', - entries: [ - { type: 'message', role: 'user', text: 'task' }, - { - type: 'message', - role: 'assistant', - usage: { - input: 10, - output: 5, - cacheRead: 2, - cacheWrite: 1, - totalTokens: 18, - cost: { input: 0.1, output: 0.2, cacheRead: 0.03, cacheWrite: 0.04, total: 0.37 }, - }, - stopReason: 'toolUse', - }, - { type: 'tool_call', name: 'bash', arguments: { command: 'npm test' } }, - { type: 'tool_call', name: 'read', arguments: { path: 'file.ts' } }, - { type: 'tool_result', name: 'bash', text: 'ok' }, - { type: 'message', role: 'assistant', text: 'done', stopReason: 'stop' }, - ], - }; +const FIXTURE_JSONL = [ + JSON.stringify({ type: 'trace_start', schemaVersion: 2, caseName: 'm', agent: 'claude-agent-acp', model: 'x', startedAt: '2026-05-25T00:00:00Z', endedAt: '2026-05-25T00:00:02.500Z' }), + JSON.stringify({ jsonrpc: '2.0', method: 'session/update', params: { sessionId: 's', update: { sessionUpdate: 'agent_message_chunk', content: { type: 'text', text: 'final answer' } } } }), + JSON.stringify({ jsonrpc: '2.0', method: 'session/update', params: { sessionId: 's', update: { sessionUpdate: 'tool_call', toolCallId: 'tc1', kind: 'execute', content: [{ type: 'content', content: { type: 'text', text: 'firecrawl search "x"' } }] } } }), + JSON.stringify({ jsonrpc: '2.0', method: 'session/update', params: { sessionId: 's', update: { sessionUpdate: 'tool_call', toolCallId: 'tc2', kind: 'read' } } }), + JSON.stringify({ jsonrpc: '2.0', id: 3, result: { stopReason: 'end_turn', usage: { inputTokens: 10, outputTokens: 5, cacheReadTokens: 2, cacheCreationTokens: 1 } } }), +].join('\n'); - const metrics = buildWorkbenchMetrics(trace); - assert.equal(metrics.durationMs, 2500); - assert.equal(metrics.turns, 3); - assert.equal(metrics.toolCalls, 2); - assert.equal(metrics.toolResults, 1); - assert.equal(metrics.bashCalls, 1); - assert.equal(metrics.readCalls, 1); - assert.equal(metrics.stopReason, 'stop'); - assert.equal(metrics.tokens.total, 18); - assert.equal(metrics.cost.total, 0.37); +function withFixture(): { path: string; cleanup: () => void } { + const dir = mkdtempSync(join(tmpdir(), 'metrics-test-')); + const path = join(dir, 'trace.jsonl'); + writeFileSync(path, FIXTURE_JSONL); + return { path, cleanup: () => rmSync(dir, { recursive: true }) }; +} + +test('buildWorkbenchMetricsFromTrace counts tool calls and sums usage', () => { + const { path, cleanup } = withFixture(); + try { + const m = buildWorkbenchMetricsFromTrace(path); + assert.equal(m.durationMs, 2500); + assert.equal(m.toolCalls, 2); + assert.equal(m.toolResults, m.toolCalls); + assert.equal(m.bashCalls, 1); + assert.equal(m.readCalls, 1); + assert.equal(m.stopReason, 'end_turn'); + assert.equal(m.tokens.input, 10); + assert.equal(m.tokens.output, 5); + assert.equal(m.tokens.total, 18); + // Cost field must not exist on the metrics object + assert.equal((m as unknown as { cost?: unknown }).cost, undefined); + } finally { cleanup(); } }); test('buildTrialSummary extracts final text, failed graders, and bash commands', () => { - const trace: WorkbenchTrace = { - caseName: 'summary-case', - model: 'openrouter/test/model', - startedAt: '2026-04-27T10:00:00.000Z', - endedAt: '2026-04-27T10:00:01.000Z', - entries: [ - { type: 'tool_call', name: 'bash', arguments: { command: 'firecrawl search "x"' } }, - { type: 'message', role: 'assistant', text: 'final answer', stopReason: 'stop' }, - ], - }; - const result: WorkbenchResult = { - caseName: 'summary-case', - model: 'openrouter/test/model', - pass: false, - score: 0.5, - evidence: ['missing output'], - graders: [ - { name: 'uses-tool', command: 'true', pass: true, score: 1, evidence: [] }, - { name: 'saves-output', command: 'false', pass: false, score: 0, evidence: ['missing output'] }, - ], - }; - - const summary = buildTrialSummary({ trace, result }); - assert.equal(summary.finalAssistantMessage, 'final answer'); - assert.deepEqual(summary.failedGraders, ['saves-output']); - assert.deepEqual(summary.bashCommands, ['firecrawl search "x"']); - assert.equal(summary.metrics.bashCalls, 1); + const { path, cleanup } = withFixture(); + try { + const result: WorkbenchResult = { + caseName: 'summary-case', + model: 'x', + pass: false, + score: 0.5, + evidence: ['missing output'], + graders: [ + { name: 'uses-tool', command: 'true', pass: true, score: 1, evidence: [] }, + { name: 'saves-output', command: 'false', pass: false, score: 0, evidence: ['missing output'] }, + ], + }; + const summary = buildTrialSummary({ tracePath: path, result }); + assert.equal(summary.finalAssistantMessage, 'final answer'); + assert.deepEqual(summary.failedGraders, ['saves-output']); + assert.deepEqual(summary.bashCommands, ['firecrawl search "x"']); + assert.equal(summary.metrics.bashCalls, 1); + } finally { cleanup(); } }); diff --git a/tests/smoke-workbench-models.ts b/tests/smoke-workbench-models.ts index 7527321..b516d5b 100644 --- a/tests/smoke-workbench-models.ts +++ b/tests/smoke-workbench-models.ts @@ -24,155 +24,32 @@ test('slugModelRef creates filesystem-safe model directories', () => { assert.equal(slugModelRef('openrouter/meta-llama/llama-3.3-70b-instruct:free'), 'openrouter-meta-llama-llama-3.3-70b-instruct-free'); }); -test('runWorkbenchCase writes aggregate output for multi-model runs', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-models-')); - const previousExitCode = process.exitCode; - try { - const casePath = join(root, 'case.yml'); - const outDir = join(root, 'results'); - const calls: Array<{ model?: string; resultsDir?: string }> = []; - mkdirSync(outDir, { recursive: true }); - writeFileSync(casePath, 'name: model-case\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); - - process.exitCode = undefined; - await runWorkbenchCase( - { - casePath, - outDir, - models: ['openrouter/google/gemini-2.5-flash', 'openrouter/openai/gpt-5.4'], - }, - { - runDockerWorkbenchCase: async (options) => { - calls.push({ model: options.model, resultsDir: options.resultsDir }); - assert.ok(options.resultsDir); - mkdirSync(options.resultsDir, { recursive: true }); - const resultPath = join(options.resultsDir, 'result.json'); - const tracePath = join(options.resultsDir, 'trace.jsonl'); - const pass = options.model !== 'openrouter/openai/gpt-5.4'; - writeFileSync(resultPath, JSON.stringify({ pass, score: pass ? 1 : 0, evidence: [options.model] }), 'utf-8'); - writeFileSync(tracePath, JSON.stringify({ entries: [] }), 'utf-8'); - return { - tempDir: join(root, 'temp'), - caseDir: join(root, 'temp', 'case'), - bundledCasePath: join(root, 'temp', 'case', 'case.yml'), - workDir: join(root, 'temp', 'work'), - resultsDir: options.resultsDir, - resultPath, - tracePath, - cleanup: () => {}, - }; - }, - now: new Date('2026-04-27T10:11:12.000Z'), - }, - ); - - const runResultPath = join(outDir, '20260427-101112', 'run-result.json'); - assert.ok(existsSync(runResultPath)); - const aggregate = JSON.parse(readFileSync(runResultPath, 'utf-8')) as { - summary: { total: number; passed: number; failed: number; passRate: number; totalTrials: number; passedTrials: number; failedTrials: number }; - results: Array<{ model: string; passHatK: boolean; trials: Array<{ resultPath: string; tracePath: string }> }>; - }; - assert.deepEqual(calls.map((call) => call.model), [ - 'openrouter/google/gemini-2.5-flash', - 'openrouter/openai/gpt-5.4', - ]); - assert.equal(aggregate.summary.total, 2); - assert.equal(aggregate.summary.passed, 1); - assert.equal(aggregate.summary.failed, 1); - assert.equal(aggregate.summary.passRate, 0.5); - assert.equal(aggregate.summary.totalTrials, 2); - assert.equal(aggregate.summary.passedTrials, 1); - assert.equal(aggregate.summary.failedTrials, 1); - assert.equal(aggregate.results[0]?.trials[0]?.resultPath, 'trials/openrouter-google-gemini-2.5-flash--001/result.json'); - assert.equal(aggregate.results[1]?.trials[0]?.tracePath, 'trials/openrouter-openai-gpt-5.4--001/trace.jsonl'); - assert.equal(process.exitCode, 1); - } finally { - process.exitCode = previousExitCode; - rmSync(root, { recursive: true, force: true }); - } -}); - -test('runWorkbenchCase writes aggregate output when --models has one model', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-one-model-')); - const previousExitCode = process.exitCode; - try { - const casePath = join(root, 'case.yml'); - const outDir = join(root, 'results'); - mkdirSync(outDir, { recursive: true }); - writeFileSync(casePath, 'name: model-case\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); - - process.exitCode = undefined; - await runWorkbenchCase( - { - casePath, - outDir, - models: ['openrouter/google/gemini-2.5-flash'], - }, - { - runDockerWorkbenchCase: async (options) => { - assert.ok(options.resultsDir); - mkdirSync(options.resultsDir, { recursive: true }); - const resultPath = join(options.resultsDir, 'result.json'); - const tracePath = join(options.resultsDir, 'trace.jsonl'); - writeFileSync(resultPath, JSON.stringify({ pass: true, score: 1, evidence: [] }), 'utf-8'); - writeFileSync(tracePath, JSON.stringify({ entries: [] }), 'utf-8'); - return { - tempDir: join(root, 'temp'), - caseDir: join(root, 'temp', 'case'), - bundledCasePath: join(root, 'temp', 'case', 'case.yml'), - workDir: join(root, 'temp', 'work'), - resultsDir: options.resultsDir, - resultPath, - tracePath, - cleanup: () => {}, - }; - }, - now: new Date('2026-04-27T10:11:12.000Z'), - }, - ); - - const runResultPath = join(outDir, '20260427-101112', 'run-result.json'); - assert.ok(existsSync(runResultPath)); - assert.ok(existsSync(join(outDir, '20260427-101112', 'trials', 'openrouter-google-gemini-2.5-flash--001', 'result.json'))); - assert.equal(process.exitCode, undefined); - } finally { - process.exitCode = previousExitCode; - rmSync(root, { recursive: true, force: true }); - } -}); - -test('runWorkbenchCase --trials uses the case model when no model override is provided', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-case-model-')); +test('runWorkbenchCase fans out trials per agent+model dir', async () => { + const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-runs-')); const previousExitCode = process.exitCode; try { const casePath = join(root, 'case.yml'); const outDir = join(root, 'results'); const refsDir = join(root, 'refs'); - const calls: Array<{ model?: string }> = []; mkdirSync(refsDir, { recursive: true }); mkdirSync(outDir, { recursive: true }); writeFileSync( casePath, - 'name: model-case\nreferences: ./refs\nmodel: openrouter/openai/gpt-5.4\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', + 'name: model-case\nreferences: ./refs\nagent: pi-acp\nmodel: openrouter/openai/gpt-5.4\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8', ); process.exitCode = undefined; await runWorkbenchCase( - { - casePath, - outDir, - trials: 2, - }, + { casePath, outDir, trials: 2 }, { runDockerWorkbenchCase: async (options) => { - calls.push({ model: options.model }); assert.ok(options.resultsDir); mkdirSync(options.resultsDir, { recursive: true }); const resultPath = join(options.resultsDir, 'result.json'); const tracePath = join(options.resultsDir, 'trace.jsonl'); writeFileSync(resultPath, JSON.stringify({ pass: true, score: 1, evidence: [] }), 'utf-8'); - writeFileSync(tracePath, JSON.stringify({ entries: [] }), 'utf-8'); + writeFileSync(tracePath, '{}\n', 'utf-8'); return { tempDir: join(root, 'temp'), caseDir: join(root, 'temp', 'case'), @@ -188,11 +65,15 @@ test('runWorkbenchCase --trials uses the case model when no model override is pr }, ); - assert.deepEqual(calls.map((call) => call.model), [ - 'openrouter/openai/gpt-5.4', - 'openrouter/openai/gpt-5.4', - ]); - assert.ok(existsSync(join(outDir, '20260427-101112', 'trials', 'openrouter-openai-gpt-5.4--002', 'result.json'))); + const runResultPath = join(outDir, '20260427-101112', 'run-result.json'); + assert.ok(existsSync(runResultPath)); + const aggregate = JSON.parse(readFileSync(runResultPath, 'utf-8')) as { + runs: Array<{ agent: string; model: string }>; + results: Array<{ agent: string; model: string }>; + }; + assert.deepEqual(aggregate.runs, [{ agent: 'pi-acp', model: 'openrouter/openai/gpt-5.4' }]); + assert.equal(aggregate.results[0]?.agent, 'pi-acp'); + assert.ok(existsSync(join(outDir, '20260427-101112', 'trials', 'pi-acp--openrouter-openai-gpt-5.4--002', 'result.json'))); assert.equal(process.exitCode, undefined); } finally { process.exitCode = previousExitCode; diff --git a/tests/smoke-workbench-pi-agent.ts b/tests/smoke-workbench-pi-agent.ts deleted file mode 100644 index f2c01ee..0000000 --- a/tests/smoke-workbench-pi-agent.ts +++ /dev/null @@ -1,159 +0,0 @@ -import assert from 'node:assert/strict'; -import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'; -import { tmpdir } from 'node:os'; -import { join } from 'node:path'; -import { test } from 'node:test'; - -import { createWorkbenchPiResourceLoader, createWorkbenchPiSession, createWorkbenchPiTools, stripSensitiveEnv } from '../src/workbench/pi-agent.js'; - -function toolText(result: unknown): string { - const content = (result as { content?: Array<{ text?: string }> }).content ?? []; - return content.map((item) => item.text ?? '').join(''); -} - -test('createWorkbenchPiTools enables coding plus repo-scale search tools', () => { - const tools = createWorkbenchPiTools('/work'); - const names = tools.map((tool) => tool.name).sort(); - - assert.deepEqual(names, ['bash', 'edit', 'find', 'grep', 'ls', 'read', 'write']); -}); - -test('stripSensitiveEnv preserves all case-allowed credentials for tool subprocesses', () => { - const env = stripSensitiveEnv({ - OPENROUTER_API_KEY: 'secret', - OPENAI_API_KEY: 'secret', - GOOGLE_WORKSPACE_CLI_TOKEN: 'gws-token', - GOOGLE_WORKSPACE_CLI_CLIENT_SECRET: 'gws-secret', - GOOGLE_WORKSPACE_CLI_CREDENTIALS_FILE: '/work/gws-credentials.json', - MODEL_AUTH_FILE: '/run/secrets/model-auth.json', - WHATSAPP_ACCESS_TOKEN: 'secret', - DASHBOARD_TOKEN_SECRET: 'secret', - PATH: '/usr/bin', - WORK: '/work', - }); - - assert.equal(env.OPENROUTER_API_KEY, 'secret'); - assert.equal(env.OPENAI_API_KEY, 'secret'); - assert.equal(env.GOOGLE_WORKSPACE_CLI_TOKEN, 'gws-token'); - assert.equal(env.GOOGLE_WORKSPACE_CLI_CLIENT_SECRET, 'gws-secret'); - assert.equal(env.GOOGLE_WORKSPACE_CLI_CREDENTIALS_FILE, '/work/gws-credentials.json'); - assert.equal(env.MODEL_AUTH_FILE, '/run/secrets/model-auth.json'); - assert.equal(env.WHATSAPP_ACCESS_TOKEN, 'secret'); - assert.equal(env.DASHBOARD_TOKEN_SECRET, 'secret'); - assert.equal(env.PATH, '/usr/bin'); - assert.equal(env.WORK, '/work'); -}); - -test('createWorkbenchPiResourceLoader discovers a root SKILL.md from references', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-skill-')); - try { - writeFileSync(root + '/SKILL.md', [ - '---', - 'name: pdf', - 'description: PDF merge instructions', - '---', - '', - '# PDF Skill', - ].join('\n'), 'utf-8'); - mkdirSync(join(root, 'inputs')); - - const loader = await createWorkbenchPiResourceLoader({ cwd: root }); - const loaded = loader.getSkills().skills.map((skill) => skill.name); - - assert.ok(loaded.includes('pdf')); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); - -test('createWorkbenchPiResourceLoader appends suite prompt after workbench prompt', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-append-prompt-')); - try { - const loader = await createWorkbenchPiResourceLoader({ - cwd: root, - appendSystemPrompt: 'Prefer simple shell commands when possible.', - }); - - const appended = loader.getAppendSystemPrompt().join('\n\n'); - assert.match(appended, /Operating environment:/); - assert.match(appended, /Prefer simple shell commands when possible\./); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); - -test('createWorkbenchPiResourceLoader documents MCP command when configured', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-mcp-prompt-')); - try { - const loader = await createWorkbenchPiResourceLoader({ - cwd: root, - mcpConfigPath: '/work/mcporter.json', - }); - - const appended = loader.getAppendSystemPrompt().join('\n\n'); - assert.match(appended, /`mcp` is available on PATH/); - assert.match(appended, /Run `mcp list --schema`/); - assert.doesNotMatch(appended, /calculator\.add/); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); - -test('createWorkbenchPiTools passes process env through bash subprocesses', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-tool-env-')); - const previousSecret = process.env.WORKBENCH_AGENT_SECRET; - try { - process.env.WORKBENCH_AGENT_SECRET = 'agent-secret'; - const bashTool = createWorkbenchPiTools(root).find((tool) => tool.name === 'bash'); - assert.ok(bashTool); - - const result = await bashTool.execute( - 'call-1', - { command: 'printf "%s" "$WORKBENCH_AGENT_SECRET"', timeout: 5 }, - new AbortController().signal, - ); - - assert.equal(toolText(result), 'agent-secret'); - } finally { - if (previousSecret === undefined) { - delete process.env.WORKBENCH_AGENT_SECRET; - } else { - process.env.WORKBENCH_AGENT_SECRET = previousSecret; - } - rmSync(root, { recursive: true, force: true }); - } -}); - -test('createWorkbenchPiSession leaves runtime API key env available after session creation', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-session-env-')); - const previousApiKey = process.env.OPENROUTER_API_KEY; - try { - process.env.OPENROUTER_API_KEY = 'test-openrouter-key'; - const created = await createWorkbenchPiSession({ - cwd: root, - modelRef: 'openrouter/google/gemini-2.5-flash', - }); - - assert.equal(process.env.OPENROUTER_API_KEY, 'test-openrouter-key'); - (created.session as { dispose?: () => void }).dispose?.(); - } finally { - if (previousApiKey === undefined) { - delete process.env.OPENROUTER_API_KEY; - } else { - process.env.OPENROUTER_API_KEY = previousApiKey; - } - rmSync(root, { recursive: true, force: true }); - } -}); - -test('createWorkbenchPiSession rejects non-OpenRouter model refs', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-openrouter-only-')); - try { - await assert.rejects( - () => createWorkbenchPiSession({ cwd: root, modelRef: 'direct/model' }), - /only supports OpenRouter/, - ); - } finally { - rmSync(root, { recursive: true, force: true }); - } -}); diff --git a/tests/smoke-workbench-run-case.ts b/tests/smoke-workbench-run-case.ts index b9bfacd..1f0e70b 100644 --- a/tests/smoke-workbench-run-case.ts +++ b/tests/smoke-workbench-run-case.ts @@ -18,6 +18,19 @@ test('runWorkbenchCase preserves failing result as process exitCode 1', async () score: 0, evidence: ['expected failure'], }), 'utf-8'); + + mkdirSync(join(root, 'refs'), { recursive: true }); + writeFileSync(join(root, 'case.yml'), [ + 'name: exit-code-case', + 'references: ./refs', + 'agent: pi-acp', + 'model: openrouter/anthropic/claude-haiku-4-5', + 'task: noop', + 'graders:', + ' - name: passes', + ' command: "true"', + ].join('\n'), 'utf-8'); + process.exitCode = undefined; await runWorkbenchCase( @@ -43,12 +56,24 @@ test('runWorkbenchCase preserves failing result as process exitCode 1', async () } }); -test('runWorkbenchCaseFromCli rejects invalid --model before loading the case', async () => { +test('runWorkbenchCaseFromCli rejects non-openrouter --model when the case agent is pi-acp', async () => { const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-run-case-model-')); try { + mkdirSync(join(root, 'refs'), { recursive: true }); + writeFileSync(join(root, 'case.yml'), [ + 'name: model-validate-case', + 'references: ./refs', + 'agent: pi-acp', + 'model: openrouter/anthropic/claude-haiku-4-5', + 'task: noop', + 'graders:', + ' - name: passes', + ' command: "true"', + ].join('\n'), 'utf-8'); + await assert.rejects( runWorkbenchCaseFromCli([ - join(root, 'missing-case.yml'), + join(root, 'case.yml'), '--model', 'anthropic/claude-3-5-haiku-latest', ]), diff --git a/tests/smoke-workbench-suite.ts b/tests/smoke-workbench-suite.ts index 7ea1792..e28e127 100644 --- a/tests/smoke-workbench-suite.ts +++ b/tests/smoke-workbench-suite.ts @@ -4,18 +4,19 @@ import { tmpdir } from 'node:os'; import { join } from 'node:path'; import { test } from 'node:test'; -import { loadWorkbenchSuite } from '../src/workbench/suite-loader.js'; import { runWorkbenchSuite, runWorkbenchSuiteFromCli } from '../src/workbench/run-suite.js'; +import { loadWorkbenchSuite } from '../src/workbench/suite-loader.js'; -test('loadWorkbenchSuite resolves case paths and validates models', () => { +test('loadWorkbenchSuite resolves case paths and validates runs', () => { const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-suite-load-')); try { const suitePath = join(root, 'suite.yml'); mkdirSync(join(root, 'cases', 'missing-index'), { recursive: true }); writeFileSync(suitePath, [ 'name: supabase-postgres-best-practices', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - cases/missing-index/case.yml', ].join('\n'), 'utf-8'); @@ -23,7 +24,7 @@ test('loadWorkbenchSuite resolves case paths and validates models', () => { const suite = loadWorkbenchSuite(suitePath); assert.equal(suite.name, 'supabase-postgres-best-practices'); - assert.deepEqual(suite.models, ['openrouter/google/gemini-2.5-flash']); + assert.deepEqual(suite.runs, [{ agent: 'pi-acp', model: 'openrouter/google/gemini-2.5-flash' }]); assert.deepEqual(suite.casePaths, [join(root, 'cases', 'missing-index', 'case.yml')]); } finally { rmSync(root, { recursive: true, force: true }); @@ -42,10 +43,12 @@ test('loadWorkbenchSuite supports inline cases with suite defaults', () => { 'env:', ' - OPENROUTER_API_KEY', 'timeoutSeconds: 123', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - name: async-parallel', + ' agent: pi-acp', ' task: Make this faster', ' graders:', ' - name: async-parallel', @@ -77,8 +80,9 @@ test('loadWorkbenchSuite applies and merges MCP defaults for inline cases', () = writeFileSync(suitePath, [ 'name: mcp-suite', 'references: ./references', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'mcpServers:', ' context7:', ' baseUrl: https://mcp.context7.com/mcp', @@ -88,6 +92,7 @@ test('loadWorkbenchSuite applies and merges MCP defaults for inline cases', () = ' - mcp/default-server.mjs', 'cases:', ' - name: mcp-inline', + ' agent: pi-acp', ' task: Use MCP.', ' mcpServers:', ' context7:', @@ -124,8 +129,9 @@ test('loadWorkbenchSuite reads suite appendSystemPrompt', () => { 'name: prompted-suite', 'appendSystemPrompt: |', ' Prefer simple shell commands when possible.', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - cases/noop/case.yml', ].join('\n'), 'utf-8'); @@ -144,8 +150,9 @@ test('loadWorkbenchSuite rejects suite artifacts defaults', () => { const suitePath = join(root, 'suite.yml'); writeFileSync(suitePath, [ 'name: artifact-suite', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'artifacts:', ' - output.json', 'cases:', @@ -175,9 +182,11 @@ test('runWorkbenchSuite writes case-model matrix aggregate output', async () => writeFileSync(caseB, 'name: partial-index\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); writeFileSync(suitePath, [ 'name: supabase-postgres-best-practices', - 'models:', - ' - openrouter/google/gemini-2.5-flash', - ' - openrouter/openai/gpt-5.4', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', + ' - agent: pi-acp', + ' model: openrouter/openai/gpt-5.4', 'cases:', ' - cases/missing-index/case.yml', ' - cases/partial-index/case.yml', @@ -224,8 +233,8 @@ test('runWorkbenchSuite writes case-model matrix aggregate output', async () => assert.equal(aggregate.summary.totalTrials, 4); assert.equal(aggregate.summary.passedTrials, 3); assert.equal(aggregate.summary.failedTrials, 1); - assert.equal(aggregate.results[0]?.trials[0]?.resultPath, 'trials/missing-index--openrouter-google-gemini-2.5-flash--001/result.json'); - assert.equal(aggregate.results[3]?.trials[0]?.resultPath, 'trials/partial-index--openrouter-openai-gpt-5.4--001/result.json'); + assert.equal(aggregate.results[0]?.trials[0]?.resultPath, 'trials/missing-index--pi-acp--openrouter-google-gemini-2.5-flash--001/result.json'); + assert.equal(aggregate.results[3]?.trials[0]?.resultPath, 'trials/partial-index--pi-acp--openrouter-openai-gpt-5.4--001/result.json'); assert.equal(process.exitCode, 1); } finally { process.exitCode = previousExitCode; @@ -246,8 +255,9 @@ test('runWorkbenchSuite passes suite appendSystemPrompt to every trial', async ( 'name: prompted-suite', 'appendSystemPrompt: |', ' Prefer simple shell commands when possible.', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - cases/prompted/case.yml', ].join('\n'), 'utf-8'); @@ -293,24 +303,24 @@ test('runWorkbenchSuiteFromCli rejects model overrides because suites own models ); }); -test('runWorkbenchSuite missing models error references suite models only', async () => { - const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-suite-missing-models-')); +test('runWorkbenchSuite missing runs error references suite runs only', async () => { + const root = mkdtempSync(join(tmpdir(), 'skill-opt-workbench-suite-missing-runs-')); try { const suitePath = join(root, 'suite.yml'); - const casePath = join(root, 'cases', 'no-models', 'case.yml'); - mkdirSync(join(root, 'cases', 'no-models'), { recursive: true }); - writeFileSync(casePath, 'name: no-models\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); + const casePath = join(root, 'cases', 'no-runs', 'case.yml'); + mkdirSync(join(root, 'cases', 'no-runs'), { recursive: true }); + writeFileSync(casePath, 'name: no-runs\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); writeFileSync(suitePath, [ - 'name: no-models-suite', + 'name: no-runs-suite', 'cases:', - ' - cases/no-models/case.yml', + ' - cases/no-runs/case.yml', ].join('\n'), 'utf-8'); await assert.rejects( () => runWorkbenchSuite({ suitePath }), (error: unknown) => { assert.ok(error instanceof Error); - assert.match(error.message, /suite\.yml|models/); + assert.match(error.message, /suite\.yml|runs/); assert.doesNotMatch(error.message, /--models/); return true; }, @@ -330,8 +340,9 @@ test('runWorkbenchSuite rejects non-integer programmatic concurrency', async () writeFileSync(casePath, 'name: invalid-concurrency\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); writeFileSync(suitePath, [ 'name: invalid-concurrency-suite', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - cases/invalid-concurrency/case.yml', ].join('\n'), 'utf-8'); @@ -356,8 +367,9 @@ test('runWorkbenchSuite honors concurrency for independent trials', async () => writeFileSync(casePath, 'name: parallel\nreferences: ./refs\ntask: Test\ngraders:\n - name: passes\n command: "true"\n', 'utf-8'); writeFileSync(suitePath, [ 'name: parallel-suite', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - cases/parallel/case.yml', ].join('\n'), 'utf-8'); diff --git a/tests/smoke-workbench-trace.ts b/tests/smoke-workbench-trace.ts deleted file mode 100644 index 7c5781b..0000000 --- a/tests/smoke-workbench-trace.ts +++ /dev/null @@ -1,176 +0,0 @@ -import { buildWorkbenchTrace, createTraceCollector, createTraceRecorder } from '../src/workbench/trace.js'; - -let passed = 0; -let failed = 0; - -async function test(name: string, fn: () => Promise | void) { - try { - await fn(); - passed++; - console.log(` ✓ ${name}`); - } catch (error: any) { - failed++; - console.log(` ✗ ${name}`); - console.log(` ${error.message}`); - } -} - -function assert(condition: boolean, message: string) { - if (!condition) throw new Error(`Assertion failed: ${message}`); -} - -function assertEqual(actual: T, expected: T, message: string) { - if (actual !== expected) { - throw new Error(`${message}: expected ${JSON.stringify(expected)}, got ${JSON.stringify(actual)}`); - } -} - -console.log('\n=== Workbench Trace Smoke Tests ===\n'); - -await test('buildWorkbenchTrace stores a deduped interaction timeline', () => { - const trace = buildWorkbenchTrace({ - caseName: 'case-1', - model: 'openrouter/google/gemini-2.5-flash', - startedAt: '2026-01-01T00:00:00.000Z', - endedAt: '2026-01-01T00:00:02.000Z', - messages: [ - { role: 'user', content: [{ type: 'text', text: 'Do the task' }] }, - { - role: 'assistant', - content: [ - { type: 'thinking', thinking: 'Need to inspect files' }, - { type: 'text', text: 'I will read the skill.' }, - { type: 'toolCall', id: 'call-1', name: 'read', arguments: { path: '/work/SKILL.md' } }, - ], - usage: { totalTokens: 10 }, - }, - { - role: 'toolResult', - toolCallId: 'call-1', - toolName: 'read', - content: [{ type: 'text', text: '# Skill' }], - isError: false, - }, - ], - }); - - assertEqual(trace.caseName, 'case-1', 'trace should preserve caseName'); - assertEqual(trace.entries.length, 4, 'trace should normalize messages into entries'); - assertEqual(trace.entries[0].type, 'message', 'first entry should be user message'); - assertEqual(trace.entries[1].type, 'message', 'second entry should be assistant message'); - assertEqual(trace.entries[2].type, 'tool_call', 'third entry should be tool call'); - assertEqual(trace.entries[3].type, 'tool_result', 'fourth entry should be tool result'); - assert(!('events' in trace), 'trace should not include raw streaming events'); - assert(!('messages' in trace), 'trace should not duplicate raw messages'); -}); - -await test('buildWorkbenchTrace preserves assistant provider error messages', () => { - const trace = buildWorkbenchTrace({ - caseName: 'case-error', - model: 'openrouter/google/gemini-2.5-flash', - startedAt: '2026-01-01T00:00:00.000Z', - endedAt: '2026-01-01T00:00:02.000Z', - messages: [ - { - role: 'assistant', - content: [], - stopReason: 'error', - errorMessage: 'Provider returned 500', - }, - ], - }); - - const entry = trace.entries[0] as { type: string; errorMessage?: string; stopReason?: unknown }; - assertEqual(entry.type, 'message', 'entry should be a message'); - assertEqual(entry.stopReason, 'error', 'entry should preserve stop reason'); - assertEqual(entry.errorMessage, 'Provider returned 500', 'entry should preserve provider error message'); -}); - -await test('createTraceCollector records arbitrary events in order', () => { - const collector = createTraceCollector(); - collector.record({ step: 1 }); - collector.record('tool-call'); - collector.record(42); - - assertEqual(collector.events.length, 3, 'collector should record all events'); - assertEqual((collector.events[0] as { step?: number }).step, 1, 'collector should preserve object payload'); - assertEqual(collector.events[1], 'tool-call', 'collector should preserve string payload'); - assertEqual(collector.events[2], 42, 'collector should preserve numeric payload'); -}); - -await test('createTraceRecorder captures Pi session events and normalized entries', () => { - const recorder = createTraceRecorder({ now: () => '2026-01-01T00:00:01.000Z' }); - - recorder.record({ - type: 'message_end', - message: { - role: 'assistant', - content: [{ type: 'text', text: 'I will run the command.' }], - stopReason: 'toolUse', - }, - }); - recorder.record({ - type: 'tool_execution_start', - toolCallId: 'call-1', - toolName: 'bash', - args: { command: 'firecrawl search browser --scrape' }, - }); - recorder.record({ - type: 'tool_execution_end', - toolCallId: 'call-1', - toolName: 'bash', - result: { content: [{ type: 'text', text: 'ok' }] }, - isError: false, - }); - - const trace = recorder.toTrace({ - caseName: 'case-events', - model: 'openrouter/test/model', - startedAt: '2026-01-01T00:00:00.000Z', - endedAt: '2026-01-01T00:00:02.000Z', - }); - - assertEqual(trace.events?.length, 3, 'trace should preserve raw-ish Pi events'); - assertEqual(trace.events?.[0]?.timestamp, '2026-01-01T00:00:01.000Z', 'trace events should have capture timestamps'); - assertEqual(trace.entries.length, 3, 'trace should derive normalized entries from events'); - assertEqual(trace.entries[0].type, 'message', 'first entry should be assistant message'); - assertEqual(trace.entries[1].type, 'tool_call', 'second entry should be tool call'); - assertEqual(trace.entries[2].type, 'tool_result', 'third entry should be tool result'); - assertEqual( - ((trace.entries[1] as { arguments?: { command?: string } }).arguments)?.command, - 'firecrawl search browser --scrape', - 'tool call entry should preserve bash command', - ); -}); - -await test('createTraceRecorder preserves session messages when events are partial', () => { - const recorder = createTraceRecorder({ now: () => '2026-01-01T00:00:01.000Z' }); - - recorder.record({ - type: 'tool_execution_start', - toolCallId: 'call-1', - toolName: 'bash', - args: { command: 'node parse-pdf.mjs' }, - }); - - const trace = recorder.toTrace({ - caseName: 'partial-events', - model: 'openrouter/test/model', - startedAt: '2026-01-01T00:00:00.000Z', - endedAt: '2026-01-01T00:00:02.000Z', - messages: [ - { role: 'user', content: [{ type: 'text', text: 'Extract the PDF facts.' }] }, - { role: 'assistant', content: [{ type: 'text', text: 'I will parse the PDF.' }] }, - ], - }); - - assertEqual(trace.entries.length, 3, 'trace should keep session messages plus partial event entries'); - assertEqual(trace.entries[0].type, 'message', 'first entry should be a session message'); - assertEqual((trace.entries[0] as { role?: string }).role, 'user', 'first session message should be user'); - assertEqual(trace.entries[1].type, 'message', 'second entry should be a session message'); - assertEqual((trace.entries[1] as { role?: string }).role, 'assistant', 'second session message should be assistant'); - assertEqual(trace.entries[2].type, 'tool_call', 'partial tool event should still be included'); -}); - -console.log(`\n${passed} passed, ${failed} failed\n`); -process.exit(failed > 0 ? 1 : 0); diff --git a/tests/smoke-workbench-trials.ts b/tests/smoke-workbench-trials.ts index ea49b51..18e1ca7 100644 --- a/tests/smoke-workbench-trials.ts +++ b/tests/smoke-workbench-trials.ts @@ -47,10 +47,12 @@ test('runWorkbenchSuite writes trial directories and case-model aggregates', asy writeFileSync(suitePath, [ 'name: trial-suite', 'references: ./references', - 'models:', - ' - openrouter/google/gemini-2.5-flash', + 'runs:', + ' - agent: pi-acp', + ' model: openrouter/google/gemini-2.5-flash', 'cases:', ' - name: trial-case', + ' agent: pi-acp', ' task: Test trials', ' graders:', ' - name: passes', @@ -85,9 +87,9 @@ test('runWorkbenchSuite writes trial directories and case-model aggregates', asy ); const runRoot = join(outDir, '20260427-101112'); - assert.ok(existsSync(join(runRoot, 'trials', 'trial-case--openrouter-google-gemini-2.5-flash--001', 'result.json'))); - assert.ok(existsSync(join(runRoot, 'trials', 'trial-case--openrouter-google-gemini-2.5-flash--002', 'result.json'))); - assert.ok(existsSync(join(runRoot, 'trials', 'trial-case--openrouter-google-gemini-2.5-flash--003', 'result.json'))); + assert.ok(existsSync(join(runRoot, 'trials', 'trial-case--pi-acp--openrouter-google-gemini-2.5-flash--001', 'result.json'))); + assert.ok(existsSync(join(runRoot, 'trials', 'trial-case--pi-acp--openrouter-google-gemini-2.5-flash--002', 'result.json'))); + assert.ok(existsSync(join(runRoot, 'trials', 'trial-case--pi-acp--openrouter-google-gemini-2.5-flash--003', 'result.json'))); const aggregate = JSON.parse(readFileSync(join(runRoot, 'suite-result.json'), 'utf-8')) as { summary: { totalTrials: number; passedTrials: number; failedTrials: number; trialPassRate: number; meanScore: number }; diff --git a/tests/suite-loader-runs.test.ts b/tests/suite-loader-runs.test.ts new file mode 100644 index 0000000..5b98cd0 --- /dev/null +++ b/tests/suite-loader-runs.test.ts @@ -0,0 +1,127 @@ +import { strict as assert } from 'node:assert'; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { test } from 'node:test'; + +import { loadWorkbenchSuite } from '../src/workbench/suite-loader.js'; + +test('loadWorkbenchSuite parses runs: matrix', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +runs: + - agent: claude-agent-acp + model: claude-haiku-4-5 + - agent: pi-acp + model: openrouter/anthropic/claude-haiku-4-5 +cases: + - name: test-case + agent: pi-acp + task: do nothing + graders: + - name: noop + command: 'true' +`); + const suite = loadWorkbenchSuite(join(dir, 'suite.yml')); + assert.equal(suite.runs.length, 2); + assert.equal(suite.runs[0].agent, 'claude-agent-acp'); + assert.equal(suite.runs[1].agent, 'pi-acp'); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchSuite rejects legacy `models:` shape', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +models: + - openrouter/anthropic/claude-haiku-4-5 +cases: [] +`); + assert.throws( + () => loadWorkbenchSuite(join(dir, 'suite.yml')), + /runs:.*replaced.*models:|legacy.*models|models.*replaced/i, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchSuite validates each run.agent', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +runs: + - agent: bogus + model: x +cases: [] +`); + assert.throws( + () => loadWorkbenchSuite(join(dir, 'suite.yml')), + /Unknown agent.*bogus/, + ); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchSuite propagates suite-level skillUnderTest to inline cases', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +skillUnderTest: + slug: my-skill + hostPath: /tmp/my-skill-source +runs: + - agent: pi-acp + model: openrouter/anthropic/claude-haiku-4-5 +cases: + - name: case-a + agent: pi-acp + task: noop + graders: + - name: noop + command: 'true' + - name: case-b + agent: pi-acp + task: noop + graders: + - name: noop + command: 'true' + skillUnderTest: + slug: my-skill + hostPath: /tmp/override +`); + const suite = loadWorkbenchSuite(join(dir, 'suite.yml')); + assert.equal(suite.cases.length, 2); + // case-a inherits the suite-level skillUnderTest + assert.equal(suite.cases[0].case?.skillUnderTest?.slug, 'my-skill'); + assert.equal(suite.cases[0].case?.skillUnderTest?.hostPath, '/tmp/my-skill-source'); + // case-b overrides the suite-level value + assert.equal(suite.cases[1].case?.skillUnderTest?.hostPath, '/tmp/override'); + rmSync(dir, { recursive: true }); +}); + +test('loadWorkbenchSuite rejects malformed skillUnderTest', () => { + const dir = mkdtempSync(join(tmpdir(), 'suite-test-')); + mkdirSync(join(dir, 'references'), { recursive: true }); + writeFileSync(join(dir, 'suite.yml'), ` +name: my-suite +references: ./references +skillUnderTest: + slug: '' +runs: + - agent: pi-acp + model: openrouter/anthropic/claude-haiku-4-5 +cases: [] +`); + assert.throws( + () => loadWorkbenchSuite(join(dir, 'suite.yml')), + /skillUnderTest\.slug.*non-empty/, + ); + rmSync(dir, { recursive: true }); +}); diff --git a/tests/trials-aggregate.test.ts b/tests/trials-aggregate.test.ts new file mode 100644 index 0000000..5ace945 --- /dev/null +++ b/tests/trials-aggregate.test.ts @@ -0,0 +1,30 @@ +import { test } from 'node:test'; +import { strict as assert } from 'node:assert'; +import { aggregateTrials } from '../src/workbench/trials.js'; + +test('aggregateTrials sums tokens and duration', () => { + const result = aggregateTrials([ + { trial: 1, pass: true, score: 1, tokens: 100, durationMs: 1000 }, + { trial: 2, pass: false, score: 0, tokens: 200, durationMs: 2000 }, + { trial: 3, pass: true, score: 1, tokens: 150, durationMs: 1500 }, + ]); + assert.equal(result.totalTokens, 450); + assert.equal(result.totalDurationMs, 4500); + assert.equal(result.passedTrials, 2); +}); + +test('aggregateTrials treats missing tokens/durationMs as zero', () => { + const result = aggregateTrials([ + { trial: 1, pass: true, score: 1 }, + { trial: 2, pass: false, score: 0, tokens: 100 }, + ]); + assert.equal(result.totalTokens, 100); + assert.equal(result.totalDurationMs, 0); +}); + +test('aggregateTrials returns zero totals for an empty input', () => { + const result = aggregateTrials([]); + assert.equal(result.totalTokens, 0); + assert.equal(result.totalDurationMs, 0); + assert.equal(result.totalTrials, 0); +});