diff --git a/.github/workflows/commitperclip-review.yml b/.github/workflows/commitperclip-review.yml
index 6b7122772941..4f8262f5f318 100644
--- a/.github/workflows/commitperclip-review.yml
+++ b/.github/workflows/commitperclip-review.yml
@@ -15,6 +15,11 @@ permissions:
jobs:
review:
+ # Upstream's review bot, and only upstream can run it. Dependency Review needs
+ # Advanced Security, which a private fork does not have, and the token step needs
+ # `COMMITPERCLIP_KEY` — upstream's GitHub App, not a secret a fork can hold. On the
+ # fork this job could only ever fail, so it is scoped to the repository that owns it.
+ if: github.repository == 'paperclipai/paperclip'
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml
index 68de5269d8af..ef4aa39dc602 100644
--- a/.github/workflows/pr.yml
+++ b/.github/workflows/pr.yml
@@ -27,10 +27,17 @@ jobs:
- name: Block manual lockfile edits
if: >-
github.head_ref != 'chore/refresh-lockfile' &&
+ !startsWith(github.head_ref, 'sync/upstream-') &&
github.event.pull_request.user.login != 'dependabot[bot]'
run: |
# Diff the PR branch against its merge base so recent base-branch commits
# do not masquerade as changes made by the PR itself.
+ #
+ # `sync/upstream-*` is exempt: those PRs merge upstream/master into the
+ # fork's integration branch, so upstream's own lockfile arrives with the
+ # merge. Blocking it would make an upstream sync unmergeable. The property
+ # that matters — the committed lockfile agrees with the manifests — is
+ # still enforced downstream by `pnpm install --frozen-lockfile`.
changed="$(git diff --name-only "${{ github.event.pull_request.base.sha }}...${{ github.event.pull_request.head.sha }}")"
if printf '%s\n' "$changed" | grep -qx 'pnpm-lock.yaml'; then
echo "Do not commit pnpm-lock.yaml in pull requests. CI owns lockfile updates."
@@ -151,26 +158,31 @@ jobs:
include:
# The server suite is pinned to maxWorkers=1 (server/vitest.config.ts),
# so it can only be parallelized across runners. Shard it to keep this
- # lane off the PR critical path. Four shards because the suite has
- # grown to ~880s of serial vitest wall time (run 30723165356,
- # 2026-08-01): at three shards the worst shard ran 313s and was the
- # slowest check in the whole PR run; four brings each shard to ~220s.
+ # lane off the PR critical path. Five shards because the suite has
+ # grown to ~946s of serial vitest wall time (run 30930345729,
+ # 2026-08-04): at four shards the worst shard ran 311s and was the
+ # slowest check in the whole PR run; five brings each shard to ~196s
+ # of suite time (~240s job), level with the other ~250-300s lanes.
- group: general-server
- group_label: server (1/4)
+ group_label: server (1/5)
shard_index: 0
- shard_count: 4
+ shard_count: 5
- group: general-server
- group_label: server (2/4)
+ group_label: server (2/5)
shard_index: 1
- shard_count: 4
+ shard_count: 5
- group: general-server
- group_label: server (3/4)
+ group_label: server (3/5)
shard_index: 2
- shard_count: 4
+ shard_count: 5
- group: general-server
- group_label: server (4/4)
+ group_label: server (4/5)
shard_index: 3
- shard_count: 4
+ shard_count: 5
+ - group: general-server
+ group_label: server (5/5)
+ shard_index: 4
+ shard_count: 5
- group: general-workspaces-a
group_label: workspaces-a
- group: general-workspaces-b
@@ -272,18 +284,25 @@ jobs:
fail-fast: false
matrix:
include:
+ # A successful PR run on 2026-08-04 (30876682788) spent 256s in
+ # serialized shard 2/4, making its 305s job the run's slowest check.
+ # Five shards reduce the measured 739s suite total to about 148s per
+ # runner before setup overhead.
- shard_index: 0
- shard_count: 4
- shard_label: 1/4
+ shard_count: 5
+ shard_label: 1/5
- shard_index: 1
- shard_count: 4
- shard_label: 2/4
+ shard_count: 5
+ shard_label: 2/5
- shard_index: 2
- shard_count: 4
- shard_label: 3/4
+ shard_count: 5
+ shard_label: 3/5
- shard_index: 3
- shard_count: 4
- shard_label: 4/4
+ shard_count: 5
+ shard_label: 4/5
+ - shard_index: 4
+ shard_count: 5
+ shard_label: 5/5
steps:
- name: Checkout repository
diff --git a/.github/workflows/release-verify.yml b/.github/workflows/release-verify.yml
index 890dd4aea4b0..cac52d75b5d6 100644
--- a/.github/workflows/release-verify.yml
+++ b/.github/workflows/release-verify.yml
@@ -109,17 +109,20 @@ jobs:
matrix:
include:
- shard_index: 0
- shard_count: 4
- shard_label: 1/4
+ shard_count: 5
+ shard_label: 1/5
- shard_index: 1
- shard_count: 4
- shard_label: 2/4
+ shard_count: 5
+ shard_label: 2/5
- shard_index: 2
- shard_count: 4
- shard_label: 3/4
+ shard_count: 5
+ shard_label: 3/5
- shard_index: 3
- shard_count: 4
- shard_label: 4/4
+ shard_count: 5
+ shard_label: 4/5
+ - shard_index: 4
+ shard_count: 5
+ shard_label: 5/5
steps:
- name: Checkout repository
diff --git a/AGENTS.md b/AGENTS.md
index cc341f6861c7..3f8581b80843 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -29,6 +29,10 @@ Before making changes, read in this order:
- `packages/adapters/`: agent adapter implementations (Claude, Codex, Cursor, etc.)
- `packages/adapter-utils/`: shared adapter utilities
- `packages/plugins/`: plugin system packages
+- `packages/skills-catalog/`: app-shipped skills catalog (`@paperclipai/skills-catalog`)
+- `packages/teams-catalog/`: app-shipped teams catalog (`@paperclipai/teams-catalog`)
+- `cli/`: `paperclipai` CLI package (published bin, agent-facing commands)
+- `skills/`: Paperclip runtime/operational skills (not part of the app catalog)
- `doc/`: operational and product docs
## 4. Dev Setup (Auto DB)
@@ -179,48 +183,6 @@ A change is done when all are true:
4. Docs updated when behavior or commands change
5. PR description follows the [PR template](.github/PULL_REQUEST_TEMPLATE.md) with all sections filled in (including Model Used)
-## 11. Fork-Specific: HenkDz/paperclip
-
-This is a fork of `paperclipai/paperclip` with QoL patches and a **built-in** Hermes adapter story on branch `feat/externalize-hermes-adapter` ([tree](https://github.com/HenkDz/paperclip/tree/feat/externalize-hermes-adapter)).
-
-### Branch Strategy
-
-- `feat/externalize-hermes-adapter` now ships `hermes_local` and `hermes_gateway` as built-in core adapters.
-- Older fork branches may still document plugin-only Hermes; treat this file as authoritative for the current branch.
-
-### Hermes (built-in)
-
-- `hermes_local` is available without Adapter manager installation and runs the local Hermes CLI.
-- `hermes_gateway` is available without Adapter manager installation and calls an already-running Hermes API server.
-- Operators may still install external Hermes packages through Adapter manager to override/shadow the built-ins.
-- Optional: `file:` entry in `~/.paperclip/adapter-plugins.json` remains useful for local development of override packages.
-
-### Local Dev
-
-- Fork runs on port 3101+ (auto-detects if 3100 is taken by upstream instance)
-- `npx vite build` hangs on NTFS — use `node node_modules/vite/bin/vite.js build` instead
-- Server startup from NTFS takes 30-60s — don't assume failure immediately
-- Kill ALL paperclip processes before starting: `pkill -f "paperclip"; pkill -f "tsx.*index.ts"`
-- Vite cache survives `rm -rf dist` — delete both: `rm -rf ui/dist ui/node_modules/.vite`
-
-### Fork QoL Patches (not in upstream)
-
-These are local modifications in the fork's UI. If re-copying source, these must be re-applied:
-
-1. **stderr_group** — amber accordion for MCP init noise in `RunTranscriptView.tsx`
-2. **tool_group** — accordion for consecutive non-terminal tools (write, read, search, browser)
-3. **Dashboard excerpt** — `LatestRunCard` strips markdown, shows first 3 lines/280 chars
-
-### Plugin System
-
-PR #2218 (`feat/external-adapter-phase1`) adds external adapter support. See root `AGENTS.md` for full details.
-
-- Adapters can be loaded as external plugins via `~/.paperclip/adapter-plugins.json`
-- The plugin-loader should have ZERO hardcoded adapter imports — pure dynamic loading
-- `createServerAdapter()` must include ALL optional fields (especially `detectModel`)
-- Built-in UI adapters can shadow external plugin parsers; external override pause/resume should restore the built-in parser.
-- Reference external adapters: Droid (npm); Hermes can also be tested as an override package.
-
## Design system
`DESIGN.md` at the repo root is the source of truth for UI design decisions. The token-only rule applies to all `ui/` changes: every color, spacing, radius, type, shadow, and motion value in `ui/src/components/**` and `ui/src/pages/**` comes from the token layer in `ui/src/index.css` — no hex, raw px, arbitrary Tailwind bracket values, or raw `font-size`/`fontSize` declarations in components, outside the documented allowlist in `ui/src/index.css`. Run `pnpm check:token-gates` (`scripts/check-token-gates.mjs`) before committing UI changes — it fails on any violation not covered by that allowlist.
diff --git a/Dockerfile b/Dockerfile
index 2769078b6683..dfa7594e1397 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -42,6 +42,11 @@ COPY --parents packages/plugins/sandbox-providers/./*/package.json packages/plug
COPY packages/plugins/paperclip-plugin-fake-sandbox/package.json packages/plugins/paperclip-plugin-fake-sandbox/
COPY packages/plugins/plugin-llm-wiki/package.json packages/plugins/plugin-llm-wiki/
COPY packages/plugins/plugin-workspace-diff/package.json packages/plugins/plugin-workspace-diff/
+# Fork plugins.
+COPY packages/plugins/paperclip-plugin-escalation/package.json packages/plugins/paperclip-plugin-escalation/
+COPY packages/plugins/paperclip-plugin-github-mirror/package.json packages/plugins/paperclip-plugin-github-mirror/
+COPY packages/plugins/paperclip-plugin-run-completion/package.json packages/plugins/paperclip-plugin-run-completion/
+COPY packages/plugins/paperclip-plugin-telegram-notify/package.json packages/plugins/paperclip-plugin-telegram-notify/
COPY patches/ patches/
COPY scripts/link-plugin-dev-sdk.mjs scripts/
diff --git a/README.md b/README.md
index 8e25e8f668db..53a7291c46e5 100644
--- a/README.md
+++ b/README.md
@@ -15,7 +15,7 @@
-
+
diff --git a/cli/src/__tests__/agent-jwt-env.test.ts b/cli/src/__tests__/agent-jwt-env.test.ts
index baf5db5128bd..b26ae018df21 100644
--- a/cli/src/__tests__/agent-jwt-env.test.ts
+++ b/cli/src/__tests__/agent-jwt-env.test.ts
@@ -76,4 +76,62 @@ describe("agent jwt env helpers", () => {
expect(contents).toContain('PAPERCLIP_WORKTREE_COLOR="#439edb"');
expect(readPaperclipEnvEntries(envPath).PAPERCLIP_WORKTREE_COLOR).toBe("#439edb");
});
+
+ it("preserves operator content and CRLF while updating only managed entries", () => {
+ const configPath = tempConfigPath();
+ const envPath = resolveAgentJwtEnvFile(configPath);
+ const original = [
+ "# operator comment",
+ "DATABASE_URL='postgres://operator:encoded@localhost/paperclip'",
+ "",
+ "export PAPERCLIP_HOME = '/old path' # managed path",
+ "PAPERCLIP_DUPLICATE=stale",
+ 'PAPERCLIP_DUPLICATE="current"',
+ "UNKNOWN_VALUE=operator-owned",
+ "",
+ ].join("\r\n");
+ fs.writeFileSync(envPath, original, { mode: 0o600 });
+
+ mergePaperclipEnvEntries(
+ {
+ PAPERCLIP_HOME: "/new path",
+ PAPERCLIP_DUPLICATE: "current",
+ PAPERCLIP_WORKTREE_COLOR: "#439edb",
+ DATABASE_URL: "postgres://paperclip-must-not-overwrite",
+ },
+ envPath,
+ );
+
+ const updated = fs.readFileSync(envPath, "utf8");
+ expect(updated).toBe([
+ "# operator comment",
+ "DATABASE_URL='postgres://operator:encoded@localhost/paperclip'",
+ "",
+ 'export PAPERCLIP_HOME = "/new path" # managed path',
+ "PAPERCLIP_DUPLICATE=current",
+ 'PAPERCLIP_DUPLICATE="current"',
+ "UNKNOWN_VALUE=operator-owned",
+ 'PAPERCLIP_WORKTREE_COLOR="#439edb"',
+ "",
+ ].join("\r\n"));
+ expect(updated.replaceAll("\r\n", "")).not.toContain("\n");
+ });
+
+ it("does not replace the env file when managed values are already current", () => {
+ const configPath = tempConfigPath();
+ const envPath = resolveAgentJwtEnvFile(configPath);
+ const original = [
+ "# preserve this file byte-for-byte",
+ "export PAPERCLIP_HOME = '/same path'",
+ "UNKNOWN=\"operator encoding\"",
+ "",
+ ].join("\n");
+ fs.writeFileSync(envPath, original, { mode: 0o600 });
+ const previousInode = fs.statSync(envPath).ino;
+
+ mergePaperclipEnvEntries({ PAPERCLIP_HOME: "/same path" }, envPath);
+
+ expect(fs.readFileSync(envPath, "utf8")).toBe(original);
+ expect(fs.statSync(envPath).ino).toBe(previousInode);
+ });
});
diff --git a/cli/src/__tests__/agent-lifecycle.test.ts b/cli/src/__tests__/agent-lifecycle.test.ts
index 4e2917ce83e9..41817823db9f 100644
--- a/cli/src/__tests__/agent-lifecycle.test.ts
+++ b/cli/src/__tests__/agent-lifecycle.test.ts
@@ -83,7 +83,15 @@ describe("agent lifecycle commands", () => {
await run(["agent", "runtime-state:reset-session", AGENT_ID, "--task-key", "task-1"]);
await run(["agent", "task-sessions", AGENT_ID]);
await run(["agent", "skills", AGENT_ID]);
- await run(["agent", "skills:sync", AGENT_ID, "--desired-skills", "paperclip,github"]);
+ await run([
+ "agent",
+ "skills:sync",
+ AGENT_ID,
+ "--desired-skills",
+ "paperclip,github",
+ "--mode",
+ "replace",
+ ]);
await run(["agent", "instructions-path:update", AGENT_ID, "--payload-json", JSON.stringify({ path: "/tmp/AGENTS.md" })]);
await run(["agent", "instructions-bundle", AGENT_ID]);
await run(["agent", "instructions-bundle:update", AGENT_ID, "--payload-json", JSON.stringify({ mode: "managed" })]);
diff --git a/cli/src/__tests__/company-delete.test.ts b/cli/src/__tests__/company-delete.test.ts
index d45fc1202260..54cc59a2315c 100644
--- a/cli/src/__tests__/company-delete.test.ts
+++ b/cli/src/__tests__/company-delete.test.ts
@@ -27,6 +27,7 @@ function makeCompany(overrides: Partial): Company {
createdAt: new Date(),
updatedAt: new Date(),
...overrides,
+ interactionResolverGovernance: overrides.interactionResolverGovernance ?? {},
};
}
diff --git a/cli/src/__tests__/company.test.ts b/cli/src/__tests__/company.test.ts
index 1532047f6f89..a548c8821a4a 100644
--- a/cli/src/__tests__/company.test.ts
+++ b/cli/src/__tests__/company.test.ts
@@ -504,6 +504,17 @@ describe("renderCompanyImportResult", () => {
{ slug: "cto", id: "agent-2", action: "updated", name: "CTO", reason: "replace strategy" },
{ slug: "ops", id: null, action: "skipped", name: "Ops", reason: "skip strategy" },
],
+ skills: [
+ {
+ originalKey: "company/source/review",
+ originalSlug: "review",
+ key: "company/target/review-2",
+ slug: "review-2",
+ id: "skill-1",
+ action: "renamed",
+ reason: "rename strategy",
+ },
+ ],
projects: [
{ slug: "app", id: "project-1", action: "created", name: "App", reason: null },
{ slug: "ops", id: "project-2", action: "updated", name: "Operations", reason: "replace strategy" },
@@ -525,8 +536,10 @@ describe("renderCompanyImportResult", () => {
expect(rendered).toContain("Company");
expect(rendered).toContain("https://paperclip.example/PAP/dashboard");
expect(rendered).toContain("3 agents total (1 created, 1 updated, 1 skipped)");
+ expect(rendered).toContain("1 skill total (1 renamed)");
expect(rendered).toContain("3 projects total (1 created, 1 updated, 1 skipped)");
expect(rendered).toContain("Agent results");
+ expect(rendered).toContain("Skill results");
expect(rendered).toContain("Project results");
expect(rendered).toContain("Using claude-local adapter");
expect(rendered).toContain("Review API keys");
diff --git a/cli/src/__tests__/config-store.test.ts b/cli/src/__tests__/config-store.test.ts
new file mode 100644
index 000000000000..21abb4bf60ee
--- /dev/null
+++ b/cli/src/__tests__/config-store.test.ts
@@ -0,0 +1,139 @@
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { afterEach, describe, expect, it, vi } from "vitest";
+import {
+ backupInvalidConfig,
+ readConfig,
+ writeConfig,
+} from "../config/store.js";
+import { paperclipConfigSchema, type PaperclipConfig } from "../config/schema.js";
+
+const roots: string[] = [];
+
+afterEach(() => {
+ vi.restoreAllMocks();
+ for (const root of roots.splice(0)) {
+ fs.rmSync(root, { recursive: true, force: true });
+ }
+});
+
+function createConfigPath(): string {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "paperclip-config-store-"));
+ roots.push(root);
+ return path.join(root, "config.json");
+}
+
+function defaultConfig(): PaperclipConfig {
+ return paperclipConfigSchema.parse({
+ $meta: {
+ version: 1,
+ updatedAt: "2026-08-06T00:00:00.000Z",
+ source: "configure",
+ },
+ database: { mode: "embedded-postgres" },
+ logging: { mode: "file" },
+ server: {},
+ });
+}
+
+describe("config store", () => {
+ it("preserves top-level and nested extension keys during a known-field update", () => {
+ const configPath = createConfigPath();
+ fs.writeFileSync(configPath, JSON.stringify({
+ ...defaultConfig(),
+ topLevelExtension: { enabled: true },
+ server: {
+ ...defaultConfig().server,
+ serverExtension: "keep",
+ },
+ storage: {
+ ...defaultConfig().storage,
+ localDisk: {
+ ...defaultConfig().storage.localDisk,
+ driverExtension: "keep",
+ },
+ },
+ }, null, 2));
+
+ const source = readConfig(configPath)!;
+ const { topLevelExtension: _topLevelExtension, ...knownConfig } = source;
+ const { serverExtension: _serverExtension, ...knownServer } = source.server;
+ const { driverExtension: _driverExtension, ...knownLocalDisk } = source.storage.localDisk;
+ const update: PaperclipConfig = {
+ ...knownConfig,
+ server: {
+ ...knownServer,
+ port: 3200,
+ },
+ storage: {
+ ...source.storage,
+ localDisk: knownLocalDisk,
+ },
+ };
+
+ expect(writeConfig(update, configPath)).toBe(true);
+ expect(JSON.parse(fs.readFileSync(configPath, "utf8"))).toMatchObject({
+ topLevelExtension: { enabled: true },
+ server: {
+ port: 3200,
+ serverExtension: "keep",
+ },
+ storage: {
+ localDisk: {
+ driverExtension: "keep",
+ },
+ },
+ });
+ });
+
+ it("skips semantic no-op writes and keeps the config mtime stable", () => {
+ const configPath = createConfigPath();
+ const source = defaultConfig();
+ fs.writeFileSync(configPath, `${JSON.stringify(source, null, 2)}\n`);
+ const stableTime = new Date("2020-01-01T00:00:00.000Z");
+ fs.utimesSync(configPath, stableTime, stableTime);
+
+ const update = {
+ ...source,
+ $meta: {
+ ...source.$meta,
+ source: "doctor" as const,
+ updatedAt: "2026-08-06T01:00:00.000Z",
+ },
+ };
+
+ expect(writeConfig(update, configPath)).toBe(false);
+ expect(fs.statSync(configPath).mtimeMs).toBe(stableTime.getTime());
+ expect(fs.existsSync(`${configPath}.backup`)).toBe(false);
+ });
+
+ it("backs up invalid bytes collision-safely and only replaces them through an atomic repair", () => {
+ const configPath = createConfigPath();
+ const invalidBytes = Buffer.from('{"server": invalid}\n', "utf8");
+ fs.writeFileSync(configPath, invalidBytes);
+ fs.writeFileSync(`${configPath}.invalid-1`, "existing backup");
+
+ const open = vi.spyOn(fs, "openSync");
+ const sync = vi.spyOn(fs, "fsyncSync");
+ const backupPath = backupInvalidConfig(configPath);
+ expect(backupPath).toBe(`${configPath}.invalid-2`);
+ expect(fs.readFileSync(backupPath)).toEqual(invalidBytes);
+ expect(open).toHaveBeenCalledWith(backupPath, "r");
+ expect(open).toHaveBeenCalledWith(path.dirname(configPath), "r");
+ expect(sync).toHaveBeenCalled();
+ expect(() => writeConfig(defaultConfig(), configPath)).toThrow(/Refusing to overwrite invalid config/);
+ expect(fs.readFileSync(configPath)).toEqual(invalidBytes);
+
+ open.mockClear();
+ sync.mockClear();
+ const rename = vi.spyOn(fs, "renameSync");
+ expect(writeConfig(defaultConfig(), configPath, { invalidBackupPath: backupPath })).toBe(true);
+ expect(rename).toHaveBeenCalledWith(expect.stringMatching(/config\.json\.tmp-\d+-\d+$/), configPath);
+ expect(open).toHaveBeenCalledWith(path.dirname(configPath), "r");
+ expect(open.mock.invocationCallOrder.at(-1)!).toBeGreaterThan(rename.mock.invocationCallOrder.at(-1)!);
+ expect(sync.mock.invocationCallOrder.at(-1)!).toBeGreaterThan(open.mock.invocationCallOrder.at(-1)!);
+ expect(readConfig(configPath)).not.toBeNull();
+ expect(fs.readdirSync(path.dirname(configPath)).some((entry) => entry.includes(".tmp-"))).toBe(false);
+ });
+});
diff --git a/cli/src/__tests__/configure-repair.test.ts b/cli/src/__tests__/configure-repair.test.ts
new file mode 100644
index 000000000000..8ffe3834695f
--- /dev/null
+++ b/cli/src/__tests__/configure-repair.test.ts
@@ -0,0 +1,83 @@
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
+import * as prompts from "@clack/prompts";
+import { configure } from "../commands/configure.js";
+import { readConfig } from "../config/store.js";
+
+vi.mock("@clack/prompts", () => ({
+ intro: vi.fn(),
+ outro: vi.fn(),
+ cancel: vi.fn(),
+ confirm: vi.fn(),
+ select: vi.fn(),
+ isCancel: vi.fn(() => false),
+ log: {
+ error: vi.fn(),
+ message: vi.fn(),
+ step: vi.fn(),
+ success: vi.fn(),
+ warn: vi.fn(),
+ },
+}));
+
+vi.mock("../prompts/server.js", () => ({
+ promptServer: vi.fn(async ({ currentServer, currentAuth }) => ({
+ server: currentServer,
+ auth: currentAuth,
+ })),
+}));
+
+const ORIGINAL_EXIT_CODE = process.exitCode;
+let originalStdinIsTTY: boolean | undefined;
+let originalStdoutIsTTY: boolean | undefined;
+
+beforeEach(() => {
+ originalStdinIsTTY = process.stdin.isTTY;
+ originalStdoutIsTTY = process.stdout.isTTY;
+ Object.defineProperty(process.stdin, "isTTY", { configurable: true, value: true });
+ Object.defineProperty(process.stdout, "isTTY", { configurable: true, value: true });
+ vi.mocked(prompts.confirm).mockResolvedValue(true);
+ vi.spyOn(console, "log").mockImplementation(() => undefined);
+});
+
+afterEach(() => {
+ Object.defineProperty(process.stdin, "isTTY", {
+ configurable: true,
+ value: originalStdinIsTTY,
+ });
+ Object.defineProperty(process.stdout, "isTTY", {
+ configurable: true,
+ value: originalStdoutIsTTY,
+ });
+ process.exitCode = ORIGINAL_EXIT_CODE;
+ vi.restoreAllMocks();
+});
+
+describe("configure invalid-config repair", () => {
+ it("repairs only after confirmation and commits the staged config atomically", async () => {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "paperclip-configure-repair-"));
+ const configPath = path.join(root, "config.json");
+ const invalidBytes = Buffer.from('{"server": invalid}\n', "utf8");
+ fs.writeFileSync(configPath, invalidBytes);
+ const rename = vi.spyOn(fs, "renameSync");
+
+ try {
+ await configure({ config: configPath, section: "server" });
+
+ expect(prompts.confirm).toHaveBeenCalledWith({
+ message: `Repair from defaults? The invalid original is backed up at ${configPath}.invalid-1.`,
+ initialValue: false,
+ });
+ expect(fs.readFileSync(`${configPath}.invalid-1`)).toEqual(invalidBytes);
+ expect(readConfig(configPath)).not.toBeNull();
+ expect(rename).toHaveBeenCalledWith(
+ expect.stringMatching(/config\.json\.tmp-\d+-\d+$/),
+ configPath,
+ );
+ } finally {
+ fs.rmSync(root, { recursive: true, force: true });
+ }
+ });
+});
diff --git a/cli/src/__tests__/configure.test.ts b/cli/src/__tests__/configure.test.ts
index 74a37fc8fccb..cbd7016879d0 100644
--- a/cli/src/__tests__/configure.test.ts
+++ b/cli/src/__tests__/configure.test.ts
@@ -96,4 +96,29 @@ describe("configure command", () => {
fs.rmSync(root, { recursive: true, force: true });
}
});
+
+ it("backs up invalid config bytes and refuses non-interactive replacement", async () => {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "paperclip-configure-invalid-"));
+ const configPath = path.join(root, "config.json");
+ const invalidBytes = Buffer.from('{"server": invalid}\n', "utf8");
+ const stdinDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "isTTY");
+ fs.writeFileSync(configPath, invalidBytes);
+ Object.defineProperty(process.stdin, "isTTY", { configurable: true, value: false });
+
+ try {
+ await configure({ config: configPath, section: "server" });
+
+ expect(process.exitCode).toBe(1);
+ expect(fs.readFileSync(configPath)).toEqual(invalidBytes);
+ expect(fs.readFileSync(`${configPath}.invalid-1`)).toEqual(invalidBytes);
+ expect(fs.existsSync(`${configPath}.backup`)).toBe(false);
+ } finally {
+ if (stdinDescriptor) {
+ Object.defineProperty(process.stdin, "isTTY", stdinDescriptor);
+ } else {
+ delete (process.stdin as { isTTY?: boolean }).isTTY;
+ }
+ fs.rmSync(root, { recursive: true, force: true });
+ }
+ });
});
diff --git a/cli/src/__tests__/onboard.test.ts b/cli/src/__tests__/onboard.test.ts
index 33890e703de7..59a578b1f86b 100644
--- a/cli/src/__tests__/onboard.test.ts
+++ b/cli/src/__tests__/onboard.test.ts
@@ -8,6 +8,7 @@ import type { PaperclipConfig } from "../config/schema.js";
const ORIGINAL_ENV = { ...process.env };
const ORIGINAL_CWD = process.cwd();
const ORIGINAL_PATH = process.env.PATH;
+const ORIGINAL_EXIT_CODE = process.exitCode;
function createExistingConfigFixture() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), "paperclip-onboard-"));
@@ -107,6 +108,7 @@ describe("onboard", () => {
afterEach(() => {
process.env = { ...ORIGINAL_ENV };
process.chdir(ORIGINAL_CWD);
+ process.exitCode = ORIGINAL_EXIT_CODE;
});
it("preserves an existing config when rerun without flags", async () => {
@@ -129,6 +131,20 @@ describe("onboard", () => {
expect(fs.existsSync(path.join(path.dirname(fixture.configPath), ".env"))).toBe(true);
});
+ it("backs up invalid config bytes and refuses --yes replacement", async () => {
+ const configPath = createFreshConfigPath();
+ const invalidBytes = Buffer.from('{"database": invalid}\n', "utf8");
+ fs.mkdirSync(path.dirname(configPath), { recursive: true });
+ fs.writeFileSync(configPath, invalidBytes);
+
+ await onboard({ config: configPath, yes: true, invokedByRun: true });
+
+ expect(process.exitCode).toBe(1);
+ expect(fs.readFileSync(configPath)).toEqual(invalidBytes);
+ expect(fs.readFileSync(`${configPath}.invalid-1`)).toEqual(invalidBytes);
+ expect(fs.existsSync(`${configPath}.backup`)).toBe(false);
+ });
+
it("keeps --yes onboarding on local trusted loopback defaults", async () => {
const configPath = createFreshConfigPath();
process.env.HOST = "0.0.0.0";
diff --git a/cli/src/__tests__/skills.test.ts b/cli/src/__tests__/skills.test.ts
index 038b9812f1b1..80d0dd3c7915 100644
--- a/cli/src/__tests__/skills.test.ts
+++ b/cli/src/__tests__/skills.test.ts
@@ -479,6 +479,8 @@ describe("skills CLI commands", () => {
"review-prs",
"--skill",
"paperclip/qa",
+ "--mode",
+ "add",
"--company-id",
"company-1",
"--api-base",
@@ -498,7 +500,7 @@ describe("skills CLI commands", () => {
"http://paperclip.test/api/agents/agent-1/skills/sync",
expect.objectContaining({
method: "POST",
- body: JSON.stringify({ desiredSkills: ["review-prs", "paperclip/qa"] }),
+ body: JSON.stringify({ desiredSkills: ["review-prs", "paperclip/qa"], mode: "add" }),
}),
);
expect(JSON.parse(String(logSpy.mock.calls[0]?.[0]))).toEqual(snapshot);
diff --git a/cli/src/__tests__/worktree.test.ts b/cli/src/__tests__/worktree.test.ts
index ce063abfd798..dab7f78bfd9e 100644
--- a/cli/src/__tests__/worktree.test.ts
+++ b/cli/src/__tests__/worktree.test.ts
@@ -31,6 +31,7 @@ import {
resolveWorktreeReseedTargetPaths,
resolveGitWorktreeAddArgs,
resolvePnpmInstallInvocation,
+ resolveCurrentWorktreeEndpoint,
resolveWorktreeSeedBackupEngine,
resolveWorktreeMakeTargetPath,
worktreeRepairCommand,
@@ -167,6 +168,49 @@ function buildSourceConfig(): PaperclipConfig {
}
describe("worktree helpers", () => {
+ it("uses the repo-local config for the current worktree", () => {
+ const targetRoot = fs.mkdtempSync(path.join(os.tmpdir(), "paperclip-current-worktree-"));
+ try {
+ const localConfig = path.join(targetRoot, ".paperclip", "config.json");
+ fs.mkdirSync(path.dirname(localConfig), { recursive: true });
+ fs.writeFileSync(localConfig, "{}\n");
+ process.env.PAPERCLIP_CONFIG = "/tmp/ambient-paperclip/config.json";
+ process.chdir(targetRoot);
+
+ expect(resolveCurrentWorktreeEndpoint()).toMatchObject({
+ rootPath: targetRoot,
+ configPath: localConfig,
+ isCurrent: true,
+ });
+ } finally {
+ process.chdir(ORIGINAL_CWD);
+ fs.rmSync(targetRoot, { recursive: true, force: true });
+ }
+ });
+
+ it("uses the repository config from a nested working directory", () => {
+ const targetRoot = fs.mkdtempSync(path.join(os.tmpdir(), "paperclip-current-worktree-nested-"));
+ try {
+ execFileSync("git", ["init", "-q"], { cwd: targetRoot });
+ const nestedDirectory = path.join(targetRoot, "packages", "example", "src");
+ const localConfig = path.join(targetRoot, ".paperclip", "config.json");
+ fs.mkdirSync(nestedDirectory, { recursive: true });
+ fs.mkdirSync(path.dirname(localConfig), { recursive: true });
+ fs.writeFileSync(localConfig, "{}\n");
+ process.env.PAPERCLIP_CONFIG = "/tmp/ambient-paperclip/config.json";
+ process.chdir(nestedDirectory);
+
+ expect(resolveCurrentWorktreeEndpoint()).toMatchObject({
+ rootPath: targetRoot,
+ configPath: localConfig,
+ isCurrent: true,
+ });
+ } finally {
+ process.chdir(ORIGINAL_CWD);
+ fs.rmSync(targetRoot, { recursive: true, force: true });
+ }
+ });
+
it("sanitizes instance ids", () => {
expect(sanitizeWorktreeInstanceId("feature/worktree-support")).toBe("feature-worktree-support");
expect(sanitizeWorktreeInstanceId(" ")).toBe("worktree");
diff --git a/cli/src/commands/client/agent.ts b/cli/src/commands/client/agent.ts
index 8144352c491d..6d6677e4244a 100644
--- a/cli/src/commands/client/agent.ts
+++ b/cli/src/commands/client/agent.ts
@@ -71,6 +71,7 @@ interface AgentResetSessionOptions extends BaseClientOptions {
interface AgentSkillsSyncOptions extends BaseClientOptions {
desiredSkills: string;
+ mode: string;
}
interface AgentInstructionsFileOptions extends BaseClientOptions {
@@ -589,10 +590,17 @@ export function registerAgentCommands(program: Command): void {
.description("Sync desired skills onto an agent")
.argument("", "Agent ID")
.requiredOption("--desired-skills ", "Desired skill names")
+ .requiredOption(
+ "--mode ",
+ "Merge mode: add keeps other skills; remove deletes only named skills; replace destructively overwrites the complete set",
+ )
.action(async (agentId: string, opts: AgentSkillsSyncOptions) => {
try {
const ctx = resolveCommandContext(opts);
- const payload = agentSkillSyncSchema.parse({ desiredSkills: parseCsv(opts.desiredSkills) });
+ const payload = agentSkillSyncSchema.parse({
+ desiredSkills: parseCsv(opts.desiredSkills),
+ mode: opts.mode,
+ });
const result = await ctx.api.post(apiPath`/api/agents/${agentId}/skills/sync`, payload);
printOutput(result, { json: ctx.json });
} catch (err) {
diff --git a/cli/src/commands/client/company.ts b/cli/src/commands/client/company.ts
index bd0d162a502c..72864f6482cb 100644
--- a/cli/src/commands/client/company.ts
+++ b/cli/src/commands/client/company.ts
@@ -548,6 +548,16 @@ function summarizeImportAgentResults(agents: CompanyPortabilityImportResult["age
return `${agents.length} ${pluralize(agents.length, "agent")} total (${parts.join(", ")})`;
}
+function summarizeImportSkillResults(skills: CompanyPortabilityImportResult["skills"]): string {
+ if (skills.length === 0) return "0 skills changed";
+ const actions = ["created", "renamed", "replaced", "skipped"] as const;
+ const parts = actions.flatMap((action) => {
+ const count = skills.filter((skill) => skill.action === action).length;
+ return count > 0 ? [`${count} ${action}`] : [];
+ });
+ return `${skills.length} ${pluralize(skills.length, "skill")} total (${parts.join(", ")})`;
+}
+
function summarizeImportProjectResults(projects: CompanyPortabilityImportResult["projects"]): string {
if (projects.length === 0) return "0 projects changed";
const created = projects.filter((project) => project.action === "created").length;
@@ -681,10 +691,12 @@ export function renderCompanyImportResult(
result: CompanyPortabilityImportResult,
meta: { targetLabel: string; companyUrl?: string; infoMessages?: string[] },
): string {
+ const skills = result.skills ?? [];
const lines: string[] = [
`${pc.bold("Target")} ${meta.targetLabel}`,
`${pc.bold("Company")} ${result.company.name} (${actionChip(result.company.action)})`,
`${pc.bold("Agents")} ${summarizeImportAgentResults(result.agents)}`,
+ `${pc.bold("Skills")} ${summarizeImportSkillResults(skills)}`,
`${pc.bold("Projects")} ${summarizeImportProjectResults(result.projects)}`,
];
@@ -701,6 +713,15 @@ export function renderCompanyImportResult(
reason: agent.reason,
})),
);
+ appendPreviewExamples(
+ lines,
+ "Skill results",
+ skills.map((skill) => ({
+ action: skill.action,
+ label: `${skill.originalSlug} -> ${skill.slug}`,
+ reason: skill.reason,
+ })),
+ );
appendPreviewExamples(
lines,
"Project results",
diff --git a/cli/src/commands/client/skills.ts b/cli/src/commands/client/skills.ts
index c97ebd18df10..f24316c9c492 100644
--- a/cli/src/commands/client/skills.ts
+++ b/cli/src/commands/client/skills.ts
@@ -1,17 +1,19 @@
import { Command } from "commander";
-import type {
- Agent,
- AgentSkillSnapshot,
- CatalogSkill,
- CompanySkill,
- CompanySkillAuditResult,
- CompanySkillDetail,
- CompanySkillFileDetail,
- CompanySkillImportResult,
- CompanySkillInstallCatalogResult,
- CompanySkillListItem,
- CompanySkillProjectScanResult,
- CompanySkillUpdateStatus,
+import {
+ agentSkillAssignmentModeSchema,
+ type AgentSkillAssignmentMode,
+ type Agent,
+ type AgentSkillSnapshot,
+ type CatalogSkill,
+ type CompanySkill,
+ type CompanySkillAuditResult,
+ type CompanySkillDetail,
+ type CompanySkillFileDetail,
+ type CompanySkillImportResult,
+ type CompanySkillInstallCatalogResult,
+ type CompanySkillListItem,
+ type CompanySkillProjectScanResult,
+ type CompanySkillUpdateStatus,
} from "@paperclipai/shared";
import { readFile } from "node:fs/promises";
import { stdin as input, stdout as output } from "node:process";
@@ -69,6 +71,7 @@ interface ConfirmedSkillOptions extends SkillsOptions {
interface AgentSkillSyncOptions extends SkillsOptions {
skill?: string[];
+ mode: AgentSkillAssignmentMode;
}
type CompanySkillReferenceTarget = Pick;
@@ -502,9 +505,13 @@ function registerAgentSkillCommands(skills: Command): void {
addCommonClientOptions(
agent
.command("sync")
- .description("Replace an agent's desired company skills and sync runtime state")
+ .description("Merge an agent's desired company skills and sync runtime state")
.argument("", "Agent ID or shortname/url-key")
.option("--skill ", "Desired company skill ID, key, or slug; may be repeated", collectOptionValue, [] as string[])
+ .requiredOption(
+ "--mode ",
+ "Merge mode: add keeps other skills; remove deletes only named skills; replace destructively overwrites the complete set",
+ )
.action(async (agentRef: string, opts: AgentSkillSyncOptions) => {
try {
const desiredSkills = opts.skill ?? [];
@@ -513,16 +520,17 @@ function registerAgentSkillCommands(skills: Command): void {
}
const ctx = resolveCommandContext(opts, { requireCompany: true });
const agentRow = await resolveAgent(ctx, agentRef);
+ const mode = agentSkillAssignmentModeSchema.parse(opts.mode);
const snapshot = await ctx.api.post(
`/api/agents/${encodeURIComponent(agentRow.id)}/skills/sync`,
- { desiredSkills },
+ { desiredSkills, mode },
);
if (ctx.json) {
printOutput(snapshot, { json: true });
return;
}
console.log(
- `Desired company skills replaced for ${agentRow.name} (${agentRow.id}); runtime sync returned ${snapshot?.entries.length ?? 0} entrie(s).`,
+ `Desired company skills updated with ${mode} mode for ${agentRow.name} (${agentRow.id}); runtime sync returned ${snapshot?.entries.length ?? 0} entrie(s).`,
);
printAgentSkillSnapshot(snapshot, agentRow);
} catch (err) {
@@ -548,7 +556,7 @@ function registerAgentSkillCommands(skills: Command): void {
);
const snapshot = await ctx.api.post(
`/api/agents/${encodeURIComponent(agentRow.id)}/skills/sync`,
- { desiredSkills: [] },
+ { desiredSkills: [], mode: "replace" },
);
if (ctx.json) {
printOutput(snapshot, { json: true });
diff --git a/cli/src/commands/configure.ts b/cli/src/commands/configure.ts
index 7c9b391bf0fb..126fcc414bf6 100644
--- a/cli/src/commands/configure.ts
+++ b/cli/src/commands/configure.ts
@@ -1,7 +1,16 @@
import * as p from "@clack/prompts";
import pc from "picocolors";
-import { readConfig, writeConfig, configExists, resolveConfigPath } from "../config/store.js";
-import type { PaperclipConfig } from "../config/schema.js";
+import {
+ backupInvalidConfig,
+ readConfig,
+ writeConfig,
+ configExists,
+ resolveConfigPath,
+} from "../config/store.js";
+import {
+ findPaperclipConfigKeyWarnings,
+ type PaperclipConfig,
+} from "../config/schema.js";
import { ensureLocalSecretsKeyFile } from "../config/secrets-key.js";
import { promptDatabase } from "../prompts/database.js";
import { promptLlm } from "../prompts/llm.js";
@@ -88,15 +97,39 @@ export async function configure(opts: {
}
let config: PaperclipConfig;
+ let invalidBackupPath: string | undefined;
try {
config = readConfig(opts.config) ?? defaultConfig();
+ for (const warning of findPaperclipConfigKeyWarnings(config)) {
+ p.log.warn(`Unknown config key ${warning.path}; did you mean ${warning.suggestion}? It will be preserved.`);
+ }
} catch (err) {
- p.log.message(
- pc.yellow(
- `Existing config is invalid. Loading defaults so you can repair it now.\n${err instanceof Error ? err.message : String(err)}`,
- ),
+ const backupPath = backupInvalidConfig(opts.config);
+ p.log.warn(
+ `Existing config is invalid. Preserved the original bytes at ${backupPath}.\n${err instanceof Error ? err.message : String(err)}`,
);
+
+ if (!process.stdin.isTTY || !process.stdout.isTTY) {
+ p.log.error(
+ `Refusing to replace ${configPath} without confirmation. Rerun interactively to repair from defaults; the original and ${backupPath} are unchanged.`,
+ );
+ p.outro("");
+ process.exitCode = 1;
+ return;
+ }
+
+ const repair = await p.confirm({
+ message: `Repair from defaults? The invalid original is backed up at ${backupPath}.`,
+ initialValue: false,
+ });
+ if (p.isCancel(repair) || !repair) {
+ p.cancel(`Configuration left unchanged. Invalid backup: ${backupPath}`);
+ process.exitCode = 1;
+ return;
+ }
+
config = defaultConfig();
+ invalidBackupPath = backupPath;
}
let section: Section | undefined = opts.section as Section | undefined;
@@ -179,8 +212,15 @@ export async function configure(opts: {
config.$meta.updatedAt = new Date().toISOString();
config.$meta.source = "configure";
- writeConfig(config, opts.config);
- p.log.success(`${SECTION_LABELS[section]} configuration updated.`);
+ const written = writeConfig(config, opts.config, {
+ invalidBackupPath,
+ });
+ invalidBackupPath = undefined;
+ if (written) {
+ p.log.success(`${SECTION_LABELS[section]} configuration updated.`);
+ } else {
+ p.log.message(pc.dim(`${SECTION_LABELS[section]} configuration unchanged.`));
+ }
// If section was provided via CLI flag, don't loop
if (opts.section) {
diff --git a/cli/src/commands/onboard.ts b/cli/src/commands/onboard.ts
index 3024360dda58..31850714d732 100644
--- a/cli/src/commands/onboard.ts
+++ b/cli/src/commands/onboard.ts
@@ -17,8 +17,17 @@ import {
type SecretProvider,
type StorageProvider,
} from "@paperclipai/shared";
-import { configExists, readConfig, resolveConfigPath, writeConfig } from "../config/store.js";
-import type { PaperclipConfig } from "../config/schema.js";
+import {
+ backupInvalidConfig,
+ configExists,
+ readConfig,
+ resolveConfigPath,
+ writeConfig,
+} from "../config/store.js";
+import {
+ findPaperclipConfigKeyWarnings,
+ type PaperclipConfig,
+} from "../config/schema.js";
import { ensureAgentJwtSecret, resolveAgentJwtEnvFile } from "../config/env.js";
import { ensureLocalSecretsKeyFile } from "../config/secrets-key.js";
import { promptDatabase } from "../prompts/database.js";
@@ -356,17 +365,45 @@ export async function onboard(opts: OnboardOptions): Promise {
);
let existingConfig: PaperclipConfig | null = null;
+ let invalidBackupPath: string | undefined;
if (configExists(opts.config)) {
p.log.message(pc.dim(`${configPath} exists`));
try {
existingConfig = readConfig(opts.config);
+ for (const warning of findPaperclipConfigKeyWarnings(existingConfig)) {
+ p.log.warn(`Unknown config key ${warning.path}; did you mean ${warning.suggestion}? It will be preserved.`);
+ }
} catch (err) {
- p.log.message(
- pc.yellow(
- `Existing config appears invalid and will be updated.\n${err instanceof Error ? err.message : String(err)}`,
- ),
+ const backupPath = backupInvalidConfig(opts.config);
+ p.log.warn(
+ `Existing config is invalid. Preserved the original bytes at ${backupPath}.\n${err instanceof Error ? err.message : String(err)}`,
);
+
+ const canConfirmRepair =
+ opts.yes !== true &&
+ opts.invokedByRun !== true &&
+ process.stdin.isTTY === true &&
+ process.stdout.isTTY === true;
+ if (!canConfirmRepair) {
+ p.log.error(
+ `Refusing to replace ${configPath} without confirmation. Rerun interactively to repair from defaults; the original and ${backupPath} are unchanged.`,
+ );
+ p.outro("");
+ process.exitCode = 1;
+ return;
+ }
+
+ const repair = await p.confirm({
+ message: `Repair from defaults? The invalid original is backed up at ${backupPath}.`,
+ initialValue: false,
+ });
+ if (p.isCancel(repair) || !repair) {
+ p.cancel(`Configuration left unchanged. Invalid backup: ${backupPath}`);
+ process.exitCode = 1;
+ return;
+ }
+ invalidBackupPath = backupPath;
}
}
@@ -646,7 +683,9 @@ export async function onboard(opts: OnboardOptions): Promise {
p.log.message(pc.dim(`Using existing local secrets key file at ${keyResult.path}`));
}
- writeConfig(config, opts.config);
+ writeConfig(config, opts.config, {
+ invalidBackupPath,
+ });
if (tc) trackInstallCompleted(tc, {
adapterType: server.deploymentMode,
diff --git a/cli/src/commands/worktree.ts b/cli/src/commands/worktree.ts
index 676ee7d52ff0..fe07f3ce618f 100644
--- a/cli/src/commands/worktree.ts
+++ b/cli/src/commands/worktree.ts
@@ -2218,10 +2218,13 @@ async function closeDb(db: ClosableDb): Promise {
await db.$client?.end?.({ timeout: 5 }).catch(() => undefined);
}
-function resolveCurrentEndpoint(): ResolvedWorktreeEndpoint {
+export function resolveCurrentWorktreeEndpoint(): ResolvedWorktreeEndpoint {
+ const cwd = path.resolve(process.cwd());
+ const rootPath = detectGitWorkspaceInfo(cwd)?.root ?? cwd;
+ const localConfigPath = path.join(rootPath, ".paperclip", "config.json");
return {
- rootPath: path.resolve(process.cwd()),
- configPath: resolveConfigPath(),
+ rootPath,
+ configPath: existsSync(localConfigPath) ? localConfigPath : resolveConfigPath(),
label: "current",
isCurrent: true,
};
@@ -2233,7 +2236,7 @@ function resolveAttachmentLookupStorages(input: {
}): ConfiguredStorage[] {
const orderedConfigPaths = [
input.sourceEndpoint.configPath,
- resolveCurrentEndpoint().configPath,
+ resolveCurrentWorktreeEndpoint().configPath,
input.targetEndpoint.configPath,
...toMergeSourceChoices(process.cwd())
.filter((choice) => choice.hasPaperclipConfig)
@@ -2804,7 +2807,7 @@ export async function worktreeListCommand(opts: WorktreeListOptions): Promise {
const excluded = excludeWorktreePath ? path.resolve(excludeWorktreePath) : null;
- const currentEndpoint = resolveCurrentEndpoint();
+ const currentEndpoint = resolveCurrentWorktreeEndpoint();
const choices = toMergeSourceChoices(process.cwd())
.filter((choice) => choice.hasPaperclipConfig || choice.isCurrent)
.filter((choice) => path.resolve(choice.worktree) !== excluded)
@@ -3295,7 +3298,7 @@ export async function worktreeMergeHistoryCommand(sourceArg: string | undefined,
const targetEndpoint = opts.to
? resolveWorktreeEndpointFromSelector(opts.to, { allowCurrent: true })
- : resolveCurrentEndpoint();
+ : resolveCurrentWorktreeEndpoint();
const sourceEndpoint = opts.from
? resolveWorktreeEndpointFromSelector(opts.from, { allowCurrent: true })
: sourceArg
@@ -3400,7 +3403,7 @@ async function runWorktreeReseed(opts: WorktreeReseedOptions): Promise {
const targetEndpoint = opts.to
? resolveWorktreeEndpointFromSelector(opts.to, { allowCurrent: true })
- : resolveCurrentEndpoint();
+ : resolveCurrentWorktreeEndpoint();
const source = resolveWorktreeReseedSource(opts);
if (path.resolve(source.configPath) === path.resolve(targetEndpoint.configPath)) {
diff --git a/cli/src/config/env.ts b/cli/src/config/env.ts
index a7266ea241cd..6787ee190a87 100644
--- a/cli/src/config/env.ts
+++ b/cli/src/config/env.ts
@@ -2,9 +2,11 @@ import fs from "node:fs";
import path from "node:path";
import { randomBytes } from "node:crypto";
import { config as loadDotenv, parse as parseEnvFileContents } from "dotenv";
+import { updateEnvFileContents, writeEnvFileAtomicallyIfChanged } from "@paperclipai/shared/env-file";
import { resolveConfigPath } from "./store.js";
const JWT_SECRET_ENV_KEY = "PAPERCLIP_AGENT_JWT_SECRET";
+const PAPERCLIP_OWNED_ENV_KEY_PATTERN = /^PAPERCLIP_[A-Z0-9_]+$/;
function resolveEnvFilePath(configPath?: string) {
return path.resolve(path.dirname(resolveConfigPath(configPath)), ".env");
}
@@ -22,21 +24,20 @@ function parseEnvFile(contents: string) {
}
}
-function formatEnvValue(value: string): string {
- if (/^[A-Za-z0-9_./:@-]+$/.test(value)) {
- return value;
- }
- return JSON.stringify(value);
-}
-
-function renderEnvFile(entries: Record) {
- const lines = [
+function emptyEnvFileContents() {
+ return [
"# Paperclip environment variables",
"# Generated by Paperclip CLI commands",
- ...Object.entries(entries).map(([key, value]) => `${key}=${formatEnvValue(value)}`),
"",
- ];
- return lines.join("\n");
+ ].join("\n");
+}
+
+function paperclipOwnedEntries(entries: Record): Record {
+ return Object.fromEntries(
+ Object.entries(entries).filter(
+ ([key, value]) => PAPERCLIP_OWNED_ENV_KEY_PATTERN.test(key) && value.trim().length > 0,
+ ),
+ );
}
export function resolvePaperclipEnvFile(configPath?: string): string {
@@ -102,11 +103,11 @@ export function readPaperclipEnvEntries(filePath = resolveEnvFilePath()): Record
}
export function writePaperclipEnvEntries(entries: Record, filePath = resolveEnvFilePath()): void {
- const dir = path.dirname(filePath);
- fs.mkdirSync(dir, { recursive: true });
- fs.writeFileSync(filePath, renderEnvFile(entries), {
- mode: 0o600,
+ const previousContents = fs.existsSync(filePath) ? fs.readFileSync(filePath, "utf8") : null;
+ const nextContents = updateEnvFileContents(previousContents ?? emptyEnvFileContents(), paperclipOwnedEntries(entries), {
+ valueEncoding: "minimal",
});
+ writeEnvFileAtomicallyIfChanged(filePath, previousContents, nextContents);
}
export function mergePaperclipEnvEntries(
@@ -114,12 +115,11 @@ export function mergePaperclipEnvEntries(
filePath = resolveEnvFilePath(),
): Record {
const current = readPaperclipEnvEntries(filePath);
+ const managedEntries = paperclipOwnedEntries(entries);
const next = {
...current,
- ...Object.fromEntries(
- Object.entries(entries).filter(([, value]) => typeof value === "string" && value.trim().length > 0),
- ),
+ ...managedEntries,
};
- writePaperclipEnvEntries(next, filePath);
+ writePaperclipEnvEntries(managedEntries, filePath);
return next;
}
diff --git a/cli/src/config/schema.ts b/cli/src/config/schema.ts
index 65ddeab73346..799d8ba0d753 100644
--- a/cli/src/config/schema.ts
+++ b/cli/src/config/schema.ts
@@ -8,11 +8,15 @@ export {
serverConfigSchema,
authConfigSchema,
telemetryConfigSchema,
+ updatesConfigSchema,
storageConfigSchema,
storageLocalDiskConfigSchema,
storageS3ConfigSchema,
secretsConfigSchema,
secretsLocalEncryptedConfigSchema,
+ mergePaperclipConfig,
+ findPaperclipConfigKeyWarnings,
+ type ConfigKeyWarning,
type PaperclipConfig,
type LlmConfig,
type DatabaseBackupConfig,
@@ -27,4 +31,5 @@ export {
type SecretsConfig,
type SecretsLocalEncryptedConfig,
type ConfigMeta,
+ type UpdatesConfig,
} from "../../../packages/shared/src/config-schema.js";
diff --git a/cli/src/config/store.ts b/cli/src/config/store.ts
index 8dddc777064e..b1ab0229d171 100644
--- a/cli/src/config/store.ts
+++ b/cli/src/config/store.ts
@@ -1,6 +1,11 @@
import fs from "node:fs";
import path from "node:path";
-import { paperclipConfigSchema, type PaperclipConfig } from "./schema.js";
+import { isDeepStrictEqual } from "node:util";
+import {
+ mergePaperclipConfig,
+ paperclipConfigSchema,
+ type PaperclipConfig,
+} from "./schema.js";
import {
resolveDefaultConfigPath,
resolvePaperclipInstanceId,
@@ -95,24 +100,132 @@ export function readConfig(configPath?: string): PaperclipConfig | null {
return parsed.data;
}
+function effectiveConfig(config: PaperclipConfig): Record {
+ const meta = { ...config.$meta } as Record;
+ delete meta.updatedAt;
+ delete meta.source;
+ return {
+ ...config,
+ $meta: meta,
+ };
+}
+
+function syncDirectory(directoryPath: string): void {
+ let directoryDescriptor: number | null = null;
+ try {
+ directoryDescriptor = fs.openSync(directoryPath, "r");
+ fs.fsyncSync(directoryDescriptor);
+ } catch (error) {
+ const code = error instanceof Error && "code" in error ? error.code : null;
+ if (process.platform !== "win32" || !["EACCES", "EINVAL", "EISDIR", "ENOTSUP", "EPERM"].includes(String(code))) {
+ throw error;
+ }
+ } finally {
+ if (directoryDescriptor !== null) fs.closeSync(directoryDescriptor);
+ }
+}
+
+function durableCopyFile(sourcePath: string, destinationPath: string, flags = 0): void {
+ fs.copyFileSync(sourcePath, destinationPath, flags);
+ fs.chmodSync(destinationPath, 0o600);
+
+ const backupDescriptor = fs.openSync(destinationPath, "r");
+ try {
+ fs.fsyncSync(backupDescriptor);
+ } finally {
+ fs.closeSync(backupDescriptor);
+ }
+ syncDirectory(path.dirname(destinationPath));
+}
+
+function atomicWriteFile(filePath: string, contents: string): void {
+ let attempt = 0;
+
+ while (true) {
+ const temporaryPath = `${filePath}.tmp-${process.pid}-${attempt}`;
+ attempt += 1;
+ let fileDescriptor: number | null = null;
+ try {
+ fileDescriptor = fs.openSync(temporaryPath, "wx", 0o600);
+ fs.writeFileSync(fileDescriptor, contents, "utf8");
+ fs.fsyncSync(fileDescriptor);
+ fs.closeSync(fileDescriptor);
+ fileDescriptor = null;
+ fs.renameSync(temporaryPath, filePath);
+ syncDirectory(path.dirname(filePath));
+ return;
+ } catch (error) {
+ if (fileDescriptor !== null) fs.closeSync(fileDescriptor);
+ fs.rmSync(temporaryPath, { force: true });
+ const code = error instanceof Error && "code" in error ? error.code : null;
+ if (code === "EEXIST") continue;
+ throw error;
+ }
+ }
+}
+
+export function backupInvalidConfig(configPath?: string): string {
+ const filePath = resolveConfigPath(configPath);
+ if (!fs.existsSync(filePath)) {
+ throw new Error(`Cannot back up missing config at ${filePath}`);
+ }
+
+ for (let suffix = 1; ; suffix += 1) {
+ const backupPath = `${filePath}.invalid-${suffix}`;
+ try {
+ durableCopyFile(filePath, backupPath, fs.constants.COPYFILE_EXCL);
+ return backupPath;
+ } catch (error) {
+ const code = error instanceof Error && "code" in error ? error.code : null;
+ if (code === "EEXIST") continue;
+ throw error;
+ }
+ }
+}
+
export function writeConfig(
config: PaperclipConfig,
configPath?: string,
-): void {
+ options: { invalidBackupPath?: string } = {},
+): boolean {
const filePath = resolveConfigPath(configPath);
const dir = path.dirname(filePath);
fs.mkdirSync(dir, { recursive: true });
+ let nextConfig = paperclipConfigSchema.parse(config);
+ if (fs.existsSync(filePath)) {
+ try {
+ const source = paperclipConfigSchema.parse(migrateLegacyConfig(parseJson(filePath)));
+ nextConfig = paperclipConfigSchema.parse(mergePaperclipConfig(source, nextConfig));
+ if (isDeepStrictEqual(effectiveConfig(source), effectiveConfig(nextConfig))) {
+ return false;
+ }
+ } catch (error) {
+ const invalidBackupPath = options.invalidBackupPath;
+ if (!invalidBackupPath) {
+ throw new Error(
+ `Refusing to overwrite invalid config at ${filePath}: ${error instanceof Error ? error.message : String(error)}`,
+ );
+ }
+ if (
+ !fs.existsSync(invalidBackupPath) ||
+ !fs.readFileSync(filePath).equals(fs.readFileSync(invalidBackupPath))
+ ) {
+ throw new Error(
+ `Refusing to overwrite ${filePath} because it changed after the invalid backup was created`,
+ );
+ }
+ }
+ }
+
// Backup existing config before overwriting
if (fs.existsSync(filePath)) {
const backupPath = filePath + ".backup";
- fs.copyFileSync(filePath, backupPath);
- fs.chmodSync(backupPath, 0o600);
+ durableCopyFile(filePath, backupPath);
}
- fs.writeFileSync(filePath, JSON.stringify(config, null, 2) + "\n", {
- mode: 0o600,
- });
+ atomicWriteFile(filePath, JSON.stringify(nextConfig, null, 2) + "\n");
+ return true;
}
export function configExists(configPath?: string): boolean {
diff --git a/doc/CLI.md b/doc/CLI.md
index 5b36f799589b..5c75aeb4e853 100644
--- a/doc/CLI.md
+++ b/doc/CLI.md
@@ -365,7 +365,7 @@ pnpm paperclipai agent runtime-state
pnpm paperclipai agent runtime-state:reset-session [--task-key ]
pnpm paperclipai agent task-sessions
pnpm paperclipai agent skills
-pnpm paperclipai agent skills:sync --desired-skills paperclip,github
+pnpm paperclipai agent skills:sync --desired-skills paperclip,github --mode add
pnpm paperclipai agent instructions-path:update --payload-json '{"path":"/path/to/AGENTS.md"}'
pnpm paperclipai agent instructions-bundle
pnpm paperclipai agent instructions-bundle:update --payload-json '{"mode":"managed"}'
@@ -465,9 +465,10 @@ By default the command creates a `todo` issue assigned to the target agent and w
1. **Company install** — adds or updates a row in `company_skills` for the
whole company. This is what `skills install`, `skills import`, `skills create`,
and `skills scan-projects` do.
-2. **Agent attach** — replaces an agent's *desired* company skill set
- (`skills agent sync`/`clear`). This is a desired-state operation on the
- agent's adapter config; it does not change the company library.
+2. **Agent attach** — merges an agent's *desired* company skill set with an
+ explicit `add`, `remove`, or `replace` mode (`skills agent sync`/`clear`).
+ This is a desired-state operation on the agent's adapter config; it does not
+ change the company library.
3. **Adapter runtime sync** — the adapter reconciles the desired skill set
with files on disk and reports an `AgentSkillSnapshot` (`skills agent list`).
`skills agent sync` triggers this automatically after updating desired state.
@@ -564,12 +565,14 @@ maintenance loop for catalog-installed skills:
```sh
pnpm paperclipai skills agent list --company-id
-pnpm paperclipai skills agent sync --skill [--skill ...] --company-id
+pnpm paperclipai skills agent sync --skill [--skill ...] --mode --company-id
pnpm paperclipai skills agent clear --yes --company-id
```
-`skills agent sync` replaces the agent's non-required desired skill set (it is
-not additive) and returns the resulting adapter `AgentSkillSnapshot`.
+`skills agent sync` requires a merge mode and returns the resulting adapter
+`AgentSkillSnapshot`. `add` preserves all unnamed assignments, `remove` deletes
+only named assignments, and `replace` destructively overwrites the complete
+non-required desired skill set.
`skills agent clear` sends an empty desired list. Required Paperclip skills are
still enforced by the server in both cases.
diff --git a/doc/DATABASE.md b/doc/DATABASE.md
index b5034991b111..6c80f1f50047 100644
--- a/doc/DATABASE.md
+++ b/doc/DATABASE.md
@@ -113,7 +113,18 @@ DATABASE_URL=postgres://postgres.[PROJECT-REF]:[PASSWORD]@aws-0-[REGION].pooler.
DATABASE_MIGRATION_URL=postgres://postgres.[PROJECT-REF]:[PASSWORD]@aws-0-[REGION].pooler.supabase.com:5432/postgres
```
-If your hosted database requires transaction-pooling-only connections, use a direct or session-pooled connection for Paperclip until runtime pooling support is documented in this guide. Do not edit database client source files as part of deployment setup.
+If your hosted database requires transaction-pooling-only connections (pgbouncer transaction mode, Supavisor port 6543, Neon `-pooler` endpoints), set `DATABASE_PREPARED_STATEMENTS=false` so the client does not rely on session-scoped prepared statements, and keep `DATABASE_MIGRATION_URL` on a direct connection. Do not edit database client source files as part of deployment setup.
+
+### Client tuning (optional)
+
+All of these are optional; when unset, the driver defaults apply and behavior is unchanged — typical self-hosted setups need none of them:
+
+```sh
+DATABASE_PREPARED_STATEMENTS=false # required for transaction-mode poolers; default: enabled
+DATABASE_POOL_MAX=25 # connection pool size; default: 10
+DATABASE_IDLE_TIMEOUT_SECONDS=60 # close idle pooled connections; default: keep open
+DATABASE_CONNECT_TIMEOUT_SECONDS=10 # default: 30
+```
### Push the schema
diff --git a/doc/DEVELOPING.md b/doc/DEVELOPING.md
index e16c40beafc9..ac53ac1b4f3b 100644
--- a/doc/DEVELOPING.md
+++ b/doc/DEVELOPING.md
@@ -136,7 +136,12 @@ at least one identity source. Supported-platform process probes fail explicitly
instead of silently treating a live PID as either the original owner or a
recycled process when identity cannot be established.
-Use `--drain-required` only when the deploy intentionally requires the old terminate-and-retry behavior. Without that flag, the old server verifies that the marker targets its own PID, snapshots currently running heartbeat run IDs and child PIDs, and skips the shutdown drain so eligible detached local-agent processes can keep running. On startup the new server writes `$PAPERCLIP_HOME/instances/${PAPERCLIP_INSTANCE_ID:-default}/hot-restart-report.json` with `previousServerPid`, `newServerPid`, `previousServerVersion`, `newServerVersion`, `adoptedRunIds`, `finalizedWhileDownRunIds`, `lostRunIds`, and per-run classifications before the normal orphan reaper runs.
+Use `--drain-required` only when the deploy intentionally requires the old terminate-and-retry behavior. Without that flag, the old server verifies that the marker targets its own PID, stops new scheduler work, waits for any queue-claim callback already in flight, snapshots currently running heartbeat run IDs and child PIDs, and skips the shutdown drain so eligible detached local-agent processes can keep running. ACP-backed local runs use server-owned stdio and cannot survive their parent server, so the old server instead persists their complete snapshot, changes the marker to `drainRequired` with `drainReason: "active_acp_run"`, and drains only those runs to queued retries. Detached CLI runs remain eligible for adoption during the same mixed restart. If an ACP process terminates but its terminal run update does not persist, startup classifies it as lost with reason `selective_drain_not_finalized` rather than treating the drain as successful. On startup the new server writes `$PAPERCLIP_HOME/instances/${PAPERCLIP_INSTANCE_ID:-default}/hot-restart-report.json` with `previousServerPid`, `newServerPid`, `previousServerVersion`, `newServerVersion`, `drainReason`, `adoptedRunIds`, `finalizedWhileDownRunIds`, `lostRunIds`, and per-run classifications before the normal orphan reaper runs.
+
+When Paperclip manages embedded PostgreSQL, it suppresses that dependency's eager
+`SIGINT`/`SIGTERM` cleanup hooks. Paperclip owns signal ordering so the heartbeat
+snapshot and any required drain complete while the database is still available;
+the coordinated shutdown path stops embedded PostgreSQL afterward.
The request command records the preflight set of running heartbeat IDs and writes
an instance-scoped marker plus a PID-targeted legacy home-root handoff marker.
@@ -187,6 +192,13 @@ An alive child appears in `adoptedRunIds`; a child that completed during the
restart window appears in `finalizedWhileDownRunIds`. Either is continuous. A
`lostRunIds` entry remains a failed deploy and must not be waived.
+For a recovery from a version that can stop embedded PostgreSQL before writing
+its shutdown snapshot, use `--drain-required` once to cross the broken boundary.
+After the fixed server is live, perform another ordinary hot restart. Require
+`lostRunIds` to be empty and every preflight run to appear in either
+`adoptedRunIds` or `finalizedWhileDownRunIds`; an ACP-backed original should be
+finalized and have a queued retry rather than be adopted.
+
Tailscale/private-auth dev mode:
```sh
@@ -324,6 +336,14 @@ Every local install keeps runtime state directly under the selected instance roo
`PAPERCLIP_HOME` and `PAPERCLIP_INSTANCE_ID` override the home root and instance id respectively. `paperclipai onboard` echoes the resolved values in its banner (`Local home: | instance: | config: `) so you can confirm where state will land before continuing.
+Config updates preserve unrecognized top-level and nested keys so provider or
+plugin extensions survive `configure` and worktree port repair. Likely
+misspellings of known keys produce a warning but are not removed. If an
+existing `config.json` is malformed, `onboard` and `configure` first create a
+byte-for-byte sibling backup named `config.json.invalid-1` (then `-2`, and so
+on). Repair from defaults requires an interactive confirmation; non-interactive
+runs stop without replacing the original.
+
## Database in Dev (Auto-Handled)
For local development, leave `DATABASE_URL` unset.
@@ -451,6 +471,7 @@ The default `worktree init` still seeds eagerly and writes `seed-complete` immed
- `pnpm paperclipai worktree ensure-seeded` performs the deferred seed **exactly once**. It is lock-guarded and idempotent: a present `seed-complete` marker or a missing `seed-pending` marker short-circuits it, so it is safe to call repeatedly and from concurrent processes. It reads the source instance from the `seed-pending` marker unless you pass `--from-config`.
- `paperclipai run` calls `ensureWorktreeSeeded` automatically before doctor/boot, so `run` transparently seeds a lean worktree on first launch.
+- Managed git-worktree runtime startup also runs `scripts/provision-worktree-runtime.sh` automatically when a legacy workspace policy has no explicit runtime provision command and the worktree is still `seed-pending`. An explicitly configured runtime provision command always takes precedence.
- Worktrees created before lazy seeding shipped have neither marker; they are treated as already-seeded for backward compatibility (never re-cloned).
**Seed-pending guard.** `pnpm dev` (the dev-runner) refuses to boot a worktree whose database is still `seed-pending` and points you at the fix:
diff --git a/doc/SPEC-implementation.md b/doc/SPEC-implementation.md
index ce4d6d28e66a..e11298ef817e 100644
--- a/doc/SPEC-implementation.md
+++ b/doc/SPEC-implementation.md
@@ -230,6 +230,7 @@ Routine execution issues add a routine-scoped env overlay after project env and
- `description` text null
- `status` enum: `backlog | todo | in_progress | in_review | done | blocked | cancelled`
- `priority` enum: `critical | high | medium | low`
+- `review_policy` nullable enum: `anyone | not_creator | human_only`; null is equivalent to `anyone`
- `assignee_agent_id` uuid fk `agents.id` null
- `assignee_user_id` text null
- checkout/execution locks: `checkout_run_id`, `execution_run_id`, `execution_agent_name_key`, `execution_locked_at`
@@ -531,8 +532,9 @@ Detailed ownership, execution, blocker, active-run watchdog, crash-recovery, and
- Bearer API key mapped to one agent and company
- Agent key scope:
- read org/task/company context for own company
- - read/write own assigned tasks and comments
- - create tasks/comments for delegation
+ - read company-visible tasks and comments
+ - comment on and update visible tasks under the shared write rule
+ - create child tasks and assign visible work for delegation under the same rule
- report heartbeat status
- report cost events
- Agent cannot:
@@ -557,6 +559,40 @@ Detailed ownership, execution, blocker, active-run watchdog, crash-recovery, and
| Manage another user's inbox state | yes | scoped `inbox:manage` grant |
| Set work-object visibility (issue/project) | no | no (pro gate) |
+### 9.3.1 Shared default-open issue writes
+
+For standard-trust agents, issue comments, issue field/status updates, child
+creation under a parent, and assignment share one authorization rule: the
+target issue must be visible to the agent and the responsible user represented
+by the run must also be authorized. In V1, issue visibility defaults to the
+whole company, so these writes are company-wide by default.
+
+The shared rule does not widen low-trust, `skill_test`, or `task_bridge` key
+scopes. It also does not replace run-lifecycle controls: checkout ownership,
+active-run conflicts, status-transition validation, interaction ownership,
+budget gates, and pause gates remain independently enforced. Comment access is
+structurally downstream of issue read access (`issue:comment` is a subset of
+`issue:read`).
+
+Cross-issue writes are contained per heartbeat run. An agent-authored comment
+may wake the target assignee, including an explicit `resume: true` comment on a
+`done` or `cancelled` issue, but the wake remains agent-class and is subject to
+the normal agent rewake throttle; comment presentation cannot give it human
+wake privileges. Agent issue comments and updates require a persisted heartbeat
+run bound to the authenticated agent and company; missing, invalid, or mismatched
+run context fails closed before mutation. A run may attempt at most 20 cross-issue comments or issue
+updates across the shared counter. The server records each attempt with its
+source issue, target issue, run, count, and rollout mode, and fails closed with
+the cap in the error once enforcement is active. Assignee self-comments do not
+wake the assignee, and a non-assignee comment cannot mint a mention grant.
+
+Agent-authored issue comments persist the responsible user derived from the
+authenticated actor; clients cannot choose that attribution. Each comment also
+records the write-policy reason, and spoof attempts fail with an audited 422.
+Every issue PATCH emits an `issue.updated` activity receipt containing the
+actor, responsible user, run, authorization reason, and field-level before/after
+changes so both agent and board edits are visible in the issue activity stream.
+
## 9.4 Permission Terminology and Default Visibility Rule
Paperclip V1 keeps a company-scoped visibility model as the default because centralized authorization and scoped work-object controls are not yet a core V1 control surface.
@@ -611,6 +647,15 @@ The approved term set is:
When multiple constraint families are present, assignment must satisfy all of them. Denials return `403` with a generic scope explanation and do not disclose details about hidden or unrelated resources.
+A protected-agent hard block is represented canonically as
+`authorizationPolicy.protectedAgent.blockAssignment: true`. It denies assignment
+even when the caller has a broad or scoped assignment grant. A company
+administrator must remove the block before assignment can be retried; no pending
+approval is created. The legacy fields `protectedAgent.requiresApproval` and
+`assignmentPolicy.protectedAgentRequiresApproval` remain fail-closed compatibility
+aliases for the same hard block, but API denial copy must describe the block and
+administrator remediation rather than promising a nonexistent approval step.
+
## 9.9 Task Watchdog Authority Contract
A task watchdog is a scoped execution capacity for a configured watchdog agent on one watched issue subtree. It is not a separate principal, does not inherit board auth, and does not expand the selected agent's company boundary. The server must enforce the watchdog contract from persisted watchdog configuration and run context; custom instructions and prompt text can narrow the mandate but cannot expand it.
@@ -848,6 +893,20 @@ All endpoints are under `/api` and return JSON.
- `PATCH /companies/:companyId/branding`
- `POST /companies/:companyId/archive`
+On a Paperclip Cloud-managed instance, `POST /companies` returns `403` with
+code `cloud_managed`; the trusted-header provisioning path and company import
+routes remain the only company-creation paths there.
+
+## 10.1.1 Cloud Stack Portfolio
+
+- `GET /cloud/stacks`
+
+The route exists only on a Cloud-managed instance, requires a trusted
+`cloud_tenant` actor, and proxies the current actor's user id plus the current
+stack id to the Cloud tenant portfolio endpoint. Client-supplied user ids are
+never forwarded. Successful responses are cached briefly per user; self-hosted
+instances return `404`.
+
## 10.2 Goals
- `GET /companies/:companyId/goals`
@@ -979,7 +1038,10 @@ The current app also exposes V1-supporting surfaces for:
`GET /companies/:companyId/search/extract`; extraction accepts a server-escaped literal `contains`, optional
server-owned URL expansion, issue/comment/document scopes, status/date filters, issue-level pagination, a
bounded `matchesPerIssue` override for machine consumers, and explicit issue/match truncation flags
-- execution workspaces, project workspaces, workspace runtime services, and workspace operations
+- execution workspaces, project workspaces, workspace runtime services, and workspace operations. Workspace reads
+ derive `deliveryState` as `merged_via_pr | merged_by_ancestry | unmerged | unknown`; terminal issue trees with a
+ merged delivery and no active checkout run become cleanup-eligible with reason `issue_terminal` and are archived
+ through the workspace cleanup path. Reopening the source issue records activity but does not restore that workspace.
- task watchdog configuration and reusable watchdog issue orchestration for explicitly watched issue subtrees
- routines and scheduled/API/webhook triggers
- plugin installation, configuration, state, jobs, logs, webhooks, and plugin database namespace migration
diff --git a/doc/connections/CONNECTOR-PLAYBOOK.md b/doc/connections/CONNECTOR-PLAYBOOK.md
index 2c954e7090f4..28b57f519da5 100644
--- a/doc/connections/CONNECTOR-PLAYBOOK.md
+++ b/doc/connections/CONNECTOR-PLAYBOOK.md
@@ -183,6 +183,114 @@ Recommended defaults for a new catalog entry:
If a gallery card cannot pass this path against a real vendor, de-list it or mark it unavailable until the missing auth, transport, or governance dependency is fixed.
+## MCP-Direct Connections (Hosted MCP + OAuth)
+
+Many vendors now expose an official hosted MCP server whose authorization
+server is discovered from the MCP endpoint itself, instead of documenting fixed
+OAuth URLs. For these connectors the manifest's `oauth` block is a hint at
+most; the broker resolves endpoints at connect time:
+
+1. `GET ` unauthenticated returns `401` with a `WWW-Authenticate`
+ header naming the protected-resource metadata URL (RFC 9728).
+2. `GET /.well-known/oauth-protected-resource[/]` names the
+ authorization server(s).
+3. `GET /.well-known/oauth-authorization-server` (RFC 8414) yields
+ `authorization_endpoint`, `token_endpoint`, and — when the vendor supports
+ dynamic registration — `registration_endpoint`.
+
+The broker implements this in `discoverOAuthEndpoints`
+(`server/src/services/tool-access.ts`), but discovery is **not**
+unconditional. `oauthEndpointsForConnection` resolves endpoints in this
+order:
+
+1. If the manifest's method `defaults` ship a **complete** pair
+ (`authorizationEndpoint` **and** `tokenEndpoint`), those are used
+ unconditionally. `discoverOAuthEndpoints` never runs in this case, so
+ endpoints stored on the connection's own OAuth config and 401 challenge
+ hints are **not consulted at all**.
+2. Otherwise, for `mcp_remote` connections, the broker calls
+ `discoverOAuthEndpoints`, which first checks endpoints already stored on
+ the connection's own OAuth config (falling back field-by-field to the 401
+ challenge hints); a complete stored/hinted pair is used as-is — no
+ `.well-known` fetch.
+3. Only when neither of the above yields a complete pair does the broker run
+ the RFC 9728 → RFC 8414 discovery chain above.
+
+Consequence: complete manifest endpoint hints are **authoritative, not
+hints** — they override even endpoints that an earlier discovery persisted
+on the connection, and if they go stale the broker keeps using them. For
+discovery-capable vendors, ship only `serverUrl` in `defaults` (as
+`notion.json` does) so the broker discovers fresh endpoints at connect
+time; add explicit `authorizationEndpoint`/`tokenEndpoint` only for vendors
+that do not publish RFC 9728/8414 metadata, and then own keeping them
+current.
+
+### Dynamic client registration (RFC 7591)
+
+Vendors whose authorization server advertises a `registration_endpoint` and
+supports public clients (`token_endpoint_auth_method: "none"` plus PKCE S256)
+need **no pre-provisioned OAuth app at all**. At first connect the broker
+registers a client on the fly and stores it on the connection:
+
+- Registration request: `client_name` `Paperclip ()`,
+ `redirect_uris` = the instance's own callback, `grant_types`
+ `["authorization_code", "refresh_token"]`, `response_types` `["code"]`,
+ `token_endpoint_auth_method` `"none"`.
+- The issued `client_id` is persisted in the connection's OAuth config and any
+ issued `client_secret` becomes a `company_secrets` ref. The registered
+ client is **reused** for every later authorize/refresh on that connection —
+ re-registering orphans prior grants on providers that bind grants to the
+ client.
+- Env-registered clients always win: when
+ `PAPERCLIP_TOOL_OAUTH__CLIENT_ID/_SECRET` are configured, the
+ broker uses them (`customer` ownership) and skips registration. List both
+ `customer` and `dcr` in the method's `ownershipModes` when the vendor
+ supports both.
+
+**DCR needs neither Paperclip ID nor Paperclip Connect.** DCR is always
+instance-local (ratified in the PAP-14828 connector-service spec, section 10
+item 8.4: "DCR is always instance-local; the service has no DCR involvement").
+Each instance registers its own public client with the vendor and uses its own
+`/api/tools/oauth/callback` redirect. **Cloud-hosted and self-hosted instances
+use the SAME path** — the only per-instance difference is the hostname inside
+the redirect URI. `id.paperclip.ing` authenticates operators only and never
+holds resource tokens; `connect.paperclip.ing` is a fallback only for
+providers that genuinely require a pre-registered public redirect, which a DCR
+provider by definition does not.
+
+### Redirect-URI constraints
+
+Vendors restrict what `redirect_uris` a dynamic client may register. Record
+the probed constraint in the `AppDefinition` `redirectConstraints` field and
+enforce it before starting OAuth. The first supported value is
+`https-or-loopback-http` (Notion's rule): HTTPS on any host — public or
+private — or plain HTTP only on loopback (`localhost`, `*.localhost`, `::1`,
+`127.0.0.0/8`). A plain-HTTP non-loopback origin fails fast with
+`oauth_redirect_origin_unsupported` ("This provider requires an HTTPS or
+loopback origin. Configure TLS before connecting.") and a pointer to the TLS
+deployment docs, instead of a confusing vendor-side `invalid_redirect_uri`.
+Probe the constraint with real registration attempts before writing the
+manifest — the redirect-URI rule and browser-reachability are independent
+axes; a private HTTPS host can be fine even when plain HTTP is not.
+
+### Documentation standards for every connection doc
+
+Every connection doc — playbook appendix, proposal, or user-facing doc —
+must include all three of the following (they are part of the template below):
+
+1. **Service involvement statement.** Say explicitly whether Paperclip ID or
+ Paperclip Connect participates in the flow. For RFC 7591 DCR providers the
+ answer is always: neither — DCR is instance-local and cloud vs self-hosted
+ use the same path.
+2. **Sequence diagram + exact endpoints.** A sequence diagram of how the
+ connection works, and the exact paths/endpoints used for auth: authorize,
+ token, registration (if DCR), and the Paperclip callback. Keep mermaid
+ sources next to the doc; do not put semicolons inside mermaid message text
+ (they parse as statement separators).
+3. **Administrator setup instructions.** Step-by-step: what (if anything) an
+ admin must register — callback URLs? client credentials? nothing, for DCR? —
+ where to register it, and how to verify the connection works end to end.
+
## Template
Copy this section into a connector proposal or implementation issue.
@@ -208,6 +316,25 @@ Copy this section into a connector proposal or implementation issue.
- Secret storage: company_secrets refs only
- Revocation behavior:
+## Connection Flow (mandatory)
+
+- Sequence diagram:
+- Auth endpoints (exact paths):
+ - Authorize:
+ - Token:
+ - Registration (if DCR):
+ - Discovery (.well-known), if any:
+ - Paperclip callback: `/api/tools/oauth/callback` (or n/a)
+- Redirect constraints (probed): none / https-or-loopback-http / requires-public-redirect
+- Paperclip ID / Paperclip Connect involvement: <"none — DCR is instance-local; cloud and self-hosted use the same path" for RFC 7591 providers; otherwise name the role>
+
+## Administrator Setup (mandatory)
+
+- What the admin must register (callback URLs? client credentials? nothing for DCR?):
+- Where to register it:
+- Instance prerequisites (TLS, base URL, feature flags):
+- How to verify the connection works:
+
## Resource Filters
- Required filters:
@@ -363,3 +490,264 @@ Linear's real-vendor evidence belongs in [PAP-12373](/PAP/issues/PAP-12373). The
Connector proposals now target the versioned `AppDefinition` contract in `packages/shared/src/types/app-definition.ts`. Seed data is one JSON file per provider under `packages/shared/src/app-definitions/`; regenerate Wave 1 with `pnpm connections:ingest-app-definitions`. The generator parses all 99 captured templates, validates required placeholders, OAuth ownership modes, and API-key placement, and produces deterministic output for review. FIRST-30 remains authoritative for `riskTier` and `requiredResourceFilters`; managed ownership modes stay data-visible but runtime-hidden until availability is injected.
+## Appendix: Notion Dry Run (MCP-Direct With DCR)
+
+This dry run applies the template to Notion, the first MCP-direct connector to
+ship with RFC 7591 dynamic client registration (PAP-16637; server
+implementation PAP-16649, PR #11009). Unlike the Linear appendix, every
+endpoint and constraint below comes from a live request log, not vendor docs
+alone.
+
+### Vendor
+
+- App key: `notion`
+- App name: Notion
+- First-30 classification: MCP-direct. Notion ships an official hosted MCP
+ server; its ~20 `notion-*` tools map directly to Paperclip grants.
+- Reason for classification: no shim or wrapper needed — the hosted server
+ speaks Streamable HTTP, which `server/src/services/mcp-http.ts` already
+ handles. The FIRST-30 matrix's "thin wrapper for block/database policy" is
+ explicitly deferred; v1 enforcement is gateway policy plus filters-as-config.
+- Security tier: S3 — workspace content read/write, but no payments, tenant
+ admin, or production infrastructure.
+- Plugin needed: No. Gallery card, OAuth connect, filters, catalog, profiles,
+ policies, and audit cover the UX.
+
+### Transport And Auth
+
+- Transport: `mcp_remote`
+- Endpoint: `https://mcp.notion.com/mcp` (Streamable HTTP; `/sse` fallback exists)
+- Auth mode: OAuth, endpoints resolved by discovery (RFC 9728 → RFC 8414),
+ public client via RFC 7591 DCR with PKCE S256 mandatory. Discovery runs
+ because `notion.json` deliberately ships only `serverUrl` — no
+ `authorizationEndpoint`/`tokenEndpoint` hints, which would otherwise take
+ precedence and be used verbatim (see "MCP-Direct Connections" above).
+- Ownership modes: `dcr` (default, zero setup) and `customer`
+ (env-registered classic integration via
+ `PAPERCLIP_TOOL_OAUTH_NOTION_CLIENT_ID/_SECRET`, which always wins when set).
+- Token behavior: access tokens last ~8 h (`expires_in` authoritative).
+ Refresh tokens **rotate on every refresh** — the old token is invalidated
+ (at most 2 valid per grant) and replaying a stale one can revoke the whole
+ grant, so the broker persists the rotated token before publishing the new
+ access token and serializes refresh per connection. Absolute expiry 180
+ days, inactivity expiry 30 days. `invalid_grant` on refresh is terminal:
+ clear tokens, require re-auth, never retry.
+- Secret storage: access/refresh tokens and any DCR `client_secret` are
+ `company_secrets` refs; the DCR `client_id` persists on the connection and
+ is reused — re-registering would orphan prior grants.
+- Revocation behavior: disabling or revoking the connection removes
+ `notion-*` tools from agent sessions and denies brokered execution on the
+ next gateway check.
+
+### Connection Flow (mandatory)
+
+Paperclip ID / Paperclip Connect involvement: **none — DCR is instance-local**
+(PAP-14828 spec section 10 item 8.4); **cloud-hosted and self-hosted use the
+same path**. The only per-instance difference is the hostname in the redirect
+URI.
+
+Auth endpoints (exact paths, from the live discovery chain):
+
+| Role | Endpoint |
+| --- | --- |
+| MCP server | `https://mcp.notion.com/mcp` |
+| Protected-resource metadata (RFC 9728) | `https://mcp.notion.com/.well-known/oauth-protected-resource/mcp` |
+| AS metadata (RFC 8414) | `https://mcp.notion.com/.well-known/oauth-authorization-server` |
+| Authorize | `https://mcp.notion.com/authorize` |
+| Token (exchange + refresh) | `https://mcp.notion.com/token` |
+| Registration (RFC 7591 DCR) | `https://mcp.notion.com/register` |
+| Paperclip connect (wizard) | `POST /api/companies/:companyId/tools/apps/connect` |
+| Paperclip OAuth start | `POST /api/tools/oauth/:connectionId/start` |
+| Paperclip callback | `GET /api/tools/oauth/callback` |
+
+Redirect constraints (probed): `https-or-loopback-http`.
+
+```mermaid
+sequenceDiagram
+ autonumber
+ actor U as User's browser
+ participant UI as Paperclip UI
/PAP/apps/connect?source=notion
+ participant S as Paperclip instance server
(cloud or self-hosted — same path)
+ participant M as mcp.notion.com
(MCP server + OAuth AS)
+ participant N as Notion web
(app.notion.com, notion.com)
+
+ U->>UI: Click "Connect" (deep link ?source=notion)
+ UI->>S: POST /companies/:id/tools/apps/connect { appKey: "notion" }
+ S->>M: GET /.well-known/oauth-protected-resource (RFC 9728)
+ M-->>S: authorization_servers → mcp.notion.com
+ S->>M: GET /.well-known/oauth-authorization-server (RFC 8414)
+ M-->>S: authorize / token / registration endpoints
+ alt First connect on this instance (no stored client, no env client)
+ S->>M: POST registration_endpoint (RFC 7591 DCR, public client, PKCE-only)
+ M-->>S: client_id (persisted, REUSED for every later connect)
+ else Client already known
+ S->>S: Reuse stored DCR client_id (or env-registered client if configured)
+ end
+ S-->>UI: auth.startUrl (authorize URL + PKCE S256 challenge + state)
+ UI->>U: Redirect browser to startUrl
+ U->>M: GET /authorize?client_id + code_challenge + state
+ M->>N: 302 to app.notion.com/install-integration
+ N->>N: notion.com/login (only if signed out)
+ N-->>U: Consent page: pick workspace, approve integration
+ U->>S: 302 to GET /api/tools/oauth/callback?code&state (instance's OWN callback)
+ S->>M: POST token_endpoint (code + code_verifier)
+ M-->>S: access_token (~8 h) + rotating refresh_token
+ S->>S: Store tokens as company_secrets refs (server-side only)
+ S-->>U: Redirect to wizard actions/review step (?oauth=connected)
+ Note over S,M: Later: agent runs reach notion-* tools via the managed MCP gateway.
Server refreshes ahead of use — each refresh ROTATES the refresh token.
+```
+
+### Dry-Run Request Log (PAP-16649, 2026-08-06/07)
+
+The verified request sequence for a first connect:
+
+1. `GET https://mcp.notion.com/mcp` → `401` with `WWW-Authenticate` naming
+ `https://mcp.notion.com/.well-known/oauth-protected-resource/mcp`.
+2. `GET https://mcp.notion.com/.well-known/oauth-protected-resource/mcp` →
+ `200`; authorization server `https://mcp.notion.com`, scope `default`.
+3. `GET https://mcp.notion.com/.well-known/oauth-authorization-server` →
+ `200`; `/authorize`, `/token`, `/register`; `token_endpoint_auth_method`
+ `none` supported; PKCE `S256` supported.
+4. `POST https://mcp.notion.com/register` (RFC 7591).
+5. Browser `GET https://mcp.notion.com/authorize` → Notion consent
+ (`app.notion.com/install-integration`, `notion.com/login` if signed out).
+6. `POST https://mcp.notion.com/token` for code exchange and every refresh.
+7. `POST https://mcp.notion.com/mcp` for MCP traffic.
+
+Redirect-URI probes against `/register`:
+
+| Probed `redirect_uris` value | Result |
+| --- | --- |
+| `http://paperclip-dev:3100/api/tools/oauth/callback` | 400 `invalid_redirect_uri` — "Redirect URI must use HTTPS unless it is a loopback HTTP URI" |
+| `https://paperclip-dev:3100/api/tools/oauth/callback` | Accepted — private host is fine over HTTPS |
+| `http://localhost:3100/api/tools/oauth/callback` | Accepted |
+| `http://127.0.0.1:3100/api/tools/oauth/callback` | Accepted |
+
+Hence `redirectConstraints: "https-or-loopback-http"` in `notion.json`, and
+the broker's fail-fast `oauth_redirect_origin_unsupported` error for
+plain-HTTP non-loopback origins.
+
+### Administrator Setup (mandatory)
+
+- What the admin must register: **nothing**. Notion's authorization server
+ supports RFC 7591 DCR, so the instance registers its own public client on
+ first connect. No Notion integration, no client credentials, no callback
+ registration, no Paperclip ID or Paperclip Connect involvement.
+- Optional escape hatch: to use a pre-registered classic Notion integration
+ instead, set `PAPERCLIP_TOOL_OAUTH_NOTION_CLIENT_ID` and
+ `PAPERCLIP_TOOL_OAUTH_NOTION_CLIENT_SECRET`; the env client always takes
+ precedence (`customer` ownership).
+- Instance prerequisites: the instance base URL must be HTTPS on any host or
+ loopback HTTP (Notion's redirect-URI rule). A plain-HTTP non-loopback origin
+ gets "This provider requires an HTTPS or loopback origin. Configure TLS
+ before connecting." — add TLS first (e.g. a tailscale cert, as
+ paperclip-dev did). The `enableApps` experimental setting must be on for
+ `/apps/*` routes. The connecting user must be allowed to install
+ integrations in their Notion workspace.
+- How to verify: visit `/PAP/apps/connect?source=notion`, complete the Notion
+ consent flow, and land on the wizard's actions step listing `notion-*`
+ tools. Then confirm an agent run sees Notion tools through the runtime MCP
+ gateway and that a write call (e.g. `notion-create-pages`) opens an
+ ask-first action request.
+
+### Resource Filters
+
+- Required filters: workspace, page, database (per FIRST-30).
+- Optional filters: object type, database/data-source scope.
+- Write-enabling filters: workspace plus page/database scope for
+ create/update.
+- Enforced by: gateway policy plus filters-as-config in v1; the FIRST-30
+ "thin wrapper for block/database policy" is explicitly deferred. Notion-side
+ scoping also applies — the consent step lets the user share only selected
+ pages/databases with the integration.
+
+### Manifest Sketch
+
+The shipped `packages/shared/src/app-definitions/notion.json` (regenerate via
+`pnpm connections:ingest-app-definitions`):
+
+```json
+{
+ "schemaVersion": 1,
+ "slug": "notion",
+ "name": "Notion",
+ "description": "Read and update pages in your Notion workspace.",
+ "urlPatterns": ["https://mcp.notion.com/*"],
+ "methods": [
+ {
+ "key": "mcp-oauth",
+ "transport": "mcp_remote",
+ "auth": "oauth",
+ "ownershipModes": ["customer", "dcr"],
+ "defaults": { "serverUrl": "https://mcp.notion.com/mcp" },
+ "riskTier": "S3",
+ "requiredResourceFilters": ["workspace", "page", "database"]
+ }
+ ],
+ "redirectConstraints": "https-or-loopback-http"
+}
+```
+
+### Actions
+
+Notion's hosted server exposes ~20 `notion-*` tools. Representative risk
+classes below; the full catalog review with per-tool defaults is PAP-16652
+(P4), and changed-action quarantine applies as usual.
+
+| Tool | Risk | Default status | Filters | Approval default | Audit fields | Negative case |
+| --- | --- | --- | --- | --- | --- | --- |
+| `notion-search` | read | active after catalog review; plan-gated by Notion (needs Notion AI) — may list but fail at call time | workspace | allow when profile includes Notion reads | query summary, result count | Ungranted agent cannot invoke. |
+| `notion-fetch` | read | active after catalog review | workspace, page, database | allow when profile includes Notion reads | page/database id | Fetch outside shared pages fails Notion-side and is audited. |
+| `notion-create-pages` | write | active only after review; changed versions quarantined | workspace, page, database | ask-first by default | parent id, title hash, created page id | Missing workspace/page filter denies. |
+| `notion-update-page` | write | active only after review; changed versions quarantined | workspace, page | ask-first by default | page id, redaction summary | Revoked connection blocks retry. |
+| `notion-query-data-sources` | read | active after catalog review | workspace, database | allow when profile includes Notion reads | data-source id, result count | Granted agent cannot query a disallowed database. |
+
+No destructive Notion action ships in the first pass; any future
+delete/archive/bulk action starts quarantined pending SecurityEngineer review.
+
+### Wizard Path
+
+1. Operator opens `/PAP/apps/connect?source=notion` (or the Notion gallery
+ card → Connect). The deep link POSTs connect immediately and redirects the
+ browser to `auth.startUrl`.
+2. Operator completes Notion consent (workspace picker → approve).
+3. Notion redirects to the instance's own `GET /api/tools/oauth/callback`;
+ Paperclip exchanges the code, stores token material in `company_secrets`,
+ and returns the operator to the wizard (`?oauth=connected`).
+4. Operator confirms resource filters and default ask-first writes.
+5. Paperclip runs health check and catalog refresh; `notion-*` tools appear
+ on the actions step.
+6. Write actions stay ask-first until the operator approves calls or creates
+ narrow trust rules.
+
+Error state: on a plain-HTTP non-loopback instance, step 1 fails fast with
+the TLS guidance error above — the operator never reaches Notion.
+
+### Governance Defaults
+
+- Default profile: Notion read actions for the selected scope; writes opt-in.
+- Policy defaults: ask-first for `notion-create-pages`, `notion-update-page`,
+ and comment writes; block any unreviewed destructive action.
+- Quarantine: new or schema-changed write actions receive
+ `quarantineReason: "pending_review"` and are hidden from agent tool lists.
+- Rate limits: per-connection search/fetch budget to protect vendor quota.
+- Audit: log connect, DCR registration, config/filter changes, grant changes,
+ action requests, allowed/denied calls, token refresh failures, revoke, and
+ catalog quarantine events.
+
+### Validation Hook
+
+End-to-end evidence belongs to PAP-16654 (P6) and the PAP-12373 matrix:
+
+- Zero-setup OAuth connect succeeds on
+ `https://paperclip-dev.tail29c1aa.ts.net/PAP/apps/connect?source=notion`
+ with no pre-provisioned OAuth env vars (proves DCR).
+- Catalog discovery lists `notion-*` tools; new/changed risky actions are
+ quarantined.
+- An agent run sees Notion tools through the managed runtime MCP gateway.
+- `notion-create-pages` opens ask-first review and executes only after
+ approval.
+- Revocation removes Notion tools and blocks execution.
+- Audit rows prove actor, run/issue context, connection, tool, decision,
+ reason code, and outcome.
+
diff --git a/docs/adapters/claude-local.md b/docs/adapters/claude-local.md
index 76cb4b337b4b..1a13d0459c1c 100644
--- a/docs/adapters/claude-local.md
+++ b/docs/adapters/claude-local.md
@@ -8,8 +8,9 @@ The `claude_local` adapter runs Anthropic's Claude Code CLI locally. It supports
## Prerequisites
- Claude Code CLI installed (`claude` command available)
-- Either `ANTHROPIC_API_KEY` in adapter env/host env, or a Claude Code
- subscription login available to the execution target
+- Either `ANTHROPIC_API_KEY` or `CLAUDE_CODE_OAUTH_TOKEN` in adapter or
+ environment env (or host env), or a Claude Code subscription login
+ available to the execution target
## Configuration Fields
@@ -72,6 +73,7 @@ The adapter creates a temporary directory with symlinks to Paperclip skills and
## Remote credential ownership
+When no API key or `CLAUDE_CODE_OAUTH_TOKEN` is configured,
`claude_local` uses a snapshot-owns-auth topology for managed sandbox execution
targets. When the run uses a sandbox execution target and no explicit
`CLAUDE_CONFIG_DIR` is configured, Paperclip creates a remote
@@ -111,5 +113,12 @@ Use the "Test Environment" button in the UI to validate the adapter config. It c
- Claude CLI is installed and accessible
- Working directory is absolute and available (auto-created if missing and permitted)
-- API key/auth mode hints (`ANTHROPIC_API_KEY` vs subscription login)
+- API key/auth mode hints (`ANTHROPIC_API_KEY` vs `CLAUDE_CODE_OAUTH_TOKEN` vs subscription login)
- A live hello probe (`claude --print - --output-format stream-json --verbose` with prompt `Respond with hello.`) to verify CLI readiness
+
+The probe sees the same layered env as a real run: when an environment is
+selected, its environment variables (secret refs included) are resolved and
+merged under the adapter config's `env`, so environment-level auth is
+reflected in the test result. A secret binding that is missing surfaces as
+an `environment_env_binding_missing` failure instead of a silently passing
+probe.
diff --git a/docs/adapters/overview.md b/docs/adapters/overview.md
index 83c14e10efb6..6fdf1723f345 100644
--- a/docs/adapters/overview.md
+++ b/docs/adapters/overview.md
@@ -39,7 +39,7 @@ before the CLI starts:
| Adapter | Credential topology | Which credential file wins on managed sandbox targets |
|---------|---------------------|-------------------------------------------------------|
| [`codex_local`](/adapters/codex-local) | Host-owns-auth for Paperclip-managed `CODEX_HOME` | A host-owned `auth.json` is symlinked into the managed `CODEX_HOME` and uploaded to the sandbox. If a per-agent `OPENAI_API_KEY` is configured, Paperclip writes an API-key `auth.json` instead and that file wins. A login baked into the sandbox image is shadowed because Codex runs with Paperclip's uploaded `CODEX_HOME`. |
-| [`claude_local`](/adapters/claude-local) | Snapshot-owns-auth for managed remote Claude config | Paperclip uploads only sanitized settings and skill/runtime assets. When the remote managed config has no Claude credential files, it copies `.credentials.json` or `credentials.json` from the sandbox image's own `$HOME/.claude`, so the image's login wins. |
+| [`claude_local`](/adapters/claude-local) | Snapshot-owns-auth for managed remote Claude config | A configured `ANTHROPIC_API_KEY` or `CLAUDE_CODE_OAUTH_TOKEN` (agent or environment env) wins over any stored login. Otherwise Paperclip uploads only sanitized settings and skill/runtime assets, and when the remote managed config has no Claude credential files it copies `.credentials.json` or `credentials.json` from the sandbox image's own `$HOME/.claude`, so the image's login wins. |
Worked examples:
diff --git a/docs/api/issues.md b/docs/api/issues.md
index e63e70c17263..bebea9f76ea0 100644
--- a/docs/api/issues.md
+++ b/docs/api/issues.md
@@ -193,6 +193,7 @@ GET /api/issues/{issueId}/interactions
POST /api/issues/{issueId}/interactions
{
"kind": "request_confirmation",
+ "resolverPolicy": "board_only",
"idempotencyKey": "confirmation:{issueId}:plan:{revisionId}",
"title": "Plan approval",
"summary": "Waiting for the board/user to accept or request changes.",
@@ -223,6 +224,12 @@ Supported `kind` values:
- `suggest_tasks`: propose child issues for the board/user to accept or reject
- `ask_user_questions`: ask structured questions and store selected answers
- `request_confirmation`: ask the board/user to accept or reject a proposal
+- `request_checkbox_confirmation`: ask for one accept/reject decision over selected option ids
+- `request_item_verdicts`: collect approve/reject/defer verdicts per item
+
+`resolverPolicy: "board_only" | "board_or_agents"`. Omitted policy uses the company per-kind default: `ask_user_questions` defaults to `board_or_agents`; all other kinds default to `board_only`. `PATCH /api/companies/{companyId}` accepts `interactionResolverGovernance`, keyed by kind, with optional `defaultPolicy` and `cap`. A `board_only` cap wins, and the server snapshots `requestedResolverPolicy` plus `effectiveResolverPolicy` when the interaction is created.
+
+`addresseeAgentId` optionally targets a same-company agent. The addressee is woken with `interaction_pending`, and only that agent or a board user may resolve the card; the creator cannot address itself, tool-action confirmations with an addressee return `400`, and all low-trust/watchdog/same-run restrictions remain. Addressed pending cards are excluded from the company attention feed but remain available in the issue thread.
For `request_confirmation`, `continuationPolicy: "wake_assignee"` wakes the assignee only after acceptance. Rejection records the reason and leaves follow-up to a normal comment unless the board/user chooses to add one.
@@ -232,9 +239,13 @@ For `request_confirmation`, `continuationPolicy: "wake_assignee"` wakes the assi
POST /api/issues/{issueId}/interactions/{interactionId}/accept
POST /api/issues/{issueId}/interactions/{interactionId}/reject
POST /api/issues/{issueId}/interactions/{interactionId}/respond
+POST /api/issues/{issueId}/interactions/{interactionId}/verdicts
+POST /api/issues/{issueId}/interactions/{interactionId}/withdraw
```
-Board users resolve interactions from the UI. Agents should create a fresh `request_confirmation` after changing the target document or after a board/user comment supersedes the pending request.
+Board users can resolve all interactions. Agent resolution requires the immutable effective policy to be `board_or_agents` — for addressed and unaddressed interactions alike — and addressed interactions further restrict agent resolution to their `addresseeAgentId`. Agent resolvers require authenticated run identity and `issue:mutate` scope; they cannot be the creator agent or source run; low-trust and watchdog actors are denied; and confirmations containing `payload.toolAction` are always board-only. Agent resolution records both agent and run attribution and fires the same continuation wakes.
+
+The creator agent or a board user may withdraw a pending interaction. Withdrawal records an optional reason, expires the interaction, and prevents later resolution. Low-trust and task-watchdog agent runs cannot withdraw interactions.
## Documents
diff --git a/docs/cli/control-plane-commands.md b/docs/cli/control-plane-commands.md
index 406b3190bd78..70adf99ad241 100644
--- a/docs/cli/control-plane-commands.md
+++ b/docs/cli/control-plane-commands.md
@@ -91,7 +91,7 @@ pnpm paperclipai skills import ./skills/my-skill --company-id
pnpm paperclipai skills import owner/repo/path/to/skill --company-id
# Attach desired company skills to an agent after install/import
-pnpm paperclipai skills agent sync --skill github-pr-workflow --company-id
+pnpm paperclipai skills agent sync --skill github-pr-workflow --mode add --company-id
```
## Approval Commands
diff --git a/docs/deploy/database.md b/docs/deploy/database.md
index a454de9b3325..0d4ad5731e9f 100644
--- a/docs/deploy/database.md
+++ b/docs/deploy/database.md
@@ -56,16 +56,14 @@ For production, use a hosted provider like [Supabase](https://supabase.com/).
Use the **direct connection** (port 5432) for migrations and the **pooled connection** (port 6543) for the application.
-If using connection pooling, disable prepared statements:
-
-```ts
-// packages/db/src/client.ts
-export function createDb(url: string) {
- const sql = postgres(url, { prepare: false });
- return drizzlePg(sql, { schema });
-}
+If using connection pooling (transaction mode), disable prepared statements via the environment — no source edits needed:
+
+```sh
+DATABASE_PREPARED_STATEMENTS=false
```
+Related optional client tuning (driver defaults apply when unset): `DATABASE_POOL_MAX`, `DATABASE_IDLE_TIMEOUT_SECONDS`, `DATABASE_CONNECT_TIMEOUT_SECONDS`.
+
## Switching Between Modes
| `DATABASE_URL` | Mode |
diff --git a/packages/adapter-utils/src/acpx-engine/execute.test.ts b/packages/adapter-utils/src/acpx-engine/execute.test.ts
index c222fa0553f5..242ca779957b 100644
--- a/packages/adapter-utils/src/acpx-engine/execute.test.ts
+++ b/packages/adapter-utils/src/acpx-engine/execute.test.ts
@@ -35,7 +35,11 @@ import {
type AcpxEngineExecutorOptions,
} from "./execute.js";
import { runChildProcess } from "../server-utils.js";
-import { SANDBOX_STARTUP_SPAN_ATTRS } from "./startup-timing.js";
+import {
+ getActiveStepContext,
+ runWithRuntimeParent,
+ SANDBOX_STARTUP_SPAN_ATTRS,
+} from "./startup-timing.js";
const tempRoots: string[] = [];
@@ -92,16 +96,7 @@ function createLocalSandboxRunner(
}) => void,
) {
let counter = 0;
- // Synthetic provider-duration accumulators so per-step payload assertions can
- // verify the `providerExecMs`/`providerGetMs` threading end-to-end (the real
- // sandbox runner sources these from the Daytona plugin's result metadata; this
- // double stands in for that with a fixed per-exec cost).
- let providerExecMs = 0;
- let providerGetMs = 0;
return {
- execCount: () => counter,
- providerExecMs: () => providerExecMs,
- providerGetMs: () => providerGetMs,
execute: async (input: {
command: string;
args?: string[];
@@ -113,8 +108,6 @@ function createLocalSandboxRunner(
onSpawn?: (meta: { pid: number; startedAt: string }) => Promise;
}) => {
counter += 1;
- providerExecMs += 600;
- providerGetMs += 15;
onExecute?.(input);
const command = input.command === "bash" ? "/bin/bash" : input.command;
return await runChildProcess(`acpx-sandbox-run-${counter}`, command, input.args ?? [], {
@@ -227,8 +220,15 @@ interface RecordingSpan {
name: string;
attributes: Record;
parent: RecordingSpan | null;
+ // The raw third `startSpan` argument — the parent-context token — before the
+ // recorder resolves it to a parent span. A root span receives `undefined`
+ // here, so a test asserts the trace-root shape from the argument itself.
+ parentContextArg: unknown;
status: { code: number } | null;
ended: boolean;
+ // The count of `end` calls. A guarded `end` closure ends the span at most
+ // once, so a test asserts this count equals one for the run root span.
+ endCalls: number;
setAttribute(key: string, value: string | number | boolean): void;
setStatus(status: { code: number; message?: string }): void;
end(): void;
@@ -256,8 +256,10 @@ function createRecordingStartupTrace() {
name,
attributes: { ...(options?.attributes ?? {}) },
parent,
+ parentContextArg: context,
status: null,
ended: false,
+ endCalls: 0,
setAttribute(key: string, value: string | number | boolean) {
span.attributes[key] = value;
},
@@ -266,6 +268,7 @@ function createRecordingStartupTrace() {
},
end() {
span.ended = true;
+ span.endCalls += 1;
},
};
spans.push(span);
@@ -279,6 +282,19 @@ function createRecordingStartupTrace() {
return { traceContext, spans };
}
+// Mirror the production host-to-sandbox exec seam for one execution. The real
+// seam reads the runtime-parent store with `getActiveStepContext()` and opens a
+// `sandbox.exec` span whose third `startSpan` argument is the stored
+// parent-context token. This test copy issues that same span through the
+// injected recorder, so a test asserts which context a startup-body exec parents
+// to at the point the run issues it. The recorder pushes the span into `spans`.
+function issueSandboxExecFromStore(
+ tracing: NonNullable,
+): void {
+ const activeStep = getActiveStepContext();
+ tracing.tracer.startSpan("sandbox.exec", undefined, activeStep?.parentContext);
+}
+
// The closed span-attribute allowlist for a sandbox-start span. A test asserts
// every recorded attribute key is in this set, so a command, path, id, or
// error-text key can never ride a span. Every key uses the closed
@@ -304,6 +320,21 @@ const ALLOWED_STARTUP_SPAN_ATTRIBUTE_KEYS = new Set([
A.batch,
]);
+// The closed attribute allowlist for the run root span. It carries only a
+// non-reversible run-id hash and its own wall time, so no command, path, id, or
+// error text can ride the `task.run` span.
+const ALLOWED_RUN_SPAN_ATTRIBUTE_KEYS = new Set([
+ "paperclip.task.run.run_id",
+ "paperclip.task.run.wall_ms",
+]);
+
+// The closed attribute allowlist for the agent turn span. It carries only its
+// own wall time, so no command, path, id, prompt, or error text can ride the
+// `agent.turn` span.
+const ALLOWED_TURN_SPAN_ATTRIBUTE_KEYS = new Set([
+ "paperclip.agent.turn.wall_ms",
+]);
+
describe("shared ACPX engine runtime behavior", () => {
it("persists ACP agent process identity before prompting and reuses it for the next warm heartbeat", async () => {
const root = await makeTempRoot();
@@ -3219,35 +3250,393 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
{ authToken: "real-run-jwt", executionTarget, startupTraceContext: traceContext },
);
- // Exactly one root span, and it is the bring-up root.
+ // Exactly one trace root, and it is the run root span.
const roots = spans.filter((span) => span.parent === null);
expect(roots).toHaveLength(1);
- const rootSpan = roots[0]!;
- expect(rootSpan.name).toBe("sandbox.startup");
- expect(rootSpan.ended).toBe(true);
+ const runRootSpan = roots[0]!;
+ expect(runRootSpan.name).toBe("task.run");
+ expect(runRootSpan.ended).toBe(true);
+
+ // The sandbox bring-up span parents to the run root span.
+ const startupSpan = spans.find((span) => span.name === "sandbox.startup");
+ expect(startupSpan).toBeTruthy();
+ expect(startupSpan!.parent).toBe(runRootSpan);
+ expect(startupSpan!.ended).toBe(true);
+
+ // The agent turn span is a sibling of the bring-up span: it parents to the
+ // run root span, not to the bring-up span.
+ const turnSpan = spans.find((span) => span.name === "agent.turn");
+ expect(turnSpan).toBeTruthy();
+ expect(turnSpan!.parent).toBe(runRootSpan);
+ expect(turnSpan!.ended).toBe(true);
// A codex bring-up over the remote sandbox lane crosses all 7 boundaries.
- const childNames = spans.filter((span) => span !== rootSpan).map((span) => span.name).sort();
+ // Each boundary span parents to the sandbox bring-up span, not to the run
+ // root or the turn span. The `stage.sync` step also opens one host `pack`
+ // span around the workspace tarball build, so it nests one level deeper.
+ const childNames = spans
+ .filter((span) => span !== runRootSpan && span !== startupSpan && span !== turnSpan)
+ .map((span) => span.name)
+ .sort();
expect(childNames).toEqual(
[
"acp.handshake",
"bridge.paperclip",
"bridge.process-session",
"codex-home.seed",
+ "pack",
"skills.reconcile",
"stage.sync",
"workspace.resolve",
],
);
- // Every child parents to the one root and ends.
+ // The host `pack` span nests under the `stage.sync` step span (the host
+ // tarball build runs inside that step), not directly under the bring-up
+ // span.
+ const stageSyncSpan = spans.find((span) => span.name === "stage.sync");
+ const packSpan = spans.find((span) => span.name === "pack");
+ expect(stageSyncSpan).toBeTruthy();
+ expect(packSpan).toBeTruthy();
+ expect(packSpan!.parent).toBe(stageSyncSpan);
+ expect(packSpan!.ended).toBe(true);
+
+ // Every boundary step span parents to the sandbox bring-up span and ends.
+ // The `pack` span is the one exception: it parents to `stage.sync` above.
for (const span of spans) {
- if (span === rootSpan) continue;
- expect(span.parent, `span "${span.name}" must parent to the root`).toBe(rootSpan);
+ if (span === runRootSpan || span === startupSpan || span === turnSpan) continue;
+ if (span === packSpan) continue;
+ expect(span.parent, `span "${span.name}" must parent to the startup span`).toBe(startupSpan);
expect(span.ended, `span "${span.name}" must end`).toBe(true);
}
});
+ it("test_task_run_span_opens_and_ends_once_for_remote", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Run the executor with a fake remote-sandbox tracer. The engine opens one
+ // `task.run` root span and ends it once at the run return.
+ await runExecutor(
+ {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ { authToken: "real-run-jwt", executionTarget, startupTraceContext: traceContext },
+ );
+
+ // Exactly one `task.run` span opens for a remote-sandbox run.
+ const runSpans = spans.filter((span) => span.name === "task.run");
+ expect(runSpans).toHaveLength(1);
+ const runSpan = runSpans[0]!;
+ // It is the trace root: it opens with no parent-context token.
+ expect(runSpan.parent).toBeNull();
+ expect(runSpan.parentContextArg).toBeUndefined();
+ // It ends exactly once. A clean run sets no error status.
+ expect(runSpan.ended).toBe(true);
+ expect(runSpan.endCalls).toBe(1);
+ expect(runSpan.status).toBeNull();
+ // The run id rides only as a non-reversible hash; the raw run id never rides
+ // the span.
+ expect(runSpan.attributes["paperclip.task.run.run_id"]).toMatch(/^[0-9a-f]{12}$/);
+ expect(String(runSpan.attributes["paperclip.task.run.run_id"])).not.toContain("run-");
+ });
+
+ it("test_sandbox_startup_parents_to_task_run", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Run the executor startup path with a fake remote-sandbox tracer. The run
+ // root opens the `task.run` span first, then the startup root opens the
+ // `sandbox.startup` span as its child.
+ await runExecutor(
+ {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ { authToken: "real-run-jwt", executionTarget, startupTraceContext: traceContext },
+ );
+
+ const runSpan = spans.find((span) => span.name === "task.run");
+ expect(runSpan).toBeTruthy();
+ const startupSpan = spans.find((span) => span.name === "sandbox.startup");
+ expect(startupSpan).toBeTruthy();
+ // The `sandbox.startup` `startSpan` call now receives the `task.run` parent
+ // context as its third argument. The recorder builds the token as
+ // `{ span }`, so the token carries the run span and the startup span parents
+ // to `task.run`.
+ expect(startupSpan!.parentContextArg).toEqual({ span: runSpan });
+ expect(startupSpan!.parent).toBe(runSpan);
+ });
+
+ it("test_task_run_span_is_noop_for_local_target", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ await fs.mkdir(localCwd, { recursive: true });
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // A local run has no sandbox, so the run root span stays a no-op even when a
+ // trace context is injected. No real `task.run` span opens.
+ const { result } = await runExecutor(
+ { agent: "custom", agentCommand: "node ./fake-acp.js", stateDir, cwd: localCwd },
+ { authToken: "real-run-jwt", startupTraceContext: traceContext },
+ );
+ expect(result.exitCode).toBe(0);
+ expect(spans.some((span) => span.name === "task.run")).toBe(false);
+ expect(spans).toHaveLength(0);
+ });
+
+ it("test_agent_turn_span_parents_to_task_run", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Run the executor turn path with a fake remote-sandbox tracer. The engine
+ // opens one `agent.turn` span around the turn as a child of `task.run` and
+ // ends it once at the turn return.
+ await runExecutor(
+ {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ { authToken: "real-run-jwt", executionTarget, startupTraceContext: traceContext },
+ );
+
+ const runSpan = spans.find((span) => span.name === "task.run");
+ expect(runSpan).toBeTruthy();
+ // Exactly one `agent.turn` span opens for a remote-sandbox run.
+ const turnSpans = spans.filter((span) => span.name === "agent.turn");
+ expect(turnSpans).toHaveLength(1);
+ const turnSpan = turnSpans[0]!;
+ // The `agent.turn` `startSpan` call receives the `task.run` parent context
+ // as its third argument. The recorder builds the token as `{ span }`, so the
+ // turn span parents to `task.run`.
+ expect(turnSpan.parentContextArg).toEqual({ span: runSpan });
+ expect(turnSpan.parent).toBe(runSpan);
+ // It ends exactly once. A clean turn sets no error status.
+ expect(turnSpan.ended).toBe(true);
+ expect(turnSpan.endCalls).toBe(1);
+ expect(turnSpan.status).toBeNull();
+ });
+
+ it("test_agent_turn_span_is_noop_for_local_target", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ await fs.mkdir(localCwd, { recursive: true });
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // A local run has no sandbox, so the turn span stays a no-op even when a
+ // trace context is injected. No real `agent.turn` span opens.
+ const { result } = await runExecutor(
+ { agent: "custom", agentCommand: "node ./fake-acp.js", stateDir, cwd: localCwd },
+ { authToken: "real-run-jwt", startupTraceContext: traceContext },
+ );
+ expect(result.exitCode).toBe(0);
+ expect(spans.some((span) => span.name === "agent.turn")).toBe(false);
+ });
+
+ it("test_run_parent_getter_tracks_task_run_then_turn", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Sample `getRuntimeParentContext` at three points of the run: at runtime
+ // construction (startup), inside `startTurn` (turn), and after the executor
+ // resolves (post-turn). The run publishes the current-run parent token
+ // through `runtimeOptions.getRuntimeParentContext`, so a fake runtime reads
+ // it to prove the holder tracks `task.run`, then `agent.turn`, then
+ // `task.run` again.
+ let startupToken: unknown;
+ let turnToken: unknown;
+ let capturedGetter: (() => unknown) | undefined;
+ const execute = createAcpxEngineExecutor({
+ createRuntime: (options) => {
+ const opts = options as unknown as { getRuntimeParentContext?: () => unknown };
+ // Keep the getter for the post-turn sample after the run resolves.
+ capturedGetter = opts.getRuntimeParentContext;
+ // Startup phase: the holder is the `task.run` token here.
+ startupToken = opts.getRuntimeParentContext?.();
+ const runtime = buildRuntime();
+ return {
+ ...runtime,
+ startTurn: (input: Record) => {
+ // Turn phase: the holder is the `agent.turn` token here.
+ turnToken = opts.getRuntimeParentContext?.();
+ return (runtime.startTurn as (input: unknown) => unknown)(input);
+ },
+ } as never;
+ },
+ });
+
+ const result = await execute({
+ runId: "run-getter",
+ agent: { id: "agent-1", companyId: "company-1" },
+ runtime: {},
+ config: {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ context: {},
+ authToken: "real-run-jwt",
+ executionTarget,
+ startupTraceContext: traceContext,
+ onLog: async () => {},
+ onMeta: async () => {},
+ onEvent: async () => {},
+ } as never);
+ expect(result.exitCode).toBe(0);
+
+ // Post-turn phase: the holder is the `task.run` token again.
+ const postToken = capturedGetter?.();
+
+ const taskRunSpan = spans.find((span) => span.name === "task.run");
+ expect(taskRunSpan).toBeTruthy();
+ const agentTurnSpan = spans.find((span) => span.name === "agent.turn");
+ expect(agentTurnSpan).toBeTruthy();
+
+ // The recorder builds a parent-context token as `{ span }`. So the getter
+ // returns the `task.run` token during startup, the `agent.turn` token during
+ // the turn, and the `task.run` token after the turn.
+ expect(startupToken).toEqual({ span: taskRunSpan });
+ expect(turnToken).toEqual({ span: agentTurnSpan });
+ expect(postToken).toEqual({ span: taskRunSpan });
+ });
+
+ it("test_root_region_exec_parents_to_sandbox_startup", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Issue one exec from the root region of the bring-up. `getServers()` runs
+ // once inside the bring-up, after the `sandbox.startup` span opens and
+ // outside every measured step, so it stands in for a startup-body exec that
+ // runs at the root region. The run publishes the `sandbox.startup` context to
+ // the runtime-parent store for the whole bring-up, so the exec reads that
+ // token here.
+ const runtimeMcp = {
+ getServers: () => {
+ issueSandboxExecFromStore(traceContext);
+ return [];
+ },
+ };
+
+ await runExecutor(
+ {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ {
+ authToken: "real-run-jwt",
+ executionTarget,
+ startupTraceContext: traceContext,
+ runtimeMcp: runtimeMcp as never,
+ },
+ );
+
+ const startupSpan = spans.find((span) => span.name === "sandbox.startup");
+ expect(startupSpan).toBeTruthy();
+ const execSpan = spans.find((span) => span.name === "sandbox.exec");
+ expect(execSpan).toBeTruthy();
+ // The root-region exec parents to `sandbox.startup`, not to a detached root.
+ // The third `startSpan` argument is the exact `sandbox.startup` child context
+ // that `contextWithSpan` built, so the exec span is a child of the bring-up.
+ expect(execSpan!.parentContextArg).toEqual({ span: startupSpan });
+ expect(execSpan!.parent).toBe(startupSpan);
+ });
+
+ it("test_in_step_exec_still_parents_to_step_after_root_wrap", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Issue one exec from inside the `stage.sync` step body, under the same root
+ // wrap. `prepareRemoteManagedHome` runs inside the measured `stage.sync`
+ // step, so the step overrides the runtime-parent store with its own context
+ // there. The in-step exec must read the step context, not the root context.
+ await runExecutor(
+ {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ {
+ authToken: "real-run-jwt",
+ executionTarget,
+ startupTraceContext: traceContext,
+ prepareRemoteManagedHome: async (input) => {
+ issueSandboxExecFromStore(traceContext);
+ const stagedRuntime = await input.stage([]);
+ return { stagedRuntime };
+ },
+ },
+ );
+
+ const stepSpan = spans.find((span) => span.name === "stage.sync");
+ expect(stepSpan).toBeTruthy();
+ const startupSpan = spans.find((span) => span.name === "sandbox.startup");
+ expect(startupSpan).toBeTruthy();
+ const execSpan = spans.find((span) => span.name === "sandbox.exec");
+ expect(execSpan).toBeTruthy();
+ // The in-step exec still parents to its step span. `measureStartupStep`
+ // overrides the store inside the wrap, so the step context wins over the
+ // root context for an exec that runs inside the step.
+ expect(execSpan!.parentContextArg).toEqual({ span: stepSpan });
+ expect(execSpan!.parent).toBe(stepSpan);
+ // The in-step exec does not parent to `sandbox.startup`.
+ expect(execSpan!.parent).not.toBe(startupSpan);
+ });
+
it("records root wall / work / diff times and the bounded context on the root span", async () => {
const root = await makeTempRoot();
const stateDir = path.join(root, "state");
@@ -3272,7 +3661,7 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
{ authToken: "real-run-jwt", executionTarget, startupTraceContext: traceContext },
);
- const rootSpan = spans.find((span) => span.name === "sandbox.startup" && span.parent === null);
+ const rootSpan = spans.find((span) => span.name === "sandbox.startup");
expect(rootSpan).toBeTruthy();
// The three timing numbers are present, finite, and non-negative.
for (const key of [A.rootWallMs, A.rootWorkMs, A.rootDiffMs]) {
@@ -3328,7 +3717,7 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
.filter((event) => event.payload?.step !== "skills.reconcile")
.reduce((total, event) => total + (event.payload?.durationMs as number), 0);
- const rootSpan = spans.find((span) => span.name === "sandbox.startup" && span.parent === null);
+ const rootSpan = spans.find((span) => span.name === "sandbox.startup");
expect(rootSpan).toBeTruthy();
expect(rootSpan!.attributes[A.rootWorkMs]).toBe(sumExceptReconcile);
});
@@ -3346,7 +3735,7 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
{ authToken: "real-run-jwt", executionTarget, startupTraceContext: traceContext },
);
- const rootSpan = spans.find((span) => span.name === "sandbox.startup" && span.parent === null);
+ const rootSpan = spans.find((span) => span.name === "sandbox.startup");
expect(rootSpan).toBeTruthy();
const paperclip = spans.find((span) => span.name === "bridge.paperclip");
const processSession = spans.find((span) => span.name === "bridge.process-session");
@@ -3396,9 +3785,17 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
expect(spans.length).toBeGreaterThan(0);
for (const span of spans) {
+ // The run root span and the agent turn span each use their own closed
+ // allowlist; every other span uses the sandbox startup allowlist.
+ const allowed =
+ span.name === "task.run"
+ ? ALLOWED_RUN_SPAN_ATTRIBUTE_KEYS
+ : span.name === "agent.turn"
+ ? ALLOWED_TURN_SPAN_ATTRIBUTE_KEYS
+ : ALLOWED_STARTUP_SPAN_ATTRIBUTE_KEYS;
for (const [key, value] of Object.entries(span.attributes)) {
expect(
- ALLOWED_STARTUP_SPAN_ATTRIBUTE_KEYS.has(key),
+ allowed.has(key),
`span "${span.name}" set a non-allowlisted attribute "${key}"`,
).toBe(true);
// No non-finite numeric attribute (no NaN, no Infinity).
@@ -3448,7 +3845,7 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
} as never);
expect(result.exitCode).toBe(1);
- const rootSpan = spans.find((span) => span.name === "sandbox.startup" && span.parent === null);
+ const rootSpan = spans.find((span) => span.name === "sandbox.startup");
expect(rootSpan).toBeTruthy();
expect(rootSpan!.ended).toBe(true);
// `2` is `SpanStatusCode.ERROR`.
@@ -3509,6 +3906,261 @@ describe("ACPX engine sandbox-start spans (opt-in root + child parenting)", () =
expect(result.exitCode).toBe(0);
expect(spans).toHaveLength(0);
});
+
+ it("test_full_trace_tree_parents_correctly", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Drive one remote-sandbox run and issue an exec at four points of the run.
+ // Each exec issues through the same host-to-sandbox seam that the run uses,
+ // so the recorder captures the exact parent of each exec span. Each point
+ // fires exactly once, so the run records exactly four `sandbox.exec` spans.
+ let rootRegionExecFired = false;
+ let syncExecFired = false;
+ let turnExecFired = false;
+
+ // A run-scoped getter holder. The engine publishes the current-run parent
+ // token through `runtimeOptions.getRuntimeParentContext`. A detached site
+ // reads it per unit of work and wraps the work in `runWithRuntimeParent`.
+ let getRuntimeParentContext: (() => unknown) | undefined;
+
+ const execute = createAcpxEngineExecutor({
+ prepareRemoteManagedHome: async (input) => {
+ // Point 2: an exec inside the measured `stage.sync` step body. The step
+ // overrides the runtime-parent store, so this exec parents to the sync
+ // step span, not to `sandbox.startup`.
+ if (!syncExecFired) {
+ syncExecFired = true;
+ issueSandboxExecFromStore(traceContext);
+ }
+ const stagedRuntime = await input.stage([]);
+ return { stagedRuntime };
+ },
+ createRuntime: (options) => {
+ const opts = options as unknown as {
+ getRuntimeParentContext?: () => unknown;
+ };
+ getRuntimeParentContext = opts.getRuntimeParentContext;
+ const runtime = buildRuntime();
+ return {
+ ...runtime,
+ startTurn: (input: Record) => {
+ // Point 3: a detached exec during the turn. The detached site reads
+ // the getter (the `agent.turn` token here) and wraps the work in
+ // `runWithRuntimeParent`, so the exec parents to `agent.turn`.
+ if (!turnExecFired) {
+ turnExecFired = true;
+ runWithRuntimeParent(opts.getRuntimeParentContext?.(), () =>
+ issueSandboxExecFromStore(traceContext),
+ );
+ }
+ return (runtime.startTurn as (input: unknown) => unknown)(input);
+ },
+ } as never;
+ },
+ });
+
+ const runtimeMcp = {
+ getServers: () => {
+ // Point 1: a root-region exec. `getServers` runs inside the bring-up,
+ // after `sandbox.startup` opens and outside every measured step, so the
+ // store holds the `sandbox.startup` context and this exec parents to it.
+ if (!rootRegionExecFired) {
+ rootRegionExecFired = true;
+ issueSandboxExecFromStore(traceContext);
+ }
+ return [];
+ },
+ };
+
+ const result = await execute({
+ runId: "run-e2e-tree",
+ agent: { id: "agent-1", companyId: "company-1" },
+ runtime: {},
+ config: {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ context: {},
+ authToken: "real-run-jwt",
+ executionTarget,
+ runtimeMcp: runtimeMcp as never,
+ startupTraceContext: traceContext,
+ onLog: async () => {},
+ onMeta: async () => {},
+ onEvent: async () => {},
+ } as never);
+ expect(result.exitCode).toBe(0);
+
+ // Point 4: a detached exec after the turn. The getter returns the `task.run`
+ // token now, so an off-turn detached exec parents to `task.run`.
+ runWithRuntimeParent(getRuntimeParentContext?.(), () =>
+ issueSandboxExecFromStore(traceContext),
+ );
+
+ // The four framing spans of the run.
+ const runSpan = spans.find((span) => span.name === "task.run");
+ expect(runSpan).toBeTruthy();
+ const startupSpan = spans.find((span) => span.name === "sandbox.startup");
+ expect(startupSpan).toBeTruthy();
+ const turnSpan = spans.find((span) => span.name === "agent.turn");
+ expect(turnSpan).toBeTruthy();
+ const syncStepSpan = spans.find((span) => span.name === "stage.sync");
+ expect(syncStepSpan).toBeTruthy();
+
+ // `task.run` is the trace root. `sandbox.startup` and `agent.turn` are its
+ // direct children.
+ expect(runSpan!.parent).toBeNull();
+ expect(startupSpan!.parent).toBe(runSpan);
+ expect(turnSpan!.parent).toBe(runSpan);
+
+ // The run records exactly the four injected exec spans.
+ const execSpans = spans.filter((span) => span.name === "sandbox.exec");
+ expect(execSpans).toHaveLength(4);
+
+ // A root-region exec parents to `sandbox.startup`.
+ const rootRegionExec = execSpans.find((span) => span.parent === startupSpan);
+ expect(rootRegionExec, "a root-region exec must parent to sandbox.startup").toBeTruthy();
+
+ // A `stage.sync` body exec parents to the sync step, not to `sandbox.startup`.
+ const syncExec = execSpans.find((span) => span.parent === syncStepSpan);
+ expect(syncExec, "a stage.sync exec must parent to the sync step").toBeTruthy();
+ expect(syncExec!.parent).not.toBe(startupSpan);
+
+ // A turn detached exec parents to `agent.turn`.
+ const turnExec = execSpans.find((span) => span.parent === turnSpan);
+ expect(turnExec, "a turn detached exec must parent to agent.turn").toBeTruthy();
+
+ // An off-turn detached exec parents to `task.run`.
+ const offTurnExec = execSpans.find((span) => span.parent === runSpan);
+ expect(offTurnExec, "an off-turn detached exec must parent to task.run").toBeTruthy();
+
+ // The four execs are distinct spans with four distinct parents.
+ const execParents = new Set([rootRegionExec, syncExec, turnExec, offTurnExec]);
+ expect(execParents.size).toBe(4);
+ });
+
+ it("test_no_sandbox_exec_parents_to_http_root", async () => {
+ const root = await makeTempRoot();
+ const stateDir = path.join(root, "state");
+ const localCwd = path.join(root, "worktree");
+ const codexHome = path.join(root, "codex-home");
+ await fs.mkdir(localCwd, { recursive: true });
+ await fs.mkdir(codexHome, { recursive: true });
+ const executionTarget = await remoteSandboxTarget(root);
+ const { traceContext, spans } = createRecordingStartupTrace();
+
+ // Open a top-level HTTP request span before the run. In production the run
+ // opens under the HTTP request span, and the buggy exec parented to it. The
+ // recorder resolves an explicit parent token to its `span`, and it resolves
+ // an `undefined` token to `null`. A `sandbox.exec` issued with an `undefined`
+ // token is exactly the exec that the real tracer would attach to the ambient
+ // HTTP request span. So the two negatives are: no exec names this request
+ // span, and no exec opens with an `undefined` token (the ambient root).
+ const httpRequestSpan = traceContext.tracer.startSpan("http.request", undefined, undefined);
+
+ let rootRegionExecFired = false;
+ let syncExecFired = false;
+ let turnExecFired = false;
+ let getRuntimeParentContext: (() => unknown) | undefined;
+
+ const execute = createAcpxEngineExecutor({
+ prepareRemoteManagedHome: async (input) => {
+ if (!syncExecFired) {
+ syncExecFired = true;
+ issueSandboxExecFromStore(traceContext);
+ }
+ const stagedRuntime = await input.stage([]);
+ return { stagedRuntime };
+ },
+ createRuntime: (options) => {
+ const opts = options as unknown as {
+ getRuntimeParentContext?: () => unknown;
+ };
+ getRuntimeParentContext = opts.getRuntimeParentContext;
+ const runtime = buildRuntime();
+ return {
+ ...runtime,
+ startTurn: (input: Record) => {
+ // The detached site reads the getter and wraps the work. The getter
+ // never returns `undefined` inside a live run, so this exec never
+ // detaches to the HTTP request span.
+ if (!turnExecFired) {
+ turnExecFired = true;
+ runWithRuntimeParent(opts.getRuntimeParentContext?.(), () =>
+ issueSandboxExecFromStore(traceContext),
+ );
+ }
+ return (runtime.startTurn as (input: unknown) => unknown)(input);
+ },
+ } as never;
+ },
+ });
+
+ const runtimeMcp = {
+ getServers: () => {
+ if (!rootRegionExecFired) {
+ rootRegionExecFired = true;
+ issueSandboxExecFromStore(traceContext);
+ }
+ return [];
+ },
+ };
+
+ const result = await execute({
+ runId: "run-e2e-http",
+ agent: { id: "agent-1", companyId: "company-1" },
+ runtime: {},
+ config: {
+ agent: "codex",
+ agentCommand: "node ./fake-acp.js",
+ stateDir,
+ cwd: localCwd,
+ env: { CODEX_HOME: codexHome },
+ },
+ context: {},
+ authToken: "real-run-jwt",
+ executionTarget,
+ runtimeMcp: runtimeMcp as never,
+ startupTraceContext: traceContext,
+ onLog: async () => {},
+ onMeta: async () => {},
+ onEvent: async () => {},
+ } as never);
+ expect(result.exitCode).toBe(0);
+
+ // The off-turn detached exec also reads the getter, so it also stays under a
+ // run span.
+ runWithRuntimeParent(getRuntimeParentContext?.(), () =>
+ issueSandboxExecFromStore(traceContext),
+ );
+
+ const execSpans = spans.filter((span) => span.name === "sandbox.exec");
+ // The run issues four exec spans, so there is something to scan.
+ expect(execSpans.length).toBeGreaterThan(0);
+
+ for (const span of execSpans) {
+ // Negative 1: no exec parents to the top-level HTTP request span.
+ expect(span.parent, "a sandbox.exec must not parent to the HTTP request span").not.toBe(
+ httpRequestSpan,
+ );
+ // Negative 2: no exec opens unparented inside the run. An `undefined`
+ // parent token is the ambient root that would attach to the HTTP request
+ // span in production, and a `null` parent is a detached span.
+ expect(span.parentContextArg, "a sandbox.exec must open with an explicit parent token")
+ .not.toBeUndefined();
+ expect(span.parent, "a sandbox.exec must not open unparented inside the run").not.toBeNull();
+ }
+ });
});
describe("ACPX engine per-step startup timing (run.startup.step events)", () => {
@@ -3605,7 +4257,7 @@ describe("ACPX engine per-step startup timing (run.startup.step events)", () =>
}
});
- it("carries roundTrips + provider durations for sequential startup steps and keeps concurrent bridge steps duration-only", async () => {
+ it("emits only the high-level fields on every startup-step payload, never the detailed timing or counts", async () => {
const root = await makeTempRoot();
const stateDir = path.join(root, "state");
const localCwd = path.join(root, "worktree");
@@ -3626,65 +4278,20 @@ describe("ACPX engine per-step startup timing (run.startup.step events)", () =>
);
const steps = stepEvents(events);
- const seen = new Map(steps.map((event) => [String(event.payload?.step), event]));
+ expect(steps.length).toBeGreaterThan(0);
- // Every timed boundary still records duration.
+ // Every timed boundary still records the high-level duration and outcome.
+ // The detailed per-step round-trip and provider-duration numbers ride the
+ // OTel spans now, so no payload carries them.
for (const event of steps) {
- expect(typeof event.payload?.durationMs).toBe("number");
- }
- // Sequential boundaries retain runner-counter attribution.
- for (const step of ["workspace.resolve", "stage.sync", "acp.handshake"]) {
- expect(typeof seen.get(step)?.payload?.roundTrips).toBe("number");
- }
- // workspace.resolve is host-only → zero host→sandbox execs.
- expect(seen.get("workspace.resolve")?.payload?.roundTrips).toBe(0);
- // stage.sync ships the workspace over the exec seam → at least one round-trip,
- // and the accumulated provider durations scale with it.
- const stageSync = seen.get("stage.sync");
- expect(stageSync?.payload?.roundTrips as number).toBeGreaterThan(0);
- expect(stageSync?.payload?.providerExecMs).toBe(
- (stageSync?.payload?.roundTrips as number) * 600,
- );
- expect(stageSync?.payload?.providerGetMs).toBe(
- (stageSync?.payload?.roundTrips as number) * 15,
- );
- // Concurrent bridge steps are duration-only so they do not double-count
- // shared runner counters while their lifecycles overlap.
- for (const step of ["bridge.paperclip", "bridge.process-session"]) {
- expect(seen.get(step)?.payload?.roundTrips).toBeUndefined();
- expect(seen.get(step)?.payload?.providerExecMs).toBeUndefined();
- expect(seen.get(step)?.payload?.providerGetMs).toBeUndefined();
+ const payload = event.payload ?? {};
+ expect(typeof payload.durationMs).toBe("number");
+ expect(payload).not.toHaveProperty("roundTrips");
+ expect(payload).not.toHaveProperty("providerExecMs");
+ expect(payload).not.toHaveProperty("providerGetMs");
+ expect(payload).not.toHaveProperty("createRuntimeMs");
+ expect(payload).not.toHaveProperty("ensureSessionMs");
}
- // The external ACP client crosses no host exec seam.
- expect(seen.get("acp.handshake")?.payload?.roundTrips).toBe(0);
- });
-
- it("splits acp.handshake into createRuntimeMs and ensureSessionMs sub-phases", async () => {
- const root = await makeTempRoot();
- const stateDir = path.join(root, "state");
- const localCwd = path.join(root, "worktree");
- const remoteCwd = path.join(root, "remote-workspace");
- await fs.mkdir(localCwd, { recursive: true });
- await fs.mkdir(remoteCwd, { recursive: true });
- const executionTarget = {
- kind: "remote",
- transport: "sandbox",
- providerKey: "fake-plugin",
- remoteCwd,
- runner: createLocalSandboxRunner(),
- };
-
- const { events } = await runExecutor(
- { agent: "custom", agentCommand: "node ./fake-acp.js", stateDir, cwd: localCwd },
- { authToken: "real-run-jwt", executionTarget },
- );
-
- const handshake = stepEvents(events).find((event) => event.payload?.step === "acp.handshake");
- expect(handshake).toBeTruthy();
- expect(typeof handshake!.payload?.createRuntimeMs).toBe("number");
- expect(handshake!.payload?.createRuntimeMs as number).toBeGreaterThanOrEqual(0);
- expect(typeof handshake!.payload?.ensureSessionMs).toBe("number");
- expect(handshake!.payload?.ensureSessionMs as number).toBeGreaterThanOrEqual(0);
});
it("emits a skipped acp.handshake event when a warm-handle hit skips the handshake", async () => {
diff --git a/packages/adapter-utils/src/acpx-engine/execute.ts b/packages/adapter-utils/src/acpx-engine/execute.ts
index 786a43d10bac..3dc9fa7260bc 100644
--- a/packages/adapter-utils/src/acpx-engine/execute.ts
+++ b/packages/adapter-utils/src/acpx-engine/execute.ts
@@ -84,18 +84,21 @@ import {
DEFAULT_ACP_ENGINE_WARM_HANDLE_IDLE_MS,
} from "./constants.js";
import {
+ createRuntimeSpanRunner,
emitSkippedStartupStep,
+ getActiveStepContext,
measureStartupStep,
NOOP_STARTUP_SPAN,
NOOP_STARTUP_TRACE_CONTEXT,
+ runWithRuntimeParent,
setSandboxRootSpanAttributes,
+ type RuntimeSpanRunner,
type SandboxRootSpanContext,
type StartupSpan,
type StartupSpanContext,
type StartupStepMeasureOptions,
type StartupTraceContext,
} from "./startup-timing.js";
-import type { CommandManagedRuntimeRunner } from "../command-managed-runtime.js";
const defaultModuleDir = path.dirname(fileURLToPath(import.meta.url));
const PAPERCLIP_MANAGED_CODEX_SKILLS_MANIFEST = ".paperclip-managed-skills.json";
@@ -137,6 +140,11 @@ type AcpxAgentProcessIdentity = { pid: number; startedAt: string };
type PaperclipAcpRuntimeOptions = AcpRuntimeOptions & {
onAgentSpawn?: (meta: AcpxAgentProcessIdentity) => Promise;
+ // Return the current-run parent-context token. It is the `task.run` token
+ // during startup and after the turn, and the `agent.turn` token during the
+ // turn. A detached exec reads this getter to parent to the live run span. The
+ // real `createAcpRuntime` ignores this optional field.
+ getRuntimeParentContext?: () => StartupSpanContext | undefined;
};
type AcpxProcessIdentitySink = {
@@ -1325,6 +1333,11 @@ async function stageAcpRemoteRuntime(input: {
additionalSources?: SandboxAdditionalSource[];
onLog: AdapterExecutionContext["onLog"];
onRuntimeProgress: AdapterExecutionContext["onRuntimeProgress"];
+ // Optional host span runner for the workspace tarball build. It rides down to
+ // prepareSandboxManagedRuntime so the host pack time shows as one `pack` span.
+ // The caller passes a runner that parents to the active `stage.sync` step, so
+ // the `pack` span nests under `stage.sync`. The default is a no-op.
+ runtimeSpan?: RuntimeSpanRunner;
}): Promise {
await input.onLog(
"stdout",
@@ -1343,25 +1356,10 @@ async function stageAcpRemoteRuntime(input: {
: {}),
onProgress: (line) => input.onLog("stdout", line),
onRuntimeProgress: input.onRuntimeProgress,
+ runtimeSpan: input.runtimeSpan,
});
}
-// Bind a startup-step round-trip/provider-duration reader set to a runner's
-// cumulative counters (Open Q1). Only the sandbox runner instruments the exec
-// seam, so a runner without `execCount` (SSH, or none) yields an empty option
-// set and the affected steps omit the fields entirely. Reader closures are
-// passed — not the runner — so `measureStartupStep` stays runner-agnostic.
-function buildStartupStepMetrics(
- runner: CommandManagedRuntimeRunner | undefined,
-): StartupStepMeasureOptions {
- if (!runner) return {};
- return {
- ...(runner.execCount ? { roundTrips: () => runner.execCount!() } : {}),
- ...(runner.providerExecMs ? { providerExecMs: () => runner.providerExecMs!() } : {}),
- ...(runner.providerGetMs ? { providerGetMs: () => runner.providerGetMs!() } : {}),
- };
-}
-
async function buildRuntime(input: {
ctx: AdapterExecutionContext;
engine: AcpxEngineSettings;
@@ -1372,6 +1370,24 @@ async function buildRuntime(input: {
// executor opens, and each step publishes its own child context for an inner
// exec span to parent to.
spanParent: Pick;
+ // Return the current-run parent-context token. `buildRuntime` threads it into
+ // the two remote bridge factories, so a run-time exec from a bridge parents to
+ // the live run span (`agent.turn` during the turn, `task.run` otherwise). The
+ // run closure passes the run-scoped getter here; when it is absent, each
+ // bridge site keeps its earlier unparented run-time behavior.
+ getRuntimeParentContext?: () => StartupSpanContext | undefined;
+ // Wrap each unit of bridge run-time work in its own named span.
+ // `buildRuntime` threads it into the two remote bridge factories, so the
+ // socket handler, the poll loop, and the callback worker each open a wrapper
+ // span per unit of work. The run closure passes the run-scoped runner here;
+ // when it is absent, each bridge site opens no wrapper span.
+ runtimeSpan?: RuntimeSpanRunner;
+ // Wrap the host workspace tarball build in one `pack` span. Unlike
+ // `runtimeSpan`, this runner parents each span to the active startup step, so
+ // the `pack` span nests under the `stage.sync` step that runs the staging
+ // seam. `buildRuntime` threads it into the staging seam. When it is absent, the
+ // staging seam opens no `pack` span.
+ stageRuntimeSpan?: RuntimeSpanRunner;
}): Promise {
const { runId, agent, config, context, authToken } = input.ctx;
// Injectable monotonic clock for per-step startup timing. Hoisted above the
@@ -1450,28 +1466,13 @@ async function buildRuntime(input: {
? remoteExecutionIdentity.remoteCwd
: cwd;
const executionTargetIsRemote = remoteExecutionIdentity !== null;
- // Round-trip / provider-duration readers for per-step attribution (Open Q1),
- // sourced from the sandbox runner's cumulative counters. `measureStartupStep`
- // reads each as a `() => number` closure (never the runner itself, Risk R1)
- // and emits the per-step delta. Empty when there is no runner (local runs,
- // the runner-less ACP→CLI fallback, or an SSH runner that does not
- // instrument the seam), so those steps simply omit the fields.
// Merge the injected tracer + root parent-context into every step option set,
// so each boundary span parents to the root span. With no injected trace
// context both fields are no-ops and the span path stays inert.
const stepMetrics: StartupStepMeasureOptions = {
- ...buildStartupStepMetrics(
- executionTarget?.kind === "remote" && executionTarget.transport === "sandbox"
- ? executionTarget.runner
- : undefined,
- ),
...input.spanParent,
};
- // The two bridge-start steps intentionally overlap, so their runner counters
- // would double-count each other if we sampled them here. Keep the shared
- // counter attribution on the sequential startup phases only; the concurrent
- // bridge steps still emit duration telemetry (and a span), just not
- // misleading per-step round-trip/provider deltas. A shared `batch` tag marks
+ // The two bridge-start steps intentionally overlap. A shared `batch` tag marks
// the two spans as one parallel batch, and `criticalPath: false` keeps their
// inner exec spans off the critical path (their wall time overlaps).
const concurrentBridgeStepMetrics: StartupStepMeasureOptions = {
@@ -1704,6 +1705,13 @@ async function buildRuntime(input: {
executionTarget.transport === "sandbox" &&
Boolean(executionTarget.runner) &&
Boolean(agentCommandShell);
+ // Stream the agent output through the persistent session log stream instead of
+ // the host output-file poll. Default OFF; an operator opts a sandbox
+ // environment in through the environment config.
+ const streamAgentSessionOutput =
+ executionTarget?.kind === "remote" &&
+ executionTarget.transport === "sandbox" &&
+ executionTarget.streamAgentSessionOutput === true;
// The ACP `session/new` cwd and every cwd-keyed session-state site
// (fingerprint, compat, persist, ensureSession, error) bind to THIS single
// value so a warm/resumable session created with the in-sandbox cwd is reused
@@ -1870,6 +1878,7 @@ async function buildRuntime(input: {
additionalSources,
onLog: input.ctx.onLog,
onRuntimeProgress: input.ctx.onRuntimeProgress,
+ runtimeSpan: input.stageRuntimeSpan,
});
// Snapshot env before the seam so we can capture exactly which keys it
// repointed onto the in-sandbox home (e.g. `CODEX_HOME`) and replay them
@@ -1971,10 +1980,8 @@ async function buildRuntime(input: {
// dir/script setup first, then awaits that thunk right before its launch, so
// the launch always observes the merged paperclip env.
//
- // Measurement caveat: both starts share ONE runner counter, so their
- // overlapping `providerExecMs`/`roundTrips` deltas are approximate (the same
- // caveat as `acp.handshake`). Both `run.startup.step` events still emit —
- // `measureStartupStep` records them in a `finally`, even on a start failure.
+ // Both `run.startup.step` events still emit — `measureStartupStep` records
+ // them in a `finally`, even on a start failure.
const paperclipStart = measureStartupStep(input.ctx, nowMs, "bridge.paperclip", () =>
startAdapterExecutionTargetPaperclipBridge({
runId,
@@ -1984,6 +1991,8 @@ async function buildRuntime(input: {
timeoutSec,
hostApiToken: env.PAPERCLIP_API_KEY,
onLog: input.ctx.onLog,
+ getRuntimeParentContext: input.getRuntimeParentContext,
+ runtimeSpan: input.runtimeSpan,
}),
concurrentBridgeStepMetrics,
);
@@ -2014,6 +2023,9 @@ async function buildRuntime(input: {
env: finalizeLaunchEnv,
timeoutSec,
onLog: input.ctx.onLog,
+ getRuntimeParentContext: input.getRuntimeParentContext,
+ runtimeSpan: input.runtimeSpan,
+ streamOutputViaSession: streamAgentSessionOutput,
}),
concurrentBridgeStepMetrics,
);
@@ -2899,16 +2911,20 @@ const STARTUP_BRIDGE_BATCH = "bridge";
/**
* Open the one root span for a sandbox bring-up and return its parent-context
- * token plus a guarded `end`. The span parents every startup boundary span:
- * the engine forwards `parentContext` to each `measureStartupStep` call. The
- * `end` closure runs at most once (bring-up complete OR a bring-up failure) and
- * swallows every tracer error, so observability never changes startup control
- * flow. With no injected trace context, the tracer is a no-op and the span is
- * a no-op.
+ * token plus a guarded `end`. The span parents to the run root span through
+ * `runParentContext`, so `sandbox.startup` becomes a child of `task.run`. The
+ * span then parents every startup boundary span in turn: the engine forwards
+ * `parentContext` to each `measureStartupStep` call. The `end` closure runs at
+ * most once (bring-up complete OR a bring-up failure) and swallows every tracer
+ * error, so observability never changes startup control flow. With no injected
+ * trace context, the tracer is a no-op and the span is a no-op.
*/
function openStartupRootSpan(
tracing: StartupTraceContext,
nowMs: () => number,
+ // The run root span parent context. `sandbox.startup` opens as a child of it,
+ // so the whole bring-up parents to `task.run`. It is an opaque token here.
+ runParentContext: StartupSpanContext,
// Return the final root-span numbers and context at end time. The work sum
// and the cold-start flag are known only after the bring-up runs, so the
// caller reads them lazily here.
@@ -2919,7 +2935,7 @@ function openStartupRootSpan(
} {
let span: StartupSpan;
try {
- span = tracing.tracer.startSpan(STARTUP_ROOT_SPAN_NAME);
+ span = tracing.tracer.startSpan(STARTUP_ROOT_SPAN_NAME, undefined, runParentContext);
} catch {
span = NOOP_STARTUP_SPAN;
}
@@ -2954,6 +2970,146 @@ function openStartupRootSpan(
};
}
+/** The stable name of the one root span for a whole run. It is a fixed
+ * low-cardinality constant, never derived from run or user data. */
+const RUN_ROOT_SPAN_NAME = "task.run";
+
+/** The stable name of the one span for the agent turn. It is a fixed
+ * low-cardinality constant, never derived from run or user data. The turn span
+ * is a child of the run root span. */
+const TURN_SPAN_NAME = "agent.turn";
+
+/** The attribute prefix for the run root span. It groups the run-level span
+ * attributes under one namespace, the same shape as the sandbox startup
+ * prefix. */
+const RUN_ROOT_SPAN_ATTR_PREFIX = "paperclip.task.run.";
+
+/** The attribute prefix for the agent turn span. It groups the turn-level span
+ * attributes under one namespace, the same shape as the run root prefix. */
+const TURN_SPAN_ATTR_PREFIX = "paperclip.agent.turn.";
+
+/** Map a run id to a non-reversible 12-hex hash for a span attribute. The raw
+ * run id never rides a span; only this hash does. This mirrors the id-hash rule
+ * that `clampSpanLabel` uses for the startup ids. */
+function hashRunId(runId: string): string {
+ return createHash("sha256").update(runId).digest("hex").slice(0, 12);
+}
+
+/**
+ * Open the one root span for a whole run and return its parent-context token
+ * plus a guarded `end`. The run root span is the trace root: the sandbox
+ * bring-up span (`sandbox.startup`) parents to it, so the engine forwards
+ * `parentContext` into `openStartupRootSpan`. The `end` closure runs at most
+ * once and swallows every tracer error, so observability never changes run
+ * control flow. With no injected trace context the tracer is a no-op and the
+ * span is a no-op.
+ *
+ * The span carries only a bounded, non-reversible run-id hash and its own wall
+ * time. It never carries the prompt, the command, or any user text, so no raw
+ * run text rides the span. This follows the same allowlist rule as
+ * `openStartupRootSpan`.
+ */
+function openRunRootSpan(
+ tracing: StartupTraceContext,
+ nowMs: () => number,
+ runId: string,
+): {
+ parentContext: StartupSpanContext;
+ end: (failed: boolean) => void;
+} {
+ let span: StartupSpan;
+ try {
+ span = tracing.tracer.startSpan(RUN_ROOT_SPAN_NAME);
+ } catch {
+ span = NOOP_STARTUP_SPAN;
+ }
+ let parentContext: StartupSpanContext;
+ try {
+ parentContext = tracing.contextWithSpan(span);
+ } catch {
+ parentContext = undefined;
+ }
+ const startedAtMs = nowMs();
+ let ended = false;
+ return {
+ parentContext,
+ end: (failed: boolean) => {
+ if (ended) return;
+ ended = true;
+ try {
+ // The run id rides only as a non-reversible short hash, never as the raw
+ // id. The wall time is a plain duration. No raw run text rides the span.
+ span.setAttribute(`${RUN_ROOT_SPAN_ATTR_PREFIX}run_id`, hashRunId(runId));
+ span.setAttribute(`${RUN_ROOT_SPAN_ATTR_PREFIX}wall_ms`, nowMs() - startedAtMs);
+ // `2` is `SpanStatusCode.ERROR`. `adapter-utils` stays OTel-free, so it
+ // uses the numeric value that a real injected span reads as the error
+ // status.
+ if (failed) span.setStatus({ code: 2 });
+ span.end();
+ } catch {
+ // Observability must not change run control flow.
+ }
+ },
+ };
+}
+
+/**
+ * Open the one span for the agent turn and return its parent-context token plus
+ * a guarded `end`. The span parents to the run root span through
+ * `runParentContext`, so `agent.turn` becomes a child of `task.run`. The
+ * executor holds the returned `parentContext` for later exec parenting. The
+ * `end` closure runs at most once and swallows every tracer error, so
+ * observability never changes turn control flow. With no injected trace context
+ * the tracer is a no-op and the span is a no-op.
+ *
+ * The span carries only its own wall time. It never carries the prompt, the
+ * command, or any user text, so no raw run text rides the span. This follows the
+ * same allowlist rule as `openStartupRootSpan`.
+ */
+function openTurnSpan(
+ tracing: StartupTraceContext,
+ nowMs: () => number,
+ // The run root span parent context. `agent.turn` opens as a child of it, so
+ // the turn parents to `task.run`. It is an opaque token here.
+ runParentContext: StartupSpanContext,
+): {
+ parentContext: StartupSpanContext;
+ end: (failed: boolean) => void;
+} {
+ let span: StartupSpan;
+ try {
+ span = tracing.tracer.startSpan(TURN_SPAN_NAME, undefined, runParentContext);
+ } catch {
+ span = NOOP_STARTUP_SPAN;
+ }
+ let parentContext: StartupSpanContext;
+ try {
+ parentContext = tracing.contextWithSpan(span);
+ } catch {
+ parentContext = undefined;
+ }
+ const startedAtMs = nowMs();
+ let ended = false;
+ return {
+ parentContext,
+ end: (failed: boolean) => {
+ if (ended) return;
+ ended = true;
+ try {
+ // The wall time is a plain duration. No raw run text rides the span.
+ span.setAttribute(`${TURN_SPAN_ATTR_PREFIX}wall_ms`, nowMs() - startedAtMs);
+ // `2` is `SpanStatusCode.ERROR`. `adapter-utils` stays OTel-free, so it
+ // uses the numeric value that a real injected span reads as the error
+ // status.
+ if (failed) span.setStatus({ code: 2 });
+ span.end();
+ } catch {
+ // Observability must not change turn control flow.
+ }
+ },
+ };
+}
+
export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) {
const createRuntime = deps.createRuntime ?? createAcpRuntime;
const now = deps.now ?? (() => Date.now());
@@ -2975,20 +3131,11 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) {
billingType: billingIdentity?.billingType ?? ("unknown" as const),
};
const warmIdleMs = asNumber(ctx.config.warmHandleIdleMs, DEFAULT_ACP_ENGINE_WARM_HANDLE_IDLE_MS);
- // Evict idle staged runtimes BEFORE building the runtime, since buildRuntime
- // consults the staged cache to decide whether a compatible resume may reuse
- // an already-staged runtime — an expired entry must not be reused.
- await cleanupIdleStagedRuntimes({
- handles: stagedRuntimes,
- locks: stagingLocks,
- now,
- idleMs: warmIdleMs,
- });
- // The `sandbox.startup` span names a sandbox bring-up. It must not cover a
- // local or SSH run: those runs have no sandbox, so they stay out of sandbox
- // telemetry. Open the real root span only when the target is a remote
- // sandbox; every other target forces the no-op trace context, so the whole
- // startup span path stays inert regardless of the injected context.
+ // The `task.run` and `sandbox.startup` spans must not cover a local or SSH
+ // run: those runs have no sandbox, so they stay out of sandbox telemetry.
+ // Open a real root span only when the target is a remote sandbox and the
+ // server injected a trace context; every other target forces the no-op
+ // trace context, so the whole span path stays inert.
const startupExecutionTarget = readAdapterExecutionTarget({
executionTarget: ctx.executionTarget,
legacyRemoteExecution: ctx.executionTransport?.remoteExecution,
@@ -2998,591 +3145,656 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) {
? startupExecutionTarget
: null;
const targetsRemoteSandbox = sandboxTarget !== null;
- // Open the one root span for this bring-up. It spans `buildRuntime` through
- // `acp.handshake`, so every startup boundary span parents to it. `spanParent`
- // carries the injected tracer + the root parent-context token into each
- // `measureStartupStep` call. With no injected trace context the whole path
- // is a no-op. `endRootSpan` runs exactly once — at bring-up completion or on
- // a bring-up failure.
const tracing =
targetsRemoteSandbox && ctx.startupTraceContext
? ctx.startupTraceContext
: NOOP_STARTUP_TRACE_CONTEXT;
- // The sum of the step wall times. The root span records it as `root.work_ms`
- // and the difference from its own wall time as `root.diff_ms` (the overlap
- // the parallel steps saved). Every step reports its wall time through
- // `onWallMs`; a skipped step adds zero.
- let stepWallSumMs = 0;
- // Whether this bring-up is a cold start (no warm handle). Set once the warm-
- // handle lookup runs below; it stays undefined on an early build failure, so
- // the root span omits the attribute (fail open).
- let coldStart: boolean | undefined;
- const rootSpan = openStartupRootSpan(tracing, now, () => ({
- workMs: stepWallSumMs,
- context: {
- coldStart,
- // The provider key and the lease id are the only low-cardinality
- // context values this provider-agnostic layer holds. The region, the
- // image id, and the sandbox id are not threaded here, so the root span
- // omits them (fail open). The lease id rides only as a hash.
- provider: sandboxTarget?.providerKey ?? undefined,
- leaseId: sandboxTarget?.leaseId ?? undefined,
- },
- }));
- const spanParent: Pick<
- StartupStepMeasureOptions,
- "tracer" | "parentContext" | "contextWithSpan" | "onWallMs"
- > = {
- tracer: tracing.tracer,
- parentContext: rootSpan.parentContext,
- // Each step uses this to publish its own child context, so an inner exec
- // span parents to the step span, not to the root.
- contextWithSpan: (span) => tracing.contextWithSpan(span),
- // Accumulate each step wall time into the root work sum.
- onWallMs: (wallMs) => {
- stepWallSumMs += wallMs;
- },
- };
- let prepared: AcpxPreparedRuntime;
- try {
- prepared = await buildRuntime({ ctx, engine, deps, spanParent });
- } catch (err) {
- rootSpan.end(true);
- throw err;
- }
- // Per-project staging outcomes for the referenced (mentioned) projects, surfaced back to the
- // server on the run result. A referenced project that failed to stage into the sandbox is a
- // first-class, counted failure in the requested-vs-synced observability, not only a warning. The
- // list is empty on a local target, on a transport that does not stage referenced projects, or
- // when every staged referenced project succeeded, so the spread adds the field only when there
- // is a failure to report.
- const referencedProjectStagingFailures = (
- prepared.stagedRuntime?.additionalSourceFailures ?? []
- ).map((failure) => ({ projectId: failure.projectId }));
- const referencedProjectStagingFailuresField =
- referencedProjectStagingFailures.length > 0 ? { referencedProjectStagingFailures } : {};
- // State the effective wall-clock timeout and its source up front so a
- // later timeout is diagnosable from the run log alone. Goes to stderr:
- // the acpx stdout log stream carries JSON acpx.* event payloads and must
- // stay machine-parseable line by line.
- await ctx.onLog(
- "stderr",
- `[paperclip] ${formatAdapterExecutionTimeoutStartLogLine(prepared.timeoutResolution)}\n`,
+ // Open the one run root span at the engine first line, before any bring-up
+ // work. It is the trace root for the whole run: the sandbox bring-up and the
+ // agent turn parent to it. The engine forwards its parent context into the
+ // startup span. `runRootSpan.end` runs exactly once, in the `finally` below,
+ // on every return and on a throw.
+ const runRootSpan = openRunRootSpan(tracing, now, ctx.runId);
+ // Hold the current-run parent-context token for the whole run. It starts as
+ // the `task.run` token, switches to the `agent.turn` token during the turn,
+ // and switches back to the `task.run` token after the turn. It is never
+ // `undefined` while the run is live. The holder is a run-scoped local, so two
+ // concurrent runs in one host process keep separate tokens. A detached exec
+ // reads it through `getRuntimeParentContext` to parent to the live span.
+ let currentRunParentContext: StartupSpanContext | undefined = runRootSpan.parentContext;
+ const getRuntimeParentContext = (): StartupSpanContext | undefined => currentRunParentContext;
+ // Wrap each unit of bridge run-time work (one outbound ACP message, one poll
+ // tick, one callback request) in its own named span, parented to the live run
+ // span. The runner reads the run parent per call through
+ // `getRuntimeParentContext`, so a wrapper span always parents to the current
+ // run span. On a no-op trace context the runner opens no real span.
+ const runRuntimeSpan = createRuntimeSpanRunner(tracing, getRuntimeParentContext);
+ // Wrap the host workspace tarball build in one `pack` span. This runner
+ // parents each span to the ACTIVE startup step (not the run span), so the
+ // `pack` span nests under the `stage.sync` step that runs the staging seam.
+ // The staging seam runs inside `stage.sync`'s measured step, so
+ // `getActiveStepContext()` returns that step's child context at pack time.
+ // On a no-op trace context the runner opens no real span.
+ const runStageSpan = createRuntimeSpanRunner(
+ tracing,
+ () => getActiveStepContext()?.parentContext,
);
- await cleanupIdleHandles({ handles: warmHandles, now: now(), idleMs: warmIdleMs });
-
- const previousParams = parseObject(ctx.runtime.sessionParams);
- const canResume = isCompatibleSession(previousParams, prepared);
- const resumeSessionId = canResume ? asString(previousParams.acpSessionId, "") || undefined : undefined;
- const cached = canResume ? warmHandles.get(prepared.sessionKey) : undefined;
- const childStderrState = cached?.childStderrState ?? { logPath: null, pendingLiveLine: "" };
- const processIdentitySink = cached?.processIdentitySink ?? {
- current: ctx.onSpawn,
- latest: null,
- };
- // ACPX runtimes can stay warm across heartbeat runs. Keep the callback
- // target mutable so a later agent respawn records identity on the current
- // heartbeat instead of the run that originally created the runtime.
- processIdentitySink.current = ctx.onSpawn;
- flushChildStderr(childStderrState);
- childStderrState.logPath = prepared.childStderrLogPath;
- const runtimeOptions: PaperclipAcpRuntimeOptions = {
- cwd: prepared.cwd,
- // Host-only spawn cwd for the relay proxy on the remote process-session
- // lane; `undefined` elsewhere so acpx falls back to `cwd` (byte-identical).
- // The advertised `session/new` cwd (`prepared.cwd` = `remoteCwd`) and the
- // fingerprint / compat key are unaffected — this redirects ONLY the host
- // `spawn()` `chdir`, not the in-sandbox data path.
- spawnCwd: prepared.hostSpawnCwd,
- sessionStore: createRuntimeStore({ stateDir: prepared.stateDir }),
- agentRegistry: prepared.agentRegistry,
- permissionMode: prepared.permissionMode,
- nonInteractivePermissions: prepared.nonInteractivePermissions,
- mcpServers: prepared.mcpServers,
- timeoutMs: prepared.timeoutSec > 0 ? prepared.timeoutSec * 1000 : undefined,
- // Scope ACPX runtime verbose logs to the claude agent only. Codex
- // and custom agents already emit their own per-tool output and don't
- // benefit from doubling the log volume.
- verbose: prepared.acpxAgent === "claude",
- onAgentStderr: prepared.childStderrLogPath
- ? (chunk) => routeChildStderr(childStderrState, chunk)
- : undefined,
- onAgentSpawn: async (meta) => {
- processIdentitySink.latest = meta;
- await processIdentitySink.current?.({
- pid: meta.pid,
- processGroupId: null,
- startedAt: meta.startedAt,
- });
- },
- };
- // Open Q2: split the ~7s `acp.handshake` into the two in-repo-observable
- // sub-phases — the ACP runtime construction (`createRuntime`) vs the session
- // establishment envelope (`ensureSession`). The patched spawn lifecycle
- // hook records process identity, but the finer spawn/`initialize`/
- // `session/new` timing split still lives inside external `acpx`.
- // `createRuntime` runs once and only on a cold start; a warm-handle hit
- // reuses `cached.runtime`, so `createRuntimeMs` stays undefined and the
- // split reports nothing for it.
- let createRuntimeMs: number | undefined;
- let runtime: AcpRuntime;
- // A warm handle reuses the running ACP runtime; a miss constructs one. The
- // root span records this as `cold_start`.
- coldStart = !cached?.runtime;
- if (cached?.runtime) {
- runtime = cached.runtime;
- } else {
- const createRuntimeStart = now();
- runtime = createRuntime(runtimeOptions);
- createRuntimeMs = now() - createRuntimeStart;
- }
- if (cached) clearWarmHandleTimer(cached);
- if (!canResume && asString(previousParams.runtimeSessionName, "")) {
+ // `runFailed` marks the run root span status at end time. It stays `true`
+ // until the run reaches a clean completed turn, so every failure and every
+ // early exit closes the span with error status.
+ let runFailed = true;
+ try {
+ // Evict idle staged runtimes BEFORE building the runtime, since buildRuntime
+ // consults the staged cache to decide whether a compatible resume may reuse
+ // an already-staged runtime — an expired entry must not be reused.
+ await cleanupIdleStagedRuntimes({
+ handles: stagedRuntimes,
+ locks: stagingLocks,
+ now,
+ idleMs: warmIdleMs,
+ });
+ // The sum of the step wall times. The root span records it as `root.work_ms`
+ // and the difference from its own wall time as `root.diff_ms` (the overlap
+ // the parallel steps saved). Every step reports its wall time through
+ // `onWallMs`; a skipped step adds zero.
+ let stepWallSumMs = 0;
+ // Whether this bring-up is a cold start (no warm handle). Set once the warm-
+ // handle lookup runs below; it stays undefined on an early build failure, so
+ // the root span omits the attribute (fail open).
+ let coldStart: boolean | undefined;
+ const rootSpan = openStartupRootSpan(tracing, now, runRootSpan.parentContext, () => ({
+ workMs: stepWallSumMs,
+ context: {
+ coldStart,
+ // The provider key and the lease id are the only low-cardinality
+ // context values this provider-agnostic layer holds. The region, the
+ // image id, and the sandbox id are not threaded here, so the root span
+ // omits them (fail open). The lease id rides only as a hash.
+ provider: sandboxTarget?.providerKey ?? undefined,
+ leaseId: sandboxTarget?.leaseId ?? undefined,
+ },
+ }));
+ const spanParent: Pick<
+ StartupStepMeasureOptions,
+ "tracer" | "parentContext" | "contextWithSpan" | "onWallMs"
+ > = {
+ tracer: tracing.tracer,
+ parentContext: rootSpan.parentContext,
+ // Each step uses this to publish its own child context, so an inner exec
+ // span parents to the step span, not to the root.
+ contextWithSpan: (span) => tracing.contextWithSpan(span),
+ // Accumulate each step wall time into the root work sum.
+ onWallMs: (wallMs) => {
+ stepWallSumMs += wallMs;
+ },
+ };
+ let prepared: AcpxPreparedRuntime;
+ try {
+ // Publish the `sandbox.startup` context to the runtime-parent store for
+ // the whole bring-up. A startup-body exec that runs outside a measured
+ // step reads this token and parents its span to `sandbox.startup`, not to
+ // a detached root. A measured step nests its own `activeStepContextStorage`
+ // run inside this wrap and overrides the store, so an in-step exec still
+ // parents to its step span. On a local or SSH target
+ // `spanParent.parentContext` is a no-op token, so the wrap is inert.
+ prepared = await runWithRuntimeParent(spanParent.parentContext, () =>
+ buildRuntime({ ctx, engine, deps, spanParent, getRuntimeParentContext, runtimeSpan: runRuntimeSpan, stageRuntimeSpan: runStageSpan }),
+ );
+ } catch (err) {
+ rootSpan.end(true);
+ throw err;
+ }
+ // Per-project staging outcomes for the referenced (mentioned) projects, surfaced back to the
+ // server on the run result. A referenced project that failed to stage into the sandbox is a
+ // first-class, counted failure in the requested-vs-synced observability, not only a warning. The
+ // list is empty on a local target, on a transport that does not stage referenced projects, or
+ // when every staged referenced project succeeded, so the spread adds the field only when there
+ // is a failure to report.
+ const referencedProjectStagingFailures = (
+ prepared.stagedRuntime?.additionalSourceFailures ?? []
+ ).map((failure) => ({ projectId: failure.projectId }));
+ const referencedProjectStagingFailuresField =
+ referencedProjectStagingFailures.length > 0 ? { referencedProjectStagingFailures } : {};
+ // State the effective wall-clock timeout and its source up front so a
+ // later timeout is diagnosable from the run log alone. Goes to stderr:
+ // the acpx stdout log stream carries JSON acpx.* event payloads and must
+ // stay machine-parseable line by line.
await ctx.onLog(
- "stdout",
- `[paperclip] ACPX session "${asString(previousParams.runtimeSessionName, "")}" does not match the current agent/cwd/mode/runtime identity; starting fresh in "${prepared.cwd}".\n`,
+ "stderr",
+ `[paperclip] ${formatAdapterExecutionTimeoutStartLogLine(prepared.timeoutResolution)}\n`,
);
- }
+ await cleanupIdleHandles({ handles: warmHandles, now: now(), idleMs: warmIdleMs });
+
+ const previousParams = parseObject(ctx.runtime.sessionParams);
+ const canResume = isCompatibleSession(previousParams, prepared);
+ const resumeSessionId = canResume ? asString(previousParams.acpSessionId, "") || undefined : undefined;
+ const cached = canResume ? warmHandles.get(prepared.sessionKey) : undefined;
+ const childStderrState = cached?.childStderrState ?? { logPath: null, pendingLiveLine: "" };
+ const processIdentitySink = cached?.processIdentitySink ?? {
+ current: ctx.onSpawn,
+ latest: null,
+ };
+ // ACPX runtimes can stay warm across heartbeat runs. Keep the callback
+ // target mutable so a later agent respawn records identity on the current
+ // heartbeat instead of the run that originally created the runtime.
+ processIdentitySink.current = ctx.onSpawn;
+ flushChildStderr(childStderrState);
+ childStderrState.logPath = prepared.childStderrLogPath;
+ const runtimeOptions: PaperclipAcpRuntimeOptions = {
+ cwd: prepared.cwd,
+ // Host-only spawn cwd for the relay proxy on the remote process-session
+ // lane; `undefined` elsewhere so acpx falls back to `cwd` (byte-identical).
+ // The advertised `session/new` cwd (`prepared.cwd` = `remoteCwd`) and the
+ // fingerprint / compat key are unaffected — this redirects ONLY the host
+ // `spawn()` `chdir`, not the in-sandbox data path.
+ spawnCwd: prepared.hostSpawnCwd,
+ sessionStore: createRuntimeStore({ stateDir: prepared.stateDir }),
+ agentRegistry: prepared.agentRegistry,
+ permissionMode: prepared.permissionMode,
+ nonInteractivePermissions: prepared.nonInteractivePermissions,
+ mcpServers: prepared.mcpServers,
+ timeoutMs: prepared.timeoutSec > 0 ? prepared.timeoutSec * 1000 : undefined,
+ // Scope ACPX runtime verbose logs to the claude agent only. Codex
+ // and custom agents already emit their own per-tool output and don't
+ // benefit from doubling the log volume.
+ verbose: prepared.acpxAgent === "claude",
+ onAgentStderr: prepared.childStderrLogPath
+ ? (chunk) => routeChildStderr(childStderrState, chunk)
+ : undefined,
+ onAgentSpawn: async (meta) => {
+ processIdentitySink.latest = meta;
+ await processIdentitySink.current?.({
+ pid: meta.pid,
+ processGroupId: null,
+ startedAt: meta.startedAt,
+ });
+ },
+ getRuntimeParentContext,
+ };
+ // Open Q2: split the ~7s `acp.handshake` into the two in-repo-observable
+ // sub-phases — the ACP runtime construction (`createRuntime`) vs the session
+ // establishment envelope (`ensureSession`). The patched spawn lifecycle
+ // hook records process identity, but the finer spawn/`initialize`/
+ // `session/new` timing split still lives inside external `acpx`.
+ // `createRuntime` runs once and only on a cold start; a warm-handle hit
+ // reuses `cached.runtime`, so `createRuntimeMs` stays undefined and the
+ // split reports nothing for it.
+ let createRuntimeMs: number | undefined;
+ let runtime: AcpRuntime;
+ // A warm handle reuses the running ACP runtime; a miss constructs one. The
+ // root span records this as `cold_start`.
+ coldStart = !cached?.runtime;
+ if (cached?.runtime) {
+ runtime = cached.runtime;
+ } else {
+ const createRuntimeStart = now();
+ runtime = createRuntime(runtimeOptions);
+ createRuntimeMs = now() - createRuntimeStart;
+ }
+ if (cached) clearWarmHandleTimer(cached);
+ if (!canResume && asString(previousParams.runtimeSessionName, "")) {
+ await ctx.onLog(
+ "stdout",
+ `[paperclip] ACPX session "${asString(previousParams.runtimeSessionName, "")}" does not match the current agent/cwd/mode/runtime identity; starting fresh in "${prepared.cwd}".\n`,
+ );
+ }
- let handle = cached?.handle ?? null;
- let resumedSession = Boolean(handle ?? resumeSessionId);
- let clearSession = false;
+ let handle = cached?.handle ?? null;
+ let resumedSession = Boolean(handle ?? resumeSessionId);
+ let clearSession = false;
- try {
- if (!handle) {
- try {
- // Step 7 — acp.handshake: ACP session establishment (session/new or
- // resume). A throwing handshake still reports its duration before the
- // resume-retry path below runs. `roundTrips` is expected to be 0 (the
- // ACP client is external, not the host exec seam); the payload also
- // carries the createRuntime/ensureSession sub-split (Open Q2).
- let ensureSessionMs: number | undefined;
- handle = await measureStartupStep(ctx, now, "acp.handshake", async () => {
- const ensureSessionStart = now();
- const established = await runtime.ensureSession({
- sessionKey: prepared.sessionKey,
- agent: prepared.acpxAgent,
- mode: prepared.mode,
- cwd: prepared.cwd,
- resumeSessionId,
- sessionOptions: { env: prepared.env },
+ try {
+ if (!handle) {
+ try {
+ // Step 7 — acp.handshake: ACP session establishment (session/new or
+ // resume). A throwing handshake still reports its duration before the
+ // resume-retry path below runs. The createRuntime/ensureSession
+ // sub-split rides the step span as fixed, closed keys (Open Q2).
+ let ensureSessionMs: number | undefined;
+ handle = await measureStartupStep(ctx, now, "acp.handshake", async () => {
+ const ensureSessionStart = now();
+ const established = await runtime.ensureSession({
+ sessionKey: prepared.sessionKey,
+ agent: prepared.acpxAgent,
+ mode: prepared.mode,
+ cwd: prepared.cwd,
+ resumeSessionId,
+ sessionOptions: { env: prepared.env },
+ });
+ ensureSessionMs = now() - ensureSessionStart;
+ return established;
+ }, {
+ ...prepared.stepMetrics,
+ // The two sub-times ride the span as fixed, closed keys.
+ spanWallTimes: () => ({
+ createRuntime: createRuntimeMs,
+ ensureSession: ensureSessionMs,
+ }),
});
- ensureSessionMs = now() - ensureSessionStart;
- return established;
- }, {
- ...prepared.stepMetrics,
- extra: () => ({
- ...(createRuntimeMs !== undefined ? { createRuntimeMs } : {}),
- ...(ensureSessionMs !== undefined ? { ensureSessionMs } : {}),
- }),
- // The same two sub-times ride the span as fixed, closed keys.
- spanWallTimes: () => ({
- createRuntime: createRuntimeMs,
- ensureSession: ensureSessionMs,
- }),
- });
- } catch (err) {
- if (!resumeSessionId || !isResumeFailure(err)) throw err;
- clearSession = true;
- resumedSession = false;
- await ctx.onLog(
- "stdout",
- `[paperclip] ACPX resume session "${resumeSessionId}" is unavailable; retrying with a fresh session.\n`,
- );
- // Fresh-session retry: the runtime was already constructed on the
- // first attempt (never re-created), so this event reports only its
- // own `ensureSessionMs` — no `createRuntimeMs`.
- let retryEnsureSessionMs: number | undefined;
- handle = await measureStartupStep(ctx, now, "acp.handshake", async () => {
- const ensureSessionStart = now();
- const established = await runtime.ensureSession({
- sessionKey: prepared.sessionKey,
- agent: prepared.acpxAgent,
- mode: prepared.mode,
- cwd: prepared.cwd,
- sessionOptions: { env: prepared.env },
+ } catch (err) {
+ if (!resumeSessionId || !isResumeFailure(err)) throw err;
+ clearSession = true;
+ resumedSession = false;
+ await ctx.onLog(
+ "stdout",
+ `[paperclip] ACPX resume session "${resumeSessionId}" is unavailable; retrying with a fresh session.\n`,
+ );
+ // Fresh-session retry: the runtime was already constructed on the
+ // first attempt (never re-created), so this event reports only its
+ // own `ensureSessionMs` — no `createRuntimeMs`.
+ let retryEnsureSessionMs: number | undefined;
+ handle = await measureStartupStep(ctx, now, "acp.handshake", async () => {
+ const ensureSessionStart = now();
+ const established = await runtime.ensureSession({
+ sessionKey: prepared.sessionKey,
+ agent: prepared.acpxAgent,
+ mode: prepared.mode,
+ cwd: prepared.cwd,
+ sessionOptions: { env: prepared.env },
+ });
+ retryEnsureSessionMs = now() - ensureSessionStart;
+ return established;
+ }, {
+ ...prepared.stepMetrics,
+ // The retry reuses the runtime from the first attempt, so it reports
+ // only its own ensure-session sub-time on the span.
+ spanWallTimes: () => ({ ensureSession: retryEnsureSessionMs }),
});
- retryEnsureSessionMs = now() - ensureSessionStart;
- return established;
- }, {
- ...prepared.stepMetrics,
- extra: () => ({
- ...(retryEnsureSessionMs !== undefined ? { ensureSessionMs: retryEnsureSessionMs } : {}),
- }),
- // The retry reuses the runtime from the first attempt, so it reports
- // only its own ensure-session sub-time on the span.
- spanWallTimes: () => ({ ensureSession: retryEnsureSessionMs }),
+ }
+ } else {
+ // Warm-handle hit: a compatible cached handle reuses the running ACP
+ // agent, so the `acp.handshake` step does no work. Emit a step span and
+ // event with `outcome = skipped` and a zero wall time, so the trace and
+ // the run log show the skip as a distinct outcome, never a misleading
+ // zero-work `ok` step.
+ await emitSkippedStartupStep(ctx, "acp.handshake", {
+ tracer: prepared.stepMetrics.tracer,
+ parentContext: prepared.stepMetrics.parentContext,
});
}
- } else {
- // Warm-handle hit: a compatible cached handle reuses the running ACP
- // agent, so the `acp.handshake` step does no work. Emit a step span and
- // event with `outcome = skipped` and a zero wall time, so the trace and
- // the run log show the skip as a distinct outcome, never a misleading
- // zero-work `ok` step.
- await emitSkippedStartupStep(ctx, "acp.handshake", {
- tracer: prepared.stepMetrics.tracer,
- parentContext: prepared.stepMetrics.parentContext,
+ // A compatible warm handle reuses the already-running ACP agent and does
+ // not emit another spawn event. Persist its known identity on this run
+ // before the next prompt starts so every running heartbeat is adoptable.
+ if (handle && cached && processIdentitySink.latest && ctx.onSpawn) {
+ await ctx.onSpawn({
+ pid: processIdentitySink.latest.pid,
+ processGroupId: null,
+ startedAt: processIdentitySink.latest.startedAt,
+ });
+ }
+ } catch (err) {
+ // Bring-up failed at the handshake — close the root span with error status.
+ rootSpan.end(true);
+ const { classified, message } = await emitAcpxFailure({
+ ctx,
+ prepared,
+ err,
+ phase: "ensure_session",
});
+ await discardStagedRuntime({ handles: stagedRuntimes, prepared });
+ await cleanupRemoteBridges(prepared);
+ return {
+ exitCode: 1,
+ signal: null,
+ timedOut: false,
+ errorMessage: message,
+ ...classified,
+ ...billingFields,
+ ...referencedProjectStagingFailuresField,
+ model: prepared.requestedModel || null,
+ clearSession,
+ resultJson: { phase: "ensure_session" },
+ summary: message,
+ };
+ }
+
+ if (!handle) {
+ // Bring-up produced no session handle — close the root span with error status.
+ rootSpan.end(true);
+ await discardStagedRuntime({ handles: stagedRuntimes, prepared });
+ await cleanupRemoteBridges(prepared);
+ return {
+ exitCode: 1,
+ signal: null,
+ timedOut: false,
+ errorMessage: "ACPX did not return a runtime session handle.",
+ errorCode: "acpx_runtime_error",
+ ...billingFields,
+ ...referencedProjectStagingFailuresField,
+ model: prepared.requestedModel || null,
+ resultJson: { phase: "ensure_session" },
+ summary: "ACPX did not return a runtime session handle.",
+ };
}
- // A compatible warm handle reuses the already-running ACP agent and does
- // not emit another spawn event. Persist its known identity on this run
- // before the next prompt starts so every running heartbeat is adoptable.
- if (handle && cached && processIdentitySink.latest && ctx.onSpawn) {
- await ctx.onSpawn({
- pid: processIdentitySink.latest.pid,
- processGroupId: null,
- startedAt: processIdentitySink.latest.startedAt,
+ // Bring-up is complete: the session handle is established. Close the root
+ // span here, so it covers `buildRuntime` through `acp.handshake` and no
+ // further. The agent turn runs after and is out of the startup root's scope.
+ rootSpan.end(false);
+ const sessionHandle = handle;
+ try {
+ await applySessionConfigOptions({
+ runtime,
+ handle: sessionHandle,
+ prepared,
+ onLog: ctx.onLog,
+ });
+ } catch (err) {
+ const { classified, message } = await emitAcpxFailure({
+ ctx,
+ prepared,
+ err,
+ phase: "configure_session",
});
+ await runtime.close({
+ handle: sessionHandle,
+ reason: "paperclip config cleanup",
+ discardPersistentState: false,
+ }).catch(() => {});
+ const existing = warmHandles.get(prepared.sessionKey);
+ if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
+ clearWarmHandleTimer(existing);
+ warmHandles.delete(prepared.sessionKey);
+ }
+ await discardStagedRuntime({ handles: stagedRuntimes, prepared });
+ await cleanupRemoteBridges(prepared);
+ return {
+ exitCode: 1,
+ signal: null,
+ timedOut: false,
+ errorMessage: message,
+ ...classified,
+ ...billingFields,
+ ...referencedProjectStagingFailuresField,
+ model: prepared.requestedModel || null,
+ clearSession,
+ resultJson: {
+ phase: "configure_session",
+ agent: prepared.acpxAgent,
+ requestedModel: prepared.requestedModel || null,
+ requestedThinkingEffort: prepared.requestedThinkingEffort || null,
+ fastMode: prepared.fastMode,
+ },
+ summary: message,
+ };
}
- } catch (err) {
- // Bring-up failed at the handshake — close the root span with error status.
- rootSpan.end(true);
- const { classified, message } = await emitAcpxFailure({
- ctx,
- prepared,
- err,
- phase: "ensure_session",
- });
- await discardStagedRuntime({ handles: stagedRuntimes, prepared });
- await cleanupRemoteBridges(prepared);
- return {
- exitCode: 1,
- signal: null,
- timedOut: false,
- errorMessage: message,
- ...classified,
- ...billingFields,
- ...referencedProjectStagingFailuresField,
- model: prepared.requestedModel || null,
- clearSession,
- resultJson: { phase: "ensure_session" },
- summary: message,
- };
- }
-
- if (!handle) {
- // Bring-up produced no session handle — close the root span with error status.
- rootSpan.end(true);
- await discardStagedRuntime({ handles: stagedRuntimes, prepared });
- await cleanupRemoteBridges(prepared);
- return {
- exitCode: 1,
- signal: null,
- timedOut: false,
- errorMessage: "ACPX did not return a runtime session handle.",
- errorCode: "acpx_runtime_error",
- ...billingFields,
- ...referencedProjectStagingFailuresField,
+ const { prompt, promptMetrics, commandNotes } = await buildPrompt(ctx, resumedSession, prepared.env);
+ const runPrompt = joinPromptSections([prepared.skillPromptInstructions, prompt]);
+ await emitAcpxLog(ctx, {
+ type: "acpx.session",
+ agent: prepared.acpxAgent,
+ sessionId: sessionHandle.backendSessionId,
+ acpSessionId: sessionHandle.backendSessionId,
+ agentSessionId: sessionHandle.agentSessionId,
+ runtimeSessionName: sessionHandle.runtimeSessionName,
+ mode: prepared.mode,
+ permissionMode: prepared.permissionMode,
model: prepared.requestedModel || null,
- resultJson: { phase: "ensure_session" },
- summary: "ACPX did not return a runtime session handle.",
- };
- }
- // Bring-up is complete: the session handle is established. Close the root
- // span here, so it covers `buildRuntime` through `acp.handshake` and no
- // further. The agent turn runs after and is out of the startup root's scope.
- rootSpan.end(false);
- const sessionHandle = handle;
- try {
- await applySessionConfigOptions({
- runtime,
- handle: sessionHandle,
- prepared,
- onLog: ctx.onLog,
+ thinkingEffort: prepared.requestedThinkingEffort || null,
+ fastMode: prepared.fastMode,
});
- } catch (err) {
- const { classified, message } = await emitAcpxFailure({
- ctx,
- prepared,
- err,
- phase: "configure_session",
- });
- await runtime.close({
- handle: sessionHandle,
- reason: "paperclip config cleanup",
- discardPersistentState: false,
- }).catch(() => {});
- const existing = warmHandles.get(prepared.sessionKey);
- if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
- clearWarmHandleTimer(existing);
- warmHandles.delete(prepared.sessionKey);
+ if (ctx.onMeta) {
+ await ctx.onMeta({
+ adapterType: engine.adapterType,
+ command: prepared.agentCommand ?? prepared.acpxAgent,
+ cwd: prepared.cwd,
+ commandNotes: [
+ `ACPX runtime embedded in Paperclip with ${prepared.mode} session mode.`,
+ `Effective ACPX permission mode: ${prepared.permissionMode}.`,
+ ...(prepared.requestedModel
+ ? [
+ prepared.acpxAgent === "claude"
+ ? `Requested ACPX model: ${prepared.requestedModel} (set via ANTHROPIC_MODEL env at startup).`
+ : prepared.acpxAgent === "codex"
+ ? `Requested ACPX model: ${prepared.requestedModel} (set via CODEX_CONFIG at startup).`
+ : `Requested ACPX model: ${prepared.requestedModel}.`,
+ ]
+ : []),
+ ...(prepared.requestedThinkingEffort ? [`Requested ACPX thinking effort: ${prepared.requestedThinkingEffort}.`] : []),
+ ...(prepared.fastMode ? ["Requested ACPX Codex fast mode."] : []),
+ ...(Array.isArray(prepared.skillsIdentity.commandNotes)
+ ? prepared.skillsIdentity.commandNotes.filter((note): note is string => typeof note === "string")
+ : []),
+ ...commandNotes,
+ ],
+ env: prepared.loggedEnv,
+ prompt: runPrompt,
+ promptMetrics,
+ context: ctx.context,
+ });
}
- await discardStagedRuntime({ handles: stagedRuntimes, prepared });
- await cleanupRemoteBridges(prepared);
- return {
- exitCode: 1,
- signal: null,
- timedOut: false,
- errorMessage: message,
- ...classified,
- ...billingFields,
- ...referencedProjectStagingFailuresField,
- model: prepared.requestedModel || null,
- clearSession,
- resultJson: {
- phase: "configure_session",
- agent: prepared.acpxAgent,
- requestedModel: prepared.requestedModel || null,
- requestedThinkingEffort: prepared.requestedThinkingEffort || null,
- fastMode: prepared.fastMode,
- },
- summary: message,
- };
- }
- const { prompt, promptMetrics, commandNotes } = await buildPrompt(ctx, resumedSession, prepared.env);
- const runPrompt = joinPromptSections([prepared.skillPromptInstructions, prompt]);
- await emitAcpxLog(ctx, {
- type: "acpx.session",
- agent: prepared.acpxAgent,
- sessionId: sessionHandle.backendSessionId,
- acpSessionId: sessionHandle.backendSessionId,
- agentSessionId: sessionHandle.agentSessionId,
- runtimeSessionName: sessionHandle.runtimeSessionName,
- mode: prepared.mode,
- permissionMode: prepared.permissionMode,
- model: prepared.requestedModel || null,
- thinkingEffort: prepared.requestedThinkingEffort || null,
- fastMode: prepared.fastMode,
- });
- if (ctx.onMeta) {
- await ctx.onMeta({
- adapterType: engine.adapterType,
- command: prepared.agentCommand ?? prepared.acpxAgent,
- cwd: prepared.cwd,
- commandNotes: [
- `ACPX runtime embedded in Paperclip with ${prepared.mode} session mode.`,
- `Effective ACPX permission mode: ${prepared.permissionMode}.`,
- ...(prepared.requestedModel
- ? [
- prepared.acpxAgent === "claude"
- ? `Requested ACPX model: ${prepared.requestedModel} (set via ANTHROPIC_MODEL env at startup).`
- : prepared.acpxAgent === "codex"
- ? `Requested ACPX model: ${prepared.requestedModel} (set via CODEX_CONFIG at startup).`
- : `Requested ACPX model: ${prepared.requestedModel}.`,
- ]
- : []),
- ...(prepared.requestedThinkingEffort ? [`Requested ACPX thinking effort: ${prepared.requestedThinkingEffort}.`] : []),
- ...(prepared.fastMode ? ["Requested ACPX Codex fast mode."] : []),
- ...(Array.isArray(prepared.skillsIdentity.commandNotes)
- ? prepared.skillsIdentity.commandNotes.filter((note): note is string => typeof note === "string")
- : []),
- ...commandNotes,
- ],
- env: prepared.loggedEnv,
- prompt: runPrompt,
- promptMetrics,
- context: ctx.context,
- });
- }
- let cancelActiveTurn: ((reason: string) => Promise) | null = null;
- let controller: AbortController | null = null;
- let timeout: NodeJS.Timeout | null = null;
- let timedOut = false;
- const textParts: string[] = [];
- let eventBreakdown: AcpRuntimeUsageBreakdown | null = null;
- let eventCostUsd: number | null = null;
- try {
- // Snapshot pre-turn usage so cumulative agent-reported cost can be
- // attributed to this run alone.
- const preTurnStatus = await readRuntimeStatus(runtime, sessionHandle);
- const timeoutMs = prepared.timeoutSec > 0 ? prepared.timeoutSec * 1000 : undefined;
- controller = new AbortController();
- if (timeoutMs) {
- timeout = setTimeout(() => {
- timedOut = true;
- controller?.abort();
- void cancelActiveTurn?.(formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution)).catch(() => {});
- }, timeoutMs);
- }
- const turn = runtime.startTurn({
- handle: sessionHandle,
- text: runPrompt,
- mode: "prompt",
- requestId: ctx.runId,
- timeoutMs,
- signal: controller?.signal,
- });
- cancelActiveTurn = async (reason: string) => {
- await turn.cancel({ reason });
- };
- const toolTitles = new Map();
- for await (const event of turn.events) {
- if (event.type === "text_delta") textParts.push(event.text);
- if (event.type === "status" && event.tag === "usage_update") {
- eventBreakdown = event.breakdown ?? eventBreakdown;
- eventCostUsd = usdCostAmount(event.cost) ?? eventCostUsd;
+ let cancelActiveTurn: ((reason: string) => Promise) | null = null;
+ let controller: AbortController | null = null;
+ let timeout: NodeJS.Timeout | null = null;
+ let timedOut = false;
+ const textParts: string[] = [];
+ let eventBreakdown: AcpRuntimeUsageBreakdown | null = null;
+ let eventCostUsd: number | null = null;
+ // Open the agent turn span as a child of the run root span. It wraps the
+ // whole turn: the executor holds `turnSpan.parentContext` for later exec
+ // parenting, and the `finally` below ends the span once on every path. The
+ // span is declared before the `try` so the `finally` can reach it.
+ const turnSpan = openTurnSpan(tracing, now, runRootSpan.parentContext);
+ // Switch the current-run holder to the `agent.turn` token for the turn, so
+ // a detached exec during the turn parents to `agent.turn`. The turn
+ // `finally` resets the holder to the `task.run` token.
+ currentRunParentContext = turnSpan.parentContext;
+ try {
+ // Snapshot pre-turn usage so cumulative agent-reported cost can be
+ // attributed to this run alone.
+ const preTurnStatus = await readRuntimeStatus(runtime, sessionHandle);
+ const timeoutMs = prepared.timeoutSec > 0 ? prepared.timeoutSec * 1000 : undefined;
+ controller = new AbortController();
+ if (timeoutMs) {
+ timeout = setTimeout(() => {
+ timedOut = true;
+ controller?.abort();
+ void cancelActiveTurn?.(formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution)).catch(() => {});
+ }, timeoutMs);
}
- await emitRuntimeEvent(ctx, event, toolTitles);
- }
- const terminal = await turn.result;
- if (timeout) clearTimeout(timeout);
- // Read usage before the close/warm-handle paths below can discard state.
- const postTurnStatus = await readRuntimeStatus(runtime, sessionHandle);
- const turnUsage = summarizeAcpxTurnUsage({
- preStatus: preTurnStatus,
- postStatus: postTurnStatus,
- eventBreakdown,
- eventCostUsd,
- });
- if (terminal.status === "failed" || terminal.status === "cancelled" || timedOut) {
- const existing = warmHandles.get(prepared.sessionKey);
- if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
- await closeWarmHandle({
- handles: warmHandles,
- key: prepared.sessionKey,
- entry: existing,
- reason: timedOut ? "paperclip timeout cleanup" : `paperclip turn ${terminal.status}`,
- discardPersistentState: terminal.status === "cancelled" || timedOut,
- });
+ const turn = runtime.startTurn({
+ handle: sessionHandle,
+ text: runPrompt,
+ mode: "prompt",
+ requestId: ctx.runId,
+ timeoutMs,
+ signal: controller?.signal,
+ });
+ cancelActiveTurn = async (reason: string) => {
+ await turn.cancel({ reason });
+ };
+ const toolTitles = new Map();
+ for await (const event of turn.events) {
+ if (event.type === "text_delta") textParts.push(event.text);
+ if (event.type === "status" && event.tag === "usage_update") {
+ eventBreakdown = event.breakdown ?? eventBreakdown;
+ eventCostUsd = usdCostAmount(event.cost) ?? eventCostUsd;
+ }
+ await emitRuntimeEvent(ctx, event, toolTitles);
+ }
+ const terminal = await turn.result;
+ if (timeout) clearTimeout(timeout);
+ // Read usage before the close/warm-handle paths below can discard state.
+ const postTurnStatus = await readRuntimeStatus(runtime, sessionHandle);
+ const turnUsage = summarizeAcpxTurnUsage({
+ preStatus: preTurnStatus,
+ postStatus: postTurnStatus,
+ eventBreakdown,
+ eventCostUsd,
+ });
+ if (terminal.status === "failed" || terminal.status === "cancelled" || timedOut) {
+ const existing = warmHandles.get(prepared.sessionKey);
+ if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
+ await closeWarmHandle({
+ handles: warmHandles,
+ key: prepared.sessionKey,
+ entry: existing,
+ reason: timedOut ? "paperclip timeout cleanup" : `paperclip turn ${terminal.status}`,
+ discardPersistentState: terminal.status === "cancelled" || timedOut,
+ });
+ } else {
+ await runtime.close({
+ handle: sessionHandle,
+ reason: timedOut ? "paperclip timeout cleanup" : `paperclip turn ${terminal.status}`,
+ discardPersistentState: terminal.status === "cancelled" || timedOut,
+ }).catch(() => {});
+ }
+ } else if (prepared.mode === "persistent" && warmIdleMs > 0 && !prepared.processSessionBridge) {
+ const existing = warmHandles.get(prepared.sessionKey);
+ if (existing && !warmHandleMatches(existing, runtime, sessionHandle)) {
+ await runtime.close({
+ handle: sessionHandle,
+ reason: "paperclip duplicate warm handle cleanup",
+ discardPersistentState: false,
+ }).catch(() => {});
+ } else {
+ const entry: RuntimeCacheEntry = {
+ runtime,
+ handle: sessionHandle,
+ childStderrState,
+ processIdentitySink,
+ fingerprint: prepared.fingerprint,
+ lastUsedAt: now(),
+ };
+ warmHandles.set(prepared.sessionKey, entry);
+ scheduleIdleHandleCleanup({
+ handles: warmHandles,
+ key: prepared.sessionKey,
+ entry,
+ idleMs: warmIdleMs,
+ now,
+ });
+ }
} else {
- await runtime.close({
- handle: sessionHandle,
- reason: timedOut ? "paperclip timeout cleanup" : `paperclip turn ${terminal.status}`,
- discardPersistentState: terminal.status === "cancelled" || timedOut,
- }).catch(() => {});
+ const existing = warmHandles.get(prepared.sessionKey);
+ if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
+ await closeWarmHandle({
+ handles: warmHandles,
+ key: prepared.sessionKey,
+ entry: existing,
+ reason: "paperclip completed turn cleanup",
+ });
+ } else {
+ await runtime.close({
+ handle: sessionHandle,
+ reason: "paperclip completed turn cleanup",
+ discardPersistentState: false,
+ }).catch(() => {});
+ }
}
- } else if (prepared.mode === "persistent" && warmIdleMs > 0 && !prepared.processSessionBridge) {
- const existing = warmHandles.get(prepared.sessionKey);
- if (existing && !warmHandleMatches(existing, runtime, sessionHandle)) {
- await runtime.close({
- handle: sessionHandle,
- reason: "paperclip duplicate warm handle cleanup",
- discardPersistentState: false,
- }).catch(() => {});
+
+ // PR 3: keep the staged runtime warm for the next compatible resume only
+ // after a clean turn; a failed/cancelled/timed-out turn discards it so the
+ // next run stages fresh instead of reusing a torn-down session's staged
+ // credentials. Copy-back still fires for every outcome via
+ // `cleanupRemoteBridges` below (unchanged from PR 2).
+ if (terminal.status === "completed" && !timedOut) {
+ saveStagedRuntimeAfterCleanTurn({ handles: stagedRuntimes, prepared, now: now() });
} else {
- const entry: RuntimeCacheEntry = {
- runtime,
- handle: sessionHandle,
- childStderrState,
- processIdentitySink,
- fingerprint: prepared.fingerprint,
- lastUsedAt: now(),
- };
- warmHandles.set(prepared.sessionKey, entry);
- scheduleIdleHandleCleanup({
- handles: warmHandles,
- key: prepared.sessionKey,
- entry,
- idleMs: warmIdleMs,
- now,
- });
+ await discardStagedRuntime({ handles: stagedRuntimes, prepared });
}
- } else {
+
+ const errorMessage = timedOut
+ ? formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution)
+ : resultErrorMessage(terminal);
+ const terminalStopReason = terminal.status === "failed" ? terminal.error.message : terminal.stopReason;
+ await emitAcpxLog(ctx, {
+ type: terminal.status === "completed" ? "acpx.result" : "acpx.error",
+ summary: terminal.status,
+ stopReason: terminalStopReason,
+ message: errorMessage,
+ });
+ await cleanupRemoteBridges(prepared);
+ flushChildStderr(childStderrState);
+ // The one clean-completion path clears the run failure flag; every other
+ // path keeps it set, so the run root span closes with error status.
+ runFailed = terminal.status === "completed" && !timedOut ? false : true;
+ return {
+ exitCode: terminal.status === "completed" ? 0 : 1,
+ signal: timedOut ? "SIGTERM" : null,
+ timedOut,
+ errorMessage,
+ errorCode: terminal.status === "failed" ? "acpx_turn_failed" : timedOut ? "acpx_timeout" : null,
+ sessionId: sessionHandle.backendSessionId ?? sessionHandle.runtimeSessionName,
+ sessionParams: buildSessionParams({ prepared, handle: sessionHandle }),
+ sessionDisplayId: sessionHandle.agentSessionId ?? sessionHandle.backendSessionId ?? sessionHandle.runtimeSessionName,
+ ...billingFields,
+ ...referencedProjectStagingFailuresField,
+ model: prepared.requestedModel || null,
+ ...(turnUsage.usage ? { usage: turnUsage.usage, usageBasis: "per_run" as const } : {}),
+ costUsd: turnUsage.costUsd,
+ resultJson: {
+ status: terminal.status,
+ stopReason: terminalStopReason,
+ permissionMode: prepared.permissionMode,
+ mode: prepared.mode,
+ requestedModel: prepared.requestedModel || null,
+ requestedThinkingEffort: prepared.requestedThinkingEffort || null,
+ fastMode: prepared.fastMode,
+ ...(turnUsage.usageDetail ? { usage: turnUsage.usageDetail } : {}),
+ ...(turnUsage.cumulativeCostUsd != null
+ ? { cumulativeCostUsd: turnUsage.cumulativeCostUsd }
+ : {}),
+ },
+ summary: textParts.join("").trim() || terminalStopReason || terminal.status,
+ clearSession,
+ };
+ } catch (err) {
+ if (timeout) clearTimeout(timeout);
+ const messageOverride = timedOut
+ ? formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution)
+ : undefined;
+ const cancel = cancelActiveTurn as ((reason: string) => Promise) | null;
+ const preEmitMessage =
+ messageOverride ?? (err instanceof Error ? err.message : String(err));
+ if (cancel) await cancel(preEmitMessage).catch(() => {});
+ await runtime.close({
+ handle: sessionHandle,
+ reason: timedOut ? "paperclip timeout cleanup" : "paperclip error cleanup",
+ discardPersistentState: timedOut,
+ }).catch(() => {});
const existing = warmHandles.get(prepared.sessionKey);
if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
- await closeWarmHandle({
- handles: warmHandles,
- key: prepared.sessionKey,
- entry: existing,
- reason: "paperclip completed turn cleanup",
- });
- } else {
- await runtime.close({
- handle: sessionHandle,
- reason: "paperclip completed turn cleanup",
- discardPersistentState: false,
- }).catch(() => {});
+ clearWarmHandleTimer(existing);
+ warmHandles.delete(prepared.sessionKey);
}
- }
-
- // PR 3: keep the staged runtime warm for the next compatible resume only
- // after a clean turn; a failed/cancelled/timed-out turn discards it so the
- // next run stages fresh instead of reusing a torn-down session's staged
- // credentials. Copy-back still fires for every outcome via
- // `cleanupRemoteBridges` below (unchanged from PR 2).
- if (terminal.status === "completed" && !timedOut) {
- saveStagedRuntimeAfterCleanTurn({ handles: stagedRuntimes, prepared, now: now() });
- } else {
await discardStagedRuntime({ handles: stagedRuntimes, prepared });
+ const { classified, message } = await emitAcpxFailure({
+ ctx,
+ prepared,
+ err,
+ phase: "turn",
+ messageOverride,
+ });
+ await cleanupRemoteBridges(prepared);
+ flushChildStderr(childStderrState);
+ return {
+ exitCode: 1,
+ signal: timedOut ? "SIGTERM" : null,
+ timedOut,
+ errorMessage: message,
+ errorCode: timedOut ? "acpx_timeout" : classified.errorCode,
+ errorMeta: classified.errorMeta,
+ ...billingFields,
+ ...referencedProjectStagingFailuresField,
+ model: prepared.requestedModel || null,
+ clearSession: clearSession || timedOut,
+ resultJson: { phase: "turn" },
+ summary: message,
+ };
+ } finally {
+ // End the agent turn span exactly once, on every return and on a throw.
+ // `runFailed` is `false` only on a completed, non-timed-out turn, so the
+ // span status is correct for success, error, and timeout.
+ turnSpan.end(runFailed);
+ // Reset the current-run holder to the `task.run` token after the turn.
+ // The run stays live here, so the holder is never `undefined`. A detached
+ // exec after the turn parents to `task.run`.
+ currentRunParentContext = runRootSpan.parentContext;
}
-
- const errorMessage = timedOut
- ? formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution)
- : resultErrorMessage(terminal);
- const terminalStopReason = terminal.status === "failed" ? terminal.error.message : terminal.stopReason;
- await emitAcpxLog(ctx, {
- type: terminal.status === "completed" ? "acpx.result" : "acpx.error",
- summary: terminal.status,
- stopReason: terminalStopReason,
- message: errorMessage,
- });
- await cleanupRemoteBridges(prepared);
- flushChildStderr(childStderrState);
- return {
- exitCode: terminal.status === "completed" ? 0 : 1,
- signal: timedOut ? "SIGTERM" : null,
- timedOut,
- errorMessage,
- errorCode: terminal.status === "failed" ? "acpx_turn_failed" : timedOut ? "acpx_timeout" : null,
- sessionId: sessionHandle.backendSessionId ?? sessionHandle.runtimeSessionName,
- sessionParams: buildSessionParams({ prepared, handle: sessionHandle }),
- sessionDisplayId: sessionHandle.agentSessionId ?? sessionHandle.backendSessionId ?? sessionHandle.runtimeSessionName,
- ...billingFields,
- ...referencedProjectStagingFailuresField,
- model: prepared.requestedModel || null,
- ...(turnUsage.usage ? { usage: turnUsage.usage, usageBasis: "per_run" as const } : {}),
- costUsd: turnUsage.costUsd,
- resultJson: {
- status: terminal.status,
- stopReason: terminalStopReason,
- permissionMode: prepared.permissionMode,
- mode: prepared.mode,
- requestedModel: prepared.requestedModel || null,
- requestedThinkingEffort: prepared.requestedThinkingEffort || null,
- fastMode: prepared.fastMode,
- ...(turnUsage.usageDetail ? { usage: turnUsage.usageDetail } : {}),
- ...(turnUsage.cumulativeCostUsd != null
- ? { cumulativeCostUsd: turnUsage.cumulativeCostUsd }
- : {}),
- },
- summary: textParts.join("").trim() || terminalStopReason || terminal.status,
- clearSession,
- };
- } catch (err) {
- if (timeout) clearTimeout(timeout);
- const messageOverride = timedOut
- ? formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution)
- : undefined;
- const cancel = cancelActiveTurn as ((reason: string) => Promise) | null;
- const preEmitMessage =
- messageOverride ?? (err instanceof Error ? err.message : String(err));
- if (cancel) await cancel(preEmitMessage).catch(() => {});
- await runtime.close({
- handle: sessionHandle,
- reason: timedOut ? "paperclip timeout cleanup" : "paperclip error cleanup",
- discardPersistentState: timedOut,
- }).catch(() => {});
- const existing = warmHandles.get(prepared.sessionKey);
- if (warmHandleMatches(existing, runtime, sessionHandle) && existing) {
- clearWarmHandleTimer(existing);
- warmHandles.delete(prepared.sessionKey);
- }
- await discardStagedRuntime({ handles: stagedRuntimes, prepared });
- const { classified, message } = await emitAcpxFailure({
- ctx,
- prepared,
- err,
- phase: "turn",
- messageOverride,
- });
- await cleanupRemoteBridges(prepared);
- flushChildStderr(childStderrState);
- return {
- exitCode: 1,
- signal: timedOut ? "SIGTERM" : null,
- timedOut,
- errorMessage: message,
- errorCode: timedOut ? "acpx_timeout" : classified.errorCode,
- errorMeta: classified.errorMeta,
- ...billingFields,
- ...referencedProjectStagingFailuresField,
- model: prepared.requestedModel || null,
- clearSession: clearSession || timedOut,
- resultJson: { phase: "turn" },
- summary: message,
- };
+ } finally {
+ // End the run root span exactly once, on every return and on a throw.
+ runRootSpan.end(runFailed);
}
};
}
diff --git a/packages/adapter-utils/src/acpx-engine/startup-timing.test.ts b/packages/adapter-utils/src/acpx-engine/startup-timing.test.ts
index f625909eb6aa..14c7e0bfe225 100644
--- a/packages/adapter-utils/src/acpx-engine/startup-timing.test.ts
+++ b/packages/adapter-utils/src/acpx-engine/startup-timing.test.ts
@@ -1,12 +1,17 @@
import { describe, expect, it, vi } from "vitest";
import type { AdapterRuntimeEvent } from "../types.js";
-import type { StartupSpan, StartupTracer } from "./startup-timing.js";
+import type { StartupSpan, StartupTraceContext, StartupTracer } from "./startup-timing.js";
import {
clampSpanLabel,
+ createRuntimeSpanRunner,
emitSkippedStartupStep,
getActiveStepContext,
measureStartupStep,
+ NOOP_STARTUP_SPAN,
+ NOOP_STARTUP_TRACE_CONTEXT,
normalizeProviderFamily,
+ runWithoutActiveStep,
+ runWithRuntimeParent,
SANDBOX_STARTUP_SPAN_ATTR_PREFIX,
SANDBOX_STARTUP_SPAN_ATTRS,
setSandboxRootSpanAttributes,
@@ -82,60 +87,9 @@ describe("measureStartupStep", () => {
expect(events[0]!.message).toBe("startup step: stage.sync (150ms)");
});
- it("includes the roundTrips delta in the payload when a round-trip reader is supplied", async () => {
+ it("emits only the high-level fields (step, durationMs, outcome) on the payload", async () => {
let t = 0;
const now = () => t;
- // Cumulative host→sandbox exec counter; the step performs 3 execs.
- let execCount = 5;
- const events: AdapterRuntimeEvent[] = [];
- const onEvent = vi.fn(async (event: AdapterRuntimeEvent) => {
- events.push(event);
- });
-
- await measureStartupStep({ onEvent }, now, "stage.sync", async () => {
- t = 90;
- execCount += 3;
- return "ok";
- }, { roundTrips: () => execCount });
-
- expect(events[0]!.payload).toMatchObject({
- step: "stage.sync",
- durationMs: 90,
- roundTrips: 3,
- });
- });
-
- it("reports roundTrips: 0 for a step that performs no execs (reader supplied)", async () => {
- const now = () => 0;
- const events: AdapterRuntimeEvent[] = [];
- const onEvent = vi.fn(async (event: AdapterRuntimeEvent) => {
- events.push(event);
- });
-
- await measureStartupStep({ onEvent }, now, "workspace.resolve", async () => "ok", {
- roundTrips: () => 7,
- });
-
- expect(events[0]!.payload).toMatchObject({ step: "workspace.resolve", roundTrips: 0 });
- });
-
- it("omits roundTrips from the payload when no reader is supplied", async () => {
- const now = () => 0;
- const events: AdapterRuntimeEvent[] = [];
- const onEvent = vi.fn(async (event: AdapterRuntimeEvent) => {
- events.push(event);
- });
-
- await measureStartupStep({ onEvent }, now, "workspace.resolve", async () => "ok");
-
- expect(events[0]!.payload).not.toHaveProperty("roundTrips");
- });
-
- it("accumulates provider exec/get durations and merges extra fields into the payload", async () => {
- let t = 0;
- const now = () => t;
- let execMs = 100;
- let getMs = 40;
const events: AdapterRuntimeEvent[] = [];
const onEvent = vi.fn(async (event: AdapterRuntimeEvent) => {
events.push(event);
@@ -143,23 +97,18 @@ describe("measureStartupStep", () => {
await measureStartupStep({ onEvent }, now, "acp.handshake", async () => {
t = 7000;
- execMs += 600; // one provider executeCommand round-trip
- getMs += 15; // one client.get re-fetch
return "handle";
- }, {
- providerExecMs: () => execMs,
- providerGetMs: () => getMs,
- extra: () => ({ createRuntimeMs: 12, ensureSessionMs: 6988 }),
});
- expect(events[0]!.payload).toMatchObject({
- step: "acp.handshake",
- durationMs: 7000,
- providerExecMs: 600,
- providerGetMs: 15,
- createRuntimeMs: 12,
- ensureSessionMs: 6988,
- });
+ // The detailed per-step round-trip and provider-duration numbers ride the
+ // OTel spans now, so the run-log payload keeps only the high-level fields.
+ const payload = events[0]!.payload as Record;
+ expect(Object.keys(payload).sort()).toEqual(["durationMs", "outcome", "step"]);
+ expect(payload).not.toHaveProperty("roundTrips");
+ expect(payload).not.toHaveProperty("providerExecMs");
+ expect(payload).not.toHaveProperty("providerGetMs");
+ expect(payload).not.toHaveProperty("createRuntimeMs");
+ expect(payload).not.toHaveProperty("ensureSessionMs");
});
it("returns the wrapped fn result unchanged", async () => {
@@ -282,64 +231,27 @@ describe("measureStartupStep", () => {
expect(spans[0]!.status?.code).toBe(2);
});
- it("keeps the roundTrips / providerExecMs / providerGetMs deltas on the payload but off the span", async () => {
- let t = 0;
- const now = () => t;
- let execCount = 5;
- let execMs = 100;
- let getMs = 40;
- const events: AdapterRuntimeEvent[] = [];
- const onEvent = vi.fn(async (event: AdapterRuntimeEvent) => {
- events.push(event);
- });
- const { tracer, spans } = makeMockTracer();
-
- await measureStartupStep({ onEvent }, now, "stage.sync", async () => {
- t = 90;
- execCount += 3;
- execMs += 600;
- getMs += 15;
- return "ok";
- }, {
- tracer,
- roundTrips: () => execCount,
- providerExecMs: () => execMs,
- providerGetMs: () => getMs,
- });
-
- // The counter deltas still ride the event payload.
- const payload = events[0]!.payload as Record;
- expect(payload.roundTrips).toBe(3);
- expect(payload.providerExecMs).toBe(600);
- expect(payload.providerGetMs).toBe(15);
- // The per-execution `sandbox.exec` spans now carry the round-trip detail, so
- // the step span no longer duplicates it.
- expect(spans[0]!.attributes).not.toHaveProperty(A.roundTripsCount);
- expect(spans[0]!.attributes).not.toHaveProperty(A.providerExecSumMs);
- expect(spans[0]!.attributes).not.toHaveProperty(A.providerGetSumMs);
- });
-
- it("sets no span attribute (and no payload field) when a reader returns undefined", async () => {
+ it("carries only the allowlisted attributes on the step span, never the removed round-trip detail", async () => {
const events: AdapterRuntimeEvent[] = [];
const onEvent = vi.fn(async (event: AdapterRuntimeEvent) => {
events.push(event);
});
const { tracer, spans } = makeMockTracer();
- await measureStartupStep({ onEvent }, () => 0, "workspace.resolve", async () => "ok", {
+ await measureStartupStep({ onEvent }, () => 0, "stage.sync", async () => "ok", {
tracer,
- // A reader may return undefined when the counter is unavailable. The guard
- // must omit the attribute rather than emit NaN or 0.
- roundTrips: () => undefined as unknown as number,
- providerExecMs: () => undefined as unknown as number,
+ provider: "daytona",
});
- expect(spans[0]!.attributes).not.toHaveProperty(A.roundTripsCount);
- expect(spans[0]!.attributes).not.toHaveProperty(A.providerExecSumMs);
- expect(Object.values(spans[0]!.attributes).some((v) => Number.isNaN(v))).toBe(false);
+ // The step span carries only the closed allowlist. The per-execution
+ // `sandbox.exec` spans carry the round-trip and provider-duration detail.
+ expect(Object.keys(spans[0]!.attributes).sort()).toEqual(
+ [A.provider, A.stepWallMs, A.outcome].sort(),
+ );
const payload = events[0]!.payload as Record;
expect(payload).not.toHaveProperty("roundTrips");
expect(payload).not.toHaveProperty("providerExecMs");
+ expect(payload).not.toHaveProperty("providerGetMs");
});
it("normalizes a plugin-backed provider key to plugin and keeps a built-in family as-is", async () => {
@@ -376,20 +288,11 @@ describe("measureStartupStep", () => {
await measureStartupStep({ onEvent }, () => 0, "acp.handshake", async () => "ok", {
tracer,
provider: "daytona",
- roundTrips: () => 3,
- providerExecMs: () => 600,
- providerGetMs: () => 15,
- // extra() carries caller-measured numbers into the EVENT payload only.
- // It must never widen the span-attribute set.
- extra: () => ({ createRuntimeMs: 12, ensureSessionMs: 6988 }),
});
expect(Object.keys(spans[0]!.attributes).sort()).toEqual(
[A.provider, A.stepWallMs, A.outcome].sort(),
);
- // extra() keys stay off the span.
- expect(spans[0]!.attributes).not.toHaveProperty("createRuntimeMs");
- expect(spans[0]!.attributes).not.toHaveProperty("ensureSessionMs");
// Every key uses the closed prefix, so no free-form command / path / id key
// can ride the span.
for (const key of Object.keys(spans[0]!.attributes)) {
@@ -411,9 +314,7 @@ describe("measureStartupStep", () => {
// No tracer supplied. The helper must still emit the event and return the
// value without throwing.
- const result = await measureStartupStep({ onEvent }, () => 0, "stage.sync", async () => "ok", {
- roundTrips: () => 3,
- });
+ const result = await measureStartupStep({ onEvent }, () => 0, "stage.sync", async () => "ok");
expect(result).toBe("ok");
expect(events[0]!.payload).toMatchObject({ step: "stage.sync" });
@@ -471,9 +372,6 @@ describe("measureStartupStep", () => {
await measureStartupStep({ onEvent: vi.fn(async () => {}) }, () => 0, "stage.sync", async () => "ok", {
tracer,
provider: "daytona",
- roundTrips: () => 3,
- providerExecMs: () => 600,
- providerGetMs: () => 15,
});
const keys = Object.keys(spans[0]!.attributes);
@@ -481,12 +379,9 @@ describe("measureStartupStep", () => {
for (const key of keys) {
expect(key.startsWith(SANDBOX_STARTUP_SPAN_ATTR_PREFIX)).toBe(true);
}
- // The step wall time and the counters use their type suffixes.
+ // The step wall time uses its type suffix.
expect(keys).toContain(A.stepWallMs);
expect(A.stepWallMs.endsWith(".wall_ms")).toBe(true);
- expect(A.roundTripsCount.endsWith(".count")).toBe(true);
- expect(A.providerExecSumMs.endsWith(".sum_ms")).toBe(true);
- expect(A.providerGetSumMs.endsWith(".sum_ms")).toBe(true);
});
});
@@ -605,6 +500,121 @@ describe("getActiveStepContext", () => {
}, { criticalPath: false });
expect(seen!.criticalPath).toBe(false);
});
+
+ it("keeps the ended step store in a timer scheduled inside the step body", async () => {
+ // Node snapshots the active store on each async resource at creation time.
+ // A timer scheduled inside a measured step body keeps that step store, even
+ // after the step span ends. A later run-time exec then reads the ended step
+ // and parents its `sandbox.exec` span to a dead startup step. This test locks
+ // that leak, so the fix and its guard test stay honest.
+ let storeInTimer: ReturnType = null;
+ let fireTimer!: () => void;
+ const timerRan = new Promise((resolve) => {
+ fireTimer = resolve;
+ });
+
+ await measureStartupStep(
+ { onEvent: vi.fn(async () => {}) },
+ () => 0,
+ "bridge.process-session",
+ async () => {
+ setTimeout(() => {
+ storeInTimer = getActiveStepContext();
+ fireTimer();
+ }, 0);
+ return "started";
+ },
+ { criticalPath: false },
+ );
+
+ // The step span already ended, so the main line reads no active step.
+ expect(getActiveStepContext()).toBeNull();
+
+ await timerRan;
+
+ // Yet the timer callback still reads the ended step context. This is the
+ // store leak the bridge boundary fix removes.
+ expect(storeInTimer).not.toBeNull();
+ expect(storeInTimer!.criticalPath).toBe(false);
+ });
+
+ it("clears the active step for a continuation wrapped in runWithoutActiveStep", async () => {
+ // The bridge boundary wraps its long-lived poll timer in `runWithoutActiveStep`.
+ // A timer scheduled inside that empty store scope reads no active step, so a
+ // later run-time exec opens an unparented span instead of one under the ended
+ // startup step.
+ let storeInTimer: ReturnType = null;
+ let sawTimer = false;
+ let fireTimer!: () => void;
+ const timerRan = new Promise((resolve) => {
+ fireTimer = resolve;
+ });
+
+ await measureStartupStep(
+ { onEvent: vi.fn(async () => {}) },
+ () => 0,
+ "bridge.process-session",
+ async () => {
+ runWithoutActiveStep(() => {
+ setTimeout(() => {
+ storeInTimer = getActiveStepContext();
+ sawTimer = true;
+ fireTimer();
+ }, 0);
+ });
+ return "started";
+ },
+ { criticalPath: false },
+ );
+
+ await timerRan;
+
+ expect(sawTimer).toBe(true);
+ expect(storeInTimer).toBeNull();
+ });
+
+ it("returns the work result and restores the previous active step", async () => {
+ // Outside any measured step the previous store is empty, so the helper both
+ // returns the work value and leaves the store empty afterward.
+ const value = runWithoutActiveStep(() => "value");
+ expect(value).toBe("value");
+ expect(getActiveStepContext()).toBeNull();
+ });
+});
+
+describe("runWithRuntimeParent", () => {
+ it("sets the parent context so inner code parents to the given token", () => {
+ // The server builds an opaque parent-context token from a run-time span. The
+ // helper publishes it, so inner code reads it through the getter and parents
+ // a child span to that token. Model the token as a plain object; the helper
+ // forwards it opaque.
+ const token = { span: "runtime-parent" };
+ const seen = runWithRuntimeParent(token, () => getActiveStepContext());
+
+ expect(seen).not.toBeNull();
+ // The published parent token is the given token, unchanged.
+ expect(seen!.parentContext).toBe(token);
+ // The helper stores the no-op span, not a real step span.
+ expect(seen!.span).toBe(NOOP_STARTUP_SPAN);
+ // A run-time exec is not on the startup critical path.
+ expect(seen!.criticalPath).toBe(false);
+ // The context clears once the work settles.
+ expect(getActiveStepContext()).toBeNull();
+ });
+
+ it("empties the store when the token is undefined, like runWithoutActiveStep", () => {
+ // A missing token means no run-time parent. The helper then empties the
+ // store, so inner code reads no active step and opens an unparented span.
+ const seen = runWithRuntimeParent(undefined, () => getActiveStepContext());
+
+ expect(seen).toBeNull();
+ expect(getActiveStepContext()).toBeNull();
+ });
+
+ it("returns the work result", () => {
+ const value = runWithRuntimeParent({ span: "runtime-parent" }, () => "value");
+ expect(value).toBe("value");
+ });
});
describe("clampSpanLabel", () => {
@@ -645,3 +655,60 @@ describe("clampSpanLabel", () => {
expect(clampSpanLabel("stdout", "secret output")).toBeUndefined();
});
});
+
+describe("createRuntimeSpanRunner", () => {
+ it("opens a named wrapper span and parents the wrapped work to it", async () => {
+ const { tracer, spans } = makeMockTracer();
+ const runParent = { marker: "run-parent" };
+ const traceContext: StartupTraceContext = {
+ tracer,
+ contextWithSpan: (span) => ({ span }),
+ };
+ const run = createRuntimeSpanRunner(traceContext, () => runParent);
+
+ let childParent: unknown = "unset";
+ const result = await run("sandbox.agentSession.sendInput", async () => {
+ childParent = getActiveStepContext()?.parentContext;
+ return "ok";
+ });
+
+ expect(result).toBe("ok");
+ expect(spans).toHaveLength(1);
+ expect(spans[0]!.name).toBe("sandbox.agentSession.sendInput");
+ expect(spans[0]!.endCount).toBe(1);
+ // The wrapped work parents to the wrapper span, not straight to the run
+ // parent, so the inner exec spans group under the wrapper span.
+ expect((childParent as { span?: unknown }).span).toBe(spans[0]);
+ });
+
+ it("marks the wrapper span failed and ends it when the work throws", async () => {
+ const { tracer, spans } = makeMockTracer();
+ const traceContext: StartupTraceContext = {
+ tracer,
+ contextWithSpan: (span) => ({ span }),
+ };
+ const run = createRuntimeSpanRunner(traceContext, () => ({ marker: "run-parent" }));
+
+ await expect(
+ run("sandbox.callbackBridge.relayRequest", async () => {
+ throw new Error("boom");
+ }),
+ ).rejects.toThrow(/boom/);
+ expect(spans).toHaveLength(1);
+ expect(spans[0]!.status?.code).toBe(2);
+ expect(spans[0]!.endCount).toBe(1);
+ });
+
+ it("runs the work unwrapped under a no-op trace context", async () => {
+ const run = createRuntimeSpanRunner(NOOP_STARTUP_TRACE_CONTEXT, () => ({ marker: "run-parent" }));
+ // The no-op `contextWithSpan` yields no child token, so the store empties for
+ // the work, exactly like the earlier unparented run-time behavior.
+ let childStep: unknown = "unset";
+ const result = await run("sandbox.agentSession.pollOutput", async () => {
+ childStep = getActiveStepContext();
+ return 7;
+ });
+ expect(result).toBe(7);
+ expect(childStep).toBeNull();
+ });
+});
diff --git a/packages/adapter-utils/src/acpx-engine/startup-timing.ts b/packages/adapter-utils/src/acpx-engine/startup-timing.ts
index ada1877feebb..ce91ca6a5bd6 100644
--- a/packages/adapter-utils/src/acpx-engine/startup-timing.ts
+++ b/packages/adapter-utils/src/acpx-engine/startup-timing.ts
@@ -53,7 +53,6 @@ export const SANDBOX_STARTUP_SPAN_ATTR_PREFIX = "paperclip.sandbox.startup.";
* the `paperclip.sandbox.startup.` prefix and a type suffix:
*
* - `*.wall_ms` — one wall-clock time in float milliseconds.
- * - `*.sum_ms` — a sum of wall-clock times in float milliseconds.
* - `*.count` — a count.
*
* The producer sets only these keys. It never sets a free-form key, so a
@@ -66,12 +65,6 @@ export const SANDBOX_STARTUP_SPAN_ATTRS = {
outcome: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}outcome`,
/** The wall-clock time of one measured step. */
stepWallMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}step.wall_ms`,
- /** The number of host-to-sandbox round trips a step made. */
- roundTripsCount: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}round_trips.count`,
- /** The sum of provider `executeCommand` wall time a step made. */
- providerExecSumMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}provider_exec.sum_ms`,
- /** The sum of provider handle-refetch wall time a step made. */
- providerGetSumMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}provider_get.sum_ms`,
/** The clamped `argv[0]` command label of one execution. */
execCommand: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}exec.command`,
/** The numeric process exit code of one execution. */
@@ -86,6 +79,8 @@ export const SANDBOX_STARTUP_SPAN_ATTRS = {
execNetworkMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}exec.network_ms`,
/** Whether one execution sits on the startup critical path. */
execCriticalPath: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}exec.critical_path`,
+ /** Whether the provider served the sandbox handle from its warm cache. */
+ execCacheHit: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}exec.cache_hit`,
/** The root-span wall time of the whole bring-up. */
rootWallMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}root.wall_ms`,
/** The sum of the step wall times of the whole bring-up. */
@@ -108,6 +103,12 @@ export const SANDBOX_STARTUP_SPAN_ATTRS = {
handshakeEnsureSessionWallMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}handshake.ensure_session.wall_ms`,
/** A shared low-cardinality tag that marks two steps as one parallel batch. */
batch: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}batch`,
+ /** The host-local wall time of the pack step (build the tarball). */
+ packWallMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}pack.wall_ms`,
+ /** The wall time of the transfer step (upload the files to the sandbox). */
+ transferWallMs: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}transfer.wall_ms`,
+ /** The number of serial guard round trips before one transfer. */
+ transferGuardCount: `${SANDBOX_STARTUP_SPAN_ATTR_PREFIX}transfer.guard.count`,
} as const;
/** The closed value set for the `outcome` attribute. */
@@ -331,7 +332,7 @@ export interface ActiveStepContext {
* a module-level singleton, so the value propagates across `await` boundaries
* and across package boundaries that share this module.
*/
-const activeStepContextStorage = new AsyncLocalStorage();
+const activeStepContextStorage = new AsyncLocalStorage();
/**
* Return the active step context, or `null` when no measured step is running.
@@ -343,6 +344,124 @@ export function getActiveStepContext(): ActiveStepContext | null {
return activeStepContextStorage.getStore() ?? null;
}
+/**
+ * Run `work` with no active step context, then restore the previous store. A
+ * bridge boundary uses this to start its long-lived poll timer and socket
+ * handlers outside the measured step store.
+ *
+ * Node snapshots the active store on each async resource at creation time. So a
+ * timer or a handler scheduled inside a measured step body keeps that step store
+ * after the step span ends. A later run-time exec then reads the ended step and
+ * parents its `sandbox.exec` span to a dead startup step, and it copies the
+ * step's `criticalPath` flag. This helper resets the store for the wrapped work,
+ * so each continuation reads an empty store. Each run-time exec then opens an
+ * unparented span with no stale `criticalPath` flag.
+ *
+ * The helper forwards only the opaque store, so this package stays free of
+ * `@opentelemetry/api`. It needs no Node version gate.
+ */
+export function runWithoutActiveStep(work: () => T): T {
+ return activeStepContextStorage.run(undefined, work);
+}
+
+/**
+ * Build the minimal active step context that the store publishes. The step path
+ * and `runWithRuntimeParent` share this builder, so both write the same shape.
+ */
+function buildActiveStepContext(
+ span: StartupSpan,
+ parentContext: StartupSpanContext,
+ criticalPath: boolean,
+): ActiveStepContext {
+ return { span, parentContext, criticalPath };
+}
+
+/**
+ * Run `work` under a given parent-context token, then restore the previous
+ * store. A run-time exec that reads `getActiveStepContext()` inside `work`
+ * parents its span to `parentContext`, not to a startup step. The store carries
+ * the no-op span, because there is no open step span at run time. It sets
+ * `criticalPath` to `false`, because a run-time exec is not on the startup
+ * critical path.
+ *
+ * When `parentContext` is `undefined`, the helper empties the store, exactly
+ * like `runWithoutActiveStep`. Inner code then reads `null` and opens an
+ * unparented span.
+ *
+ * The helper forwards only the opaque token, so this package stays free of
+ * `@opentelemetry/api`.
+ */
+export function runWithRuntimeParent(
+ parentContext: StartupSpanContext,
+ work: () => T,
+): T {
+ if (parentContext === undefined) {
+ return activeStepContextStorage.run(undefined, work);
+ }
+ const activeStep = buildActiveStepContext(NOOP_SPAN, parentContext, false);
+ return activeStepContextStorage.run(activeStep, work);
+}
+
+/**
+ * Run one run-time operation inside its own wrapper span. The runner opens a
+ * wrapper span parented to the current run span, publishes the wrapper span as
+ * the runtime parent while `work` runs, and ends the span when `work` settles.
+ * A child `sandbox.exec` span inside `work` parents to the wrapper span, so the
+ * trace groups the operation's execs under one named span. A throwing `work`
+ * sets the wrapper span error status before the span ends.
+ *
+ * The runner reads the run parent per call, so it always parents to the live
+ * span (`agent.turn` during the turn, `task.run` otherwise). The default runner
+ * opens no real span; it only runs `work` under the current run parent, so the
+ * span path stays a no-op until the server injects a real tracer.
+ */
+export type RuntimeSpanRunner = (name: string, work: () => Promise) => Promise;
+
+/**
+ * Build a {@link RuntimeSpanRunner} from a trace context and the run-parent
+ * getter. The runner opens the wrapper span through `traceContext.tracer`, and
+ * it derives the wrapper span's child parent token through
+ * `traceContext.contextWithSpan`. A no-op trace context yields a runner that
+ * opens no real span and runs `work` under the current run parent, so the span
+ * path stays inert until the server injects a real tracer. Every tracer call
+ * sits inside an error swallow, so a throwing tracer never changes control flow.
+ */
+export function createRuntimeSpanRunner(
+ traceContext: StartupTraceContext,
+ getRuntimeParentContext: () => StartupSpanContext | undefined,
+): RuntimeSpanRunner {
+ return async (name: string, work: () => Promise): Promise => {
+ const parentContext = getRuntimeParentContext();
+ let span: StartupSpan;
+ try {
+ span = traceContext.tracer.startSpan(name, undefined, parentContext);
+ } catch {
+ // A throwing tracer must not change control flow; run `work` unwrapped.
+ return runWithRuntimeParent(parentContext, work);
+ }
+ let childContext: StartupSpanContext;
+ try {
+ childContext = traceContext.contextWithSpan(span);
+ } catch {
+ childContext = parentContext;
+ }
+ let failed = false;
+ try {
+ return await runWithRuntimeParent(childContext, work);
+ } catch (err) {
+ failed = true;
+ throw err;
+ } finally {
+ try {
+ if (failed) span.setStatus({ code: SPAN_STATUS_CODE_ERROR });
+ span.end();
+ } catch {
+ // Observability must not change control flow.
+ }
+ }
+ };
+}
+
/**
* Set a numeric span attribute only when the value is a finite number. A reader
* that returns `undefined` (the counter is unavailable) yields no attribute,
@@ -360,47 +479,20 @@ function setFiniteNumberAttr(
}
/**
- * Compute a counter delta from a reader. Return `undefined` when the reader is
- * absent, or when either the start or the end snapshot is not a finite number.
- * A `undefined` result yields no payload field and no span attribute.
- */
-function finiteDelta(
- read: (() => number) | undefined,
- start: number | undefined,
-): number | undefined {
- if (!read) return undefined;
- const end = read();
- if (typeof end !== "number" || !Number.isFinite(end)) return undefined;
- const base = typeof start === "number" && Number.isFinite(start) ? start : 0;
- return end - base;
-}
-
-/**
- * Optional per-step attribution attached to a `run.startup.step` event, all
- * additive to the free-form jsonb payload (no schema change). Each reader is a
- * plain `() => number` closure so the timing helper stays decoupled from the
- * runner/provider it reads (Risk R1): `measureStartupStep` snapshots the reader
- * before `fn` and again in `finally`, emitting the delta.
+ * Optional per-step attribution for a `run.startup.step` event and its span.
+ * The event payload carries only the high-level fields (`step`, `durationMs`,
+ * `outcome`). The detailed per-step round-trip and provider-duration numbers
+ * ride the OTel spans (the per-execution `sandbox.exec` child spans and the
+ * step span), not the payload. These options configure the span path and the
+ * step context.
*
- * - `roundTrips` — cumulative host→sandbox `runner.execute` count; the delta is
- * how many round-trips the step performed (Open Q1, host boundary).
- * - `providerExecMs` / `providerGetMs` — cumulative provider-reported wall-time
- * (ms) for the `executeCommand` REST call vs the `client.get` sandbox
- * re-fetch; the delta attributes the step's round-trip time to its parts
- * (Open Q1, finer provider attribution).
- * - `extra` — a reader (called once in `finally`, after `fn` settles) returning
- * any additional numeric fields to merge into the payload; used by
- * `acp.handshake` to carry its `createRuntimeMs` / `ensureSessionMs` sub-split
- * (Open Q2), which are measured by the caller rather than read from a counter.
- * The `extra` map feeds the EVENT payload only. Its keys never become span
- * attributes, so a free-form key cannot widen the closed span allowlist.
* - `tracer` — an injected structural tracer. It defaults to a no-op, so the
* span path changes no runtime behavior until the server injects a real
* tracer. The span carries only the closed attribute allowlist from
* `SANDBOX_STARTUP_SPAN_ATTRS`: the normalized `provider`, the step wall time,
* and the outcome. The step name rides the span name, not an attribute. The
- * round-trip and provider-duration detail stays on the payload and on the
- * per-execution `sandbox.exec` child spans.
+ * round-trip and provider-duration detail rides the per-execution
+ * `sandbox.exec` child spans.
* - `parentContext` — an opaque parent-context token from the root span. When
* set, the step's span parents to that root. `measureStartupStep` forwards it
* to `startSpan` and never inspects it, so parenting stays explicit and does
@@ -410,10 +502,6 @@ function finiteDelta(
* low-cardinality `provider` span attribute. It never sets the raw key.
*/
export interface StartupStepMeasureOptions {
- roundTrips?: () => number;
- providerExecMs?: () => number;
- providerGetMs?: () => number;
- extra?: () => Record;
tracer?: StartupTracer;
parentContext?: StartupSpanContext;
provider?: string;
@@ -469,8 +557,8 @@ function buildStepEvent(payload: Record): AdapterRuntimeEvent {
/**
* Time `fn` with the injected `now` clock and emit exactly one
- * `run.startup.step` event carrying `{ step, durationMs }` plus any counters
- * supplied via `options`. The event fires in a `finally`, so a throwing step
+ * `run.startup.step` event carrying only the high-level `{ step, durationMs,
+ * outcome }`. The event fires in a `finally`, so a throwing step
* still reports its duration before the error is re-thrown. `now` is injected
* (never `Date.now()` here) so callers/tests stay deterministic, and
* `ctx.onEvent` is optional — a missing sink is a no-op that neither throws nor
@@ -482,7 +570,8 @@ function buildStepEvent(payload: Record): AdapterRuntimeEvent {
* from `SANDBOX_STARTUP_SPAN_ATTRS`: the normalized `provider`, the step wall
* time, and the outcome (`ok` or `failed`). The step name rides the span name.
* A throwing `fn` sets the span error status before the span ends and the
- * outcome is `failed`. The counter deltas stay on the event payload. The tracer
+ * outcome is `failed`. The round-trip and provider-duration detail rides the
+ * spans, not the payload. The tracer
* defaults to a no-op, so a caller with no tracer changes nothing. Every span
* call sits inside the same error swallow as the event sink, so a throwing
* tracer never changes startup control flow.
@@ -495,9 +584,6 @@ export async function measureStartupStep(
options: StartupStepMeasureOptions = {},
): Promise {
const start = now();
- const roundTripsStart = options.roundTrips?.();
- const providerExecStart = options.providerExecMs?.();
- const providerGetStart = options.providerGetMs?.();
// Open the span with only the low-cardinality allowlisted attributes known at
// the start: the normalized provider family. The span name already carries
@@ -531,11 +617,11 @@ export async function measureStartupStep(
} catch {
stepChildContext = undefined;
}
- const activeStep: ActiveStepContext = {
+ const activeStep = buildActiveStepContext(
span,
- parentContext: stepChildContext,
- criticalPath: options.criticalPath ?? true,
- };
+ stepChildContext,
+ options.criticalPath ?? true,
+ );
let stepFailed = false;
try {
@@ -546,28 +632,16 @@ export async function measureStartupStep(
} finally {
const durationMs = now() - start;
- // One attribute-build block feeds both the event payload and the span, so
- // the two paths never drift. `undefined` deltas produce neither a payload
- // field nor a span attribute (fail open — never `NaN`, never `0`).
- const roundTrips = finiteDelta(options.roundTrips, roundTripsStart);
- const providerExecMs = finiteDelta(options.providerExecMs, providerExecStart);
- const providerGetMs = finiteDelta(options.providerGetMs, providerGetStart);
-
// The step outcome. A throwing `fn` is `failed`; a settled `fn` is `ok`. A
// step that a warm cache skips uses `emitSkippedStartupStep` instead.
const outcome: SandboxStartupOutcome = stepFailed
? SANDBOX_STARTUP_OUTCOME.failed
: SANDBOX_STARTUP_OUTCOME.ok;
+ // The payload carries only the high-level fields. The detailed per-step
+ // round-trip and provider-duration numbers ride the OTel spans now, so the
+ // run-log copy is gone.
const payload: Record = { step, durationMs, outcome };
- if (roundTrips !== undefined) payload.roundTrips = roundTrips;
- if (providerExecMs !== undefined) payload.providerExecMs = providerExecMs;
- if (providerGetMs !== undefined) payload.providerGetMs = providerGetMs;
- if (options.extra) {
- // `extra` feeds the EVENT payload only. Its keys never become span
- // attributes, so it cannot widen the closed span allowlist.
- Object.assign(payload, options.extra());
- }
try {
if (stepFailed) span.setStatus({ code: SPAN_STATUS_CODE_ERROR });
diff --git a/packages/adapter-utils/src/command-managed-runtime.ts b/packages/adapter-utils/src/command-managed-runtime.ts
index fc9b0f0cc856..32e459a21fb0 100644
--- a/packages/adapter-utils/src/command-managed-runtime.ts
+++ b/packages/adapter-utils/src/command-managed-runtime.ts
@@ -16,6 +16,7 @@ import {
import { preferredShellForSandbox, shellCommandArgs } from "./sandbox-shell.js";
import type { RunProcessResult } from "./server-utils.js";
import type { RuntimeProgressSink, RuntimeStatusSink } from "./runtime-progress.js";
+import type { RuntimeSpanRunner } from "./acpx-engine/startup-timing.js";
export interface CommandManagedRuntimeRunner {
/**
@@ -25,25 +26,6 @@ export interface CommandManagedRuntimeRunner {
* and let the caller choose a chunked upload path when progress is requested.
*/
supportsSingleStreamStdinProgress?: boolean;
- /**
- * Cumulative count of host→sandbox `execute` round-trips this runner has
- * performed (Open Q1). Present only on runners that instrument the single
- * exec seam (the sandbox runner); the per-step delta is emitted as
- * `run.startup.step` `payload.roundTrips`. A `() => number` reader, never the
- * runner itself, is threaded into `measureStartupStep` so the timing helper
- * stays runner-agnostic.
- */
- execCount?(): number;
- /**
- * Cumulative provider-reported wall-time (ms) for the `executeCommand` REST
- * call ({@link providerExecMs}) vs the `client.get` sandbox re-fetch that
- * precedes it ({@link providerGetMs}), accumulated across every `execute`
- * round-trip (Open Q1, finer attribution). Present only when the provider
- * surfaces these durations on its result metadata; the per-step deltas are
- * emitted as `payload.providerExecMs` / `payload.providerGetMs`.
- */
- providerExecMs?(): number;
- providerGetMs?(): number;
execute(input: {
command: string;
args?: string[];
@@ -53,6 +35,27 @@ export interface CommandManagedRuntimeRunner {
timeoutMs?: number;
onLog?: (stream: "stdout" | "stderr", chunk: string) => Promise;
onSpawn?: (meta: { pid: number; startedAt: string }) => Promise;
+ /**
+ * Run this command through the lease's persistent session even when no run
+ * step is active. A sandbox provider opens the session on the first
+ * non-bypassed command; the ACP process session bridge sets this so the
+ * long-lived agent command streams its output through the session log
+ * stream. The default keeps the context-based session selection.
+ */
+ useSession?: boolean;
+ /**
+ * Run this command outside the lease's persistent session even when a run
+ * step is active. The persistent session is a single serialized shell. In
+ * streamed mode the agent runs as one long-lived foreground command that
+ * holds the session for the whole run. The bridge control-plane execs
+ * (input delivery, output read, callback relay, and the queue/setup
+ * bookkeeping) must run concurrently with the agent, so they run as
+ * independent one-shot commands. On the session they queue behind the agent
+ * command that never returns — a permanent deadlock. An explicit bypass
+ * always wins over the context-based session selection and over
+ * `useSession`. The default keeps the context-based session selection.
+ */
+ bypassSession?: boolean;
}): Promise;
/**
* Optional native inbound file transfer. Present only when the sandbox
@@ -354,8 +357,7 @@ export function createCommandManagedRuntimeClient(input: {
// replace untar for directories, direct `writeFile` for single files), then run
// the operation's ordered `postUploadCommands` fail-fast. Byte-for-byte
// behavior-equivalent to the caller-inlined tar path it will replace. All exec
- // rides the shared `execute` seam so `execCount`/`providerExecMs` still
- // attribute (Open Q1).
+ // rides the shared `execute` seam.
const fallbackSyncIn = async (operations: SandboxSyncOperation[]): Promise => {
const resultOperations: SandboxSyncResult["operations"] = [];
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "paperclip-syncin-fallback-"));
@@ -477,6 +479,10 @@ export async function prepareCommandManagedRuntime(input: {
// task wires it into the byte-counting writeFile/readFile transport.
onProgress?: RuntimeProgressSink;
onRuntimeProgress?: RuntimeStatusSink;
+ // Optional host span runner for the workspace tarball build. Forwarded to
+ // prepareSandboxManagedRuntime so the host pack time rides one `pack` span
+ // under the `stage.sync` step. The default is a no-op.
+ runtimeSpan?: RuntimeSpanRunner;
}): Promise {
const timeoutMs = input.spec.timeoutMs && input.spec.timeoutMs > 0 ? input.spec.timeoutMs : 300_000;
const workspaceRemoteDir = input.workspaceRemoteDir ?? input.spec.remoteCwd;
@@ -528,6 +534,7 @@ export async function prepareCommandManagedRuntime(input: {
additionalSources: input.additionalSources,
onProgress: input.onProgress,
onRuntimeProgress: input.onRuntimeProgress,
+ runtimeSpan: input.runtimeSpan,
});
}
}
@@ -566,5 +573,6 @@ export async function prepareCommandManagedRuntime(input: {
additionalSources: input.additionalSources,
onProgress: input.onProgress,
onRuntimeProgress: input.onRuntimeProgress,
+ runtimeSpan: input.runtimeSpan,
});
}
diff --git a/packages/adapter-utils/src/env-bindings.test.ts b/packages/adapter-utils/src/env-bindings.test.ts
new file mode 100644
index 000000000000..a3d1dd78d3b4
--- /dev/null
+++ b/packages/adapter-utils/src/env-bindings.test.ts
@@ -0,0 +1,82 @@
+import { describe, expect, it } from "vitest";
+import { buildAdapterEnvConfig, parseEnvBindings, parseEnvVars } from "./env-bindings.js";
+
+describe("parseEnvBindings", () => {
+ it("keeps plain, company secret, and user-scoped bindings", () => {
+ const env = parseEnvBindings({
+ PLAIN: { type: "plain", value: "on" },
+ LEGACY_STRING: "raw",
+ COMPANY: { type: "secret_ref", secretId: "11111111-1111-1111-1111-111111111111", version: "latest" },
+ USER: { type: "user_secret_ref", key: "github_token", version: "latest", required: true },
+ });
+
+ expect(env).toEqual({
+ PLAIN: { type: "plain", value: "on" },
+ LEGACY_STRING: { type: "plain", value: "raw" },
+ COMPANY: { type: "secret_ref", secretId: "11111111-1111-1111-1111-111111111111", version: "latest" },
+ USER: { type: "user_secret_ref", key: "github_token", version: "latest", required: true },
+ });
+ });
+
+ it("keeps a user-scoped binding without an optional version or required flag", () => {
+ const env = parseEnvBindings({
+ USER: { type: "user_secret_ref", key: "github_token" },
+ });
+
+ expect(env).toEqual({ USER: { type: "user_secret_ref", key: "github_token" } });
+ });
+
+ it("preserves allowMissingOverride when present", () => {
+ const env = parseEnvBindings({
+ USER: { type: "user_secret_ref", key: "k", allowMissingOverride: true },
+ });
+
+ expect(env.USER).toEqual({ type: "user_secret_ref", key: "k", allowMissingOverride: true });
+ });
+
+ it("drops invalid keys, unknown shapes, and incomplete refs", () => {
+ const env = parseEnvBindings({
+ "1BAD": { type: "plain", value: "x" },
+ MISSING_KEY: { type: "user_secret_ref" },
+ MISSING_ID: { type: "secret_ref" },
+ UNKNOWN: { type: "mystery", value: "y" },
+ });
+
+ expect(env).toEqual({});
+ });
+
+ it("returns an empty map for non-object input", () => {
+ expect(parseEnvBindings(null)).toEqual({});
+ expect(parseEnvBindings([])).toEqual({});
+ expect(parseEnvBindings("nope")).toEqual({});
+ });
+});
+
+describe("parseEnvVars", () => {
+ it("reads KEY=value lines and skips blanks, comments, and invalid keys", () => {
+ // The key is trimmed; the value keeps every character after the first "=".
+ const env = parseEnvVars("A=1\n# comment\n\nB = two = three\n1BAD=x\nNOEQ");
+
+ expect(env).toEqual({ A: "1", B: " two = three" });
+ });
+});
+
+describe("buildAdapterEnvConfig", () => {
+ it("lets a structured binding win over a legacy plain-text entry with the same key", () => {
+ const env = buildAdapterEnvConfig(
+ { SHARED: { type: "user_secret_ref", key: "token" } },
+ "SHARED=legacy\nONLY_LEGACY=x",
+ );
+
+ expect(env).toEqual({
+ SHARED: { type: "user_secret_ref", key: "token" },
+ ONLY_LEGACY: { type: "plain", value: "x" },
+ });
+ });
+
+ it("tolerates missing legacy text", () => {
+ expect(buildAdapterEnvConfig({ A: { type: "plain", value: "1" } }, undefined)).toEqual({
+ A: { type: "plain", value: "1" },
+ });
+ });
+});
diff --git a/packages/adapter-utils/src/env-bindings.ts b/packages/adapter-utils/src/env-bindings.ts
new file mode 100644
index 000000000000..efdd57b85871
--- /dev/null
+++ b/packages/adapter-utils/src/env-bindings.ts
@@ -0,0 +1,99 @@
+// Shared parser for the agent configuration form environment fields.
+//
+// The form supplies two sources for adapter `env`:
+// - `envBindings`: structured bindings keyed by variable name. A binding is a
+// plain value, a company `secret_ref`, or a user-scoped `user_secret_ref`.
+// - `envVars`: a legacy plain-text block of `KEY=value` lines.
+//
+// The server resolves every binding to a string, both at run time and at test
+// time. This parser must keep each binding shape intact so the resolver gets
+// it. A dropped binding makes the agent "Test" action miss a variable that a
+// real run receives. So this parser keeps all three binding types. It never
+// resolves a secret and never reads a secret value; it only preserves the
+// binding shape for the server resolver.
+
+const ENV_KEY_RE = /^[A-Za-z_][A-Za-z0-9_]*$/;
+
+function isVersionSelector(value: unknown): value is number | "latest" {
+ return typeof value === "number" || value === "latest";
+}
+
+/**
+ * Convert the structured `envBindings` map into an adapter `env` map. Keep
+ * `plain`, `secret_ref`, and `user_secret_ref` bindings. Drop entries with an
+ * invalid variable name or an unknown binding shape.
+ */
+export function parseEnvBindings(bindings: unknown): Record {
+ if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
+ const env: Record = {};
+ for (const [key, raw] of Object.entries(bindings)) {
+ if (!ENV_KEY_RE.test(key)) continue;
+ if (typeof raw === "string") {
+ env[key] = { type: "plain", value: raw };
+ continue;
+ }
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
+ const rec = raw as Record;
+ if (rec.type === "plain" && typeof rec.value === "string") {
+ env[key] = { type: "plain", value: rec.value };
+ continue;
+ }
+ if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
+ env[key] = {
+ type: "secret_ref",
+ secretId: rec.secretId,
+ ...(isVersionSelector(rec.version) ? { version: rec.version } : {}),
+ };
+ continue;
+ }
+ if (rec.type === "user_secret_ref" && typeof rec.key === "string") {
+ env[key] = {
+ type: "user_secret_ref",
+ key: rec.key,
+ ...(isVersionSelector(rec.version) ? { version: rec.version } : {}),
+ ...(typeof rec.required === "boolean" ? { required: rec.required } : {}),
+ ...(typeof rec.allowMissingOverride === "boolean"
+ ? { allowMissingOverride: rec.allowMissingOverride }
+ : {}),
+ };
+ }
+ }
+ return env;
+}
+
+/**
+ * Parse the legacy plain-text `KEY=value` block. Skip blank lines, comment
+ * lines, and lines with an invalid variable name.
+ */
+export function parseEnvVars(text: string): Record {
+ const env: Record = {};
+ for (const line of text.split(/\r?\n/)) {
+ const trimmed = line.trim();
+ if (!trimmed || trimmed.startsWith("#")) continue;
+ const eq = trimmed.indexOf("=");
+ if (eq <= 0) continue;
+ const key = trimmed.slice(0, eq).trim();
+ const value = trimmed.slice(eq + 1);
+ if (!ENV_KEY_RE.test(key)) continue;
+ env[key] = value;
+ }
+ return env;
+}
+
+/**
+ * Build the adapter `env` map from both form sources. A structured binding wins
+ * over a legacy plain-text entry with the same key.
+ */
+export function buildAdapterEnvConfig(
+ envBindings: unknown,
+ envVars: string | undefined | null,
+): Record {
+ const env = parseEnvBindings(envBindings);
+ const legacy = parseEnvVars(envVars ?? "");
+ for (const [key, value] of Object.entries(legacy)) {
+ if (!Object.prototype.hasOwnProperty.call(env, key)) {
+ env[key] = { type: "plain", value };
+ }
+ }
+ return env;
+}
diff --git a/packages/adapter-utils/src/execution-target-sandbox.test.ts b/packages/adapter-utils/src/execution-target-sandbox.test.ts
index 8f939c654020..db7eeeda2062 100644
--- a/packages/adapter-utils/src/execution-target-sandbox.test.ts
+++ b/packages/adapter-utils/src/execution-target-sandbox.test.ts
@@ -23,6 +23,7 @@ import {
startAdapterExecutionTargetPaperclipBridge,
type AdapterSandboxExecutionTarget,
} from "./execution-target.js";
+import { getActiveStepContext } from "./acpx-engine/startup-timing.js";
import { createSandboxRunLogTailFactory } from "./sandbox-run-log-stream.js";
import { runChildProcess } from "./server-utils.js";
import { shellQuote } from "./ssh.js";
@@ -248,6 +249,363 @@ describe("sandbox adapter execution targets", () => {
}
});
+ it("test_process_session_poll_exec_parents_to_run_context", async () => {
+ // The poll timer runs run-time execs for the whole run. Its `sandbox.exec`
+ // span must parent to the live run span, not to the ended startup step. The
+ // bridge reads `getRuntimeParentContext` per tick and runs the poll under
+ // that token. This test drives the bridge with a getter that returns a known
+ // token, lets the first poll tick fire, and proves the poll exec reads that
+ // token from the active step store.
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-poll-parent-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "noop-acp-child.mjs");
+ await writeFile(childPath, "process.stdin.on('data', () => {});\n", "utf8");
+
+ const runParentToken = { marker: "process-session-run-parent" };
+ let bridgeStarted = false;
+ let pollStep: ReturnType | "unset" = "unset";
+ let resolvePoll: () => void = () => {};
+ const pollObserved = new Promise((resolve) => {
+ resolvePoll = resolve;
+ });
+
+ const delegate = createLocalSandboxRunner();
+ const runner = {
+ execute: async (input: Parameters[0]) => {
+ // Record the active step for the first exec that runs after the bridge
+ // start resolves. The setup execs run during the measured start; the
+ // poll timer fires later, under the run parent context.
+ if (bridgeStarted && pollStep === "unset") {
+ pollStep = getActiveStepContext();
+ resolvePoll();
+ }
+ return delegate.execute(input);
+ },
+ };
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner,
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-process-session-poll-parent",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ getRuntimeParentContext: () => runParentToken,
+ });
+ expect(bridge).not.toBeNull();
+ bridgeStarted = true;
+
+ try {
+ await pollObserved;
+ // The poll exec ran under the run parent context, so its exec span parents
+ // to the run token, not to a detached root or an ended startup step.
+ expect(pollStep).not.toBe("unset");
+ expect(pollStep).not.toBeNull();
+ expect((pollStep as { parentContext?: unknown }).parentContext).toBe(runParentToken);
+ expect((pollStep as { criticalPath?: boolean }).criticalPath).toBe(false);
+ } finally {
+ await bridge?.stop();
+ }
+ });
+
+ it("test_process_session_poll_exec_stays_unparented_without_getter", async () => {
+ // With no `getRuntimeParentContext`, the poll tick runs with an empty active
+ // step store, exactly like the earlier `runWithoutActiveStep` behavior. So a
+ // poll `sandbox.exec` span opens unparented with no stale startup flag.
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-poll-nogetter-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "noop-acp-child.mjs");
+ await writeFile(childPath, "process.stdin.on('data', () => {});\n", "utf8");
+
+ let bridgeStarted = false;
+ let pollStep: ReturnType | "unset" = "unset";
+ let resolvePoll: () => void = () => {};
+ const pollObserved = new Promise((resolve) => {
+ resolvePoll = resolve;
+ });
+
+ const delegate = createLocalSandboxRunner();
+ const runner = {
+ execute: async (input: Parameters[0]) => {
+ if (bridgeStarted && pollStep === "unset") {
+ pollStep = getActiveStepContext();
+ resolvePoll();
+ }
+ return delegate.execute(input);
+ },
+ };
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner,
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-process-session-poll-nogetter",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ });
+ expect(bridge).not.toBeNull();
+ bridgeStarted = true;
+
+ try {
+ await pollObserved;
+ expect(pollStep).toBeNull();
+ } finally {
+ await bridge?.stop();
+ }
+ });
+
+ it("test_process_session_stdin_exec_reads_send_time_run_parent", async () => {
+ // A persistent socket can open under one run parent and receive stdin later,
+ // under a different parent. The stdin-write `sandbox.exec` span must parent
+ // to the parent that is live at send time, not to the parent that was live
+ // when the socket opened. The bridge reads `getRuntimeParentContext` per
+ // message in the `data` handler, not once at connect time. This test opens a
+ // socket while `connectParent` is live, switches the getter to `turnParent`,
+ // sends one stdin line, and proves the stdin write ran under `turnParent`.
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-stdin-parent-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "noop-acp-child.mjs");
+ await writeFile(childPath, "process.stdin.on('data', () => {});\n", "utf8");
+
+ const connectParent = { marker: "process-session-connect-parent" };
+ const turnParent = { marker: "process-session-turn-parent" };
+ let currentParent: unknown = connectParent;
+
+ let stdinWriteStep: ReturnType | "unset" = "unset";
+ let resolveStdinWrite: () => void = () => {};
+ const stdinWriteObserved = new Promise((resolve) => {
+ resolveStdinWrite = resolve;
+ });
+
+ const delegate = createLocalSandboxRunner();
+ const runner = {
+ execute: async (input: Parameters[0]) => {
+ // Record the active step for the first exec that writes the stdin file.
+ // The `.paperclip-upload` temp path under the `stdin` directory is unique
+ // to the stdin-write path; the poll loop reads the `events` directory.
+ const script = (input.args ?? []).join("\n");
+ if (stdinWriteStep === "unset" && /\/stdin\/[^\s']*paperclip-upload/.test(script)) {
+ stdinWriteStep = getActiveStepContext();
+ resolveStdinWrite();
+ }
+ return delegate.execute(input);
+ },
+ };
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner,
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-process-session-stdin-parent",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ getRuntimeParentContext: () => currentParent as never,
+ });
+ expect(bridge).not.toBeNull();
+
+ let peer: net.Socket | null = null;
+ try {
+ const proxySource = await readFile(bridge!.agentCommand, "utf8");
+ const port = Number(/port: (\d+)/.exec(proxySource)?.[1] ?? Number.NaN);
+ const tokenLiteral = /const token = (".*?");/.exec(proxySource)?.[1];
+ expect(Number.isFinite(port)).toBe(true);
+ expect(typeof tokenLiteral).toBe("string");
+ const token = JSON.parse(tokenLiteral as string) as string;
+
+ // Open the socket while `connectParent` is the live run parent.
+ const peerSocket = net.createConnection({ host: "127.0.0.1", port });
+ peer = peerSocket;
+ peerSocket.on("error", () => undefined);
+ await new Promise((resolve, reject) => {
+ peerSocket.once("connect", () => resolve());
+ peerSocket.once("error", reject);
+ });
+ // Let the server accept the connection and register the `data` handler
+ // under the connect-time parent before the getter switches.
+ await new Promise((resolve) => setImmediate(resolve));
+
+ // The run enters an agent turn: the live run parent switches.
+ currentParent = turnParent;
+
+ // Send one stdin line. The first token-bearing message authenticates and
+ // writes the stdin file. That write must read `turnParent` at send time.
+ peerSocket.write(`${JSON.stringify({ token, type: "stdin", data: Buffer.from("hi").toString("base64") })}\n`);
+
+ await stdinWriteObserved;
+ // The stdin write ran under the send-time parent, not the connect-time
+ // parent captured when the socket opened.
+ expect(stdinWriteStep).not.toBe("unset");
+ expect(stdinWriteStep).not.toBeNull();
+ expect((stdinWriteStep as { parentContext?: unknown }).parentContext).toBe(turnParent);
+ expect((stdinWriteStep as { parentContext?: unknown }).parentContext).not.toBe(connectParent);
+ expect((stdinWriteStep as { criticalPath?: boolean }).criticalPath).toBe(false);
+ } finally {
+ peer?.destroy();
+ await bridge?.stop();
+ }
+ });
+
+ it("wraps a stdin write in a sandbox.agentSession.sendInput span", async () => {
+ // With a span runner injected, the socket handler wraps one outbound ACP
+ // message to the agent in a `sandbox.agentSession.sendInput` span. This test
+ // connects a socket, sends one stdin line, and proves the handler opens that
+ // wrapper span around the write.
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-sendinput-span-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "noop-acp-child.mjs");
+ await writeFile(childPath, "process.stdin.on('data', () => {});\n", "utf8");
+
+ const spanNames: string[] = [];
+ let resolveSendInput: () => void = () => {};
+ const sendInputObserved = new Promise((resolve) => {
+ resolveSendInput = resolve;
+ });
+
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner: createLocalSandboxRunner(),
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-process-session-sendinput-span",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ // Record each wrapper span name, then run the wrapped work.
+ runtimeSpan: async (name, work) => {
+ spanNames.push(name);
+ if (name === "sandbox.agentSession.sendInput") resolveSendInput();
+ return work();
+ },
+ });
+ expect(bridge).not.toBeNull();
+
+ let peer: net.Socket | null = null;
+ try {
+ const proxySource = await readFile(bridge!.agentCommand, "utf8");
+ const port = Number(/port: (\d+)/.exec(proxySource)?.[1] ?? Number.NaN);
+ const tokenLiteral = /const token = (".*?");/.exec(proxySource)?.[1];
+ const token = JSON.parse(tokenLiteral as string) as string;
+
+ const peerSocket = net.createConnection({ host: "127.0.0.1", port });
+ peer = peerSocket;
+ peerSocket.on("error", () => undefined);
+ await new Promise((resolve, reject) => {
+ peerSocket.once("connect", () => resolve());
+ peerSocket.once("error", reject);
+ });
+
+ // The first token-bearing message authenticates and writes the stdin file.
+ peerSocket.write(
+ `${JSON.stringify({ token, type: "stdin", data: Buffer.from("hi").toString("base64") })}\n`,
+ );
+
+ await sendInputObserved;
+ expect(spanNames).toContain("sandbox.agentSession.sendInput");
+ } finally {
+ peer?.destroy();
+ await bridge?.stop();
+ }
+ });
+
+ it("wraps each poll tick in a sandbox.agentSession.pollOutput span", async () => {
+ // With a span runner injected, the poll timer wraps each 100 ms poll tick in
+ // a `sandbox.agentSession.pollOutput` span. This test lets the first poll tick
+ // fire and proves the timer opens that wrapper span.
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-poll-span-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "noop-acp-child.mjs");
+ await writeFile(childPath, "process.stdin.on('data', () => {});\n", "utf8");
+
+ const spanNames: string[] = [];
+ let resolvePoll: () => void = () => {};
+ const pollObserved = new Promise((resolve) => {
+ resolvePoll = resolve;
+ });
+
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner: createLocalSandboxRunner(),
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-process-session-poll-span",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ // Record each wrapper span name, then run the wrapped work.
+ runtimeSpan: async (name, work) => {
+ spanNames.push(name);
+ if (name === "sandbox.agentSession.pollOutput") resolvePoll();
+ return work();
+ },
+ });
+ expect(bridge).not.toBeNull();
+
+ try {
+ await pollObserved;
+ expect(spanNames).toContain("sandbox.agentSession.pollOutput");
+ } finally {
+ await bridge?.stop();
+ }
+ });
+
it("bridges bidirectional sandbox process sessions through a local ACPX-spawnable proxy", async () => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-"));
cleanupDirs.push(rootDir);
@@ -543,6 +901,344 @@ describe("sandbox adapter execution targets", () => {
}
});
+ describe("streamed output (streamOutputViaSession)", () => {
+ it("bridges bidirectional sessions when the wrapper streams output to stdout", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-stream-echo-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "echo-acp-child.mjs");
+ await writeFile(
+ childPath,
+ [
+ "process.stdin.on('data', (chunk) => {",
+ " process.stdout.write('out:' + chunk.toString());",
+ " process.stderr.write('err:' + chunk.toString());",
+ "});",
+ ].join("\n"),
+ "utf8",
+ );
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner: createLocalSandboxRunner(),
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-stream-echo",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ streamOutputViaSession: true,
+ });
+ expect(bridge).not.toBeNull();
+
+ try {
+ const result = await runProxyWithInput(bridge!.agentCommand, "hello\n");
+ expect(result.code).toBe(0);
+ expect(result.stdout).toBe("out:hello\n");
+ expect(result.stderr).toBe("err:hello\n");
+ } finally {
+ await bridge?.stop();
+ }
+ });
+
+ it("buffers streamed output until the local proxy connects", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-stream-buffer-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "fast-stream-child.mjs");
+ await writeFile(
+ childPath,
+ [
+ "process.stdout.write('early-out\\n');",
+ "process.stderr.write('early-err\\n');",
+ "setTimeout(() => process.exit(0), 20);",
+ ].join("\n"),
+ "utf8",
+ );
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner: createLocalSandboxRunner(),
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-stream-buffer",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ streamOutputViaSession: true,
+ });
+ expect(bridge).not.toBeNull();
+
+ try {
+ await new Promise((resolve) => setTimeout(resolve, 300));
+ const result = await runProxyWithInput(bridge!.agentCommand, "");
+ expect(result.code).toBe(0);
+ // The seq guard delivers the early output exactly once even though the
+ // live stream and the terminal result both carry it.
+ expect(result.stdout).toBe("early-out\n");
+ expect(result.stderr).toBe("early-err\n");
+ } finally {
+ await bridge?.stop();
+ }
+ });
+
+ it("delivers full streamed output when the sandbox child exits immediately", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-stream-fast-exit-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "instant-stream-child.mjs");
+ await writeFile(
+ childPath,
+ [
+ "process.stdout.write('final-out\\n');",
+ "process.stderr.write('final-err\\n');",
+ ].join("\n"),
+ "utf8",
+ );
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner: createLocalSandboxRunner(),
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-stream-fast-exit",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ streamOutputViaSession: true,
+ });
+ expect(bridge).not.toBeNull();
+
+ try {
+ const result = await runProxyWithInput(bridge!.agentCommand, "");
+ expect(result.code).toBe(0);
+ expect(result.stdout).toBe("final-out\n");
+ expect(result.stderr).toBe("final-err\n");
+ } finally {
+ await bridge?.stop();
+ }
+ });
+
+ it("streams live output before the child exits and never writes output event files", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-stream-live-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "live-stream-child.mjs");
+ await writeFile(
+ childPath,
+ [
+ "process.stdin.setEncoding('utf8');",
+ "process.stdin.on('data', (chunk) => {",
+ " if (chunk.includes('ping')) {",
+ " process.stdout.write('delta:ping\\n');",
+ " process.stderr.write('trace:ping\\n');",
+ " }",
+ " if (chunk.includes('finish')) process.exit(0);",
+ "});",
+ "process.stdin.resume();",
+ ].join("\n"),
+ "utf8",
+ );
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner: createLocalSandboxRunner(),
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-stream-live",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ streamOutputViaSession: true,
+ });
+ expect(bridge).not.toBeNull();
+
+ const child = spawn(bridge!.agentCommand, [], { stdio: ["pipe", "pipe", "pipe"] });
+ let stdout = "";
+ let stderr = "";
+ let exited = false;
+ child.stdout.setEncoding("utf8");
+ child.stderr.setEncoding("utf8");
+ child.stdout.on("data", (chunk) => {
+ stdout += chunk;
+ });
+ child.stderr.on("data", (chunk) => {
+ stderr += chunk;
+ });
+ const exitPromise = new Promise((resolve, reject) => {
+ const timeout = setTimeout(() => {
+ child.kill("SIGKILL");
+ reject(new Error("Timed out waiting for streamed process session proxy."));
+ }, 5000);
+ child.on("error", (error) => {
+ clearTimeout(timeout);
+ reject(error);
+ });
+ child.on("exit", (exitCode) => {
+ exited = true;
+ clearTimeout(timeout);
+ resolve(exitCode);
+ });
+ });
+
+ try {
+ child.stdin.write("ping\n");
+ await waitForCondition(
+ () => stdout.includes("delta:ping\n") && stderr.includes("trace:ping\n"),
+ "Timed out waiting for live streamed process session output.",
+ 3000,
+ );
+ expect(exited).toBe(false);
+
+ child.stdin.end("finish\n");
+ await expect(exitPromise).resolves.toBe(0);
+
+ // The streamed path uses the stdout wrapper, not the output-file poll, so
+ // no `events` directory is ever created under the session runtime tree.
+ const hasEventsDir = await readdir(
+ path.posix.join(rootDir, ".paperclip-runtime", "acpx", "process-sessions"),
+ { withFileTypes: true, recursive: true },
+ )
+ .then((entries) => entries.some((entry) => entry.isDirectory() && entry.name === "events"))
+ .catch(() => false);
+ expect(hasEventsDir).toBe(false);
+ } finally {
+ if (!exited) {
+ child.kill("SIGKILL");
+ await exitPromise.catch(() => undefined);
+ }
+ await bridge?.stop();
+ }
+ });
+
+ it("keeps the agent command on the persistent session and forces bridge control execs off it", async () => {
+ // Regression guard for the streamed-mode startup deadlock. The persistent
+ // session is one serialized shell. In streamed mode the agent runs as a
+ // long-lived foreground session command that holds the session for the
+ // whole run. The bridge control-plane execs (script sync, stdin delivery,
+ // teardown) must run concurrently with the agent, so each must force
+ // itself off the session. On the session they queue behind the agent
+ // command that never returns, and the first handshake write never drains.
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-process-session-stream-isolation-"));
+ cleanupDirs.push(rootDir);
+ const childPath = path.join(rootDir, "echo-acp-child.mjs");
+ await writeFile(
+ childPath,
+ [
+ "process.stdin.on('data', (chunk) => {",
+ " process.stdout.write('out:' + chunk.toString());",
+ "});",
+ ].join("\n"),
+ "utf8",
+ );
+
+ const delegate = createLocalSandboxRunner();
+ const execs: Array<{ useSession?: boolean; bypassSession?: boolean; script: string }> = [];
+ const runner = {
+ execute: vi.fn(
+ async (
+ input: Parameters[0] & {
+ useSession?: boolean;
+ bypassSession?: boolean;
+ },
+ ) => {
+ execs.push({
+ useSession: input.useSession,
+ bypassSession: input.bypassSession,
+ script: input.args?.[1] ?? "",
+ });
+ return delegate.execute(input);
+ },
+ ),
+ };
+ const target: AdapterSandboxExecutionTarget = {
+ kind: "remote",
+ transport: "sandbox",
+ providerKey: "local-test",
+ remoteCwd: rootDir,
+ timeoutMs: 30_000,
+ runner,
+ };
+
+ const bridge = await startAdapterExecutionTargetProcessSessionBridge({
+ runId: "run-stream-isolation",
+ target,
+ runtimeRootDir: path.posix.join(rootDir, ".paperclip-runtime", "acpx"),
+ adapterKey: "acpx",
+ command: process.execPath,
+ args: [childPath],
+ cwd: rootDir,
+ env: {},
+ timeoutSec: 5,
+ onLog: async () => {},
+ streamOutputViaSession: true,
+ });
+ expect(bridge).not.toBeNull();
+
+ try {
+ // Round-trip one input so a stdin-delivery control exec runs and gets
+ // recorded before the assertions below.
+ const result = await runProxyWithInput(bridge!.agentCommand, "hello\n");
+ expect(result.stdout).toBe("out:hello\n");
+
+ // Exactly one exec runs on the persistent session: the long-lived agent
+ // command. It streams its output through the session log stream, so it
+ // must not also bypass the session.
+ const sessionExecs = execs.filter((exec) => exec.useSession === true);
+ expect(sessionExecs).toHaveLength(1);
+ expect(sessionExecs[0]!.bypassSession).not.toBe(true);
+ expect(sessionExecs[0]!.script).toContain("node ");
+
+ // Every other exec is bridge control-plane plumbing. Each must force
+ // itself off the persistent session so it never queues behind the agent
+ // command that holds it.
+ const controlExecs = execs.filter((exec) => exec.useSession !== true);
+ expect(controlExecs.length).toBeGreaterThan(0);
+ for (const exec of controlExecs) {
+ expect(exec.bypassSession).toBe(true);
+ }
+ } finally {
+ await bridge?.stop();
+ }
+ });
+ });
+
it("applies the remote sandbox fallback when adapter timeoutSec is unset", () => {
const sandboxTarget: AdapterSandboxExecutionTarget = {
kind: "remote",
diff --git a/packages/adapter-utils/src/execution-target.ts b/packages/adapter-utils/src/execution-target.ts
index d1d878d7c240..a525b2f29001 100644
--- a/packages/adapter-utils/src/execution-target.ts
+++ b/packages/adapter-utils/src/execution-target.ts
@@ -46,6 +46,11 @@ import {
} from "./server-utils.js";
import { sanitizeRemoteExecutionEnv } from "./remote-execution-env.js";
import { preferredShellForSandbox, shellCommandArgs } from "./sandbox-shell.js";
+import {
+ runWithRuntimeParent,
+ type RuntimeSpanRunner,
+ type StartupSpanContext,
+} from "./acpx-engine/startup-timing.js";
import type { RuntimeProgressSink, RuntimeStatusSink } from "./runtime-progress.js";
import type { LocalProcessSandboxOptions } from "./local-process-sandbox.js";
@@ -101,6 +106,13 @@ export interface AdapterSandboxExecutionTarget extends AdapterExecutionTargetWor
* set to `false` to explicitly opt out back to batch-at-end delivery.
*/
streamRunLogs?: boolean | null;
+ /**
+ * Stream the interactive ACP agent output through the persistent session log
+ * stream instead of the host-side output-file poll. The process session
+ * bridge runs the agent as one long-lived session command and reads its
+ * output frames from the stream. Default OFF: the bridge keeps the poll path.
+ */
+ streamAgentSessionOutput?: boolean | null;
}
export type AdapterExecutionTarget =
@@ -1138,6 +1150,11 @@ export async function prepareAdapterExecutionTargetRuntime(input: {
// counters without further changes here.
onProgress?: RuntimeProgressSink;
onRuntimeProgress?: RuntimeStatusSink;
+ // Optional host span runner for the workspace tarball build. Only the confined
+ // sandbox lane uses it: it forwards the runner to prepareCommandManagedRuntime
+ // so the host pack time rides one `pack` span under the `stage.sync` step. The
+ // SSH and local lanes ignore it. The default is a no-op.
+ runtimeSpan?: RuntimeSpanRunner;
}): Promise {
const target = input.target ?? { kind: "local" as const };
if (target.kind === "local") {
@@ -1201,6 +1218,7 @@ export async function prepareAdapterExecutionTargetRuntime(input: {
detectCommand: input.detectCommand,
onProgress: input.onProgress,
onRuntimeProgress: input.onRuntimeProgress,
+ runtimeSpan: input.runtimeSpan,
});
return {
target,
@@ -1273,6 +1291,10 @@ async function readBridgeForwardResponseBody(response: Response, maxBodyBytes: n
const PROCESS_SESSION_PROXY_SCRIPT = "paperclip-process-session-proxy.mjs";
const PROCESS_SESSION_REMOTE_SCRIPT = "paperclip-process-session-remote.mjs";
+// The streamed variant writes its output frames to stdout, so it rides a
+// separate remote path. A sandbox can hold both scripts without the content
+// hash-skip gate thrashing when a run switches output mode.
+const PROCESS_SESSION_REMOTE_STREAM_SCRIPT = "paperclip-process-session-remote-stream.mjs";
const PROCESS_SESSION_AUTH_TIMEOUT_MS = 5_000;
function jsonLine(value: unknown): string {
@@ -1304,13 +1326,14 @@ async function syncProcessSessionRemoteScript(input: {
remoteScriptPath: string;
timeoutMs?: number | null;
shellCommand?: "bash" | "sh" | null;
+ outputToStdout?: boolean;
}): Promise<{ uploaded: boolean }> {
const { uploaded } = await syncRemoteTextFileWithHashSkip({
runner: input.runner,
remoteCwd: input.remoteCwd,
remoteDir: input.remoteScriptDir,
remotePath: input.remoteScriptPath,
- body: getProcessSessionRemoteSource(),
+ body: getProcessSessionRemoteSource({ outputToStdout: input.outputToStdout === true }),
label: "Process session remote script",
action: "sync process session remote script",
lockDir: path.posix.join(input.remoteScriptDir, ".paperclip-process-session-script.lock"),
@@ -1350,6 +1373,14 @@ async function waitForLocalServerListen(server: net.Server): Promise {
return address.port;
}
+/** Span name that wraps the socket handler's one `writeTextFile` exec — one
+ * outbound ACP message to the agent. */
+const AGENT_SESSION_SEND_INPUT_SPAN = "sandbox.agentSession.sendInput";
+
+/** Span name that wraps one 100 ms poll tick — `list`, then `read`+`remove` per
+ * file found (`1 + 2n` execs). */
+const AGENT_SESSION_POLL_OUTPUT_SPAN = "sandbox.agentSession.pollOutput";
+
export async function startAdapterExecutionTargetProcessSessionBridge(input: {
runId: string;
target: AdapterExecutionTarget | null | undefined;
@@ -1366,6 +1397,23 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
env: Record | (() => Promise>);
timeoutSec?: number | null;
onLog?: (stream: "stdout" | "stderr", chunk: string) => Promise;
+ // Return the current-run parent-context token. The socket handlers and the
+ // poll timer read it per unit of work and run under it, so their run-time
+ // `sandbox.exec` spans parent to the live run span (`agent.turn` during the
+ // turn, `task.run` otherwise). When it is absent, the work runs with an empty
+ // store, exactly like the earlier `runWithoutActiveStep` behavior.
+ getRuntimeParentContext?: () => StartupSpanContext | undefined;
+ // Wrap each unit of run-time work in its own named span. The socket handler
+ // uses it for `sandbox.agentSession.sendInput` and the poll timer for
+ // `sandbox.agentSession.pollOutput`, so each unit's inner `sandbox.exec` spans
+ // group under one wrapper span. When it is absent, the work runs under the run
+ // parent with no wrapper span, exactly like the earlier behavior.
+ runtimeSpan?: RuntimeSpanRunner;
+ // Stream the agent output through the persistent session log stream instead of
+ // the host output-file poll. When true, the bridge runs the wrapper as one
+ // long-lived session command and reads its stdout frames from the stream, and
+ // it does not start the 100 ms poll. Default OFF: the bridge keeps the poll.
+ streamOutputViaSession?: boolean;
}): Promise {
if (!input.target || input.target.kind !== "remote" || input.target.transport !== "sandbox") {
return null;
@@ -1374,6 +1422,14 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
const target = input.target;
const onLog = input.onLog ?? (async () => {});
const runner = requireSandboxRunner(target);
+ // Run one unit of run-time work under its named wrapper span when a span
+ // runner is injected. Without a runner, run the work under the current run
+ // parent, so the inner `sandbox.exec` spans parent to the live run span,
+ // exactly like the earlier behavior.
+ const runRuntimeWork = (name: string, work: () => Promise): Promise =>
+ input.runtimeSpan
+ ? input.runtimeSpan(name, work)
+ : runWithRuntimeParent(input.getRuntimeParentContext?.(), work);
const shellCommand = preferredSandboxShell(target);
const timeoutMs =
typeof input.timeoutSec === "number" && Number.isFinite(input.timeoutSec) && input.timeoutSec > 0
@@ -1387,7 +1443,14 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
const sessionDir = path.posix.join(bridgeRuntimeDir, sessionId);
const stdinDir = path.posix.join(sessionDir, "stdin");
const eventsDir = path.posix.join(sessionDir, "events");
- const remoteScriptPath = path.posix.join(bridgeRuntimeDir, PROCESS_SESSION_REMOTE_SCRIPT);
+ // The streamed wrapper writes its frames to stdout and rides a separate remote
+ // path, so a warm sandbox can hold both wrapper scripts without the content
+ // hash-skip gate thrashing when a run switches output mode.
+ const streamOutput = input.streamOutputViaSession === true;
+ const remoteScriptPath = path.posix.join(
+ bridgeRuntimeDir,
+ streamOutput ? PROCESS_SESSION_REMOTE_STREAM_SCRIPT : PROCESS_SESSION_REMOTE_SCRIPT,
+ );
const client = createCommandManagedSandboxCallbackBridgeQueueClient({
runner,
remoteCwd: target.remoteCwd,
@@ -1405,6 +1468,7 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
remoteScriptPath,
timeoutMs,
shellCommand,
+ outputToStdout: streamOutput,
});
// Resolve the launch env AFTER the env-independent setup above, so a caller
@@ -1418,26 +1482,34 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
env: sanitizeRemoteExecutionEnv(launchEnv),
}), "utf8").toString("base64");
- await onLog("stdout", `[paperclip] Starting ACP process session bridge in sandbox (${target.providerKey ?? "provider"}).\n`);
- const startResult = await runner.execute({
- command: shellCommand,
- args: shellCommandArgs(
- [
- `mkdir -p ${shellQuote(stdinDir)} ${shellQuote(eventsDir)}`,
- `PAPERCLIP_PROCESS_SESSION_DIR=${shellQuote(sessionDir)} ` +
- `PAPERCLIP_PROCESS_SESSION_COMMAND_B64=${shellQuote(commandPayload)} ` +
- `nohup node ${shellQuote(remoteScriptPath)} >/dev/null 2>&1 < /dev/null &`,
- "printf '%s\\n' \"$!\"",
- ].join("\n"),
- ),
- cwd: target.remoteCwd,
- env: {
- PAPERCLIP_SANDBOX_EXEC_CHANNEL: "bridge",
- },
- timeoutMs,
- });
- if (startResult.timedOut || (startResult.exitCode ?? 1) !== 0) {
- throw new Error(`Failed to start sandbox ACP process session bridge: ${startResult.stderr || startResult.stdout}`);
+ // Legacy poll path: background the wrapper with `nohup` and read its output
+ // event files with the host poll below. The streamed path launches the wrapper
+ // as one foreground session command further down instead, so skip this.
+ if (!streamOutput) {
+ await onLog("stdout", `[paperclip] Starting ACP process session bridge in sandbox (${target.providerKey ?? "provider"}).\n`);
+ const startResult = await runner.execute({
+ command: shellCommand,
+ args: shellCommandArgs(
+ [
+ `mkdir -p ${shellQuote(stdinDir)} ${shellQuote(eventsDir)}`,
+ `PAPERCLIP_PROCESS_SESSION_DIR=${shellQuote(sessionDir)} ` +
+ `PAPERCLIP_PROCESS_SESSION_COMMAND_B64=${shellQuote(commandPayload)} ` +
+ `nohup node ${shellQuote(remoteScriptPath)} >/dev/null 2>&1 < /dev/null &`,
+ "printf '%s\\n' \"$!\"",
+ ].join("\n"),
+ ),
+ cwd: target.remoteCwd,
+ env: {
+ PAPERCLIP_SANDBOX_EXEC_CHANNEL: "bridge",
+ },
+ timeoutMs,
+ // The wrapper launch is bridge plumbing. Keep it off the persistent
+ // session so it never queues behind an in-run session command.
+ bypassSession: true,
+ });
+ if (startResult.timedOut || (startResult.exitCode ?? 1) !== 0) {
+ throw new Error(`Failed to start sandbox ACP process session bridge: ${startResult.stderr || startResult.stdout}`);
+ }
}
let socket: net.Socket | null = null;
@@ -1488,6 +1560,13 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
};
const liveSockets = new Set();
+ // Register the per-connection socket handlers with no run parent context.
+ // A stdin write from a socket handler is a run-time exec, not startup work.
+ // The connection can open under `task.run` and receive stdin later, during an
+ // `agent.turn`. So the handler must read the current-run parent at send time,
+ // not at connect time. A connect-time read captures the parent that was live
+ // when the socket opened, and every later exec span parents to that stale
+ // parent. The `data` handler below reads the getter per message instead.
const server = net.createServer((nextSocket) => {
liveSockets.add(nextSocket);
nextSocket.setEncoding("utf8");
@@ -1531,20 +1610,28 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
socket = nextSocket;
flushPendingRemoteEvents();
}
- void (async () => {
- if (message.type === "stdin" && typeof message.data === "string") {
- stdinSeq += 1;
- const name = `${String(stdinSeq).padStart(12, "0")}.json`;
- await client.writeTextFile(path.posix.join(stdinDir, name), jsonLine({ type: "stdin", data: message.data }));
- } else if (message.type === "stdinEnd") {
- stdinSeq += 1;
- const name = `${String(stdinSeq).padStart(12, "0")}.json`;
- await client.writeTextFile(path.posix.join(stdinDir, name), jsonLine({ type: "stdinEnd" }));
- }
- })().catch((error) => {
- nextSocket.write(jsonLine({ type: "error", message: error instanceof Error ? error.message : String(error) }));
- nextSocket.destroy();
- });
+ // Wrap one outbound ACP message to the agent in a
+ // `sandbox.agentSession.sendInput` span, so its one `writeTextFile` exec
+ // groups under one named span. The span runner reads the current-run
+ // parent at send time: the live parent switches to `agent.turn` during
+ // the turn and back to `task.run` after it. A message that is neither
+ // `stdin` nor `stdinEnd` writes nothing, so it opens no span.
+ const stdinPayload =
+ message.type === "stdin" && typeof message.data === "string"
+ ? { type: "stdin", data: message.data }
+ : message.type === "stdinEnd"
+ ? { type: "stdinEnd" }
+ : null;
+ if (stdinPayload) {
+ stdinSeq += 1;
+ const name = `${String(stdinSeq).padStart(12, "0")}.json`;
+ void runRuntimeWork(AGENT_SESSION_SEND_INPUT_SPAN, () =>
+ client.writeTextFile(path.posix.join(stdinDir, name), jsonLine(stdinPayload)),
+ ).catch((error) => {
+ nextSocket.write(jsonLine({ type: "error", message: error instanceof Error ? error.message : String(error) }));
+ nextSocket.destroy();
+ });
+ }
}
});
});
@@ -1572,16 +1659,125 @@ export async function startAdapterExecutionTargetProcessSessionBridge(input: {
return;
} finally {
if (!stopping) {
- pollTimer = setTimeout(() => void poll(), 100);
- pollTimer.unref?.();
+ schedulePoll();
}
}
};
+ // Schedule the long-lived poll timer. Wrap each 100 ms poll tick in a
+ // `sandbox.agentSession.pollOutput` span, so the tick's `list` plus per-file
+ // `read`/`remove` execs group under one named span. The poll loop reads remote
+ // event files with run-time execs, not startup work, so the wrapper span and
+ // its child execs parent to the live run span, not to the ended bridge step.
+ // The span runner reads the run parent per tick, because the re-arm timer that
+ // the poll body schedules opens a new tick span: the live parent switches to
+ // `agent.turn` during the turn and back to `task.run` after it.
+ const schedulePoll = () => {
+ pollTimer = setTimeout(() => void runRuntimeWork(AGENT_SESSION_POLL_OUTPUT_SPAN, poll), 100);
+ pollTimer.unref?.();
+ };
+
const port = await waitForLocalServerListen(server);
const agentCommand = await writeProcessSessionProxyScript(proxyDir, port, token);
- pollTimer = setTimeout(() => void poll(), 100);
- pollTimer.unref?.();
+
+ if (streamOutput) {
+ // Streamed output path. Run the wrapper as one long-lived session command;
+ // its stdout carries newline-delimited JSON frames that reach the host
+ // through the provider session log stream. Deliver each frame exactly once
+ // by its monotonic `seq`, so a frame that arrives both live and in the final
+ // result is not repeated. There is no host output-file poll here.
+ let streamBuffer = "";
+ let lastSeq = 0;
+ let sawTerminal = false;
+ const deliverFrame = (frame: (typeof pendingRemoteEvents)[number] & { seq?: number }) => {
+ if (typeof frame.seq === "number") {
+ if (frame.seq <= lastSeq) return;
+ lastSeq = frame.seq;
+ }
+ if (frame.type === "exit" || frame.type === "error") sawTerminal = true;
+ deliverRemoteEvent(frame);
+ };
+ const parseFrameLine = (line: string) => {
+ if (!line.trim()) return;
+ let frame: (typeof pendingRemoteEvents)[number] & { seq?: number };
+ try {
+ frame = JSON.parse(line) as typeof frame;
+ } catch {
+ return;
+ }
+ deliverFrame(frame);
+ };
+ // Live delivery: buffer partial lines across stream chunks, deliver each
+ // complete frame line as it arrives.
+ const ingestStreamChunk = (text: string) => {
+ streamBuffer += text;
+ const split = splitJsonLines(streamBuffer);
+ streamBuffer = split.rest;
+ for (const line of split.lines) parseFrameLine(line);
+ };
+ // Terminal delivery (the defined fallback to the poll): the resolved result
+ // carries the full wrapper stdout even when the live stream degraded to the
+ // provider session-log poll. The text is complete and self-contained, so
+ // re-parse it on its own; the `seq` guard drops every frame the live stream
+ // already delivered. Drop any partial live line — its complete form is in the
+ // full text.
+ const ingestFinalText = (text: string) => {
+ streamBuffer = "";
+ for (const line of text.split(/\n/)) parseFrameLine(line);
+ };
+
+ const launchEnvForStream =
+ typeof input.env === "function" ? await input.env() : input.env;
+ const streamCommandPayload = Buffer.from(JSON.stringify({
+ command: input.command,
+ args: input.args,
+ cwd: input.cwd || target.remoteCwd,
+ env: sanitizeRemoteExecutionEnv(launchEnvForStream),
+ }), "utf8").toString("base64");
+ await onLog(
+ "stdout",
+ `[paperclip] Starting streamed ACP process session bridge in sandbox (${target.providerKey ?? "provider"}).\n`,
+ );
+ // Fire the long-lived command; do NOT await it here. `useSession` forces the
+ // persistent session so the provider streams the wrapper stdout back through
+ // `onLog`. On resolve, the terminal re-parse fills any frames the live stream
+ // missed; on reject, deliver one error frame so the local proxy fails loud.
+ void runner
+ .execute({
+ command: shellCommand,
+ args: shellCommandArgs(`node ${shellQuote(remoteScriptPath)}`),
+ cwd: target.remoteCwd,
+ env: {
+ PAPERCLIP_PROCESS_SESSION_DIR: sessionDir,
+ PAPERCLIP_PROCESS_SESSION_COMMAND_B64: streamCommandPayload,
+ PAPERCLIP_SANDBOX_EXEC_CHANNEL: "bridge",
+ },
+ timeoutMs,
+ useSession: true,
+ onLog: async (stream, chunk) => {
+ if (stream === "stdout") ingestStreamChunk(chunk);
+ },
+ })
+ .then((result) => {
+ ingestFinalText(result.stdout);
+ if (!sawTerminal && !stopping) {
+ deliverRemoteEvent({
+ type: "exit",
+ code: typeof result.exitCode === "number" ? result.exitCode : null,
+ });
+ }
+ })
+ .catch((error) => {
+ if (!stopping) {
+ deliverRemoteEvent({
+ type: "error",
+ message: error instanceof Error ? error.message : String(error),
+ });
+ }
+ });
+ } else {
+ schedulePoll();
+ }
return {
agentCommand,
@@ -1647,7 +1843,94 @@ socket.on("close", () => {
`;
}
-function getProcessSessionRemoteSource(): string {
+function getProcessSessionRemoteSource(input?: { outputToStdout?: boolean }): string {
+ return input?.outputToStdout === true
+ ? getProcessSessionRemoteStreamSource()
+ : getProcessSessionRemoteEventFileSource();
+}
+
+// The shared stdin drain. Both wrappers read newline-delimited stdin messages
+// from the stdin file queue and write them to the child, then end the child
+// stdin on `stdinEnd`. A write to a closed child stdin only emits an `error`
+// event, so the wrapper installs a no-op handler at the call site.
+const PROCESS_SESSION_STDIN_POLL_TAIL = `child.stdin.on("error", () => {});
+
+async function pollStdin() {
+ while (!stdinClosed) {
+ const entries = (await fs.readdir(stdinDir).catch(() => [])).filter((name) => name.endsWith(".json")).sort();
+ for (const name of entries) {
+ const file = path.posix.join(stdinDir, name);
+ const raw = await fs.readFile(file, "utf8").catch(() => null);
+ await fs.rm(file, { force: true }).catch(() => undefined);
+ if (!raw) continue;
+ const message = JSON.parse(raw);
+ if (message.type === "stdin" && typeof message.data === "string") {
+ if (!stdinClosed) child.stdin.write(Buffer.from(message.data, "base64"));
+ } else if (message.type === "stdinEnd") {
+ stdinClosed = true;
+ child.stdin.end();
+ break;
+ }
+ }
+ if (!stdinClosed) await new Promise((resolve) => setTimeout(resolve, 50));
+ }
+}
+
+void pollStdin().catch((error) => void writeEvent({ type: "error", message: error instanceof Error ? error.message : String(error) }));
+`;
+
+// Streamed variant: the wrapper writes each output frame as one newline-
+// delimited JSON line to its stdout. The host runs this wrapper as one
+// long-lived session command and reads the frames from the session log stream,
+// so there is no host output-file poll. Each frame carries a monotonic `seq`,
+// so the host delivers every frame exactly once whether it arrives live or in
+// the final result. The wrapper exits when the child closes, so the session
+// command settles and the session shell (the subshell wrap around it) survives.
+function getProcessSessionRemoteStreamSource(): string {
+ return `import { spawn } from "node:child_process";
+import { promises as fs } from "node:fs";
+import path from "node:path";
+
+const sessionDir = process.env.PAPERCLIP_PROCESS_SESSION_DIR;
+const commandPayload = process.env.PAPERCLIP_PROCESS_SESSION_COMMAND_B64;
+if (!sessionDir || !commandPayload) throw new Error("Missing process session bridge env.");
+
+const stdinDir = path.posix.join(sessionDir, "stdin");
+let seq = 0;
+let stdinClosed = false;
+
+const config = JSON.parse(Buffer.from(commandPayload, "base64").toString("utf8"));
+await fs.mkdir(stdinDir, { recursive: true });
+
+// One newline-delimited JSON frame per event. Node keeps process.stdout writes
+// ordered, and the base64 payload holds no newline, so each frame is one line.
+function writeEvent(event) {
+ seq += 1;
+ process.stdout.write(JSON.stringify({ seq, ...event }) + "\\n");
+}
+
+const child = spawn(config.command, Array.isArray(config.args) ? config.args : [], {
+ cwd: config.cwd || process.cwd(),
+ env: { ...process.env, ...(config.env || {}) },
+ stdio: ["pipe", "pipe", "pipe"],
+});
+
+child.stdout.on("data", (chunk) => writeEvent({ type: "data", stream: "stdout", data: Buffer.from(chunk).toString("base64") }));
+child.stderr.on("data", (chunk) => writeEvent({ type: "data", stream: "stderr", data: Buffer.from(chunk).toString("base64") }));
+child.on("error", (error) => writeEvent({ type: "error", message: error.message }));
+// "close" (not "exit") so stdout/stderr fully drain before the exit frame.
+// Stop the stdin poll and set the exit code, then let the event loop drain: a
+// natural exit flushes the stdout pipe, so the exit frame always lands.
+child.on("close", (code, signal) => {
+ writeEvent({ type: "exit", code, signal });
+ stdinClosed = true;
+ process.exitCode = typeof code === "number" ? code : 1;
+});
+
+${PROCESS_SESSION_STDIN_POLL_TAIL}`;
+}
+
+function getProcessSessionRemoteEventFileSource(): string {
return `import { spawn } from "node:child_process";
import { promises as fs } from "node:fs";
import path from "node:path";
@@ -1691,29 +1974,7 @@ child.on("error", (error) => void writeEvent({ type: "error", message: error.mes
// the write chain then guarantees the exit file lands after every data file.
child.on("close", (code, signal) => void writeEvent({ type: "exit", code, signal }));
-async function pollStdin() {
- while (!stdinClosed) {
- const entries = (await fs.readdir(stdinDir).catch(() => [])).filter((name) => name.endsWith(".json")).sort();
- for (const name of entries) {
- const file = path.posix.join(stdinDir, name);
- const raw = await fs.readFile(file, "utf8").catch(() => null);
- await fs.rm(file, { force: true }).catch(() => undefined);
- if (!raw) continue;
- const message = JSON.parse(raw);
- if (message.type === "stdin" && typeof message.data === "string") {
- child.stdin.write(Buffer.from(message.data, "base64"));
- } else if (message.type === "stdinEnd") {
- stdinClosed = true;
- child.stdin.end();
- break;
- }
- }
- if (!stdinClosed) await new Promise((resolve) => setTimeout(resolve, 50));
- }
-}
-
-void pollStdin().catch((error) => void writeEvent({ type: "error", message: error instanceof Error ? error.message : String(error) }));
-`;
+${PROCESS_SESSION_STDIN_POLL_TAIL}`;
}
export async function startAdapterExecutionTargetPaperclipBridge(input: {
@@ -1726,6 +1987,16 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: {
hostApiUrl?: string | null;
onLog?: (stream: "stdout" | "stderr", chunk: string) => Promise;
maxBodyBytes?: number | null;
+ // Return the current-run parent-context token. The factory threads it into the
+ // callback bridge worker, which reads it per request so each request
+ // `sandbox.exec` span parents to the live run span. When it is absent, the
+ // request work runs with an empty store, exactly like the earlier behavior.
+ getRuntimeParentContext?: () => StartupSpanContext | undefined;
+ // Wrap each callback request in a `sandbox.callbackBridge.relayRequest` span.
+ // The factory threads it into the worker, which uses it per request so each
+ // request's execs group under one wrapper span. When it is absent, the request
+ // work runs under the run parent with no wrapper span.
+ runtimeSpan?: RuntimeSpanRunner;
}): Promise {
if (!adapterExecutionTargetUsesPaperclipBridge(input.target)) {
return null;
@@ -1787,10 +2058,17 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: {
// this flag is enabled. Only intended for active debugging in trusted
// environments.
const bridgeDebugEnabled = isBridgeDebugEnabled(process.env);
+ // `startSandboxCallbackBridgeWorker` keeps its awaited queue-directory
+ // setup on the active `bridge.paperclip` step, and runs each request under
+ // the run parent context (see `runWithRuntimeParent` inside that function).
+ // So the startup `mkdir` execs stay parented to the step, and every later
+ // request `sandbox.exec` span parents to the live run span.
worker = await startSandboxCallbackBridgeWorker({
client,
queueDir,
maxBodyBytes,
+ getRuntimeParentContext: input.getRuntimeParentContext,
+ runtimeSpan: input.runtimeSpan,
handleRequest: async (request) => {
const method = request.method.trim().toUpperCase() || "GET";
if (bridgeDebugEnabled) {
diff --git a/packages/adapter-utils/src/git-workspace-sync.test.ts b/packages/adapter-utils/src/git-workspace-sync.test.ts
index 596cd709816b..3f5b70ca8c63 100644
--- a/packages/adapter-utils/src/git-workspace-sync.test.ts
+++ b/packages/adapter-utils/src/git-workspace-sync.test.ts
@@ -14,6 +14,7 @@ import {
isMissingGitPrerequisiteError,
readGitWorkspaceSnapshot,
runLocalGit,
+ sanitizeGitRemoteUrl,
withShallowGitWorkspaceClone,
} from "./git-workspace-sync.js";
@@ -73,6 +74,101 @@ describe("git workspace sync", () => {
});
});
+ it("copies the workspace origin remote into the shallow clone", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-origin-"));
+ cleanupDirs.push(rootDir);
+ const repo = await createRepo(rootDir);
+ await git(repo, ["remote", "add", "origin", "https://github.com/example/repo.git"]);
+
+ const snapshot = await readGitWorkspaceSnapshot(repo);
+ await withShallowGitWorkspaceClone({
+ localDir: repo,
+ snapshot: snapshot!,
+ }, async (cloneDir) => {
+ expect(await git(cloneDir, ["remote", "get-url", "origin"])).toBe("https://github.com/example/repo.git");
+ });
+ });
+
+ it("scrubs credentials from the origin remote before copying it", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-origin-scrub-"));
+ cleanupDirs.push(rootDir);
+ const repo = await createRepo(rootDir);
+ await git(repo, ["remote", "add", "origin", "https://x-access-token:sekret@github.com/example/repo.git"]);
+
+ const snapshot = await readGitWorkspaceSnapshot(repo);
+ await withShallowGitWorkspaceClone({
+ localDir: repo,
+ snapshot: snapshot!,
+ }, async (cloneDir) => {
+ expect(await git(cloneDir, ["remote", "get-url", "origin"])).toBe("https://github.com/example/repo.git");
+ });
+ });
+
+ it("leaves the shallow clone remote-less when the workspace has no origin", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-no-origin-"));
+ cleanupDirs.push(rootDir);
+ const repo = await createRepo(rootDir);
+
+ const snapshot = await readGitWorkspaceSnapshot(repo);
+ await withShallowGitWorkspaceClone({
+ localDir: repo,
+ snapshot: snapshot!,
+ }, async (cloneDir) => {
+ await expect(git(cloneDir, ["remote", "get-url", "origin"])).rejects.toThrow();
+ });
+ });
+
+ it("drops a filesystem-path origin instead of copying it into the shallow clone", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-path-origin-"));
+ cleanupDirs.push(rootDir);
+ const repo = await createRepo(rootDir);
+ await git(repo, ["remote", "add", "origin", path.join(rootDir, "elsewhere.git")]);
+
+ const snapshot = await readGitWorkspaceSnapshot(repo);
+ await withShallowGitWorkspaceClone({
+ localDir: repo,
+ snapshot: snapshot!,
+ }, async (cloneDir) => {
+ await expect(git(cloneDir, ["remote", "get-url", "origin"])).rejects.toThrow();
+ });
+ });
+
+ it("pushes new commits from the shallow clone to an origin that holds the base commit", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-shallow-push-"));
+ cleanupDirs.push(rootDir);
+ const repo = await createRepo(rootDir);
+ const upstream = path.join(rootDir, "upstream.git");
+ await mkdir(upstream, { recursive: true });
+ await git(upstream, ["init", "--bare"]);
+ await git(repo, ["remote", "add", "origin", upstream]);
+ await git(repo, ["push", "origin", "main"]);
+ const baseHead = await git(repo, ["rev-parse", "HEAD"]);
+
+ const snapshot = await readGitWorkspaceSnapshot(repo);
+ await withShallowGitWorkspaceClone({
+ localDir: repo,
+ snapshot: snapshot!,
+ }, async (cloneDir) => {
+ await git(cloneDir, ["config", "user.name", "Paperclip Sandbox"]);
+ await git(cloneDir, ["config", "user.email", "sandbox@paperclip.dev"]);
+ await writeFile(path.join(cloneDir, "change.txt"), "sandbox change\n", "utf8");
+ await git(cloneDir, ["add", "change.txt"]);
+ await git(cloneDir, ["commit", "-m", "sandbox change"]);
+ const cloneHead = await git(cloneDir, ["rev-parse", "HEAD"]);
+
+ // A filesystem-path origin is dropped by the allowlist, so configure the
+ // remote explicitly — the property under test is the push itself: the
+ // clone is shallow (single grafted commit), but the boundary commit
+ // already exists on the origin, so the push pack closes without full
+ // ancestry. That is what makes transported branches publishable.
+ await git(cloneDir, ["remote", "add", "origin", upstream]);
+ await git(cloneDir, ["push", "origin", "HEAD:refs/heads/sandbox-change"]);
+
+ expect(await git(upstream, ["rev-parse", "refs/heads/sandbox-change"])).toBe(cloneHead);
+ expect(await git(upstream, ["merge-base", "refs/heads/main", "refs/heads/sandbox-change"])).toBe(baseHead);
+ });
+ });
+
it("builds thin git delta bundles relative to the imported base", async () => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-delta-"));
cleanupDirs.push(rootDir);
@@ -300,3 +396,47 @@ describe("git workspace sync", () => {
}
});
});
+
+describe("sanitizeGitRemoteUrl", () => {
+ it("strips userinfo, query, and fragment from http(s) URLs", () => {
+ expect(sanitizeGitRemoteUrl("https://x-access-token:sekret@github.com/example/repo.git"))
+ .toBe("https://github.com/example/repo.git");
+ expect(sanitizeGitRemoteUrl("https://sekret-token@github.com/example/repo.git"))
+ .toBe("https://github.com/example/repo.git");
+ expect(sanitizeGitRemoteUrl("http://user:pass@git.internal/example/repo.git"))
+ .toBe("http://git.internal/example/repo.git");
+ expect(sanitizeGitRemoteUrl("https://github.com/example/repo.git?private_token=sekret#fragment"))
+ .toBe("https://github.com/example/repo.git");
+ });
+
+ it("strips password and query from ssh-scheme URLs but keeps the username", () => {
+ expect(sanitizeGitRemoteUrl("ssh://git@github.com/example/repo.git"))
+ .toBe("ssh://git@github.com/example/repo.git");
+ expect(sanitizeGitRemoteUrl("ssh://git:sekret@github.com/example/repo.git"))
+ .toBe("ssh://git@github.com/example/repo.git");
+ expect(sanitizeGitRemoteUrl("git+ssh://git@github.com/example/repo.git?key=sekret"))
+ .toBe("git+ssh://git@github.com/example/repo.git");
+ });
+
+ it("keeps credential-free scp-like remotes unchanged", () => {
+ expect(sanitizeGitRemoteUrl("git@github.com:example/repo.git"))
+ .toBe("git@github.com:example/repo.git");
+ expect(sanitizeGitRemoteUrl("https://github.com/example/repo.git"))
+ .toBe("https://github.com/example/repo.git");
+ });
+
+ it("drops every shape whose credential surface is unknown", () => {
+ // Filesystem paths are useless on the execution host and could leak
+ // host-layout details; unknown schemes and malformed userinfo could carry
+ // embedded secrets the sanitizer cannot recognize. All fail closed.
+ expect(sanitizeGitRemoteUrl("/tmp/local/upstream.git")).toBeNull();
+ expect(sanitizeGitRemoteUrl("ftp://user:pass@host/repo.git")).toBeNull();
+ expect(sanitizeGitRemoteUrl("user:pass@host:path/repo.git")).toBeNull();
+ expect(sanitizeGitRemoteUrl("host.example:path/repo.git")).toBeNull();
+ });
+
+ it("returns null for empty input", () => {
+ expect(sanitizeGitRemoteUrl("")).toBeNull();
+ expect(sanitizeGitRemoteUrl(" ")).toBeNull();
+ });
+});
diff --git a/packages/adapter-utils/src/git-workspace-sync.ts b/packages/adapter-utils/src/git-workspace-sync.ts
index 0e36dee982af..dd0517ec41d9 100644
--- a/packages/adapter-utils/src/git-workspace-sync.ts
+++ b/packages/adapter-utils/src/git-workspace-sync.ts
@@ -110,6 +110,67 @@ export async function readGitWorkspaceSnapshot(localDir: string): Promise {
+ try {
+ const result = await runLocalGit(localDir, ["remote", "get-url", "origin"], {
+ timeout: 10_000,
+ maxBuffer: 16 * 1024,
+ });
+ return sanitizeGitRemoteUrl(result.stdout.trim());
+ } catch {
+ return null;
+ }
+}
+
export async function withShallowGitWorkspaceClone(
input: {
localDir: string;
@@ -120,6 +181,7 @@ export async function withShallowGitWorkspaceClone(
const cloneDir = await fs.mkdtemp(path.join(os.tmpdir(), "paperclip-git-workspace-"));
const tempRef = `refs/paperclip/git-sync/import/${randomUUID()}`;
try {
+ const originUrl = await readSanitizedOriginRemoteUrl(input.localDir);
await runLocalGit(input.localDir, ["update-ref", tempRef, input.snapshot.headCommit], {
timeout: 10_000,
maxBuffer: 16 * 1024,
@@ -128,6 +190,19 @@ export async function withShallowGitWorkspaceClone(
timeout: 10_000,
maxBuffer: 64 * 1024,
});
+ if (originUrl) {
+ // The clone is what lands in the sandbox. Without `origin`, the branch
+ // there reads as an unpublishable root snapshot even though its head is a
+ // commit the upstream remote already holds — so fetch (to reconnect
+ // ancestry) and push (to publish the branch; the shallow boundary commit
+ // is already on the remote, so the pack closes) are both mechanically
+ // possible once the remote is carried over. Best-effort: a failure to
+ // record the remote must not fail the transport.
+ await runLocalGit(cloneDir, ["remote", "add", "origin", originUrl], {
+ timeout: 10_000,
+ maxBuffer: 16 * 1024,
+ }).catch(() => undefined);
+ }
await runLocalGit(cloneDir, ["fetch", "--depth=1", input.localDir, tempRef], {
timeout: 60_000,
maxBuffer: 1024 * 1024,
diff --git a/packages/adapter-utils/src/index.ts b/packages/adapter-utils/src/index.ts
index dc5db8865f08..5df19260aabd 100644
--- a/packages/adapter-utils/src/index.ts
+++ b/packages/adapter-utils/src/index.ts
@@ -65,6 +65,11 @@ export {
redactCommandText,
} from "./command-redaction.js";
export { buildSandboxNpmInstallCommand } from "./sandbox-install-command.js";
+export {
+ buildAdapterEnvConfig,
+ parseEnvBindings,
+ parseEnvVars,
+} from "./env-bindings.js";
export { createRuntimeProgressReporter } from "./runtime-progress.js";
export type {
RuntimeProgressSink,
diff --git a/packages/adapter-utils/src/sandbox-callback-bridge.test.ts b/packages/adapter-utils/src/sandbox-callback-bridge.test.ts
index 73fc3b472a47..e69fef32c8c1 100644
--- a/packages/adapter-utils/src/sandbox-callback-bridge.test.ts
+++ b/packages/adapter-utils/src/sandbox-callback-bridge.test.ts
@@ -5,6 +5,7 @@ import path from "node:path";
import { promisify } from "node:util";
import { afterEach, describe, expect, it, vi } from "vitest";
+import { getActiveStepContext, measureStartupStep } from "./acpx-engine/startup-timing.js";
import { prepareCommandManagedRuntime } from "./command-managed-runtime.js";
import {
authorizeSandboxCallbackBridgeRequestWithRoutes,
@@ -471,6 +472,242 @@ describe("sandbox callback bridge", () => {
}
});
+ it("keeps the queue-directory setup on the startup step but resets the poll loop store", async () => {
+ // The worker starts inside the measured `bridge.paperclip` step. Its awaited
+ // queue-directory setup is startup work, so a `makeDir` `sandbox.exec` span
+ // must keep the active step and its `criticalPath` flag. The long-lived poll
+ // loop runs run-time execs for the whole run, so a loop `sandbox.exec` span
+ // must open unparented with no stale flag. This test reads the active step in
+ // both places and proves the boundary sits at the loop, not the whole worker.
+ let setupStep: ReturnType | "unset" = "unset";
+ let loopStep: ReturnType | "unset" = "unset";
+ let resolveFirstPoll: () => void = () => {};
+ const firstPoll = new Promise((resolve) => {
+ resolveFirstPoll = resolve;
+ });
+
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-bridge-step-store-"));
+ cleanupDirs.push(rootDir);
+ const queueDir = path.posix.join(rootDir, "queue");
+
+ const worker = await measureStartupStep(
+ {},
+ () => 0,
+ "bridge.paperclip",
+ () =>
+ startSandboxCallbackBridgeWorker({
+ client: {
+ makeDir: async () => {
+ setupStep = getActiveStepContext();
+ },
+ makeDirs: async () => {
+ setupStep = getActiveStepContext();
+ },
+ listJsonFiles: async () => {
+ loopStep = getActiveStepContext();
+ resolveFirstPoll();
+ return [];
+ },
+ readTextFile: async () => {
+ throw new Error("unexpected readTextFile");
+ },
+ writeTextFile: async () => {
+ throw new Error("unexpected writeTextFile");
+ },
+ rename: async () => {
+ throw new Error("unexpected rename");
+ },
+ remove: async () => {},
+ },
+ queueDir,
+ authorizeRequest: async () => null,
+ handleRequest: async () => ({ status: 200, body: "ok" }),
+ }),
+ { criticalPath: false },
+ );
+
+ await firstPoll;
+ await worker.stop();
+
+ // The setup ran on the active step, so its exec span parents to the step.
+ expect(setupStep).not.toBe("unset");
+ expect(setupStep).not.toBeNull();
+ expect((setupStep as { criticalPath?: boolean }).criticalPath).toBe(false);
+
+ // The loop ran outside that store, so its exec span opens unparented with no
+ // stale `criticalPath` flag.
+ expect(loopStep).toBeNull();
+ });
+
+ it("test_paperclip_loop_exec_parents_to_run_context", async () => {
+ // The worker starts inside the measured `bridge.paperclip` step. Its awaited
+ // queue-directory setup is startup work and keeps the active step. The poll
+ // loop shell stays outside that store. But a per-request unit of work is
+ // run-time work, so the worker runs each request under the current-run
+ // parent context. A request `sandbox.exec` span then parents to the live run
+ // span, not to the ended startup step. This test drives the worker with a
+ // `getRuntimeParentContext` that returns a known token, queues one request,
+ // and proves the request work reads that token from the active step store.
+ const runParentToken = { marker: "run-parent-token" };
+ let setupStep: ReturnType | "unset" = "unset";
+ let requestStep: ReturnType | "unset" = "unset";
+ let served = false;
+ let resolveServed: () => void = () => {};
+ const requestServed = new Promise((resolve) => {
+ resolveServed = resolve;
+ });
+
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-bridge-run-parent-"));
+ cleanupDirs.push(rootDir);
+ const queueDir = path.posix.join(rootDir, "queue");
+
+ const worker = await measureStartupStep(
+ {},
+ () => 0,
+ "bridge.paperclip",
+ () =>
+ startSandboxCallbackBridgeWorker({
+ client: {
+ makeDir: async () => {
+ setupStep = getActiveStepContext();
+ },
+ makeDirs: async () => {
+ setupStep = getActiveStepContext();
+ },
+ // Return one request on the first poll, then nothing.
+ listJsonFiles: async () => (served ? [] : ["000000000001.json"]),
+ readTextFile: async () =>
+ JSON.stringify({ id: "req-1", method: "GET", path: "/", query: "", headers: {}, body: "" }),
+ writeTextFile: async () => {},
+ rename: async () => {},
+ remove: async () => {},
+ },
+ queueDir,
+ authorizeRequest: async () => null,
+ handleRequest: async () => {
+ requestStep = getActiveStepContext();
+ served = true;
+ resolveServed();
+ return { status: 200, body: "ok" };
+ },
+ getRuntimeParentContext: () => runParentToken,
+ }),
+ { criticalPath: false },
+ );
+
+ await requestServed;
+ await worker.stop();
+
+ // The setup ran on the active step, so its exec span parents to the step.
+ expect(setupStep).not.toBe("unset");
+ expect(setupStep).not.toBeNull();
+
+ // The request work ran under the run parent context. Its exec span parents
+ // to the run token, not to the ended startup step, and it carries no
+ // startup `criticalPath` flag.
+ expect(requestStep).not.toBe("unset");
+ expect(requestStep).not.toBeNull();
+ expect((requestStep as { parentContext?: unknown }).parentContext).toBe(runParentToken);
+ expect((requestStep as { criticalPath?: boolean }).criticalPath).toBe(false);
+ });
+
+ it("wraps each request in a sandbox.callbackBridge.relayRequest span", async () => {
+ // With a span runner injected, the worker wraps each request in one
+ // `sandbox.callbackBridge.relayRequest` span, so the request's read, write,
+ // and remove execs group under one named span. This test drives the worker
+ // with a recording runner and proves it opens the wrapper span around the
+ // request work.
+ const wrapped: string[] = [];
+ let served = false;
+ let resolveServed: () => void = () => {};
+ const requestServed = new Promise((resolve) => {
+ resolveServed = resolve;
+ });
+
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-bridge-relay-span-"));
+ cleanupDirs.push(rootDir);
+ const queueDir = path.posix.join(rootDir, "queue");
+
+ const worker = await startSandboxCallbackBridgeWorker({
+ client: {
+ makeDir: async () => {},
+ makeDirs: async () => {},
+ listJsonFiles: async () => (served ? [] : ["000000000001.json"]),
+ readTextFile: async () =>
+ JSON.stringify({ id: "req-1", method: "GET", path: "/", query: "", headers: {}, body: "" }),
+ writeTextFile: async () => {},
+ rename: async () => {},
+ remove: async () => {},
+ },
+ queueDir,
+ authorizeRequest: async () => null,
+ handleRequest: async () => {
+ served = true;
+ resolveServed();
+ return { status: 200, body: "ok" };
+ },
+ // Record each wrapper span name, then run the wrapped work.
+ runtimeSpan: async (name, work) => {
+ wrapped.push(name);
+ return work();
+ },
+ });
+
+ await requestServed;
+ await worker.stop();
+
+ expect(wrapped).toContain("sandbox.callbackBridge.relayRequest");
+ });
+
+ it("test_paperclip_loop_exec_stays_unparented_without_getter", async () => {
+ // With no `getRuntimeParentContext`, a request runs with an empty active
+ // step store, exactly like the earlier `runWithoutActiveStep` behavior. So a
+ // request `sandbox.exec` span opens unparented with no stale startup flag.
+ let requestStep: ReturnType | "unset" = "unset";
+ let served = false;
+ let resolveServed: () => void = () => {};
+ const requestServed = new Promise((resolve) => {
+ resolveServed = resolve;
+ });
+
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-bridge-no-getter-"));
+ cleanupDirs.push(rootDir);
+ const queueDir = path.posix.join(rootDir, "queue");
+
+ const worker = await measureStartupStep(
+ {},
+ () => 0,
+ "bridge.paperclip",
+ () =>
+ startSandboxCallbackBridgeWorker({
+ client: {
+ makeDir: async () => {},
+ makeDirs: async () => {},
+ listJsonFiles: async () => (served ? [] : ["000000000001.json"]),
+ readTextFile: async () =>
+ JSON.stringify({ id: "req-1", method: "GET", path: "/", query: "", headers: {}, body: "" }),
+ writeTextFile: async () => {},
+ rename: async () => {},
+ remove: async () => {},
+ },
+ queueDir,
+ authorizeRequest: async () => null,
+ handleRequest: async () => {
+ requestStep = getActiveStepContext();
+ served = true;
+ resolveServed();
+ return { status: 200, body: "ok" };
+ },
+ }),
+ { criticalPath: false },
+ );
+
+ await requestServed;
+ await worker.stop();
+
+ expect(requestStep).toBeNull();
+ });
+
it("serializes remote response writes so stop does not recreate a late orphaned response", async () => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-bridge-response-lock-"));
cleanupDirs.push(rootDir);
@@ -1036,6 +1273,8 @@ describe("sandbox callback bridge", () => {
{ method: "POST", path: "/api/issues/issue-1/interactions/inter-1/accept" },
{ method: "POST", path: "/api/issues/issue-1/interactions/inter-1/reject" },
{ method: "POST", path: "/api/issues/issue-1/interactions/inter-1/respond" },
+ { method: "POST", path: "/api/issues/issue-1/interactions/inter-1/verdicts" },
+ { method: "POST", path: "/api/issues/issue-1/interactions/inter-1/withdraw" },
{ method: "POST", path: "/api/companies/co-1/issues" },
{ method: "GET", path: "/api/approvals/ap-1" },
{ method: "GET", path: "/api/approvals/ap-1/issues" },
diff --git a/packages/adapter-utils/src/sandbox-callback-bridge.ts b/packages/adapter-utils/src/sandbox-callback-bridge.ts
index d990adbef3b1..35ee47db7022 100644
--- a/packages/adapter-utils/src/sandbox-callback-bridge.ts
+++ b/packages/adapter-utils/src/sandbox-callback-bridge.ts
@@ -3,6 +3,12 @@ import { promises as fs } from "node:fs";
import os from "node:os";
import path from "node:path";
+import {
+ runWithoutActiveStep,
+ runWithRuntimeParent,
+ type RuntimeSpanRunner,
+ type StartupSpanContext,
+} from "./acpx-engine/startup-timing.js";
import type { CommandManagedRuntimeRunner } from "./command-managed-runtime.js";
import { preferredShellForSandbox, shellCommandArgs } from "./sandbox-shell.js";
import type { RunProcessResult } from "./server-utils.js";
@@ -18,6 +24,10 @@ const SANDBOX_CALLBACK_BRIDGE_ENTRYPOINT = "paperclip-bridge-server.mjs";
const SANDBOX_EXEC_CHANNEL_ENV = "PAPERCLIP_SANDBOX_EXEC_CHANNEL";
const SANDBOX_EXEC_CHANNEL_BRIDGE = "bridge";
+/** Span name that wraps one Paperclip-API callback request — read the request,
+ * write the response, and remove the request file. */
+const CALLBACK_BRIDGE_RELAY_REQUEST_SPAN = "sandbox.callbackBridge.relayRequest";
+
export const DEFAULT_SANDBOX_CALLBACK_BRIDGE_MAX_BODY_BYTES = DEFAULT_BRIDGE_MAX_BODY_BYTES;
export interface SandboxCallbackBridgeRouteRule {
@@ -72,10 +82,10 @@ export const DEFAULT_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST: readonly SandboxCa
{ method: "POST", path: /^\/api\/issues\/[^/]+\/work-products$/ },
{ method: "PATCH", path: /^\/api\/work-products\/[^/]+$/ },
- // Issue-thread interactions (suggest tasks, ask questions, request confirmation)
+ // Issue-thread interactions (create, resolve, verdict, and withdraw)
{ method: "GET", path: /^\/api\/issues\/[^/]+\/interactions(?:\/[^/]+)?$/ },
{ method: "POST", path: /^\/api\/issues\/[^/]+\/interactions$/ },
- { method: "POST", path: /^\/api\/issues\/[^/]+\/interactions\/[^/]+\/(?:accept|reject|respond)$/ },
+ { method: "POST", path: /^\/api\/issues\/[^/]+\/interactions\/[^/]+\/(?:accept|reject|respond|verdicts|withdraw)$/ },
// Subtasks / delegation
{ method: "POST", path: /^\/api\/companies\/[^/]+\/issues$/ },
@@ -226,6 +236,13 @@ async function runShell(
},
timeoutMs,
stdin,
+ // Every command that rides this helper is bridge control-plane plumbing:
+ // input delivery, output read, callback relay, and queue/setup bookkeeping.
+ // It must run concurrently with the agent, so force it off the persistent
+ // session. In streamed mode the agent holds that single serialized session
+ // for the whole run; a control write on the same session queues behind the
+ // agent command that never returns — a permanent deadlock.
+ bypassSession: true,
});
}
@@ -619,6 +636,18 @@ export async function startSandboxCallbackBridgeWorker(input: {
body?: string;
}>;
maxBodyBytes?: number | null;
+ // Return the current-run parent-context token. The worker reads it per request
+ // and runs the request work under it, so the request `sandbox.exec` span
+ // parents to the live run span (`agent.turn` during the turn, `task.run`
+ // otherwise). When it is absent, the request work runs with an empty store,
+ // exactly like the earlier `runWithoutActiveStep` behavior.
+ getRuntimeParentContext?: () => StartupSpanContext | undefined;
+ // Wrap each Paperclip-API callback request in a
+ // `sandbox.callbackBridge.relayRequest` span, so the request's read, write, and
+ // remove execs group under one named span. When it is absent, the request work
+ // runs under the run parent with no wrapper span, exactly like the earlier
+ // behavior.
+ runtimeSpan?: RuntimeSpanRunner;
}): Promise {
const pollIntervalMs = normalizeTimeoutMs(input.pollIntervalMs, DEFAULT_BRIDGE_POLL_INTERVAL_MS);
const maxBodyBytes = normalizeTimeoutMs(input.maxBodyBytes, DEFAULT_BRIDGE_MAX_BODY_BYTES);
@@ -745,7 +774,13 @@ export async function startSandboxCallbackBridgeWorker(input: {
}
};
- const loop = (async () => {
+ // Start the long-lived poll loop outside the measured startup-step store.
+ // The `makeDir` calls above are startup work and must keep the active
+ // `bridge.paperclip` step. The loop runs run-time execs for the whole run,
+ // so each loop `sandbox.exec` span must not parent to the ended step or copy
+ // its `criticalPath` flag. `runWithoutActiveStep` empties the store for the
+ // loop only; Node keeps the empty store on every later poll continuation.
+ const loop = runWithoutActiveStep(() => (async () => {
try {
while (true) {
const fileNames = await input.client.listJsonFiles(directories.requestsDir);
@@ -760,7 +795,20 @@ export async function startSandboxCallbackBridgeWorker(input: {
if (stopping && Date.now() >= stopDeadline) break;
inFlight += 1;
try {
- await processRequestFile(fileName);
+ // A request is run-time work, not startup work. Wrap it in a
+ // `sandbox.callbackBridge.relayRequest` span, so its read, write, and
+ // remove execs group under one named span that parents to the live
+ // run span. The span runner reads the run parent per request: the
+ // live parent switches to `agent.turn` during the turn and back to
+ // `task.run` after it. Without a runner, the request runs under the
+ // run parent with no wrapper span, exactly like the earlier behavior.
+ await (input.runtimeSpan
+ ? input.runtimeSpan(CALLBACK_BRIDGE_RELAY_REQUEST_SPAN, () =>
+ processRequestFile(fileName),
+ )
+ : runWithRuntimeParent(input.getRuntimeParentContext?.(), () =>
+ processRequestFile(fileName),
+ ));
} finally {
inFlight -= 1;
}
@@ -785,7 +833,7 @@ export async function startSandboxCallbackBridgeWorker(input: {
settleResolve();
}
}
- })();
+ })());
void loop;
diff --git a/packages/adapter-utils/src/sandbox-managed-runtime.test.ts b/packages/adapter-utils/src/sandbox-managed-runtime.test.ts
index 4e82cb0d86c8..52c3cb65db0c 100644
--- a/packages/adapter-utils/src/sandbox-managed-runtime.test.ts
+++ b/packages/adapter-utils/src/sandbox-managed-runtime.test.ts
@@ -19,6 +19,15 @@ import {
prepareCommandManagedRuntime,
type CommandManagedRuntimeRunner,
} from "./command-managed-runtime.js";
+import {
+ createRuntimeSpanRunner,
+ getActiveStepContext,
+ measureStartupStep,
+ type RuntimeSpanRunner,
+ type StartupSpan,
+ type StartupTraceContext,
+ type StartupTracer,
+} from "./acpx-engine/startup-timing.js";
import type { RunProcessResult } from "./server-utils.js";
function toArrayBuffer(bytes: Buffer): ArrayBuffer {
@@ -156,6 +165,87 @@ async function listTarMembers(rootDir: string, name: string, bytes: Buffer): Pro
return stdout.split("\n").map((line) => line.trim()).filter(Boolean);
}
+// Build a filesystem-backed managed-runtime client. The host tarball path runs
+// unchanged; the client just materializes the mappings on the local disk, so a
+// pack-span test needs no provider.
+function makeFilesystemClient(): SandboxManagedRuntimeClient {
+ const client: SandboxManagedRuntimeClient = {
+ makeDir: async (remotePath) => {
+ await mkdir(remotePath, { recursive: true });
+ },
+ writeFile: async (remotePath, bytes) => {
+ await mkdir(path.dirname(remotePath), { recursive: true });
+ await writeFile(remotePath, Buffer.from(bytes));
+ },
+ readFile: async (remotePath) => await readFile(remotePath),
+ listFiles: async (remotePath) => {
+ const entries = await readdir(remotePath, { withFileTypes: true }).catch(() => []);
+ return entries
+ .filter((entry) => entry.isFile())
+ .map((entry) => entry.name)
+ .sort((left, right) => left.localeCompare(right));
+ },
+ remove: async (remotePath) => {
+ await rm(remotePath, { recursive: true, force: true });
+ },
+ run: async (command) => {
+ await execFile("sh", ["-c", command], { maxBuffer: 32 * 1024 * 1024 });
+ },
+ };
+ attachFallbackSyncIn(client);
+ return client;
+}
+
+// One recorded span from the fake tracer. `parentName` is the name of the span
+// that the start context carried, so a test can assert the parent relationship.
+interface RecordedSpan {
+ name: string;
+ parentName: string | null;
+ ended: boolean;
+ attributes: Record;
+}
+
+// A fake trace context that records every span and its parent by name. It
+// satisfies the structural `StartupTraceContext` contract, so the real
+// `createRuntimeSpanRunner` and `measureStartupStep` drive it unchanged. The
+// opaque parent token is the parent's `RecordedSpan`, so a child span reads its
+// parent name from the start context.
+function createRecordingTraceContext(): {
+ traceContext: StartupTraceContext;
+ spans: RecordedSpan[];
+} {
+ const spans: RecordedSpan[] = [];
+ const byHandle = new WeakMap();
+ const tracer: StartupTracer = {
+ startSpan(name, options, context) {
+ const parent = context as RecordedSpan | undefined;
+ const record: RecordedSpan = {
+ name,
+ parentName: parent?.name ?? null,
+ ended: false,
+ attributes: { ...(options?.attributes ?? {}) },
+ };
+ spans.push(record);
+ const handle: StartupSpan = {
+ setAttribute(key, value) {
+ record.attributes[key] = value;
+ },
+ setStatus() {},
+ end() {
+ record.ended = true;
+ },
+ };
+ byHandle.set(handle, record);
+ return handle;
+ },
+ };
+ const traceContext: StartupTraceContext = {
+ tracer,
+ contextWithSpan: (span) => byHandle.get(span),
+ };
+ return { traceContext, spans };
+}
+
describe("sandbox managed runtime", () => {
const cleanupDirs: string[] = [];
@@ -1929,4 +2019,106 @@ describe("sandbox managed runtime", () => {
else process.env[flagKey] = priorFlag;
}
});
+
+ it("builds the workspace tarball inside one host pack span for a usual workspace sync", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-pack-span-"));
+ cleanupDirs.push(rootDir);
+ const localWorkspaceDir = path.join(rootDir, "local-workspace");
+ const remoteWorkspaceDir = path.join(rootDir, "remote-workspace");
+ await mkdir(localWorkspaceDir, { recursive: true });
+ await writeFile(path.join(localWorkspaceDir, "README.md"), "workspace body\n", "utf8");
+
+ // Record every span name the runner opens and run the wrapped work, so the
+ // test proves the host opens exactly one `pack` span around the tarball
+ // build for the usual (plain) workspace sync.
+ const openedSpans: string[] = [];
+ const runtimeSpan: RuntimeSpanRunner = async (name, work) => {
+ openedSpans.push(name);
+ return await work();
+ };
+
+ const prepared = await prepareSandboxManagedRuntime({
+ spec: {
+ transport: "sandbox",
+ provider: "test",
+ sandboxId: "sandbox-pack",
+ remoteCwd: remoteWorkspaceDir,
+ timeoutMs: 30_000,
+ apiKey: null,
+ },
+ adapterKey: "test-adapter",
+ client: makeFilesystemClient(),
+ workspaceLocalDir: localWorkspaceDir,
+ runtimeSpan,
+ });
+
+ expect(openedSpans).toEqual(["pack"]);
+ // The tarball build still lands the workspace inside the span, so the wrap
+ // changes no staging behavior.
+ await expect(readFile(path.join(remoteWorkspaceDir, "README.md"), "utf8")).resolves.toBe("workspace body\n");
+ expect(prepared.workspaceRemoteDir).toBe(remoteWorkspaceDir);
+ });
+
+ it("nests the host pack span under the stage.sync step span", async () => {
+ const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-pack-nest-"));
+ cleanupDirs.push(rootDir);
+ const localWorkspaceDir = path.join(rootDir, "local-workspace");
+ const remoteWorkspaceDir = path.join(rootDir, "remote-workspace");
+ await mkdir(localWorkspaceDir, { recursive: true });
+ await writeFile(path.join(localWorkspaceDir, "README.md"), "workspace body\n", "utf8");
+
+ const { traceContext, spans } = createRecordingTraceContext();
+ // The root span stands in for `sandbox.startup`. Its child context is the
+ // step span's parent, exactly as the executor wires it.
+ const rootHandle = traceContext.tracer.startSpan("sandbox.startup", undefined, undefined);
+ const rootContext = traceContext.contextWithSpan(rootHandle);
+
+ // The stage runner parents each span to the ACTIVE startup step, so the
+ // `pack` span nests under `stage.sync`. This is the exact runner the
+ // executor threads into the staging seam.
+ const stageRuntimeSpan = createRuntimeSpanRunner(
+ traceContext,
+ () => getActiveStepContext()?.parentContext,
+ );
+
+ // A deterministic monotonic clock, so the step timing stays test-stable.
+ let clock = 0;
+ const now = () => (clock += 1000);
+
+ await measureStartupStep(
+ {},
+ now,
+ "stage.sync",
+ async () => {
+ await prepareSandboxManagedRuntime({
+ spec: {
+ transport: "sandbox",
+ provider: "test",
+ sandboxId: "sandbox-nest",
+ remoteCwd: remoteWorkspaceDir,
+ timeoutMs: 30_000,
+ apiKey: null,
+ },
+ adapterKey: "test-adapter",
+ client: makeFilesystemClient(),
+ workspaceLocalDir: localWorkspaceDir,
+ runtimeSpan: stageRuntimeSpan,
+ });
+ },
+ {
+ tracer: traceContext.tracer,
+ parentContext: rootContext,
+ contextWithSpan: (span) => traceContext.contextWithSpan(span),
+ },
+ );
+
+ const stageSpan = spans.find((span) => span.name === "stage.sync");
+ const packSpan = spans.find((span) => span.name === "pack");
+ expect(stageSpan).toBeDefined();
+ expect(packSpan).toBeDefined();
+ expect(packSpan!.ended).toBe(true);
+ // The `pack` span parents to `stage.sync`, not to the root span, so it nests
+ // under the step in a real trace.
+ expect(packSpan!.parentName).toBe("stage.sync");
+ });
});
diff --git a/packages/adapter-utils/src/sandbox-managed-runtime.ts b/packages/adapter-utils/src/sandbox-managed-runtime.ts
index 2f9db88dc8f6..4169ced4a77a 100644
--- a/packages/adapter-utils/src/sandbox-managed-runtime.ts
+++ b/packages/adapter-utils/src/sandbox-managed-runtime.ts
@@ -27,6 +27,7 @@ import {
type RuntimeStatusSink,
} from "./runtime-progress.js";
import { isRelativePathOrDescendant, shouldExcludePath } from "./exclude-patterns.js";
+import type { RuntimeSpanRunner } from "./acpx-engine/startup-timing.js";
const execFile = promisify(execFileCallback);
const SANDBOX_WORKSPACE_HEAVY_DIR_NAMES = [
@@ -703,6 +704,12 @@ export async function prepareSandboxManagedRuntime(input: {
// child task wires it into writeFile/readFile.
onProgress?: RuntimeProgressSink;
onRuntimeProgress?: RuntimeStatusSink;
+ // Optional host span runner for the workspace tarball build. When present, the
+ // host builds both workspace tarballs inside one span named `pack`, so the
+ // host pack time is visible under the `stage.sync` step. The default is a
+ // no-op that keeps the current behavior and control flow. A throwing runner
+ // never changes control flow (see `createRuntimeSpanRunner`).
+ runtimeSpan?: RuntimeSpanRunner;
}): Promise {
const workspaceRemoteDir = input.workspaceRemoteDir ?? input.spec.remoteCwd;
const runtimeRootDir = path.posix.join(workspaceRemoteDir, ".paperclip-runtime", input.adapterKey);
@@ -835,73 +842,84 @@ export async function prepareSandboxManagedRuntime(input: {
const workspacePostUploadCommands: SandboxPostUploadCommand[] = [];
let workspaceUploadBytes = 0;
- // 1. git-history tar (git-backed workspace only). Both tar targets live under
- // `runtimeRootDir` (`.paperclip-runtime/`). The git extract
- // wipes the target tree EXCEPT `.paperclip-runtime`, so the overlay tar,
- // which sits under `.paperclip-runtime`, survives to run its own extract.
- if (gitSnapshot) {
- await emitRuntimeStatus(input.onRuntimeProgress, "git_sync", "Syncing git history to sandbox");
- const gitTarPath = path.join(tempDir, "git-workspace.tar");
- const remoteGitTar = path.posix.join(runtimeRootDir, "git-workspace-upload.tar");
- await withShallowGitWorkspaceClone({
- localDir: input.workspaceLocalDir,
- snapshot: gitSnapshot,
- }, async (cloneDir) => {
- await createTarballFromDirectory({
- localDir: cloneDir,
- archivePath: gitTarPath,
- exclude: [".paperclip-runtime"],
+ // Build both host tarballs (the git-history tar and the workspace-overlay
+ // tar) inside one host span named `pack`. This span makes the host pack
+ // time visible under the `stage.sync` step, where it is otherwise a hidden
+ // gap with no span. The transfer (`stageConfinedSyncIn`) runs after this
+ // span, so `pack` measures only the host tar-build cost. The runner
+ // defaults to a no-op, so a caller with no injected runner keeps the
+ // current behavior and control flow.
+ const runPackSpan = (work: () => Promise): Promise =>
+ input.runtimeSpan ? input.runtimeSpan("pack", work) : work();
+ await runPackSpan(async () => {
+ // 1. git-history tar (git-backed workspace only). Both tar targets live under
+ // `runtimeRootDir` (`.paperclip-runtime/`). The git extract
+ // wipes the target tree EXCEPT `.paperclip-runtime`, so the overlay tar,
+ // which sits under `.paperclip-runtime`, survives to run its own extract.
+ if (gitSnapshot) {
+ await emitRuntimeStatus(input.onRuntimeProgress, "git_sync", "Syncing git history to sandbox");
+ const gitTarPath = path.join(tempDir, "git-workspace.tar");
+ const remoteGitTar = path.posix.join(runtimeRootDir, "git-workspace-upload.tar");
+ await withShallowGitWorkspaceClone({
+ localDir: input.workspaceLocalDir,
+ snapshot: gitSnapshot,
+ }, async (cloneDir) => {
+ await createTarballFromDirectory({
+ localDir: cloneDir,
+ archivePath: gitTarPath,
+ exclude: [".paperclip-runtime"],
+ });
+ });
+ workspaceFiles.push({ sourcePath: gitTarPath, targetPath: remoteGitTar, kind: "file", access: "rw", writablePath: workspaceRemoteDir });
+ workspacePostUploadCommands.push({
+ command: buildWorkspaceTarExtractCommand({
+ workspaceRemoteDir,
+ remoteTar: remoteGitTar,
+ wipeExceptNames: [".paperclip-runtime"],
+ }),
+ });
+ workspaceUploadBytes += (await fs.stat(gitTarPath)).size;
+ }
+
+ // 2. workspace-overlay tar. A git-backed overlay merges on top of the just
+ // extracted git tree (no wipe); a plain workspace wipes every child except
+ // the preserved names first. The extract runs AFTER the git extract.
+ await emitRuntimeStatus(input.onRuntimeProgress, "config_sync", "Syncing workspace to sandbox");
+ const workspaceTarPath = path.join(tempDir, "workspace.tar");
+ const workspaceArchiveDir = gitSnapshot ? path.join(tempDir, "workspace-overlay") : input.workspaceLocalDir;
+ if (gitSnapshot) {
+ await copySelectedWorkspaceEntries({
+ sourceDir: input.workspaceLocalDir,
+ targetDir: workspaceArchiveDir,
+ relativePaths: gitSnapshot.overlayPaths,
+ exclude: workspaceArchiveExclude,
});
+ }
+ await createTarballFromDirectory({
+ localDir: workspaceArchiveDir,
+ archivePath: workspaceTarPath,
+ exclude: gitSnapshot ? undefined : workspaceArchiveExclude,
});
- workspaceFiles.push({ sourcePath: gitTarPath, targetPath: remoteGitTar, kind: "file", access: "rw", writablePath: workspaceRemoteDir });
+ const remoteWorkspaceTar = path.posix.join(runtimeRootDir, "workspace-upload.tar");
+ workspaceFiles.push({ sourcePath: workspaceTarPath, targetPath: remoteWorkspaceTar, kind: "file", access: "rw", writablePath: workspaceRemoteDir });
workspacePostUploadCommands.push({
command: buildWorkspaceTarExtractCommand({
workspaceRemoteDir,
- remoteTar: remoteGitTar,
- wipeExceptNames: [".paperclip-runtime"],
+ remoteTar: remoteWorkspaceTar,
+ wipeExceptNames: gitSnapshot ? null : [...preservedNames],
}),
});
- workspaceUploadBytes += (await fs.stat(gitTarPath)).size;
- }
-
- // 2. workspace-overlay tar. A git-backed overlay merges on top of the just
- // extracted git tree (no wipe); a plain workspace wipes every child except
- // the preserved names first. The extract runs AFTER the git extract.
- await emitRuntimeStatus(input.onRuntimeProgress, "config_sync", "Syncing workspace to sandbox");
- const workspaceTarPath = path.join(tempDir, "workspace.tar");
- const workspaceArchiveDir = gitSnapshot ? path.join(tempDir, "workspace-overlay") : input.workspaceLocalDir;
- if (gitSnapshot) {
- await copySelectedWorkspaceEntries({
- sourceDir: input.workspaceLocalDir,
- targetDir: workspaceArchiveDir,
- relativePaths: gitSnapshot.overlayPaths,
- exclude: workspaceArchiveExclude,
- });
- }
- await createTarballFromDirectory({
- localDir: workspaceArchiveDir,
- archivePath: workspaceTarPath,
- exclude: gitSnapshot ? undefined : workspaceArchiveExclude,
- });
- const remoteWorkspaceTar = path.posix.join(runtimeRootDir, "workspace-upload.tar");
- workspaceFiles.push({ sourcePath: workspaceTarPath, targetPath: remoteWorkspaceTar, kind: "file", access: "rw", writablePath: workspaceRemoteDir });
- workspacePostUploadCommands.push({
- command: buildWorkspaceTarExtractCommand({
- workspaceRemoteDir,
- remoteTar: remoteWorkspaceTar,
- wipeExceptNames: gitSnapshot ? null : [...preservedNames],
- }),
+ // 3. Optional remove-deleted-paths command runs LAST, after both extracts.
+ if (gitSnapshot && gitSnapshot.deletedPaths.length > 0) {
+ workspacePostUploadCommands.push({
+ command: buildRemoveDeletedPathsCommand({
+ remoteDir: workspaceRemoteDir,
+ deletedPaths: gitSnapshot.deletedPaths,
+ }),
+ });
+ }
+ workspaceUploadBytes += (await fs.stat(workspaceTarPath)).size;
});
- // 3. Optional remove-deleted-paths command runs LAST, after both extracts.
- if (gitSnapshot && gitSnapshot.deletedPaths.length > 0) {
- workspacePostUploadCommands.push({
- command: buildRemoveDeletedPathsCommand({
- remoteDir: workspaceRemoteDir,
- deletedPaths: gitSnapshot.deletedPaths,
- }),
- });
- }
- workspaceUploadBytes += (await fs.stat(workspaceTarPath)).size;
// One confined `syncIn` for the whole merged workspace file set. The confine
// guard covers every mapping BEFORE any bytes upload (fail-closed): a source
diff --git a/packages/adapter-utils/src/server-utils.test.ts b/packages/adapter-utils/src/server-utils.test.ts
index 6467973ab01d..2dd38ccdf418 100644
--- a/packages/adapter-utils/src/server-utils.test.ts
+++ b/packages/adapter-utils/src/server-utils.test.ts
@@ -742,6 +742,37 @@ describe("renderPaperclipWakePrompt", () => {
);
});
+ it("renders the simplified-english interaction directive only when the payload enables it", () => {
+ const payload = {
+ reason: "issue_commented",
+ issue: {
+ id: "issue-1",
+ identifier: "PAP-15936",
+ title: "Interaction language",
+ description: null,
+ descriptionTruncated: false,
+ status: "in_progress",
+ },
+ commentWindow: { requestedCount: 0, includedCount: 0, missingCount: 0 },
+ comments: [],
+ fallbackFetchNeeded: false,
+ };
+
+ expect(renderPaperclipWakePrompt(payload)).not.toContain("ASD-STE100");
+
+ const enabled = { ...payload, simplifiedEnglishInteractions: true };
+ const fresh = renderPaperclipWakePrompt(enabled);
+ expect(fresh).toContain("ASD-STE100 Simplified Technical English");
+ expect(fresh).toContain("what happens for each choice");
+ // Resume deltas carry the directive too: the setting can change between wakes.
+ expect(renderPaperclipWakePrompt(enabled, { resumedSession: true })).toContain(
+ "ASD-STE100 Simplified Technical English",
+ );
+ expect(JSON.parse(stringifyPaperclipWakePayload(enabled) ?? "{}")).toMatchObject({
+ simplifiedEnglishInteractions: true,
+ });
+ });
+
it("suppresses the issue description when the prompt already carries the task-context markdown", () => {
const payload = {
reason: "issue_assigned",
diff --git a/packages/adapter-utils/src/server-utils.ts b/packages/adapter-utils/src/server-utils.ts
index 401ae2c27e2f..6c51a3cd322d 100644
--- a/packages/adapter-utils/src/server-utils.ts
+++ b/packages/adapter-utils/src/server-utils.ts
@@ -666,6 +666,9 @@ type PaperclipWakePayload = {
recovery: PaperclipWakeRecovery | null;
issue: PaperclipWakeIssue | null;
checkedOutByHarness: boolean;
+ // Experimental: write user-interaction content in ASD-STE100 Simplified
+ // Technical English with brief decision context.
+ simplifiedEnglishInteractions: boolean;
dependencyBlockedInteraction: boolean;
treeHoldInteraction: boolean;
activeTreeHold: PaperclipWakeTreeHoldSummary | null;
@@ -1317,6 +1320,7 @@ export function normalizePaperclipWakePayload(value: unknown): PaperclipWakePayl
recovery,
issue: normalizePaperclipWakeIssue(payload.issue),
checkedOutByHarness: asBoolean(payload.checkedOutByHarness, false),
+ simplifiedEnglishInteractions: asBoolean(payload.simplifiedEnglishInteractions, false),
dependencyBlockedInteraction: asBoolean(payload.dependencyBlockedInteraction, false),
treeHoldInteraction: asBoolean(payload.treeHoldInteraction, false),
activeTreeHold,
@@ -1624,6 +1628,11 @@ export function renderPaperclipWakePrompt(
`- execution workspace branch: you are running in an execution workspace on branch ${markdownInlineCode(normalized.executionWorkspace.branchName)}. Do not switch, rename, or re-point this branch; keep all commits on it.`,
);
}
+ if (normalized.simplifiedEnglishInteractions) {
+ lines.push(
+ "- interaction language (experimental): write every user interaction you post (request_confirmation, ask_user_questions, suggest_tasks, checkbox prompts and options, and any other content rendered inside an interaction block) in ASD-STE100 Simplified Technical English. In each interaction, briefly tell the user what information they need to make the decision and what happens for each choice. This applies only to interaction content — write your thinking, comments, documents, and other responses in your usual style.",
+ );
+ }
if (normalized.dependencyBlockedInteraction) {
lines.push("- dependency-blocked interaction: yes");
lines.push("- execution scope: respond or triage the human comment; do not treat blocker-dependent deliverable work as unblocked");
diff --git a/packages/adapter-utils/src/ssh.ts b/packages/adapter-utils/src/ssh.ts
index b2aa909c9a84..de86cdb678ad 100644
--- a/packages/adapter-utils/src/ssh.ts
+++ b/packages/adapter-utils/src/ssh.ts
@@ -6,6 +6,7 @@ import os from "node:os";
import path from "node:path";
import { Transform } from "node:stream";
import type { CommandManagedRuntimeRunner } from "./command-managed-runtime.js";
+import { readSanitizedOriginRemoteUrl } from "./git-workspace-sync.js";
import type { RunProcessResult } from "./server-utils.js";
import type { DirectorySnapshot } from "./workspace-restore-merge.js";
import { mergeDirectoryWithBaseline } from "./workspace-restore-merge.js";
@@ -776,6 +777,7 @@ async function importGitWorkspaceToSsh(input: {
timeout: 60_000,
maxBuffer: 1024 * 1024,
});
+ const originUrl = await readSanitizedOriginRemoteUrl(input.localDir);
const remoteSetupScript = [
"set -e",
@@ -784,6 +786,15 @@ async function importGitWorkspaceToSsh(input: {
'trap \'rm -f "$tmp_bundle"\' EXIT',
'cat > "$tmp_bundle"',
`if [ ! -d ${shellQuote(path.posix.join(input.remoteDir, ".git"))} ]; then git init ${shellQuote(input.remoteDir)} >/dev/null; fi`,
+ // Carry the workspace's (credential-scrubbed) origin into the transported
+ // repo so branches there keep a publishable remote instead of reading as
+ // remote-less snapshots. set-url covers a reused workspace whose origin
+ // changed; add covers the fresh-init case. Best-effort under `set -e`.
+ ...(originUrl
+ ? [
+ `{ git -C ${shellQuote(input.remoteDir)} remote set-url origin ${shellQuote(originUrl)} >/dev/null 2>&1 || git -C ${shellQuote(input.remoteDir)} remote add origin ${shellQuote(originUrl)} >/dev/null 2>&1; } || true`,
+ ]
+ : []),
`git -C ${shellQuote(input.remoteDir)} fetch --force "$tmp_bundle" '${tempRef}:${tempRef}' >/dev/null`,
input.snapshot.branchName
? `git -C ${shellQuote(input.remoteDir)} checkout --force -B ${shellQuote(input.snapshot.branchName)} ${shellQuote(input.snapshot.headCommit)} >/dev/null`
diff --git a/packages/adapters/AUTHORING.md b/packages/adapters/AUTHORING.md
index 3448e994b583..0b32365653eb 100644
--- a/packages/adapters/AUTHORING.md
+++ b/packages/adapters/AUTHORING.md
@@ -38,6 +38,18 @@ How to apply:
`workspace_finalize=failed` on the execution workspace, which gates
dependent issue wakes until the next successful finalize. Do not swallow
restore errors.
+- A transported workspace copy *may* carry the local workspace's `origin`
+ remote URL so that branches in the copy stay publishable by the agent or an
+ operator who holds credentials — the transport helpers copy the URL as
+ metadata only. The copy is allowlist-based and fails closed
+ (`sanitizeGitRemoteUrl`): http(s) URLs are stripped of userinfo, query, and
+ fragment; `ssh:`/`git:` scheme URLs are stripped of password and query;
+ scp-like `user@host:path` passes through (the syntax has no password slot);
+ every other shape — filesystem paths, unknown schemes — is dropped rather
+ than risk persisting an embedded secret. This does not weaken the contract:
+ sync-back through the local cwd remains the only cross-run persistence
+ path, the helpers never fetch from or push to that remote, and a workspace
+ without an `origin` transports exactly as before.
The invariant is pinned by the `no-remote-git contract` case in
[`packages/adapter-utils/src/ssh-fixture.test.ts`](../adapter-utils/src/ssh-fixture.test.ts),
diff --git a/packages/adapters/claude-local/src/server/execute.remote.test.ts b/packages/adapters/claude-local/src/server/execute.remote.test.ts
index fad8dc2bb4ee..ab27ab03f6d2 100644
--- a/packages/adapters/claude-local/src/server/execute.remote.test.ts
+++ b/packages/adapters/claude-local/src/server/execute.remote.test.ts
@@ -169,11 +169,16 @@ describe("claude remote execution", () => {
localDir: workspaceDir,
remoteDir: managedRemoteWorkspace,
}));
- expect(syncDirectoryToSsh).toHaveBeenCalledTimes(1);
+ // One sync per registered runtime asset: skills and mcp-config.
+ expect(syncDirectoryToSsh).toHaveBeenCalledTimes(2);
expect(syncDirectoryToSsh).toHaveBeenCalledWith(expect.objectContaining({
remoteDir: `${managedRemoteWorkspace}/.paperclip-runtime/claude/skills`,
followSymlinks: true,
}));
+ expect(syncDirectoryToSsh).toHaveBeenCalledWith(expect.objectContaining({
+ remoteDir: `${managedRemoteWorkspace}/.paperclip-runtime/claude/mcp-config`,
+ followSymlinks: true,
+ }));
expect(runChildProcess).toHaveBeenCalledTimes(1);
const call = runChildProcess.mock.calls[0] as unknown as
| [string, string, string[], { env: Record; remoteExecution?: { remoteCwd: string } | null }]
diff --git a/packages/adapters/claude-local/src/server/test.probe.test.ts b/packages/adapters/claude-local/src/server/test.probe.test.ts
index 9ff7b5f0e8e5..58088fa47d08 100644
--- a/packages/adapters/claude-local/src/server/test.probe.test.ts
+++ b/packages/adapters/claude-local/src/server/test.probe.test.ts
@@ -101,7 +101,7 @@ describe("claude sandbox hello probe diagnostics", () => {
expect(failed?.detail).not.toContain('"subtype":"init"');
});
- it("classifies rate-limit/overload failures as a transient warning, not a hard fail", async () => {
+ it("classifies subscription usage-limit failures as a usage-limited warning, not a hard fail", async () => {
probeResult.value = {
exitCode: 1,
stdout: [
@@ -119,7 +119,31 @@ describe("claude sandbox hello probe diagnostics", () => {
environmentName: "Daytona",
});
+ expect(result.checks.some((check) => check.code === "claude_hello_probe_usage_limited")).toBe(true);
+ expect(result.checks.some((check) => check.code === "claude_hello_probe_transient_upstream")).toBe(false);
+ expect(result.checks.some((check) => check.code === "claude_hello_probe_failed")).toBe(false);
+ });
+
+ it("classifies overload failures as a transient warning, not a hard fail", async () => {
+ probeResult.value = {
+ exitCode: 1,
+ stdout: [
+ initLine,
+ '{"type":"result","subtype":"error_during_execution","is_error":true,"result":"API Error: 529 overloaded_error","session_id":"abc"}',
+ ].join("\n"),
+ stderr: "",
+ };
+
+ const result = await testEnvironment({
+ companyId: "company-1",
+ adapterType: "claude_local",
+ config: { engine: "cli", command: "claude" },
+ executionTarget: sandboxTarget,
+ environmentName: "Daytona",
+ });
+
expect(result.checks.some((check) => check.code === "claude_hello_probe_transient_upstream")).toBe(true);
+ expect(result.checks.some((check) => check.code === "claude_hello_probe_usage_limited")).toBe(false);
expect(result.checks.some((check) => check.code === "claude_hello_probe_failed")).toBe(false);
});
@@ -161,3 +185,58 @@ describe("claude sandbox hello probe diagnostics", () => {
expect(failed?.detail).toBeUndefined();
});
});
+
+describe("claude auth mode hints", () => {
+ const successStdout = [
+ initLine,
+ '{"type":"result","subtype":"success","is_error":false,"result":"hello","session_id":"abc"}',
+ ].join("\n");
+
+ it("reports the configured subscription token for remote targets", async () => {
+ probeResult.value = { exitCode: 0, stdout: successStdout, stderr: "" };
+
+ const result = await testEnvironment({
+ companyId: "company-1",
+ adapterType: "claude_local",
+ config: {
+ engine: "cli",
+ command: "claude",
+ env: { CLAUDE_CODE_OAUTH_TOKEN: "oauth-test-token" },
+ },
+ executionTarget: sandboxTarget,
+ environmentName: "Daytona",
+ });
+
+ const hint = result.checks.find((check) => check.code === "claude_oauth_token_configured");
+ expect(hint).toBeTruthy();
+ expect(hint?.level).toBe("info");
+ expect(hint?.detail).toContain("configured environment variables");
+ expect(
+ result.checks.some((check) => check.code === "claude_anthropic_api_key_overrides_subscription"),
+ ).toBe(false);
+ });
+
+ it("keeps the API-key warning authoritative when both ANTHROPIC_API_KEY and the token are set", async () => {
+ probeResult.value = { exitCode: 0, stdout: successStdout, stderr: "" };
+
+ const result = await testEnvironment({
+ companyId: "company-1",
+ adapterType: "claude_local",
+ config: {
+ engine: "cli",
+ command: "claude",
+ env: {
+ ANTHROPIC_API_KEY: "api-test-key",
+ CLAUDE_CODE_OAUTH_TOKEN: "oauth-test-token",
+ },
+ },
+ executionTarget: sandboxTarget,
+ environmentName: "Daytona",
+ });
+
+ expect(
+ result.checks.some((check) => check.code === "claude_anthropic_api_key_overrides_subscription"),
+ ).toBe(true);
+ expect(result.checks.some((check) => check.code === "claude_oauth_token_configured")).toBe(false);
+ });
+});
diff --git a/packages/adapters/claude-local/src/server/test.ts b/packages/adapters/claude-local/src/server/test.ts
index caedc9d7325e..1c50a7ae4724 100644
--- a/packages/adapters/claude-local/src/server/test.ts
+++ b/packages/adapters/claude-local/src/server/test.ts
@@ -28,6 +28,7 @@ import {
import {
describeClaudeFailure,
detectClaudeLoginRequired,
+ isClaudeProviderQuotaError,
isClaudeTransientUpstreamError,
parseClaudeStreamJson,
} from "./parse.js";
@@ -278,6 +279,20 @@ export async function testEnvironment(
detail: `Detected in ${source}.`,
hint: "Unset ANTHROPIC_API_KEY if you want subscription-based Claude login behavior.",
});
+ } else if (
+ isNonEmpty(env.CLAUDE_CODE_OAUTH_TOKEN) ||
+ (considerHostEnv && isNonEmpty(process.env.CLAUDE_CODE_OAUTH_TOKEN))
+ ) {
+ const source = isNonEmpty(env.CLAUDE_CODE_OAUTH_TOKEN)
+ ? "configured environment variables"
+ : "server environment";
+ checks.push({
+ code: "claude_oauth_token_configured",
+ level: "info",
+ message:
+ "CLAUDE_CODE_OAUTH_TOKEN is set. Claude will authenticate with the configured subscription token; no stored login is needed on the execution target.",
+ detail: `Detected in ${source}.`,
+ });
} else if (!targetIsRemote) {
checks.push({
code: "claude_subscription_mode_possible",
@@ -428,27 +443,44 @@ export async function testEnvironment(
(stdoutFallback ? truncateDetail(stdoutFallback) : "") ||
detail ||
"";
+ // Provider-quota exhaustion (usage/session limit) is classified
+ // separately from generic transient upstream errors: auth works, the
+ // subscription's usage window is just spent. Surface it as its own
+ // warning instead of a hard probe failure.
+ const usageLimited = isClaudeProviderQuotaError({
+ parsed,
+ stdout: probe.stdout,
+ stderr: probe.stderr,
+ });
const transient = isClaudeTransientUpstreamError({
parsed,
stdout: probe.stdout,
stderr: probe.stderr,
});
checks.push(
- transient
+ usageLimited
? {
- code: "claude_hello_probe_transient_upstream",
+ code: "claude_hello_probe_usage_limited",
level: "warn",
- message: "Claude hello probe hit a transient upstream error (rate limit or overload).",
+ message: "Claude hello probe hit the subscription usage limit.",
...(failureDetail ? { detail: failureDetail } : {}),
- hint: "This is usually temporary. Wait a moment and re-run Test.",
+ hint: "Authentication works; the account's usage window is exhausted. Wait for the limit to reset and re-run Test.",
}
- : {
- code: "claude_hello_probe_failed",
- level: "error",
- message: "Claude hello probe failed.",
- ...(failureDetail ? { detail: failureDetail } : {}),
- hint: `Exit code ${probe.exitCode ?? "unknown"}. Run \`claude --print - --output-format stream-json --verbose\` manually in this directory and prompt \`Respond with hello\` to debug.`,
- },
+ : transient
+ ? {
+ code: "claude_hello_probe_transient_upstream",
+ level: "warn",
+ message: "Claude hello probe hit a transient upstream error (rate limit or overload).",
+ ...(failureDetail ? { detail: failureDetail } : {}),
+ hint: "This is usually temporary. Wait a moment and re-run Test.",
+ }
+ : {
+ code: "claude_hello_probe_failed",
+ level: "error",
+ message: "Claude hello probe failed.",
+ ...(failureDetail ? { detail: failureDetail } : {}),
+ hint: `Exit code ${probe.exitCode ?? "unknown"}. Run \`claude --print - --output-format stream-json --verbose\` manually in this directory and prompt \`Respond with hello\` to debug.`,
+ },
);
}
}
diff --git a/packages/adapters/claude-local/src/ui/build-config.test.ts b/packages/adapters/claude-local/src/ui/build-config.test.ts
index 3f8d3626bfd1..30eb04c9d71c 100644
--- a/packages/adapters/claude-local/src/ui/build-config.test.ts
+++ b/packages/adapters/claude-local/src/ui/build-config.test.ts
@@ -47,4 +47,34 @@ describe("buildClaudeLocalConfig", () => {
expect(buildClaudeLocalConfig(makeValues({ claudeEngine: "cli" }))).toMatchObject({ engine: "cli" });
expect(buildClaudeLocalConfig(makeValues({ claudeEngine: "acp" }))).toMatchObject({ engine: "acp" });
});
+
+ it("keeps user-scoped env bindings so the server resolves them at test time", () => {
+ const config = buildClaudeLocalConfig(
+ makeValues({
+ envBindings: {
+ GH_TOKEN: { type: "user_secret_ref", key: "github_token", version: "latest", required: true },
+ },
+ }),
+ );
+
+ expect(config.env).toEqual({
+ GH_TOKEN: { type: "user_secret_ref", key: "github_token", version: "latest", required: true },
+ });
+ });
+
+ it("keeps company secret and plain env bindings", () => {
+ const config = buildClaudeLocalConfig(
+ makeValues({
+ envBindings: {
+ API_KEY: { type: "secret_ref", secretId: "11111111-1111-1111-1111-111111111111", version: "latest" },
+ FLAG: { type: "plain", value: "on" },
+ },
+ }),
+ );
+
+ expect(config.env).toEqual({
+ API_KEY: { type: "secret_ref", secretId: "11111111-1111-1111-1111-111111111111", version: "latest" },
+ FLAG: { type: "plain", value: "on" },
+ });
+ });
});
diff --git a/packages/adapters/claude-local/src/ui/build-config.ts b/packages/adapters/claude-local/src/ui/build-config.ts
index c17fcd717899..9262984806ca 100644
--- a/packages/adapters/claude-local/src/ui/build-config.ts
+++ b/packages/adapters/claude-local/src/ui/build-config.ts
@@ -1,4 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
+import { buildAdapterEnvConfig, type CreateConfigValues } from "@paperclipai/adapter-utils";
function parseCommaArgs(value: string): string[] {
return value
@@ -7,49 +7,6 @@ function parseCommaArgs(value: string): string[] {
.filter(Boolean);
}
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
-
function parseJsonObject(text: string): Record | null {
const trimmed = text.trim();
if (!trimmed) return null;
@@ -84,13 +41,7 @@ export function buildClaudeLocalConfig(v: CreateConfigValues): Record 0) ac.env = env;
ac.maxTurnsPerRun = v.maxTurnsPerRun;
ac.dangerouslySkipPermissions = v.dangerouslySkipPermissions;
diff --git a/packages/adapters/codex-local/CODEX-AUTH-CACHE.md b/packages/adapters/codex-local/CODEX-AUTH-CACHE.md
new file mode 100644
index 000000000000..fcffe6957d5e
--- /dev/null
+++ b/packages/adapters/codex-local/CODEX-AUTH-CACHE.md
@@ -0,0 +1,101 @@
+# Codex identity-keyed host credential cache
+
+This document describes the host credential cache for Codex authentication. The
+cache keeps one usable subscription credential per identity (`account_id`) in a
+separate host store. The cache is additive. It does not change the copy-back
+path, the fail-closed decision predicate, or the host default store overwrite.
+
+## Where the cache lives
+
+The cache root is company-scoped, under the same isolation boundary as the
+managed Codex home:
+
+```
+/companies//codex-auth-cache//auth.json
+```
+
+The root sits outside the shared Codex home (`resolveSharedCodexHomeDir`) and
+outside the symlink allowlist. The `account_id` is sanitized to one safe path
+segment. Each cache root and each identity directory is private (mode `0700`).
+
+## The two directions
+
+The board set one rule for both directions. A side that has no credential can
+receive one, but only when a real credential on one side names the expected
+identity. The harness never picks a credential from the cache at random.
+
+- **Host to sandbox (inbound).** A sandbox that starts with no Codex credential
+ takes the host credential. The cache does not change this. At provision the
+ cache also refreshes the host credential with a strictly-newer cached copy of
+ the same identity (the vend, below).
+- **Sandbox to host (copy-back).** A host that already holds an identity keeps a
+ strictly-newer same-identity credential from the sandbox. The cache does not
+ change this. At teardown the cache also writes the sandbox credential into its
+ per-identity slot (the cache write, below).
+
+## The identity anchor rule
+
+The identity anchor rule is the load-bearing constraint:
+
+- The **cache vend** only refreshes an identity the host **already holds** in the
+ shared home. It replaces the staged host credential with a strictly-newer
+ cached credential of the **same** `account_id`. It never introduces a new
+ identity.
+- When the host shared home holds **no** credential, the vend does **nothing**.
+ The harness never selects a cache entry to seed an empty host. This is the
+ "no random pick from the cache when the host is empty" rule.
+- The **cache write** keys each entry by the real `account_id` of the credential
+ that flows back from the sandbox. It writes to a per-identity cache slot, never
+ to the host default store. The cache write is best-effort: it runs after the
+ host copy-back finishes, so a cache-write failure never replaces the successful
+ copy-back result. The failure is logged with its errno code and the next
+ teardown re-attempts the write.
+
+The host never learns an identity from the cache. The host only refreshes an
+identity a real credential already states.
+
+## State matrix
+
+"Host has auth" means the shared source store `resolveSharedCodexHomeDir(env)/auth.json`
+holds a usable subscription credential. "Sandbox has auth" means the run's
+sandbox `auth.json` holds a usable credential at teardown. `X` and `Y` are two
+different subscription identities (`account_id`).
+
+| # | Host store | Sandbox cred | Inbound: sandbox home gets | Copy-back: host store | Cache write | Cache vend | Identity anchor |
+|---|---|---|---|---|---|---|---|
+| 1a | HAS `X` | HAS `X`, newer | fresher of the two (`X`) | overwrite with newer `X` | write slot `X` | may stage a strictly-newer cached `X` | host and sandbox both name `X` |
+| 1b | HAS `X` | HAS `Y` (`Y != X`) | host `X` (predicate rejects `Y`) | keep host `X` | write slot `Y` (per identity) | may stage a strictly-newer cached `X` | host names `X`; `Y` is cached, never adopted |
+| 2 | HAS `X` | NONE | host `X` | keep host (sandbox absent) | no write (no source) | may stage a strictly-newer cached `X` | host names `X` |
+| 3 | NONE | HAS `Y` | image-login fallback; host store **never seeded** | keep host **empty** (never seed) | write slot `Y` | **none** (host empty, no random pick) | sandbox names `Y`; host is silent |
+| 4 | NONE | NONE | image-login fallback, or the run fails | keep host **empty** | no write | **none** | no side names an identity |
+
+The cache changes an outcome only in the "stage a strictly-newer cached copy of
+an identity the host already holds" cases (rows 1a, 1b, 2, vend column). It never
+changes which identity a side uses. It never seeds an empty host store (rows 3,
+4).
+
+## Off-switch
+
+The cache is on by default. Set the environment flag `PAPERCLIP_CODEX_AUTH_CACHE`
+to an explicit falsy value (`0`, `false`, `no`, or `off`) to turn it off. When
+off, the teardown cache write and the provision vend become no-ops. The host
+default overwrite is unchanged in both states.
+
+## Cache-clear action
+
+Two operator actions remove cached credentials:
+
+- `clearCodexAuthCacheEntry(env, accountId, companyId)` removes exactly one
+ identity slot.
+- `clearCodexAuthCache(env, companyId)` removes every slot in the company-scoped
+ cache root.
+
+To disable the cache without a code revert, use the off-switch. To remove a
+single cached identity, use `clearCodexAuthCacheEntry`. To remove every cached
+credential, delete the cache root or use `clearCodexAuthCache`. Neither action
+affects the host default store.
+
+## No secret in logs
+
+The cache logs the decision and the outcome only. It never logs token bytes and
+never logs a raw `account_id`.
diff --git a/packages/adapters/codex-local/src/server/codex-auth-cache.test.ts b/packages/adapters/codex-local/src/server/codex-auth-cache.test.ts
new file mode 100644
index 000000000000..f3818ecd8acc
--- /dev/null
+++ b/packages/adapters/codex-local/src/server/codex-auth-cache.test.ts
@@ -0,0 +1,333 @@
+import { chmod, lstat, mkdir, mkdtemp, readdir, readFile, rm, symlink, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { afterEach, describe, expect, it } from "vitest";
+
+import {
+ clearCodexAuthCache,
+ clearCodexAuthCacheEntry,
+ ensureCodexAuthCacheEntryDir,
+ isCodexAuthCacheEnabled,
+ resolveCodexAuthCacheDir,
+ resolveCodexAuthCacheEntryPath,
+ selectVendCredential,
+ toCacheKey,
+} from "./codex-auth-cache.js";
+import { resolveSharedCodexHomeDir } from "./codex-home.js";
+
+// This suite proves the per-identity host credential cache store. The cache is a
+// separate directory outside the shared Codex home and outside the symlink
+// allowlist. It keys one usable subscription credential per identity
+// (`account_id`). The suite drives the real path resolver, the real directory
+// guards, and the real decision predicate (`codex-auth-merge-decision.cjs`)
+// against a real host tmp filesystem.
+describe("codex auth cache store", () => {
+ const cleanupDirs: string[] = [];
+
+ afterEach(async () => {
+ while (cleanupDirs.length > 0) {
+ const dir = cleanupDirs.pop();
+ if (!dir) continue;
+ await chmod(dir, 0o700).catch(() => undefined);
+ await rm(dir, { recursive: true, force: true }).catch(() => undefined);
+ }
+ });
+
+ async function makeInstanceRoot(): Promise {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "paperclip-codex-cache-"));
+ cleanupDirs.push(dir);
+ return dir;
+ }
+
+ function envFor(instanceHome: string, extra: Record = {}): NodeJS.ProcessEnv {
+ return {
+ PAPERCLIP_HOME: instanceHome,
+ PAPERCLIP_INSTANCE_ID: "default",
+ ...extra,
+ };
+ }
+
+ function subscriptionAuth(input: { accountId: string; lastRefresh?: string; marker?: string }): string {
+ const suffix = input.marker ?? input.accountId;
+ return JSON.stringify({
+ tokens: {
+ id_token: `id-token-${suffix}`,
+ access_token: `access-token-${suffix}`,
+ refresh_token: `refresh-token-${suffix}`,
+ account_id: input.accountId,
+ },
+ ...(input.lastRefresh ? { last_refresh: input.lastRefresh } : {}),
+ });
+ }
+
+ const NEWER = "2026-07-09T02:00:00Z";
+ const OLDER = "2026-07-09T01:00:00Z";
+
+ describe("Phase 1: cache store location and path scheme", () => {
+ it("resolveCodexAuthCacheDir returns a path outside resolveSharedCodexHomeDir", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home, { CODEX_HOME: path.join(home, "shared-codex") });
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ const sharedHome = resolveSharedCodexHomeDir(env);
+ expect(cacheDir.startsWith(sharedHome + path.sep)).toBe(false);
+ expect(cacheDir).not.toBe(sharedHome);
+ });
+
+ it("resolveCodexAuthCacheDir returns a company-scoped path under the instance companies directory when companyId is set", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ expect(cacheDir).toBe(
+ path.resolve(home, "instances", "default", "companies", "company-a", "codex-auth-cache"),
+ );
+ });
+
+ it("resolveCodexAuthCacheEntryPath keys the entry by a sanitized account_id and ends with auth.json", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ const entryPath = resolveCodexAuthCacheEntryPath(env, "acct-42", "company-a");
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ expect(entryPath).toBe(path.join(cacheDir, "acct-42", "auth.json"));
+ });
+
+ it("toCacheKey rejects an empty account_id, a path separator, and ..", () => {
+ expect(() => toCacheKey("")).toThrow();
+ expect(() => toCacheKey(" ")).toThrow();
+ expect(() => toCacheKey("..")).toThrow();
+ expect(() => toCacheKey(".")).toThrow();
+ expect(() => toCacheKey("a/b")).toThrow();
+ expect(() => toCacheKey("a\\b")).toThrow();
+ expect(() => toCacheKey("../escape")).toThrow();
+ expect(() => toCacheKey("a\0b")).toThrow();
+ expect(toCacheKey("acct-42")).toBe("acct-42");
+ });
+
+ it("resolveCodexAuthCacheDir rejects a traversal companyId before it builds the path", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ // A traversal companyId must never make the cache root escape the
+ // companies/ directory. Sanitization runs before path construction, so a
+ // relative segment, a path separator, an absolute path, and a NUL byte all
+ // fail loud.
+ expect(() => resolveCodexAuthCacheDir(env, "")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, " ")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, "..")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, ".")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, "../../etc")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, "/etc")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, "a/b")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, "a\\b")).toThrow();
+ expect(() => resolveCodexAuthCacheDir(env, "a\0b")).toThrow();
+ // The clear path inherits the same guard, so a traversal companyId can
+ // never reach rm() on an escaped root.
+ await expect(clearCodexAuthCache(env, "../../etc")).rejects.toThrow();
+ const safeDir = resolveCodexAuthCacheDir(env, "company-a");
+ expect(safeDir).toBe(
+ path.resolve(home, "instances", "default", "companies", "company-a", "codex-auth-cache"),
+ );
+ });
+
+ it("resolveCodexAuthCacheEntryPath verifies the resolved path stays under the cache root", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ // A traversal account_id is rejected by toCacheKey, so the resolved entry
+ // path can never escape the cache root.
+ expect(() => resolveCodexAuthCacheEntryPath(env, "../../etc", "company-a")).toThrow();
+ const entryPath = resolveCodexAuthCacheEntryPath(env, "acct-ok", "company-a");
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ expect(entryPath.startsWith(cacheDir + path.sep)).toBe(true);
+ expect(entryPath.endsWith(path.join("acct-ok", "auth.json"))).toBe(true);
+ });
+
+ it("ensureCodexAuthCacheEntryDir creates the cache root and the entry directory private (0700)", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ const entryPath = await ensureCodexAuthCacheEntryDir(env, "acct-priv", "company-a");
+ const entryDir = path.dirname(entryPath);
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ expect((await lstat(entryDir)).mode & 0o777).toBe(0o700);
+ expect((await lstat(cacheDir)).mode & 0o777).toBe(0o700);
+ });
+
+ it("the cache root fails closed (lstat) when it is a symlink or a non-directory", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ // Make the cache root a symlink to another directory. lstat (not stat)
+ // must catch it and fail closed.
+ await mkdir(path.dirname(cacheDir), { recursive: true });
+ const target = path.join(home, "elsewhere");
+ await mkdir(target, { recursive: true });
+ await symlink(target, cacheDir);
+ await expect(ensureCodexAuthCacheEntryDir(env, "acct-x", "company-a")).rejects.toThrow();
+ });
+
+ it("the entry directory fails closed (lstat) when it is a symlink or a non-directory", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ await mkdir(cacheDir, { recursive: true, mode: 0o700 });
+ // Put a regular file where the entry directory must be.
+ await writeFile(path.join(cacheDir, "acct-file"), "not a dir", { mode: 0o600 });
+ await expect(ensureCodexAuthCacheEntryDir(env, "acct-file", "company-a")).rejects.toThrow();
+ });
+ });
+
+ describe("Phase 4: identity-anchored vend", () => {
+ async function stageHostAndCache(input: {
+ hostAuth?: string;
+ cacheAuthByAccount?: Record;
+ }): Promise<{ env: NodeJS.ProcessEnv; sharedHomeAuthPath: string }> {
+ const home = await makeInstanceRoot();
+ const env = envFor(home, { CODEX_HOME: path.join(home, "shared-codex") });
+ const sharedHome = resolveSharedCodexHomeDir(env);
+ await mkdir(sharedHome, { recursive: true });
+ const sharedHomeAuthPath = path.join(sharedHome, "auth.json");
+ if (input.hostAuth !== undefined) {
+ await writeFile(sharedHomeAuthPath, input.hostAuth, { mode: 0o600 });
+ }
+ for (const [accountId, auth] of Object.entries(input.cacheAuthByAccount ?? {})) {
+ const entryPath = await ensureCodexAuthCacheEntryDir(env, accountId, "company-a");
+ await writeFile(entryPath, auth, { mode: 0o600 });
+ }
+ return { env, sharedHomeAuthPath };
+ }
+
+ const resolveEntry =
+ (env: NodeJS.ProcessEnv) => (accountId: string) =>
+ resolveCodexAuthCacheEntryPath(env, accountId, "company-a");
+
+ it("vend refreshes the host identity when the cached copy is strictly newer for the same identity", async () => {
+ const { env, sharedHomeAuthPath } = await stageHostAndCache({
+ hostAuth: subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" }),
+ cacheAuthByAccount: {
+ "acct-x": subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER, marker: "cache" }),
+ },
+ });
+ const outcome = await selectVendCredential(sharedHomeAuthPath, resolveEntry(env), () => undefined);
+ expect(outcome).toBe("vended");
+ const finalHost = await readFile(sharedHomeAuthPath, "utf8");
+ expect(finalHost).toContain("acct-x");
+ expect(finalHost).toContain("cache");
+ });
+
+ it("vend keeps the shared-home credential when the cache copy is older or a tie", async () => {
+ for (const cacheRefresh of [OLDER, NEWER]) {
+ const hostRefresh = cacheRefresh === OLDER ? NEWER : NEWER; // host newer or tie
+ const { env, sharedHomeAuthPath } = await stageHostAndCache({
+ hostAuth: subscriptionAuth({ accountId: "acct-x", lastRefresh: hostRefresh, marker: "host" }),
+ cacheAuthByAccount: {
+ "acct-x": subscriptionAuth({ accountId: "acct-x", lastRefresh: cacheRefresh, marker: "cache" }),
+ },
+ });
+ const before = await readFile(sharedHomeAuthPath, "utf8");
+ const outcome = await selectVendCredential(sharedHomeAuthPath, resolveEntry(env), () => undefined);
+ expect(outcome).toBe("kept-host");
+ expect(await readFile(sharedHomeAuthPath, "utf8")).toBe(before);
+ }
+ });
+
+ it("vend never selects a cache slot with a different account_id than the host holds", async () => {
+ const { env, sharedHomeAuthPath } = await stageHostAndCache({
+ hostAuth: subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" }),
+ cacheAuthByAccount: {
+ "acct-y": subscriptionAuth({ accountId: "acct-y", lastRefresh: NEWER, marker: "other" }),
+ },
+ });
+ const before = await readFile(sharedHomeAuthPath, "utf8");
+ const outcome = await selectVendCredential(sharedHomeAuthPath, resolveEntry(env), () => undefined);
+ expect(outcome).toBe("kept-host");
+ expect(await readFile(sharedHomeAuthPath, "utf8")).toBe(before);
+ });
+
+ it("vend does nothing when the host shared home has no auth.json (no random pick from the cache)", async () => {
+ const { env, sharedHomeAuthPath } = await stageHostAndCache({
+ cacheAuthByAccount: {
+ "acct-y": subscriptionAuth({ accountId: "acct-y", lastRefresh: NEWER, marker: "other" }),
+ },
+ });
+ const outcome = await selectVendCredential(sharedHomeAuthPath, resolveEntry(env), () => undefined);
+ expect(outcome).toBe("no-host-identity");
+ await expect(lstat(sharedHomeAuthPath)).rejects.toThrow();
+ });
+
+ it("vend does nothing when the host shared home holds an apikey (no subscription identity)", async () => {
+ const { env, sharedHomeAuthPath } = await stageHostAndCache({
+ hostAuth: JSON.stringify({ OPENAI_API_KEY: "sk-host" }),
+ cacheAuthByAccount: {
+ "acct-y": subscriptionAuth({ accountId: "acct-y", lastRefresh: NEWER, marker: "other" }),
+ },
+ });
+ const before = await readFile(sharedHomeAuthPath, "utf8");
+ const outcome = await selectVendCredential(sharedHomeAuthPath, resolveEntry(env), () => undefined);
+ expect(outcome).toBe("no-host-identity");
+ expect(await readFile(sharedHomeAuthPath, "utf8")).toBe(before);
+ });
+
+ it("the vend never emits token bytes or a raw account_id to the log", async () => {
+ const { env, sharedHomeAuthPath } = await stageHostAndCache({
+ hostAuth: subscriptionAuth({ accountId: "SECRET-ACCT", lastRefresh: OLDER, marker: "HOST-SENTINEL" }),
+ cacheAuthByAccount: {
+ "SECRET-ACCT": subscriptionAuth({ accountId: "SECRET-ACCT", lastRefresh: NEWER, marker: "CACHE-SENTINEL" }),
+ },
+ });
+ const logs: string[] = [];
+ await selectVendCredential(sharedHomeAuthPath, resolveEntry(env), (line) => {
+ logs.push(line);
+ });
+ const combined = logs.join("\n");
+ expect(combined).not.toContain("SENTINEL");
+ expect(combined).not.toContain("SECRET-ACCT");
+ expect(combined).not.toContain("id-token");
+ });
+ });
+
+ describe("Phase 5: cache-clear operator action and off-switch", () => {
+ it("clearCodexAuthCacheEntry removes one identity slot and leaves other slots intact", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ const keepPath = await ensureCodexAuthCacheEntryDir(env, "acct-keep", "company-a");
+ const dropPath = await ensureCodexAuthCacheEntryDir(env, "acct-drop", "company-a");
+ await writeFile(keepPath, subscriptionAuth({ accountId: "acct-keep" }), { mode: 0o600 });
+ await writeFile(dropPath, subscriptionAuth({ accountId: "acct-drop" }), { mode: 0o600 });
+
+ await clearCodexAuthCacheEntry(env, "acct-drop", "company-a");
+ await expect(lstat(path.dirname(dropPath))).rejects.toThrow();
+ expect(await readFile(keepPath, "utf8")).toContain("acct-keep");
+ });
+
+ it("clearCodexAuthCacheEntry rejects an account_id that escapes the cache root", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ await expect(clearCodexAuthCacheEntry(env, "../../etc", "company-a")).rejects.toThrow();
+ await expect(clearCodexAuthCacheEntry(env, "a/b", "company-a")).rejects.toThrow();
+ });
+
+ it("clearCodexAuthCacheEntry is a benign no-op for a missing slot", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ await expect(clearCodexAuthCacheEntry(env, "acct-absent", "company-a")).resolves.toBeUndefined();
+ });
+
+ it("clearCodexAuthCache removes every slot", async () => {
+ const home = await makeInstanceRoot();
+ const env = envFor(home);
+ await ensureCodexAuthCacheEntryDir(env, "acct-1", "company-a");
+ await ensureCodexAuthCacheEntryDir(env, "acct-2", "company-a");
+ const cacheDir = resolveCodexAuthCacheDir(env, "company-a");
+ expect((await readdir(cacheDir)).sort()).toEqual(["acct-1", "acct-2"]);
+ await clearCodexAuthCache(env, "company-a");
+ await expect(lstat(cacheDir)).rejects.toThrow();
+ });
+
+ it("isCodexAuthCacheEnabled defaults to on and turns off on an explicit falsy flag", () => {
+ expect(isCodexAuthCacheEnabled({})).toBe(true);
+ expect(isCodexAuthCacheEnabled({ PAPERCLIP_CODEX_AUTH_CACHE: "1" })).toBe(true);
+ expect(isCodexAuthCacheEnabled({ PAPERCLIP_CODEX_AUTH_CACHE: "on" })).toBe(true);
+ expect(isCodexAuthCacheEnabled({ PAPERCLIP_CODEX_AUTH_CACHE: "0" })).toBe(false);
+ expect(isCodexAuthCacheEnabled({ PAPERCLIP_CODEX_AUTH_CACHE: "false" })).toBe(false);
+ expect(isCodexAuthCacheEnabled({ PAPERCLIP_CODEX_AUTH_CACHE: "off" })).toBe(false);
+ expect(isCodexAuthCacheEnabled({ PAPERCLIP_CODEX_AUTH_CACHE: "no" })).toBe(false);
+ });
+ });
+});
diff --git a/packages/adapters/codex-local/src/server/codex-auth-cache.ts b/packages/adapters/codex-local/src/server/codex-auth-cache.ts
new file mode 100644
index 000000000000..3283cfd68066
--- /dev/null
+++ b/packages/adapters/codex-local/src/server/codex-auth-cache.ts
@@ -0,0 +1,415 @@
+import { execFile as execFileCallback } from "node:child_process";
+import { lstat, mkdir, open, readFile, rename, rm } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { promisify } from "node:util";
+import { randomUUID } from "node:crypto";
+import { resolvePaperclipInstanceRootForAdapter } from "@paperclipai/adapter-utils/server-utils";
+import { withDirectoryMergeLock } from "@paperclipai/adapter-utils/workspace-restore-merge";
+
+const execFile = promisify(execFileCallback);
+
+// The identity-keyed host credential cache keeps one usable subscription
+// credential per identity (`account_id`) in a SEPARATE host store, outside the
+// shared Codex home and outside the symlink allowlist. The cache is additive: it
+// never changes the copy-back path, the fail-closed decision predicate, or the
+// host default store overwrite. It only refreshes an identity the host already
+// holds. It never seeds an empty host store and never picks a credential at
+// random.
+
+const CACHE_DIR_NAME = "codex-auth-cache";
+const CACHE_ENTRY_FILE = "auth.json";
+// A private directory (owner rwx only). 0o700 has no group/other bits, so a
+// standard umask can never widen it.
+const PRIVATE_DIR_MODE = 0o700;
+
+// One default-on off-switch. When the flag is an explicit falsy value the cache
+// write and the cache vend become no-ops. The host default overwrite is
+// unchanged in both states.
+export const CODEX_AUTH_CACHE_OFF_SWITCH_ENV = "PAPERCLIP_CODEX_AUTH_CACHE";
+const FALSY_ENV_RE = /^(0|false|no|off)$/i;
+
+// The cache reuses the same direction-agnostic decision predicate the copy-back
+// and inbound restore run. The predicate answers one question — "should the
+// caller replace `destination` with `source`?" — purely by argument order (first
+// = source, second = destination). Exit 10 = use source; exit 20 = keep
+// destination. A leading `--seed-if-dest-absent` flag adds one opt-in behaviour:
+// fill an ABSENT destination slot from a usable subscription source. The
+// predicate only reads the two files and exits with a code; it never prints
+// token bytes.
+const DECISION_SCRIPT_PATH = fileURLToPath(
+ new URL("./codex-auth-merge-decision.cjs", import.meta.url),
+);
+const SEED_IF_DEST_ABSENT_FLAG = "--seed-if-dest-absent";
+const USE_SOURCE_EXIT = 10;
+const KEEP_DESTINATION_EXIT = 20;
+
+function nonEmpty(value: string | undefined): string | null {
+ return typeof value === "string" && value.trim().length > 0 ? value.trim() : null;
+}
+
+/**
+ * True when the cache is enabled. The cache is on by default. It turns off only
+ * when {@link CODEX_AUTH_CACHE_OFF_SWITCH_ENV} is an explicit falsy value
+ * (`0`, `false`, `no`, or `off`).
+ */
+export function isCodexAuthCacheEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
+ const raw = env[CODEX_AUTH_CACHE_OFF_SWITCH_ENV];
+ if (typeof raw !== "string") return true;
+ return !FALSY_ENV_RE.test(raw.trim());
+}
+
+/**
+ * Sanitizes one raw value to a single safe path segment. Rejects an empty value,
+ * a relative segment (`.` or `..`), a path separator (`/` or `\`), and a NUL
+ * byte, so the value can never become a path traversal. Returns the trimmed,
+ * safe segment. The `label` names the value in the error message. (Security
+ * condition 3.)
+ */
+function toSafePathSegment(value: string, label: string): string {
+ const trimmed = typeof value === "string" ? value.trim() : "";
+ if (trimmed.length === 0) {
+ throw new Error(`codex auth cache: ${label} is empty`);
+ }
+ if (trimmed === "." || trimmed === "..") {
+ throw new Error(`codex auth cache: ${label} is a relative path segment`);
+ }
+ if (trimmed.includes("/") || trimmed.includes("\\") || trimmed.includes("\0")) {
+ throw new Error(`codex auth cache: ${label} contains a path separator`);
+ }
+ // Defense in depth: a safe segment is exactly its own basename. Anything else
+ // carries a separator or a relative segment the checks above must have caught.
+ if (path.basename(trimmed) !== trimmed) {
+ throw new Error(`codex auth cache: ${label} is not a single path segment`);
+ }
+ return trimmed;
+}
+
+/**
+ * Sanitizes an `account_id` to one safe path segment. Rejects an empty value, a
+ * relative segment (`.` or `..`), a path separator, and a NUL byte, so a raw
+ * `account_id` can never become a path traversal. Returns the trimmed, safe
+ * segment. (Security condition 3.)
+ */
+export function toCacheKey(accountId: string): string {
+ return toSafePathSegment(accountId, "account_id");
+}
+
+/**
+ * Resolves the company-scoped cache root under the same isolation boundary as
+ * the managed Codex home (`resolveManagedCodexHomeDir`). The root is always
+ * company-scoped, so a Codex credential can never cross a company boundary.
+ * There is no instance-global fallback root: `companyId` is required, and an
+ * empty value is a fail-loud error. `companyId` is sanitized to a single safe
+ * path segment, so a traversal value (`..`, `/etc`, `a/b`) can never make the
+ * cache root escape the `companies/` directory. Every downstream path (the
+ * entry path, the vend, and the clear) inherits this guard. (Security
+ * condition 1.)
+ */
+export function resolveCodexAuthCacheDir(
+ env: NodeJS.ProcessEnv = process.env,
+ companyId: string,
+): string {
+ const safeCompanyId = toSafePathSegment(companyId, "companyId");
+ const instanceRoot = resolvePaperclipInstanceRootForAdapter({
+ homeDir: nonEmpty(env.PAPERCLIP_HOME) ?? undefined,
+ instanceId: nonEmpty(env.PAPERCLIP_INSTANCE_ID) ?? undefined,
+ env,
+ });
+ return path.resolve(instanceRoot, "companies", safeCompanyId, CACHE_DIR_NAME);
+}
+
+/**
+ * Resolves the entry path for one identity: `//auth.json`.
+ * The `account_id` is sanitized by {@link toCacheKey}. After the join, this
+ * verifies the resolved entry path stays under the cache root and ends at exactly
+ * `/auth.json`. This function does no filesystem work; it is safe
+ * for a read path (the vend and the clear). (Security condition 3.)
+ */
+export function resolveCodexAuthCacheEntryPath(
+ env: NodeJS.ProcessEnv = process.env,
+ accountId: string,
+ companyId: string,
+): string {
+ const resolvedRoot = resolveCodexAuthCacheDir(env, companyId);
+ const safeKey = toCacheKey(accountId);
+ const entryDir = path.resolve(resolvedRoot, safeKey);
+ const entryPath = path.resolve(entryDir, CACHE_ENTRY_FILE);
+ const expectedEntryPath = path.join(resolvedRoot, safeKey, CACHE_ENTRY_FILE);
+ if (
+ !entryDir.startsWith(resolvedRoot + path.sep) ||
+ path.dirname(entryDir) !== resolvedRoot ||
+ entryPath !== expectedEntryPath
+ ) {
+ throw new Error("codex auth cache: resolved entry path escapes the cache root");
+ }
+ return entryPath;
+}
+
+/**
+ * Ensures one directory exists and is private (mode 0700). Fails closed with
+ * `lstat` (not `stat`) when the existing path is a symlink or a non-directory,
+ * so the cache never writes through a planted symlink. (Security condition 3.)
+ */
+async function ensurePrivateDir(dir: string): Promise {
+ const existing = await lstat(dir).catch((error: NodeJS.ErrnoException) => {
+ if (error.code === "ENOENT") return null;
+ throw error;
+ });
+ if (existing) {
+ if (existing.isSymbolicLink() || !existing.isDirectory()) {
+ throw new Error("codex auth cache: cache path is a symlink or a non-directory");
+ }
+ return;
+ }
+ await mkdir(dir, { recursive: true, mode: PRIVATE_DIR_MODE });
+}
+
+/**
+ * Resolves the entry path and creates the cache root and the entry directory
+ * private (mode 0700), each guarded by `lstat`. Use this on the write path
+ * before the cache slot is written.
+ */
+export async function ensureCodexAuthCacheEntryDir(
+ env: NodeJS.ProcessEnv = process.env,
+ accountId: string,
+ companyId: string,
+): Promise {
+ const entryPath = resolveCodexAuthCacheEntryPath(env, accountId, companyId);
+ await ensurePrivateDir(resolveCodexAuthCacheDir(env, companyId));
+ await ensurePrivateDir(path.dirname(entryPath));
+ return entryPath;
+}
+
+/**
+ * Reads the usable subscription `account_id` from an `auth.json` payload. Returns
+ * `null` for an absent, unusable, or api-key credential (no subscription
+ * identity). This mirrors `parseAuth` in `codex-auth-merge-decision.cjs`; keep
+ * the two in step when the auth format changes.
+ */
+export function readSubscriptionAccountId(bytes: Buffer): string | null {
+ let parsed: unknown;
+ try {
+ parsed = JSON.parse(bytes.toString("utf8"));
+ } catch {
+ return null;
+ }
+ if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
+ return null;
+ }
+ const record = parsed as Record;
+ const apiKey = record.OPENAI_API_KEY;
+ if (typeof apiKey === "string" && apiKey.trim().length > 0) {
+ return null;
+ }
+ const tokens = record.tokens;
+ if (tokens === null || typeof tokens !== "object" || Array.isArray(tokens)) {
+ return null;
+ }
+ const tokenRecord = tokens as Record;
+ const accountId = typeof tokenRecord.account_id === "string" ? tokenRecord.account_id.trim() : "";
+ const hasTokenMaterial = ["id_token", "access_token", "refresh_token"].some((key) => {
+ const value = tokenRecord[key];
+ return typeof value === "string" && value.trim().length > 0;
+ });
+ if (!accountId || !hasTokenMaterial) {
+ return null;
+ }
+ return accountId;
+}
+
+async function decideExitCode(
+ sourcePath: string,
+ destinationPath: string,
+ options: { seedIfDestAbsent?: boolean } = {},
+): Promise {
+ const args = options.seedIfDestAbsent
+ ? [DECISION_SCRIPT_PATH, SEED_IF_DEST_ABSENT_FLAG, sourcePath, destinationPath]
+ : [DECISION_SCRIPT_PATH, sourcePath, destinationPath];
+ try {
+ await execFile("node", args);
+ } catch (error) {
+ const code = (error as { code?: unknown }).code;
+ if (code === USE_SOURCE_EXIT || code === KEEP_DESTINATION_EXIT) {
+ return code;
+ }
+ const detail =
+ typeof code === "string"
+ ? `node could not be executed (${code})`
+ : typeof code === "number"
+ ? `unexpected predicate exit code ${code}`
+ : error instanceof Error
+ ? error.message
+ : String(error);
+ throw new Error(`codex auth cache decision predicate failed: ${detail}`);
+ }
+ throw new Error("codex auth cache decision predicate exited 0 (expected 10 or 20)");
+}
+
+export type WriteCodexAuthCacheEntryOutcome = "written" | "kept-slot";
+
+/**
+ * Writes `sandboxAuthBytes` into its per-identity cache slot when the slot is
+ * absent, or when the source is a strictly-newer same-identity subscription
+ * credential (Phase 2 seed mode). The mutation runs under the merge lock on the
+ * slot directory. Never logs token bytes or a raw `account_id`.
+ */
+export async function writeCodexAuthCacheEntry(input: {
+ sandboxAuthBytes: Buffer;
+ cacheEntryPath: string;
+ log: (line: string) => void | Promise;
+}): Promise {
+ const { sandboxAuthBytes, cacheEntryPath, log } = input;
+ const cacheEntryDir = path.dirname(cacheEntryPath);
+ return withDirectoryMergeLock(cacheEntryDir, async () => {
+ const stagedTempPath = path.join(
+ cacheEntryDir,
+ `.auth.json.cache-source-${process.pid}-${randomUUID()}.tmp`,
+ );
+ const handle = await open(stagedTempPath, "wx", 0o600);
+ try {
+ await handle.writeFile(sandboxAuthBytes);
+ await handle.close();
+ const decision = await decideExitCode(stagedTempPath, cacheEntryPath, {
+ seedIfDestAbsent: true,
+ });
+ if (decision === USE_SOURCE_EXIT) {
+ await rename(stagedTempPath, cacheEntryPath);
+ await log("[paperclip] Codex auth cache: wrote the per-identity cache slot at mode 0600.");
+ return "written";
+ }
+ await log(
+ "[paperclip] Codex auth cache: kept the cache slot (source is not a strictly-newer same-identity subscription credential).",
+ );
+ return "kept-slot";
+ } finally {
+ await handle.close().catch(() => undefined);
+ await rm(stagedTempPath, { force: true }).catch(() => undefined);
+ }
+ });
+}
+
+export type VendCodexAuthOutcome = "vended" | "kept-host" | "no-host-identity";
+
+/**
+ * Refreshes the identity the host already holds with a strictly-newer cached
+ * credential of the same `account_id`. Identity-anchored:
+ *
+ * 1. Read the host identity from `sharedHomeAuthPath` first.
+ * 2. When the host holds no usable subscription identity, do nothing. The vend
+ * never picks a credential from the cache at random (matrix rows 3, 4).
+ * 3. Resolve ONLY that exact identity's cache slot. The vend never scans the
+ * cache root. (Security condition 4.)
+ * 4. Run the DEFAULT-mode predicate (cache slot as source, shared home as
+ * destination). Install the cache copy only when it is strictly newer for the
+ * same identity.
+ *
+ * The vend never changes identity and never stages a spent single-use refresh
+ * token (the strictly-newer rule guards it). Never logs token bytes or a raw
+ * `account_id`.
+ */
+export async function selectVendCredential(
+ sharedHomeAuthPath: string,
+ resolveCacheEntryPath: (accountId: string) => string,
+ log: (line: string) => void | Promise,
+): Promise {
+ const hostBytes = await readFile(sharedHomeAuthPath).catch((error: NodeJS.ErrnoException) => {
+ if (error.code === "ENOENT") return null;
+ throw error;
+ });
+ const hostAccountId = hostBytes ? readSubscriptionAccountId(hostBytes) : null;
+ if (!hostAccountId) {
+ // No host identity: the vend does nothing (identity-anchor rule).
+ return "no-host-identity";
+ }
+
+ let cacheEntryPath: string;
+ try {
+ cacheEntryPath = resolveCacheEntryPath(hostAccountId);
+ } catch {
+ // A host identity that cannot form a safe cache key never vends.
+ return "no-host-identity";
+ }
+
+ const hostDir = path.dirname(sharedHomeAuthPath);
+ return withDirectoryMergeLock(hostDir, async () => {
+ // Read the cached bytes under the lock, then stage exactly those bytes into a
+ // private (0600) temp next to the host target. The predicate reads the temp,
+ // so the vend installs exactly the bytes the predicate approved. This closes
+ // the read-after-validate skew: a separate reader of `cacheEntryPath` could
+ // otherwise see different bytes than the ones installed. The rename over the
+ // host target stays device-local and atomic.
+ const cacheBytes = await readFile(cacheEntryPath).catch((error: NodeJS.ErrnoException) => {
+ if (error.code === "ENOENT") return null;
+ throw error;
+ });
+ if (!cacheBytes) {
+ await log("[paperclip] Codex auth cache: no cached credential for the host identity; host credential kept.");
+ return "kept-host";
+ }
+ const stagedTempPath = path.join(
+ hostDir,
+ `.auth.json.vend-${process.pid}-${randomUUID()}.tmp`,
+ );
+ const handle = await open(stagedTempPath, "wx", 0o600);
+ try {
+ await handle.writeFile(cacheBytes);
+ await handle.close();
+ // Default-mode predicate: install the cache copy only when it is strictly
+ // newer for the SAME identity. Same-identity + strictly-newer keeps the
+ // change additive and semantics-preserving.
+ const decision = await decideExitCode(stagedTempPath, sharedHomeAuthPath);
+ if (decision === USE_SOURCE_EXIT) {
+ await rename(stagedTempPath, sharedHomeAuthPath);
+ await log(
+ "[paperclip] Codex auth cache: refreshed the host credential with a strictly-newer cached copy of the same identity at mode 0600.",
+ );
+ return "vended";
+ }
+ await log(
+ "[paperclip] Codex auth cache: host credential kept (the cached copy is not strictly newer for the same identity).",
+ );
+ return "kept-host";
+ } finally {
+ await handle.close().catch(() => undefined);
+ await rm(stagedTempPath, { force: true }).catch(() => undefined);
+ }
+ });
+}
+
+/**
+ * Removes exactly one identity slot from the cache. The `account_id` is
+ * sanitized by {@link toCacheKey}, so the removal can never escape the cache
+ * root. A missing slot is a benign no-op. Any other removal error is fail-loud.
+ */
+export async function clearCodexAuthCacheEntry(
+ env: NodeJS.ProcessEnv = process.env,
+ accountId: string,
+ companyId: string,
+): Promise {
+ const entryPath = resolveCodexAuthCacheEntryPath(env, accountId, companyId);
+ const entryDir = path.dirname(entryPath);
+ // A missing slot is a benign no-op. Pre-check before the lock: the merge lock
+ // sits next to the slot under the cache root, so locking a slot whose cache
+ // root does not exist would fail on the lock directory itself.
+ const existing = await lstat(entryDir).catch((error: NodeJS.ErrnoException) => {
+ if (error.code === "ENOENT") return null;
+ throw error;
+ });
+ if (!existing) return;
+ await withDirectoryMergeLock(entryDir, async () => {
+ await rm(entryDir, { recursive: true, force: true });
+ });
+}
+
+/**
+ * Removes every slot in the company-scoped cache root. A missing cache root is a
+ * benign no-op.
+ */
+export async function clearCodexAuthCache(
+ env: NodeJS.ProcessEnv = process.env,
+ companyId: string,
+): Promise {
+ const cacheDir = resolveCodexAuthCacheDir(env, companyId);
+ await rm(cacheDir, { recursive: true, force: true });
+}
diff --git a/packages/adapters/codex-local/src/server/codex-auth-copyback.test.ts b/packages/adapters/codex-local/src/server/codex-auth-copyback.test.ts
index 4aff672ddb48..e238eba08548 100644
--- a/packages/adapters/codex-local/src/server/codex-auth-copyback.test.ts
+++ b/packages/adapters/codex-local/src/server/codex-auth-copyback.test.ts
@@ -1,9 +1,15 @@
-import { chmod, lstat, mkdtemp, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
+import { chmod, lstat, mkdir, mkdtemp, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, describe, expect, it } from "vitest";
import { copyBackCodexAuth } from "./codex-auth-copyback.js";
+import {
+ ensureCodexAuthCacheEntryDir,
+ resolveCodexAuthCacheDir,
+ resolveCodexAuthCacheEntryPath,
+} from "./codex-auth-cache.js";
+import { resolveSharedCodexHomeDir } from "./codex-home.js";
// The copy-back module reuses the exact same direction-agnostic decision
// predicate (`codex-auth-merge-decision.cjs`) that the inbound extract path
@@ -313,3 +319,299 @@ describe("copyBackCodexAuth", () => {
expect(combined).not.toContain("refresh-token");
});
});
+
+// The teardown copy-back also writes the fresher, usable subscription credential
+// into its per-identity cache slot, keyed by the real `account_id`. The cache is
+// a SEPARATE store; the host default overwrite and the cache slot are asserted
+// independently. This suite drives the real `.cjs` predicate (default mode for
+// the host store, seed mode for the cache slot) against a real host tmp
+// filesystem.
+describe("copyBackCodexAuth identity-keyed cache write", () => {
+ const cleanupDirs: string[] = [];
+ const COMPANY_ID = "company-a";
+
+ afterEach(async () => {
+ while (cleanupDirs.length > 0) {
+ const dir = cleanupDirs.pop();
+ if (!dir) continue;
+ await chmod(dir, 0o700).catch(() => undefined);
+ await rm(dir, { recursive: true, force: true }).catch(() => undefined);
+ }
+ });
+
+ function subscriptionAuth(input: { accountId: string; lastRefresh?: string; marker: string }): string {
+ return JSON.stringify({
+ tokens: {
+ id_token: `id-token-${input.marker}`,
+ access_token: `access-token-${input.marker}`,
+ refresh_token: `refresh-token-${input.marker}`,
+ account_id: input.accountId,
+ },
+ ...(input.lastRefresh ? { last_refresh: input.lastRefresh } : {}),
+ });
+ }
+
+ function apiKeyAuth(marker: string): string {
+ return JSON.stringify({ OPENAI_API_KEY: `sk-${marker}` });
+ }
+
+ const NEWER = "2026-07-09T02:00:00Z";
+ const OLDER = "2026-07-09T01:00:00Z";
+
+ async function makeEnv(extra: Record = {}): Promise<{
+ env: NodeJS.ProcessEnv;
+ sharedHomeAuthPath: string;
+ sharedHome: string;
+ }> {
+ const home = await mkdtemp(path.join(os.tmpdir(), "paperclip-codex-copyback-cache-"));
+ cleanupDirs.push(home);
+ const env: NodeJS.ProcessEnv = {
+ PAPERCLIP_HOME: home,
+ PAPERCLIP_INSTANCE_ID: "default",
+ CODEX_HOME: path.join(home, "shared-codex"),
+ ...extra,
+ };
+ const sharedHome = resolveSharedCodexHomeDir(env);
+ await mkdir(sharedHome, { recursive: true });
+ return { env, sharedHomeAuthPath: path.join(sharedHome, "auth.json"), sharedHome };
+ }
+
+ async function runWithCache(input: {
+ sandboxAuth: string;
+ hostAuth?: string;
+ env: NodeJS.ProcessEnv;
+ sharedHomeAuthPath: string;
+ cacheEnabledEnv?: NodeJS.ProcessEnv;
+ }): Promise<{
+ outcome: Awaited>;
+ finalHostAuth: string | null;
+ logs: string[];
+ }> {
+ if (input.hostAuth !== undefined) {
+ await writeFile(input.sharedHomeAuthPath, input.hostAuth, { mode: 0o600 });
+ }
+ const logs: string[] = [];
+ const outcome = await copyBackCodexAuth({
+ readSandboxAuth: async () => Buffer.from(input.sandboxAuth, "utf8"),
+ hostAuthPath: input.sharedHomeAuthPath,
+ log: (line) => {
+ logs.push(line);
+ },
+ resolveCacheEntryPath: (accountId) =>
+ ensureCodexAuthCacheEntryDir(input.env, accountId, COMPANY_ID),
+ env: input.cacheEnabledEnv ?? input.env,
+ });
+ const finalHostAuth = await readFile(input.sharedHomeAuthPath, "utf8").catch(
+ (error: NodeJS.ErrnoException) => (error.code === "ENOENT" ? null : Promise.reject(error)),
+ );
+ return { outcome, finalHostAuth, logs };
+ }
+
+ async function readCacheSlot(env: NodeJS.ProcessEnv, accountId: string): Promise {
+ const entryPath = resolveCodexAuthCacheEntryPath(env, accountId, COMPANY_ID);
+ return readFile(entryPath, "utf8").catch((error: NodeJS.ErrnoException) =>
+ error.code === "ENOENT" ? null : Promise.reject(error),
+ );
+ }
+
+ it("host absent (matrix row 3): host default store stays empty, cache slot holds the credential keyed by account_id", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({ accountId: "acct-y", lastRefresh: NEWER, marker: "row3" });
+
+ const result = await runWithCache({ sandboxAuth, env, sharedHomeAuthPath });
+
+ expect(result.outcome).toBe("kept-host");
+ expect(result.finalHostAuth).toBeNull(); // host never seeded
+ expect(await readCacheSlot(env, "acct-y")).toBe(sandboxAuth);
+ });
+
+ it("host present, same identity, sandbox newer (matrix row 1a): host default overwritten AND cache slot updated", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER, marker: "sandbox" });
+ const hostAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" });
+
+ const result = await runWithCache({ sandboxAuth, hostAuth, env, sharedHomeAuthPath });
+
+ expect(result.outcome).toBe("copied");
+ expect(result.finalHostAuth).toBe(sandboxAuth);
+ expect(await readCacheSlot(env, "acct-x")).toBe(sandboxAuth);
+ });
+
+ it("host present, different identity (matrix row 1b): host default untouched, cache slot holds the new identity only", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({ accountId: "acct-y", lastRefresh: NEWER, marker: "sandbox" });
+ const hostAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" });
+
+ const result = await runWithCache({ sandboxAuth, hostAuth, env, sharedHomeAuthPath });
+
+ expect(result.outcome).toBe("kept-host");
+ expect(result.finalHostAuth).toBe(hostAuth); // host keeps its own identity X
+ expect(await readCacheSlot(env, "acct-y")).toBe(sandboxAuth); // slot Y written
+ expect(await readCacheSlot(env, "acct-x")).toBeNull(); // no slot for host identity
+ });
+
+ it("sandbox apikey or unusable: neither the host default nor the cache changes", async () => {
+ for (const sandboxAuth of [apiKeyAuth("sbx"), "{not valid json"]) {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const hostAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" });
+ const result = await runWithCache({ sandboxAuth, hostAuth, env, sharedHomeAuthPath });
+ expect(result.outcome).toBe("kept-host");
+ expect(result.finalHostAuth).toBe(hostAuth);
+ const cacheDir = resolveCodexAuthCacheDir(env, COMPANY_ID);
+ // No cache slot directory is created at all for a credential with no identity.
+ await expect(readdir(cacheDir)).rejects.toThrow();
+ }
+ });
+
+ it("off-switch off (matrix row 1a inputs): teardown cache write is a no-op and the host default overwrite still runs", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER, marker: "sandbox" });
+ const hostAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" });
+
+ const result = await runWithCache({
+ sandboxAuth,
+ hostAuth,
+ env,
+ sharedHomeAuthPath,
+ cacheEnabledEnv: { ...env, PAPERCLIP_CODEX_AUTH_CACHE: "0" },
+ });
+
+ // Host overwrite still runs with the off-switch off.
+ expect(result.outcome).toBe("copied");
+ expect(result.finalHostAuth).toBe(sandboxAuth);
+ // No cache slot is written.
+ const cacheDir = resolveCodexAuthCacheDir(env, COMPANY_ID);
+ await expect(readdir(cacheDir)).rejects.toThrow();
+ });
+
+ it("two concurrent teardown writes for the same identity leave one valid cache slot (no partial file)", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ await writeFile(
+ sharedHomeAuthPath,
+ subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" }),
+ { mode: 0o600 },
+ );
+ const run = (marker: string) =>
+ copyBackCodexAuth({
+ readSandboxAuth: async () =>
+ Buffer.from(subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER, marker }), "utf8"),
+ hostAuthPath: sharedHomeAuthPath,
+ log: () => {},
+ resolveCacheEntryPath: (accountId) => ensureCodexAuthCacheEntryDir(env, accountId, COMPANY_ID),
+ env,
+ });
+ await Promise.all([run("a"), run("b")]);
+
+ const entryPath = resolveCodexAuthCacheEntryPath(env, "acct-x", COMPANY_ID);
+ const slotDir = path.dirname(entryPath);
+ // Exactly one valid slot file, no leftover staging temp.
+ expect(await readdir(slotDir)).toEqual(["auth.json"]);
+ const finalSlot = await readFile(entryPath, "utf8");
+ expect(JSON.parse(finalSlot).tokens.account_id).toBe("acct-x");
+ });
+
+ it("the cache write never emits source token bytes or a raw account_id to the log", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({
+ accountId: "SECRET-ACCT",
+ lastRefresh: NEWER,
+ marker: "TOKEN-SENTINEL",
+ });
+ const result = await runWithCache({ sandboxAuth, env, sharedHomeAuthPath });
+ expect(await readCacheSlot(env, "SECRET-ACCT")).toBe(sandboxAuth);
+ const combined = result.logs.join("\n");
+ expect(combined).not.toContain("SENTINEL");
+ expect(combined).not.toContain("SECRET-ACCT");
+ expect(combined).not.toContain("id-token");
+ });
+
+ it("a read-only cache directory does not fail the copy-back: the successful host result is kept and no partial slot remains", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER, marker: "sandbox" });
+ const hostAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" });
+ await writeFile(sharedHomeAuthPath, hostAuth, { mode: 0o600 });
+
+ // Pre-create the entry directory, then make it read-only so the cache-slot
+ // temp create fails. The host overwrite runs first and must stay intact. The
+ // additive cache write is best-effort, so its failure must not throw.
+ const entryPath = await ensureCodexAuthCacheEntryDir(env, "acct-x", COMPANY_ID);
+ const slotDir = path.dirname(entryPath);
+ const logs: string[] = [];
+ await chmod(slotDir, 0o500);
+ let outcome: string;
+ try {
+ outcome = await copyBackCodexAuth({
+ readSandboxAuth: async () => Buffer.from(sandboxAuth, "utf8"),
+ hostAuthPath: sharedHomeAuthPath,
+ log: (line) => {
+ logs.push(line);
+ },
+ resolveCacheEntryPath: (accountId) => ensureCodexAuthCacheEntryDir(env, accountId, COMPANY_ID),
+ env,
+ });
+ } finally {
+ await chmod(slotDir, 0o700);
+ }
+
+ // The host copy-back succeeded and its result is returned unchanged.
+ expect(outcome).toBe("copied");
+ // Host default overwrite ran before the cache write and stays applied.
+ expect(await readFile(sharedHomeAuthPath, "utf8")).toBe(sandboxAuth);
+ // No partial slot file; only the (empty) slot directory remains.
+ expect(await readdir(slotDir)).toEqual([]);
+ // The failure is visible in the log with only the errno code. The raw
+ // account_id (which the failing slot path embeds) never reaches the log.
+ const combined = logs.join("\n");
+ expect(combined).toContain("additive cache write failed (EACCES)");
+ expect(combined).not.toContain("acct-x");
+ });
+
+ it("a rejecting cache-failure log does not override the successful host copy-back result", async () => {
+ const { env, sharedHomeAuthPath } = await makeEnv();
+ const sandboxAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER, marker: "sandbox" });
+ const hostAuth = subscriptionAuth({ accountId: "acct-x", lastRefresh: OLDER, marker: "host" });
+ await writeFile(sharedHomeAuthPath, hostAuth, { mode: 0o600 });
+
+ // Force the additive cache write to fail: pre-create the slot directory,
+ // then make it read-only so the slot temp create fails with EACCES.
+ const entryPath = await ensureCodexAuthCacheEntryDir(env, "acct-x", COMPANY_ID);
+ const slotDir = path.dirname(entryPath);
+ await chmod(slotDir, 0o500);
+
+ // The logger rejects for the cache-failure diagnostic line only. The host
+ // copy-back already installed the sandbox credential on disk, so this
+ // rejection must not propagate and must not override the "copied" result.
+ const logs: string[] = [];
+ let outcome: Awaited> | undefined;
+ let thrown: unknown;
+ try {
+ outcome = await copyBackCodexAuth({
+ readSandboxAuth: async () => Buffer.from(sandboxAuth, "utf8"),
+ hostAuthPath: sharedHomeAuthPath,
+ log: (line) => {
+ logs.push(line);
+ if (line.includes("additive cache write failed")) {
+ return Promise.reject(new Error("log sink boom"));
+ }
+ },
+ resolveCacheEntryPath: (accountId) => ensureCodexAuthCacheEntryDir(env, accountId, COMPANY_ID),
+ env,
+ }).catch((error: unknown) => {
+ thrown = error;
+ return undefined;
+ });
+ } finally {
+ await chmod(slotDir, 0o700);
+ }
+
+ // The rejecting cache-failure log never surfaces as a thrown error.
+ expect(thrown).toBeUndefined();
+ // The host copy-back result is kept intact.
+ expect(outcome).toBe("copied");
+ expect(await readFile(sharedHomeAuthPath, "utf8")).toBe(sandboxAuth);
+ // No partial slot file remains after the failed cache write.
+ expect(await readdir(slotDir)).toEqual([]);
+ // The cache-failure diagnostic was attempted even though the sink rejected.
+ expect(logs.some((line) => line.includes("additive cache write failed (EACCES)"))).toBe(true);
+ });
+});
diff --git a/packages/adapters/codex-local/src/server/codex-auth-copyback.ts b/packages/adapters/codex-local/src/server/codex-auth-copyback.ts
index bf960701b7ea..19028af02176 100644
--- a/packages/adapters/codex-local/src/server/codex-auth-copyback.ts
+++ b/packages/adapters/codex-local/src/server/codex-auth-copyback.ts
@@ -5,6 +5,11 @@ import { fileURLToPath } from "node:url";
import { promisify } from "node:util";
import { randomUUID } from "node:crypto";
import { withDirectoryMergeLock } from "@paperclipai/adapter-utils/workspace-restore-merge";
+import {
+ isCodexAuthCacheEnabled,
+ readSubscriptionAccountId,
+ writeCodexAuthCacheEntry,
+} from "./codex-auth-cache.js";
const execFile = promisify(execFileCallback);
@@ -41,6 +46,19 @@ export interface CopyBackCodexAuthInput {
hostAuthPath: string;
/** Non-leaking progress sink: receives decision/outcome lines only. */
log: (line: string) => void | Promise;
+ /**
+ * Resolves and ensures the per-identity cache slot path for a sandbox
+ * `account_id`. When this is provided AND the cache off-switch is on, the
+ * copy-back also writes the fresher, usable subscription credential into its
+ * per-identity cache slot as a second, additive write, keyed by the real
+ * `account_id`. This is independent of the host default overwrite: it can
+ * write a cache slot for a different identity than the host holds (matrix rows
+ * 1b, 3), and it never touches the host default store. When absent, no cache
+ * write happens.
+ */
+ resolveCacheEntryPath?: (accountId: string) => Promise;
+ /** Environment for the cache off-switch read. Defaults to `process.env`. */
+ env?: NodeJS.ProcessEnv;
}
async function decideExitCode(sourcePath: string, destinationPath: string): Promise {
@@ -93,7 +111,7 @@ async function decideExitCode(sourcePath: string, destinationPath: string): Prom
* Never logs token bytes — only the decision outcome.
*/
export async function copyBackCodexAuth(input: CopyBackCodexAuthInput): Promise {
- const { readSandboxAuth, hostAuthPath, log } = input;
+ const { readSandboxAuth, hostAuthPath, log, resolveCacheEntryPath, env } = input;
// Read first (outside the lock) — a read never mutates the host, so there is
// nothing to serialize yet. A genuinely absent sandbox `auth.json` (ENOENT —
@@ -116,7 +134,7 @@ export async function copyBackCodexAuth(input: CopyBackCodexAuthInput): Promise<
const hostDir = path.dirname(hostAuthPath);
await mkdir(hostDir, { recursive: true });
- return withDirectoryMergeLock(hostDir, async () => {
+ const hostOutcome = await withDirectoryMergeLock(hostDir, async () => {
// Stage on the same filesystem as the host target so both the predicate read
// and the final rename stay device-local (rename across devices is not
// atomic and would fail with EXDEV).
@@ -151,4 +169,43 @@ export async function copyBackCodexAuth(input: CopyBackCodexAuthInput): Promise<
await rm(stagedTempPath, { force: true }).catch(() => undefined);
}
});
+
+ // Additive cache write. Independent of the host default overwrite above: it
+ // runs on its own directory lock, keys the slot by the real sandbox
+ // `account_id`, and can write a slot for a different identity than the host
+ // holds. It never touches the host default store. The off-switch (default on)
+ // makes this a no-op when disabled. Only a usable subscription credential has
+ // an identity to key; an api-key or unusable sandbox credential is skipped.
+ //
+ // The cache write is best-effort. The host copy-back above already finished
+ // and set `hostOutcome`, so a failure of this additive write must not replace
+ // that successful result. Catch the error, log it, and return `hostOutcome`.
+ // The cache stays a hint: the next teardown re-attempts the write.
+ if (resolveCacheEntryPath && isCodexAuthCacheEnabled(env)) {
+ try {
+ const sandboxAccountId = readSubscriptionAccountId(sandboxAuthBytes);
+ if (sandboxAccountId) {
+ const cacheEntryPath = await resolveCacheEntryPath(sandboxAccountId);
+ await writeCodexAuthCacheEntry({ sandboxAuthBytes, cacheEntryPath, log });
+ }
+ } catch (error) {
+ // Log only the errno code, never the error message. The message embeds the
+ // cache slot path, and the slot path embeds the raw `account_id`; the code
+ // (for example EACCES or ENOSPC) makes the failure diagnosable without a
+ // leak. Token bytes never reach the log.
+ const code = (error as NodeJS.ErrnoException | null)?.code ?? "unknown";
+ // The host copy-back above already finished and set `hostOutcome`. This
+ // diagnostic log is the last step, so a rejecting logger must not throw
+ // and turn that successful result into a failed copy-back. Guard the log:
+ // a rejection here is swallowed, and the function still returns
+ // `hostOutcome` below.
+ await Promise.resolve(
+ log(
+ `[paperclip] Codex auth cache: additive cache write failed (${code}); host copy-back result kept.`,
+ ),
+ ).catch(() => undefined);
+ }
+ }
+
+ return hostOutcome;
}
diff --git a/packages/adapters/codex-local/src/server/codex-auth-merge-decision.cjs b/packages/adapters/codex-local/src/server/codex-auth-merge-decision.cjs
index 358de3731ec6..fe557795a201 100644
--- a/packages/adapters/codex-local/src/server/codex-auth-merge-decision.cjs
+++ b/packages/adapters/codex-local/src/server/codex-auth-merge-decision.cjs
@@ -50,15 +50,45 @@ function parseAuth(filePath) {
// argv[0] (first positional) = source auth.json path
// argv[1] (second positional) = destination auth.json path
//
+// The predicate has two modes, selected by a leading positional flag:
+//
+// default (no flag) — fail closed to the destination; used by the host default
+// store, whose fail-closed behavior must never change. An absent or
+// unusable destination keeps the destination (no seed from empty).
+// --seed-if-dest-absent — the opt-in cache-slot mode; used only by the
+// per-identity cache slot helper. It ADDS one behavior on top of the default:
+// when the destination is unusable (absent or unparseable) AND the source is
+// a usable subscription credential, use the source (fill the empty slot). It
+// never relaxes the different-identity, api-key, or unusable-source guards.
+//
+// A leading positional flag (not an environment variable) keeps the mode
+// explicit per call, so a host-default two-path call can never enter seed mode.
+//
// Exit 10 = use source; exit 20 = keep destination. The predicate only ever
// reads the two files and exits with a code — it never prints token bytes.
const USE_SOURCE = 10;
const KEEP_DESTINATION = 20;
+const SEED_IF_DEST_ABSENT_FLAG = "--seed-if-dest-absent";
-const [sourceAuthPath, destinationAuthPath] = process.argv.slice(2);
+const rawArgs = process.argv.slice(2);
+const seedIfDestAbsent = rawArgs[0] === SEED_IF_DEST_ABSENT_FLAG;
+const [sourceAuthPath, destinationAuthPath] = seedIfDestAbsent ? rawArgs.slice(1) : rawArgs;
const sourceAuth = parseAuth(sourceAuthPath);
const destinationAuth = parseAuth(destinationAuthPath);
+// Seed mode only: fill an ABSENT (unusable) destination slot from a usable
+// subscription source. A subscription-kind source is guaranteed usable and to
+// carry a real account_id (parseAuth returns "subscription" only then), so this
+// is never a random pick. This branch changes ONLY the destination-unusable
+// case; the api-key and unusable-source guards below still keep the destination.
+if (
+ seedIfDestAbsent &&
+ destinationAuth.kind === "unusable" &&
+ sourceAuth.kind === "subscription"
+) {
+ process.exit(USE_SOURCE);
+}
+
// Fail closed to the destination unless both sides are the same usable,
// subscription-kind identity — an unusable side, an api-key credential, a kind
// mismatch, or a different account_id all keep the destination copy.
diff --git a/packages/adapters/codex-local/src/server/codex-auth-merge-decision.test.ts b/packages/adapters/codex-local/src/server/codex-auth-merge-decision.test.ts
new file mode 100644
index 000000000000..5c05352cda96
--- /dev/null
+++ b/packages/adapters/codex-local/src/server/codex-auth-merge-decision.test.ts
@@ -0,0 +1,167 @@
+import { execFile as execFileCallback } from "node:child_process";
+import { mkdtemp, rm, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { promisify } from "node:util";
+import { afterEach, describe, expect, it } from "vitest";
+
+const execFile = promisify(execFileCallback);
+
+// This suite pins the opt-in seed mode of the single decision predicate. The
+// default (no-flag) call keeps the fail-closed host-default contract unchanged.
+// The leading positional `--seed-if-dest-absent` flag adds one behaviour: fill
+// an ABSENT destination slot from a usable subscription source. The flag never
+// relaxes the different-identity, api-key, or unusable-source guards, and a
+// default two-path call can never enter seed mode.
+describe("codex-auth-merge-decision predicate seed mode", () => {
+ const cleanupDirs: string[] = [];
+
+ afterEach(async () => {
+ while (cleanupDirs.length > 0) {
+ const dir = cleanupDirs.pop();
+ if (!dir) continue;
+ await rm(dir, { recursive: true, force: true }).catch(() => undefined);
+ }
+ });
+
+ const decisionScriptPath = fileURLToPath(
+ new URL("./codex-auth-merge-decision.cjs", import.meta.url),
+ );
+
+ const USE_SOURCE = 10;
+ const KEEP_DESTINATION = 20;
+ const NEWER = "2026-07-09T02:00:00Z";
+ const OLDER = "2026-07-09T01:00:00Z";
+
+ function subscriptionAuth(input: { accountId: string; lastRefresh?: string; marker?: string }): string {
+ const suffix = input.marker ?? input.accountId;
+ return JSON.stringify({
+ tokens: {
+ id_token: `id-token-${suffix}`,
+ access_token: `access-token-${suffix}`,
+ refresh_token: `refresh-token-${suffix}`,
+ account_id: input.accountId,
+ },
+ ...(input.lastRefresh ? { last_refresh: input.lastRefresh } : {}),
+ });
+ }
+
+ function apiKeyAuth(marker: string): string {
+ return JSON.stringify({ OPENAI_API_KEY: `sk-${marker}` });
+ }
+
+ const ABSENT = Symbol("absent");
+
+ async function runDecision(input: {
+ seed?: boolean;
+ sourceAuth: string;
+ destinationAuth: string | typeof ABSENT;
+ }): Promise {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "paperclip-codex-seed-decision-"));
+ cleanupDirs.push(dir);
+ const sourcePath = path.join(dir, "source-auth.json");
+ const destinationPath = path.join(dir, "destination-auth.json");
+ await writeFile(sourcePath, input.sourceAuth, { mode: 0o600 });
+ if (input.destinationAuth !== ABSENT) {
+ await writeFile(destinationPath, input.destinationAuth, { mode: 0o600 });
+ }
+ // The flag, when present, is the leading positional argument, parsed before
+ // the two path arguments.
+ const args = input.seed
+ ? [decisionScriptPath, "--seed-if-dest-absent", sourcePath, destinationPath]
+ : [decisionScriptPath, sourcePath, destinationPath];
+ try {
+ await execFile("node", args);
+ return 0;
+ } catch (error) {
+ const failure = error as { code?: unknown };
+ if (typeof failure.code === "number") return failure.code;
+ throw error;
+ }
+ }
+
+ it("default mode keeps destination when destination is absent (host-default fail-closed)", async () => {
+ const code = await runDecision({
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER }),
+ destinationAuth: ABSENT,
+ });
+ expect(code).toBe(KEEP_DESTINATION);
+ });
+
+ it("seed mode uses source when destination is absent and source is a usable subscription credential", async () => {
+ const code = await runDecision({
+ seed: true,
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER }),
+ destinationAuth: ABSENT,
+ });
+ expect(code).toBe(USE_SOURCE);
+ });
+
+ it("seed mode uses source when destination is unparseable and source is a usable subscription credential", async () => {
+ const code = await runDecision({
+ seed: true,
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER }),
+ destinationAuth: "{not valid json",
+ });
+ expect(code).toBe(USE_SOURCE);
+ });
+
+ it("seed mode still keeps destination when source is apikey or unusable", async () => {
+ const apikeyCode = await runDecision({
+ seed: true,
+ sourceAuth: apiKeyAuth("source"),
+ destinationAuth: ABSENT,
+ });
+ expect(apikeyCode).toBe(KEEP_DESTINATION);
+
+ const unusableCode = await runDecision({
+ seed: true,
+ sourceAuth: "{not valid json",
+ destinationAuth: ABSENT,
+ });
+ expect(unusableCode).toBe(KEEP_DESTINATION);
+ });
+
+ it("seed mode still keeps destination when destination holds a different account_id", async () => {
+ const code = await runDecision({
+ seed: true,
+ sourceAuth: subscriptionAuth({ accountId: "acct-x", lastRefresh: NEWER }),
+ destinationAuth: subscriptionAuth({ accountId: "acct-y", lastRefresh: OLDER }),
+ });
+ expect(code).toBe(KEEP_DESTINATION);
+ });
+
+ it("seed mode keeps the same-identity strictly-newer contract for a present destination", async () => {
+ const newerCode = await runDecision({
+ seed: true,
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER, marker: "src" }),
+ destinationAuth: subscriptionAuth({ accountId: "acct", lastRefresh: OLDER, marker: "dst" }),
+ });
+ expect(newerCode).toBe(USE_SOURCE);
+
+ const tieCode = await runDecision({
+ seed: true,
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER, marker: "src" }),
+ destinationAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER, marker: "dst" }),
+ });
+ expect(tieCode).toBe(KEEP_DESTINATION);
+ });
+
+ it("the leading positional --seed-if-dest-absent flag is parsed before the two path arguments; a default two-path call never enters seed mode", async () => {
+ // Identical inputs; only the leading flag differs. Absent destination:
+ // default keeps, seed uses source. This proves a host-default two-path call
+ // (no flag) can never seed.
+ const defaultCode = await runDecision({
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER }),
+ destinationAuth: ABSENT,
+ });
+ const seedCode = await runDecision({
+ seed: true,
+ sourceAuth: subscriptionAuth({ accountId: "acct", lastRefresh: NEWER }),
+ destinationAuth: ABSENT,
+ });
+ expect(defaultCode).toBe(KEEP_DESTINATION);
+ expect(seedCode).toBe(USE_SOURCE);
+ });
+});
diff --git a/packages/adapters/codex-local/src/server/execute.ts b/packages/adapters/codex-local/src/server/execute.ts
index 10d4e83f7368..a39b7469297a 100644
--- a/packages/adapters/codex-local/src/server/execute.ts
+++ b/packages/adapters/codex-local/src/server/execute.ts
@@ -4,6 +4,12 @@ import { fileURLToPath } from "node:url";
import { inferOpenAiCompatibleBiller, type AdapterExecutionContext, type AdapterExecutionResult } from "@paperclipai/adapter-utils";
import { buildCodexAuthInboundProvision } from "./codex-auth-merge-scripts.js";
import { copyBackCodexAuth } from "./codex-auth-copyback.js";
+import {
+ ensureCodexAuthCacheEntryDir,
+ isCodexAuthCacheEnabled,
+ resolveCodexAuthCacheEntryPath,
+ selectVendCredential,
+} from "./codex-auth-cache.js";
import {
adapterExecutionTargetIsRemote,
adapterExecutionTargetRemoteCwd,
@@ -619,6 +625,29 @@ export async function execute(ctx: AdapterExecutionContext): Promise resolveCodexAuthCacheEntryPath(process.env, accountId, agent.companyId),
+ (line) => onLog("stdout", `${line}\n`),
+ ).catch(async (error) => {
+ // The vend is best-effort and additive. A vend failure must never block a
+ // run: log and fall through to seed from the unrefreshed shared credential.
+ await onLog(
+ "stderr",
+ `[paperclip] Codex auth cache: vend skipped after an error; using the shared credential as-is.\n`,
+ );
+ void error;
+ });
+ }
if (configuredCodexHome == null) {
await prepareManagedCodexHome(process.env, onLog, agent.companyId, {
apiKey: configuredOpenAiApiKey,
@@ -771,6 +800,14 @@ export async function execute(ctx: AdapterExecutionContext): Promise readFile(path.posix.join(assetDir, "auth.json")),
hostAuthPath: path.join(resolveSharedCodexHomeDir(process.env), "auth.json"),
log: (line) => onLog("stdout", `${line}\n`),
+ // Additive cache write (sandbox to host): also cache the
+ // sandbox subscription credential in its per-identity slot,
+ // keyed by the real `account_id`. Company-scoped root; the
+ // helper ensures the slot directory private and containment-
+ // guarded. The off-switch (default on) is read inside.
+ resolveCacheEntryPath: (accountId) =>
+ ensureCodexAuthCacheEntryDir(process.env, accountId, agent.companyId),
+ env: process.env,
})),
// No `exclude` denylist: `stagedCodexHomeDir` already contains
// ONLY the allowlisted files (auth/config/skills), so there is
@@ -1470,11 +1507,29 @@ export async function execute(ctx: AdapterExecutionContext): Promise undefined);
+ }
}
}
} finally {
diff --git a/packages/adapters/codex-local/src/ui/build-config.ts b/packages/adapters/codex-local/src/ui/build-config.ts
index e82293fefe9e..1a1da822b8df 100644
--- a/packages/adapters/codex-local/src/ui/build-config.ts
+++ b/packages/adapters/codex-local/src/ui/build-config.ts
@@ -1,4 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
+import { buildAdapterEnvConfig, type CreateConfigValues } from "@paperclipai/adapter-utils";
import { DEFAULT_CODEX_LOCAL_BYPASS_APPROVALS_AND_SANDBOX } from "../index.js";
function parseCommaArgs(value: string): string[] {
@@ -8,49 +8,6 @@ function parseCommaArgs(value: string): string[] {
.filter(Boolean);
}
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
-
function parseJsonObject(text: string): Record | null {
const trimmed = text.trim();
if (!trimmed) return null;
@@ -79,13 +36,7 @@ export function buildCodexLocalConfig(v: CreateConfigValues): Record 0) ac.env = env;
ac.search = v.search;
ac.fastMode = v.fastMode;
diff --git a/packages/adapters/cursor-cloud/src/ui/build-config.ts b/packages/adapters/cursor-cloud/src/ui/build-config.ts
index 3e37804028c6..7ffecbaa245d 100644
--- a/packages/adapters/cursor-cloud/src/ui/build-config.ts
+++ b/packages/adapters/cursor-cloud/src/ui/build-config.ts
@@ -1,47 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
-
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
+import { buildAdapterEnvConfig, type CreateConfigValues } from "@paperclipai/adapter-utils";
export function buildCursorCloudConfig(values: CreateConfigValues): Record {
const config: Record = {
@@ -52,13 +9,7 @@ export function buildCursorCloudConfig(values: CreateConfigValues): Record 0) {
config.env = env;
}
diff --git a/packages/adapters/cursor-local/src/ui/build-config.ts b/packages/adapters/cursor-local/src/ui/build-config.ts
index 0bc09e831234..a6da4ab9304a 100644
--- a/packages/adapters/cursor-local/src/ui/build-config.ts
+++ b/packages/adapters/cursor-local/src/ui/build-config.ts
@@ -1,4 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
+import { buildAdapterEnvConfig, type CreateConfigValues } from "@paperclipai/adapter-utils";
import { DEFAULT_CURSOR_LOCAL_MODEL } from "../index.js";
function parseCommaArgs(value: string): string[] {
@@ -8,49 +8,6 @@ function parseCommaArgs(value: string): string[] {
.filter(Boolean);
}
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
-
function normalizeMode(value: string): "plan" | "ask" | null {
const mode = value.trim().toLowerCase();
if (mode === "plan" || mode === "ask") return mode;
@@ -66,13 +23,7 @@ export function buildCursorLocalConfig(v: CreateConfigValues): Record 0) ac.env = env;
if (v.command) ac.command = v.command;
if (v.extraArgs) ac.extraArgs = parseCommaArgs(v.extraArgs);
diff --git a/packages/adapters/gemini-local/src/ui/build-config.ts b/packages/adapters/gemini-local/src/ui/build-config.ts
index baa54183087c..d500b8ef65a5 100644
--- a/packages/adapters/gemini-local/src/ui/build-config.ts
+++ b/packages/adapters/gemini-local/src/ui/build-config.ts
@@ -1,4 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
+import { buildAdapterEnvConfig, type CreateConfigValues } from "@paperclipai/adapter-utils";
import { DEFAULT_GEMINI_LOCAL_MODEL } from "../index.js";
function parseCommaArgs(value: string): string[] {
@@ -8,49 +8,6 @@ function parseCommaArgs(value: string): string[] {
.filter(Boolean);
}
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
-
export function buildGeminiLocalConfig(v: CreateConfigValues): Record {
const ac: Record = {};
if (v.cwd) ac.cwd = v.cwd;
@@ -66,13 +23,7 @@ export function buildGeminiLocalConfig(v: CreateConfigValues): Record 0) ac.env = env;
ac.sandbox = !v.dangerouslyBypassSandbox;
diff --git a/packages/adapters/grok-local/src/ui/build-config.ts b/packages/adapters/grok-local/src/ui/build-config.ts
index 6c9e9a66d5a2..37c9a4886349 100644
--- a/packages/adapters/grok-local/src/ui/build-config.ts
+++ b/packages/adapters/grok-local/src/ui/build-config.ts
@@ -1,4 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
+import { buildAdapterEnvConfig, type CreateConfigValues } from "@paperclipai/adapter-utils";
import { DEFAULT_GROK_LOCAL_MODEL } from "../index.js";
function parseCommaArgs(value: string): string[] {
@@ -8,49 +8,6 @@ function parseCommaArgs(value: string): string[] {
.filter(Boolean);
}
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
-
export function buildGrokLocalConfig(v: CreateConfigValues): Record {
const ac: Record = {};
if (v.cwd) ac.cwd = v.cwd;
@@ -59,13 +16,7 @@ export function buildGrokLocalConfig(v: CreateConfigValues): Record 0) ac.env = env;
if (v.command) ac.command = v.command;
diff --git a/packages/adapters/openclaw-gateway/src/server/execute.test.ts b/packages/adapters/openclaw-gateway/src/server/execute.test.ts
index 316b81c7d5cc..3689e412458b 100644
--- a/packages/adapters/openclaw-gateway/src/server/execute.test.ts
+++ b/packages/adapters/openclaw-gateway/src/server/execute.test.ts
@@ -1,5 +1,5 @@
import { describe, expect, it } from "vitest";
-import { buildAgentParams, resolveSessionKey } from "./execute.js";
+import { buildAgentParams, resolveClaimedApiKeyPath, resolveSessionKey } from "./execute.js";
describe("resolveSessionKey", () => {
it("prefixes run-scoped session keys with the configured agent", () => {
@@ -98,3 +98,28 @@ describe("buildAgentParams", () => {
});
});
});
+
+describe("resolveClaimedApiKeyPath", () => {
+ const DEFAULT_PATH = "~/.openclaw/workspace/paperclip-claimed-api-key.json";
+
+ it("returns the configured per-agent path when set", () => {
+ expect(
+ resolveClaimedApiKeyPath("~/.openclaw/workspace/paperclip-keys/happy.json"),
+ ).toBe("~/.openclaw/workspace/paperclip-keys/happy.json");
+ });
+
+ it("falls back to the shared default when value is empty", () => {
+ expect(resolveClaimedApiKeyPath("")).toBe(DEFAULT_PATH);
+ expect(resolveClaimedApiKeyPath(" ")).toBe(DEFAULT_PATH);
+ });
+
+ it("falls back to the shared default when value is missing", () => {
+ expect(resolveClaimedApiKeyPath(undefined)).toBe(DEFAULT_PATH);
+ expect(resolveClaimedApiKeyPath(null)).toBe(DEFAULT_PATH);
+ });
+
+ it("falls back to the shared default when value is not a string", () => {
+ expect(resolveClaimedApiKeyPath(42)).toBe(DEFAULT_PATH);
+ expect(resolveClaimedApiKeyPath({})).toBe(DEFAULT_PATH);
+ });
+});
diff --git a/packages/adapters/openclaw-gateway/src/server/execute.ts b/packages/adapters/openclaw-gateway/src/server/execute.ts
index ab84aa7de02e..7771254de4d1 100644
--- a/packages/adapters/openclaw-gateway/src/server/execute.ts
+++ b/packages/adapters/openclaw-gateway/src/server/execute.ts
@@ -337,7 +337,7 @@ function resolvePaperclipApiUrlOverride(value: unknown): string | null {
const DEFAULT_CLAIMED_API_KEY_PATH = "~/.openclaw/workspace/paperclip-claimed-api-key.json";
-function resolveClaimedApiKeyPath(value: unknown): string {
+export function resolveClaimedApiKeyPath(value: unknown): string {
return nonEmpty(value) ?? DEFAULT_CLAIMED_API_KEY_PATH;
}
@@ -369,8 +369,8 @@ function buildWakeText(
payload: WakePayload,
paperclipEnv: Record,
structuredWakePrompt: string,
+ claimedApiKeyPath: string,
): string {
- const claimedApiKeyPath = "~/.openclaw/workspace/paperclip-claimed-api-key.json";
const orderedKeys = [
"PAPERCLIP_RUN_ID",
"PAPERCLIP_AGENT_ID",
@@ -1096,6 +1096,7 @@ export async function execute(ctx: AdapterExecutionContext): Promise {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record = {};
- for (const [key, raw] of Object.entries(bindings)) {
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- if (typeof raw === "string") {
- env[key] = { type: "plain", value: raw };
- continue;
- }
- if (typeof raw !== "object" || raw === null || Array.isArray(raw)) continue;
- const rec = raw as Record;
- if (rec.type === "plain" && typeof rec.value === "string") {
- env[key] = { type: "plain", value: rec.value };
- continue;
- }
- if (rec.type === "secret_ref" && typeof rec.secretId === "string") {
- env[key] = {
- type: "secret_ref",
- secretId: rec.secretId,
- ...(typeof rec.version === "number" || rec.version === "latest"
- ? { version: rec.version }
- : {}),
- };
- }
- }
- return env;
-}
-
export function buildOpenCodeLocalConfig(v: CreateConfigValues): Record {
const ac: Record = {};
if (v.cwd) ac.cwd = v.cwd;
@@ -61,13 +18,7 @@ export function buildOpenCodeLocalConfig(v: CreateConfigValues): Record 0) ac.env = env;
if (v.command) ac.command = v.command;
if (v.extraArgs) ac.extraArgs = parseCommaArgs(v.extraArgs);
diff --git a/packages/adapters/pi-local/src/ui/build-config.ts b/packages/adapters/pi-local/src/ui/build-config.ts
index 470ecedaf0f6..06927f6f8d85 100644
--- a/packages/adapters/pi-local/src/ui/build-config.ts
+++ b/packages/adapters/pi-local/src/ui/build-config.ts
@@ -1,47 +1,4 @@
-import type { CreateConfigValues } from "@paperclipai/adapter-utils";
-
-function parseEnvVars(text: string): Record {
- const env: Record = {};
- for (const line of text.split(/\r?\n/)) {
- const trimmed = line.trim();
- if (!trimmed || trimmed.startsWith("#")) continue;
- const eq = trimmed.indexOf("=");
- if (eq <= 0) continue;
- const key = trimmed.slice(0, eq).trim();
- const value = trimmed.slice(eq + 1);
- if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(key)) continue;
- env[key] = value;
- }
- return env;
-}
-
-function parseEnvBindings(bindings: unknown): Record {
- if (typeof bindings !== "object" || bindings === null || Array.isArray(bindings)) return {};
- const env: Record