From 3c95610f944714b6b6a29faeed5a6d34c682c726 Mon Sep 17 00:00:00 2001 From: Sebastian Otaegui Date: Mon, 13 Jul 2026 10:22:13 -0300 Subject: [PATCH 1/4] docs(agent-journal): freeze the V3 pre-task contract --- ...rk-journal-v3-infrastructure-manifest.json | 144 +++++++++++++++++ docs/evaluations/agent-work-journal-v3.md | 92 +++++++++++ ...t-agent-work-journal-v3-evaluation-plan.md | 148 ++++++++++++++++++ 3 files changed, 384 insertions(+) create mode 100644 docs/evaluations/agent-work-journal-v3-infrastructure-manifest.json create mode 100644 docs/evaluations/agent-work-journal-v3.md create mode 100644 docs/plans/2026-07-13-001-feat-agent-work-journal-v3-evaluation-plan.md diff --git a/docs/evaluations/agent-work-journal-v3-infrastructure-manifest.json b/docs/evaluations/agent-work-journal-v3-infrastructure-manifest.json new file mode 100644 index 0000000..0b7c3c8 --- /dev/null +++ b/docs/evaluations/agent-work-journal-v3-infrastructure-manifest.json @@ -0,0 +1,144 @@ +{ + "schemaVersion": 3, + "status": "pre-task-infrastructure", + "concreteTasksSelected": false, + "taskIds": [], + "promptDigests": [], + "categories": [ + "automated-multi-file-continuation", + "material-dependency-revalidation", + "append-only-conflict-resolution" + ], + "model": "openai-codex/gpt-5.6-sol", + "reasoning": "high", + "runtime": { + "nodeVersion": "24.14.1", + "piCodingAgentVersion": "0.80.6", + "repositoryLockedExecutableRequired": true, + "darwinSandboxExecRequiredForUntrustedCommands": true + }, + "readTolerance": 1, + "exactRunsPerCondition": 3, + "exactTotalTraces": 18, + "budgetPolicy": { + "taskAgentMaximumAssistantTurnsPerPhase": 12, + "taskAgentMaximumToolCallsPerPhase": 40, + "phaseWallTimeoutSeconds": 1800, + "ownerSimulatorMaximumAssistantTurns": 2, + "ownerSimulatorMaximumToolCalls": 8, + "ownerSimulatorWallTimeoutSeconds": 600, + "statusByteLimit": 4000, + "capsuleByteLimit": 4000, + "tokensReportedPostHoc": true, + "budgetExhaustionResult": "terminal_fail" + }, + "attemptPolicy": { + "scheduleAndRunIdsPreRegistered": true, + "firstProviderRequestStartsAttempt": true, + "postRequestFailureResult": "terminal_fail_no_retry", + "preRequestInfrastructureRetries": 1, + "extraMissingDuplicateOrReplacementTraceResult": "terminal_fail" + }, + "ownerProtocol": { + "fields": [ + "objective", + "current_status", + "settled_decisions", + "evidence", + "open_questions", + "next_action", + "material_dependencies" + ], + "phaseAActions": ["status_create", "status_refresh", "status_correction"], + "phaseBActions": ["known_context_clarification", "resume_restatement"], + "eachActionRecorded": true, + "statusCreateAlwaysCountsAsAvoidableMaintenance": true + }, + "materialCaseMatrix": { + "journalSafetyAssertionsOnly": true, + "baselineMustDetectAndRevalidateForTaskCorrectness": true, + "dependencyTaskPerRepetition": ["one_material_stale_or_missing", "one_unaffected_control"], + "conflictTaskPerRepetition": ["one_material_conflict", "one_unaffected_control"], + "falsePositiveUniverse": "all_declared_unaffected_controls" + }, + "taskLifecycle": { + "antiTuningEmbargoStartsBeforeCandidateGeneration": true, + "maximumCandidateAttemptsPerCategory": 3, + "firstStructurallyValidCandidateMustFreeze": true, + "candidateRejectableBeforeAnyModelTrial": true, + "allowedCandidateRejections": [ + "missing_bytes_or_digest", + "invalid_fixture", + "nondeterministic_mutation", + "broken_grader", + "category_mismatch", + "prior_task_reuse", + "unsupported_runner_requirement" + ], + "safeRejectionReceiptRequired": true, + "atomicThreeTaskFreezeRequired": true, + "frozenTasksOneShot": true + }, + "quarantine": { + "twoIndependentImplementationsRequired": true, + "duplicateReconstructionRequired": true, + "duplicateMutationRequired": true, + "expectedHashesRequired": true, + "mutationPreconditionFailureRequired": true, + "positiveAndNegativeGraderChecksRequired": true, + "independentReceiptReviewRequired": true, + "proseOnlyMutationForbidden": true + }, + "sandbox": { + "privateRootMode": "0700", + "sanitizedEnvironmentAllowlistRequired": true, + "ambientCredentialsForbidden": true, + "mutatorAndGraderNetworkDisabled": true, + "readOnlyInputsAndSingleWritableTrialRoot": true, + "commandsSpawnedWithoutShellConcatenation": true, + "resourceLimitsAndProcessTreeTerminationRequired": true + }, + "productBoundary": { + "toolNames": ["journal_record", "journal_inspect", "journal_checkpoint", "journal_session"], + "materialFileObservationWithinJournalRecord": true, + "callerSuppliedFileProvenanceMustBeRecomputed": true, + "predecessorCutoverIncluded": false + }, + "provenance": { + "canonicalSerialization": "RFC8785-JCS", + "digest": "SHA-256", + "separateInfrastructureAcceptanceReceiptRequired": true, + "separateFrozenTaskSetReceiptRequired": true, + "bindRepositoryCommitAndTree": true, + "bindContractManifestRunnerScorerNormalizerValidatorOwnerProtocolGate": true, + "bindScheduleAttemptsRawDerivedRecomputationCleanup": true + }, + "historicalEvidenceGuard": { + "docs/evaluations/agent-work-journal-v1.md": "cb4e88d65ecba49bd12669c03c8a865641b1538516dca1e056258ebbca0bdbe1", + "docs/evaluations/agent-work-journal-v1-results.json": "ff6f8b1389e0667ca410fd6d544247901ebc2955aaa1a6fa1a47404d00e0315b", + "docs/plans/2026-07-12-001-feat-agent-work-journal-plan.md": "d74dee96690d5f125ee53001b182100209587c4e207173272c86d604e5206748", + "docs/evaluations/agent-work-journal-v2.md": "939f479276af7038b2d1512d71d0bdd822b7ed7b56f95cd4632cb089603c63f2", + "docs/evaluations/agent-work-journal-v2-results.json": "1f0aafc726bab5c6482a56cc187d588c51d7081c3e2dfef7af3ad42cc23e0f7c", + "docs/evaluations/agent-work-journal-v2-infrastructure-manifest.json": "bad596b7909c3b85aa6cafec7dd893516384eb1f157e44fe33e709c22a1e4e34", + "docs/plans/2026-07-12-002-feat-agent-work-journal-v2-redesign-plan.md": "c67417dca8747a1310f6419ef2f51b4a6173985d1390c8415fe5c25d07fda141" + }, + "priorTaskNovelty": { + "v1SemanticDenylist": [ + "partial_multi_file_investigation", + "material_dependency_change", + "settled_competing_alternative" + ], + "v2SafeTaskDigestsFromHistoricalResultRequired": true, + "independentSemanticNoveltyReviewRequired": true + }, + "privacy": { + "rawPromptsCommitted": false, + "fixtureSourceCommitted": false, + "graderSourceCommitted": false, + "mutationSourceCommitted": false, + "rawToolPayloadsCommitted": false, + "rawModelMessagesCommitted": false, + "absolutePrivatePathsCommitted": false, + "privateEvidenceDeletedAfterFinalGateAndRecomputation": true + } +} diff --git a/docs/evaluations/agent-work-journal-v3.md b/docs/evaluations/agent-work-journal-v3.md new file mode 100644 index 0000000..aabf1d8 --- /dev/null +++ b/docs/evaluations/agent-work-journal-v3.md @@ -0,0 +1,92 @@ +# Agent Work Journal V3 Evaluation Contract + +Status: **pre-task infrastructure implementation; held-out selection forbidden**. + +## Product claim + +V3 evaluates automatically maintained trustworthy status with correctness parity, bounded exploration cost, less avoidable owner maintenance, and perfect handling of planted material stale/conflict cases. + +## Immutable history + +V1 and V2 plans, contracts, manifests, and results are historical evidence and must not change. Their tasks, prompts, rubrics, fixtures, mutations, and traces cannot be reused. Both predecessor packages remain active. + +## Task lifecycle + +- **Candidate:** private, unexposed to any model trial, and rejectable only for enumerated structural defects. +- **Frozen:** independently quarantine-validated, atomically included in the three-task set, immutable, exposed, one-shot, and non-replaceable. + +The anti-tuning embargo starts before any candidate is generated or viewed. Generate at most three candidates per category in a precommitted order; the first structurally valid candidate must freeze. Allowed candidate rejection reasons are missing bytes/digests, invalid fixture, non-deterministic mutation, broken positive/negative grader, category mismatch, prior-task reuse, or unsupported runner requirement. Every rejection has an ordered safe receipt. Model outcomes are never an allowed rejection reason. + +## Quarantine contract + +Each candidate contains exact bytes or content-addressed references for both prompts, rubric, fixture, hidden graders, executable deterministic mutation, mutation inputs, expected pre/post hashes, phase boundaries, unsafe-continuation rule, material cases, and runner budgets. Prose-only mutation instructions are invalid. + +Before freeze, two independently implemented validators reconstruct, mutate, run golden passing/failing implementations, and compare canonical receipts. They verify identical trees and expected hashes, prove mutation precondition failure, check every digest, and confirm category novelty against the V1 semantic denylist and V2 safe digests. All three tasks then freeze atomically under one task-set digest bound to the independently accepted infrastructure receipt. + +## Real trial contract + +- Categories: `automated-multi-file-continuation`, `material-dependency-revalidation`, `append-only-conflict-resolution`. +- Exactly three baseline and three journal traces per category: exactly 18 total; extra, missing, duplicate, or replaced traces fail. +- Detached worktree and distinct phase-A/phase-B processes per condition. +- Same task prompts, rubric, snapshot, model, reasoning, pause point, turn/tool/wall budgets across conditions. +- Baseline phase A generates the seven-field owner status through the same model; baseline B starts fresh with only that status. +- Journal B reopens the actual phase-A session/store, clears prior transcript from model context, and receives only the runtime-generated capsule as continuation context. +- Status and capsule are each at most 4,000 UTF-8 bytes. +- Native traces from both phases remain private until the final gate and independent recomputation attest the result. +- All 18 run IDs and interleaved schedule are preregistered before launch. A provider request starts an attempt. Any post-request crash, timeout, budget/provenance failure is terminal with no retry; one mechanically proven pre-provider infrastructure retry is allowed and recorded. + +## Intervention taxonomy + +Avoidable maintenance: +- `status_create` +- `status_refresh` +- `status_correction` +- `known_context_clarification` +- `resume_restatement` + +Necessary safety: +- `material_stale_resolution` +- `material_conflict_resolution` +- `binding_ambiguity_resolution` +- `credential_exclusion_resolution` + +Necessary safety never counts as avoidable maintenance. Unknown kinds fail closed. + +## Frozen owner protocol + +Baseline status contains exactly these seven sections: objective, current status, settled decisions, evidence, open questions, next action, and material dependencies. Phase A permits only `status_create`, `status_refresh`, and `status_correction`; phase B permits only `known_context_clarification` and `resume_restatement`. Every action is recorded, and `status_create` always counts as avoidable owner maintenance. The canonical protocol digest is bound into every parity receipt. + +## Frozen gate + +All clauses must pass: + +1. Journal median task score is at least baseline median in every scenario, with no journal material-correctness failure. +2. Journal median normalized repository reads are no more than baseline median plus 1 in every scenario. +3. Journal median avoidable maintenance is lower than baseline in at least 2 of 3 scenarios. +4. Journal strict-majority no-restatement outcome is no worse than baseline in every scenario. +5. Journal traces must handle every planted positive material case before unsafe continuation: affected support is withheld, durable history remains unchanged, resolution appends new evidence/state, and the required safety intervention is recorded. Baseline must detect and revalidate mutations for task correctness but has no journal-history assertion. Every dependency/conflict repetition includes one unaffected control; any safety notice/intervention on a declared control is a false positive and fails. +6. Every quarantine, parity, budget, trace, provenance, recomputation, and retention receipt validates. + +Failure records FAIL and stops. Pass records only cutover eligibility. Neither outcome performs cutover. + +## Enforceable budgets + +- Model: `openai-codex/gpt-5.6-sol`. +- Reasoning: `high`. +- Task agent per phase: at most 12 assistant turns, 40 tool calls, and 1,800 wall-clock seconds. +- Owner simulator: at most 2 assistant turns, 8 tool calls, and 600 wall-clock seconds. +- Exhaustion is terminal FAIL. Retries and parallel calls count toward the same observed budget. +- Baseline status and journal capsule: at most 4,000 UTF-8 bytes each. +- Token usage is reported post hoc, not used as an enforced gate unless termination can precede violation. + +## Safe evidence and retention + +Committed evidence contains only opaque IDs, categories, SHA-256 digests, normalized counts, typed outcomes, medians, safe provenance, cleanup receipts, and the final decision. It never contains prompts, fixture/grader/mutation source, raw model messages, reasoning, tool arguments/results, credentials, or absolute private paths. + +Private roots are canonical owner-controlled mode-0700 directories. Mutators and graders run without shell concatenation in a disposable sandbox with sanitized allowlisted environment, no ambient credentials/network, read-only inputs, one writable trial root, resource limits, and process-tree termination. The model/provider runner receives only the separately required provider connectivity; tool subprocesses remain sandboxed. + +Use RFC 8785 JCS and SHA-256 receipts binding repository commit/tree, contract, manifest, runtime, owner protocol, runner, normalizer, scorer, validators, gate, frozen schedule/task set, every attempt/raw/derived trace, final result, recomputation, and cleanup. Apply and independently attest the complete gate before deletion. Then verify cleanup and write the terminal result/cleanup receipt. Rejected candidates, crashes, timeouts, partial trials, and recomputation failures follow the same bounded recovery state machine; failed recomputation preserves encrypted/private evidence for manual adjudication rather than claiming cleanup. + +## Pre-task boundary + +[`agent-work-journal-v3-infrastructure-manifest.json`](./agent-work-journal-v3-infrastructure-manifest.json) is the immutable pre-task candidate. Until U1–U5 pass independent acceptance, it must retain empty task and prompt arrays and `concreteTasksSelected: false`. U5 writes a separate immutable infrastructure-acceptance receipt. U6 writes a separate immutable frozen-task-set receipt; neither mutates the pre-task manifest. diff --git a/docs/plans/2026-07-13-001-feat-agent-work-journal-v3-evaluation-plan.md b/docs/plans/2026-07-13-001-feat-agent-work-journal-v3-evaluation-plan.md new file mode 100644 index 0000000..7337002 --- /dev/null +++ b/docs/plans/2026-07-13-001-feat-agent-work-journal-v3-evaluation-plan.md @@ -0,0 +1,148 @@ +--- +title: Agent Work Journal V3 Evaluation +status: active +plan_type: implementation +artifact_contract: ce-unified-plan/v1 +artifact_readiness: implementation-ready +execution: code +created: 2026-07-13 +source: owner-approved V3 LFG contract +--- + +# Agent Work Journal V3 Evaluation + +## Goal Capsule + +Build a third-generation evaluation that can validly test Agent Journal as automatically maintained trustworthy status. First make material file dependencies usable through the existing `journal_record` tool, then prove a real two-process Pi runner and a byte-complete quarantine validator with synthetic tasks. Only after independent infrastructure acceptance may a fresh selector create candidates. Candidates become immutable one-shot held-out tasks only after complete quarantine validation. Run exactly 18 valid traces, independently recompute the frozen all-of gate, delete private evidence, and stop before predecessor cutover. + +## Product Contract + +- Product thesis: automatically maintained trustworthy status with correctness parity, bounded exploration cost, less owner maintenance, and perfect handling of planted material stale/conflict cases. +- Keep exactly four tools: `journal_record`, `journal_inspect`, `journal_checkpoint`, and `journal_session`. +- V1/V2 plans, contracts, manifests, and results are immutable historical evidence. +- `pi-code-reasoning` and `pi-sequential-thinking` remain active. +- No raw prompts, model messages, reasoning, tool payloads, credentials, fixture source, grader source, mutations, or absolute private paths may be committed. +- Every behavior change follows strict observable RED→GREEN TDD. +- No held-out task is exposed before independent U1–U5 acceptance. +- No product, task, rubric, grader, threshold, or runner tuning occurs after the frozen task set is exposed. +- A passing V3 gate authorizes only a separate owner cutover decision. + +## Architecture and Decisions + +### Candidate versus frozen + +The anti-tuning embargo begins before candidate generation. Precommit an order and generate at most three candidates per category; the first structurally valid candidate must freeze. A candidate is private and may be rejected only before any model trial for missing bytes/digests, invalid or non-deterministic fixtures/mutations, broken graders, category mismatch, prior-task reuse, or unsupported runner requirements, with an ordered safe rejection receipt. A candidate becomes frozen only after two independently implemented reconstructions, deterministic mutation/hash checks, positive/negative grader checks, digest verification, and receipt agreement. The three candidates freeze atomically under one task-set digest bound to the U5 infrastructure receipt. Frozen tasks are immutable and non-replaceable. + +### Material file observation + +Extend `journal_record` entries with optional `observe_files: [{ path, material }]`. The caller supplies no hash, timestamp, workspace ID, or originating ID. `JournalService` computes a safe file dependency using the entry ID. Existing explicit typed dependencies remain supported, but public `kind: file` inputs are always safely reopened and recomputed; caller provenance can never be treated as observed. Sensitive paths are denied before reading, bounded in-memory bytes are scanned before hashing, and rejection writes nothing or echoes no candidate path/content/hash. File contents never persist. + +### Enforceable budget + +Freeze equal model (`openai-codex/gpt-5.6-sol`), high reasoning, 12 assistant turns/40 tool calls/1,800 seconds per task phase, 2 turns/8 calls/600 seconds for the owner simulator, and 4,000-byte baseline-status/journal-capsule bounds. Exhaustion is terminal FAIL. Token usage is reported post hoc and is not a gate unless the runner can stop before violation. + +### Safe evidence + +Private bundles and raw traces live under an owner-controlled mode-0700 trial root. Mutators/graders run without shell concatenation in a no-network sandbox with sanitized environment, no ambient credentials, read-only inputs, one writable trial root, resource limits, and process-tree termination. Safe RFC-8785-JCS/SHA-256 receipts contain opaque IDs, digests, counts, typed outcomes, medians, and cleanup state only. Apply and independently attest the gate before private deletion; write the terminal cleanup receipt afterward. + +## Requirements + +- **R1:** Exactly four public tools remain registered. +- **R2:** `observe_files` computes dependencies inside `JournalService` and rejects escapes, symlinks, special/oversized files, and secret-bearing outputs. +- **R3:** Autonomous checkpoints preserve observed dependencies and resume withholds stale/missing material support. +- **R4:** The real runner uses detached worktrees and distinct phase processes with actual Pi session/store continuity. +- **R5:** Baseline receives same-model phase-A-generated seven-field status; journal receives only the runtime capsule after transcript clearing. +- **R6:** Scores, reads, maintenance, restatement, pause order, and material safety derive from observable evidence, never constants. +- **R7:** Bundle schema requires exact bytes/content digests for prompts, rubric, fixture, graders, executable mutation, expected hashes, boundaries, cases, and budgets. +- **R8:** Quarantine reconstructs/mutates twice and exercises graders against known pass/fail implementations. +- **R9:** Task set contains one novel task per frozen category and has no V1/V2 digest reuse. +- **R10:** Preregister an interleaved schedule and all run IDs, then run exactly three repetitions per condition per task: exactly 18 globally unique traces. A provider request starts an attempt; post-request failure is terminal with no retry, while one mechanically proven pre-provider retry is allowed. +- **R11:** Gate requires correctness, read parity (`journal <= baseline + 1`), maintenance improvement in 2/3 scenarios, no-restatement parity, perfect material safety, and evidence integrity. +- **R12:** Failure records honestly and stops; pass records cutover eligibility only. + +## Implementation Units + +### U1 — Freeze V3 contract and infrastructure boundary + +Files: +- `docs/plans/2026-07-13-001-feat-agent-work-journal-v3-evaluation-plan.md` +- `docs/evaluations/agent-work-journal-v3.md` +- `docs/evaluations/agent-work-journal-v3-infrastructure-manifest.json` + +Create candidate/frozen semantics, enforceable budgets, taxonomy, safe evidence, gate, and immutable pre-task manifest with empty task IDs/prompts. Independently review coherence, feasibility, security, evaluation integrity, and adversarial post-hoc risk. + +Verification: `specdocs_validate`; manifest regression tests; V1/V2 SHA-256 guard. + +### U2 — Add safe material file observation + +Files: +- `packages/pi-agent-journal/extensions/domain.ts` +- `packages/pi-agent-journal/extensions/journal-service.ts` +- `packages/pi-agent-journal/extensions/tools.ts` +- corresponding tests and README + +RED first for schema, computed metadata, path/symlink/special/size rejection, explicit-dependency compatibility, batch atomicity, secret boundaries, checkpoint propagation, stale withholding, and append-only resolution. Implement minimal GREEN without adding a tool or persisting file contents. + +Verification: focused tool/service/runtime tests, full package suite, typecheck, Biome. + +### U3 — Build real Pi runner and synthetic preflight + +Files: +- repository-only `packages/pi-agent-journal/evaluation/` sources +- evaluation harness/trace/scorer sources +- focused tests and package-boundary test + +RED first for ambient/global Pi usage, transcript leakage, favorable constant facts, missing/extra events, budget omission/exhaustion, process failure, raw loss, sandbox escape, and provenance mismatch. Then add the minimal real Pi process adapter, private sessions/stores, phase-specific prompts, baseline owner simulator, provider-bound transcript-cleared journal resume, complete native JSONL, objective graders, enforceable budget receipts, derived classifiers, material/store proofs, provenance, and failure-safe cleanup. Use unrelated synthetic tasks only. + +Verification: real-Pi synthetic smoke across both conditions; no hard-coded favorable outcomes; package dry-run excludes evaluation sources/raw files. + +### U4 — Implement byte-complete bundle quarantine + +RED first for every omitted byte/digest, prose mutation, non-determinism, precondition drift, grader false acceptance/rejection, semantic predecessor reuse, sandbox escape, and validator disagreement. Then create schema plus two independent validator implementations for exact bundle components and canonical digests. Reconstruct/mutate independently, verify hashes, exercise known pass/fail implementations, detect V1 semantic-category and V2 digest reuse, and compare safe receipts. + +Verification: adversarial validator matrix and deterministic integration test. + +### U5 — Independently accept infrastructure + +Run synthetic end-to-end preflight, all package verification, leak review, and five independent document/code lenses. The manifest remains pre-task with empty IDs/prompts. Write a separate immutable infrastructure-acceptance receipt binding the commit/tree and all executable/configuration digests. Commit, push, open the infrastructure PR, and require green CI before any selector process is launched. No held-out selection occurs until every blocker is resolved. + +### U6 — Select, quarantine, freeze, and evaluate + +Only after the green U5 infrastructure commit/CI checkpoint, use a fresh independent selector under the precommitted generation order. Validate candidates privately; rejected candidates leave no model trial and produce safe receipts. Write a separate frozen-task-set receipt, preregister the 18-run interleaved schedule/IDs, and run every launch fail-closed. Independently recompute and apply/attest the gate while raw evidence remains. Then verify private cleanup, write `docs/evaluations/agent-work-journal-v3-results.json` plus executable regression and cleanup receipt, and stop before cutover. + +## Verification Contract + +| Gate | Required evidence | +|---|---| +| Frozen-history guard | V1/V2 file SHA-256 values equal the branch-start receipt | +| Material observation | Focused RED/GREEN tests plus real stale/missing resume chain | +| Four-tool boundary | Portable and MCP list-tools tests report exactly four names | +| Real runner | Synthetic real-Pi baseline/journal two-phase receipt | +| Quarantine | Duplicate reconstruction/mutation hashes and positive/negative grader receipts | +| Privacy | Recursive scan of committed results and package tarball | +| Package tests | `npx vitest run packages/pi-agent-journal/__tests__` | +| Coverage | Repository thresholds pass | +| Type/lint | package TypeScript and Biome pass | +| Package | MCP build and npm dry-run exclude evaluation/private artifacts | +| Specs | `specdocs_validate` passes | +| Product | 18 valid traces and all frozen V3 clauses pass | + +## Risks + +- Model-authored observation may be underused; synthetic preflight must prove discoverability before selection, without task-specific tuning. +- Clearing transcript while preserving branch binding is subtle; prove actual capsule-only model context. +- Native Pi events may omit needed facts; fail closed instead of synthesizing favorable values. +- Candidate rejection can become cherry-picking; permit only enumerated structural failures before any model trial and retain safe rejection counts. +- Private evidence can leak through diagnostics, paths, package exports, or test fixtures; recursively scan all outputs. + +## Definition of Done + +- U1–U5 pass before task exposure. +- Every behavior change has recorded RED then GREEN evidence. +- Three candidates pass quarantine and freeze atomically. +- Exactly 18 valid traces are independently recomputed. +- V3 records an honest PASS or FAIL with no private leakage. +- Private evidence is deleted only after recomputation. +- V1/V2 evidence and predecessor packages are unchanged. +- Simplify, review, browser applicability, incremental commits, PR, and CI stages complete. +- No predecessor cutover occurs. From f33ed7cbfddbeee0251bab0b56f33c8af1ccf830 Mon Sep 17 00:00:00 2001 From: Sebastian Otaegui Date: Mon, 13 Jul 2026 10:29:54 -0300 Subject: [PATCH 2/4] feat(agent-journal): compute material file observations safely --- packages/pi-agent-journal/README.md | 4 +- .../__tests__/capture-policy.test.ts | 50 +++++++++ .../pi-agent-journal/__tests__/mcp.test.ts | 11 +- .../__tests__/tools.portable.test.ts | 106 +++++++++++++++++- .../pi-agent-journal/extensions/domain.ts | 7 ++ .../extensions/journal-service.ts | 54 +++++++-- packages/pi-agent-journal/extensions/tools.ts | 7 ++ 7 files changed, 226 insertions(+), 13 deletions(-) diff --git a/packages/pi-agent-journal/README.md b/packages/pi-agent-journal/README.md index 075a10a..954a152 100644 --- a/packages/pi-agent-journal/README.md +++ b/packages/pi-agent-journal/README.md @@ -20,7 +20,7 @@ Pi and MCP use separate default stores and cannot be configured to the same cano ## Tools -- `journal_record` — append bounded typed entries (`observation`, `evidence`, `assumption`, `decision`, `rejected_alternative`, `validation`, `next_action`) and optional `supersedes` / `alternative-to` links. +- `journal_record` — append bounded typed entries (`observation`, `evidence`, `assumption`, `decision`, `rejected_alternative`, `validation`, `next_action`), optional `supersedes` / `alternative-to` links, and optional `observe_files` declarations. For observed files, callers provide only a workspace-relative path and material flag; the service safely computes the dependency hash, workspace identity, timestamp, and entry binding. - `journal_inspect` — read a bounded current projection, append-only history, or durable notices. - `journal_checkpoint` — create a compact checkpoint or resume with referenced entries and freshness results. - `journal_session` — list, create, select, inspect, or close sessions. Close never deletes history. @@ -49,7 +49,7 @@ Pi storage can be overridden by `AGENT_JOURNAL_STORAGE_DIR` or `--agent-journal- ## Privacy and limits -Journal files are local plaintext JSON with private filesystem permissions where supported. Credential detection is best-effort; detected candidate bytes are excluded from journal-owned files, temp files, outputs, and notices. Pi's own session transcript is separate plaintext storage outside this guarantee. Do not pass suspected secrets in tool arguments. +Journal files are local plaintext JSON with private filesystem permissions where supported. Credential detection is best-effort; detected candidate bytes are excluded from journal-owned files, temp files, outputs, and notices. File observation denies sensitive path names, scans bounded bytes before hashing, rejects symlinks/special/oversized files, and never stores file contents. Pi's own session transcript is separate plaintext storage outside this guarantee. Do not pass suspected secrets in tool arguments. V1 supports one process/writer per store. Reads and outputs are bounded, artifact freshness reads stay inside the workspace and reject symlinks/special files, and unresolved conflicts remain available in headless modes. Existing Sequential Thinking data is never scanned, imported, migrated, or deleted. diff --git a/packages/pi-agent-journal/__tests__/capture-policy.test.ts b/packages/pi-agent-journal/__tests__/capture-policy.test.ts index 7bcad41..910707b 100644 --- a/packages/pi-agent-journal/__tests__/capture-policy.test.ts +++ b/packages/pi-agent-journal/__tests__/capture-policy.test.ts @@ -219,6 +219,56 @@ describe("capture policy and journal service", () => { await expect(service.observeFileDependency("../outside", "evidence", true)).rejects.toThrow(/workspace/i); }); + it("propagates computed observations through checkpoints and resolves stale history append-only", async () => { + writeFileSync(join(workspace, "observed.txt"), "v1\n"); + const original = await service.record("work", { + id: "observed-entry", + type: "evidence", + content: "Observed contract", + observeFiles: [{ path: "observed.txt", material: true }], + }); + expect(original.dependencies).toEqual([ + expect.objectContaining({ + kind: "file", + path: "observed.txt", + material: true, + originatingEntryId: "observed-entry", + }), + ]); + await service.createCheckpoint("work", { + objective: "observe", + status: "active", + evidenceEntryIds: [original.id], + artifactDependencies: original.dependencies, + supportEntryIds: [original.id], + }); + writeFileSync(join(workspace, "observed.txt"), "v2\n"); + const stale = await service.resume("work"); + expect(stale.entries).toEqual([]); + expect(stale.notices).toEqual([expect.objectContaining({ category: "stale", requiresJudgment: true })]); + const before = await service.inspectHistory("work"); + const noticeBefore = (await service.inspectNotices("work"))[0]; + + const fresh = await service.record("work", { + id: "fresh-entry", + type: "evidence", + content: "Revalidated contract", + relationships: [{ type: "supersedes", targetEntryId: original.id }], + observeFiles: [{ path: "observed.txt", material: true }], + }); + await service.createCheckpoint("work", { + objective: "observe", + status: "revalidated", + evidenceEntryIds: [fresh.id], + artifactDependencies: fresh.dependencies, + supportEntryIds: [fresh.id], + }); + await service.resume("work"); + const after = await service.inspectHistory("work"); + expect(after.slice(0, before.length)).toEqual(before); + expect((await service.inspectNotices("work"))[0]).toMatchObject({ ...noticeBefore, requiresJudgment: false }); + }); + it("excludes stale material supporting entries from resumable current state", async () => { writeFileSync(join(workspace, "state.txt"), "one", "utf8"); const dependency = await service.observeFileDependency("state.txt", "file-evidence", true); diff --git a/packages/pi-agent-journal/__tests__/mcp.test.ts b/packages/pi-agent-journal/__tests__/mcp.test.ts index 3157350..1e5d76a 100644 --- a/packages/pi-agent-journal/__tests__/mcp.test.ts +++ b/packages/pi-agent-journal/__tests__/mcp.test.ts @@ -41,7 +41,16 @@ function normalize(value: unknown): unknown { Object.entries(value as Record) .filter( ([key]) => - !["id", "createdAt", "updatedAt", "closedAt", "activeCheckpointId", "fingerprint", "timestamp"].includes(key), + ![ + "id", + "createdAt", + "updatedAt", + "closedAt", + "activeCheckpointId", + "fingerprint", + "timestamp", + "observedAt", + ].includes(key), ) .map(([key, item]) => [key, normalize(item)]), ); diff --git a/packages/pi-agent-journal/__tests__/tools.portable.test.ts b/packages/pi-agent-journal/__tests__/tools.portable.test.ts index e13fc1a..89a164d 100644 --- a/packages/pi-agent-journal/__tests__/tools.portable.test.ts +++ b/packages/pi-agent-journal/__tests__/tools.portable.test.ts @@ -1,4 +1,5 @@ -import { mkdtempSync, rmSync } from "node:fs"; +import { createHash } from "node:crypto"; +import { mkdirSync, mkdtempSync, realpathSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { executePortableTool, type PortableTool } from "@feniix/bridgekit"; @@ -11,14 +12,17 @@ import { createJournalTools } from "../extensions/tools.js"; let root: string; let storage: JournalStorage; let service: JournalService; +let workspace: string; beforeEach(async () => { root = mkdtempSync(join(tmpdir(), "agent-journal-tools-")); - storage = new JournalStorage(root); + workspace = join(root, "workspace"); + mkdirSync(workspace); + storage = new JournalStorage(join(root, "store")); await storage.createSession("work"); service = new JournalService({ storage, - workspaceRoot: process.cwd(), + workspaceRoot: workspace, idGenerator: (() => { let i = 0; return () => `id-${++i}`; @@ -59,6 +63,102 @@ describe("Agent Journal portable tools", () => { } }); + it("computes material file observations and recomputes forged public file provenance", async () => { + writeFileSync(join(workspace, "state.txt"), "current\n"); + const record = tool("journal_record"); + const observed = await executePortableTool( + record, + { + entries: [ + { + id: "observed", + type: "evidence", + content: "Contract evidence", + observe_files: [{ path: "state.txt", material: true }], + dependencies: [ + { + kind: "file", + path: "state.txt", + workspaceId: "forged", + observedHash: "forged", + observedAt: "forged", + originatingEntryId: "observed", + material: false, + }, + { + kind: "repository_state", + value: "head-1", + observedAt: "2026-07-13T00:00:00.000Z", + originatingEntryId: "observed", + material: false, + }, + ], + }, + ], + }, + { host: "test" }, + ); + expect(observed.isError).toBeFalsy(); + const [entry] = (await storage.getSession("work")).entries; + expect(entry.dependencies).toEqual([ + { + kind: "repository_state", + value: "head-1", + observedAt: "2026-07-13T00:00:00.000Z", + originatingEntryId: "observed", + material: false, + }, + { + kind: "file", + path: "state.txt", + workspaceId: createHash("sha256").update(realpathSync(workspace)).digest("hex"), + observedHash: createHash("sha256").update("current\n").digest("hex"), + observedAt: expect.any(String), + originatingEntryId: "observed", + material: true, + }, + ]); + expect(JSON.stringify(entry)).not.toContain("forged"); + }); + + it("rejects unsafe observed files atomically without leaking candidate data", async () => { + writeFileSync(join(workspace, "good.txt"), "safe\n"); + writeFileSync(join(workspace, ".env"), "PUBLIC=true\n"); + const token = `ghp_${"s".repeat(30)}`; + writeFileSync(join(workspace, "payload.txt"), token); + writeFileSync(join(workspace, "large.bin"), Buffer.alloc(2 * 1024 * 1024 + 1)); + mkdirSync(join(workspace, "directory")); + symlinkSync("good.txt", join(workspace, "link.txt")); + const record = tool("journal_record"); + for (const [index, path] of ["../outside", ".env", "payload.txt", "large.bin", "directory", "link.txt"].entries()) { + const result = await executePortableTool( + record, + { + entries: [ + { + id: `good-${index}`, + type: "evidence", + content: "first", + observe_files: [{ path: "good.txt", material: true }], + }, + { id: `bad-${index}`, type: "evidence", content: "second", observe_files: [{ path, material: true }] }, + ], + }, + { host: "test" }, + ); + expect(result.isError, path).toBe(true); + expect(JSON.stringify(result)).not.toContain(token); + expect((await storage.getSession("work")).entries, path).toEqual([]); + } + }); + + it("guides Pi to observe only materially supporting files", () => { + const record = tool("journal_record"); + expect(record.hostExtras?.pi?.promptGuidelines).toEqual( + expect.arrayContaining([expect.stringMatching(/observe.*file.*material/i)]), + ); + }); + it("records entries, inspects bounded history, and resumes a checkpoint", async () => { const factory = tools(); const find = (name: string) => factory.find((item) => item.name === name) as PortableTool; diff --git a/packages/pi-agent-journal/extensions/domain.ts b/packages/pi-agent-journal/extensions/domain.ts index d29701c..0df4ace 100644 --- a/packages/pi-agent-journal/extensions/domain.ts +++ b/packages/pi-agent-journal/extensions/domain.ts @@ -112,12 +112,19 @@ export interface JournalSession { fingerprint: string; } +export interface FileObservationInput { + path: string; + material: boolean; +} + export interface EntryInput { id?: string; type: EntryType; content: string; relationships?: Relationship[]; dependencies?: FreshnessDependency[]; + /** Safe service-computed file dependencies; never persisted as a separate field. */ + observeFiles?: FileObservationInput[]; } export const EVALUATION_SCENARIO_CATEGORIES = [ diff --git a/packages/pi-agent-journal/extensions/journal-service.ts b/packages/pi-agent-journal/extensions/journal-service.ts index 8e2660a..04ce4ae 100644 --- a/packages/pi-agent-journal/extensions/journal-service.ts +++ b/packages/pi-agent-journal/extensions/journal-service.ts @@ -82,13 +82,40 @@ export class JournalService { async recordBatch(sessionId: string, inputs: EntryInput[]): Promise { if (inputs.length === 0) throw new JournalValidationError("journal record batch must not be empty"); for (const input of inputs) await this.rejectSecretCandidate(sessionId, input); - const entries = inputs.map((input) => - normalizeEntryInput(input, { - id: input.id ?? this.idGenerator(), - timestamp: this.clock(), - maxEntryBytes: this.maxEntryBytes, - }), - ); + const entries: JournalEntry[] = []; + for (const input of inputs) { + const id = input.id ?? this.idGenerator(); + const explicit = input.dependencies ?? []; + const base = normalizeEntryInput( + { + ...input, + id, + dependencies: explicit.filter((dependency) => dependency.kind !== "file"), + observeFiles: undefined, + }, + { id, timestamp: this.clock(), maxEntryBytes: this.maxEntryBytes }, + ); + const observations = [ + ...(input.observeFiles ?? []), + ...explicit.filter((dependency): dependency is FileDependency => dependency.kind === "file"), + ]; + if (observations.length > 20) throw new JournalValidationError("entry file observations exceed item limit"); + const byPath = new Map(); + for (const observation of observations) { + if ( + typeof observation.path !== "string" || + !observation.path.trim() || + typeof observation.material !== "boolean" + ) { + throw new JournalValidationError("file observation requires path and material boolean"); + } + const path = observation.path.trim(); + byPath.set(path, (byPath.get(path) ?? false) || observation.material); + } + const computed: FileDependency[] = []; + for (const [path, material] of byPath) computed.push(await this.observeFileDependency(path, id, material)); + entries.push({ ...base, dependencies: [...base.dependencies, ...computed] }); + } for (const entry of entries) { await this.rejectSecretCandidate(sessionId, entry); if (entry.dependencies.some((dependency) => dependency.originatingEntryId !== entry.id)) { @@ -295,6 +322,9 @@ export class JournalService { async observeFileDependency(path: string, originatingEntryId: string, material: boolean): Promise { const { path: safePath, bytes } = await this.readSafeFile(path); + if (containsLikelySecretValue(bytes.toString("utf8"))) { + throw new JournalValidationError("artifact content rejected sensitive data"); + } return { kind: "file", path: relative(this.workspaceRoot, safePath), @@ -482,6 +512,16 @@ export class JournalService { } private resolveSafeFile(path: string): string { + const sensitiveSegment = path + .replaceAll("\\", "/") + .split("/") + .some( + (segment) => + /^\.env(?:\.|$)/i.test(segment) || + /credential|secret|private[-_.]?key/i.test(segment) || + /^id_(?:rsa|ed25519)$/i.test(segment), + ); + if (sensitiveSegment) throw new JournalValidationError("artifact path rejected by sensitive-path policy"); const absolute = resolve(this.workspaceRoot, path); if (absolute !== this.workspaceRoot && !absolute.startsWith(`${this.workspaceRoot}${sep}`)) { throw new JournalValidationError("artifact path escapes workspace"); diff --git a/packages/pi-agent-journal/extensions/tools.ts b/packages/pi-agent-journal/extensions/tools.ts index 2958420..e0b0d72 100644 --- a/packages/pi-agent-journal/extensions/tools.ts +++ b/packages/pi-agent-journal/extensions/tools.ts @@ -38,6 +38,10 @@ const relationship = Type.Object({ type: Type.Union([Type.Literal("supersedes"), Type.Literal("alternative-to")]), targetEntryId: entryId, }); +const fileObservation = Type.Object({ + path: Type.String({ minLength: 1, maxLength: 512 }), + material: Type.Boolean(), +}); const dependency = Type.Union([ Type.Object({ kind: Type.Literal("file"), @@ -78,6 +82,7 @@ const entry = Type.Object({ content: Type.String({ minLength: 1, maxLength: 20000 }), relationships: Type.Optional(Type.Array(relationship, { maxItems: 20 })), dependencies: Type.Optional(Type.Array(dependency, { maxItems: 20 })), + observe_files: Type.Optional(Type.Array(fileObservation, { maxItems: 20 })), }); export const recordParams = Type.Object({ @@ -369,6 +374,7 @@ export function createJournalTools(deps: JournalToolDeps): PortableTool promptSnippet: "Record only durable decisions, evidence, assumptions, validations, and next actions.", promptGuidelines: [ "Use journal_record for semantic durable state; omit exploratory narration and raw tool output.", + "Observe a file only when it materially supports a durable entry; provide only its workspace-relative path and material flag.", "Use supersedes or alternative-to to relate append-only entries.", ], }, @@ -384,6 +390,7 @@ export function createJournalTools(deps: JournalToolDeps): PortableTool content: value.content as string, relationships: value.relationships as Relationship[] | undefined, dependencies: value.dependencies as FreshnessDependency[] | undefined, + observeFiles: value.observe_files as EntryInput["observeFiles"], })); const entries = await deps.service.recordBatch(id, inputs); return { sessionId: id, persisted: entries.length, entryIds: entries.map((item) => item.id) }; From 2a46e075939dca4f9264addd2aa95e880893d2c8 Mon Sep 17 00:00:00 2001 From: Sebastian Otaegui Date: Mon, 13 Jul 2026 11:14:47 -0300 Subject: [PATCH 3/4] test(agent-journal): record V3 infrastructure stop --- .../agent-work-journal-v3-results.json | 45 ++++++++++++++ docs/evaluations/agent-work-journal-v3.md | 8 ++- .../__tests__/evaluation-v3-result.test.ts | 61 +++++++++++++++++++ .../__tests__/tools.portable.test.ts | 43 +++++++++++++ .../extensions/journal-service.ts | 57 +++++++++++++---- 5 files changed, 202 insertions(+), 12 deletions(-) create mode 100644 docs/evaluations/agent-work-journal-v3-results.json create mode 100644 packages/pi-agent-journal/__tests__/evaluation-v3-result.test.ts diff --git a/docs/evaluations/agent-work-journal-v3-results.json b/docs/evaluations/agent-work-journal-v3-results.json new file mode 100644 index 0000000..438b4a2 --- /dev/null +++ b/docs/evaluations/agent-work-journal-v3-results.json @@ -0,0 +1,45 @@ +{ + "schemaVersion": 3, + "status": "FAIL", + "failureStage": "infrastructure-acceptance", + "releaseAuthorized": false, + "productOutcome": "not-adjudicated", + "concreteTasksSelected": false, + "infrastructureAccepted": false, + "gateEvaluated": false, + "validTraceCount": 0, + "repositorySnapshot": "f33ed7cbfddbeee0251bab0b56f33c8af1ccf830", + "preTaskManifest": { + "digest": "5d328c49382539ff117f1de8dcf6ed99139a1a5f200394be65c2b3de91649ac7", + "taskIds": [], + "promptDigests": [] + }, + "completedUnits": [ + "v3_contract", + "material_file_observation" + ], + "blockers": [ + "real_two_condition_two_phase_capsule_only_provenance_not_proven", + "byte_complete_independent_quarantine_not_accepted", + "sandbox_and_attempt_provenance_not_accepted", + "raw_to_derived_scorer_chain_not_accepted" + ], + "selectedTaskIds": [], + "runIds": [], + "scenarioMedians": null, + "gateClauses": { + "taskCorrectness": "NOT_EVALUATED", + "repositoryReadParity": "NOT_EVALUATED", + "ownerMaintenance": "NOT_EVALUATED", + "noRestatementParity": "NOT_EVALUATED", + "materialSafety": "NOT_EVALUATED", + "evidenceIntegrity": "FAIL" + }, + "privateEvidenceCleanup": { + "syntheticWorktreesDeleted": true, + "syntheticSessionsAndTracesDeleted": true, + "heldOutEvidenceCreated": false + }, + "predecessorCutoverPerformed": false, + "terminalDecision": "stop-before-held-out-selection" +} diff --git a/docs/evaluations/agent-work-journal-v3.md b/docs/evaluations/agent-work-journal-v3.md index aabf1d8..a28be3c 100644 --- a/docs/evaluations/agent-work-journal-v3.md +++ b/docs/evaluations/agent-work-journal-v3.md @@ -1,6 +1,6 @@ # Agent Work Journal V3 Evaluation Contract -Status: **pre-task infrastructure implementation; held-out selection forbidden**. +Status: **terminal infrastructure FAIL; held-out selection never occurred and product outcome is not adjudicated**. ## Product claim @@ -90,3 +90,9 @@ Use RFC 8785 JCS and SHA-256 receipts binding repository commit/tree, contract, ## Pre-task boundary [`agent-work-journal-v3-infrastructure-manifest.json`](./agent-work-journal-v3-infrastructure-manifest.json) is the immutable pre-task candidate. Until U1–U5 pass independent acceptance, it must retain empty task and prompt arrays and `concreteTasksSelected: false`. U5 writes a separate immutable infrastructure-acceptance receipt. U6 writes a separate immutable frozen-task-set receipt; neither mutates the pre-task manifest. + +## Terminal infrastructure outcome + +U1 froze the V3 contract and U2 added safe service-computed material file observations without adding a fifth tool. U3/U4 synthetic infrastructure passed unit tests but failed independent acceptance: the real four-process smoke could not prove provider-bound capsule-only continuation, and independent reviews rejected the quarantine independence, sandbox/attempt provenance, and raw-to-derived scorer chain. All candidate U3/U4 code was discarded rather than weakening the contract. + +No selector was launched, no held-out task or prompt was created, and no product trial ran. V3 therefore fails closed at infrastructure acceptance with product performance not adjudicated. Synthetic worktrees, sessions, stores, and traces were deleted. Safe terminal evidence is recorded in [`agent-work-journal-v3-results.json`](./agent-work-journal-v3-results.json). Both predecessors remain active and cutover remains unauthorized. diff --git a/packages/pi-agent-journal/__tests__/evaluation-v3-result.test.ts b/packages/pi-agent-journal/__tests__/evaluation-v3-result.test.ts new file mode 100644 index 0000000..749f200 --- /dev/null +++ b/packages/pi-agent-journal/__tests__/evaluation-v3-result.test.ts @@ -0,0 +1,61 @@ +import { createHash } from "node:crypto"; +import { readFileSync } from "node:fs"; +import { describe, expect, it } from "vitest"; + +const result = JSON.parse(readFileSync("docs/evaluations/agent-work-journal-v3-results.json", "utf8")); + +describe("Agent Journal V3 terminal infrastructure result", () => { + it("fails before held-out selection without claiming a product outcome", () => { + expect(result).toMatchObject({ + schemaVersion: 3, + status: "FAIL", + failureStage: "infrastructure-acceptance", + releaseAuthorized: false, + productOutcome: "not-adjudicated", + concreteTasksSelected: false, + infrastructureAccepted: false, + gateEvaluated: false, + validTraceCount: 0, + terminalDecision: "stop-before-held-out-selection", + }); + expect(result.selectedTaskIds).toEqual([]); + expect(result.runIds).toEqual([]); + expect(result.scenarioMedians).toBeNull(); + }); + + it("records the real-preflight and independent-review blockers", () => { + expect(result.blockers).toEqual([ + "real_two_condition_two_phase_capsule_only_provenance_not_proven", + "byte_complete_independent_quarantine_not_accepted", + "sandbox_and_attempt_provenance_not_accepted", + "raw_to_derived_scorer_chain_not_accepted", + ]); + expect(result.completedUnits).toEqual(["v3_contract", "material_file_observation"]); + expect(result.repositorySnapshot).toBe("f33ed7cbfddbeee0251bab0b56f33c8af1ccf830"); + expect(result.preTaskManifest).toEqual({ + digest: createHash("sha256") + .update(readFileSync("docs/evaluations/agent-work-journal-v3-infrastructure-manifest.json")) + .digest("hex"), + taskIds: [], + promptDigests: [], + }); + }); + + it("contains no private or held-out material and records cleanup", () => { + const encoded = JSON.stringify(result); + expect(encoded).not.toMatch(/rawPrompt|phaseA|phaseB|toolArguments|toolResults|credential/i); + const strings: string[] = []; + const visit = (value: unknown): void => { + if (typeof value === "string") strings.push(value); + else if (Array.isArray(value)) value.forEach(visit); + else if (value && typeof value === "object") Object.values(value as Record).forEach(visit); + }; + visit(result); + expect(strings.some((value) => /^\/(?!\/)|^[A-Za-z]:[\\/]/.test(value))).toBe(false); + expect(result.privateEvidenceCleanup).toEqual({ + syntheticWorktreesDeleted: true, + syntheticSessionsAndTracesDeleted: true, + heldOutEvidenceCreated: false, + }); + }); +}); diff --git a/packages/pi-agent-journal/__tests__/tools.portable.test.ts b/packages/pi-agent-journal/__tests__/tools.portable.test.ts index 89a164d..102d18a 100644 --- a/packages/pi-agent-journal/__tests__/tools.portable.test.ts +++ b/packages/pi-agent-journal/__tests__/tools.portable.test.ts @@ -121,6 +121,49 @@ describe("Agent Journal portable tools", () => { expect(JSON.stringify(entry)).not.toContain("forged"); }); + it("canonicalizes forged checkpoint file provenance from the persisted originating entry", async () => { + writeFileSync(join(workspace, "checkpoint.txt"), "bound\n"); + await executePortableTool( + tool("journal_record"), + { + entries: [ + { + id: "checkpoint-source", + type: "evidence", + content: "Checkpoint support", + observe_files: [{ path: "checkpoint.txt", material: true }], + }, + ], + }, + { host: "test" }, + ); + const persistedEntry = (await storage.getSession("work")).entries[0]; + const forged = { + kind: "file", + path: "checkpoint.txt", + workspaceId: "forged", + observedHash: "forged", + observedAt: "forged", + originatingEntryId: "checkpoint-source", + material: false, + }; + const result = await executePortableTool( + tool("journal_checkpoint"), + { + action: "create", + objective: "Bind support", + status: "paused", + evidence_entry_ids: ["checkpoint-source"], + artifact_dependencies: [forged], + }, + { host: "test" }, + ); + expect(result.isError).toBeFalsy(); + const checkpoint = (await storage.getSession("work")).checkpoints[0]; + expect(checkpoint.artifactDependencies).toEqual([persistedEntry.dependencies[0]]); + expect(JSON.stringify(checkpoint)).not.toContain("forged"); + }); + it("rejects unsafe observed files atomically without leaking candidate data", async () => { writeFileSync(join(workspace, "good.txt"), "safe\n"); writeFileSync(join(workspace, ".env"), "PUBLIC=true\n"); diff --git a/packages/pi-agent-journal/extensions/journal-service.ts b/packages/pi-agent-journal/extensions/journal-service.ts index 04ce4ae..a2a3518 100644 --- a/packages/pi-agent-journal/extensions/journal-service.ts +++ b/packages/pi-agent-journal/extensions/journal-service.ts @@ -439,13 +439,22 @@ export class JournalService { `checkpoint next action must reference a next_action entry: '${nextActionEntryId}'`, ); } - for (const dependency of draft.artifactDependencies ?? []) { - if (!entryIds.has(dependency.originatingEntryId)) { + const artifactDependencies = (draft.artifactDependencies ?? []).map((dependency) => { + const origin = entriesById.get(dependency.originatingEntryId); + if (!origin) { throw new JournalValidationError( `checkpoint dependency entry '${dependency.originatingEntryId}' does not exist`, ); } - } + if (dependency.kind !== "file") return dependency; + const observed = origin.dependencies.filter( + (candidate): candidate is FileDependency => candidate.kind === "file" && candidate.path === dependency.path, + ); + if (observed.length !== 1) { + throw new JournalValidationError("checkpoint file dependency must reference one persisted observation"); + } + return observed[0]; + }); const checkpoint = validateCheckpointShape({ id: draft.id ?? this.idGenerator(), objective: draft.objective.trim(), @@ -453,7 +462,7 @@ export class JournalService { settledDecisionEntryIds, openQuestions: draft.openQuestions ?? [], evidenceEntryIds, - artifactDependencies: draft.artifactDependencies ?? [], + artifactDependencies, nextActionEntryId, supportEntryIds, createdAt: this.clock(), @@ -498,15 +507,41 @@ export class JournalService { private async readSafeFile(path: string): Promise<{ path: string; bytes: Buffer }> { const safePath = this.resolveSafeFile(path); const handle = await open(safePath, constants.O_RDONLY | constants.O_NOFOLLOW | constants.O_NONBLOCK); - const controller = new AbortController(); - const timeout = setTimeout(() => controller.abort(), this.freshnessTimeoutMs); try { - const stats = await handle.stat(); - if (!stats.isFile()) throw new JournalValidationError("artifact must be a regular file"); - if (stats.size > MAX_HASH_BYTES) throw new JournalValidationError("artifact exceeds hash byte limit"); - return { path: safePath, bytes: await handle.readFile({ signal: controller.signal }) }; + const before = await handle.stat(); + if (!before.isFile()) throw new JournalValidationError("artifact must be a regular file"); + if (before.size > MAX_HASH_BYTES) throw new JournalValidationError("artifact exceeds hash byte limit"); + const assertPathIdentity = (): void => { + const canonical = realpathSync(safePath); + if (canonical !== safePath || (canonical !== this.workspaceRoot && !canonical.startsWith(`${this.workspaceRoot}${sep}`))) { + throw new JournalValidationError("opened artifact escapes workspace"); + } + const pathStats = statSync(safePath); + if (pathStats.dev !== before.dev || pathStats.ino !== before.ino) { + throw new JournalValidationError("artifact identity changed while being observed"); + } + }; + assertPathIdentity(); + const buffer = Buffer.allocUnsafe(MAX_HASH_BYTES + 1); + let offset = 0; + while (offset <= MAX_HASH_BYTES) { + const { bytesRead } = await handle.read(buffer, offset, buffer.length - offset, null); + if (bytesRead === 0) break; + offset += bytesRead; + } + if (offset > MAX_HASH_BYTES) throw new JournalValidationError("artifact exceeds hash byte limit"); + const after = await handle.stat(); + assertPathIdentity(); + if ( + before.dev !== after.dev || + before.ino !== after.ino || + before.size !== after.size || + before.mtimeMs !== after.mtimeMs + ) { + throw new JournalValidationError("artifact changed while being observed"); + } + return { path: safePath, bytes: Buffer.from(buffer.subarray(0, offset)) }; } finally { - clearTimeout(timeout); await handle.close(); } } From 07cdae619ed49102e44ff22b6c1c3a88a8037234 Mon Sep 17 00:00:00 2001 From: Sebastian Otaegui Date: Mon, 13 Jul 2026 11:19:22 -0300 Subject: [PATCH 4/4] style(agent-journal): format observation identity guard --- packages/pi-agent-journal/extensions/journal-service.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/pi-agent-journal/extensions/journal-service.ts b/packages/pi-agent-journal/extensions/journal-service.ts index a2a3518..a19c61b 100644 --- a/packages/pi-agent-journal/extensions/journal-service.ts +++ b/packages/pi-agent-journal/extensions/journal-service.ts @@ -513,7 +513,10 @@ export class JournalService { if (before.size > MAX_HASH_BYTES) throw new JournalValidationError("artifact exceeds hash byte limit"); const assertPathIdentity = (): void => { const canonical = realpathSync(safePath); - if (canonical !== safePath || (canonical !== this.workspaceRoot && !canonical.startsWith(`${this.workspaceRoot}${sep}`))) { + if ( + canonical !== safePath || + (canonical !== this.workspaceRoot && !canonical.startsWith(`${this.workspaceRoot}${sep}`)) + ) { throw new JournalValidationError("opened artifact escapes workspace"); } const pathStats = statSync(safePath);