From 9312f4d97efe1f5f247356d622771dde6b9bd6d2 Mon Sep 17 00:00:00 2001 From: screenleon Date: Tue, 28 Jul 2026 10:19:58 +0900 Subject: [PATCH 1/4] feat(gate): add canonical policy resolution --- .gitignore | 2 + BACKLOG.md | 198 +++ commands/pr-gate.md | 23 +- commands/ship.md | 4 +- core/README.md | 11 + core/policy/gate-policy-consumers.tsv | 6 + core/policy/gate-policy-signals.tsv | 18 + core/schema/gate-assurance.schema.json | 411 +++++ core/schema/gate-policy-override.schema.json | 94 ++ .../script-variable-consumers.tsv | 2 + docs/dispatch-brief.md | 2 +- docs/model-tier-policy.md | 3 +- docs/review-model.md | 61 +- ops/setup/setup-project.sh | 8 +- runtime/bin/gate-supervisor.sh | 29 +- runtime/bin/pr-gate.sh | 1377 +++++++++++++++-- runtime/lib/gate-result-verify.sh | 121 +- runtime/lib/pmctl-gate.sh | 37 +- runtime/lib/pmctl-ship.sh | 4 +- skills/pr-gate-review/SKILL.md | 13 +- tests/bin/run-tests.sh | 9 +- tests/shell/test-commands.sh | 12 + tests/shell/test-core-schemas.sh | 149 +- tests/shell/test-gate-lifecycle.sh | 92 +- tests/shell/test-pmctl-gate.sh | 91 +- tests/shell/test-pmctl-ship.sh | 7 +- tests/shell/test-pr-gate-profile.sh | 13 +- tests/shell/test-pr-gate.sh | 1307 +++++++++++++--- tests/shell/test-run-tests.sh | 42 +- tests/shell/test-setup-project.sh | 8 +- 30 files changed, 3743 insertions(+), 411 deletions(-) create mode 100644 core/policy/gate-policy-consumers.tsv create mode 100644 core/policy/gate-policy-signals.tsv create mode 100644 core/schema/gate-policy-override.schema.json diff --git a/.gitignore b/.gitignore index 9487a919..6fb9340c 100644 --- a/.gitignore +++ b/.gitignore @@ -15,6 +15,8 @@ pm/rollup/.PORTFOLIO.md.* # Repo-local pm-dispatch state (context index, trace events) # Set PM_DISPATCH_STATE_ROOT=.pm-dispatch to use this location. .pm-dispatch/ +# Sandbox-local durable state used by ship worktree lanes. +.pm-dispatch-state/ # SQLite context index files — global scope is intentional: covers custom # PM_DISPATCH_STATE_ROOT paths inside the repo in addition to .pm-dispatch/. *.db diff --git a/BACKLOG.md b/BACKLOG.md index 54b18ee7..3fd53074 100644 --- a/BACKLOG.md +++ b/BACKLOG.md @@ -42,6 +42,10 @@ CC-001/CC-002 were consumed by PR #24 fix bundle inline, with no standalone entr | CC-520 | 🔵 active | synthesis parity 與 remediation seed:findings union、root-cause grouping、coverage matrix 與 no-silent-drop | ops/gate | 2026-07-23 | — | P1 | design | | CC-521 | 🔵 active | test-gap matrix、protocol recovery 與 live recall evaluation 分層 | ops/test | 2026-07-23 | — | P2 | design | | CC-522 | 🔵 active | 任意 `--test-cmd` 的 opaque/structured capability negotiation、執行失敗分類與外部 evidence recovery | ops/test | 2026-07-27 | feedback:2026-07-27 | P1 | design | +| CC-523 | 🔵 active | `pmctl gate cancel` 必須終止 reviewer 派發前仍在執行的 foreground preflight 與其 process tree | arch/gate | 2026-07-27 | feedback:2026-07-27 | P1 | hygiene | +| CC-524 | 🔵 active | `pmctl artifacts show` 顯示 canonical absolute run root 並提供穩定 machine-readable locator | ux/ops | 2026-07-27 | feedback:2026-07-27 | P2 | hygiene | +| CC-525 | 🔵 active | copy-mode verifier fallback 的 generated provenance 必須指向實際 generator,並由 parity ratchet 防止再次漂移 | ops/test | 2026-07-28 | feedback:2026-07-28 | P3 | hygiene | +| CC-526 | 🔵 active | reviewer override file 的 symlink trust-boundary hardening 與相容性契約 | security/gate | 2026-07-28 | feedback:2026-07-28 | P2 | hygiene | | CC-465 | 🔵 active | memory/context 關鍵詞管線 CJK 支援:抽出共用零依賴斷詞 lib,取代三處各自 ASCII-only 抽詞;工作序列起點(465→467→468→466)(2026-07-07 記憶系統深入分析) | memory | 2026-07-07 | feedback:2026-07-07 | P2 | retrieval | | CC-466 | ⏸ deferred | 記憶卡片生命週期閉環:expires_at 執行 + 關窗式 supersede + usage sidecar 休眠偵測 + doctor→distill 接線;僅在 CC-467 證明 stale/dormant card 已形成實際問題時啟動 | memory | 2026-07-07 | feedback:2026-07-07 | P2 | retrieval | | CC-467 | 🔵 active | `pmctl memory stats`:注入效益可視化(唯讀聚合器)——注入 bytes/卡片命中分佈/從未命中卡/episode 填寫率,回答「記憶有跟沒有差在哪」;排在 CC-466 之前(2026-07-07;業界僅離線 recall 評測,無 per-injection 遙測) | DX/memory | 2026-07-07 | — | P2 | retrieval | @@ -2218,6 +2222,200 @@ protocol recovery contract保持正交。P1。 --- +## CC-523 — gate cancel 終止 pre-review foreground producer work 🔵 active + +**Framing**: 本票是 [[CC-508]] parent-operation cancellation 契約的 regression +closure,不重做 operation control plane、`pmctl dispatch cancel` 或 gate workflow。 +範圍限於 gate 已建立 parent operation、但尚未派發第一個 reviewer child 時仍由 +producer 擁有的執行工作;foreground preflight 是必須通過的原始重現,detached +lifecycle 若共用同一 pre-review seam 也不得保留分歧。允許加入最小、可重用的 +producer execution identity/cancellation metadata,但不建立泛用 job-control +framework,也不接受使用者提供的裸 PID。 + +**Problem**: 現行 `pmctl_operation_cancel` 只遍歷 parent record 已記錄的 child +dispatch runs。foreground `pmctl gate run` 在 reviewer 派發前會同步執行 +`--test-cmd` preflight;此時 operation 已存在但 `children.jsonl` 仍為空。 +`pmctl gate cancel ` 因而可把 operation 直接寫成 `cancelled`,卻沒有 +停止仍在執行的 `pr-gate.sh`、`timeout` wrapper 或測試 process tree。2026-07-27 +實際操作已觀察到 cancel 回報後 foreground preflight 仍持續執行。這使 durable +state 與真實 liveness 互相矛盾,也可能讓已取消 producer 繼續寫 artifact、晚到 +派發 reviewer,違反 [[CC-508]]「任一 in-flight producer 可驗證取消且無孤兒」的 +Done-when。 + +**Requirement**: + +1. gate 必須在進入任何 pre-review 工作前,發布與 operation ID、canonical + repository/workdir 綁定的 producer execution identity;若以 PID/process + group 表示,必須含 starttime 或等價 anti-reuse evidence,並由 producer 自行 + 寫入 trusted state,cancel caller 不得注入任意 PID。 +2. `pmctl gate cancel` 遇到尚無 reviewer child、但 producer/preflight 仍 live + 的 operation 時,必須先要求 producer 停止並終止該次 preflight 的完整 process + tree;沿用 bounded grace 後 escalation 的取消語意。只有確認 producer 與其 + owned descendants 已停止後才能寫 `cancelled`;identity mismatch、無法確認 + termination 或部分停止一律收斂為 `indeterminate`/非零。 +3. gate producer 必須在 preflight 完成後、每次 reviewer dispatch 前與 + finalization 前檢查 durable cancellation intent。cancel 已勝出的 operation + 不得再派發 child、不得以 late GO/NO-GO/failed 覆寫 `cancelled`,也不得把 + cancelled preflight 誤報成程式碼 test failure。 +4. foreground caller 必須以明確非成功狀態返回,並輸出 operation ID、取消結果與 + 可查 evidence;若 detached gate 在同一階段被取消,supervisor、sentinel 與 + wait 結論必須使用相同 terminal semantics,不得只殺 child 或只改 state。 +5. 已派發 reviewer 後的既有 child ownership/`pmctl dispatch cancel` 路徑、 + foreign-project 拒絕、cancel-vs-complete 單一終態與 reconcile 規則必須保持; + producer cancellation primitive 應為窄 API,不在 gate/ship 各自複製未驗證的 + signal 邏輯。 +6. deterministic regression 使用 FIFO/readiness handshake 啟動會阻塞的 + foreground preflight,從另一 process 執行 `pmctl gate cancel`,驗證 bounded + return、preflight 與 descendants 全部死亡、operation 為 `cancelled`、零 + reviewer dispatch、零 late artifact overwrite。另覆蓋 cancel/finish race、 + PID reuse/identity mismatch、重複 cancel、preflight 已退出與 termination + 失敗轉 `indeterminate`;禁止以裸 sleep 猜時序。 + +**Done-when**: 對 reviewer 派發前仍在執行的 foreground gate operation 執行 +`pmctl gate cancel ` 後,cancel 只有在 producer-owned preflight +process tree 已可驗證停止時才回報 `cancelled`;原 foreground caller bounded +返回、沒有 reviewer child 或孤兒程序、沒有 late terminal overwrite,且 +foreground/detached 共用一致的 cancellation terminal contract。 + +**Non-goals**: 不重寫 preflight evidence classification(→ [[CC-522]]);不擴張 +為任意 shell job manager;不允許 PID-only cancellation;不變更使用者未要求的 +timeout 預設。 + +**Dependencies**: regression boundary 直接承接 [[CC-508]],並複用 [[CC-495]] +dispatch cancellation 與 [[CC-509]] supervisor identity/liveness evidence。P1, +應先於下一次依賴 foreground gate cancellation 的 maintainer delivery 處理。 + +--- + +## CC-524 — artifacts show canonical absolute run root 🔵 active + +**Framing**: 本票補齊 [[CC-418]] observer/discoverability 已交付後暴露的 locator +缺口,不搬動 artifact、不改 state partition layout,也不把 `artifacts show` +擴張成檔案內容 viewer。human output 與 machine output 都必須由同一 canonical +state-path resolver 產生,不能另做 repo-local fallback 或掃描猜測。 + +**Problem**: `pmctl artifacts show --cd ` 已能從 canonical project +partition 找到 run directory,成功時卻只列出 ``。 +使用者因此知道 `.gate-results/foo.md` 存在,卻不知道它實際位於哪個絕對根目錄; +尤其 artifact 已搬離 target repo、`PM_DISPATCH_STATE_ROOT`/XDG/HOME precedence +可能不同時,只能再次猜測或全檔案系統搜尋。錯誤路徑反而會印出 resolved run +directory,形成成功與失敗輸出的可發現性倒置。 + +**Requirement**: + +1. 成功的 human output 必須在任何 file rows 前明確印出 canonical physical + `run root: `,其值為同一 `--cd` project partition 下 + `runs/` 的實際目錄;即使 run directory 為空也必須印 root。 +2. 保留現有檔案 size 與 relative-path 資訊;新增穩定 `--json` locator contract, + 至少含 schema/kind、run ID、canonical repository root、canonical absolute run + root,以及依穩定順序排列的 `{relative_path,size_bytes}` files。human label 與 + JSON field 的 root 必須完全一致。 +3. root 必須經 canonical path resolution,且驗證仍位於 resolver 選出的 project + `runs/` containment 下;symlinked state root、relative `--cd`、git subdirectory + 與 custom `PM_DISPATCH_STATE_ROOT` 都不得產生 lexical-only、foreign partition + 或不存在的 success locator。 +4. state-root precedence 繼續使用現行 + `PM_DISPATCH_STATE_ROOT` → `XDG_DATA_HOME` → HOME fallback;不得因 target repo + 找不到 `.gate-results` 就掃描其他 project partitions。unknown run/wrong + `--cd` 維持非零,並在不越權搜尋的前提下給出可複製的 `artifacts list/show` + recovery 指令。 +5. regression fixtures 覆蓋 default 與 custom state root、含空白路徑、空 run、 + 多層 artifact、human/JSON parity、stable ordering、wrong project、symlink + canonicalization 與 containment rejection;現有 size/relative-path consumer + 必須有明確 compatibility 測試或 migration 說明。 + +**Done-when**: 操作者只執行一次 +`pmctl artifacts show --cd ` 就能複製 canonical absolute run root +並直接定位列出的 artifact;automation 可用 `--json` 取得同一 locator,不需猜測 +state store、搜尋 target repo 或解析錯誤訊息。 + +**Non-goals**: 不新增 `cat`/download/open 子指令;不改 artifact retention/GC; +不遷移既有 runs;不放寬 project partition containment;不替 missing artifact +重建內容。 + +**Dependencies**: 延伸 [[CC-418]] 已建立的 artifacts observer 與 +`state-paths.sh` canonical partition seam;與 [[CC-515]] artifact +freshness/applicability verifier 正交。P2。 + +--- + +## CC-525 — generated verifier fallback provenance path ratchet 🔵 active + +**Framing**: 本票是 copy-mode 維護資訊的窄幅清理,不改 verifier 行為、gate +verdict 或 bundle layout。`runtime/bin/pr-gate.sh` 的 inline fallback 仍由唯一既有 +generator 管理;修正與 ratchet 應併入後續小型 maintenance change。 + +**Problem**: inline fallback 的 generated block 註解目前宣稱由不存在的 +`scripts/sync-gate-result-verifier-fallback.sh` 產生,實際 canonical generator +是 `tools/generate-gate-result-verifier-fallback.sh`。現有 `--check` 能驗證內容 +parity,卻沒有驗證 provenance 指向可執行、存在且唯一的 generator;維護者依註解 +操作時會走到錯誤 recovery path。 + +**Requirement**: + +1. generated block 的 provenance 必須指向 repo 內實際 canonical generator, + 路徑可由 repository root 穩定解析,且文件與測試不得另宣告第二個同步工具。 +2. 擴充既有 generator `--check` 或相鄰 contract test,同時驗證 marker、generated + body parity 與 provenance path;不存在、不可執行或漂移到非 canonical 路徑時 + 必須 fail-loud。 +3. copy-mode standalone fallback、repo-layout shared verifier 與現有 ShellCheck + source annotation 均保持;註解修正不得手動改寫 generated verifier body。 + +**Done-when**: 維護者可直接依 inline 註解執行實際 generator;CI 在 provenance +再次指向不存在或非 canonical 工具時失敗,而目前 verifier parity 與 copy-mode +行為完全不變。 + +**Non-goals**: 不新增 generator;不改 verdict parser、artifact schema 或 fallback +內容;不把 generator 搬到另一個目錄;不併入 [[CC-513]] 的 policy resolver。 + +**Dependencies**: 延伸 [[CC-512]] 的 shared verifier/fallback parity seam,與 +[[CC-513]] 僅共享發現時點、沒有交付依賴。P3,適合後續小型 maintenance PR。 + +--- + +## CC-526 — reviewer override symlink trust-boundary hardening 🔵 active + +**Framing**: 本票只處理 free-form reviewer override channel +(auto-discovered `.gate-overrides.md` 與 explicit `--override-file`)的檔案信任 +邊界。它與 [[CC-513]] 的 machine-validated policy override 是不同輸入面;因為 +拒絕 symlink 會改變既有信任/相容行為,必須獨立交付並明確測試。 + +**Problem**: reviewer override 目前只以 `-f` 接受檔案,再 canonicalize parent +並讀取內容;symlink 指向 regular target 仍會通過。workspace 內的 +`.gate-overrides.md` 或 explicit path 因此可把 reviewer prompt content +重新導向其他位置,而現有 provenance 只記錄 symlink lexical path 與 target +content hash,沒有把這項 redirect semantics 當成可見的 trust decision。 + +**Requirement**: + +1. auto-discovered 與 explicit reviewer override 必須使用同一窄 validator,只接受 + readable、non-empty、regular、non-symlink file;拒絕訊息需指出輸入與違反的 + contract,且在任何 reviewer dispatch/brief injection 前 fail-closed。 +2. validation、canonical path、讀取與 sha256 provenance 的順序必須避免 + check/use 間把 symlink 或 target 換入;若 shell primitive 無法提供原子 open, + 必須以可驗證 identity/content stability check 收斂,而非只增加一次 `-L`。 +3. regular file 的 auto-discovery、relative explicit path、含空白路徑、content + injection 與 provenance 行為保持相容;若 empty/unreadable file 原先可用而 + 新契約改為拒絕,需在 CLI/review docs 明示 migration。 +4. deterministic regression 覆蓋 auto 與 explicit symlink、absolute/relative + external target、dangling link、empty/unreadable file、正常檔案及 validation + 後置換情境;測試必須證明被拒內容不會出現在 reviewer brief 或 result + provenance。 + +**Done-when**: symlink 或 validation 後遭置換的 reviewer override 無法影響任何 +reviewer brief;regular override 的既有使用方式維持,新的拒絕行為有契約測試與 +相容性說明。 + +**Non-goals**: 不重新設計 accepted-risk 語法;不把 reviewer override 升格為 +policy downgrade;不宣稱防禦具有同一 OS 帳號寫入權限的攻擊者;不順帶修改 +[[CC-513]] policy override validation。 + +**Dependencies**: 與 [[CC-513]] 的 policy override trust boundary 保持正交,並 +可參考 [[CC-258]] 的 symlink-safe install contract,但不得假設 realpath-only +即足夠。P2,需獨立 review 與 compatibility evidence。 + +--- + ## CC-508 — 所有間接 dispatch 的 parent-operation control plane ✅ 2026-07-25 **Problem**: `pmctl gate run`、`pmctl ship --parallel`/adapter 路徑、`pmctl task dispatch` 與任何未來 producer 都可能以一個 parent operation 間接啟動一或多個 detached dispatch;但產品控制面主要只暴露個別 `pmctl dispatch cancel `。parent ID 與其子 run 沒有強制、可查的 ownership relation,也沒有一致的 producer-level cancel surface。當任一 producer 卡住、選錯 executor 或需中止時,操作者無法透過 pmctl 取消整個 operation;直接對 supervisor PID 操作會繞過 run state、sentinel 與 cancel-vs-complete 單一終態契約,並可能留下無法判定的 stale operation。 diff --git a/commands/pr-gate.md b/commands/pr-gate.md index 0709652b..78b5ae7f 100644 --- a/commands/pr-gate.md +++ b/commands/pr-gate.md @@ -5,6 +5,9 @@ argument-hint: "[express|standard|full] [--targeted r1,r2 --initial-result path] Run the PR gate via `pmctl gate run`. The `runtime/bin/pr-gate.sh` script is the internal implementation; `pmctl gate run` is the preferred invocation surface. +This command uses the `generic` consumer policy: one canonical resolver combines +the diff, trusted brief metadata, requested tier/mode/pass/coverage, and +repository policy before any reviewer is dispatched. **Sequential mode (default):** all reviewers run in one combined session. Low main-thread token cost (~5k dispatch + read result). @@ -17,8 +20,17 @@ paths or when reviewer independence matters. |---|---| | Routine code / seed / docs changes | _(none)_ | | Re-gate after fixing specific findings | `--targeted qa-tester,risk-reviewer --initial-result ` | -| Auth / payment / migration / sensitive paths | `--mode parallel` | -| Force a specific tier | `express` / `standard` / `full` | +| Auth / payment / migration / sensitive paths | _(none; policy adds the matching reviewer and recommends parallel)_ | +| Input/evaluation/command execution boundary | _(none; policy requires parallel)_ | +| Request independent reviewer sessions | `--mode parallel` | +| Request a specific tier | `express` / `standard` / `full` (cannot lower the policy floor) | + +A policy rejection happens before reviewer dispatch and is an execution/policy +failure, not a `Final: NO-GO` reviewer verdict. Scope-bound downgrades use the +runtime's explicit structured policy-override contract. Its scope fingerprint +binds the actual tracked patch plus in-scope untracked content, so an approval +cannot be replayed after a same-shape content change. `.gate-overrides.md` only +supplies reviewer finding context and cannot lower policy. ## Step 1 - Invoke pmctl directly @@ -179,7 +191,7 @@ SCOPE="${SCOPE_TOKENS[*]:-}" # `pmctl:*` prefix match. The quotes around the placeholder keep this block # valid, executable Bash even before that substitution (bare, unquoted # `` would be parsed as I/O redirection and fail to parse). -GATE_ARGS=(--cd "" --executor "$GATE_EXECUTOR") +GATE_ARGS=(--cd "" --executor "$GATE_EXECUTOR" --policy generic) [[ -n "$GATE_MODEL" ]] && GATE_ARGS+=(--model "$GATE_MODEL") [[ -n "$TIER_OVERRIDE" ]] && GATE_ARGS+=(--tier "$TIER_OVERRIDE") [[ -n "$TARGETED_REVIEWERS" ]] && GATE_ARGS+=(--targeted "$TARGETED_REVIEWERS" --initial-result "$INITIAL_RESULT") @@ -289,7 +301,10 @@ When the `pmctl gate wait` background Bash completion notification arrives: and matching canonical terminal run records. The producer publishes the sidecar before the v2 result that references it; verification briefly retries when it observes an in-flight v2 result before its protected - attestation rename completes. + attestation rename completes. Current envelopes also carry the shell-owned + policy resolution (classification, matched signals, floors, resolved + coordinates, and override provenance); earlier v2 envelopes without that + optional block remain readable. 5. Prepend `PR-gate complete.` to completion relay and include the full gate result (including `Final: GO` / `Final: NO-GO`) unchanged. 6. On failure, avoid collapsing findings; relay the actual stderr summary and diff --git a/commands/ship.md b/commands/ship.md index 129bcac0..4b3f11a1 100644 --- a/commands/ship.md +++ b/commands/ship.md @@ -135,7 +135,7 @@ handoff, and keep the same resolved pair for targeted re-runs. Add `--model ""` below only when the executor default does not already resolve to the selected gate model. -Run `pmctl gate run --executor --cd "" --lifecycle foreground` +Run `pmctl gate run --executor --policy maintainer --cd "" --lifecycle foreground` (substitute `` with the literal absolute working directory, not `"$PWD"` — a shell-variable expansion makes the command unanalyzable statically and forces a manual approval every time even though a bare @@ -172,7 +172,7 @@ once the call returns. preserve the established structure, such as wording/comments, a narrow assertion or fixture adjustment, or a small guard/error-handling fix. - Then re-run `pmctl gate run --executor --cd "" + Then re-run `pmctl gate run --executor --policy maintainer --cd "" --lifecycle foreground --targeted --initial-result ""` (same literal-path substitution as Step 3's first call — never `"$PWD"`). The initial-result path is the comprehensive diff --git a/core/README.md b/core/README.md index 558479fb..db655a7b 100644 --- a/core/README.md +++ b/core/README.md @@ -10,6 +10,17 @@ This directory contains the canonical PM-runtime data contract. **`core/` is def (`~/.local/share/pm-dispatch/state/`). **Definitions, not the writer.** - `context-pack/` — source-interface contract for ContextPack assembly. +Gate assurance definitions are split deliberately: + +- `policy/gate-tiers.tsv`, `gate-modes.tsv`, and `gate-pass-kinds.tsv` define + the independent assurance coordinates. +- `policy/gate-policy-consumers.tsv` and `gate-policy-signals.tsv` define + consumer-specific coverage and deterministic risk floors. +- `schema/gate-policy-override.schema.json` defines explicit scope-bound user + approval for a downgrade. +- `schema/gate-assurance.schema.json` defines the portable envelope that records + both resolved coordinates and the policy resolution that produced them. + ## Invariants 1. **`core/*` may NOT import / reference `runtime/`, `scripts/`, `adapters/`, diff --git a/core/policy/gate-policy-consumers.tsv b/core/policy/gate-policy-consumers.tsv new file mode 100644 index 00000000..1e3c94b8 --- /dev/null +++ b/core/policy/gate-policy-consumers.tsv @@ -0,0 +1,6 @@ +# Gate policy consumers. Consumer policy does not rewrite tier, mode, or pass semantics. +policy_pass policy pass_kind minimum_tier required_reviewers recommended_mode required_mode +generic:initial generic initial express critic,qa-tester sequential none +generic:targeted generic targeted express none sequential none +maintainer:initial maintainer initial express critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +maintainer:targeted maintainer targeted express none parallel none diff --git a/core/policy/gate-policy-signals.tsv b/core/policy/gate-policy-signals.tsv new file mode 100644 index 00000000..5c167527 --- /dev/null +++ b/core/policy/gate-policy-signals.tsv @@ -0,0 +1,18 @@ +# Gate policy signals. Reviewer requirements apply to initial discovery; targeted passes retain tier/mode signals but use requested remediation coverage. +signal match_source pattern minimum_tier required_reviewers recommended_mode required_mode +docs-only classification docs-only express none sequential none +bounded-runtime classification bounded-runtime express none sequential none +medium-change classification medium-change standard architecture-reviewer parallel none +large-change classification large-change full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +binary-change classification binary-change standard architecture-reviewer parallel none +renamed-input classification renamed express none sequential none +untracked-input classification untracked express none sequential none +generated-input classification generated express none sequential none +cross-boundary classification cross-boundary standard architecture-reviewer parallel none +security-sensitive-path path-regex (^|[/_.-])(auth|oauth|jwt|sessions?|secrets?|passwords?|tokens?|credentials?|cors|csrf|webhooks?|sudo|ssh|payments?|billing)([/_.-]|$) express security-reviewer parallel none +input-execution-path path-regex (^|[/_.-])(eval|exec|execute|command|shell|hook|guard|allowlist)([/_.-]|$)|(^|/)(\.github|workflows?|ci)(/|$) standard security-reviewer parallel parallel +risk-sensitive-path path-regex (^|[/_.-])(migrations?|migrate|destructive|deletions?|delete|removals?|remove|rollback|concurrency|concurrent|race|locks?|cancel|reconcile)([/_.-]|$) express risk-reviewer parallel none +public-contract-path path-regex (^|/)(cli|commands|skills|core/schema)(/|$)|(^|[/_.-])(apis?|schemas?|contracts?)([/_.-]|$) standard architecture-reviewer parallel none +policy-source-path path-regex (^|/)core/policy(/|$) full architecture-reviewer,security-reviewer,risk-reviewer parallel none +brief-architecture-minor brief-value minor standard architecture-reviewer parallel none +brief-architecture-major brief-value major full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none diff --git a/core/schema/gate-assurance.schema.json b/core/schema/gate-assurance.schema.json index 18ba364d..e71b4362 100644 --- a/core/schema/gate-assurance.schema.json +++ b/core/schema/gate-assurance.schema.json @@ -103,6 +103,9 @@ }, "additionalProperties": false }, + "policy": { + "$ref": "#/definitions/policyResolution" + }, "dispatch": { "type": "object", "required": ["outcomes"], @@ -142,6 +145,414 @@ } }, "definitions": { + "policyResolution": { + "type": "object", + "required": [ + "kind", + "schema_version", + "consumer_policy", + "policy_source", + "scope_fingerprint", + "request", + "classification", + "resolution", + "matched_signals", + "resolved", + "enforcement", + "override", + "reviewer_override" + ], + "properties": { + "kind": { + "const": "gate_policy_resolution_v1" + }, + "schema_version": { + "const": 1 + }, + "consumer_policy": { + "enum": [ + "generic", + "maintainer" + ] + }, + "policy_source": { + "enum": [ + "canonical", + "generated-snapshot", + "mixed" + ] + }, + "scope_fingerprint": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "request": { + "type": "object", + "required": [ + "tier", + "mode", + "pass_kind", + "reviewers" + ], + "properties": { + "tier": { + "enum": [ + "auto", + "express", + "standard", + "full" + ] + }, + "mode": { + "enum": [ + "default", + "sequential", + "parallel" + ] + }, + "pass_kind": { + "enum": [ + "initial", + "targeted" + ] + }, + "reviewers": { + "oneOf": [ + { + "type": "null" + }, + { + "$ref": "#/definitions/reviewerSet" + } + ] + } + }, + "additionalProperties": false + }, + "classification": { + "type": "object", + "required": [ + "architecture_impact", + "line_changes", + "binary_or_unknown_count", + "layer_roots" + ], + "properties": { + "architecture_impact": { + "enum": [ + "unknown", + "none", + "minor", + "major" + ] + }, + "line_changes": { + "type": "integer", + "minimum": 0 + }, + "binary_or_unknown_count": { + "type": "integer", + "minimum": 0 + }, + "layer_roots": { + "$ref": "#/definitions/stringSet" + } + }, + "additionalProperties": false + }, + "resolution": { + "type": "object", + "required": [ + "minimum_tier", + "required_reviewers", + "recommended_mode", + "required_mode", + "downgrade_requested", + "downgrade_allowed" + ], + "properties": { + "minimum_tier": { + "enum": [ + "express", + "standard", + "full" + ] + }, + "required_reviewers": { + "$ref": "#/definitions/reviewerSet" + }, + "recommended_mode": { + "enum": [ + "sequential", + "parallel" + ] + }, + "required_mode": { + "type": [ + "string", + "null" + ], + "enum": [ + "sequential", + "parallel", + null + ] + }, + "downgrade_requested": { + "type": "boolean" + }, + "downgrade_allowed": { + "type": "boolean" + } + }, + "additionalProperties": false + }, + "matched_signals": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "required": [ + "id", + "source", + "matches", + "minimum_tier", + "required_reviewers", + "recommended_mode", + "required_mode" + ], + "properties": { + "id": { + "type": "string", + "minLength": 1 + }, + "source": { + "enum": [ + "consumer-policy", + "classification", + "path-regex", + "brief-value" + ] + }, + "matches": { + "$ref": "#/definitions/nonEmptyStringSet" + }, + "minimum_tier": { + "enum": [ + "express", + "standard", + "full" + ] + }, + "required_reviewers": { + "$ref": "#/definitions/reviewerSet" + }, + "recommended_mode": { + "enum": [ + "sequential", + "parallel" + ] + }, + "required_mode": { + "type": [ + "string", + "null" + ], + "enum": [ + "sequential", + "parallel", + null + ] + } + }, + "additionalProperties": false + } + }, + "resolved": { + "type": "object", + "required": [ + "tier", + "mode", + "reviewers" + ], + "properties": { + "tier": { + "enum": [ + "express", + "standard", + "full" + ] + }, + "mode": { + "enum": [ + "sequential", + "parallel" + ] + }, + "reviewers": { + "$ref": "#/definitions/reviewerSet" + } + }, + "additionalProperties": false + }, + "enforcement": { + "type": "object", + "required": [ + "status", + "violations" + ], + "properties": { + "status": { + "const": "pass" + }, + "violations": { + "type": "array", + "items": { + "type": "object", + "required": [ + "coordinate", + "requested", + "required" + ], + "properties": { + "coordinate": { + "enum": [ + "tier", + "coverage", + "mode" + ] + }, + "requested": {}, + "required": {} + }, + "additionalProperties": false + } + } + }, + "additionalProperties": false + }, + "override": { + "type": "object", + "required": [ + "status", + "source", + "sha256", + "reason", + "approver" + ], + "properties": { + "status": { + "enum": [ + "not_provided", + "not_needed", + "applied", + "scope_mismatch", + "allowance_mismatch" + ] + }, + "source": { + "type": [ + "string", + "null" + ] + }, + "sha256": { + "type": [ + "string", + "null" + ], + "pattern": "^[a-f0-9]{64}$" + }, + "reason": { + "type": [ + "string", + "null" + ] + }, + "approver": { + "oneOf": [ + { + "type": "null" + }, + { + "$ref": "#/definitions/policyApprover" + } + ] + } + }, + "additionalProperties": false + }, + "reviewer_override": { + "type": "object", + "required": [ + "status", + "source", + "sha256" + ], + "properties": { + "status": { + "enum": [ + "not_provided", + "provided" + ] + }, + "source": { + "type": [ + "string", + "null" + ] + }, + "sha256": { + "type": [ + "string", + "null" + ], + "pattern": "^[a-f0-9]{64}$" + } + }, + "additionalProperties": false + } + }, + "additionalProperties": false + }, + "stringSet": { + "type": "array", + "uniqueItems": true, + "items": { + "type": "string" + } + }, + "nonEmptyStringSet": { + "type": "array", + "minItems": 1, + "uniqueItems": true, + "items": { + "type": "string", + "minLength": 1 + } + }, + "policyApprover": { + "type": "object", + "required": [ + "kind", + "identity", + "approval_ref" + ], + "properties": { + "kind": { + "const": "user" + }, + "identity": { + "type": "string", + "minLength": 1 + }, + "approval_ref": { + "type": "string", + "minLength": 1 + } + }, + "additionalProperties": false + }, "reviewerSet": { "type": "array", "uniqueItems": true, diff --git a/core/schema/gate-policy-override.schema.json b/core/schema/gate-policy-override.schema.json new file mode 100644 index 00000000..29f7e9ad --- /dev/null +++ b/core/schema/gate-policy-override.schema.json @@ -0,0 +1,94 @@ +{ + "title": "Gate policy downgrade override", + "description": "Explicit user authorization for a scope-bound gate policy downgrade.", + "type": "object", + "required": [ + "kind", + "schema_version", + "scope_fingerprint", + "allow", + "reason", + "approver" + ], + "properties": { + "kind": { + "const": "gate_policy_override_v1" + }, + "schema_version": { + "const": 1 + }, + "scope_fingerprint": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "allow": { + "type": "object", + "required": [ + "tier", + "omit_reviewers", + "mode" + ], + "properties": { + "tier": { + "type": [ + "string", + "null" + ], + "enum": [ + "express", + "standard", + "full", + null + ] + }, + "omit_reviewers": { + "type": "array", + "uniqueItems": true, + "items": { + "type": "string", + "pattern": "^[a-z0-9][a-z0-9-]*$" + } + }, + "mode": { + "type": [ + "string", + "null" + ], + "enum": [ + "sequential", + "parallel", + null + ] + } + }, + "additionalProperties": false + }, + "reason": { + "type": "string", + "minLength": 1 + }, + "approver": { + "type": "object", + "required": [ + "kind", + "identity", + "approval_ref" + ], + "properties": { + "kind": { + "const": "user" + }, + "identity": { + "type": "string", + "minLength": 1 + }, + "approval_ref": { + "type": "string", + "minLength": 1 + } + }, + "additionalProperties": false + } + }, + "additionalProperties": false +} diff --git a/docs/architecture/script-variable-consumers.tsv b/docs/architecture/script-variable-consumers.tsv index 24615237..92ae7f02 100644 --- a/docs/architecture/script-variable-consumers.tsv +++ b/docs/architecture/script-variable-consumers.tsv @@ -41,9 +41,11 @@ CODEX_DISPATCH_TIMEOUT CODEX_DISPATCH_TIMEOUT runtime/lib/pmctl-dispatch.sh prod CODEX_GATE_STUB_* CODEX_GATE_STUB_ tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_BOLD_FINAL tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_CALLED_MARKER tests/shell/test-pr-gate-profile.sh test +CODEX_GATE_STUB_* CODEX_GATE_STUB_CONFLICTING_VERDICT tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_CROSS_TAMPER_REVIEWER tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_CROSS_TAMPER_VICTIM tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_FRONTMATTER_FINAL tests/shell/test-pr-gate.sh test +CODEX_GATE_STUB_* CODEX_GATE_STUB_HEADER_ONLY_VERDICT tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_INJECT_FILE tests/shell/test-pr-gate.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_MODE tests/shell/test-pr-gate-profile.sh test CODEX_GATE_STUB_* CODEX_GATE_STUB_MODE tests/shell/test-pr-gate.sh test diff --git a/docs/dispatch-brief.md b/docs/dispatch-brief.md index c238e375..41582d93 100644 --- a/docs/dispatch-brief.md +++ b/docs/dispatch-brief.md @@ -87,7 +87,7 @@ This applies to both `codex` and `claude` executors — Claude has more context Use as needed; not all briefs require all of them. -- **`architecture_impact`** — `none | minor | major`. Declares the architectural weight of this change. When `architecture_impact: major`, `conceptual_map` is **required** — `brief-validate.sh` will FAIL without it. When `architecture_impact: minor`, `conceptual_map` is recommended — `brief-validate.sh` will WARN if absent. Drives `pr-gate` tier suggestion: `none` → express; `minor` → standard; `major` → full. +- **`architecture_impact`** — `none | minor | major`. Declares the architectural weight of this change. When `architecture_impact: major`, `conceptual_map` is **required** — `brief-validate.sh` will FAIL without it. When `architecture_impact: minor`, `conceptual_map` is recommended — `brief-validate.sh` will WARN if absent. When supplied to `pr-gate`, this trusted metadata participates in canonical policy resolution: `minor` establishes at least a `standard` tier, `major` establishes `full`, and `none` adds no floor beyond the diff's own signals. | Value | Definition | Examples | |---|---|---| diff --git a/docs/model-tier-policy.md b/docs/model-tier-policy.md index e2b6d5e8..865a7a9f 100644 --- a/docs/model-tier-policy.md +++ b/docs/model-tier-policy.md @@ -108,7 +108,8 @@ flow above) only when **all three** hold: 1. Tier is `full` 2. Diff exceeds 1000 changed lines -3. At least one sensitive path triggered `full` (auth, payments, migrations, CI, etc.) +3. At least one sensitive-path policy signal contributed required coverage + (auth, payments, migrations, CI, etc.) Example ask: > "This PR is large and sensitive (>1000 lines, touches [path]). Opus reviewers diff --git a/docs/review-model.md b/docs/review-model.md index 45eb309c..a63d4ef5 100644 --- a/docs/review-model.md +++ b/docs/review-model.md @@ -145,38 +145,73 @@ When an artifact is missing — no `conceptual_map`, no `self_verify cmd:`, no i ## pr-gate assurance coordinates -The `/pr-gate` `--tier` flag selects the **rigor level** required for this change — not just the number of reviewers. Choose based on `architecture_impact` and blast radius: +The gate resolves one canonical policy result from the complete diff +classification, trusted brief metadata, requested coordinates, consumer policy, +and any explicit scope-bound downgrade authorization. `--tier` requests a +**rigor level**; it cannot silently lower the resolver's minimum floor. | Tier | When | Default reviewer coverage | |---|---|---| -| `express` | hotfix, docs-only, `architecture_impact: none` | critic + qa | -| `standard` | feature, `architecture_impact: minor` | conceptual map required + critic + qa + architecture-reviewer | -| `full` | architectural change, `architecture_impact: major`, sensitive path | critic + qa + architecture-reviewer + security + risk | +| `express` | docs-only or bounded low-blast-radius change | critic + qa | +| `standard` | medium/binary/cross-boundary change, `architecture_impact: minor` | critic + qa + architecture-reviewer | +| `full` | large change or `architecture_impact: major` | critic + qa + architecture-reviewer + security + risk | -**Tier suggestion**: when a `--brief` is passed to `pr-gate.sh`, it reads `architecture_impact` from the brief and emits an advisory to stderr before dispatch if the auto-detected tier is lower than the impact level implies. The user-selected or auto-detected tier always takes precedence; the advisory is informational only and does not block or alter the tier. +Sensitive paths add the corresponding security, risk, or architecture reviewer +to required coverage without automatically converting every bounded change to +`full`. Trusted `architecture_impact` is an enforced policy input: +`minor` establishes a `standard` floor and `major` establishes a `full` floor. **Tier, mode, pass kind, and coverage are independent**: -- Tier records rigor intent and supplies default reviewer coverage. `--reviewers` - may override the selected coverage without rewriting the tier. +- Tier records rigor intent and supplies default reviewer coverage. An explicit + `--reviewers` list does not rewrite the tier, but it must still include every + reviewer required by the matched risk signals. - Mode records execution topology. The default is `sequential`; select - `--mode parallel` when separate reviewer sessions and synthesis are required. - A `full` tier does not select parallel mode by itself. + `--mode parallel` when separate reviewer sessions and synthesis are desired. + Policy records recommendation separately from requirement; only an explicit + isolation signal requires parallel mode. A `full` tier does not select + parallel mode by itself. - Pass kind records whether the review is initial or a remediation-delta targeted pass. `--targeted ` requires `--initial-result ` and is not a tier alias. +The generic consumer policy keeps risk-based coverage: initial passes require +critic and QA plus signal-specific dimensions. The maintainer `/ship` initial +pass fixes coverage at all five reviewer dimensions while preserving the +independently resolved tier and mode. Targeted passes under either consumer +remain scoped to the requested remediation reviewers. + +Any requested tier, coverage, or mode below the policy floor fails before +reviewer dispatch. A downgrade is accepted only through an explicitly supplied +`gate_policy_override_v1` JSON file bound to the exact scope fingerprint and +recording user approval. The fingerprint includes the content-addressed tracked +patch and every in-scope untracked file, not only file names or aggregate line +counts. The free-form `.gate-overrides.md` file remains reviewer +finding/suppression context; it is recorded separately and cannot authorize a +policy downgrade. + The portable policy sources are [`core/policy/gate-tiers.tsv`](../core/policy/gate-tiers.tsv), [`core/policy/gate-modes.tsv`](../core/policy/gate-modes.tsv), and -[`core/policy/gate-pass-kinds.tsv`](../core/policy/gate-pass-kinds.tsv). +[`core/policy/gate-pass-kinds.tsv`](../core/policy/gate-pass-kinds.tsv), plus +the consumer and risk-signal tables +[`core/policy/gate-policy-consumers.tsv`](../core/policy/gate-policy-consumers.tsv) +and +[`core/policy/gate-policy-signals.tsv`](../core/policy/gate-policy-signals.tsv). +Changes under `core/policy/` are themselves a full-tier signal with +architecture, security, and risk coverage, so the governance tables cannot +quietly lower their own future review floor through a small edit. The final producer writes `pr_gate_result_v2` Markdown plus a sibling `gate_assurance_v2` JSON envelope. The Markdown contains human findings and a bounded relative `gate_assurance` pointer; the shell-owned envelope records requested/resolved coordinates, selected/skipped coverage, actual dispatch outcomes, run IDs, subject commits/fingerprint, and the evidence status behind -independence claims. Repo-layout results with verified independence also carry +independence claims. New envelopes also embed the canonical policy result: +classification facts, every matched signal and path, minimum tier, required +coverage, recommended versus required mode, enforcement status, and both +policy-override and reviewer-override provenance. Repo-layout results with +verified independence also carry a shell-owned attestation in the protected gate run directory. `pmctl gate verify` validates result/sidecar digests and resolves every claimed run ID against the canonical terminal records for the invoking repository and rejects @@ -188,7 +223,9 @@ published afterward; verification uses a bounded retry when it observes that in-flight canonical v2 finalization. Legacy `pr_gate_result_v1` and unbound `gate_assurance_v1` artifacts remain structurally readable, but verification reports `assurance: unavailable`; -consumers must not infer mode, coverage, or independence from them. +consumers must not infer mode, coverage, or independence from them. Earlier +`gate_assurance_v2` envelopes without the optional policy block remain +readable, while current producers always emit it. --- diff --git a/ops/setup/setup-project.sh b/ops/setup/setup-project.sh index facbaf2b..d282b943 100755 --- a/ops/setup/setup-project.sh +++ b/ops/setup/setup-project.sh @@ -28,7 +28,13 @@ done PROJECT_DIR="$(cd "$PROJECT_DIR" && pwd)" HEADER="# Claude agent / codex output — not for VCS or Docker" -ENTRIES=(".agent-trace/" ".gate-briefs/" ".gate-results/" ".agents/") +ENTRIES=( + ".agent-trace/" + ".gate-briefs/" + ".gate-results/" + ".agents/" + ".pm-dispatch-state/" +) # ── helpers ────────────────────────────────────────────────────────────────── diff --git a/runtime/bin/gate-supervisor.sh b/runtime/bin/gate-supervisor.sh index feb0eafb..2ac4e920 100755 --- a/runtime/bin/gate-supervisor.sh +++ b/runtime/bin/gate-supervisor.sh @@ -122,16 +122,25 @@ _rc=0 "$REPO_ROOT/runtime/bin/pr-gate.sh" --run-dir "$run_dir" --cd "$cd_arg" ${native[@]+"${native[@]}"} \ > "$_log" 2>&1 || _rc=$? -# pr-gate.sh prints `result: ` on completion (runtime/bin/pr-gate.sh:1493, -# both the GO and integrity-checked NO-GO paths); extract it for the sentinel -# so `pmctl gate wait` can surface it without re-deriving OUTPUT_FILE naming. +# pr-gate.sh prints `result: ` on both the GO and integrity-checked +# NO-GO paths; extract it for the sentinel so `pmctl gate wait` can surface it +# without re-deriving OUTPUT_FILE naming. _result_file="$(grep -m1 '^result: ' "$_log" 2>/dev/null | sed 's/^result: //')" || _result_file="" -case "$_rc" in - 0) _state="GO" ;; - 1) _state="NO-GO" ;; - *) _state="failed" ;; -esac +_terminal_rc="$_rc" +if [[ -z "$_result_file" && ( "$_rc" -eq 0 || "$_rc" -eq 1 ) ]]; then + # Exit 0/1 is only a GO/NO-GO verdict after pr-gate publishes the verified + # `result:` handoff. Without it, finalization or result publication failed; + # never encode that infrastructure failure as a verdict in the sentinel. + _state="failed" + _terminal_rc=2 +else + case "$_rc" in + 0) _state="GO" ;; + 1) _state="NO-GO" ;; + *) _state="failed" ;; + esac +fi -_write_sentinel "$_state" "$_rc" "$_result_file" -exit "$_rc" +_write_sentinel "$_state" "$_terminal_rc" "$_result_file" +exit "$_terminal_rc" diff --git a/runtime/bin/pr-gate.sh b/runtime/bin/pr-gate.sh index 56989d33..b242413f 100755 --- a/runtime/bin/pr-gate.sh +++ b/runtime/bin/pr-gate.sh @@ -56,6 +56,42 @@ targeted remediation-delta true false GATE_ASSURANCE_PASS_KINDS_TSV # END GENERATED from core/policy/gate-pass-kinds.tsv ;; + consumers) + # BEGIN GENERATED from core/policy/gate-policy-consumers.tsv + cat <<'GATE_POLICY_CONSUMERS_TSV' +# Gate policy consumers. Consumer policy does not rewrite tier, mode, or pass semantics. +policy_pass policy pass_kind minimum_tier required_reviewers recommended_mode required_mode +generic:initial generic initial express critic,qa-tester sequential none +generic:targeted generic targeted express none sequential none +maintainer:initial maintainer initial express critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +maintainer:targeted maintainer targeted express none parallel none +GATE_POLICY_CONSUMERS_TSV + # END GENERATED from core/policy/gate-policy-consumers.tsv + ;; + signals) + # BEGIN GENERATED from core/policy/gate-policy-signals.tsv + cat <<'GATE_POLICY_SIGNALS_TSV' +# Gate policy signals. Reviewer requirements apply to initial discovery; targeted passes retain tier/mode signals but use requested remediation coverage. +signal match_source pattern minimum_tier required_reviewers recommended_mode required_mode +docs-only classification docs-only express none sequential none +bounded-runtime classification bounded-runtime express none sequential none +medium-change classification medium-change standard architecture-reviewer parallel none +large-change classification large-change full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +binary-change classification binary-change standard architecture-reviewer parallel none +renamed-input classification renamed express none sequential none +untracked-input classification untracked express none sequential none +generated-input classification generated express none sequential none +cross-boundary classification cross-boundary standard architecture-reviewer parallel none +security-sensitive-path path-regex (^|[/_.-])(auth|oauth|jwt|sessions?|secrets?|passwords?|tokens?|credentials?|cors|csrf|webhooks?|sudo|ssh|payments?|billing)([/_.-]|$) express security-reviewer parallel none +input-execution-path path-regex (^|[/_.-])(eval|exec|execute|command|shell|hook|guard|allowlist)([/_.-]|$)|(^|/)(\.github|workflows?|ci)(/|$) standard security-reviewer parallel parallel +risk-sensitive-path path-regex (^|[/_.-])(migrations?|migrate|destructive|deletions?|delete|removals?|remove|rollback|concurrency|concurrent|race|locks?|cancel|reconcile)([/_.-]|$) express risk-reviewer parallel none +public-contract-path path-regex (^|/)(cli|commands|skills|core/schema)(/|$)|(^|[/_.-])(apis?|schemas?|contracts?)([/_.-]|$) standard architecture-reviewer parallel none +policy-source-path path-regex (^|/)core/policy(/|$) full architecture-reviewer,security-reviewer,risk-reviewer parallel none +brief-architecture-minor brief-value minor standard architecture-reviewer parallel none +brief-architecture-major brief-value major full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +GATE_POLICY_SIGNALS_TSV + # END GENERATED from core/policy/gate-policy-signals.tsv + ;; *) printf 'pr-gate: unknown assurance policy table: %s\n' "${1:-empty}" >&2 return 2 @@ -68,6 +104,8 @@ _gate_assurance_policy_filename() { tiers) printf 'gate-tiers.tsv\n' ;; modes) printf 'gate-modes.tsv\n' ;; pass-kinds) printf 'gate-pass-kinds.tsv\n' ;; + consumers) printf 'gate-policy-consumers.tsv\n' ;; + signals) printf 'gate-policy-signals.tsv\n' ;; *) return 2 ;; esac } @@ -163,6 +201,740 @@ _gate_assurance_policy_values() { ' } +_gate_policy_tier_rank() { + case "${1:-}" in + express) printf '1\n' ;; + standard) printf '2\n' ;; + full) printf '3\n' ;; + *) return 2 ;; + esac +} + +_gate_policy_order_reviewers() { + local selected="${1:-}" vocabulary="${2:-}" reviewer ordered="" + for reviewer in $vocabulary; do + if [[ " $selected " == *" $reviewer "* ]]; then + ordered="${ordered:+$ordered }$reviewer" + fi + done + printf '%s\n' "$ordered" +} + +_gate_policy_add_reviewers() { + local selected="${1:-}" csv="${2:-}" vocabulary="${3:-}" reviewer + [[ "$csv" != none ]] || { + _gate_policy_order_reviewers "$selected" "$vocabulary" + return + } + for reviewer in $(printf '%s' "$csv" | tr ',' ' '); do + if [[ " $vocabulary " != *" $reviewer "* ]]; then + printf 'Error: gate policy names unknown reviewer %s (allowed: %s)\n' \ + "$reviewer" "$vocabulary" >&2 + return 2 + fi + if [[ " $selected " != *" $reviewer "* ]]; then + selected="${selected:+$selected }$reviewer" + fi + done + _gate_policy_order_reviewers "$selected" "$vocabulary" +} + +_gate_policy_words_json() { + jq -nc --arg words "${1:-}" '$words | split(" ") | map(select(length > 0))' +} + +_gate_policy_lines_json() { + jq -Rsc 'split("\n") | map(select(length > 0))' +} + +# Validate the complete policy sources before resolving any one consumer or +# signal. Looking up only the rows that happen to match this invocation can +# leave a dormant typo or duplicate signal undiscovered until final artifact +# verification, after reviewer dispatch has already spent work. +_gate_policy_source_shape_validate() { + local table="${1:-}" expected_header="${2:-}" + [[ $# -eq 2 ]] || return 2 + if ! _gate_assurance_policy_emit "$table" | awk -F '\t' \ + -v expected_header="$expected_header" ' + /^[[:space:]]*#/ || /^[[:space:]]*$/ { next } + !header_seen { + sub(/\r$/, "") + if ($0 != expected_header) exit 2 + header_seen=1 + width=NF + next + } + { + sub(/\r$/, "", $NF) + if (NF != width) exit 2 + for (i=1; i<=NF; i++) if ($i == "") exit 2 + if (seen[$1]++) exit 2 + rows++ + } + END { + if (!header_seen || rows == 0) exit 2 + } + '; then + printf 'Error: invalid gate policy %s source (header, row width, non-empty cells, and unique IDs are required)\n' \ + "$table" >&2 + return 2 + fi +} + +_gate_policy_validate_reviewer_csv() { + local csv="${1:-}" vocabulary="${2:-}" source_label="${3:-policy}" + local reviewer seen="" + [[ "$csv" != none ]] || return 0 + if [[ -z "$csv" || "$csv" == ,* || "$csv" == *, || "$csv" == *,,* ]]; then + printf 'Error: gate policy %s has an invalid reviewer list: %s\n' \ + "$source_label" "$csv" >&2 + return 2 + fi + for reviewer in $(printf '%s' "$csv" | tr ',' ' '); do + if [[ ! "$reviewer" =~ ^[a-z0-9][a-z0-9-]*$ \ + || " $vocabulary " != *" $reviewer "* ]]; then + printf 'Error: gate policy %s names unknown reviewer %s (allowed: %s)\n' \ + "$source_label" "$reviewer" "$vocabulary" >&2 + return 2 + fi + if [[ " $seen " == *" $reviewer "* ]]; then + printf 'Error: gate policy %s repeats reviewer %s\n' \ + "$source_label" "$reviewer" >&2 + return 2 + fi + seen="${seen:+$seen }$reviewer" + done +} + +_gate_policy_validate_sources() { + local vocabulary="${1:-}" policy_pass policy pass_kind minimum_tier + local required_reviewers recommended_mode required_mode consumer_keys="" + local signal match_source pattern signal_tier signal_reviewers + local signal_recommended signal_required grep_status + [[ $# -eq 1 && -n "$vocabulary" ]] || return 2 + + _gate_policy_source_shape_validate consumers \ + $'policy_pass\tpolicy\tpass_kind\tminimum_tier\trequired_reviewers\trecommended_mode\trequired_mode' \ + || return 2 + _gate_policy_source_shape_validate signals \ + $'signal\tmatch_source\tpattern\tminimum_tier\trequired_reviewers\trecommended_mode\trequired_mode' \ + || return 2 + + while IFS=$'\t' read -r policy_pass policy pass_kind minimum_tier \ + required_reviewers recommended_mode required_mode; do + [[ -n "$policy_pass" && "$policy_pass" != \#* \ + && "$policy_pass" != policy_pass ]] || continue + case "$policy" in generic|maintainer) ;; *) + printf 'Error: gate policy consumer %s has invalid policy: %s\n' \ + "$policy_pass" "$policy" >&2 + return 2 + ;; + esac + case "$pass_kind" in initial|targeted) ;; *) + printf 'Error: gate policy consumer %s has invalid pass kind: %s\n' \ + "$policy_pass" "$pass_kind" >&2 + return 2 + ;; + esac + if [[ "$policy_pass" != "${policy}:${pass_kind}" ]]; then + printf 'Error: gate policy consumer key %s does not match %s:%s\n' \ + "$policy_pass" "$policy" "$pass_kind" >&2 + return 2 + fi + _gate_assurance_policy_lookup tiers tier "$minimum_tier" evidence_floor >/dev/null \ + || { + printf 'Error: gate policy consumer %s has invalid minimum tier: %s\n' \ + "$policy_pass" "$minimum_tier" >&2 + return 2 + } + _gate_policy_validate_reviewer_csv "$required_reviewers" "$vocabulary" \ + "consumer $policy_pass" || return 2 + _gate_assurance_policy_lookup modes mode "$recommended_mode" topology >/dev/null \ + || { + printf 'Error: gate policy consumer %s has invalid recommended mode: %s\n' \ + "$policy_pass" "$recommended_mode" >&2 + return 2 + } + if [[ "$required_mode" != none ]]; then + _gate_assurance_policy_lookup modes mode "$required_mode" topology >/dev/null \ + || { + printf 'Error: gate policy consumer %s has invalid required mode: %s\n' \ + "$policy_pass" "$required_mode" >&2 + return 2 + } + fi + consumer_keys="${consumer_keys:+$consumer_keys }$policy_pass" + done < <(_gate_assurance_policy_emit consumers) + + for policy_pass in generic:initial generic:targeted \ + maintainer:initial maintainer:targeted; do + if [[ " $consumer_keys " != *" $policy_pass "* ]]; then + printf 'Error: gate policy consumers source is missing %s\n' \ + "$policy_pass" >&2 + return 2 + fi + done + if [[ "$(printf '%s\n' "$consumer_keys" | awk '{print NF}')" -ne 4 ]]; then + printf 'Error: gate policy consumers source contains unsupported rows: %s\n' \ + "$consumer_keys" >&2 + return 2 + fi + + while IFS=$'\t' read -r signal match_source pattern signal_tier \ + signal_reviewers signal_recommended signal_required; do + [[ -n "$signal" && "$signal" != \#* && "$signal" != signal ]] || continue + if [[ ! "$signal" =~ ^[a-z0-9][a-z0-9-]*$ \ + || "$signal" == consumer-policy ]]; then + printf 'Error: gate policy signal has invalid or reserved ID: %s\n' \ + "$signal" >&2 + return 2 + fi + case "$match_source" in + classification) + case "$pattern" in + docs-only|bounded-runtime|medium-change|large-change|binary-change|\ + renamed|untracked|generated|cross-boundary) ;; + *) + printf 'Error: gate policy signal %s names unknown classification: %s\n' \ + "$signal" "$pattern" >&2 + return 2 + ;; + esac + ;; + path-regex) + grep_status=0 + LC_ALL=C grep -E -- "$pattern" /dev/null 2>&1 \ + || grep_status=$? + if [[ "$grep_status" -gt 1 ]]; then + printf 'Error: gate policy signal %s has invalid path regex: %s\n' \ + "$signal" "$pattern" >&2 + return 2 + fi + ;; + brief-value) + case "$pattern" in unknown|none|minor|major) ;; + *) + printf 'Error: gate policy signal %s has invalid brief value: %s\n' \ + "$signal" "$pattern" >&2 + return 2 + ;; + esac + ;; + *) + printf 'Error: gate policy signal %s has invalid match source: %s\n' \ + "$signal" "$match_source" >&2 + return 2 + ;; + esac + _gate_assurance_policy_lookup tiers tier "$signal_tier" evidence_floor >/dev/null \ + || { + printf 'Error: gate policy signal %s has invalid minimum tier: %s\n' \ + "$signal" "$signal_tier" >&2 + return 2 + } + _gate_policy_validate_reviewer_csv "$signal_reviewers" "$vocabulary" \ + "signal $signal" || return 2 + _gate_assurance_policy_lookup modes mode "$signal_recommended" topology >/dev/null \ + || { + printf 'Error: gate policy signal %s has invalid recommended mode: %s\n' \ + "$signal" "$signal_recommended" >&2 + return 2 + } + if [[ "$signal_required" != none ]]; then + _gate_assurance_policy_lookup modes mode "$signal_required" topology >/dev/null \ + || { + printf 'Error: gate policy signal %s has invalid required mode: %s\n' \ + "$signal" "$signal_required" >&2 + return 2 + } + fi + done < <(_gate_assurance_policy_emit signals) +} + +_gate_sha256_stream() { + if command -v sha256sum >/dev/null 2>&1 \ + && printf '' | sha256sum >/dev/null 2>&1; then + sha256sum | awk '{print $1}' + return + fi + if command -v shasum >/dev/null 2>&1 \ + && printf '' | shasum -a 256 >/dev/null 2>&1; then + shasum -a 256 | awk '{print $1}' + return + fi + printf 'Error: no sha256sum or shasum found -- cannot fingerprint gate inputs.\n' >&2 + return 2 +} + +# Emit a deterministic, content-addressed representation of the exact diff +# covered by policy resolution. The outer scope fingerprint also binds the +# requested policy/pass/brief coordinates; this digest prevents an approved +# downgrade from being replayed against a shape-identical but content-different +# patch. Working-tree scopes additionally bind every non-ignored untracked +# file's path, kind, executable bit, and content (or symlink target). +_gate_policy_scope_content_digest() { + local diff_kind="${1:-}" base="${2:-}" head_ref="${3:-HEAD}" + local include_untracked="${4:-false}" path quoted kind executable digest + { + printf 'gate-policy-scope-content-v1\0' + case "$diff_kind" in + fixed-head) + git diff --binary --full-index "$base"..."$head_ref" -- + ;; + allow-dirty) + git diff --binary --full-index "$base" -- + ;; + committed) + git diff --binary --full-index "$base"...HEAD -- + ;; + working-tree) + git diff --binary --full-index HEAD -- + ;; + *) + printf 'Error: unknown gate policy diff kind: %s\n' "$diff_kind" >&2 + return 2 + ;; + esac || return 2 + + if [[ "$include_untracked" == true ]]; then + while IFS= read -r -d '' path; do + quoted="$(printf '%q' "$path")" + if [[ -L "$WORK_DIR/$path" ]]; then + kind=symlink + executable=false + digest="$(printf '%s' "$(readlink "$WORK_DIR/$path")" \ + | _gate_sha256_stream)" || return 2 + elif [[ -f "$WORK_DIR/$path" ]]; then + kind=file + [[ -x "$WORK_DIR/$path" ]] && executable=true || executable=false + digest="$(_gate_result_sha256_file "$WORK_DIR/$path")" || return 2 + else + printf 'Error: unsupported untracked gate policy input: %s\n' \ + "$path" >&2 + return 2 + fi + printf 'untracked\0path=%s\0kind=%s\0executable=%s\0sha256=%s\0' \ + "$quoted" "$kind" "$executable" "$digest" + done < <(git ls-files --others --exclude-standard -z) + fi + } | _gate_sha256_stream +} + +# Resolve risk/policy once for every gate consumer. The function owns the +# relationship between change signals and assurance coordinates; caller paths +# consume its JSON result instead of copying path regexes or policy floors. +# +# _gate_policy_resolve [structured-policy-override] +_gate_policy_resolve() { + local input_json="${1:-}" policy_override="${2:-}" + local policy pass_kind policy_pass scope_fingerprint vocabulary + local minimum_tier required_reviewers recommended_mode required_mode + local signal match_source pattern signal_tier signal_reviewers + local normalized_signal_reviewers effective_signal_reviewers + local signal_recommended signal_required matches_json matches_text + local current_rank candidate_rank signal_json signals_file + local requested_tier requested_mode requested_reviewers_json + local resolved_tier resolved_mode tier_defaults selected_reviewers + local missing_reviewers="" reviewer tier_violation=false mode_violation=false + local violations_json downgrade_requested=false downgrade_allowed=false + local enforcement_status=pass override_status=not_provided + local override_sha="" override_reason="" override_approver_json=null + local expected_tier=null expected_mode=null missing_json override_json=null + local reviewer_override_json classification_json policy_source + + [[ $# -ge 1 && $# -le 2 ]] || return 2 + jq -e . >/dev/null 2>&1 <<<"$input_json" || { + printf 'Error: invalid gate policy resolver input\n' >&2 + return 2 + } + + policy="$(jq -r '.policy' <<<"$input_json")" + pass_kind="$(jq -r '.requested.pass_kind' <<<"$input_json")" + policy_pass="${policy}:${pass_kind}" + scope_fingerprint="$(jq -r '.scope_fingerprint' <<<"$input_json")" + vocabulary="$(jq -r '.reviewer_vocabulary | join(" ")' <<<"$input_json")" + requested_tier="$(jq -r '.requested.tier' <<<"$input_json")" + requested_mode="$(jq -r '.requested.mode' <<<"$input_json")" + requested_reviewers_json="$(jq -c '.requested.reviewers' <<<"$input_json")" + reviewer_override_json="$(jq -c '.reviewer_override' <<<"$input_json")" + classification_json="$(jq -c '.classification' <<<"$input_json")" + policy_source="$(jq -r '.policy_source' <<<"$input_json")" + + case "$policy" in generic|maintainer) ;; *) + printf 'Error: --policy must be generic or maintainer (got: %s)\n' "$policy" >&2 + return 2 + esac + [[ "$scope_fingerprint" =~ ^[a-f0-9]{64}$ ]] || { + printf 'Error: gate policy scope fingerprint is invalid\n' >&2 + return 2 + } + [[ -n "$vocabulary" ]] || { + printf 'Error: gate policy reviewer vocabulary is empty\n' >&2 + return 2 + } + + minimum_tier="$(_gate_assurance_policy_lookup consumers policy_pass "$policy_pass" minimum_tier)" \ + || { + printf 'Error: gate policy consumer source has no unique row for %s\n' \ + "$policy_pass" >&2 + return 2 + } + required_reviewers="$(_gate_assurance_policy_lookup consumers policy_pass "$policy_pass" required_reviewers)" \ + || return 2 + recommended_mode="$(_gate_assurance_policy_lookup consumers policy_pass "$policy_pass" recommended_mode)" \ + || return 2 + required_mode="$(_gate_assurance_policy_lookup consumers policy_pass "$policy_pass" required_mode)" \ + || return 2 + required_reviewers="$(_gate_policy_add_reviewers "" "$required_reviewers" "$vocabulary")" \ + || return 2 + + _gate_assurance_policy_lookup tiers tier "$minimum_tier" evidence_floor >/dev/null \ + || { + printf 'Error: gate policy consumer has invalid minimum tier: %s\n' \ + "$minimum_tier" >&2 + return 2 + } + _gate_assurance_policy_lookup modes mode "$recommended_mode" topology >/dev/null \ + || { + printf 'Error: gate policy consumer has invalid recommended mode: %s\n' \ + "$recommended_mode" >&2 + return 2 + } + if [[ "$required_mode" != none ]]; then + _gate_assurance_policy_lookup modes mode "$required_mode" topology >/dev/null \ + || { + printf 'Error: gate policy consumer has invalid required mode: %s\n' \ + "$required_mode" >&2 + return 2 + } + fi + + signals_file="$(mktemp "${TMPDIR:-/tmp}/gate-policy-signals.XXXXXX")" || return 2 + signal_json="$(jq -nc \ + --arg id "consumer-policy" --arg source "consumer-policy" \ + --arg match "$policy_pass" --arg minimum_tier "$minimum_tier" \ + --argjson required_reviewers "$(_gate_policy_words_json "$required_reviewers")" \ + --arg recommended_mode "$recommended_mode" --arg required_mode "$required_mode" '{ + id:$id,source:$source,matches:[$match],minimum_tier:$minimum_tier, + required_reviewers:$required_reviewers,recommended_mode:$recommended_mode, + required_mode:(if $required_mode == "none" then null else $required_mode end) + }')" || { + rm -f "$signals_file" + return 2 + } + printf '%s\n' "$signal_json" > "$signals_file" + + while IFS=$'\t' read -r signal match_source pattern signal_tier \ + signal_reviewers signal_recommended signal_required; do + [[ -n "$signal" && "$signal" != \#* && "$signal" != signal ]] || continue + if [[ -z "$match_source" || -z "$pattern" || -z "$signal_tier" \ + || -z "$signal_reviewers" || -z "$signal_recommended" \ + || -z "$signal_required" ]]; then + printf 'Error: malformed gate policy signal row: %s\n' "$signal" >&2 + rm -f "$signals_file" + return 2 + fi + + matches_json='[]' + case "$match_source" in + classification) + matches_json="$(jq -c --arg pattern "$pattern" \ + '[.classifications[] | select(.id == $pattern) | .matches[]]' \ + <<<"$input_json")" || { + rm -f "$signals_file" + return 2 + } + ;; + path-regex) + matches_text="$(jq -r '.changed_paths[]' <<<"$input_json" \ + | { grep -iE -- "$pattern" || true; })" + matches_json="$(printf '%s\n' "$matches_text" | _gate_policy_lines_json)" \ + || { + rm -f "$signals_file" + return 2 + } + ;; + brief-value) + if [[ "$(jq -r '.classification.architecture_impact' <<<"$input_json")" == "$pattern" ]]; then + matches_json="$(jq -nc --arg value "$pattern" '[$value]')" + fi + ;; + *) + printf 'Error: unsupported gate policy match source: %s\n' "$match_source" >&2 + rm -f "$signals_file" + return 2 + ;; + esac + [[ "$(jq -r 'length' <<<"$matches_json")" -gt 0 ]] || continue + + current_rank="$(_gate_policy_tier_rank "$minimum_tier")" || { + rm -f "$signals_file" + return 2 + } + candidate_rank="$(_gate_policy_tier_rank "$signal_tier")" || { + printf 'Error: gate policy signal %s has invalid minimum tier: %s\n' \ + "$signal" "$signal_tier" >&2 + rm -f "$signals_file" + return 2 + } + if (( candidate_rank > current_rank )); then + minimum_tier="$signal_tier" + fi + normalized_signal_reviewers="$(_gate_policy_add_reviewers "" \ + "$signal_reviewers" "$vocabulary")" || { + rm -f "$signals_file" + return 2 + } + effective_signal_reviewers="$normalized_signal_reviewers" + # A targeted pass is an explicit remediation-delta confirmation, not a + # second comprehensive discovery pass. Preserve every matched risk signal + # and its tier/mode implications, but let the requested targeted reviewers + # own coverage. Initial-pass evidence/closure is verified by its consumer. + if [[ "$pass_kind" == targeted ]]; then + effective_signal_reviewers="" + fi + required_reviewers="$(_gate_policy_add_reviewers "$required_reviewers" \ + "$(printf '%s' "$effective_signal_reviewers" | tr ' ' ',')" \ + "$vocabulary")" || { + rm -f "$signals_file" + return 2 + } + case "$signal_recommended" in + parallel) recommended_mode=parallel ;; + sequential) : ;; + *) + printf 'Error: gate policy signal %s has invalid recommended mode: %s\n' \ + "$signal" "$signal_recommended" >&2 + rm -f "$signals_file" + return 2 + ;; + esac + if [[ "$signal_required" != none ]]; then + if [[ "$required_mode" != none && "$required_mode" != "$signal_required" ]]; then + printf 'Error: conflicting required gate modes: %s and %s\n' \ + "$required_mode" "$signal_required" >&2 + rm -f "$signals_file" + return 2 + fi + required_mode="$signal_required" + fi + signal_json="$(jq -nc \ + --arg id "$signal" --arg source "$match_source" \ + --argjson matches "$matches_json" --arg minimum_tier "$signal_tier" \ + --argjson required_reviewers \ + "$(_gate_policy_words_json "$effective_signal_reviewers")" \ + --arg recommended_mode "$signal_recommended" \ + --arg required_mode "$signal_required" '{ + id:$id,source:$source,matches:$matches,minimum_tier:$minimum_tier, + required_reviewers:$required_reviewers,recommended_mode:$recommended_mode, + required_mode:(if $required_mode == "none" then null else $required_mode end) + }')" || { + rm -f "$signals_file" + return 2 + } + printf '%s\n' "$signal_json" >> "$signals_file" + done < <(_gate_assurance_policy_emit signals) + + required_reviewers="$(_gate_policy_order_reviewers "$required_reviewers" "$vocabulary")" + if [[ "$requested_tier" == auto ]]; then + resolved_tier="$minimum_tier" + else + resolved_tier="$requested_tier" + current_rank="$(_gate_policy_tier_rank "$minimum_tier")" || { + rm -f "$signals_file" + return 2 + } + candidate_rank="$(_gate_policy_tier_rank "$resolved_tier")" || { + rm -f "$signals_file" + return 2 + } + if (( candidate_rank < current_rank )); then + tier_violation=true + fi + fi + + if [[ "$requested_mode" == default ]]; then + if [[ "$required_mode" == none ]]; then + resolved_mode="$GATE_MODE_DEFAULT" + else + resolved_mode="$required_mode" + fi + else + resolved_mode="$requested_mode" + if [[ "$required_mode" != none && "$resolved_mode" != "$required_mode" ]]; then + mode_violation=true + fi + fi + + if [[ "$requested_reviewers_json" == null ]]; then + tier_defaults="$(_gate_assurance_policy_lookup tiers tier "$resolved_tier" default_reviewers)" \ + || { + rm -f "$signals_file" + return 2 + } + selected_reviewers="$(_gate_policy_add_reviewers "" "$tier_defaults" "$vocabulary")" \ + || { + rm -f "$signals_file" + return 2 + } + selected_reviewers="$(_gate_policy_add_reviewers "$selected_reviewers" \ + "$(printf '%s' "$required_reviewers" | tr ' ' ',')" "$vocabulary")" || { + rm -f "$signals_file" + return 2 + } + else + selected_reviewers="$(jq -r 'join(" ")' <<<"$requested_reviewers_json")" + selected_reviewers="$(_gate_policy_order_reviewers "$selected_reviewers" "$vocabulary")" + fi + + for reviewer in $required_reviewers; do + if [[ " $selected_reviewers " != *" $reviewer "* ]]; then + missing_reviewers="${missing_reviewers:+$missing_reviewers }$reviewer" + fi + done + missing_json="$(_gate_policy_words_json "$missing_reviewers")" + violations_json="$(jq -nc \ + --argjson tier_violation "$tier_violation" \ + --arg requested_tier "$requested_tier" --arg minimum_tier "$minimum_tier" \ + --argjson missing_reviewers "$missing_json" \ + --argjson mode_violation "$mode_violation" \ + --arg requested_mode "$requested_mode" --arg required_mode "$required_mode" '[ + if $tier_violation then { + coordinate:"tier",requested:$requested_tier,required:$minimum_tier + } else empty end, + if ($missing_reviewers | length) > 0 then { + coordinate:"coverage",requested:"explicit",required:$missing_reviewers + } else empty end, + if $mode_violation then { + coordinate:"mode",requested:$requested_mode,required:$required_mode + } else empty end + ]')" + if [[ "$(jq -r 'length' <<<"$violations_json")" -gt 0 ]]; then + downgrade_requested=true + enforcement_status=fail + fi + + if [[ -n "$policy_override" ]]; then + if [[ ! -r "$policy_override" || ! -s "$policy_override" ]]; then + printf 'Error: --policy-override must name a readable, non-empty JSON file: %s\n' \ + "$policy_override" >&2 + rm -f "$signals_file" + return 2 + fi + override_sha="$(_gate_result_sha256_file "$policy_override")" || { + rm -f "$signals_file" + return 2 + } + if ! jq -e ' + (keys | sort) == + (["allow","approver","kind","reason","schema_version","scope_fingerprint"] | sort) and + .kind == "gate_policy_override_v1" and .schema_version == 1 and + (.scope_fingerprint | type == "string" and test("^[a-f0-9]{64}$")) and + (.allow | type == "object" and + (keys | sort) == (["mode","omit_reviewers","tier"] | sort)) and + (.allow.tier == null or (.allow.tier | IN("express","standard","full"))) and + (.allow.mode == null or (.allow.mode | IN("sequential","parallel"))) and + (.allow.omit_reviewers | type == "array" and + all(.[]; type == "string" and test("^[a-z0-9][a-z0-9-]*$")) and + length == (unique | length)) and + (.reason | type == "string" and length > 0) and + (.approver | type == "object" and + (keys | sort) == (["approval_ref","identity","kind"] | sort)) and + .approver.kind == "user" and + (.approver.identity | type == "string" and length > 0) and + (.approver.approval_ref | type == "string" and length > 0) + ' "$policy_override" >/dev/null 2>&1; then + printf 'Error: invalid gate policy override contract: %s\n' "$policy_override" >&2 + rm -f "$signals_file" + return 2 + fi + override_reason="$(jq -r '.reason' "$policy_override")" + override_approver_json="$(jq -c '.approver' "$policy_override")" + if [[ "$downgrade_requested" == false ]]; then + override_status=not_needed + elif [[ "$(jq -r '.scope_fingerprint' "$policy_override")" != "$scope_fingerprint" ]]; then + override_status=scope_mismatch + else + [[ "$tier_violation" == true ]] && expected_tier="$(jq -nc --arg value "$requested_tier" '$value')" + [[ "$mode_violation" == true ]] && expected_mode="$(jq -nc --arg value "$requested_mode" '$value')" + if jq -e --argjson expected_tier "$expected_tier" \ + --argjson expected_mode "$expected_mode" \ + --argjson expected_reviewers "$missing_json" ' + .allow.tier == $expected_tier and + .allow.mode == $expected_mode and + (.allow.omit_reviewers | sort) == ($expected_reviewers | sort) + ' "$policy_override" >/dev/null; then + override_status=applied + downgrade_allowed=true + enforcement_status=pass + else + override_status=allowance_mismatch + fi + fi + override_json="$(jq -nc --arg status "$override_status" \ + --arg source "$policy_override" --arg sha256 "$override_sha" \ + --arg reason "$override_reason" --argjson approver "$override_approver_json" '{ + status:$status,source:$source,sha256:$sha256,reason:$reason,approver:$approver + }')" + else + override_json='{"status":"not_provided","source":null,"sha256":null,"reason":null,"approver":null}' + fi + + signal_json="$(jq -s '.' "$signals_file")" || { + rm -f "$signals_file" + return 2 + } + rm -f "$signals_file" + + jq -nc \ + --arg policy "$policy" --arg policy_source "$policy_source" \ + --arg scope_fingerprint "$scope_fingerprint" \ + --arg requested_tier "$requested_tier" --arg requested_mode "$requested_mode" \ + --arg pass_kind "$pass_kind" --argjson requested_reviewers "$requested_reviewers_json" \ + --argjson classification "$classification_json" \ + --arg minimum_tier "$minimum_tier" \ + --argjson required_reviewers "$(_gate_policy_words_json "$required_reviewers")" \ + --arg recommended_mode "$recommended_mode" --arg required_mode "$required_mode" \ + --argjson downgrade_requested "$downgrade_requested" \ + --argjson downgrade_allowed "$downgrade_allowed" \ + --argjson matched_signals "$signal_json" \ + --arg resolved_tier "$resolved_tier" --arg resolved_mode "$resolved_mode" \ + --argjson selected_reviewers "$(_gate_policy_words_json "$selected_reviewers")" \ + --arg enforcement_status "$enforcement_status" \ + --argjson violations "$violations_json" --argjson override "$override_json" \ + --argjson reviewer_override "$reviewer_override_json" '{ + kind:"gate_policy_resolution_v1", + schema_version:1, + consumer_policy:$policy, + policy_source:$policy_source, + scope_fingerprint:$scope_fingerprint, + request:{ + tier:$requested_tier, + mode:$requested_mode, + pass_kind:$pass_kind, + reviewers:$requested_reviewers + }, + classification:$classification, + resolution:{ + minimum_tier:$minimum_tier, + required_reviewers:$required_reviewers, + recommended_mode:$recommended_mode, + required_mode:(if $required_mode == "none" then null else $required_mode end), + downgrade_requested:$downgrade_requested, + downgrade_allowed:$downgrade_allowed + }, + matched_signals:$matched_signals, + resolved:{ + tier:$resolved_tier, + mode:$resolved_mode, + reviewers:$selected_reviewers + }, + enforcement:{status:$enforcement_status,violations:$violations}, + override:$override, + reviewer_override:$reviewer_override + }' +} + _gate_set_mode_requested() { local candidate="$1" spelling="$2" if [[ "$MODE_OPTION_SEEN" == false ]]; then @@ -192,6 +964,48 @@ verify_reviewer_artifact_hashes() { done } +# _gate_reviewer_verdict_extract +# +# Parallel reviewer briefs already require a canonical heading. Base-pinned +# reviewer definitions additionally use a lower-case, narrative `verdict:` +# field, so requiring a second upper-case Verdict line creates two competing +# output contracts. Treat the unique, reviewer-matched heading as authoritative. +# An optional legacy `Verdict:` marker remains accepted only when it is itself +# unique, valid, and agrees with the heading. +_gate_reviewer_verdict_extract() { + local reviewer="${1:-}" artifact="${2:-}" + local heading_count valid_heading_count explicit_count valid_explicit_count + local heading_verdict explicit_verdict + [[ $# -eq 2 && -n "$reviewer" && -s "$artifact" ]] || return 1 + + heading_count="$(grep -cE "^## ${reviewer} -- " "$artifact" || true)" + valid_heading_count="$( + grep -cE "^## ${reviewer} -- (approve|advise|block-soft|block)$" \ + "$artifact" || true + )" + [[ "$heading_count" -eq 1 && "$valid_heading_count" -eq 1 ]] || return 1 + heading_verdict="$( + grep -oE "^## ${reviewer} -- (approve|advise|block-soft|block)$" \ + "$artifact" | awk '{print $4}' + )" + + explicit_count="$(grep -cE '^Verdict:' "$artifact" || true)" + if [[ "$explicit_count" -gt 0 ]]; then + valid_explicit_count="$( + grep -cE '^Verdict: (approve|advise|block-soft|block)([. ]|$)' \ + "$artifact" || true + )" + [[ "$explicit_count" -eq 1 && "$valid_explicit_count" -eq 1 ]] || return 1 + explicit_verdict="$( + grep -oE '^Verdict: (approve|advise|block-soft|block)([. ]|$)' \ + "$artifact" | awk '{print $2}' | tr -d '. ' + )" + [[ "$explicit_verdict" == "$heading_verdict" ]] || return 1 + fi + + printf '%s\n' "$heading_verdict" +} + # _kill_process_tree [signal] -- signal a process AND all its descendants. # A plain `kill ` only reaches the `eval` subshell / dispatch.sh wrapper we # backgrounded; the grandchild executor (`codex exec`, or a test `sleep` stub) @@ -233,7 +1047,8 @@ _kill_process_tree() { # --cd working directory (required) # --tier express|standard|full -- overrides auto-detection # --mode sequential|parallel execution topology (default: sequential) -# --brief dispatch brief for this change; architecture_impact field informs tier suggestion +# --brief dispatch brief; trusted architecture_impact contributes to policy resolution +# --policy generic|maintainer consumer policy (default: generic) # --reviewers comma-separated requested coverage; does not change tier or pass kind # --targeted remediation-delta pass over these reviewers; requires --initial-result # --initial-result initial gate result referenced by a --targeted pass; relative to --cd @@ -268,6 +1083,8 @@ _kill_process_tree() { # is resolved against the working dir (--cd), not the caller's CWD, since # the file is loaded after the gate cd's into the work dir. The loaded source # and content are recorded in the gate result (## Gate Overrides Applied). +# --policy-override explicit gate_policy_override_v1 JSON for a scope-bound policy +# downgrade. It is never auto-discovered and requires recorded user approval. # --test-cmd pre-flight test command run in plain bash BEFORE dispatch, independent of # --timeout; pass/fail is recorded mechanically (frontmatter test_suite: field) # and a FAIL forces Final: NO-GO regardless of what any reviewer LLM writes. @@ -299,6 +1116,8 @@ MODE_OPTION_SPELLING="" PASS_KIND_REQUESTED="initial" INITIAL_RESULT_INPUT="" INITIAL_RESULT_OPTION_SEEN=false +POLICY_CONSUMER="generic" +POLICY_OVERRIDE_FILE="" REVIEWER_DIR_OVERRIDE="" SCOPE="" BASE_OVERRIDE="" @@ -322,7 +1141,7 @@ DISPATCH_APPROVAL="never" # runtime/lib/reasoning-effort.sh). Validated inline (not sourced from the lib) # so copy-mode pr-gate.sh has no extra file dependency for this flag. DISPATCH_EFFORT="" -BRIEF_FILE="" +INPUT_BRIEF_FILE="" TEST_CMD_OVERRIDE="" # --test-cmd: explicit pre-flight test command (see CC-470 Part 3) TEST_TIMEOUT="1800" # --test-timeout: independent of --timeout (dispatch budget) SKIP_PREFLIGHT_TESTS=false @@ -342,7 +1161,16 @@ while [[ $# -gt 0 ]]; do [[ $# -ge 2 && -n "$2" && "$2" != --* ]] || { printf 'Error: --mode requires sequential or parallel\n' >&2; exit 2; } _gate_set_mode_requested "$2" "--mode" || exit 2 shift 2;; - --brief) BRIEF_FILE="$2"; shift 2;; + --brief) + [[ $# -ge 2 && -n "$2" && "$2" != --* ]] || { printf 'Error: --brief requires a file path\n' >&2; exit 2; } + INPUT_BRIEF_FILE="$2"; shift 2;; + --policy) + [[ $# -ge 2 && -n "$2" && "$2" != --* ]] || { printf 'Error: --policy requires generic or maintainer\n' >&2; exit 2; } + case "$2" in + generic|maintainer) POLICY_CONSUMER="$2" ;; + *) printf 'Error: --policy must be generic or maintainer (got: %s)\n' "$2" >&2; exit 2 ;; + esac + shift 2;; --reviewers) [[ $# -ge 2 && -n "$2" && "$2" != --* ]] || { printf 'Error: --reviewers requires a reviewer list\n' >&2; exit 2; } [[ -z "$REVIEWERS_OPTION_SOURCE" ]] || { @@ -401,6 +1229,9 @@ while [[ $# -gt 0 ]]; do # script's controlled CLI error style. [[ $# -ge 2 ]] || { printf 'Error: --override-file requires a file path\n' >&2; exit 2; } OVERRIDE_FILE="$2"; shift 2;; + --policy-override) + [[ $# -ge 2 && -n "$2" && "$2" != --* ]] || { printf 'Error: --policy-override requires a JSON file path\n' >&2; exit 2; } + POLICY_OVERRIDE_FILE="$2"; shift 2;; --test-cmd) [[ $# -ge 2 ]] || { printf 'Error: --test-cmd requires a shell command\n' >&2; exit 2; } TEST_CMD_OVERRIDE="$2"; shift 2;; @@ -421,7 +1252,7 @@ while [[ $# -gt 0 ]]; do exit 0;; *) printf 'Unknown arg: %s\n' "$1" >&2 - printf 'Accepted: --cd --run-dir --tier --mode --brief --reviewers --targeted --initial-result --reviewer-dir --scope --base --head --output --executor --model --effort --isolation --timeout --parallel --sequential --allow-hooks --allow-dirty --override-file --test-cmd --test-timeout --skip-preflight-tests (-h for help)\n' >&2 + printf 'Accepted: --cd --run-dir --tier --mode --brief --policy --reviewers --targeted --initial-result --reviewer-dir --scope --base --head --output --executor --model --effort --isolation --timeout --parallel --sequential --allow-hooks --allow-dirty --override-file --policy-override --test-cmd --test-timeout --skip-preflight-tests (-h for help)\n' >&2 exit 2;; esac done @@ -561,7 +1392,7 @@ else (length == (unique | length)); def same_set($a; $b): ($a | sort) == ($b | sort); only_keys(["kind","schema_version","result","bindings","coordinates", - "dispatch","provenance"]) and + "policy","dispatch","provenance"]) and (.result | only_keys(["final"])) and (.bindings | only_keys(["result_sha256","repo_root","repo_identity", "base_commit","head_commit","subject_fingerprint"])) and @@ -574,6 +1405,118 @@ else (.coordinates.independence | only_keys(["implementation_context_isolated","reviewer_topology", "per_reviewer_independent","evidence_status"])) and + (if has("policy") then + (.policy | + only_keys(["kind","schema_version","consumer_policy","policy_source", + "scope_fingerprint","request","classification","resolution", + "matched_signals","resolved","enforcement","override", + "reviewer_override"])) and + (.policy.request | + only_keys(["tier","mode","pass_kind","reviewers"])) and + (.policy.classification | + only_keys(["architecture_impact","line_changes", + "binary_or_unknown_count","layer_roots"])) and + (.policy.resolution | + only_keys(["minimum_tier","required_reviewers","recommended_mode", + "required_mode","downgrade_requested","downgrade_allowed"])) and + (.policy.resolved | only_keys(["tier","mode","reviewers"])) and + (.policy.enforcement | only_keys(["status","violations"])) and + (.policy.override | + only_keys(["status","source","sha256","reason","approver"])) and + (.policy.reviewer_override | + only_keys(["status","source","sha256"])) and + .policy.kind == "gate_policy_resolution_v1" and + .policy.schema_version == 1 and + (.policy.consumer_policy | IN("generic","maintainer")) and + .policy.policy_source == .provenance.policy_source and + (.policy.scope_fingerprint | test("^[a-f0-9]{64}$")) and + .policy.request.tier == .coordinates.tier.requested and + .policy.request.mode == .coordinates.mode.requested and + .policy.request.pass_kind == .coordinates.pass.resolved and + ((.policy.request.reviewers == null and + .coordinates.coverage.requested == null) or + (same_set(.policy.request.reviewers; + .coordinates.coverage.requested))) and + (.policy.classification.architecture_impact | + IN("unknown","none","minor","major")) and + (.policy.classification.line_changes | + type == "number" and . >= 0 and floor == .) and + (.policy.classification.binary_or_unknown_count | + type == "number" and . >= 0 and floor == .) and + (.policy.classification.layer_roots | strings_unique) and + (.policy.resolution.minimum_tier | + IN("express","standard","full")) and + (.policy.resolution.required_reviewers | strings_unique) and + (.policy as $policy | + all($policy.resolution.required_reviewers[]; + . as $reviewer | + ($policy.resolved.reviewers | index($reviewer)) != null or + $policy.resolution.downgrade_allowed)) and + (.policy.resolution.recommended_mode | + IN("sequential","parallel")) and + (.policy.resolution.required_mode == null or + (.policy.resolution.required_mode | + IN("sequential","parallel"))) and + (.policy.resolution.downgrade_requested | type == "boolean") and + (.policy.resolution.downgrade_allowed | type == "boolean") and + (.policy.matched_signals | type == "array" and length > 0) and + ([.policy.matched_signals[].id] | strings_unique) and + (all(.policy.matched_signals[]; + only_keys(["id","source","matches","minimum_tier", + "required_reviewers","recommended_mode","required_mode"]) and + (.id | type == "string" and length > 0) and + (.source | + IN("consumer-policy","classification","path-regex","brief-value")) and + (.matches | strings_unique and length > 0) and + (.minimum_tier | IN("express","standard","full")) and + (.required_reviewers | strings_unique) and + (.recommended_mode | IN("sequential","parallel")) and + (.required_mode == null or + (.required_mode | IN("sequential","parallel"))))) and + .policy.resolved.tier == .coordinates.tier.resolved and + .policy.resolved.mode == .coordinates.mode.resolved and + same_set(.policy.resolved.reviewers; + .coordinates.coverage.selected) and + .policy.enforcement.status == "pass" and + (.policy.enforcement.violations | type == "array") and + (all(.policy.enforcement.violations[]; + only_keys(["coordinate","requested","required"]) and + (.coordinate | IN("tier","coverage","mode")))) and + (.policy.override.status | + IN("not_provided","not_needed","applied","scope_mismatch", + "allowance_mismatch")) and + (if .policy.resolution.downgrade_requested + then + .policy.resolution.downgrade_allowed == true and + .policy.override.status == "applied" and + (.policy.override.source | + type == "string" and startswith("/")) and + (.policy.override.sha256 | test("^[a-f0-9]{64}$")) and + (.policy.override.reason | type == "string" and length > 0) and + (.policy.override.approver | + only_keys(["kind","identity","approval_ref"])) and + .policy.override.approver.kind == "user" and + (.policy.override.approver.identity | + type == "string" and length > 0) and + (.policy.override.approver.approval_ref | + type == "string" and length > 0) + else + .policy.resolution.downgrade_allowed == false and + (.policy.override.status | + IN("not_provided","not_needed")) + end) and + (.policy.reviewer_override.status | + IN("not_provided","provided")) and + (if .policy.reviewer_override.status == "provided" + then + (.policy.reviewer_override.source | + type == "string" and startswith("/")) and + (.policy.reviewer_override.sha256 | test("^[a-f0-9]{64}$")) + else + .policy.reviewer_override.source == null and + .policy.reviewer_override.sha256 == null + end) + else true end) and (.dispatch | only_keys(["outcomes"])) and (all(.dispatch.outcomes[]; only_keys(["role","reviewer","status","run_id","evidence_status"]))) and @@ -782,27 +1725,13 @@ if [[ -n "$TIER_OVERRIDE" ]] \ exit 2 fi -if [[ "$MODE_OPTION_SEEN" == false ]]; then - MODE_RESOLVED="$GATE_MODE_DEFAULT" -else - MODE_RESOLVED="$MODE_REQUESTED" -fi -if ! MODE_TOPOLOGY="$(_gate_assurance_policy_lookup modes mode "$MODE_RESOLVED" topology)" \ - || ! MODE_SYNTHESIS="$(_gate_assurance_policy_lookup modes mode "$MODE_RESOLVED" synthesis)"; then +if [[ "$MODE_OPTION_SEEN" == true ]] \ + && ! _gate_assurance_policy_lookup modes mode "$MODE_REQUESTED" topology >/dev/null; then printf 'Error: --mode must be one of: %s (got: %s)\n' \ "$(printf '%s\n' "$GATE_MODE_VALUES" | awk 'BEGIN{ORS=" "} {print} END{print "\n"}' | sed 's/[[:space:]]*$//')" \ - "$MODE_RESOLVED" >&2 + "$MODE_REQUESTED" >&2 exit 2 fi -case "$MODE_TOPOLOGY:$MODE_SYNTHESIS" in - combined-session:inline) SEQUENTIAL=true ;; - per-reviewer-sessions:separate-session) SEQUENTIAL=false ;; - *) - printf 'Error: unsupported gate mode topology for %s: %s + %s\n' \ - "$MODE_RESOLVED" "$MODE_TOPOLOGY" "$MODE_SYNTHESIS" >&2 - exit 2 - ;; -esac PASS_KIND_RESOLVED="$PASS_KIND_REQUESTED" if ! PASS_SCOPE="$(_gate_assurance_policy_lookup pass-kinds pass_kind "$PASS_KIND_RESOLVED" scope)" \ @@ -883,6 +1812,7 @@ if [[ -z "$ALL_REVIEWERS" ]]; then printf 'Error: gate tier policy did not declare any reviewers\n' >&2 exit 2 fi +_gate_policy_validate_sources "$ALL_REVIEWERS" || exit 2 _gate_normalize_reviewer_list() { local raw="$1" source_label="$2" normalized="" reviewer @@ -908,13 +1838,13 @@ _gate_normalize_reviewer_list() { } _gate_policy_source_count=0 -for _gate_policy_table in tiers modes pass-kinds; do +for _gate_policy_table in tiers modes pass-kinds consumers signals; do if _gate_assurance_policy_path "$_gate_policy_table" >/dev/null; then _gate_policy_source_count=$((_gate_policy_source_count + 1)) fi done case "$_gate_policy_source_count" in - 3) GATE_ASSURANCE_POLICY_SOURCE="canonical" ;; + 5) GATE_ASSURANCE_POLICY_SOURCE="canonical" ;; 0) GATE_ASSURANCE_POLICY_SOURCE="generated-snapshot" ;; *) GATE_ASSURANCE_POLICY_SOURCE="mixed" ;; esac @@ -1339,14 +2269,38 @@ if [[ -z "$OVERRIDE_FILE" && -f "$WORK_DIR/.gate-overrides.md" ]]; then say 'pr-gate: discovered override file: .gate-overrides.md\n' fi GATE_OVERRIDES_CONTENT="" +REVIEWER_OVERRIDE_PROVENANCE_JSON='{"status":"not_provided","source":null,"sha256":null}' if [[ -n "$OVERRIDE_FILE" ]]; then if [[ ! -f "$OVERRIDE_FILE" ]]; then printf 'Error: override file not found: %s\n' "$OVERRIDE_FILE" >&2 exit 2 fi + _override_parent="$(cd "$(dirname "$OVERRIDE_FILE")" && pwd -P)" || exit 2 + OVERRIDE_FILE="$_override_parent/$(basename "$OVERRIDE_FILE")" + unset _override_parent GATE_OVERRIDES_CONTENT=$(cat "$OVERRIDE_FILE") + _reviewer_override_sha="$(_gate_result_sha256_file "$OVERRIDE_FILE")" || exit 2 + REVIEWER_OVERRIDE_PROVENANCE_JSON="$(jq -nc \ + --arg source "$OVERRIDE_FILE" --arg sha256 "$_reviewer_override_sha" \ + '{status:"provided",source:$source,sha256:$sha256}')" + unset _reviewer_override_sha say 'pr-gate: override file loaded: %s (%d bytes)\n' "$OVERRIDE_FILE" "${#GATE_OVERRIDES_CONTENT}" fi +if [[ -n "$POLICY_OVERRIDE_FILE" ]]; then + _policy_override_candidate="$POLICY_OVERRIDE_FILE" + [[ "$_policy_override_candidate" == /* ]] \ + || _policy_override_candidate="$WORK_DIR/$_policy_override_candidate" + if [[ ! -f "$_policy_override_candidate" || ! -r "$_policy_override_candidate" \ + || ! -s "$_policy_override_candidate" || -L "$_policy_override_candidate" ]]; then + printf 'Error: --policy-override must name a readable, non-empty, regular non-symlink JSON file: %s\n' \ + "$POLICY_OVERRIDE_FILE" >&2 + exit 2 + fi + _policy_override_parent="$(cd "$(dirname "$_policy_override_candidate")" && pwd -P)" \ + || exit 2 + POLICY_OVERRIDE_FILE="$_policy_override_parent/$(basename "$_policy_override_candidate")" + unset _policy_override_candidate _policy_override_parent +fi # ── Detect base branch ──────────────────────────────────────────────────────── if [[ -n "$BASE_OVERRIDE" ]]; then @@ -1457,21 +2411,22 @@ if [[ "$HEAD_REF" == "HEAD" ]] && ! git diff "$BASE"...HEAD --quiet 2>/dev/null printf 'pr-gate: --allow-dirty set -- folding uncommitted working-tree changes into review scope\n' >&2 fi -# ── Collect diff ────────────────────────────────────────────────────────────── -# Use --name-status so renames expose BOTH old and new paths for sensitive matching. -# Use --numstat to detect binary files (shown as -\t-\t). +# ── Collect diff and policy inputs ──────────────────────────────────────────── +# Keep the status-bearing form until policy resolution so renamed and untracked +# inputs remain machine-visible. Use --numstat to detect binary files +# (shown as -\t-\t). +UNTRACKED_PATHS="" if [[ "$HEAD_REF" != "HEAD" ]]; then # Fixed head ref (e.g. tag-to-tag, or a branch reviewed before a PR exists) # -- no working tree involved, so no dirty/fallback branches apply. Three-dot # (merge-base) diff, matching the default HEAD path below: reviews what # changed on HEAD_REF since it diverged from BASE, not a literal two-dot # tree diff -- so BASE moving forward independently does not appear here. - DIFF_FILES=$(git diff "$BASE"..."$HEAD_REF" --name-status | awk ' - /^R/ { print $2; print $3; next } - /^[AMDCT]/ { print $2 } - ') + DIFF_NAME_STATUS="$(git diff "$BASE"..."$HEAD_REF" --name-status)" DIFF_STAT=$(git diff "$BASE"..."$HEAD_REF" --stat) BINARY_HIT=$(git diff "$BASE"..."$HEAD_REF" --numstat | { grep -c $'^-\t-\t' || true; }) + POLICY_DIFF_KIND="fixed-head" + POLICY_SCOPE_INCLUDE_UNTRACKED=false LINES=$(git diff "$BASE"..."$HEAD_REF" --numstat | awk ' /^-\t-\t/ { next } { s += $1 + $2 } @@ -1480,12 +2435,15 @@ if [[ "$HEAD_REF" != "HEAD" ]]; then elif [[ "$ALLOW_DIRTY" == true ]] && _worktree_is_dirty; then # --allow-dirty: fold the working tree into scope. Two-dot diff vs BASE # captures committed + uncommitted tracked changes; untracked listed separately. - DIFF_FILES=$( { git diff "$BASE" --name-status | awk ' - /^R/ { print $2; print $3; next } - /^[AMDCT]/ { print $2 } - '; git ls-files --others --exclude-standard; } ) + UNTRACKED_PATHS="$(git ls-files --others --exclude-standard)" + DIFF_NAME_STATUS="$( + git diff "$BASE" --name-status + printf '%s\n' "$UNTRACKED_PATHS" | awk 'NF { print "?\t" $0 }' + )" DIFF_STAT=$(git diff "$BASE" --stat) BINARY_HIT=$(git diff "$BASE" --numstat | { grep -c $'^-\t-\t' || true; }) + POLICY_DIFF_KIND="allow-dirty" + POLICY_SCOPE_INCLUDE_UNTRACKED=true LINES=$(git diff "$BASE" --numstat | awk ' /^-\t-\t/ { next } { s += $1 + $2 } @@ -1497,12 +2455,11 @@ elif [[ "$ALLOW_DIRTY" == true ]] && _worktree_is_dirty; then elif ! git diff "$BASE"...HEAD --quiet 2>/dev/null; then # For renames (R* status lines), emit both old and new path so sensitive # keywords in the old name (e.g. auth.ts → login.ts) are not lost. - DIFF_FILES=$(git diff "$BASE"...HEAD --name-status | awk ' - /^R/ { print $2; print $3; next } - /^[AMDCT]/ { print $2 } - ') + DIFF_NAME_STATUS="$(git diff "$BASE"...HEAD --name-status)" DIFF_STAT=$(git diff "$BASE"...HEAD --stat) BINARY_HIT=$(git diff "$BASE"...HEAD --numstat | { grep -c $'^-\t-\t' || true; }) + POLICY_DIFF_KIND="committed" + POLICY_SCOPE_INCLUDE_UNTRACKED=false LINES=$(git diff "$BASE"...HEAD --numstat | awk ' /^-\t-\t/ { next } { s += $1 + $2 } @@ -1510,9 +2467,15 @@ elif ! git diff "$BASE"...HEAD --quiet 2>/dev/null; then ') else # No branch commits -- fall back to working tree changes - DIFF_FILES=$(git diff HEAD --name-only; git ls-files --others --exclude-standard) + UNTRACKED_PATHS="$(git ls-files --others --exclude-standard)" + DIFF_NAME_STATUS="$( + git diff HEAD --name-status + printf '%s\n' "$UNTRACKED_PATHS" | awk 'NF { print "?\t" $0 }' + )" DIFF_STAT=$(git diff HEAD --stat) BINARY_HIT=$(git diff HEAD --numstat | { grep -c $'^-\t-\t' || true; }) + POLICY_DIFF_KIND="working-tree" + POLICY_SCOPE_INCLUDE_UNTRACKED=true LINES=$(git diff HEAD --numstat | awk ' /^-\t-\t/ { next } { s += $1 + $2 } @@ -1527,69 +2490,206 @@ else BINARY_HIT=$((BINARY_HIT + UNTRACKED_NONDOC)) fi +DIFF_FILES="$(printf '%s\n' "$DIFF_NAME_STATUS" | awk -F '\t' ' + $1 ~ /^[RC]/ { print $2; print $3; next } + $1 ~ /^[AMDCT?]/ { print $2 } +' | awk 'NF && !seen[$0]++')" if [[ -z "$DIFF_FILES" ]]; then printf 'Error: no changed files detected against %s\n' "$BASE" >&2; exit 1 fi -# ── Detect tier ─────────────────────────────────────────────────────────────── -if [[ -n "$TIER_OVERRIDE" ]]; then - TIER_RESOLVED="$TIER_OVERRIDE" -else - NON_DOCS=$(printf '%s\n' "$DIFF_FILES" | grep -vE '\.(md|jsonl|txt)$|^\.gitignore$|^audits/|^docs/' || true) - SENSITIVE_HIT=$(printf '%s\n' "$DIFF_FILES" | { grep -iE '(^|[/_.-])(auth|oauth|jwt|session|secret|password|token|credential|cors|csrf|webhook|sudo|ssh|payment|billing)([/_.-]|$)|(^|/)migrations?/|^\.github/' || true; } | wc -l) - - if [[ -z "$NON_DOCS" ]]; then - TIER_RESOLVED=express - elif [[ "$SENSITIVE_HIT" -gt 0 || "$LINES" -gt 500 ]]; then - TIER_RESOLVED=full - elif [[ "$LINES" -lt 100 && "${BINARY_HIT:-0}" -eq 0 ]]; then - # Binary files have no line count but represent real changes -- treat as standard+ - TIER_RESOLVED=express - else - TIER_RESOLVED=standard +# Preserve the complete status-derived policy inputs. Scope-manifest expansion +# remains a later concern; this resolver records only deterministic facts it +# owns and a fingerprint over the complete status stream. +RENAMED_PATHS="$(printf '%s\n' "$DIFF_NAME_STATUS" | awk -F '\t' ' + $1 ~ /^R/ { print $2; print $3 } +' | awk 'NF && !seen[$0]++')" +[[ -n "$UNTRACKED_PATHS" ]] || UNTRACKED_PATHS="$(printf '%s\n' "$DIFF_NAME_STATUS" \ + | awk -F '\t' '$1 == "?" { print $2 }')" +GENERATED_PATHS="" +while IFS= read -r _policy_path; do + [[ -n "$_policy_path" ]] || continue + case "$_policy_path" in + generated/*|*/generated/*|dist/*|*/dist/*|vendor/*|*/vendor/*|*.generated.*) + GENERATED_PATHS="${GENERATED_PATHS:+$GENERATED_PATHS$'\n'}$_policy_path" + continue + ;; + esac + if [[ -f "$WORK_DIR/$_policy_path" ]] \ + && head -n 8 "$WORK_DIR/$_policy_path" 2>/dev/null \ + | grep -qiE 'do not edit.*generated|generated.*do not edit'; then + GENERATED_PATHS="${GENERATED_PATHS:+$GENERATED_PATHS$'\n'}$_policy_path" fi -fi -if ! _gate_assurance_policy_lookup tiers tier "$TIER_RESOLVED" default_reviewers >/dev/null; then - printf 'Error: detected gate tier is absent from policy: %s\n' "$TIER_RESOLVED" >&2 - exit 2 -fi -TIER="$TIER_RESOLVED" -TIER_EVIDENCE_FLOOR="$(_gate_assurance_policy_lookup tiers tier "$TIER" evidence_floor)" || { - printf 'Error: gate tier policy has no evidence floor for: %s\n' "$TIER" >&2 - exit 2 -} +done <<< "$DIFF_FILES" -# ── Brief-based tier suggestion (advisory; never overrides --tier) ──────────── -if [[ -n "$BRIEF_FILE" && -f "$BRIEF_FILE" && -z "$TIER_OVERRIDE" ]]; then - _brief_arch_impact="$(awk '/^architecture_impact:[[:space:]]*/{sub(/^architecture_impact:[[:space:]]*/,""); gsub(/[[:space:]]/,""); print; exit}' "$BRIEF_FILE")" - case "$_brief_arch_impact" in - major) - if [[ "$TIER" != "full" ]]; then - printf 'pr-gate: brief architecture_impact:major — suggested tier: full (detected: %s). Override with --tier full or continue with current tier.\n' "$TIER" >&2 - fi - ;; - minor) - if [[ "$TIER" == "express" ]]; then - printf 'pr-gate: brief architecture_impact:minor — suggested tier: standard (detected: %s). Override with --tier standard or continue with current tier.\n' "$TIER" >&2 - fi +NON_DOCS="$(printf '%s\n' "$DIFF_FILES" \ + | grep -vE '\.(md|jsonl|txt)$|^\.gitignore$|^audits/|^docs/' || true)" +LAYER_ROOTS="$(printf '%s\n' "$NON_DOCS" | awk -F/ ' + $1 ~ /^(core|runtime|cli|adapters|hosts|commands|skills)$/ { print $1; next } + $1 ~ /^(install|uninstall)\.sh$/ { print $1 } +' | LC_ALL=C sort -u)" +LAYER_ROOT_COUNT="$(printf '%s\n' "$LAYER_ROOTS" | grep -c '[^[:space:]]' || true)" + +ARCHITECTURE_IMPACT="unknown" +if [[ -n "$INPUT_BRIEF_FILE" ]]; then + _input_brief_candidate="$INPUT_BRIEF_FILE" + [[ "$_input_brief_candidate" == /* ]] \ + || _input_brief_candidate="$WORK_DIR/$_input_brief_candidate" + if [[ ! -f "$_input_brief_candidate" || ! -r "$_input_brief_candidate" ]]; then + printf 'Error: --brief must name a readable file: %s\n' "$INPUT_BRIEF_FILE" >&2 + exit 2 + fi + _input_brief_parent="$(cd "$(dirname "$_input_brief_candidate")" && pwd -P)" || exit 2 + INPUT_BRIEF_FILE="$_input_brief_parent/$(basename "$_input_brief_candidate")" + ARCHITECTURE_IMPACT="$(awk ' + /^architecture_impact:[[:space:]]*/ { + sub(/^architecture_impact:[[:space:]]*/, "") + gsub(/[[:space:]]/, "") + print + exit + } + ' "$INPUT_BRIEF_FILE")" + : "${ARCHITECTURE_IMPACT:=unknown}" + case "$ARCHITECTURE_IMPACT" in + none|minor|major|unknown) ;; + *) + printf 'Error: --brief has invalid architecture_impact: %s\n' \ + "$ARCHITECTURE_IMPACT" >&2 + exit 2 ;; esac + unset _input_brief_candidate _input_brief_parent fi -# ── Determine reviewer list ─────────────────────────────────────────────────── +DIFF_FILES_JSON="$(printf '%s\n' "$DIFF_FILES" | _gate_policy_lines_json)" +NON_DOCS_JSON="$(printf '%s\n' "$NON_DOCS" | _gate_policy_lines_json)" +RENAMED_PATHS_JSON="$(printf '%s\n' "$RENAMED_PATHS" | _gate_policy_lines_json)" +UNTRACKED_PATHS_JSON="$(printf '%s\n' "$UNTRACKED_PATHS" | _gate_policy_lines_json)" +GENERATED_PATHS_JSON="$(printf '%s\n' "$GENERATED_PATHS" | _gate_policy_lines_json)" +LAYER_ROOTS_JSON="$(printf '%s\n' "$LAYER_ROOTS" | _gate_policy_lines_json)" +_policy_docs_only=false +[[ -z "$NON_DOCS" ]] && _policy_docs_only=true +_policy_cross_boundary=false +[[ "$LAYER_ROOT_COUNT" -gt 1 ]] && _policy_cross_boundary=true +CLASSIFICATIONS_JSON="$(jq -nc \ + --argjson docs_only "$_policy_docs_only" \ + --argjson changed_paths "$DIFF_FILES_JSON" --argjson non_docs "$NON_DOCS_JSON" \ + --argjson renamed "$RENAMED_PATHS_JSON" --argjson untracked "$UNTRACKED_PATHS_JSON" \ + --argjson generated "$GENERATED_PATHS_JSON" --argjson layer_roots "$LAYER_ROOTS_JSON" \ + --argjson cross_boundary "$_policy_cross_boundary" \ + --argjson lines "$LINES" --argjson binary_or_unknown "${BINARY_HIT:-0}" '[ + if $docs_only then {id:"docs-only",matches:$changed_paths} + else {id:"bounded-runtime",matches:$non_docs} end, + if $lines > 500 then {id:"large-change",matches:[("changed-lines:" + ($lines|tostring))]} + elif $lines >= 100 then {id:"medium-change",matches:[("changed-lines:" + ($lines|tostring))]} + else empty end, + if $binary_or_unknown > 0 then { + id:"binary-change",matches:[("binary-or-unknown:" + ($binary_or_unknown|tostring))] + } else empty end, + if ($renamed|length) > 0 then {id:"renamed",matches:$renamed} else empty end, + if ($untracked|length) > 0 then {id:"untracked",matches:$untracked} else empty end, + if ($generated|length) > 0 then {id:"generated",matches:$generated} else empty end, + if $cross_boundary then {id:"cross-boundary",matches:$layer_roots} else empty end + ]')" +POLICY_SCOPE_CONTENT_DIGEST="$( + _gate_policy_scope_content_digest \ + "$POLICY_DIFF_KIND" "$BASE" "$HEAD_REF" "$POLICY_SCOPE_INCLUDE_UNTRACKED" +)" || exit 2 +POLICY_SCOPE_FINGERPRINT="$( + { + printf 'policy=%s\npass=%s\narchitecture_impact=%s\nlines=%s\nbinary_or_unknown=%s\ncontent=%s\n' \ + "$POLICY_CONSUMER" "$PASS_KIND_RESOLVED" "$ARCHITECTURE_IMPACT" \ + "$LINES" "${BINARY_HIT:-0}" "$POLICY_SCOPE_CONTENT_DIGEST" + printf '%s\n' "$DIFF_NAME_STATUS" + } | _gate_sha256_stream +)" || exit 2 + +REQUESTED_REVIEWERS_JSON=null if [[ -n "$REVIEWERS_OVERRIDE" ]]; then - REVIEWERS="$(_gate_normalize_reviewer_list "$REVIEWERS_OVERRIDE" "$REVIEWERS_OPTION_SOURCE")" || exit 2 - COVERAGE_REQUESTED_DISPLAY="$(printf '%s' "$REVIEWERS" | tr ' ' ',')" + _requested_reviewer_words="$(_gate_normalize_reviewer_list \ + "$REVIEWERS_OVERRIDE" "$REVIEWERS_OPTION_SOURCE")" || exit 2 + REQUESTED_REVIEWERS_JSON="$(_gate_policy_words_json "$_requested_reviewer_words")" + COVERAGE_REQUESTED_DISPLAY="$(printf '%s' "$_requested_reviewer_words" | tr ' ' ',')" + unset _requested_reviewer_words else - _tier_default_reviewers="$(_gate_assurance_policy_lookup tiers tier "$TIER" default_reviewers)" || { - printf 'Error: gate tier policy has no default reviewer coverage for: %s\n' "$TIER" >&2 - exit 2 - } - REVIEWERS="$(_gate_normalize_reviewer_list "$_tier_default_reviewers" "tier $TIER default_reviewers")" || exit 2 COVERAGE_REQUESTED_DISPLAY="default" - unset _tier_default_reviewers fi +GATE_POLICY_INPUT="$(jq -nc \ + --arg policy "$POLICY_CONSUMER" --arg policy_source "$GATE_ASSURANCE_POLICY_SOURCE" \ + --arg scope_fingerprint "$POLICY_SCOPE_FINGERPRINT" \ + --arg tier "$TIER_REQUESTED" --arg mode "$MODE_REQUESTED" \ + --arg pass_kind "$PASS_KIND_RESOLVED" \ + --argjson reviewers "$REQUESTED_REVIEWERS_JSON" \ + --argjson vocabulary "$(_gate_policy_words_json "$ALL_REVIEWERS")" \ + --arg architecture_impact "$ARCHITECTURE_IMPACT" \ + --argjson line_changes "$LINES" \ + --argjson binary_or_unknown "${BINARY_HIT:-0}" \ + --argjson layer_roots "$LAYER_ROOTS_JSON" \ + --argjson classifications "$CLASSIFICATIONS_JSON" \ + --argjson changed_paths "$DIFF_FILES_JSON" \ + --argjson reviewer_override "$REVIEWER_OVERRIDE_PROVENANCE_JSON" '{ + policy:$policy, + policy_source:$policy_source, + scope_fingerprint:$scope_fingerprint, + requested:{tier:$tier,mode:$mode,pass_kind:$pass_kind,reviewers:$reviewers}, + reviewer_vocabulary:$vocabulary, + changed_paths:$changed_paths, + classifications:$classifications, + classification:{ + architecture_impact:$architecture_impact, + line_changes:$line_changes, + binary_or_unknown_count:$binary_or_unknown, + layer_roots:$layer_roots + }, + reviewer_override:$reviewer_override + }')" +GATE_POLICY_RESOLUTION="$(_gate_policy_resolve \ + "$GATE_POLICY_INPUT" "$POLICY_OVERRIDE_FILE")" || exit 2 +if [[ "$(jq -r '.enforcement.status' <<<"$GATE_POLICY_RESOLUTION")" != pass ]]; then + { + printf 'Error: requested gate assurance is below the canonical %s policy floor.\n' \ + "$POLICY_CONSUMER" + printf ' policy scope fingerprint: %s\n' "$POLICY_SCOPE_FINGERPRINT" + jq -r '.enforcement.violations[] | + " - " + .coordinate + ": requested=" + (.requested|tostring) + + " required=" + (.required|tostring)' <<<"$GATE_POLICY_RESOLUTION" + printf ' A downgrade requires an explicit --policy-override gate_policy_override_v1 JSON\n' + printf ' bound to this exact scope and carrying recorded user approval.\n' + if [[ -n "$POLICY_OVERRIDE_FILE" ]]; then + printf ' supplied override status: %s\n' \ + "$(jq -r '.override.status' <<<"$GATE_POLICY_RESOLUTION")" + fi + } >&2 + exit 3 +fi + +TIER_RESOLVED="$(jq -r '.resolved.tier' <<<"$GATE_POLICY_RESOLUTION")" +TIER="$TIER_RESOLVED" +TIER_EVIDENCE_FLOOR="$(_gate_assurance_policy_lookup tiers tier "$TIER" evidence_floor)" || { + printf 'Error: gate tier policy has no evidence floor for: %s\n' "$TIER" >&2 + exit 2 +} +MODE_RESOLVED="$(jq -r '.resolved.mode' <<<"$GATE_POLICY_RESOLUTION")" +MODE_TOPOLOGY="$(_gate_assurance_policy_lookup modes mode "$MODE_RESOLVED" topology)" \ + || exit 2 +MODE_SYNTHESIS="$(_gate_assurance_policy_lookup modes mode "$MODE_RESOLVED" synthesis)" \ + || exit 2 +case "$MODE_TOPOLOGY:$MODE_SYNTHESIS" in + combined-session:inline) SEQUENTIAL=true ;; + per-reviewer-sessions:separate-session) SEQUENTIAL=false ;; + *) + printf 'Error: unsupported gate mode topology for %s: %s + %s\n' \ + "$MODE_RESOLVED" "$MODE_TOPOLOGY" "$MODE_SYNTHESIS" >&2 + exit 2 + ;; +esac +REVIEWERS="$(jq -r '.resolved.reviewers | join(" ")' <<<"$GATE_POLICY_RESOLUTION")" +[[ -n "$REVIEWERS" ]] || { + printf 'Error: gate policy resolved empty reviewer coverage\n' >&2 + exit 2 +} + REVIEWER_DISPLAY=$(printf '%s' "$REVIEWERS" | tr ' ' ',') NUM_REVIEWERS=$(printf '%s\n' "$REVIEWERS" | awk '{print NF}') @@ -1801,7 +2901,7 @@ gate_finalize_assurance() { local result_file="$1" assurance_file="$2" local final requested_json outcomes_json independence_status implementation_isolated local per_reviewer_independent expected_count capture_count assurance_tmp result_tmp - local result_sha assurance_sha attestation_tmp run_ids_json + local result_sha assurance_sha attestation_tmp run_ids_json attestation_pointer local -a capture_files=() final="$(grep -E '^Final: (GO|NO-GO)$' "$result_file" | awk '{print $2}')" @@ -1858,6 +2958,10 @@ gate_finalize_assurance() { evidence_status:"unavailable"}])')" fi fi + attestation_pointer="" + if [[ "$independence_status" == verified ]]; then + attestation_pointer="$ASSURANCE_ATTESTATION_POINTER" + fi result_tmp="$(mktemp "${result_file}.assurance-tmp.XXXXXX")" || { printf 'Error: unable to create gate result temporary file beside: %s\n' \ @@ -1910,8 +3014,9 @@ gate_finalize_assurance() { --arg reviewer_topology "$MODE_TOPOLOGY" \ --arg independence_status "$independence_status" \ --arg policy_source "$GATE_ASSURANCE_POLICY_SOURCE" \ - --arg attestation "$ASSURANCE_ATTESTATION_POINTER" \ + --arg attestation "$attestation_pointer" \ --argjson requested "$requested_json" --argjson outcomes "$outcomes_json" \ + --argjson policy_resolution "$GATE_POLICY_RESOLUTION" \ --argjson implementation_isolated "$implementation_isolated" \ --argjson per_reviewer_independent "$per_reviewer_independent" ' { @@ -1945,6 +3050,7 @@ gate_finalize_assurance() { evidence_status:$independence_status } }, + policy:$policy_resolution, dispatch:{outcomes:$outcomes}, provenance:{ producer:"pr-gate.sh", @@ -2108,21 +3214,46 @@ render_gate_overrides_block() { GATE_OVERRIDES_CONTEXT_BLOCK="$(render_gate_overrides_block "$GATE_OVERRIDES_CONTENT")" INITIAL_RESULT_DISPLAY="${INITIAL_RESULT_RESOLVED:-none}" +POLICY_REQUIRED_REVIEWERS_DISPLAY="$(jq -r \ + '.resolution.required_reviewers | if length == 0 then "none" else join(",") end' \ + <<<"$GATE_POLICY_RESOLUTION")" +POLICY_REQUIRED_MODE_DISPLAY="$(jq -r '.resolution.required_mode // "none"' \ + <<<"$GATE_POLICY_RESOLUTION")" +POLICY_ESCALATION_SIGNALS_DISPLAY="$(jq -c '[ + .matched_signals[] + | select(.source != "consumer-policy") + | select((.required_reviewers | length) > 0 or .required_mode != null) + | { + id, + required_reviewers, + required_mode + } +]' <<<"$GATE_POLICY_RESOLUTION")" printf -v GATE_ASSURANCE_CONTEXT_BLOCK \ - ' Assurance coordinates (resolved by the gate shell; do not reinterpret):\n tier.requested: %s\n tier.resolved: %s\n tier.evidence_floor: %s\n mode.requested: %s\n mode.resolved: %s\n mode.topology: %s\n mode.synthesis: %s\n pass.requested: %s\n pass.resolved: %s\n pass.scope: %s\n pass.initial_result: %s\n coverage.requested: %s\n coverage.selected: %s\n coverage.skipped: %s\n policy.source: %s\n' \ + ' Assurance coordinates (resolved by the gate shell; do not reinterpret):\n tier.requested: %s\n tier.resolved: %s\n tier.evidence_floor: %s\n mode.requested: %s\n mode.resolved: %s\n mode.topology: %s\n mode.synthesis: %s\n pass.requested: %s\n pass.resolved: %s\n pass.scope: %s\n pass.initial_result: %s\n coverage.requested: %s\n coverage.selected: %s\n coverage.skipped: %s\n policy.consumer: %s\n policy.minimum_tier: %s\n policy.required_reviewers: %s\n policy.recommended_mode: %s\n policy.required_mode: %s\n policy.escalation_signals: %s\n policy.scope_fingerprint: %s\n policy.source: %s\n' \ "$TIER_REQUESTED" "$TIER_RESOLVED" "$TIER_EVIDENCE_FLOOR" \ "$MODE_REQUESTED" "$MODE_RESOLVED" "$MODE_TOPOLOGY" "$MODE_SYNTHESIS" \ "$PASS_KIND_REQUESTED" "$PASS_KIND_RESOLVED" "$PASS_SCOPE" "$INITIAL_RESULT_DISPLAY" \ "$COVERAGE_REQUESTED_DISPLAY" \ "$COVERAGE_SELECTED_DISPLAY" "$COVERAGE_SKIPPED_DISPLAY" \ + "$POLICY_CONSUMER" "$(jq -r '.resolution.minimum_tier' <<<"$GATE_POLICY_RESOLUTION")" \ + "$POLICY_REQUIRED_REVIEWERS_DISPLAY" \ + "$(jq -r '.resolution.recommended_mode' <<<"$GATE_POLICY_RESOLUTION")" \ + "$POLICY_REQUIRED_MODE_DISPLAY" "$POLICY_ESCALATION_SIGNALS_DISPLAY" \ + "$POLICY_SCOPE_FINGERPRINT" \ "$GATE_ASSURANCE_POLICY_SOURCE" say 'pr-gate: tier %s -> %s; mode %s -> %s; pass %s -> %s\n' \ "$TIER_REQUESTED" "$TIER_RESOLVED" "$MODE_REQUESTED" "$MODE_RESOLVED" \ "$PASS_KIND_REQUESTED" "$PASS_KIND_RESOLVED" -say 'pr-gate: coverage requested=%s selected=%s skipped=%s; policy=%s\n' \ +say 'pr-gate: coverage requested=%s selected=%s skipped=%s; policy=%s/%s\n' \ "$COVERAGE_REQUESTED_DISPLAY" "$COVERAGE_SELECTED_DISPLAY" \ - "$COVERAGE_SKIPPED_DISPLAY" "$GATE_ASSURANCE_POLICY_SOURCE" + "$COVERAGE_SKIPPED_DISPLAY" "$POLICY_CONSUMER" "$GATE_ASSURANCE_POLICY_SOURCE" +say 'pr-gate: policy minimum-tier=%s required-reviewers=%s recommended-mode=%s required-mode=%s scope=%s\n' \ + "$(jq -r '.resolution.minimum_tier' <<<"$GATE_POLICY_RESOLUTION")" \ + "$POLICY_REQUIRED_REVIEWERS_DISPLAY" \ + "$(jq -r '.resolution.recommended_mode' <<<"$GATE_POLICY_RESOLUTION")" \ + "$POLICY_REQUIRED_MODE_DISPLAY" "$POLICY_SCOPE_FINGERPRINT" [[ "${ADJ_COUNT:-0}" -gt 0 ]] && say ' adjacent test files added: %d\n' "$ADJ_COUNT" say 'result will be written to: %s\n\n' "$OUTPUT_FILE" @@ -2179,17 +3310,6 @@ _preflight_log_display_path() { fi } -_preflight_sha256_stream() { - if command -v sha256sum >/dev/null 2>&1; then - sha256sum | awk '{print $1}' - elif command -v shasum >/dev/null 2>&1; then - shasum -a 256 | awk '{print $1}' - else - printf 'Error: pre-flight evidence requires sha256sum or shasum\n' >&2 - return 2 - fi -} - _preflight_sha256_file() { local file="$1" digest="" if command -v sha256sum >/dev/null 2>&1 \ @@ -2221,7 +3341,7 @@ _preflight_tree_fingerprint() { quoted="$(printf '%q' "$path")" if [[ -L "$WORK_DIR/$path" ]]; then kind=symlink; executable=false - digest="$(printf '%s' "$(readlink "$WORK_DIR/$path")" | _preflight_sha256_stream)" || { rm -f "$manifest"; return 2; } + digest="$(printf '%s' "$(readlink "$WORK_DIR/$path")" | _gate_sha256_stream)" || { rm -f "$manifest"; return 2; } elif [[ -f "$WORK_DIR/$path" ]]; then kind=file; [[ -x "$WORK_DIR/$path" ]] && executable=true || executable=false digest="$(_preflight_sha256_file "$WORK_DIR/$path")" || { rm -f "$manifest"; return 2; } @@ -2230,7 +3350,7 @@ _preflight_tree_fingerprint() { fi printf '%s\t%s\t%s\t%s\n' "$quoted" "$kind" "$executable" "$digest" >> "$manifest" done < <(git -C "$WORK_DIR" ls-files --cached --others --exclude-standard -z) - LC_ALL=C sort "$manifest" | _preflight_sha256_stream + LC_ALL=C sort "$manifest" | _gate_sha256_stream local rc=$? rm -f "$manifest" return "$rc" @@ -2239,7 +3359,7 @@ _preflight_tree_fingerprint() { _preflight_repo_identity() { local remote remote="$(git -C "$WORK_DIR" config --get remote.origin.url 2>/dev/null || true)" - printf '%s\n%s\n' "$WORK_DIR" "$remote" | _preflight_sha256_stream + printf '%s\n%s\n' "$WORK_DIR" "$remote" | _gate_sha256_stream } GATE_BINDING_SUBJECT_FINGERPRINT="$(_preflight_tree_fingerprint)" || exit 2 @@ -2260,7 +3380,7 @@ if [[ "$SKIP_PREFLIGHT_TESTS" != "true" && -n "$TEST_CMD_OVERRIDE" ]]; then PREFLIGHT_LOG_PATH="$WORK_DIR/.gate-results/preflight-tests-${TIMESTAMP}.log" PREFLIGHT_EVIDENCE_PATH="$WORK_DIR/.gate-results/preflight-evidence-${TIMESTAMP}.json" PREFLIGHT_RICH_RESULT_PATH="$WORK_DIR/.gate-results/preflight-rich-result-${TIMESTAMP}.json" - _preflight_command_digest="$(printf '%s' "$TEST_CMD_OVERRIDE" | _preflight_sha256_stream)" || exit 2 + _preflight_command_digest="$(printf '%s' "$TEST_CMD_OVERRIDE" | _gate_sha256_stream)" || exit 2 _preflight_before="$GATE_BINDING_SUBJECT_FINGERPRINT" _preflight_repo_id="$GATE_BINDING_REPO_IDENTITY" _preflight_base_commit="$GATE_BINDING_BASE_COMMIT" @@ -2680,7 +3800,8 @@ output_format: | - (or "none" when recommended=false) Escalation is recommended when: - (a) any diff file matches (^|[/_.-])(auth|oauth|jwt|session|secret|password|token|credential|cors|csrf|webhook|sudo|ssh|payment|billing)([/_.-]|\$)|(^|/)migrations?/|^\.github/ + (a) policy.escalation_signals above is non-empty; use this canonical resolver + output and do not re-match paths with a separate regex (b) at least one reviewer returned advise|block-soft. self_verify: @@ -2833,22 +3954,24 @@ task: Do not attempt to cover other reviewer dimensions. 3. Write a structured findings block with: - Findings: [severity] file:line -- description (low/medium/high) - - Explicit verdict: approve | advise | block-soft | block + - Exactly one heading with the canonical verdict: + ## ${r} -- approve | advise | block-soft | block - One-sentence rationale for your verdict Write your complete review to ${REVIEWER_OUTPUT}. output_format: | ## ${r} -- {verdict} - - [{severity}] {file:line} -- {finding description} + {structured findings that follow the agent definition} - Verdict: {approve | advise | block-soft | block}. {One-sentence rationale.} + The heading is the machine verdict. If you also emit an upper-case + Verdict: line, it must appear exactly once and match the heading. self_verify: - cmd: "test -f ${REVIEWER_OUTPUT}" acceptance: - - ${REVIEWER_OUTPUT} exists with at least one findings line and an explicit Verdict line + - ${REVIEWER_OUTPUT} exists with exactly one canonical ${r} verdict heading RBRIEF_EOF REVIEWER_DISPATCH_CMD="$(dispatch_via "$EXECUTOR" "$REVIEWER_BRIEF" "$WORK_DIR" "$DISPATCH_MODEL" "$DISPATCH_SANDBOX" "$DISPATCH_APPROVAL" "$TIMEOUT" "$DISPATCH_ISOLATION" "$DISPATCH_EFFORT")" || exit 2 @@ -2932,21 +4055,25 @@ RBRIEF_EOF exit 1 fi - # Verify every reviewer output contains exactly one parseable verdict line before synthesis. - # Zero lines → malformed output; two or more lines → ambiguous (first-match would silently - # ignore a later more-severe verdict). Both cases must be rejected fail-closed. + # Verify every reviewer output contains one unambiguous canonical verdict + # before synthesis. The role-matched heading is authoritative; an optional + # upper-case Verdict marker must be unique and agree with it. INVALID_OUTPUTS=() + REVIEWER_VERDICTS=() for i in "${!REVIEWER_OUTPUT_FILES[@]}"; do rf="${REVIEWER_OUTPUT_FILES[$i]}" r="${REVIEWER_NAMES[$i]}" - verdict_count=$(grep -cE '^Verdict: (approve|advise|block-soft|block)([. ]|$)' "$rf" || true) - if [[ "$verdict_count" -ne 1 ]]; then - INVALID_OUTPUTS+=("$r (found $verdict_count)") + if reviewer_verdict="$(_gate_reviewer_verdict_extract "$r" "$rf")"; then + REVIEWER_VERDICTS+=("$reviewer_verdict") + else + INVALID_OUTPUTS+=("$r") fi done if [[ "${#INVALID_OUTPUTS[@]}" -gt 0 ]]; then - printf 'Error: reviewer output must contain exactly one valid Verdict line for: %s\n' "${INVALID_OUTPUTS[*]}" >&2 - printf 'Expected: exactly one of: Verdict: approve|advise|block-soft|block\n' >&2 + printf 'Error: reviewer output has an invalid or ambiguous canonical verdict for: %s\n' \ + "${INVALID_OUTPUTS[*]}" >&2 + printf 'Expected: exactly one matching heading: ## -- approve|advise|block-soft|block\n' >&2 + printf 'Any upper-case Verdict: marker must be unique and match that heading.\n' >&2 printf 'Gate aborted -- use --sequential to diagnose.\n' >&2 exit 1 fi @@ -2995,8 +4122,7 @@ RBRIEF_EOF # Compute the final verdict deterministically in shell before synthesis. # Synthesis is treated as prose-only; the shell verdict is the authoritative gate result. SHELL_VERDICT="approve" - for rf in "${REVIEWER_OUTPUT_FILES[@]}"; do - rv=$(grep -oE '^Verdict: (approve|advise|block-soft|block)([. ]|$)' "$rf" | awk '{print $2}' | tr -d '. ' || true) + for rv in "${REVIEWER_VERDICTS[@]}"; do case "$rv" in block) SHELL_VERDICT="block" ;; block-soft) [[ "$SHELL_VERDICT" != "block" ]] && SHELL_VERDICT="block-soft" ;; @@ -3130,7 +4256,8 @@ output_format: | - (or "none" when recommended=false) Escalation is recommended when: - (a) any diff file matches (^|[/_.-])(auth|oauth|jwt|session|secret|password|token|credential|cors|csrf|webhook|sudo|ssh|payment|billing)([/_.-]|\$)|(^|/)migrations?/|^\.github/ + (a) policy.escalation_signals above is non-empty; use this canonical resolver + output and do not re-match paths with a separate regex (b) at least one reviewer returned advise|block-soft. Recommended follow-ups: @@ -3349,7 +4476,7 @@ fi # the machine-owned assurance sidecar only after every deterministic rewrite # and explicitly enabled post-gate hook is complete. The shared verifier then # checks result/pointer/envelope parity before publication or relocation. -gate_finalize_assurance "$OUTPUT_FILE" "$ASSURANCE_FILE" || exit 1 +gate_finalize_assurance "$OUTPUT_FILE" "$ASSURANCE_FILE" || exit 2 # ── Relocate result to run dir (post-verification) ─────────────────────────── # OUTPUT_FILE was written by the executor in WORK_DIR (workspace-write sandbox diff --git a/runtime/lib/gate-result-verify.sh b/runtime/lib/gate-result-verify.sh index 2f2a1c88..3157d6d6 100644 --- a/runtime/lib/gate-result-verify.sh +++ b/runtime/lib/gate-result-verify.sh @@ -101,7 +101,7 @@ gate_assurance_verify() { (length == (unique | length)); def same_set($a; $b): ($a | sort) == ($b | sort); only_keys(["kind","schema_version","result","bindings","coordinates", - "dispatch","provenance"]) and + "policy","dispatch","provenance"]) and (.result | only_keys(["final"])) and (.bindings | only_keys(["result_sha256","repo_root","repo_identity", "base_commit","head_commit","subject_fingerprint"])) and @@ -114,6 +114,118 @@ gate_assurance_verify() { (.coordinates.independence | only_keys(["implementation_context_isolated","reviewer_topology", "per_reviewer_independent","evidence_status"])) and + (if has("policy") then + (.policy | + only_keys(["kind","schema_version","consumer_policy","policy_source", + "scope_fingerprint","request","classification","resolution", + "matched_signals","resolved","enforcement","override", + "reviewer_override"])) and + (.policy.request | + only_keys(["tier","mode","pass_kind","reviewers"])) and + (.policy.classification | + only_keys(["architecture_impact","line_changes", + "binary_or_unknown_count","layer_roots"])) and + (.policy.resolution | + only_keys(["minimum_tier","required_reviewers","recommended_mode", + "required_mode","downgrade_requested","downgrade_allowed"])) and + (.policy.resolved | only_keys(["tier","mode","reviewers"])) and + (.policy.enforcement | only_keys(["status","violations"])) and + (.policy.override | + only_keys(["status","source","sha256","reason","approver"])) and + (.policy.reviewer_override | + only_keys(["status","source","sha256"])) and + .policy.kind == "gate_policy_resolution_v1" and + .policy.schema_version == 1 and + (.policy.consumer_policy | IN("generic","maintainer")) and + .policy.policy_source == .provenance.policy_source and + (.policy.scope_fingerprint | test("^[a-f0-9]{64}$")) and + .policy.request.tier == .coordinates.tier.requested and + .policy.request.mode == .coordinates.mode.requested and + .policy.request.pass_kind == .coordinates.pass.resolved and + ((.policy.request.reviewers == null and + .coordinates.coverage.requested == null) or + (same_set(.policy.request.reviewers; + .coordinates.coverage.requested))) and + (.policy.classification.architecture_impact | + IN("unknown","none","minor","major")) and + (.policy.classification.line_changes | + type == "number" and . >= 0 and floor == .) and + (.policy.classification.binary_or_unknown_count | + type == "number" and . >= 0 and floor == .) and + (.policy.classification.layer_roots | strings_unique) and + (.policy.resolution.minimum_tier | + IN("express","standard","full")) and + (.policy.resolution.required_reviewers | strings_unique) and + (.policy as $policy | + all($policy.resolution.required_reviewers[]; + . as $reviewer | + ($policy.resolved.reviewers | index($reviewer)) != null or + $policy.resolution.downgrade_allowed)) and + (.policy.resolution.recommended_mode | + IN("sequential","parallel")) and + (.policy.resolution.required_mode == null or + (.policy.resolution.required_mode | + IN("sequential","parallel"))) and + (.policy.resolution.downgrade_requested | type == "boolean") and + (.policy.resolution.downgrade_allowed | type == "boolean") and + (.policy.matched_signals | type == "array" and length > 0) and + ([.policy.matched_signals[].id] | strings_unique) and + (all(.policy.matched_signals[]; + only_keys(["id","source","matches","minimum_tier", + "required_reviewers","recommended_mode","required_mode"]) and + (.id | type == "string" and length > 0) and + (.source | + IN("consumer-policy","classification","path-regex","brief-value")) and + (.matches | strings_unique and length > 0) and + (.minimum_tier | IN("express","standard","full")) and + (.required_reviewers | strings_unique) and + (.recommended_mode | IN("sequential","parallel")) and + (.required_mode == null or + (.required_mode | IN("sequential","parallel"))))) and + .policy.resolved.tier == .coordinates.tier.resolved and + .policy.resolved.mode == .coordinates.mode.resolved and + same_set(.policy.resolved.reviewers; + .coordinates.coverage.selected) and + .policy.enforcement.status == "pass" and + (.policy.enforcement.violations | type == "array") and + (all(.policy.enforcement.violations[]; + only_keys(["coordinate","requested","required"]) and + (.coordinate | IN("tier","coverage","mode")))) and + (.policy.override.status | + IN("not_provided","not_needed","applied","scope_mismatch", + "allowance_mismatch")) and + (if .policy.resolution.downgrade_requested + then + .policy.resolution.downgrade_allowed == true and + .policy.override.status == "applied" and + (.policy.override.source | + type == "string" and startswith("/")) and + (.policy.override.sha256 | test("^[a-f0-9]{64}$")) and + (.policy.override.reason | type == "string" and length > 0) and + (.policy.override.approver | + only_keys(["kind","identity","approval_ref"])) and + .policy.override.approver.kind == "user" and + (.policy.override.approver.identity | + type == "string" and length > 0) and + (.policy.override.approver.approval_ref | + type == "string" and length > 0) + else + .policy.resolution.downgrade_allowed == false and + (.policy.override.status | + IN("not_provided","not_needed")) + end) and + (.policy.reviewer_override.status | + IN("not_provided","provided")) and + (if .policy.reviewer_override.status == "provided" + then + (.policy.reviewer_override.source | + type == "string" and startswith("/")) and + (.policy.reviewer_override.sha256 | test("^[a-f0-9]{64}$")) + else + .policy.reviewer_override.source == null and + .policy.reviewer_override.sha256 == null + end) + else true end) and (.dispatch | only_keys(["outcomes"])) and (all(.dispatch.outcomes[]; only_keys(["role","reviewer","status","run_id","evidence_status"]))) and @@ -339,5 +451,8 @@ gate_result_verify() { esac } -export -f gate_result_verdict_verify gate_assurance_verify \ - gate_assurance_authorization_verify gate_result_verify +# These functions are a sourced-library API, not a `bash -c` API. Exporting +# only the public entry points leaks an incomplete closure into descendants: +# gate_result_verify also needs private helpers such as +# _gate_result_frontmatter_value. A nested pmctl could then mistake the +# inherited entry point for a fully loaded verifier and reject valid results. diff --git a/runtime/lib/pmctl-gate.sh b/runtime/lib/pmctl-gate.sh index 17e6b7c4..fd29d51c 100644 --- a/runtime/lib/pmctl-gate.sh +++ b/runtime/lib/pmctl-gate.sh @@ -64,6 +64,22 @@ _pmctl_gate_operation_ensure_loaded() { && _pmctl_operation_ensure_loaded "$repo_root" } +# Load the verifier from the selected pmctl repository even when the caller +# exported a function with the same name. Bash function exports do not carry a +# dependency manifest, so trusting an inherited gate_result_verify can leave +# its private helpers missing or mix functions from different revisions. +_pmctl_gate_result_verifier_load() { + local repo_root="$1" verifier_lib + verifier_lib="$repo_root/runtime/lib/gate-result-verify.sh" + [[ -r "$verifier_lib" ]] || return 1 + # shellcheck disable=SC1090,SC1091 + . "$verifier_lib" || return 1 + declare -F gate_result_verify >/dev/null 2>&1 \ + && declare -F gate_result_verdict_verify >/dev/null 2>&1 \ + && declare -F _gate_result_frontmatter_value >/dev/null 2>&1 \ + && declare -F _gate_result_sha256_file >/dev/null 2>&1 +} + # A v2 result and its bound sidecar cannot be renamed into place as one # filesystem operation. If a verifier races the producer between those # renames, wait briefly for the sidecar (and, for verified independence, @@ -613,14 +629,7 @@ pmctl_gate_wait() { return 2 fi fi - if ! declare -F gate_result_verify >/dev/null 2>&1; then - local _gr_lib="$repo_root/runtime/lib/gate-result-verify.sh" - if [[ -r "$_gr_lib" ]]; then - # shellcheck disable=SC1090,SC1091 - . "$_gr_lib" 2>/dev/null || true - fi - fi - if ! declare -F gate_result_verify >/dev/null 2>&1; then + if ! _pmctl_gate_result_verifier_load "$repo_root"; then printf 'pmctl gate wait: FAIL: gate_result_verify unavailable -- cannot confirm result integrity for %s, treating as failed wait\n' "$_result" >&2 return 2 fi @@ -708,14 +717,10 @@ pmctl_gate_verify() { local attestation_file runs_file canonical_repo_root assurance_repo_root local canonical_run_root - if ! declare -F gate_result_verify >/dev/null; then - local lib="$repo_root/runtime/lib/gate-result-verify.sh" - if [[ ! -r "$lib" ]]; then - printf 'pmctl gate verify: required library not found: %s\n' "$lib" >&2 - return 2 - fi - # shellcheck source=runtime/lib/gate-result-verify.sh - . "$lib" + if ! _pmctl_gate_result_verifier_load "$repo_root"; then + printf 'pmctl gate verify: required verifier library is missing or incomplete: %s\n' \ + "$repo_root/runtime/lib/gate-result-verify.sh" >&2 + return 2 fi _pmctl_gate_wait_for_assurance_publication "$result_file" diff --git a/runtime/lib/pmctl-ship.sh b/runtime/lib/pmctl-ship.sh index e53a6594..0cc5ad25 100644 --- a/runtime/lib/pmctl-ship.sh +++ b/runtime/lib/pmctl-ship.sh @@ -187,7 +187,9 @@ pmctl_ship_finish() { local pre_gate_head pre_gate_head="$(git -C "$work_dir" rev-parse HEAD 2>/dev/null)" - local gate_args=(--executor codex --cd "$work_dir" --lifecycle foreground) + # This repo-owned finish path requests the maintainer consumer policy. The + # generic `pmctl gate run` default remains risk-based and composable. + local gate_args=(--executor codex --policy maintainer --cd "$work_dir" --lifecycle foreground) [[ -n "$reviewers" ]] && gate_args+=(--reviewers "$reviewers") local gate_out gate_status=0 gate_out="$(pmctl_gate_run "$repo_root" "${gate_args[@]}" 2>&1)" || gate_status=$? diff --git a/skills/pr-gate-review/SKILL.md b/skills/pr-gate-review/SKILL.md index 462bb167..4c58b785 100644 --- a/skills/pr-gate-review/SKILL.md +++ b/skills/pr-gate-review/SKILL.md @@ -18,16 +18,17 @@ implement → pr-gate → fix NO-GO → push → PR). - Slash command: `/pr-gate` (see `commands/pr-gate.md`). It dispatches the reviewers and writes a typed result to `.gate-results/`. -- Direct: `bash runtime/bin/pr-gate.sh --cd --executor auto [--mode sequential|parallel]`. +- Direct: `pmctl gate run --cd --executor auto --policy generic [--mode sequential|parallel]`. - Reasoning effort defaults to `medium` (`--effort low|medium|high`, independent of `--model`/`--executor`). Only reach for `--effort high` when you need deeper analysis — e.g. a hard-to-diagnose finding, or escalating after repeated NO-GO rounds on the same issue. **Tier / mode** (tiers reflect rigor level, not reviewer count): - `express` — hotfix, docs-only, `architecture_impact: none`; machine verify + critic + qa. - `standard` — feature, `architecture_impact: minor`; adds architecture-reviewer with conceptual map. -- `full` — architectural change, `architecture_impact: major`, sensitive path; defaults to all reviewer dimensions. +- `full` — large or architectural change, `architecture_impact: major`; defaults to all reviewer dimensions. +- Sensitive paths add their security/risk/architecture reviewer without automatically forcing `full`. - Default = **sequential**, low token cost. `--mode parallel` gives each reviewer an independent session; `--parallel` remains a compatibility spelling. -- Pass `--brief ` to get an advisory tier suggestion based on the brief's `architecture_impact`. -- Force a tier with `--tier express|standard|full`; re-gate a remediation subset with `--targeted r1,r2 --initial-result `. +- Trusted `architecture_impact` from `--brief ` is enforced by the canonical policy resolver (`minor` → at least `standard`, `major` → `full`). +- Request a tier with `--tier express|standard|full`; a request below the policy floor fails before dispatch. Re-gate a remediation subset with `--targeted r1,r2 --initial-result `. ## Reading the result @@ -44,7 +45,9 @@ valid envelope; treat its verdict as structurally valid without inferring implementation isolation or independent reviewer sessions. Repo-layout independence is authoritative only when verification also confirms the protected producer attestation, the invoking repository's canonical state -partition, and every claimed canonical run record. +partition, and every claimed canonical run record. Current v2 envelopes also +record the shell-owned policy classification, matched signals, floors, resolved +coordinates, and any scope-bound user override. - **NO-GO** (a reviewer returned `block`): fix the blocking finding. Per project convention, clear **every** finding (high/med/low/advise) on a NO-GO, not just diff --git a/tests/bin/run-tests.sh b/tests/bin/run-tests.sh index 1b8c44b1..abb221b3 100755 --- a/tests/bin/run-tests.sh +++ b/tests/bin/run-tests.sh @@ -187,6 +187,8 @@ map_path() { esac case "$path" in + .gitignore) + add_suite test-setup-project; behavioral=1 ;; tools/lint/lint-shellcheck.sh|tools/lint/shellcheck-domains.tsv|tools/lint/shellcheck-ignores.tsv|tests/shell/test-lint-shellcheck.sh) add_suite lint-scripts; add_suite test-lint-shellcheck; behavioral=1 ;; tools/lint/lint-script-domain-inventory.sh|tests/shell/test-script-domain-inventory.sh|docs/architecture/script-domain-ownership.md|docs/architecture/script-domain-inventory.tsv|docs/architecture/script-domain-reference-allowlist.tsv|docs/architecture/script-variable-inventory.tsv|docs/architecture/script-variable-consumers.tsv) @@ -215,9 +217,14 @@ map_path() { add_suite test-pmctl-task; behavioral=1 ;; core/schema/preflight-evidence.schema.json) add_suite test-pr-gate; behavioral=1 ;; + core/schema/gate-assurance.schema.json|core/schema/gate-policy-override.schema.json) + add_suite test-core-schemas; add_suite test-pr-gate + add_suite test-pmctl-gate; behavioral=1 ;; + runtime/lib/gate-result-verify.sh) + add_suite test-pr-gate; add_suite test-pmctl-gate; behavioral=1 ;; tools/generate-gate-result-verifier-fallback.sh) add_suite test-pr-gate; behavioral=1 ;; - core/policy/gate-tiers.tsv|core/policy/gate-modes.tsv|core/policy/gate-pass-kinds.tsv) + core/policy/gate-tiers.tsv|core/policy/gate-modes.tsv|core/policy/gate-pass-kinds.tsv|core/policy/gate-policy-consumers.tsv|core/policy/gate-policy-signals.tsv) add_suite test-pr-gate; add_suite test-pr-gate-profile; behavioral=1 ;; runtime/lib/pmctl-config.sh) add_suite test-pmctl-dispatch; add_suite test-pmctl-memory; add_suite test-pmctl-context; behavioral=1 ;; diff --git a/tests/shell/test-commands.sh b/tests/shell/test-commands.sh index b372cfdf..ac672cda 100755 --- a/tests/shell/test-commands.sh +++ b/tests/shell/test-commands.sh @@ -502,6 +502,8 @@ STUB ! grep -qx '/tmp/initial.md' "$args_log" || ! grep -qx -- '--mode' "$args_log" || ! grep -qx 'parallel' "$args_log" || + ! grep -qx -- '--policy' "$args_log" || + ! grep -qx 'generic' "$args_log" || grep -qx -- '--reviewers' "$args_log"; then fail "$name" "valid targeted invocation was not forwarded canonically" return @@ -586,6 +588,16 @@ if should_run "ship: every gate invocation uses --lifecycle foreground"; then fail "ship: every gate invocation uses --lifecycle foreground" "found $ship_gate_calls occurrence(s) of the gate call but only $ship_foreground_calls paired with --lifecycle foreground in $SHIP" fi fi +if should_run "ship: every gate invocation uses maintainer policy"; then + ship_flat=$(tr '\n' ' ' < "$SHIP" | tr -s ' ') + ship_gate_calls=$(grep -oE 'pmctl gate run --executor ' <<< "$ship_flat" | wc -l) + ship_policy_calls=$(grep -oE 'pmctl gate run --executor --policy maintainer' <<< "$ship_flat" | wc -l) + if [[ "$ship_gate_calls" -gt 0 && "$ship_gate_calls" -eq "$ship_policy_calls" ]]; then + pass "ship: every gate invocation uses maintainer policy" + else + fail "ship: every gate invocation uses maintainer policy" "found $ship_gate_calls occurrence(s) of the gate call but only $ship_policy_calls paired with --policy maintainer in $SHIP" + fi +fi should_run "ship: explains why detached+wait is unnecessary here" && assert_file_contains "ship: explains why detached+wait is unnecessary here" "$SHIP" "nothing else for the main thread to do while it waits" && pass "ship: explains why detached+wait is unnecessary here" should_run "ship: reads Final GO/NO-GO verdict" && assert_file_contains "ship: reads Final GO/NO-GO verdict" "$SHIP" "Final:" && pass "ship: reads Final GO/NO-GO verdict" should_run "ship: NO-GO fixes every finding not only blocking ones" && assert_file_contains "ship: NO-GO fixes every finding not only blocking ones" "$SHIP" "the blocking ones" && pass "ship: NO-GO fixes every finding not only blocking ones" diff --git a/tests/shell/test-core-schemas.sh b/tests/shell/test-core-schemas.sh index 7a3816e7..22cfbffb 100755 --- a/tests/shell/test-core-schemas.sh +++ b/tests/shell/test-core-schemas.sh @@ -770,9 +770,9 @@ _gate_assurance_valid_instance() { }, coverage:{ requested:null, - selected:["critic","qa-tester"], + selected:["critic","qa-tester","architecture-reviewer"], skipped:["security"], - vocabulary:["critic","qa-tester","security"] + vocabulary:["critic","qa-tester","architecture-reviewer","security"] }, independence:{ implementation_context_isolated:null, @@ -781,6 +781,67 @@ _gate_assurance_valid_instance() { evidence_status:"unavailable" } }, + policy:{ + kind:"gate_policy_resolution_v1", + schema_version:1, + consumer_policy:"generic", + policy_source:"canonical", + scope_fingerprint:("f" * 64), + request:{ + tier:"standard", + mode:"sequential", + pass_kind:"initial", + reviewers:null + }, + classification:{ + architecture_impact:"unknown", + line_changes:120, + binary_or_unknown_count:0, + layer_roots:["runtime"] + }, + resolution:{ + minimum_tier:"standard", + required_reviewers:["critic","qa-tester","architecture-reviewer"], + recommended_mode:"parallel", + required_mode:null, + downgrade_requested:false, + downgrade_allowed:false + }, + matched_signals:[ + { + id:"consumer-policy", + source:"consumer-policy", + matches:["generic:initial"], + minimum_tier:"express", + required_reviewers:["critic","qa-tester"], + recommended_mode:"sequential", + required_mode:null + }, + { + id:"medium-change", + source:"classification", + matches:["changed-lines:120"], + minimum_tier:"standard", + required_reviewers:["architecture-reviewer"], + recommended_mode:"parallel", + required_mode:null + } + ], + resolved:{ + tier:"standard", + mode:"sequential", + reviewers:["critic","qa-tester","architecture-reviewer"] + }, + enforcement:{status:"pass",violations:[]}, + override:{ + status:"not_provided", + source:null, + sha256:null, + reason:null, + approver:null + }, + reviewer_override:{status:"not_provided",source:null,sha256:null} + }, dispatch:{ outcomes:[{ role:"combined", @@ -823,6 +884,86 @@ case_gate_assurance_invalid_outcome_rejected() { rm -f "$tmpf" } +case_gate_assurance_non_user_policy_approver_rejected() { + local name="gate-assurance: non-user policy approver is rejected" + should_run "$name" || return 0 + local schema_file="$CORE_DIR/schema/gate-assurance.schema.json" tmpf + tmpf="$(mktemp /tmp/gate-assurance-policy-approver-XXXXXX.json)" + _gate_assurance_valid_instance | + jq '.policy.override.approver = { + kind:"project-pm",identity:"fixture-pm",approval_ref:"self:approval" + }' > "$tmpf" + if jsonschema -i "$tmpf" "$schema_file" >/dev/null 2>&1; then + fail "$name" "schema accepted a project-PM policy self-approval" + else + pass "$name" + fi + rm -f "$tmpf" +} + +_gate_policy_override_valid_instance() { + jq -n '{ + kind:"gate_policy_override_v1", + schema_version:1, + scope_fingerprint:("a" * 64), + allow:{ + tier:"express", + omit_reviewers:["security-reviewer"], + mode:null + }, + reason:"User accepted this exact bounded downgrade.", + approver:{ + kind:"user", + identity:"fixture-user", + approval_ref:"conversation:fixture" + } + }' +} + +case_gate_policy_override_valid_instance() { + local name="gate-policy-override: canonical user approval validates" + should_run "$name" || return 0 + local schema_file="$CORE_DIR/schema/gate-policy-override.schema.json" tmpf + tmpf="$(mktemp /tmp/gate-policy-override-valid-XXXXXX.json)" + _gate_policy_override_valid_instance > "$tmpf" + if jsonschema -i "$tmpf" "$schema_file" >/dev/null 2>&1; then + pass "$name" + else + fail "$name" "schema rejected a canonical user-approved override" + fi + rm -f "$tmpf" +} + +case_gate_policy_override_non_user_approver_rejected() { + local name="gate-policy-override: project-PM self-approval is rejected" + should_run "$name" || return 0 + local schema_file="$CORE_DIR/schema/gate-policy-override.schema.json" tmpf + tmpf="$(mktemp /tmp/gate-policy-override-invalid-XXXXXX.json)" + _gate_policy_override_valid_instance | + jq '.approver.kind = "project-pm"' > "$tmpf" + if jsonschema -i "$tmpf" "$schema_file" >/dev/null 2>&1; then + fail "$name" "schema accepted a project-PM policy self-approval" + else + pass "$name" + fi + rm -f "$tmpf" +} + +case_gate_policy_override_extra_key_rejected() { + local name="gate-policy-override: extra contract key is rejected" + should_run "$name" || return 0 + local schema_file="$CORE_DIR/schema/gate-policy-override.schema.json" tmpf + tmpf="$(mktemp /tmp/gate-policy-override-extra-key-XXXXXX.json)" + _gate_policy_override_valid_instance | + jq '.unexpected = true' > "$tmpf" + if jsonschema -i "$tmpf" "$schema_file" >/dev/null 2>&1; then + fail "$name" "schema accepted an undeclared top-level override key" + else + pass "$name" + fi + rm -f "$tmpf" +} + case_context_pack_v1_still_valid case_context_pack_v2_new_fields_valid case_context_pack_memory_source_domain_valid @@ -832,5 +973,9 @@ case_preflight_basic_evidence_needs_no_git_provenance case_preflight_reusable_evidence_requires_fingerprint case_gate_assurance_valid_instance case_gate_assurance_invalid_outcome_rejected +case_gate_assurance_non_user_policy_approver_rejected +case_gate_policy_override_valid_instance +case_gate_policy_override_non_user_approver_rejected +case_gate_policy_override_extra_key_rejected th_summary diff --git a/tests/shell/test-gate-lifecycle.sh b/tests/shell/test-gate-lifecycle.sh index a0879cab..bc99e3f2 100755 --- a/tests/shell/test-gate-lifecycle.sh +++ b/tests/shell/test-gate-lifecycle.sh @@ -335,6 +335,43 @@ case_wait_resolves_go() { fi } +# A parent gate used to export gate_result_verify without its private helper +# closure. The nested wait saw the inherited public function, skipped its own +# library load, and rejected a valid result. The selected pmctl repository must +# replace any inherited same-name function with its complete verifier library. +case_wait_reloads_verifier_over_incomplete_export() { + local name="gate-lifecycle/gate wait reloads verifier over incomplete inherited function" + should_run "$name" || return 0 + + local fixture="$tmp_root/c2b/fixture" work="$tmp_root/c2b/work" + mkdir -p "$work" + _mk_fixture_repo "$fixture" + _mk_fake_gate "$fixture" 0 + + local run_wrapper="$tmp_root/c2b/run" wait_wrapper="$tmp_root/c2b/wait" + _run_gate_wrapper "$fixture" "$run_wrapper" + _wait_wrapper "$fixture" "$wait_wrapper" + + local gate_id out code + gate_id="$("$run_wrapper" --cd "$work" --lifecycle detached)" + set +e + out="$( + # shellcheck disable=SC2317 # exported fixture is invoked by the child shell + gate_result_verify() { _incomplete_inherited_gate_verifier "$@"; } + export -f gate_result_verify + "$wait_wrapper" "$gate_id" --cd "$work" --timeout "$_WAIT_OK" 2>&1 + )" + code=$? + set -e + + if [[ "$code" -eq 0 ]] && [[ "$out" == *"state: GO"* ]] \ + && [[ "$out" == *"Final: GO"* ]]; then + pass "$name" + else + fail "$name" "code=$code out=$out" + fi +} + # ---- 3: gate wait resolves NO-GO (exit 1) ------------------------------------- case_wait_resolves_nogo() { local name="gate-lifecycle/gate wait resolves NO-GO" @@ -569,34 +606,40 @@ case_detached_requires_state_paths() { fi } -# ---- 9: GO sentinel with no result file fails the wait (exit 2) -------------- +# ---- 9: verdict-like exits without a result are infrastructure failures ------ case_wait_fails_on_missing_result() { # CC-423 pr-gate finding (risk-reviewer, high): a wait must not report - # success on a GO/NO-GO state when the sentinel recorded no result file -- - # that state is unverifiable and must not be trusted. - local name="gate-lifecycle/gate wait fails when GO sentinel has no result file" + # GO/NO-GO when the supervisor did not observe the verified `result:` + # handoff. Both an apparent GO (0) and NO-GO (1) are execution failures in + # that state and must be normalized to failed/2 at the producer boundary. + local name="gate-lifecycle/gate wait treats verdict exits without result as failed" should_run "$name" || return 0 - local fixture="$tmp_root/c9/fixture" work="$tmp_root/c9/work" - mkdir -p "$work" - _mk_fixture_repo "$fixture" - _mk_fake_gate_no_result "$fixture" 0 - - local run_wrapper="$tmp_root/c9/run" wait_wrapper="$tmp_root/c9/wait" - _run_gate_wrapper "$fixture" "$run_wrapper" - _wait_wrapper "$fixture" "$wait_wrapper" - - local gate_id - gate_id="$("$run_wrapper" --cd "$work" --lifecycle detached)" - - local out code - set +e; out="$("$wait_wrapper" "$gate_id" --cd "$work" --timeout "$_WAIT_OK" 2>&1)"; code=$?; set -e - - if [[ "$code" -eq 2 ]] && [[ "$out" == *"no result file"* ]]; then - pass "$name" - else - fail "$name" "code=$code out=$out" - fi + local gate_rc fixture work run_wrapper wait_wrapper gate_id out code + for gate_rc in 0 1; do + fixture="$tmp_root/c9-$gate_rc/fixture" + work="$tmp_root/c9-$gate_rc/work" + mkdir -p "$work" + _mk_fixture_repo "$fixture" + _mk_fake_gate_no_result "$fixture" "$gate_rc" + + run_wrapper="$tmp_root/c9-$gate_rc/run" + wait_wrapper="$tmp_root/c9-$gate_rc/wait" + _run_gate_wrapper "$fixture" "$run_wrapper" + _wait_wrapper "$fixture" "$wait_wrapper" + + gate_id="$("$run_wrapper" --cd "$work" --lifecycle detached)" + set +e + out="$("$wait_wrapper" "$gate_id" --cd "$work" --timeout "$_WAIT_OK" 2>&1)" + code=$? + set -e + if [[ "$code" -ne 2 || "$out" != *"state: failed exit: 2"* \ + || "$out" == *"state: GO"* || "$out" == *"state: NO-GO"* ]]; then + fail "$name" "gate_rc=$gate_rc code=$code out=$out" + return + fi + done + pass "$name" } # ---- 10: GO sentinel with a structurally invalid result fails the wait ------- @@ -718,6 +761,7 @@ case_detached_launch_rejects_invalid_ready_timeout case_detached_launch_accepts_terminal_evidence_after_capture_race case_detached_launch_accepts_terminal_evidence_after_liveness_race case_wait_resolves_go +case_wait_reloads_verifier_over_incomplete_export case_wait_resolves_nogo case_wait_resolves_failed case_wait_indeterminate_on_consumed_sentinel diff --git a/tests/shell/test-pmctl-gate.sh b/tests/shell/test-pmctl-gate.sh index f39d33f0..dfd1f9ea 100755 --- a/tests/shell/test-pmctl-gate.sh +++ b/tests/shell/test-pmctl-gate.sh @@ -425,12 +425,64 @@ _mk_gate_result_v2() { tier:{requested:"auto",resolved:"express",evidence_floor:"reviewer-verdicts"}, mode:{requested:"default",resolved:"sequential",topology:"combined-session",synthesis:"inline"}, pass:{requested:"initial",resolved:"initial",scope:"comprehensive",initial_result:null}, - coverage:{requested:null,selected:["critic"],skipped:["qa-tester"], + coverage:{requested:null,selected:["critic","qa-tester"],skipped:[], vocabulary:["critic","qa-tester"]}, independence:{implementation_context_isolated:null, reviewer_topology:"combined-session",per_reviewer_independent:null, evidence_status:"unavailable"} }, + policy:{ + kind:"gate_policy_resolution_v1", + schema_version:1, + consumer_policy:"generic", + policy_source:"canonical", + scope_fingerprint:("f" * 64), + request:{tier:"auto",mode:"default",pass_kind:"initial",reviewers:null}, + classification:{ + architecture_impact:"unknown", + line_changes:1, + binary_or_unknown_count:0, + layer_roots:[] + }, + resolution:{ + minimum_tier:"express", + required_reviewers:["critic","qa-tester"], + recommended_mode:"sequential", + required_mode:null, + downgrade_requested:false, + downgrade_allowed:false + }, + matched_signals:[ + { + id:"consumer-policy", + source:"consumer-policy", + matches:["generic:initial"], + minimum_tier:"express", + required_reviewers:["critic","qa-tester"], + recommended_mode:"sequential", + required_mode:null + }, + { + id:"docs-only", + source:"classification", + matches:["README.md"], + minimum_tier:"express", + required_reviewers:[], + recommended_mode:"sequential", + required_mode:null + } + ], + resolved:{ + tier:"express", + mode:"sequential", + reviewers:["critic","qa-tester"] + }, + enforcement:{status:"pass",violations:[]}, + override:{ + status:"not_provided",source:null,sha256:null,reason:null,approver:null + }, + reviewer_override:{status:"not_provided",source:null,sha256:null} + }, dispatch:{outcomes:[{role:"combined",reviewer:null,status:"passed", run_id:null,evidence_status:"unavailable"}]}, provenance:{producer:"pr-gate.sh",policy_source:"canonical",attestation:null} @@ -511,7 +563,7 @@ _mk_gate_result_v2_legacy_assurance() { jq ' .kind = "gate_assurance_v1" | .schema_version = 1 | - del(.bindings) | + del(.bindings, .policy) | .coordinates.independence = { implementation_context_isolated:true, reviewer_topology:"combined-session", @@ -557,6 +609,21 @@ case_verify_v2_assurance() { fi } +case_verify_v2_without_policy_remains_readable() { + local name="gate/verify: pre-policy v2 assurance remains readable" + should_run "$name" || return 0 + local result="$tmp_root/v2-pre-policy/result.md" out code + _mk_gate_result_v2 "$result" + jq 'del(.policy)' "${result}.assurance.json" > "${result}.assurance.tmp" + mv "${result}.assurance.tmp" "${result}.assurance.json" + set +e; out="$("$PMCTL" gate verify "$result" 2>&1)"; code=$?; set -e + if [[ "$code" -eq 0 && "$out" == *"assurance: verified"* ]]; then + pass "$name" + else + fail "$name" "code=$code out=$out" + fi +} + case_verify_v2_canonical_authorization() { local name="gate/verify: v2 protected attestation and canonical runs exit 0" should_run "$name" || return 0 @@ -620,7 +687,23 @@ case_verify_v2_claim_mismatch() { should_run "$name" || return 0 local result="$tmp_root/v2-mismatch/result.md" out code _mk_gate_result_v2 "$result" - jq '.coordinates.coverage.skipped = []' "${result}.assurance.json" \ + jq '.coordinates.coverage.selected = ["critic"]' "${result}.assurance.json" \ + > "${result}.assurance.tmp" + mv "${result}.assurance.tmp" "${result}.assurance.json" + set +e; out="$("$PMCTL" gate verify "$result" 2>&1)"; code=$?; set -e + if [[ "$code" -eq 1 && "$out" == *"structural/claim verification"* ]]; then + pass "$name" + else + fail "$name" "code=$code out=$out" + fi +} + +case_verify_v2_policy_claim_tamper() { + local name="gate/verify: v2 policy coordinate tamper exits 1" + should_run "$name" || return 0 + local result="$tmp_root/v2-policy-tamper/result.md" out code + _mk_gate_result_v2 "$result" + jq '.policy.resolved.reviewers = ["critic"]' "${result}.assurance.json" \ > "${result}.assurance.tmp" mv "${result}.assurance.tmp" "${result}.assurance.json" set +e; out="$("$PMCTL" gate verify "$result" 2>&1)"; code=$?; set -e @@ -1298,11 +1381,13 @@ case_pmctl_routing case_help_bypasses_detached_default case_verify_valid case_verify_v2_assurance +case_verify_v2_without_policy_remains_readable case_verify_v2_canonical_authorization case_verify_v2_forged_state_tree_rejected case_verify_v2_repo_binding_rejected case_verify_v2_legacy_assurance_is_unavailable case_verify_v2_claim_mismatch +case_verify_v2_policy_claim_tamper case_verify_v2_surplus_topology_record case_verify_v2_unknown_fields_rejected case_verify_v2_result_binding_tamper diff --git a/tests/shell/test-pmctl-ship.sh b/tests/shell/test-pmctl-ship.sh index c7f58bf2..02920577 100755 --- a/tests/shell/test-pmctl-ship.sh +++ b/tests/shell/test-pmctl-ship.sh @@ -1384,10 +1384,13 @@ case_finish_reviewers_flag_reaches_gate_call() { ' _ "$REPO_ROOT" "$work" "CC-9001" "$argv_file" >/dev/null 2>&1 || true local argv argv="$(cat "$argv_file" 2>/dev/null)" - if grep -q -- '--reviewers' <<<"$argv" && grep -Fxq 'critic,qa-tester' <<<"$argv"; then + if grep -q -- '--reviewers' <<<"$argv" \ + && grep -Fxq 'critic,qa-tester' <<<"$argv" \ + && grep -q -- '--policy' <<<"$argv" \ + && grep -Fxq 'maintainer' <<<"$argv"; then pass "$name" else - fail "$name" "expected --reviewers critic,qa-tester in captured gate argv, got: $argv" + fail "$name" "expected maintainer policy plus --reviewers critic,qa-tester in captured gate argv, got: $argv" fi } diff --git a/tests/shell/test-pr-gate-profile.sh b/tests/shell/test-pr-gate-profile.sh index b8859790..4c591239 100755 --- a/tests/shell/test-pr-gate-profile.sh +++ b/tests/shell/test-pr-gate-profile.sh @@ -89,6 +89,10 @@ if [[ -n "$output_path" ]]; then mkdir -p "$(dirname "$output_path")" if [[ "$brief_file" == *-synthesis.md ]]; then printf -- '---\ngate_result_version: pr_gate_result_v1\nfinal: GO\ntier: standard\nmode: parallel\nmost_severe: advise\nreviewers:\n critic: advise\nescalation:\n recommended: false\n reviewers: []\n reason: []\n---\n\n# PR-Gate Result — stub tier\n**Date**: 2026-05-17\n**Reviewers**: stub\n**Not reviewed**: none\n\n## cross-check\nnone\n\n## Gate Conclusion\n**Overall verdict**: advise\n**Most severe individual verdict**: advise\nFinal: GO\n' > "$output_path" + elif reviewer_name="$(awk '$1 == "Reviewer:" { print $2; exit }' "$brief_file")" \ + && [[ -n "$reviewer_name" ]]; then + printf '## %s -- advise\n\nstatus: advise\nfindings: []\nverdict: Stub output.\n' \ + "$reviewer_name" > "$output_path" else printf -- '---\ngate_result_version: pr_gate_result_v1\nfinal: GO\ntier: standard\nmode: sequential\nmost_severe: advise\nreviewers:\n critic: advise\nescalation:\n recommended: false\n reviewers: []\n reason: []\n---\n\n## stub-reviewer — advise\nVerdict: advise. Stub output.\nFinal: GO\n' > "$output_path" fi @@ -140,6 +144,10 @@ if [[ -n "$output_path" ]]; then mkdir -p "$(dirname "$output_path")" if [[ "$brief_file" == *-synthesis.md ]]; then printf -- '---\ngate_result_version: pr_gate_result_v1\nfinal: GO\ntier: standard\nmode: parallel\nmost_severe: advise\nreviewers:\n critic: advise\nescalation:\n recommended: false\n reviewers: []\n reason: []\n---\n\n# PR-Gate Result — stub tier\n**Date**: 2026-05-17\n**Reviewers**: stub\n**Not reviewed**: none\n\n## cross-check\nnone\n\n## Gate Conclusion\n**Overall verdict**: advise\n**Most severe individual verdict**: advise\nFinal: GO\n' > "$output_path" + elif reviewer_name="$(awk '$1 == "Reviewer:" { print $2; exit }' "$brief_file")" \ + && [[ -n "$reviewer_name" ]]; then + printf '## %s -- advise\n\nstatus: advise\nfindings: []\nverdict: Stub output.\n' \ + "$reviewer_name" > "$output_path" else printf -- '---\ngate_result_version: pr_gate_result_v1\nfinal: GO\ntier: standard\nmode: sequential\nmost_severe: advise\nreviewers:\n critic: advise\nescalation:\n recommended: false\n reviewers: []\n reason: []\n---\n\n## stub-reviewer — advise\nVerdict: advise. Stub output.\nFinal: GO\n' > "$output_path" fi @@ -286,7 +294,10 @@ test_executor_claude_parallel_dispatches_subprocess() { create_repo "$repo" set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" "$runner" --executor claude --reviewers critic,qa-tester --parallel --base main + run_gate "$home" "$runner" "$repo" "$out" "$err" "$runner" \ + --executor claude \ + --reviewers critic,qa-tester,architecture-reviewer \ + --parallel --base main local code=$? set -e if [[ "$code" -ne 0 ]]; then diff --git a/tests/shell/test-pr-gate.sh b/tests/shell/test-pr-gate.sh index 962575cf..b04ca11b 100755 --- a/tests/shell/test-pr-gate.sh +++ b/tests/shell/test-pr-gate.sh @@ -71,6 +71,8 @@ create_runner() { cp "$REPO_ROOT/core/policy/gate-tiers.tsv" "$dir/core/policy/gate-tiers.tsv" cp "$REPO_ROOT/core/policy/gate-modes.tsv" "$dir/core/policy/gate-modes.tsv" cp "$REPO_ROOT/core/policy/gate-pass-kinds.tsv" "$dir/core/policy/gate-pass-kinds.tsv" + cp "$REPO_ROOT/core/policy/gate-policy-consumers.tsv" "$dir/core/policy/gate-policy-consumers.tsv" + cp "$REPO_ROOT/core/policy/gate-policy-signals.tsv" "$dir/core/policy/gate-policy-signals.tsv" mkdir -p "$dir/adapters/codex" cat > "$dir/adapters/codex/dispatch.sh" <<'STUB_EOF' #!/usr/bin/env bash @@ -90,6 +92,9 @@ while [[ $# -gt 0 ]]; do esac done +reviewer_name="$(awk '$1 == "Reviewer:" { print $2; exit }' "$brief_file")" +: "${reviewer_name:=stub-reviewer}" + printf 'DISPATCH_STUB:%s\n' "${CODEX_GATE_STUB_MODE:-success}" if [[ -n "${CODEX_GATE_BRIEF_EXISTS_MARKER:-}" ]]; then @@ -116,7 +121,10 @@ if [[ -n "${CODEX_GATE_CAPTURE_BRIEF:-}" ]]; then fi if [[ -n "${CODEX_GATE_CAPTURE_REVIEWER_BRIEF:-}" && "$brief_file" != *-synthesis.md ]]; then - cp "$brief_file" "$CODEX_GATE_CAPTURE_REVIEWER_BRIEF" + if [[ -z "${CODEX_GATE_CAPTURE_REVIEWER_FILTER:-}" \ + || "$brief_file" == *-"${CODEX_GATE_CAPTURE_REVIEWER_FILTER}".md ]]; then + cp "$brief_file" "$CODEX_GATE_CAPTURE_REVIEWER_BRIEF" + fi fi if [[ -n "${CODEX_GATE_REVIEWER_DEFS_MARKER:-}" && "$brief_file" != *-synthesis.md ]]; then @@ -173,7 +181,35 @@ if [[ "${CODEX_GATE_STUB_VERDICT_PREFIX_ONLY:-}" == "1" && "$brief_file" != *-sy output_path=$(grep -o '\- new:.*' "$brief_file" | head -1 | awk '{print $NF}') if [[ -n "$output_path" ]]; then mkdir -p "$(dirname "$output_path")" - printf '## stub-reviewer -- approved\nVerdict: approved. Prefix-only bypass attempt.\n' > "$output_path" + printf '## %s -- approved\nVerdict: approved. Prefix-only bypass attempt.\n' \ + "$reviewer_name" > "$output_path" + fi + exit 0 +fi + +# Simulate the structured output emitted by base-pinned reviewer definitions: +# the machine verdict is in the canonical heading and the definition's +# lower-case `verdict:` field is narrative rather than a second token. +if [[ "${CODEX_GATE_STUB_HEADER_ONLY_VERDICT:-}" == "1" \ + && "$brief_file" != *-synthesis.md ]]; then + output_path=$(grep -o '\- new:.*' "$brief_file" | head -1 | awk '{print $NF}') + if [[ -n "$output_path" ]]; then + mkdir -p "$(dirname "$output_path")" + printf '## %s -- advise\n\nstatus: advise\nfindings: []\nverdict: Structured narrative.\n' \ + "$reviewer_name" > "$output_path" + fi + exit 0 +fi + +# Simulate a conflicting optional legacy Verdict marker. The heading remains +# authoritative, but disagreement must abort rather than silently choose one. +if [[ "${CODEX_GATE_STUB_CONFLICTING_VERDICT:-}" == "1" \ + && "$brief_file" != *-synthesis.md ]]; then + output_path=$(grep -o '\- new:.*' "$brief_file" | head -1 | awk '{print $NF}') + if [[ -n "$output_path" ]]; then + mkdir -p "$(dirname "$output_path")" + printf '## %s -- approve\nVerdict: block. Conflicting marker.\n' \ + "$reviewer_name" > "$output_path" fi exit 0 fi @@ -185,7 +221,8 @@ if [[ "${CODEX_GATE_STUB_MULTIPLE_VERDICTS:-}" == "1" && "$brief_file" != *-synt output_path=$(grep -o '\- new:.*' "$brief_file" | head -1 | awk '{print $NF}') if [[ -n "$output_path" ]]; then mkdir -p "$(dirname "$output_path")" - printf '## stub-reviewer -- approve\nVerdict: approve. First verdict line.\nSome additional content.\nVerdict: block. Second verdict line.\n' > "$output_path" + printf '## %s -- approve\nVerdict: approve. First verdict line.\nSome additional content.\nVerdict: block. Second verdict line.\n' \ + "$reviewer_name" > "$output_path" fi exit 0 fi @@ -348,7 +385,8 @@ PARTIAL_EOF fi # Reviewer brief: CODEX_GATE_STUB_VERDICT controls the verdict line (default advise). stub_verdict="${CODEX_GATE_STUB_VERDICT:-advise}" - printf '## stub-reviewer -- %s\nVerdict: %s. Stub output.\n' "$stub_verdict" "$stub_verdict" > "$output_path" + printf '## %s -- %s\nVerdict: %s. Stub output.\n' \ + "$reviewer_name" "$stub_verdict" "$stub_verdict" > "$output_path" if [[ "$(basename "$output_path")" == pr-gate-result-* || "$(basename "$output_path")" == gate-* ]]; then printf 'Final: GO\n' >> "$output_path" fi @@ -491,7 +529,7 @@ create_repo_with_branch() { git commit -q -m "add large code" ;; full-sensitive) - # sensitive filename → full tier regardless of line count + # Bounded sensitive filename → signal-specific security coverage. printf 'package main\n' > auth-handler.go git add auth-handler.go git commit -q -m "add auth handler" @@ -536,14 +574,14 @@ INITIAL_GATE_EOF } # Behavior: the bounded copy-mode policy snapshot is byte-for-byte equivalent -# to all three canonical gate policy TSV sources. +# to all canonical gate policy TSV sources. # Steps: extract each generated heredoc from pr-gate.sh, compare it with the # matching core/policy file, and fail on any drift. test_gate_assurance_policy_snapshot_matches_sources() { local name="gate-assurance-policy-snapshot-matches-sources" should_run "$name" || return 0 local table delimiter source snapshot - for table in tiers modes pass-kinds; do + for table in tiers modes pass-kinds consumers signals; do case "$table" in tiers) delimiter="GATE_ASSURANCE_TIERS_TSV" @@ -557,6 +595,14 @@ test_gate_assurance_policy_snapshot_matches_sources() { delimiter="GATE_ASSURANCE_PASS_KINDS_TSV" source="$REPO_ROOT/core/policy/gate-pass-kinds.tsv" ;; + consumers) + delimiter="GATE_POLICY_CONSUMERS_TSV" + source="$REPO_ROOT/core/policy/gate-policy-consumers.tsv" + ;; + signals) + delimiter="GATE_POLICY_SIGNALS_TSV" + source="$REPO_ROOT/core/policy/gate-policy-signals.tsv" + ;; esac snapshot="$(awk -v marker="$delimiter" ' index($0, "cat <<\047" marker "\047") { inside=1; next } @@ -571,16 +617,17 @@ test_gate_assurance_policy_snapshot_matches_sources() { pass "$name" } -# Behavior: a repo-layout/copy-bundle policy source controls tier-default -# reviewer selection instead of the generated fallback or a hardcoded case. -# Steps: change only the copied express default to critic, run a docs gate, and -# assert the captured brief selects critic while retaining express tier. -test_gate_tier_policy_source_controls_default_coverage() { - local name="gate-tier-policy-source-controls-default-coverage" +# Behavior: repo-layout policy sources jointly control tier-default and consumer +# required coverage instead of a generated fallback or hardcoded branch. +# Steps: narrow both the copied express default and generic-initial requirement +# to critic, then assert the captured brief selects only critic. +test_gate_policy_sources_control_default_coverage() { + local name="gate-policy-sources-control-default-coverage" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" rewritten="$dir/gate-tiers.tsv" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local rewritten="$dir/gate-tiers.tsv" rewritten_consumer="$dir/gate-policy-consumers.tsv" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer @@ -590,6 +637,11 @@ test_gate_tier_policy_source_controls_default_coverage() { { print } ' "$runner/core/policy/gate-tiers.tsv" > "$rewritten" mv "$rewritten" "$runner/core/policy/gate-tiers.tsv" + awk -F '\t' -v OFS='\t' ' + $1 == "generic:initial" { $5="critic" } + { print } + ' "$runner/core/policy/gate-policy-consumers.tsv" > "$rewritten_consumer" + mv "$rewritten_consumer" "$runner/core/policy/gate-policy-consumers.tsv" set +e CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main @@ -602,7 +654,7 @@ test_gate_tier_policy_source_controls_default_coverage() { assert_file_contains "$name" "$brief" "Tier: express" || return assert_file_contains "$name" "$brief" "coverage.selected: critic" || return assert_file_contains "$name" "$brief" "Reviewers: critic" || return - assert_not_contains "$name" "$brief" "Process each reviewer IN ORDER: critic,qa-tester" || return + assert_file_contains "$name" "$brief" "policy.required_reviewers: critic" || return pass "$name" } @@ -623,7 +675,9 @@ test_gate_assurance_policy_snapshot_is_copy_mode_fallback() { rm -f \ "$runner/core/policy/gate-tiers.tsv" \ "$runner/core/policy/gate-modes.tsv" \ - "$runner/core/policy/gate-pass-kinds.tsv" + "$runner/core/policy/gate-pass-kinds.tsv" \ + "$runner/core/policy/gate-policy-consumers.tsv" \ + "$runner/core/policy/gate-policy-signals.tsv" set +e CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main @@ -640,6 +694,207 @@ test_gate_assurance_policy_snapshot_is_copy_mode_fallback() { pass "$name" } +# Behavior: every policy row is validated before dispatch, even when its signal +# would not match the current diff. +test_dormant_policy_signal_with_unknown_reviewer_fails_before_dispatch() { + local name="dormant-policy-signal-with-unknown-reviewer-fails-before-dispatch" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + printf '%s\n' \ + $'dormant-signal\tpath-regex\tnever-match-this-fixture\tstandard\tunknown-reviewer\tparallel\tnone' \ + >> "$runner/core/policy/gate-policy-signals.tsv" + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main + local code=$? + set -e + if [[ "$code" -ne 2 ]]; then + fail "$name" "exit $code, expected policy-source failure 2" + return + fi + assert_file_contains "$name" "$err" \ + "signal dormant-signal names unknown reviewer unknown-reviewer" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: signal IDs form a closed unique inventory before any one signal is +# matched or copied into an assurance artifact. +test_duplicate_policy_signal_id_fails_before_dispatch() { + local name="duplicate-policy-signal-id-fails-before-dispatch" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + printf '%s\n' \ + $'docs-only\tpath-regex\tnever-match-this-fixture\texpress\tnone\tsequential\tnone' \ + >> "$runner/core/policy/gate-policy-signals.tsv" + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main + local code=$? + set -e + if [[ "$code" -ne 2 ]]; then + fail "$name" "exit $code, expected policy-source failure 2" + return + fi + assert_file_contains "$name" "$err" \ + "invalid gate policy signals source" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: the maintainer initial-pass policy fixes reviewer coverage at all +# five dimensions without rewriting express tier or requiring parallel mode. +test_maintainer_initial_policy_fixes_coverage_only() { + local name="maintainer-initial-policy-fixes-coverage-only" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" result_path + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --policy maintainer + local code=$? + set -e + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" + return + fi + assert_file_contains "$name" "$brief" "tier.resolved: express" || return + assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return + assert_file_contains "$name" "$brief" "policy.consumer: maintainer" || return + assert_file_contains "$name" "$brief" "policy.recommended_mode: parallel" || return + assert_file_contains "$name" "$brief" "policy.required_mode: none" || return + assert_file_contains "$name" "$brief" \ + "coverage.selected: critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer" || return + result_path="$(awk '/^result: / {sub(/^result: /, ""); print; exit}' "$out")" + jq -e ' + .policy.consumer_policy == "maintainer" and + .policy.resolved.tier == "express" and + .policy.resolved.mode == "sequential" and + .policy.resolved.reviewers == + ["critic","qa-tester","architecture-reviewer","security-reviewer","risk-reviewer"] + ' "${result_path}.assurance.json" >/dev/null || { + fail "$name" "maintainer policy coordinates were not preserved in assurance" + return + } + pass "$name" +} + +# Behavior: the maintainer targeted-pass policy scopes coverage to requested +# remediation reviewers instead of silently expanding back to all five. +test_maintainer_targeted_policy_preserves_remediation_scope() { + local name="maintainer-targeted-policy-preserves-remediation-scope" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local initial="$dir/initial.md" result_path + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" clean + printf 'package auth\n' > "$repo/auth-handler.go" + write_valid_initial_gate_result "$initial" + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --policy maintainer --targeted critic --initial-result "$initial" + local code=$? + set -e + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" + return + fi + assert_file_contains "$name" "$brief" "policy.consumer: maintainer" || return + assert_file_contains "$name" "$brief" "pass.resolved: targeted" || return + assert_file_contains "$name" "$brief" "coverage.selected: critic" || return + assert_file_contains "$name" "$brief" "policy.required_reviewers: none" || return + result_path="$(awk '/^result: / {sub(/^result: /, ""); print; exit}' "$out")" + jq -e ' + any(.policy.matched_signals[]; + .id == "security-sensitive-path" and + .matches == ["auth-handler.go"] and + .required_reviewers == []) + ' "${result_path}.assurance.json" >/dev/null || { + fail "$name" "targeted policy did not retain the security signal as non-expanding evidence" + return + } + pass "$name" +} + +# Behavior: an explicit input/execution boundary signal requires parallel +# isolation independently of its standard tier and security coverage. +test_input_execution_signal_requires_parallel_mode() { + local name="input-execution-signal-requires-parallel-mode" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" result_path + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" clean + mkdir -p "$repo/.github/workflows" + printf '#!/usr/bin/env bash\nprintf safe\n' > "$repo/command-runner.sh" + printf 'name: fixture\n' > "$repo/.github/workflows/ci.yml" + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --output "$dir/parallel-result.md" + local code=$? + set -e + if [[ "$code" -ne 0 ]]; then + fail "$name" "auto mode exited $code, expected 0" + return + fi + assert_file_contains "$name" "$brief" "tier.resolved: standard" || return + assert_file_contains "$name" "$brief" "mode.resolved: parallel" || return + assert_file_contains "$name" "$brief" "policy.required_mode: parallel" || return + assert_file_contains "$name" "$brief" "policy.escalation_signals:" || return + assert_file_contains "$name" "$brief" '"id":"input-execution-path"' || return + assert_not_contains "$name" "$brief" "any diff file matches (" || return + assert_file_contains "$name" "$brief" \ + "coverage.selected: critic,qa-tester,architecture-reviewer,security-reviewer" || return + result_path="$dir/parallel-result.md" + jq -e ' + any(.policy.matched_signals[]; + .id == "input-execution-path" and + (.matches | index(".github/workflows/ci.yml")) != null) + ' "${result_path}.assurance.json" >/dev/null || { + fail "$name" "CI execution path was not recorded as an isolation signal" + return + } + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --mode sequential + code=$? + set -e + if [[ "$code" -ne 3 ]]; then + fail "$name" "explicit sequential mode exited $code, expected policy rejection 3" + return + fi + assert_file_contains "$name" "$err" "requested=sequential required=parallel" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + # Behavior: express-tier diff with no overrides routes to codex with the # express reviewer set (critic, qa-tester). # Steps: run the gate on a docs-only diff, assert stderr shows dispatch @@ -1101,77 +1356,374 @@ test_no_changed_files() { pass "$name" } -# Behavior: an explicit --reviewers list overrides requested coverage without -# changing auto-detected tier or review pass kind. The parallel synthesis brief -# embeds only selected reviewer findings inline. -# Steps: run the gate with --reviewers critic --parallel against a diff that -# tier-detects to standard, and assert standard/initial coordinates, critic-only -# coverage, inline findings, and no reviewer output read paths. -test_reviewers_override_preserves_tier_detection() { - local name="reviewers-override" +# Behavior: an explicit reviewer selection below the canonical risk floor fails +# before dispatch instead of being mistaken for policy-sufficient coverage. +# Steps: request critic-only coverage for a medium runtime diff whose policy +# requires critic, QA, and architecture; assert the violation is diagnostic. +test_reviewers_override_below_policy_floor_fails_closed() { + local name="reviewers-override-below-policy-floor" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local out="$dir/out" err="$dir/err" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic create_repo_with_branch "$repo" standard set +e - CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel local code=$? set -e + if [[ "$code" -eq 0 ]]; then + fail "$name" "critic-only policy downgrade was accepted without user authorization" + return + fi + assert_file_contains "$name" "$err" "below the canonical generic policy floor" || return + assert_file_contains "$name" "$err" "coverage" || return + assert_file_contains "$name" "$err" "qa-tester" || return + assert_file_contains "$name" "$err" "architecture-reviewer" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: the public CLI rejects an unknown policy consumer before any +# repository work or reviewer dispatch. +test_invalid_policy_consumer_fails_before_dispatch() { + local name="invalid-policy-consumer-fails-before-dispatch" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester + create_repo "$repo" docs + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --policy bogus + local code=$? + set -e + if [[ "$code" -ne 2 ]]; then + fail "$name" "exit $code, expected CLI failure 2" + return + fi + assert_file_contains "$name" "$err" \ + "Error: --policy must be generic or maintainer (got: bogus)" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: an empty policy-override file is rejected at the CLI trust +# boundary before policy resolution or reviewer dispatch. +test_empty_policy_override_fails_before_dispatch() { + local name="empty-policy-override-fails-before-dispatch" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" policy_override="$dir/empty.json" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester + create_repo "$repo" docs + printf '' > "$policy_override" + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --policy-override "$policy_override" + local code=$? + set -e + if [[ "$code" -ne 2 ]]; then + fail "$name" "exit $code, expected CLI failure 2" + return + fi + assert_file_contains "$name" "$err" \ + "--policy-override must name a readable, non-empty, regular non-symlink JSON file" \ + || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: the runtime's inline policy-override validator rejects a non-empty +# JSON document that does not satisfy gate_policy_override_v1. +test_malformed_policy_override_contract_fails_before_dispatch() { + local name="malformed-policy-override-contract-fails-before-dispatch" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" policy_override="$dir/malformed.json" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester + create_repo "$repo" docs + jq -n '{ + kind:"gate_policy_override_v1", + schema_version:1, + scope_fingerprint:("a" * 64), + allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:test"} + }' > "$policy_override" + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --policy-override "$policy_override" + local code=$? + set -e + if [[ "$code" -ne 2 ]]; then + fail "$name" "exit $code, expected contract failure 2" + return + fi + assert_file_contains "$name" "$err" \ + "Error: invalid gate policy override contract: $policy_override" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: a structured user-approved override may authorize an exact +# scope-bound reviewer omission without rewriting full-tier intent. +# Steps: capture the rejected scope fingerprint, bind a user approval to that +# scope and qa-tester omission, then assert the policy audit is embedded. +test_scope_bound_policy_override_authorizes_exact_coverage_downgrade() { + local name="scope-bound-policy-override" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local policy_override="$dir/policy-override.json" scope_fingerprint result_path + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic + create_repo "$repo" docs + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --tier full --reviewers critic --mode sequential + local code=$? + set -e + if [[ "$code" -eq 0 ]]; then + fail "$name" "coverage downgrade unexpectedly passed without an override" + return + fi + scope_fingerprint="$(awk '/policy scope fingerprint:/ {print $NF; exit}' "$err")" + [[ "$scope_fingerprint" =~ ^[a-f0-9]{64}$ ]] || { + fail "$name" "rejection did not disclose a usable scope fingerprint" + return + } + jq -n --arg scope "$scope_fingerprint" '{ + kind:"gate_policy_override_v1", + schema_version:1, + scope_fingerprint:$scope, + allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + reason:"User accepts critic-only coverage for this bounded fixture.", + approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:test"} + }' > "$policy_override" + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --tier full --reviewers critic --mode sequential \ + --policy-override "$policy_override" + code=$? + set -e if [[ "$code" -ne 0 ]]; then - fail "$name" "exit $code, expected 0" + fail "$name" "scope-bound user override was rejected (exit $code): $(cat "$err")" return fi - # Parallel mode: CAPTURE_BRIEF receives the synthesis brief (last dispatch) - assert_file_contains "$name" "$brief" "Tier: standard" || return - assert_file_contains "$name" "$brief" "tier.requested: auto" || return - assert_file_contains "$name" "$brief" "tier.resolved: standard" || return - assert_file_contains "$name" "$brief" "pass.resolved: initial" || return + assert_file_contains "$name" "$brief" "tier.requested: full" || return + assert_file_contains "$name" "$brief" "tier.resolved: full" || return + assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return assert_file_contains "$name" "$brief" "coverage.requested: critic" || return assert_file_contains "$name" "$brief" "coverage.selected: critic" || return - assert_file_contains "$name" "$brief" "Executor: codex" || return assert_file_contains "$name" "$brief" "Reviewers: critic" || return - # Synthesis brief embeds reviewer findings inline — no read: paths to reviewer output files - assert_file_contains "$name" "$brief" "--- critic findings ---" || return - assert_not_contains "$name" "$brief" "reviewer-critic-" || return - assert_not_contains "$name" "$brief" "read: $home/.claude/agents/qa-tester.md" || return + result_path="$(awk '/^result: / {sub(/^result: /, ""); print; exit}' "$out")" + jq -e ' + .policy.resolution.downgrade_requested == true and + .policy.resolution.downgrade_allowed == true and + .policy.override.status == "applied" and + .policy.override.approver.kind == "user" and + .policy.enforcement.status == "pass" + ' "${result_path}.assurance.json" >/dev/null || { + fail "$name" "assurance sidecar did not retain applied override provenance" + return + } + pass "$name" +} + +# Behavior: policy approval for any other scope cannot authorize the current +# downgrade, even when its allowance fields exactly match the violation. +test_policy_override_scope_mismatch_fails_closed() { + local name="policy-override-scope-mismatch-fails-closed" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" policy_override="$dir/policy-override.json" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic + create_repo "$repo" docs + jq -n '{ + kind:"gate_policy_override_v1", + schema_version:1, + scope_fingerprint:("0" * 64), + allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + reason:"Approval belongs to a different change scope.", + approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:other"} + }' > "$policy_override" + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic --policy-override "$policy_override" + local code=$? + set -e + if [[ "$code" -ne 3 ]]; then + fail "$name" "exit $code, expected policy rejection 3" + return + fi + assert_file_contains "$name" "$err" "supplied override status: scope_mismatch" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: a policy override cannot be replayed after diff content changes, +# even when the changed path, status, total line count, and byte count stay the +# same. +test_policy_override_scope_binds_diff_content() { + local name="policy-override-scope-binds-diff-content" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" policy_override="$dir/policy-override.json" + local original_scope current_scope + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic + create_repo "$repo" docs + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic + local code=$? + set -e + if [[ "$code" -ne 3 ]]; then + fail "$name" "initial downgrade exited $code, expected policy rejection 3" + return + fi + original_scope="$(awk '/policy scope fingerprint:/ {print $NF; exit}' "$err")" + [[ "$original_scope" =~ ^[a-f0-9]{64}$ ]] || { + fail "$name" "initial rejection did not disclose a usable scope fingerprint" + return + } + jq -n --arg scope "$original_scope" '{ + kind:"gate_policy_override_v1", + schema_version:1, + scope_fingerprint:$scope, + allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + reason:"Approval is intentionally bound to the original fixture content.", + approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:content"} + }' > "$policy_override" + + # Same path, status, lines, and bytes; only the patch content changes. + printf 'initial\ndocs CHANGE\n' > "$repo/README.md" + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic --policy-override "$policy_override" + code=$? + set -e + if [[ "$code" -ne 3 ]]; then + fail "$name" "content-changed scope exited $code, expected policy rejection 3" + return + fi + current_scope="$(awk '/policy scope fingerprint:/ {print $NF; exit}' "$err")" + if [[ ! "$current_scope" =~ ^[a-f0-9]{64}$ \ + || "$current_scope" == "$original_scope" ]]; then + fail "$name" "scope fingerprint did not change with same-shape diff content" + return + fi + assert_file_contains "$name" "$err" \ + "supplied override status: scope_mismatch" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + pass "$name" +} + +# Behavior: an override bound to the current scope still fails closed when its +# allowance does not exactly cover the requested downgrade. +test_policy_override_allowance_mismatch_fails_closed() { + local name="policy-override-allowance-mismatch-fails-closed" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" policy_override="$dir/policy-override.json" + local scope_fingerprint + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic + create_repo "$repo" docs + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic + local code=$? + set -e + if [[ "$code" -ne 3 ]]; then + fail "$name" "initial downgrade exited $code, expected policy rejection 3" + return + fi + scope_fingerprint="$(awk '/policy scope fingerprint:/ {print $NF; exit}' "$err")" + [[ "$scope_fingerprint" =~ ^[a-f0-9]{64}$ ]] || { + fail "$name" "initial rejection did not disclose a usable scope fingerprint" + return + } + jq -n --arg scope "$scope_fingerprint" '{ + kind:"gate_policy_override_v1", + schema_version:1, + scope_fingerprint:$scope, + allow:{tier:null,omit_reviewers:[],mode:null}, + reason:"This allowance intentionally omits none of the missing reviewers.", + approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:mismatch"} + }' > "$policy_override" + + set +e + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic --policy-override "$policy_override" + code=$? + set -e + if [[ "$code" -ne 3 ]]; then + fail "$name" "exit $code, expected policy rejection 3" + return + fi + assert_file_contains "$name" "$err" \ + "supplied override status: allowance_mismatch" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return pass "$name" } -# Behavior: explicit full-tier intent and critic-only requested coverage remain -# independent; neither value rewrites the other. -# Steps: run a docs diff with --tier full --reviewers critic, then assert the -# combined brief records full requested/resolved tier and critic-only coverage. -test_full_tier_with_critic_only_coverage_is_truthful() { - local name="full-tier-with-critic-only-coverage-is-truthful" +# Behavior: policy resolution has one producer call site and its fail-closed +# enforcement check precedes every resolved-coordinate consumer. +test_policy_enforcement_precedes_resolved_coordinate_consumers() { + local name="policy-enforcement-precedes-resolved-coordinate-consumers" should_run "$name" || return 0 - local dir="$TMP_ROOT/$name" - local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" - mkdir -p "$dir" - create_runner "$runner" - create_agents "$home" critic - create_repo "$repo" docs + local gate="$REPO_ROOT/runtime/bin/pr-gate.sh" + local assignment_count assignment_line enforcement_count enforcement_line + local first_consumer_line + assignment_count="$(grep -c '^GATE_POLICY_RESOLUTION=' "$gate" || true)" + enforcement_count="$(grep -c "jq -r '.enforcement.status'" "$gate" || true)" + assignment_line="$(grep -n '^GATE_POLICY_RESOLUTION=' "$gate" \ + | cut -d: -f1 | head -1)" + enforcement_line="$(grep -n "jq -r '.enforcement.status'" "$gate" \ + | cut -d: -f1 | head -1)" + first_consumer_line="$(grep -n '^TIER_RESOLVED=' "$gate" \ + | cut -d: -f1 | head -1)" - set +e - CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ - --base main --tier full --reviewers critic --mode sequential - local code=$? - set -e - if [[ "$code" -ne 0 ]]; then - fail "$name" "exit $code, expected 0" + if [[ "$assignment_count" -ne 1 || "$enforcement_count" -ne 1 \ + || ! "$assignment_line" =~ ^[0-9]+$ \ + || ! "$enforcement_line" =~ ^[0-9]+$ \ + || ! "$first_consumer_line" =~ ^[0-9]+$ \ + || "$assignment_line" -ge "$enforcement_line" \ + || "$enforcement_line" -ge "$first_consumer_line" ]]; then + fail "$name" \ + "assignment=$assignment_count@$assignment_line enforcement=$enforcement_count@$enforcement_line first-consumer=$first_consumer_line" return fi - assert_file_contains "$name" "$brief" "tier.requested: full" || return - assert_file_contains "$name" "$brief" "tier.resolved: full" || return - assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return - assert_file_contains "$name" "$brief" "coverage.requested: critic" || return - assert_file_contains "$name" "$brief" "coverage.selected: critic" || return - assert_file_contains "$name" "$brief" "Reviewers: critic" || return pass "$name" } @@ -1403,7 +1955,7 @@ test_parallel_timeout_kills_hanging_reviewer() { local out="$dir/out" err="$dir/err" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs set +e @@ -1414,7 +1966,8 @@ test_parallel_timeout_kills_hanging_reviewer() { _PM_DISPATCH_GATE_WATCHDOG_TIMEOUT=2 \ CODEX_GATE_STUB_MODE=hang \ CODEX_GATE_HANG_SECONDS="$marker" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -eq 0 ]]; then @@ -1442,7 +1995,7 @@ test_parallel_timeout_kills_hanging_synthesis() { local out="$dir/out" err="$dir/err" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs set +e @@ -1453,7 +2006,8 @@ test_parallel_timeout_kills_hanging_synthesis() { _PM_DISPATCH_GATE_SYNTHESIS_WATCHDOG_TIMEOUT=2 \ CODEX_GATE_STUB_SYNTHESIS_MODE=hang \ CODEX_GATE_HANG_SECONDS="$marker" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -eq 0 ]]; then @@ -1536,8 +2090,8 @@ test_sequential_combined_brief_validates() { # Behavior: each parallel per-reviewer brief satisfies brief-validate.sh # (same dispatch contract the reviewer executor validates first). -# Steps: run the gate with --reviewers critic --parallel, capture the -# per-reviewer brief, run brief-validate.sh on it, and assert it exits 0. +# Steps: run the gate with the generic docs policy coverage in parallel, +# capture the critic brief, run brief-validate.sh on it, and assert it exits 0. test_parallel_reviewer_brief_validates() { local name="parallel-reviewer-brief-validates" should_run "$name" || return 0 @@ -1546,12 +2100,14 @@ test_parallel_reviewer_brief_validates() { local out="$dir/out" err="$dir/err" reviewer_brief="$dir/reviewer-brief.md" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs set +e CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$reviewer_brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + CODEX_GATE_CAPTURE_REVIEWER_FILTER=critic \ + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -2026,7 +2582,63 @@ test_reviewer_invalid_verdict_aborts_gate() { fail "$name" "expected non-zero exit when reviewer output has no valid Verdict line" return fi - assert_file_contains "$name" "$err" "exactly one valid Verdict line" || return + assert_file_contains "$name" "$err" \ + "invalid or ambiguous canonical verdict" || return + pass "$name" +} + +# Behavior: base-pinned reviewer definitions may emit their narrative verdict +# as lower-case YAML while the canonical machine verdict remains in the +# reviewer-matched heading. +test_reviewer_heading_only_verdict_is_accepted() { + local name="reviewer-heading-only-verdict-is-accepted" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + + set +e + CODEX_GATE_STUB_HEADER_ONLY_VERDICT=1 \ + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --parallel + local code=$? + set -e + if [[ "$code" -ne 0 ]]; then + fail "$name" "heading-only structured verdict exited $code: $(cat "$err")" + return + fi + assert_file_contains "$name" "$out" "result: " || return + pass "$name" +} + +# Behavior: when an optional upper-case Verdict marker conflicts with the +# canonical heading, the gate fails closed before synthesis. +test_reviewer_heading_and_explicit_verdict_must_agree() { + local name="reviewer-heading-and-explicit-verdict-must-agree" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + + set +e + CODEX_GATE_STUB_CONFLICTING_VERDICT=1 \ + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --parallel + local code=$? + set -e + if [[ "$code" -eq 0 ]]; then + fail "$name" "conflicting heading and Verdict marker unexpectedly passed" + return + fi + assert_file_contains "$name" "$err" \ + "invalid or ambiguous canonical verdict" || return + assert_not_contains "$name" "$out" "[synthesis]" || return pass "$name" } @@ -2905,7 +3517,10 @@ if [[ "$brief_file" == *-synthesis.md ]]; then printf -- '---\ngate_result_version: pr_gate_result_v1\nfinal: GO\ntier: full\nmode: parallel\nmost_severe: advise\nreviewers:\n critic: skipped\n qa-tester: skipped\n architecture-reviewer: skipped\n security-reviewer: skipped\n risk-reviewer: skipped\nescalation:\n recommended: false\n reviewers: []\n reason: []\n---\n# PR-Gate Result\n**Date**: 2026-01-01\n**Reviewers**: stub\n**Not reviewed**: none\n\n## stub-reviewer -- advise\n- stub finding\n\nVerdict: advise. Stub.\n\n## Cross-Reviewer Overlaps\nnone\n\n## Coverage Notes\n**Dimensions not covered**: none\n\n## Gate Conclusion\n**Overall verdict**: advise\n**Most severe individual verdict**: advise\nFinal: GO\n\n## Escalation\n**Recommended**: false\n**Reviewers**: none\n**Reason**:\n- none\n\nRequired fixes before GO: none\n\nRecommended follow-ups:\n- none\n\nRationale: Stub.\n' > "$output_path" else stub_verdict="${CODEX_GATE_STUB_VERDICT:-advise}" - printf '## stub-reviewer -- %s\nVerdict: %s. Stub output.\n' "$stub_verdict" "$stub_verdict" > "$output_path" + reviewer_name="$(awk '$1 == "Reviewer:" { print $2; exit }' "$brief_file")" + : "${reviewer_name:=stub-reviewer}" + printf '## %s -- %s\nVerdict: %s. Stub output.\n' \ + "$reviewer_name" "$stub_verdict" "$stub_verdict" > "$output_path" fi exit 0 TWRAP_EOF @@ -2954,7 +3569,8 @@ test_verdict_prefix_rejected() { fail "$name" "expected non-zero exit when verdict uses invalid prefix-only token" return fi - assert_file_contains "$name" "$err" "exactly one valid Verdict line" || return + assert_file_contains "$name" "$err" \ + "invalid or ambiguous canonical verdict" || return pass "$name" } @@ -3131,7 +3747,8 @@ test_multiple_verdict_lines_aborts_gate() { fail "$name" "expected non-zero exit when reviewer artifact has multiple valid Verdict lines" return fi - assert_file_contains "$name" "$err" "exactly one valid Verdict line" || return + assert_file_contains "$name" "$err" \ + "invalid or ambiguous canonical verdict" || return pass "$name" } @@ -3366,6 +3983,79 @@ STUB_PMCTL pass "$name" } +# Behavior: a repo-layout gate whose pre-flight command fails publishes a +# complete NO-GO result with unavailable dispatch evidence. The run-dir makes +# an attestation destination available, but no reviewer was dispatched, so the +# sidecar must leave provenance.attestation null instead of pointing at a file +# that cannot and must not exist. +# Steps: run a repo-layout fixture with --run-dir and a failing --test-cmd, +# then assert exit 1, a verified relocated result, the preflight-only outcome, +# null attestation provenance, and no protected attestation artifact. +test_repo_layout_preflight_failure_publishes_unattested_nogo() { + local name="gate-assurance/repo-layout-preflight-failure-publishes-unattested-nogo" + should_run "$name" || return 0 + local dir source_runner layout home repo out err result run_dir code + dir="$TMP_ROOT/$name" + source_runner="$dir/source-runner" + layout="$dir/layout" + home="$dir/home" + repo="$dir/repo" + out="$dir/out" + err="$dir/err" + run_dir="$dir/gate-run" + mkdir -p "$dir" "$layout/runtime/bin" "$layout/runtime/lib" \ + "$layout/core/policy" "$run_dir" + create_runner "$source_runner" + cp "$source_runner/pr-gate.sh" "$layout/runtime/bin/pr-gate.sh" + cp -R "$source_runner/lib/." "$layout/runtime/lib/" + cp -R "$source_runner/core/policy/." "$layout/core/policy/" + cp -R "$REPO_ROOT/agents" "$layout/agents" + cp -R "$REPO_ROOT/adapters" "$layout/adapters" + cp "$source_runner/adapters/codex/dispatch.sh" "$layout/adapters/codex/dispatch.sh" + chmod +x "$layout/runtime/bin/pr-gate.sh" "$layout/adapters/codex/dispatch.sh" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + + code=0 + set +e + HOME="$home" PM_DISPATCH_STATE_ROOT="$dir/state" \ + "$layout/runtime/bin/pr-gate.sh" --cd "$repo" --base main --executor codex \ + --run-dir "$run_dir" --test-cmd "exit 1" > "$out" 2> "$err" + code=$? + set -e + if [[ "$code" -ne 1 ]]; then + fail "$name" "exit $code, expected published NO-GO exit 1: $(tail -n 20 "$err" 2>/dev/null)" + return + fi + result="$(awk -F'result: ' '/^result: /{path=$2} END{print path}' "$out")" + if [[ -z "$result" || ! -s "$result" || "$result" != "$run_dir"/.gate-results/* ]]; then + fail "$name" "verified result handoff missing or outside run-dir: result=$result" + return + fi + if ! jq -e ' + .result.final == "NO-GO" and + .coordinates.independence.evidence_status == "unavailable" and + .dispatch.outcomes == [{ + role:"preflight",reviewer:null,status:"failed",run_id:null, + evidence_status:"unavailable" + }] and + .provenance.attestation == null + ' "${result}.assurance.json" >/dev/null; then + fail "$name" "preflight assurance claimed a nonexistent dispatch attestation" + return + fi + if find "$run_dir" -maxdepth 1 -name 'gate-assurance-*.attestation.json' -print -quit \ + | grep -q .; then + fail "$name" "preflight-only gate unexpectedly wrote a dispatch attestation" + return + fi + if ! "$REPO_ROOT/cli/pmctl" gate verify "$result" >/dev/null 2>&1; then + fail "$name" "published preflight NO-GO failed shared verification" + return + fi + pass "$name" +} + # Behavior: the parallel gate result body still carries exactly one # plain-text Final: (GO|NO-GO) line, preserving the pre-frontmatter # back-compat contract that downstream consumers grep for. @@ -3571,12 +4261,12 @@ test_full_tier_line_count() { pass "$name" } -# Behavior: a sensitive filename (auth-*) triggers full tier regardless of -# how few lines changed. -# Steps: create a repo/branch with a tiny diff to a sensitive filename, run -# the gate, and assert the captured brief has Tier: full. -test_full_tier_sensitive_file() { - local name="full-tier-sensitive-file" +# Behavior: a bounded auth-path change adds the security reviewer without +# conflating sensitive coverage with full-tier or parallel topology. +# Steps: create a tiny auth diff and assert express intent plus the security +# dimension, while parallel remains only a recommendation. +test_sensitive_file_adds_security_without_forcing_full() { + local name="sensitive-file-adds-security" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" @@ -3594,7 +4284,96 @@ test_full_tier_sensitive_file() { fail "$name" "exit $code, expected 0" return fi - assert_file_contains "$name" "$brief" "Tier: full" || return + assert_file_contains "$name" "$brief" "Tier: express" || return + assert_file_contains "$name" "$brief" "Reviewers: critic,qa-tester,security-reviewer" || return + assert_file_contains "$name" "$brief" "policy.required_reviewers: critic,qa-tester,security-reviewer" || return + assert_file_contains "$name" "$brief" "policy.recommended_mode: parallel" || return + assert_file_contains "$name" "$brief" "policy.required_mode: none" || return + assert_file_contains "$name" "$brief" \ + 'policy.escalation_signals: [{"id":"security-sensitive-path"' || return + assert_not_contains "$name" "$brief" "any diff file matches (" || return + assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return + pass "$name" +} + +# Behavior: pluralized security/risk directories and a public schema path map +# to their three signal-specific reviewer dimensions through one resolver. +test_plural_signal_paths_add_required_dimensions() { + local name="plural-signal-paths-add-required-dimensions" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" result_path + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" clean + mkdir -p "$repo/credentials" "$repo/migrations" "$repo/api" + printf 'package credentials\n' > "$repo/credentials/store.go" + printf 'select 1;\n' > "$repo/migrations/001.sql" + printf '{}\n' > "$repo/api/schema.json" + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main + local code=$? + set -e + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" + return + fi + assert_file_contains "$name" "$brief" "tier.resolved: standard" || return + assert_file_contains "$name" "$brief" \ + "coverage.selected: critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer" || return + assert_file_contains "$name" "$brief" "policy.required_mode: none" || return + result_path="$(awk -F'result: ' '/^result: /{path=$2} END{print path}' "$out")" + jq -e ' + any(.policy.matched_signals[]; + .id == "security-sensitive-path" and + (.matches | index("credentials/store.go")) != null) and + any(.policy.matched_signals[]; + .id == "risk-sensitive-path" and + (.matches | index("migrations/001.sql")) != null) and + any(.policy.matched_signals[]; + .id == "public-contract-path" and + (.matches | index("api/schema.json")) != null) + ' "${result_path}.assurance.json" >/dev/null || { + fail "$name" "policy artifact omitted a plural-path signal match" + return + } + pass "$name" +} + +# Behavior: changes to the canonical policy tables force the gate's highest +# rigor and the architecture/security/risk dimensions, preventing a small +# policy edit from quietly weakening its own future review floor. +test_policy_source_path_is_self_protecting() { + local name="policy-source-path-is-self-protecting" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" clean + mkdir -p "$repo/core/policy" + printf 'policy fixture\n' > "$repo/core/policy/example.tsv" + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" \ + run_gate "$home" "$runner" "$repo" "$out" "$err" --base main + local code=$? + set -e + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0: $(cat "$err")" + return + fi + assert_file_contains "$name" "$brief" "tier.resolved: full" || return + assert_file_contains "$name" "$brief" \ + "coverage.selected: critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer" \ + || return + assert_file_contains "$name" "$brief" '"id":"policy-source-path"' || return pass "$name" } @@ -3633,10 +4412,10 @@ test_via_symlink() { pass "$name" } -# Behavior: renaming a sensitive file (auth.ts -> login.ts) still triggers -# full tier by matching the rename's old name, not just the new one. -# Steps: commit auth.ts on main, then rename it to login.ts on a feature -# branch, run the gate, and assert the captured brief has Tier: full. +# Behavior: renaming a sensitive file (auth.ts -> login.ts) preserves both the +# rename fact and the security signal without conflating either with full tier. +# Steps: commit auth.ts on main, rename it on a feature branch, and assert the +# policy artifact records the old path under both matched signals. test_rename_sensitive_old_name() { local name="rename-sensitive-old-name" should_run "$name" || return 0 @@ -3668,7 +4447,19 @@ test_rename_sensitive_old_name() { fail "$name" "exit $code, expected 0" return fi - assert_file_contains "$name" "$brief" "Tier: full" || return + assert_file_contains "$name" "$brief" "Tier: express" || return + assert_file_contains "$name" "$brief" "Reviewers: critic,qa-tester,security-reviewer" || return + local result_path + result_path="$(awk -F'result: ' '/^result: /{path=$2} END{print path}' "$out")" + jq -e ' + any(.policy.matched_signals[]; + .id == "renamed-input" and (.matches | index("auth.ts")) != null) and + any(.policy.matched_signals[]; + .id == "security-sensitive-path" and (.matches | index("auth.ts")) != null) + ' "${result_path}.assurance.json" >/dev/null || { + fail "$name" "policy artifact did not preserve rename-origin security evidence" + return + } pass "$name" } @@ -3762,8 +4553,13 @@ test_untracked_binary_routes_to_standard() { } run_test test_gate_assurance_policy_snapshot_matches_sources -run_test test_gate_tier_policy_source_controls_default_coverage +run_test test_gate_policy_sources_control_default_coverage run_test test_gate_assurance_policy_snapshot_is_copy_mode_fallback +run_test test_dormant_policy_signal_with_unknown_reviewer_fails_before_dispatch +run_test test_duplicate_policy_signal_id_fails_before_dispatch +run_test test_maintainer_initial_policy_fixes_coverage_only +run_test test_maintainer_targeted_policy_preserves_remediation_scope +run_test test_input_execution_signal_requires_parallel_mode run_test test_tier_detection run_test test_pr_gate_does_not_mutate_gitignore run_test test_artifact_filter_drops_gate_artifacts @@ -3776,8 +4572,15 @@ run_test test_copy_mode_artifact_fallback_body_parity run_test test_missing_reviewer_agent run_test test_invalid_base_ref run_test test_no_changed_files -run_test test_reviewers_override_preserves_tier_detection -run_test test_full_tier_with_critic_only_coverage_is_truthful +run_test test_reviewers_override_below_policy_floor_fails_closed +run_test test_invalid_policy_consumer_fails_before_dispatch +run_test test_empty_policy_override_fails_before_dispatch +run_test test_malformed_policy_override_contract_fails_before_dispatch +run_test test_scope_bound_policy_override_authorizes_exact_coverage_downgrade +run_test test_policy_override_scope_mismatch_fails_closed +run_test test_policy_override_scope_binds_diff_content +run_test test_policy_override_allowance_mismatch_fails_closed +run_test test_policy_enforcement_precedes_resolved_coordinate_consumers run_test test_brief_file_snapshot_exists_at_dispatch run_test test_reviewer_definitions_are_workspace_snapshots run_test test_brief_cleanup_on_dispatch_failure @@ -3785,7 +4588,9 @@ run_test test_output_directory_created run_test test_claude_adapter_dispatches_subprocess run_test test_standard_tier_detection run_test test_full_tier_line_count -run_test test_full_tier_sensitive_file +run_test test_sensitive_file_adds_security_without_forcing_full +run_test test_plural_signal_paths_add_required_dimensions +run_test test_policy_source_path_is_self_protecting run_test test_via_symlink run_test test_rename_sensitive_old_name run_test test_binary_file_routes_to_standard @@ -3799,6 +4604,7 @@ run_test test_parallel_reviewer_brief_validates run_test test_parallel_synthesis_brief_validates run_test test_gate_result_frontmatter_and_escalation run_test test_repo_layout_captures_dispatch_run_id +run_test test_repo_layout_preflight_failure_publishes_unattested_nogo run_test test_gate_result_final_line_back_compat run_test test_frontmatter_escalation_parity run_test test_failed_reviewer_aborts_gate @@ -3806,6 +4612,8 @@ run_test test_synthesis_verdict_mismatch_aborts_gate run_test test_post_synthesis_injection_detected run_test test_synthesis_no_output_aborts_gate run_test test_reviewer_invalid_verdict_aborts_gate +run_test test_reviewer_heading_only_verdict_is_accepted +run_test test_reviewer_heading_and_explicit_verdict_must_agree run_test test_reviewer_no_output_aborts_gate run_test test_sequential_no_output_aborts_gate run_test test_sequential_no_final_line_aborts_gate @@ -4437,6 +5245,8 @@ test_help_output_is_bounded_and_current() { fi assert_file_contains "$name" "$out" "Usage:" || return assert_file_contains "$name" "$out" "--mode " || return + assert_file_contains "$name" "$out" "--policy " || return + assert_file_contains "$name" "$out" "--policy-override " || return assert_file_contains "$name" "$out" "--targeted " || return assert_file_contains "$name" "$out" "--initial-result " || return assert_not_contains "$name" "$out" "_gate_assurance_policy_snapshot" || return @@ -4532,6 +5342,40 @@ test_targeted_pass_references_initial_result() { pass "$name" } +# Behavior: a targeted pass with auto mode and no input brief resolves all +# policy coordinates before constructing the default sequential reviewer brief. +# Steps: run a critic-only targeted gate without --mode or --brief and assert +# successful dispatch plus initialized sequential coordinates. This covers the +# real runtime path that previously aborted on unbound MODE_RESOLVED/BRIEF_FILE. +test_targeted_auto_mode_initializes_brief_coordinates() { + local name="targeted-auto-mode-initializes-brief-coordinates" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/brief.md" initial="$dir/initial.md" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + write_valid_initial_gate_result "$initial" + + set +e + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --targeted critic --initial-result "$initial" + local code=$? + set -e + + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0 (stderr: $(head -3 "$err" 2>/dev/null))" + return + fi + assert_file_contains "$name" "$brief" "mode.requested: default" || return + assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return + assert_file_contains "$name" "$brief" "pass.resolved: targeted" || return + assert_not_contains "$name" "$err" "unbound variable" || return + pass "$name" +} + # Behavior: a targeted pass without an initial result fails before dispatch. # Steps: invoke --targeted critic without --initial-result and assert exit 2, # the explicit requirement error, and no dispatch marker. @@ -4715,12 +5559,12 @@ test_equivalent_mode_spellings_are_accepted() { local out="$dir/out" err="$dir/err" brief="$dir/brief.md" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs set +e CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ - --base main --reviewers critic --mode parallel --parallel + --base main --reviewers critic,qa-tester --mode parallel --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -4850,10 +5694,10 @@ test_parallel_synthesis_brief_ascii_separator() { # Behavior: the per-reviewer brief emitted in parallel mode by pr-gate.sh # uses ASCII -- separators and contains no em dash (U+2014). -# CODEX_GATE_CAPTURE_REVIEWER_BRIEF captures the last non-synthesis brief -# dispatched during a parallel run. -# Steps: run the gate with --reviewers critic --parallel, and assert the -# captured reviewer brief contains "Executor: codex" and "file:line --" +# CODEX_GATE_CAPTURE_REVIEWER_FILTER selects the critic brief from the +# policy-complete parallel dispatch. +# Steps: run the gate with generic docs coverage, and assert the captured +# critic brief contains "Executor: codex" and "file:line --" # using ASCII dashes, and no UTF-8 em dash byte sequence is present. test_parallel_reviewer_brief_ascii_separator() { local name="parallel-reviewer-brief-ascii-separator" @@ -4863,12 +5707,14 @@ test_parallel_reviewer_brief_ascii_separator() { local out="$dir/out" err="$dir/err" reviewer_brief="$dir/reviewer-brief.md" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs set +e CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$reviewer_brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + CODEX_GATE_CAPTURE_REVIEWER_FILTER=critic \ + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -4928,8 +5774,8 @@ test_sequential_brief_has_citation_guard() { # Behavior: the per-reviewer parallel brief contains the citation-guard # preamble ("Verified reference files") and the explicit constraint ("do # not invent citations"), listing real repo files. -# Steps: commit a fixture agent file, run the gate with --reviewers critic -# --parallel, and assert the captured reviewer brief contains "Verified +# Steps: commit a fixture agent file, run the generic docs coverage in +# parallel, and assert the captured critic brief contains "Verified # reference files", "do not invent citations", and the fixture path. test_parallel_reviewer_brief_has_citation_guard() { local name="parallel-reviewer-brief-has-citation-guard" @@ -4948,7 +5794,9 @@ test_parallel_reviewer_brief_has_citation_guard() { set +e CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$reviewer_brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + CODEX_GATE_CAPTURE_REVIEWER_FILTER=critic \ + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -4968,8 +5816,8 @@ test_parallel_reviewer_brief_has_citation_guard() { # Behavior: the parallel synthesis brief contains the citation-guard # preamble ("Verified reference files") and the explicit constraint ("do # not invent citations"), listing real repo files. -# Steps: commit a fixture agent file, run the gate with --reviewers critic -# --parallel, and assert the captured synthesis brief contains "Verified +# Steps: commit a fixture agent file, run the generic docs coverage in +# parallel, and assert the captured synthesis brief contains "Verified # reference files", "do not invent citations", and the fixture path. test_parallel_synthesis_brief_has_citation_guard() { local name="parallel-synthesis-brief-has-citation-guard" @@ -4988,7 +5836,8 @@ test_parallel_synthesis_brief_has_citation_guard() { set +e CODEX_GATE_CAPTURE_BRIEF="$brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -5203,10 +6052,10 @@ test_seq_brief_has_reviewer_guard_constraint() { # Behavior: each per-reviewer parallel brief contains the explicit pmctl # guard check constraint that must be called before writing the reviewer -# output file. CODEX_GATE_CAPTURE_REVIEWER_BRIEF captures the last -# non-synthesis brief dispatched during a parallel run. -# Steps: run the gate with --reviewers critic --parallel, and assert the -# captured reviewer brief contains "pmctl guard check --role reviewer" and +# output file. CODEX_GATE_CAPTURE_REVIEWER_FILTER selects the critic brief +# from the policy-complete parallel dispatch. +# Steps: run the generic docs coverage in parallel, and assert the captured +# critic brief contains "pmctl guard check --role reviewer" and # "--event pre-write". test_parallel_reviewer_brief_has_guard_constraint() { local name="parallel-reviewer-brief-has-guard-constraint" @@ -5216,12 +6065,14 @@ test_parallel_reviewer_brief_has_guard_constraint() { local out="$dir/out" err="$dir/err" reviewer_brief="$dir/reviewer-brief.md" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs set +e CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$reviewer_brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + CODEX_GATE_CAPTURE_REVIEWER_FILTER=critic \ + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -5330,9 +6181,9 @@ test_seq_brief_guard_absolute_path_when_pmctl_not_on_path() { # Behavior: same as the sequential case above, but for the --parallel # per-reviewer brief's guard-check instruction. -# Steps: strip pmctl from PATH, stage runner/cli/pmctl, run --parallel with a -# single reviewer, assert the captured reviewer brief's guard-check line uses -# the absolute path. +# Steps: strip pmctl from PATH, stage runner/cli/pmctl, run the generic docs +# coverage in parallel, and assert the captured critic brief's guard-check line +# uses the absolute path. test_parallel_reviewer_brief_guard_absolute_path_when_pmctl_not_on_path() { local name="parallel-reviewer-brief-guard-absolute-path-when-pmctl-not-on-path" should_run "$name" || return 0 @@ -5341,7 +6192,7 @@ test_parallel_reviewer_brief_guard_absolute_path_when_pmctl_not_on_path() { local out="$dir/out" err="$dir/err" reviewer_brief="$dir/reviewer-brief.md" mkdir -p "$dir" create_runner "$runner" - create_agents "$home" critic + create_agents "$home" critic qa-tester create_repo "$repo" docs local REPLY @@ -5349,8 +6200,10 @@ test_parallel_reviewer_brief_guard_absolute_path_when_pmctl_not_on_path() { local minpath="$REPLY" set +e - CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$reviewer_brief" HOME="$home" PATH="$minpath" \ - "$runner/pr-gate.sh" --cd "$repo" --base main --reviewers critic --parallel > "$out" 2> "$err" + CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$reviewer_brief" \ + CODEX_GATE_CAPTURE_REVIEWER_FILTER=critic HOME="$home" PATH="$minpath" \ + "$runner/pr-gate.sh" --cd "$repo" --base main \ + --reviewers critic,qa-tester --parallel > "$out" 2> "$err" local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -5420,6 +6273,7 @@ run_test test_copy_mode_dispatches_via_adapter run_test test_help_output_is_bounded_and_current run_test test_unknown_arg_message run_test test_targeted_pass_references_initial_result +run_test test_targeted_auto_mode_initializes_brief_coordinates run_test test_targeted_requires_initial_result run_test test_targeted_output_cannot_overwrite_initial_result run_test test_targeted_sidecar_cannot_overwrite_initial_result @@ -5440,23 +6294,23 @@ run_test test_dirty_preflight_allow_dirty_includes_worktree run_test test_allow_dirty_includes_uncommitted_tracked run_test test_clean_committed_tree_passes_preflight run_test test_dirty_only_no_commit_still_reviewed -# Behavior: a brief with architecture_impact:major emits a tier advisory -# to stderr when the auto-detected tier is not full. -# Steps: run the gate with --brief pointing at a major-impact brief on a -# docs-only diff, and assert stderr contains "architecture_impact:major" -# and "suggested tier: full". -test_brief_major_suggests_full() { - local name="brief-major-suggests-full" +# Behavior: trusted architecture_impact:major metadata is a canonical full-tier +# policy input, not a reviewer advisory. +# Steps: run a docs-only gate with a major-impact brief and assert the resolved +# tier and reviewer coverage satisfy the full floor. +test_brief_major_resolves_full() { + local name="brief-major-resolves-full" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local out="$dir/out" err="$dir/err" + local input_brief="$dir/input-brief.md" gate_brief="$dir/gate-brief.md" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer create_repo "$repo" docs - cat > "$brief" <<'BRIEF_EOF' + cat > "$input_brief" <<'BRIEF_EOF' schema_version: 1 working_dir: /tmp goal: test @@ -5468,32 +6322,37 @@ acceptance: BRIEF_EOF set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$brief" + CODEX_GATE_CAPTURE_BRIEF="$gate_brief" run_gate \ + "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$input_brief" + local code=$? set -e - assert_file_contains "$name" "$err" "architecture_impact:major" || return - assert_file_contains "$name" "$err" "suggested tier: full" || return + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" + return + fi + assert_file_contains "$name" "$gate_brief" "Tier: full" || return + assert_file_contains "$name" "$gate_brief" "policy.minimum_tier: full" || return + assert_file_contains "$name" "$gate_brief" \ + "Reviewers: critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer" || return pass "$name" } -# Behavior: a brief with architecture_impact:minor emits a standard-tier -# advisory to stderr when the auto-detected tier is express (docs-only -# diff). -# Steps: run the gate with --brief pointing at a minor-impact brief on a -# docs-only diff, and assert stderr contains "architecture_impact:minor" -# and "suggested tier: standard". -test_brief_minor_express_suggests_standard() { - local name="brief-minor-express-suggests-standard" +# Behavior: trusted architecture_impact:minor metadata raises a docs-only diff +# to the canonical standard floor and architecture coverage. +test_brief_minor_resolves_standard() { + local name="brief-minor-resolves-standard" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local out="$dir/out" err="$dir/err" + local input_brief="$dir/input-brief.md" gate_brief="$dir/gate-brief.md" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer create_repo "$repo" docs - cat > "$brief" <<'BRIEF_EOF' + cat > "$input_brief" <<'BRIEF_EOF' schema_version: 1 working_dir: /tmp goal: test @@ -5505,25 +6364,69 @@ acceptance: BRIEF_EOF set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$brief" + CODEX_GATE_CAPTURE_BRIEF="$gate_brief" run_gate \ + "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$input_brief" + local code=$? set -e - assert_file_contains "$name" "$err" "architecture_impact:minor" || return - assert_file_contains "$name" "$err" "suggested tier: standard" || return + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" + return + fi + assert_file_contains "$name" "$gate_brief" "Tier: standard" || return + assert_file_contains "$name" "$gate_brief" "policy.minimum_tier: standard" || return + assert_file_contains "$name" "$gate_brief" \ + "Reviewers: critic,qa-tester,architecture-reviewer" || return pass "$name" } -# Behavior: when --tier is explicitly set, the brief advisory is -# suppressed because TIER_OVERRIDE is populated and the advisory block is -# skipped. -# Steps: run the gate with --tier full and --brief pointing at a -# major-impact brief, and assert stderr does not contain "suggested tier". -test_brief_explicit_tier_suppresses_advisory() { - local name="brief-explicit-tier-suppresses-advisory" +# Behavior: an explicit tier at the major-impact floor is accepted unchanged. +test_brief_explicit_full_satisfies_policy_floor() { + local name="brief-explicit-full-satisfies-policy-floor" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local out="$dir/out" err="$dir/err" + local input_brief="$dir/input-brief.md" gate_brief="$dir/gate-brief.md" + mkdir -p "$dir" + create_runner "$runner" + create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer + create_repo "$repo" docs + + cat > "$input_brief" <<'BRIEF_EOF' +schema_version: 1 +working_dir: /tmp +goal: test +files: + - read: README.md +architecture_impact: major +acceptance: + - test +BRIEF_EOF + + set +e + CODEX_GATE_CAPTURE_BRIEF="$gate_brief" run_gate "$home" "$runner" "$repo" \ + "$out" "$err" --base main --tier full --brief "$input_brief" + local code=$? + set -e + + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" + return + fi + assert_file_contains "$name" "$gate_brief" "tier.requested: full" || return + assert_file_contains "$name" "$gate_brief" "tier.resolved: full" || return + pass "$name" +} + +# Behavior: an explicit tier below the major-impact floor fails before dispatch +# unless a separately validated, scope-bound policy override is supplied. +test_brief_explicit_tier_below_policy_floor_fails() { + local name="brief-explicit-tier-below-policy-floor-fails" + should_run "$name" || return 0 + local dir="$TMP_ROOT/$name" + local home="$dir/home" repo="$dir/repo" runner="$dir/runner" + local out="$dir/out" err="$dir/err" brief="$dir/input-brief.md" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer @@ -5541,23 +6444,24 @@ acceptance: BRIEF_EOF set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --tier full --brief "$brief" + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --tier express --brief "$brief" + local code=$? set -e - if grep -q "suggested tier" "$err"; then - fail "$name" "--tier full should suppress the advisory but stderr contains 'suggested tier'" + if [[ "$code" -ne 3 ]]; then + fail "$name" "exit $code, expected policy rejection 3" return fi + assert_file_contains "$name" "$err" "requested=express required=full" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return pass "$name" } -# Behavior: passing a --brief path that does not exist is benign -- the -# gate runs normally and emits no advisory (the brief block checks -f -# before reading). -# Steps: run the gate with --brief pointing at a nonexistent file, and -# assert stderr does not contain "suggested tier". -test_brief_nonexistent_file_is_benign() { - local name="brief-nonexistent-file-is-benign" +# Behavior: explicitly supplied brief metadata is trusted input only when the +# file exists and is readable; a missing path fails before policy resolution. +test_brief_nonexistent_file_fails_closed() { + local name="brief-nonexistent-file-fails-closed" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" @@ -5569,31 +6473,33 @@ test_brief_nonexistent_file_is_benign() { set +e run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$dir/no-such-brief.md" + local code=$? set -e - if grep -q "suggested tier" "$err"; then - fail "$name" "missing brief should produce no advisory but stderr contains 'suggested tier'" + if [[ "$code" -ne 2 ]]; then + fail "$name" "exit $code, expected 2" return fi + assert_file_contains "$name" "$err" "--brief must name a readable file" || return + assert_not_contains "$name" "$err" "DISPATCH_STUB" || return pass "$name" } -# Behavior: a brief with architecture_impact:none produces no tier -# advisory in stderr. -# Steps: run the gate with --brief pointing at a none-impact brief, and -# assert stderr does not contain "suggested tier". -test_brief_none_no_advisory() { - local name="brief-none-no-advisory" +# Behavior: architecture_impact:none adds no risk floor beyond the diff's +# own docs-only classification. +test_brief_none_preserves_docs_floor() { + local name="brief-none-preserves-docs-floor" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" - local out="$dir/out" err="$dir/err" brief="$dir/brief.md" + local out="$dir/out" err="$dir/err" + local input_brief="$dir/input-brief.md" gate_brief="$dir/gate-brief.md" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer create_repo "$repo" docs - cat > "$brief" <<'BRIEF_EOF' + cat > "$input_brief" <<'BRIEF_EOF' schema_version: 1 working_dir: /tmp goal: test @@ -5605,13 +6511,17 @@ acceptance: BRIEF_EOF set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$brief" + CODEX_GATE_CAPTURE_BRIEF="$gate_brief" run_gate \ + "$home" "$runner" "$repo" "$out" "$err" --base main --brief "$input_brief" + local code=$? set -e - if grep -q "suggested tier" "$err"; then - fail "$name" "architecture_impact:none should produce no advisory but stderr contains 'suggested tier'" + if [[ "$code" -ne 0 ]]; then + fail "$name" "exit $code, expected 0" return fi + assert_file_contains "$name" "$gate_brief" "Tier: express" || return + assert_file_contains "$name" "$gate_brief" "policy.minimum_tier: express" || return pass "$name" } @@ -5847,9 +6757,8 @@ test_no_overrides_brief_unchanged() { # Behavior: the override block is injected into the --parallel per-reviewer # brief too -- a distinct insertion site (pr-gate.sh:906) from the # sequential one, needing its own coverage. -# Steps: write a .gate-overrides.md, run the gate with --reviewers critic -# --parallel (a single reviewer avoids a last-writer-wins race on the -# capture target), and assert the captured reviewer brief (not synthesis) +# Steps: write a .gate-overrides.md, run the generic docs coverage in parallel, +# capture only the critic brief, and assert the reviewer brief (not synthesis) # contains "Accepted-risk overrides" and the override's content. test_override_file_injected_into_parallel_reviewer_brief() { local name="override-file-injected-parallel-reviewer" @@ -5864,7 +6773,10 @@ test_override_file_injected_into_parallel_reviewer_brief() { printf '## Gate Overrides\n\n- [risk] Accepted: storage cleanup may fail. Owner: test.\n' > "$repo/.gate-overrides.md" set +e - CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + CODEX_GATE_CAPTURE_REVIEWER_BRIEF="$brief" \ + CODEX_GATE_CAPTURE_REVIEWER_FILTER=critic \ + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -5880,8 +6792,8 @@ test_override_file_injected_into_parallel_reviewer_brief() { # Behavior: the override block is injected into the parallel synthesis # brief too -- a third distinct insertion site (pr-gate.sh:1128). -# Steps: write a .gate-overrides.md, run the gate with --reviewers critic -# --parallel (CODEX_GATE_CAPTURE_BRIEF receives the synthesis brief, the +# Steps: write a .gate-overrides.md, run the generic docs coverage in parallel +# (CODEX_GATE_CAPTURE_BRIEF receives the synthesis brief, the # last dispatch), and assert the captured brief contains the synthesis # marker "Reviewer findings (embedded" plus "Accepted-risk overrides" and # the override's content. @@ -5898,7 +6810,9 @@ test_override_file_injected_into_parallel_synthesis_brief() { printf '## Gate Overrides\n\n- [risk] Accepted: storage cleanup may fail. Owner: test.\n' > "$repo/.gate-overrides.md" set +e - CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic --parallel + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate \ + "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester --parallel local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -5920,11 +6834,12 @@ run_test test_claude_seq_brief_guard_stays_bare_pmctl_when_pmctl_not_on_path run_test test_missing_jq_fails_before_dispatch run_test test_relative_output_normalized_to_absolute run_test test_inline_fallback_matches_lib -run_test test_brief_major_suggests_full -run_test test_brief_minor_express_suggests_standard -run_test test_brief_explicit_tier_suppresses_advisory -run_test test_brief_nonexistent_file_is_benign -run_test test_brief_none_no_advisory +run_test test_brief_major_resolves_full +run_test test_brief_minor_resolves_standard +run_test test_brief_explicit_full_satisfies_policy_floor +run_test test_brief_explicit_tier_below_policy_floor_fails +run_test test_brief_nonexistent_file_fails_closed +run_test test_brief_none_preserves_docs_floor run_test test_override_file_injected_into_sequential_brief run_test test_override_file_autodiscovery run_test test_override_file_explicit_flag @@ -6914,14 +7829,15 @@ test_workspace_reviewer_definitions_are_base_pinned() { create_repo "$repo" docs mkdir -p "$repo/agents" printf '# trusted-base-reviewer\n' > "$repo/agents/critic.md" - git -C "$repo" add agents/critic.md - git -C "$repo" commit -q -m 'add trusted reviewer definition' + printf '# trusted-base-qa-reviewer\n' > "$repo/agents/qa-tester.md" + git -C "$repo" add agents/critic.md agents/qa-tester.md + git -C "$repo" commit -q -m 'add trusted reviewer definitions' printf '# malicious-working-tree-reviewer\n' > "$repo/agents/critic.md" set +e CODEX_GATE_CAPTURE_REVIEWER_DEFS="$captured" run_gate \ "$home" "$runner" "$repo" "$out" "$err" --base main --executor codex \ - --reviewers critic --reviewer-dir "$repo/agents" --allow-dirty + --reviewers critic,qa-tester --reviewer-dir "$repo/agents" --allow-dirty local code=$? set -e [[ "$code" -eq 0 ]] || { fail "$name" "exit $code, expected 0: $(cat "$err" 2>/dev/null)"; return; } @@ -6945,8 +7861,9 @@ test_relative_work_dir_preserves_base_pinned_reviewer_boundary() { create_repo "$repo" docs mkdir -p "$repo/agents" printf '# trusted-relative-base-reviewer\n' > "$repo/agents/critic.md" - git -C "$repo" add agents/critic.md - git -C "$repo" commit -q -m 'add relative-path reviewer definition' + printf '# trusted-relative-base-qa-reviewer\n' > "$repo/agents/qa-tester.md" + git -C "$repo" add agents/critic.md agents/qa-tester.md + git -C "$repo" commit -q -m 'add relative-path reviewer definitions' printf '# malicious-relative-working-tree-reviewer\n' > "$repo/agents/critic.md" local code=0 @@ -6955,7 +7872,7 @@ test_relative_work_dir_preserves_base_pinned_reviewer_boundary() { cd "$repo" HOME="$home" CODEX_GATE_CAPTURE_REVIEWER_DEFS="$captured" \ "$runner/pr-gate.sh" --cd . --base main --executor codex \ - --reviewers critic --reviewer-dir "$repo/agents" --allow-dirty + --reviewers critic,qa-tester --reviewer-dir "$repo/agents" --allow-dirty ) > "$out" 2> "$err" code=$? set -e @@ -6982,7 +7899,8 @@ test_trusted_reviewer_symlink_is_rejected() { rm -f "$runner/agents/critic.md" ln -s "$dir/foreign.md" "$runner/agents/critic.md" set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester local code=$? set -e [[ "$code" -ne 0 ]] || { fail "$name" "symlinked reviewer was accepted"; return; } @@ -7005,7 +7923,8 @@ test_trusted_reviewer_hardlink_is_rejected() { create_repo "$repo" docs ln "$runner/agents/critic.md" "$dir/critic-hardlink.md" set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --reviewers critic + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --reviewers critic,qa-tester local code=$? set -e [[ "$code" -ne 0 ]] || { fail "$name" "hardlinked reviewer was accepted"; return; } @@ -7038,7 +7957,7 @@ STUB_EOF local code=0 set +e PATH="$shim_bin:$PATH" run_gate "$home" "$runner" "$repo" "$out" "$err" \ - --base main --reviewers critic + --base main --reviewers critic,qa-tester code=$? set -e [[ "$code" -ne 0 ]] || { fail "$name" "mid-copy mutation was accepted"; return; } diff --git a/tests/shell/test-run-tests.sh b/tests/shell/test-run-tests.sh index a47cb064..79099d02 100755 --- a/tests/shell/test-run-tests.sh +++ b/tests/shell/test-run-tests.sh @@ -27,11 +27,13 @@ make_fixture() { cp "$REPO_ROOT/core/policy/gate-tiers.tsv" "$repo/core/policy/gate-tiers.tsv" cp "$REPO_ROOT/core/policy/gate-modes.tsv" "$repo/core/policy/gate-modes.tsv" cp "$REPO_ROOT/core/policy/gate-pass-kinds.tsv" "$repo/core/policy/gate-pass-kinds.tsv" + cp "$REPO_ROOT/core/policy/gate-policy-consumers.tsv" "$repo/core/policy/gate-policy-consumers.tsv" + cp "$REPO_ROOT/core/policy/gate-policy-signals.tsv" "$repo/core/policy/gate-policy-signals.tsv" chmod +x "$repo/tests/bin/run-tests.sh" cat > "$repo/tests/lib/test-suite-runner.sh" <<'RUNNER' #!/usr/bin/env bash set -euo pipefail -suites=(lint-agents lint-scripts lint-script-domain-inventory lint-portable-repo-paths test-lint-shellcheck test-script-domain-inventory test-lint-portable-repo-paths test-lint-frontmatter test-commands test-check-docs-freshness test-guards test-migrate test-install test-doctor test-pmctl-dispatch test-pmctl-context test-pmctl-memory test-pr-gate test-pr-gate-profile test-pmctl-operation test-host-manifest test-host-write-codex test-codex-dispatch-continuation test-host-write-parity test-core-schemas test-layer-boundaries test-pm-scripts test-run-tests test-state-store test-state-layout-parity) +suites=(lint-agents lint-scripts lint-script-domain-inventory lint-portable-repo-paths test-lint-shellcheck test-script-domain-inventory test-lint-portable-repo-paths test-lint-frontmatter test-commands test-check-docs-freshness test-guards test-migrate test-install test-doctor test-pmctl-dispatch test-pmctl-context test-pmctl-memory test-pmctl-gate test-pr-gate test-pr-gate-profile test-pmctl-operation test-host-manifest test-host-write-codex test-codex-dispatch-continuation test-host-write-parity test-core-schemas test-layer-boundaries test-pm-scripts test-run-tests test-setup-project test-state-store test-state-layout-parity) for arg in "$@"; do if [[ "$arg" == --list ]]; then printf '%s\n' "${suites[@]}"; exit 0; fi done @@ -151,6 +153,20 @@ case_operational_docs_map_to_stale_reference_lint() { fi } +case_gitignore_maps_to_setup_project() { + local name=gitignore-maps-to-setup-project repo out status=0 args + args="$TMP_ROOT/$name.args" + repo="$(make_fixture "$name")" + out=$(RUN_TESTS_ARGS_LOG="$args" "$repo/tests/bin/run-tests.sh" \ + --path .gitignore --list 2>&1) || status=$? + if [[ "$status" -eq 0 && "$out" == *"test-setup-project"* \ + && "$out" != *"coverage gaps"* && ! -e "$args" ]]; then + pass "$name" + else + fail "$name" "status=$status out=$out" + fi +} + case_agent_mapping_uses_registered_frontmatter_suite() { local name=agent-mapping-uses-registered-frontmatter-suite repo out status=0 args args="$TMP_ROOT/$name.args" @@ -237,7 +253,9 @@ case_gate_assurance_policy_maps_gate_consumers() { repo="$(make_fixture "$name")" out=$(RUN_TESTS_ARGS_LOG="$args" "$repo/tests/bin/run-tests.sh" \ --path core/policy/gate-tiers.tsv --path core/policy/gate-modes.tsv \ - --path core/policy/gate-pass-kinds.tsv --list 2>&1) || status=$? + --path core/policy/gate-pass-kinds.tsv \ + --path core/policy/gate-policy-consumers.tsv \ + --path core/policy/gate-policy-signals.tsv --list 2>&1) || status=$? if [[ "$status" -eq 0 && "$out" == *"test-pr-gate"* && "$out" == *"test-pr-gate-profile"* && "$out" == *"test-core-schemas"* && "$out" == *"test-layer-boundaries"* && "$out" != *"coverage gaps"* && @@ -248,6 +266,24 @@ case_gate_assurance_policy_maps_gate_consumers() { fi } +case_gate_assurance_contract_maps_runtime_verifiers() { + local name=gate-assurance-contract-maps-runtime-verifiers repo out status=0 args + args="$TMP_ROOT/$name.args" + repo="$(make_fixture "$name")" + out=$(RUN_TESTS_ARGS_LOG="$args" "$repo/tests/bin/run-tests.sh" \ + --path core/schema/gate-assurance.schema.json \ + --path core/schema/gate-policy-override.schema.json \ + --path runtime/lib/gate-result-verify.sh --list 2>&1) || status=$? + if [[ "$status" -eq 0 && "$out" == *"test-core-schemas"* && + "$out" == *"test-pr-gate"* && "$out" == *"test-pmctl-gate"* && + "$out" == *"test-layer-boundaries"* && "$out" != *"coverage gaps"* && + ! -e "$args" ]]; then + pass "$name" + else + fail "$name" "status=$status out=$out" + fi +} + case_high_fanout_escalates_full() { local name=high-fanout-escalates-full repo out status=0 args args="$TMP_ROOT/$name.args" @@ -416,6 +452,7 @@ case_state_writer_mapping_runs_operation_parity case_state_layout_mapping_runs_parity case_docs_mapping_list_only case_operational_docs_map_to_stale_reference_lint +case_gitignore_maps_to_setup_project case_agent_mapping_uses_registered_frontmatter_suite case_command_mapping_uses_registered_frontmatter_suite case_skill_mapping_uses_registered_frontmatter_suite @@ -423,6 +460,7 @@ case_guard_family_maps_to_guard_suite case_prompt_context_timeout_contract_maps_all_consumers case_evidence_contract_maps_to_runner_regression case_gate_assurance_policy_maps_gate_consumers +case_gate_assurance_contract_maps_runtime_verifiers case_high_fanout_escalates_full case_repeated_high_fanout_escalation_succeeds case_unknown_path_fails_without_test_evidence diff --git a/tests/shell/test-setup-project.sh b/tests/shell/test-setup-project.sh index e2fe974a..41a423cb 100755 --- a/tests/shell/test-setup-project.sh +++ b/tests/shell/test-setup-project.sh @@ -5,7 +5,13 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" SETUP_SCRIPT="$REPO_ROOT/ops/setup/setup-project.sh" -EXPECTED_ENTRIES=(".agent-trace/" ".gate-briefs/" ".gate-results/" ".agents/") +EXPECTED_ENTRIES=( + ".agent-trace/" + ".gate-briefs/" + ".gate-results/" + ".agents/" + ".pm-dispatch-state/" +) # shellcheck source=tests/lib/test-harness.sh . "$SCRIPT_DIR/../lib/test-harness.sh" th_init "$@" From 225b0d49c3af49b59c20c0b7eb414edcf2e0842e Mon Sep 17 00:00:00 2001 From: screenleon Date: Tue, 28 Jul 2026 11:51:47 +0900 Subject: [PATCH 2/4] fix(gate): respect explicit execution mode --- BACKLOG.md | 41 ++-- DECISIONS.md | 24 +- MILESTONES.md | 2 +- commands/pr-gate.md | 29 ++- commands/ship.md | 6 +- core/README.md | 3 +- core/policy/gate-modes.tsv | 7 +- core/policy/gate-policy-consumers.tsv | 12 +- core/policy/gate-policy-signals.tsv | 34 +-- core/schema/gate-assurance.schema.json | 31 +-- core/schema/gate-policy-override.schema.json | 18 +- docs/review-model.md | 32 +-- runtime/bin/pr-gate.sh | 221 ++++++++----------- runtime/lib/gate-result-verify.sh | 27 ++- skills/pr-gate-review/SKILL.md | 7 +- tests/shell/test-core-schemas.sh | 28 ++- tests/shell/test-pmctl-gate.sh | 9 +- tests/shell/test-pr-gate.sh | 144 +++++++----- 18 files changed, 354 insertions(+), 321 deletions(-) diff --git a/BACKLOG.md b/BACKLOG.md index 3fd53074..f72ff865 100644 --- a/BACKLOG.md +++ b/BACKLOG.md @@ -1661,7 +1661,8 @@ authorization。 `core/policy/gate-tiers.tsv`、`core/policy/gate-modes.tsv`、 `core/policy/gate-pass-kinds.tsv`);repo-layout runtime 直接讀 source, standalone/copy-mode fallback 使用 bounded generated snapshot + freshness ratchet, - 不手寫第二份 policy。三表分別擁有 defaults/topology/initial-reference requirement, + 不手寫第二份 policy。三表分別擁有 reviewer defaults/topology/ + initial-reference requirement, 不建立合併 profile。 2. CLI canonicalize: - 新增 `--mode sequential|parallel`;既有 `--parallel`/`--sequential` 為 @@ -1670,8 +1671,9 @@ authorization。 - `--targeted ` 表示 `pass.kind=targeted`,不再只是 alias;必須搭配 `--initial-result `。Initial-result 的結構存在性在本票驗,subject freshness 與 applicability 留給 [[CC-515]]。 - - omitted tier/mode/pass 分別 resolve 為 `auto`→detected tier、 - `default`→sequential、`initial`。 + - omitted tier/mode/pass 分別記錄為 `auto`、`default`、`initial`;tier 由 + detector resolve,`default` mode 代表未明確選擇並由 [[CC-513]] policy + recommendation resolve,pass 預設為 initial。 3. Gate shell 在 dispatch 前決定 coordinates,並在 dispatch/wait 時機械擷取實際 reviewer/synthesis session evidence;reviewer LLM 不得自行宣稱 tier、mode、 coverage 或 independence。 @@ -1757,8 +1759,9 @@ generic 使用者會被不必要強制,maintainer 路徑則可能漏掉必要 1. 建立單一可測 resolver,輸入 diff classification、trusted brief metadata、 generated/untracked/renamed paths、requested tier/mode/pass/reviewers、repo policy 與 accepted-risk override;輸出: - `minimum_tier`、`required_reviewers`、`recommended_mode`、`required_mode|null`、 - `downgrade_allowed`、matched signals 與 override provenance。 + `minimum_tier`、`required_reviewers`、`recommended_mode`、mode selection + source/recommendation divergence、`downgrade_allowed`、matched signals 與 + override provenance。 2. **Generic `pmctl gate` risk-based floor**: - docs-only:critic/qa; - bounded runtime:critic/qa,依 matched signal增加 dimensions; @@ -1770,19 +1773,21 @@ generic 使用者會被不必要強制,maintainer 路徑則可能漏掉必要 3. **Maintainer `/ship` policy**:primary comprehensive review 固定要求 critic、 qa、architecture、security、risk 全 coverage,因 [[CC-517]] 預設只做一次 primary discovery;這是 repo-owned recipe,不強迫 generic gate。 -4. Mode 與 tier 分離:resolver 通常輸出 `recommended_mode: parallel`;只有 policy - 明確要求 reviewer isolation 才輸出 `required_mode: parallel`。`minimum_tier: - full` 本身不得暗示 parallel。 -5. requested tier/reviewer/mode 低於 resolver floor 時 fail closed,除非使用者提供 - scope-bounded accepted-risk override;security/risk hard-gate override 不能由 PM - 自行接受。未來 [[CC-065]] repo config 可加嚴,不得靜默降低 canonical floor。 +4. Mode 與 tier 分離且 mode 為 user-owned:使用者未指定時,resolver 才採用 + `recommended_mode` 自動選擇;明確 sequential/parallel 一律優先,偏離建議只記錄 + selection source 與 divergence,不視為 downgrade,也不要求 policy override。 + `minimum_tier: full` 本身不得強制 parallel。 +5. requested tier/reviewer 低於 resolver floor 時 fail closed,除非使用者提供 + scope-bounded accepted-risk override;mode 不屬於 downgrade allowance。 + security/risk hard-gate override 不能由 PM 自行接受。未來 [[CC-065]] repo config + 可加嚴 tier/coverage,不得靜默降低 canonical floor。 6. classification/resolution artifact 列出每個 matched path/field/signal、selected/ - skipped dimensions、recommendation vs requirement、downgrade reason 與 override - provenance;所有 consumer 使用同一 output,不各自複製 regex。 + skipped dimensions、mode recommendation/selection source/divergence、downgrade + reason 與 override provenance;所有 consumer 使用同一 output,不各自複製 regex。 -**Done-when**: 每份 gate artifact 都能機械回答「為什麼需要這個 minimum tier、 -reviewers 與 recommended/required mode」;generic 與 maintainer policy 可獨立測試, -full 不再隱含 parallel。 +**Done-when**: 每份 gate artifact 都能機械回答「為什麼需要這個 minimum tier/ +reviewers、policy 建議哪個 mode、以及最終是 user 或 policy 選擇」;generic 與 +maintainer policy 可獨立測試,full 不再隱含或強制 parallel。 **Non-goals**: 不以分類 signal 取代真正 review;不把 architecture reviewer 全域 升為 hard gate;不讓 maintainer recipe 改寫 generic defaults。 @@ -1941,8 +1946,8 @@ project 使用。 1. 更新 `commands/ship.md` 與 maintainer review model:primary implementation、 affected tests、refactor/reuse audit 完成後,只執行一次 comprehensive PR gate。 - [[CC-513]] maintainer policy 固定要求五 reviewer coverage;mode 由 policy - recommended/required mode 與 caller 選擇解析,不把 full coverage 寫成必然 + [[CC-513]] maintainer policy 固定要求五 reviewer coverage;mode 未指定時採 + policy recommendation,caller 明確選擇則優先,不把 full coverage 寫成必然 parallel。Gate 使用 [[CC-518]]~[[CC-521]] 的 structured outputs。 2. 產生 `remediation_closure_v1` evidence,至少含 primary gate/result/subject、 final subject、每個 stable finding ID 的 disposition、changed files、affected-test diff --git a/DECISIONS.md b/DECISIONS.md index f08cfdc8..a843d0c4 100644 --- a/DECISIONS.md +++ b/DECISIONS.md @@ -55,6 +55,10 @@ CC-520, CC-521 > 2026-07-27 clarification:本條目的 `targeted`-as-tier 部分已由 > `targeted-review-is-a-pass-kind-not-a-tier` 取代;其餘六維正交決策不變。 +> +> 2026-07-28 clarification:mode 是使用者擁有的成本/獨立性選擇。policy 只輸出 +> recommendation;使用者未指定時才自動採用,明確 sequential/parallel 永遠優先。 +> recommendation divergence 只記錄於 evidence,不是 downgrade,也不需要 override。 **Context**: 現有 runtime 已將 tier detection、reviewer selection 與 `SEQUENTIAL=true|false` 分開處理,但 review 文件與規畫曾把 `full`、五個 reviewers、 @@ -70,15 +74,16 @@ local closure、targeted confirmation 或拆票,但不建立新的 gate 或 wo 1. **Tier** 表示審查深度與 evidence floor(express/standard/full/targeted),不決定 execution topology。 2. **Mode** 表示 sequential combined session 或 parallel per-reviewer sessions; - 不提升 tier,也不保證 reviewer coverage。policy 可分別輸出 recommended mode 與 - required mode。 + 不提升 tier,也不保證 reviewer coverage。policy 輸出 recommended mode; + 明確 user choice 優先,只有 omitted mode 才採 recommendation。 3. **Reviewer coverage/independence** 記錄實際 selected/skipped reviewers、 implementation-context isolation 與 reviewer-to-reviewer context model;不得由 tier 或 reviewer 數量推論。 4. **Policy classification** 由 canonical resolver 根據 diff、brief、敏感 surface - 與 override 算出 minimum tier、required reviewers、recommended/required mode - 與 downgrade audit;generic `pmctl gate` risk-based policy 與 maintainer `/ship` - primary full-coverage policy分開。 + 與 override 算出 minimum tier、required reviewers、recommended mode、 + mode selection source/recommendation divergence 與 downgrade audit;generic + `pmctl gate` risk-based policy 與 maintainer `/ship` primary full-coverage + policy分開。 5. **Artifact subject** 以 stable repository identity、base/head commit、tree fingerprint 與 subject kind 說明 evidence 審查/測試了什麼;artifact validity、 subject freshness 與 consumer policy applicability 分開判斷。 @@ -87,10 +92,11 @@ local closure、targeted confirmation 或拆票,但不建立新的 gate 或 wo + closed remediation ledger + required targeted confirmations。branch/HEAD/tree、 manual evidence與 accepted-risk override 仍須符合 consumer policy。 -合法組合包含 express/standard/full/targeted × sequential/parallel;只有 policy -明確要求 reviewer isolation 時 `required_mode: parallel` 才是 hard requirement。 -Maintainer `/ship` 使用一次 primary comprehensive review、structured remediation -closure、必要時一次 targeted confirmation,再對 final tree 執行 affected/full tests。 +合法組合包含 express/standard/full/targeted × sequential/parallel;policy 不得把 +mode 變成 hard requirement。未指定時可自動採 recommendation,明確 user choice +不得被 resolver 改寫。Maintainer `/ship` 使用一次 primary comprehensive review、 +structured remediation closure、必要時一次 targeted confirmation,再對 final tree +執行 affected/full tests。 **Alternatives considered**: (a) `full => parallel => five reviewers => publishable` 單一 profile——否決,因深度、拓撲、coverage 與發布證據是不同問題,且與現有 runtime diff --git a/MILESTONES.md b/MILESTONES.md index 8ca85d75..16659bb6 100644 --- a/MILESTONES.md +++ b/MILESTONES.md @@ -108,7 +108,7 @@ | 票 | 摘要 | 狀態 | |----|------|------| | CC-512 | Slices A/B/C:coordinate sources/CLI resolution、machine-owned assurance envelope/evidence capture、shared verifier/parity ratchets;targeted 不再是 tier | ✅ pr:#451 | -| CC-513 | canonical resolver:minimum tier、required reviewers、recommended/required mode、generic vs maintainer policy 與 downgrade audit | 🔵 | +| CC-513 | canonical resolver:minimum tier、required reviewers、mode recommendation/user-choice provenance、generic vs maintainer policy 與 tier/coverage downgrade audit | 🔵 | | CC-515 | immutable subject;artifact validity、subject freshness、policy applicability 三軸 shared verifier | 🔵 | ### Phase 8 — existing gate structured evidence diff --git a/commands/pr-gate.md b/commands/pr-gate.md index 78b5ae7f..65ec0c3d 100644 --- a/commands/pr-gate.md +++ b/commands/pr-gate.md @@ -9,28 +9,35 @@ This command uses the `generic` consumer policy: one canonical resolver combines the diff, trusted brief metadata, requested tier/mode/pass/coverage, and repository policy before any reviewer is dispatched. -**Sequential mode (default):** all reviewers run in one combined session. -Low main-thread token cost (~5k dispatch + read result). +**Sequential mode (`--mode sequential`; `--sequential` is compatible):** all +reviewers run in one combined session. Lower token cost. **Parallel mode (`--mode parallel`; `--parallel` is compatible):** each reviewer runs in its own independent session followed by a PM synthesis session. Higher token cost — use for auth/payment/migration paths or when reviewer independence matters. +When mode is omitted, policy automatically selects its recommendation from the +consumer and matched risk signals. Any explicit sequential or parallel choice +wins; the assurance records when that choice differs from the recommendation, +but does not treat it as a downgrade. + | Situation | Args | |---|---| | Routine code / seed / docs changes | _(none)_ | | Re-gate after fixing specific findings | `--targeted qa-tester,risk-reviewer --initial-result ` | | Auth / payment / migration / sensitive paths | _(none; policy adds the matching reviewer and recommends parallel)_ | -| Input/evaluation/command execution boundary | _(none; policy requires parallel)_ | +| Input/evaluation/command execution boundary | _(none; policy recommends parallel)_ | +| Conserve reviewer-session token usage | `--mode sequential` | | Request independent reviewer sessions | `--mode parallel` | | Request a specific tier | `express` / `standard` / `full` (cannot lower the policy floor) | -A policy rejection happens before reviewer dispatch and is an execution/policy -failure, not a `Final: NO-GO` reviewer verdict. Scope-bound downgrades use the -runtime's explicit structured policy-override contract. Its scope fingerprint -binds the actual tracked patch plus in-scope untracked content, so an approval -cannot be replayed after a same-shape content change. `.gate-overrides.md` only -supplies reviewer finding context and cannot lower policy. +A tier or reviewer-coverage policy rejection happens before reviewer dispatch +and is an execution/policy failure, not a `Final: NO-GO` reviewer verdict. +Scope-bound tier/coverage downgrades use the runtime's explicit structured +policy-override contract. Its scope fingerprint binds the actual tracked patch +plus in-scope untracked content, so an approval cannot be replayed after a +same-shape content change. `.gate-overrides.md` only supplies reviewer finding +context and cannot lower policy. ## Step 1 - Invoke pmctl directly @@ -312,6 +319,10 @@ When the `pmctl gate wait` background Bash completion notification arrives: ## Local verification after gate findings +For a follow-up gate after a complete analysis or `Final: NO-GO`, preserve the +user's explicit `--mode sequential` or `--mode parallel` choice. Omit the flag +only when the user has not chosen a mode and wants policy to infer it again. + After fixing a NO-GO finding, run only the affected tests before re-gating: ```bash diff --git a/commands/ship.md b/commands/ship.md index 4b3f11a1..b50e6b4c 100644 --- a/commands/ship.md +++ b/commands/ship.md @@ -133,7 +133,11 @@ using the authoritative [gate model diversity policy](../docs/review-model.md#ga Base the choice on actual model identities, record both identities in the handoff, and keep the same resolved pair for targeted re-runs. Add `--model ""` below only when the executor default does not already -resolve to the selected gate model. +resolve to the selected gate model. Mode is user-owned: append the literal +`--mode sequential` when token budget favors one combined reviewer session, or +`--mode parallel` when independent sessions are desired. If neither is present, +the gate auto-selects the policy recommendation. Preserve the user's explicit +choice on targeted re-runs unless the user changes it. Run `pmctl gate run --executor --policy maintainer --cd "" --lifecycle foreground` (substitute `` with the literal absolute working directory, not diff --git a/core/README.md b/core/README.md index db655a7b..a3d54842 100644 --- a/core/README.md +++ b/core/README.md @@ -17,7 +17,8 @@ Gate assurance definitions are split deliberately: - `policy/gate-policy-consumers.tsv` and `gate-policy-signals.tsv` define consumer-specific coverage and deterministic risk floors. - `schema/gate-policy-override.schema.json` defines explicit scope-bound user - approval for a downgrade. + approval for a tier or reviewer-coverage downgrade; mode remains a direct + user choice. - `schema/gate-assurance.schema.json` defines the portable envelope that records both resolved coordinates and the policy resolution that produced them. diff --git a/core/policy/gate-modes.tsv b/core/policy/gate-modes.tsv index 96ffc411..922c7faa 100644 --- a/core/policy/gate-modes.tsv +++ b/core/policy/gate-modes.tsv @@ -1,4 +1,5 @@ # Gate execution topology. Mode does not imply tier or reviewer coverage. -mode topology synthesis is_default -sequential combined-session inline true -parallel per-reviewer-sessions separate-session false +# Omitted mode is resolved from policy recommendations, not from this table. +mode topology synthesis +sequential combined-session inline +parallel per-reviewer-sessions separate-session diff --git a/core/policy/gate-policy-consumers.tsv b/core/policy/gate-policy-consumers.tsv index 1e3c94b8..0bad7fca 100644 --- a/core/policy/gate-policy-consumers.tsv +++ b/core/policy/gate-policy-consumers.tsv @@ -1,6 +1,6 @@ -# Gate policy consumers. Consumer policy does not rewrite tier, mode, or pass semantics. -policy_pass policy pass_kind minimum_tier required_reviewers recommended_mode required_mode -generic:initial generic initial express critic,qa-tester sequential none -generic:targeted generic targeted express none sequential none -maintainer:initial maintainer initial express critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none -maintainer:targeted maintainer targeted express none parallel none +# Gate policy consumers. Mode is inferred from recommended_mode only when the user does not select one explicitly. +policy_pass policy pass_kind minimum_tier required_reviewers recommended_mode +generic:initial generic initial express critic,qa-tester sequential +generic:targeted generic targeted express none sequential +maintainer:initial maintainer initial express critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel +maintainer:targeted maintainer targeted express none parallel diff --git a/core/policy/gate-policy-signals.tsv b/core/policy/gate-policy-signals.tsv index 5c167527..5e0ff6b0 100644 --- a/core/policy/gate-policy-signals.tsv +++ b/core/policy/gate-policy-signals.tsv @@ -1,18 +1,18 @@ # Gate policy signals. Reviewer requirements apply to initial discovery; targeted passes retain tier/mode signals but use requested remediation coverage. -signal match_source pattern minimum_tier required_reviewers recommended_mode required_mode -docs-only classification docs-only express none sequential none -bounded-runtime classification bounded-runtime express none sequential none -medium-change classification medium-change standard architecture-reviewer parallel none -large-change classification large-change full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none -binary-change classification binary-change standard architecture-reviewer parallel none -renamed-input classification renamed express none sequential none -untracked-input classification untracked express none sequential none -generated-input classification generated express none sequential none -cross-boundary classification cross-boundary standard architecture-reviewer parallel none -security-sensitive-path path-regex (^|[/_.-])(auth|oauth|jwt|sessions?|secrets?|passwords?|tokens?|credentials?|cors|csrf|webhooks?|sudo|ssh|payments?|billing)([/_.-]|$) express security-reviewer parallel none -input-execution-path path-regex (^|[/_.-])(eval|exec|execute|command|shell|hook|guard|allowlist)([/_.-]|$)|(^|/)(\.github|workflows?|ci)(/|$) standard security-reviewer parallel parallel -risk-sensitive-path path-regex (^|[/_.-])(migrations?|migrate|destructive|deletions?|delete|removals?|remove|rollback|concurrency|concurrent|race|locks?|cancel|reconcile)([/_.-]|$) express risk-reviewer parallel none -public-contract-path path-regex (^|/)(cli|commands|skills|core/schema)(/|$)|(^|[/_.-])(apis?|schemas?|contracts?)([/_.-]|$) standard architecture-reviewer parallel none -policy-source-path path-regex (^|/)core/policy(/|$) full architecture-reviewer,security-reviewer,risk-reviewer parallel none -brief-architecture-minor brief-value minor standard architecture-reviewer parallel none -brief-architecture-major brief-value major full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +signal match_source pattern minimum_tier required_reviewers recommended_mode +docs-only classification docs-only express none sequential +bounded-runtime classification bounded-runtime express none sequential +medium-change classification medium-change standard architecture-reviewer parallel +large-change classification large-change full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel +binary-change classification binary-change standard architecture-reviewer parallel +renamed-input classification renamed express none sequential +untracked-input classification untracked express none sequential +generated-input classification generated express none sequential +cross-boundary classification cross-boundary standard architecture-reviewer parallel +security-sensitive-path path-regex (^|[/_.-])(auth|oauth|jwt|sessions?|secrets?|passwords?|tokens?|credentials?|cors|csrf|webhooks?|sudo|ssh|payments?|billing)([/_.-]|$) express security-reviewer parallel +input-execution-path path-regex (^|[/_.-])(eval|exec|execute|command|shell|hook|guard|allowlist)([/_.-]|$)|(^|/)(\.github|workflows?|ci)(/|$) standard security-reviewer parallel +risk-sensitive-path path-regex (^|[/_.-])(migrations?|migrate|destructive|deletions?|delete|removals?|remove|rollback|concurrency|concurrent|race|locks?|cancel|reconcile)([/_.-]|$) express risk-reviewer parallel +public-contract-path path-regex (^|/)(cli|commands|skills|core/schema)(/|$)|(^|[/_.-])(apis?|schemas?|contracts?)([/_.-]|$) standard architecture-reviewer parallel +policy-source-path path-regex (^|/)core/policy(/|$) full architecture-reviewer,security-reviewer,risk-reviewer parallel +brief-architecture-minor brief-value minor standard architecture-reviewer parallel +brief-architecture-major brief-value major full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel diff --git a/core/schema/gate-assurance.schema.json b/core/schema/gate-assurance.schema.json index e71b4362..99cb59ef 100644 --- a/core/schema/gate-assurance.schema.json +++ b/core/schema/gate-assurance.schema.json @@ -266,7 +266,8 @@ "minimum_tier", "required_reviewers", "recommended_mode", - "required_mode", + "mode_selection_source", + "mode_recommendation_overridden", "downgrade_requested", "downgrade_allowed" ], @@ -287,17 +288,15 @@ "parallel" ] }, - "required_mode": { - "type": [ - "string", - "null" - ], + "mode_selection_source": { "enum": [ - "sequential", - "parallel", - null + "user", + "policy" ] }, + "mode_recommendation_overridden": { + "type": "boolean" + }, "downgrade_requested": { "type": "boolean" }, @@ -318,8 +317,7 @@ "matches", "minimum_tier", "required_reviewers", - "recommended_mode", - "required_mode" + "recommended_mode" ], "properties": { "id": { @@ -352,17 +350,6 @@ "sequential", "parallel" ] - }, - "required_mode": { - "type": [ - "string", - "null" - ], - "enum": [ - "sequential", - "parallel", - null - ] } }, "additionalProperties": false diff --git a/core/schema/gate-policy-override.schema.json b/core/schema/gate-policy-override.schema.json index 29f7e9ad..2f86686b 100644 --- a/core/schema/gate-policy-override.schema.json +++ b/core/schema/gate-policy-override.schema.json @@ -1,6 +1,6 @@ { - "title": "Gate policy downgrade override", - "description": "Explicit user authorization for a scope-bound gate policy downgrade.", + "title": "Gate tier and coverage downgrade override", + "description": "Explicit user authorization for a scope-bound gate tier or reviewer-coverage downgrade. Execution mode is selected directly by the user and is not an override coordinate.", "type": "object", "required": [ "kind", @@ -25,8 +25,7 @@ "type": "object", "required": [ "tier", - "omit_reviewers", - "mode" + "omit_reviewers" ], "properties": { "tier": { @@ -48,17 +47,6 @@ "type": "string", "pattern": "^[a-z0-9][a-z0-9-]*$" } - }, - "mode": { - "type": [ - "string", - "null" - ], - "enum": [ - "sequential", - "parallel", - null - ] } }, "additionalProperties": false diff --git a/docs/review-model.md b/docs/review-model.md index a63d4ef5..f9a95a42 100644 --- a/docs/review-model.md +++ b/docs/review-model.md @@ -41,11 +41,11 @@ The point is not to produce documentation — it is to catch a wrong direction b **When**: after the executor finishes; triggered by `/pr-gate`. -**Mechanism**: `/pr-gate` runs its reviewers (`critic`, `qa-tester`, `security-reviewer`, `risk-reviewer`, `architecture-reviewer`) in a review session *separate from the one that produced the diff*. Each reviewer receives a brief with the diff, task context, and (if present) the conceptual map — but not the dispatch session's reasoning, prior attempts, or implicit anchors. The default sequential gate runs the reviewers together in one fresh session (codex or Claude, per `--executor`); the `--parallel` path gives each reviewer its own independent session followed by a PM synthesis pass. +**Mechanism**: `/pr-gate` runs its reviewers (`critic`, `qa-tester`, `security-reviewer`, `risk-reviewer`, `architecture-reviewer`) in a review session *separate from the one that produced the diff*. Each reviewer receives a brief with the diff, task context, and (if present) the conceptual map — but not the dispatch session's reasoning, prior attempts, or implicit anchors. A sequential gate runs the reviewers together in one fresh session (codex or Claude, per `--executor`); the parallel path gives each reviewer its own independent session followed by a PM synthesis pass. An explicit user choice wins; when mode is omitted, policy selects its recommendation. This isolation matters because context-window anchoring is real: a reviewer who watched the code being written will tend to evaluate the approach rather than the outcome. A reviewer who arrives at the finished diff cold asks "does this make sense in isolation?" — a harder and more valuable question. -The isolation *from the dispatch session* is structural, not advisory: the review session is a separate process and cannot read the implementer's context. Per-reviewer independence — each reviewer also blind to the others — is provided by the `--parallel` path; the sequential default trades that for a single combined review session. See [pr-gate-handover-schema.md](pr-gate-handover-schema.md) for the handover protocol. +The isolation *from the dispatch session* is structural, not advisory: the review session is a separate process and cannot read the implementer's context. Per-reviewer independence — each reviewer also blind to the others — is provided by the parallel path; sequential mode trades that for a single combined review session. See [pr-gate-handover-schema.md](pr-gate-handover-schema.md) for the handover protocol. ### Layer 3 — Conceptual Map review @@ -166,11 +166,11 @@ to required coverage without automatically converting every bounded change to - Tier records rigor intent and supplies default reviewer coverage. An explicit `--reviewers` list does not rewrite the tier, but it must still include every reviewer required by the matched risk signals. -- Mode records execution topology. The default is `sequential`; select - `--mode parallel` when separate reviewer sessions and synthesis are desired. - Policy records recommendation separately from requirement; only an explicit - isolation signal requires parallel mode. A `full` tier does not select - parallel mode by itself. +- Mode records execution topology and remains a user-owned cost/independence + choice. When mode is omitted, policy auto-selects its recommendation from the + consumer and matched signals. Explicit `--mode sequential` or + `--mode parallel` always wins; the envelope records recommendation divergence + without treating it as a downgrade. A `full` tier does not force parallel. - Pass kind records whether the review is initial or a remediation-delta targeted pass. `--targeted ` requires `--initial-result ` and is not a tier alias. @@ -181,12 +181,13 @@ pass fixes coverage at all five reviewer dimensions while preserving the independently resolved tier and mode. Targeted passes under either consumer remain scoped to the requested remediation reviewers. -Any requested tier, coverage, or mode below the policy floor fails before -reviewer dispatch. A downgrade is accepted only through an explicitly supplied -`gate_policy_override_v1` JSON file bound to the exact scope fingerprint and -recording user approval. The fingerprint includes the content-addressed tracked -patch and every in-scope untracked file, not only file names or aggregate line -counts. The free-form `.gate-overrides.md` file remains reviewer +Any requested tier or coverage below the policy floor fails before reviewer +dispatch. A tier/coverage downgrade is accepted only through an explicitly +supplied `gate_policy_override_v1` JSON file bound to the exact scope fingerprint +and recording user approval. Mode is not a downgrade coordinate: explicit user +selection needs no override. The fingerprint includes the content-addressed +tracked patch and every in-scope untracked file, not only file names or aggregate +line counts. The free-form `.gate-overrides.md` file remains reviewer finding/suppression context; it is recorded separately and cannot authorize a policy downgrade. @@ -209,8 +210,9 @@ requested/resolved coordinates, selected/skipped coverage, actual dispatch outcomes, run IDs, subject commits/fingerprint, and the evidence status behind independence claims. New envelopes also embed the canonical policy result: classification facts, every matched signal and path, minimum tier, required -coverage, recommended versus required mode, enforcement status, and both -policy-override and reviewer-override provenance. Repo-layout results with +coverage, recommended mode, whether policy or the user selected the mode, +recommendation divergence, enforcement status, and both policy-override and +reviewer-override provenance. Repo-layout results with verified independence also carry a shell-owned attestation in the protected gate run directory. `pmctl gate verify` validates result/sidecar digests and resolves every claimed run ID diff --git a/runtime/bin/pr-gate.sh b/runtime/bin/pr-gate.sh index b242413f..765b4c03 100755 --- a/runtime/bin/pr-gate.sh +++ b/runtime/bin/pr-gate.sh @@ -40,9 +40,10 @@ GATE_ASSURANCE_TIERS_TSV # BEGIN GENERATED from core/policy/gate-modes.tsv cat <<'GATE_ASSURANCE_MODES_TSV' # Gate execution topology. Mode does not imply tier or reviewer coverage. -mode topology synthesis is_default -sequential combined-session inline true -parallel per-reviewer-sessions separate-session false +# Omitted mode is resolved from policy recommendations, not from this table. +mode topology synthesis +sequential combined-session inline +parallel per-reviewer-sessions separate-session GATE_ASSURANCE_MODES_TSV # END GENERATED from core/policy/gate-modes.tsv ;; @@ -59,12 +60,12 @@ GATE_ASSURANCE_PASS_KINDS_TSV consumers) # BEGIN GENERATED from core/policy/gate-policy-consumers.tsv cat <<'GATE_POLICY_CONSUMERS_TSV' -# Gate policy consumers. Consumer policy does not rewrite tier, mode, or pass semantics. -policy_pass policy pass_kind minimum_tier required_reviewers recommended_mode required_mode -generic:initial generic initial express critic,qa-tester sequential none -generic:targeted generic targeted express none sequential none -maintainer:initial maintainer initial express critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none -maintainer:targeted maintainer targeted express none parallel none +# Gate policy consumers. Mode is inferred from recommended_mode only when the user does not select one explicitly. +policy_pass policy pass_kind minimum_tier required_reviewers recommended_mode +generic:initial generic initial express critic,qa-tester sequential +generic:targeted generic targeted express none sequential +maintainer:initial maintainer initial express critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel +maintainer:targeted maintainer targeted express none parallel GATE_POLICY_CONSUMERS_TSV # END GENERATED from core/policy/gate-policy-consumers.tsv ;; @@ -72,23 +73,23 @@ GATE_POLICY_CONSUMERS_TSV # BEGIN GENERATED from core/policy/gate-policy-signals.tsv cat <<'GATE_POLICY_SIGNALS_TSV' # Gate policy signals. Reviewer requirements apply to initial discovery; targeted passes retain tier/mode signals but use requested remediation coverage. -signal match_source pattern minimum_tier required_reviewers recommended_mode required_mode -docs-only classification docs-only express none sequential none -bounded-runtime classification bounded-runtime express none sequential none -medium-change classification medium-change standard architecture-reviewer parallel none -large-change classification large-change full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none -binary-change classification binary-change standard architecture-reviewer parallel none -renamed-input classification renamed express none sequential none -untracked-input classification untracked express none sequential none -generated-input classification generated express none sequential none -cross-boundary classification cross-boundary standard architecture-reviewer parallel none -security-sensitive-path path-regex (^|[/_.-])(auth|oauth|jwt|sessions?|secrets?|passwords?|tokens?|credentials?|cors|csrf|webhooks?|sudo|ssh|payments?|billing)([/_.-]|$) express security-reviewer parallel none -input-execution-path path-regex (^|[/_.-])(eval|exec|execute|command|shell|hook|guard|allowlist)([/_.-]|$)|(^|/)(\.github|workflows?|ci)(/|$) standard security-reviewer parallel parallel -risk-sensitive-path path-regex (^|[/_.-])(migrations?|migrate|destructive|deletions?|delete|removals?|remove|rollback|concurrency|concurrent|race|locks?|cancel|reconcile)([/_.-]|$) express risk-reviewer parallel none -public-contract-path path-regex (^|/)(cli|commands|skills|core/schema)(/|$)|(^|[/_.-])(apis?|schemas?|contracts?)([/_.-]|$) standard architecture-reviewer parallel none -policy-source-path path-regex (^|/)core/policy(/|$) full architecture-reviewer,security-reviewer,risk-reviewer parallel none -brief-architecture-minor brief-value minor standard architecture-reviewer parallel none -brief-architecture-major brief-value major full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel none +signal match_source pattern minimum_tier required_reviewers recommended_mode +docs-only classification docs-only express none sequential +bounded-runtime classification bounded-runtime express none sequential +medium-change classification medium-change standard architecture-reviewer parallel +large-change classification large-change full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel +binary-change classification binary-change standard architecture-reviewer parallel +renamed-input classification renamed express none sequential +untracked-input classification untracked express none sequential +generated-input classification generated express none sequential +cross-boundary classification cross-boundary standard architecture-reviewer parallel +security-sensitive-path path-regex (^|[/_.-])(auth|oauth|jwt|sessions?|secrets?|passwords?|tokens?|credentials?|cors|csrf|webhooks?|sudo|ssh|payments?|billing)([/_.-]|$) express security-reviewer parallel +input-execution-path path-regex (^|[/_.-])(eval|exec|execute|command|shell|hook|guard|allowlist)([/_.-]|$)|(^|/)(\.github|workflows?|ci)(/|$) standard security-reviewer parallel +risk-sensitive-path path-regex (^|[/_.-])(migrations?|migrate|destructive|deletions?|delete|removals?|remove|rollback|concurrency|concurrent|race|locks?|cancel|reconcile)([/_.-]|$) express risk-reviewer parallel +public-contract-path path-regex (^|/)(cli|commands|skills|core/schema)(/|$)|(^|[/_.-])(apis?|schemas?|contracts?)([/_.-]|$) standard architecture-reviewer parallel +policy-source-path path-regex (^|/)core/policy(/|$) full architecture-reviewer,security-reviewer,risk-reviewer parallel +brief-architecture-minor brief-value minor standard architecture-reviewer parallel +brief-architecture-major brief-value major full critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer parallel GATE_POLICY_SIGNALS_TSV # END GENERATED from core/policy/gate-policy-signals.tsv ;; @@ -308,20 +309,20 @@ _gate_policy_validate_reviewer_csv() { _gate_policy_validate_sources() { local vocabulary="${1:-}" policy_pass policy pass_kind minimum_tier - local required_reviewers recommended_mode required_mode consumer_keys="" + local required_reviewers recommended_mode consumer_keys="" local signal match_source pattern signal_tier signal_reviewers - local signal_recommended signal_required grep_status + local signal_recommended grep_status [[ $# -eq 1 && -n "$vocabulary" ]] || return 2 _gate_policy_source_shape_validate consumers \ - $'policy_pass\tpolicy\tpass_kind\tminimum_tier\trequired_reviewers\trecommended_mode\trequired_mode' \ + $'policy_pass\tpolicy\tpass_kind\tminimum_tier\trequired_reviewers\trecommended_mode' \ || return 2 _gate_policy_source_shape_validate signals \ - $'signal\tmatch_source\tpattern\tminimum_tier\trequired_reviewers\trecommended_mode\trequired_mode' \ + $'signal\tmatch_source\tpattern\tminimum_tier\trequired_reviewers\trecommended_mode' \ || return 2 while IFS=$'\t' read -r policy_pass policy pass_kind minimum_tier \ - required_reviewers recommended_mode required_mode; do + required_reviewers recommended_mode; do [[ -n "$policy_pass" && "$policy_pass" != \#* \ && "$policy_pass" != policy_pass ]] || continue case "$policy" in generic|maintainer) ;; *) @@ -355,14 +356,6 @@ _gate_policy_validate_sources() { "$policy_pass" "$recommended_mode" >&2 return 2 } - if [[ "$required_mode" != none ]]; then - _gate_assurance_policy_lookup modes mode "$required_mode" topology >/dev/null \ - || { - printf 'Error: gate policy consumer %s has invalid required mode: %s\n' \ - "$policy_pass" "$required_mode" >&2 - return 2 - } - fi consumer_keys="${consumer_keys:+$consumer_keys }$policy_pass" done < <(_gate_assurance_policy_emit consumers) @@ -381,7 +374,7 @@ _gate_policy_validate_sources() { fi while IFS=$'\t' read -r signal match_source pattern signal_tier \ - signal_reviewers signal_recommended signal_required; do + signal_reviewers signal_recommended; do [[ -n "$signal" && "$signal" != \#* && "$signal" != signal ]] || continue if [[ ! "$signal" =~ ^[a-z0-9][a-z0-9-]*$ \ || "$signal" == consumer-policy ]]; then @@ -440,14 +433,6 @@ _gate_policy_validate_sources() { "$signal" "$signal_recommended" >&2 return 2 } - if [[ "$signal_required" != none ]]; then - _gate_assurance_policy_lookup modes mode "$signal_required" topology >/dev/null \ - || { - printf 'Error: gate policy signal %s has invalid required mode: %s\n' \ - "$signal" "$signal_required" >&2 - return 2 - } - fi done < <(_gate_assurance_policy_emit signals) } @@ -528,18 +513,19 @@ _gate_policy_scope_content_digest() { _gate_policy_resolve() { local input_json="${1:-}" policy_override="${2:-}" local policy pass_kind policy_pass scope_fingerprint vocabulary - local minimum_tier required_reviewers recommended_mode required_mode + local minimum_tier required_reviewers recommended_mode local signal match_source pattern signal_tier signal_reviewers local normalized_signal_reviewers effective_signal_reviewers - local signal_recommended signal_required matches_json matches_text + local signal_recommended matches_json matches_text local current_rank candidate_rank signal_json signals_file local requested_tier requested_mode requested_reviewers_json - local resolved_tier resolved_mode tier_defaults selected_reviewers - local missing_reviewers="" reviewer tier_violation=false mode_violation=false + local resolved_tier resolved_mode mode_selection_source + local mode_recommendation_overridden=false tier_defaults selected_reviewers + local missing_reviewers="" reviewer tier_violation=false local violations_json downgrade_requested=false downgrade_allowed=false local enforcement_status=pass override_status=not_provided local override_sha="" override_reason="" override_approver_json=null - local expected_tier=null expected_mode=null missing_json override_json=null + local expected_tier=null missing_json override_json=null local reviewer_override_json classification_json policy_source [[ $# -ge 1 && $# -le 2 ]] || return 2 @@ -583,8 +569,6 @@ _gate_policy_resolve() { || return 2 recommended_mode="$(_gate_assurance_policy_lookup consumers policy_pass "$policy_pass" recommended_mode)" \ || return 2 - required_mode="$(_gate_assurance_policy_lookup consumers policy_pass "$policy_pass" required_mode)" \ - || return 2 required_reviewers="$(_gate_policy_add_reviewers "" "$required_reviewers" "$vocabulary")" \ || return 2 @@ -600,24 +584,14 @@ _gate_policy_resolve() { "$recommended_mode" >&2 return 2 } - if [[ "$required_mode" != none ]]; then - _gate_assurance_policy_lookup modes mode "$required_mode" topology >/dev/null \ - || { - printf 'Error: gate policy consumer has invalid required mode: %s\n' \ - "$required_mode" >&2 - return 2 - } - fi - signals_file="$(mktemp "${TMPDIR:-/tmp}/gate-policy-signals.XXXXXX")" || return 2 signal_json="$(jq -nc \ --arg id "consumer-policy" --arg source "consumer-policy" \ --arg match "$policy_pass" --arg minimum_tier "$minimum_tier" \ --argjson required_reviewers "$(_gate_policy_words_json "$required_reviewers")" \ - --arg recommended_mode "$recommended_mode" --arg required_mode "$required_mode" '{ + --arg recommended_mode "$recommended_mode" '{ id:$id,source:$source,matches:[$match],minimum_tier:$minimum_tier, - required_reviewers:$required_reviewers,recommended_mode:$recommended_mode, - required_mode:(if $required_mode == "none" then null else $required_mode end) + required_reviewers:$required_reviewers,recommended_mode:$recommended_mode }')" || { rm -f "$signals_file" return 2 @@ -625,11 +599,10 @@ _gate_policy_resolve() { printf '%s\n' "$signal_json" > "$signals_file" while IFS=$'\t' read -r signal match_source pattern signal_tier \ - signal_reviewers signal_recommended signal_required; do + signal_reviewers signal_recommended; do [[ -n "$signal" && "$signal" != \#* && "$signal" != signal ]] || continue if [[ -z "$match_source" || -z "$pattern" || -z "$signal_tier" \ - || -z "$signal_reviewers" || -z "$signal_recommended" \ - || -z "$signal_required" ]]; then + || -z "$signal_reviewers" || -z "$signal_recommended" ]]; then printf 'Error: malformed gate policy signal row: %s\n' "$signal" >&2 rm -f "$signals_file" return 2 @@ -709,25 +682,14 @@ _gate_policy_resolve() { return 2 ;; esac - if [[ "$signal_required" != none ]]; then - if [[ "$required_mode" != none && "$required_mode" != "$signal_required" ]]; then - printf 'Error: conflicting required gate modes: %s and %s\n' \ - "$required_mode" "$signal_required" >&2 - rm -f "$signals_file" - return 2 - fi - required_mode="$signal_required" - fi signal_json="$(jq -nc \ --arg id "$signal" --arg source "$match_source" \ --argjson matches "$matches_json" --arg minimum_tier "$signal_tier" \ --argjson required_reviewers \ "$(_gate_policy_words_json "$effective_signal_reviewers")" \ - --arg recommended_mode "$signal_recommended" \ - --arg required_mode "$signal_required" '{ + --arg recommended_mode "$signal_recommended" '{ id:$id,source:$source,matches:$matches,minimum_tier:$minimum_tier, - required_reviewers:$required_reviewers,recommended_mode:$recommended_mode, - required_mode:(if $required_mode == "none" then null else $required_mode end) + required_reviewers:$required_reviewers,recommended_mode:$recommended_mode }')" || { rm -f "$signals_file" return 2 @@ -754,16 +716,13 @@ _gate_policy_resolve() { fi if [[ "$requested_mode" == default ]]; then - if [[ "$required_mode" == none ]]; then - resolved_mode="$GATE_MODE_DEFAULT" - else - resolved_mode="$required_mode" - fi + resolved_mode="$recommended_mode" + mode_selection_source=policy else resolved_mode="$requested_mode" - if [[ "$required_mode" != none && "$resolved_mode" != "$required_mode" ]]; then - mode_violation=true - fi + mode_selection_source=user + [[ "$resolved_mode" == "$recommended_mode" ]] \ + || mode_recommendation_overridden=true fi if [[ "$requested_reviewers_json" == null ]]; then @@ -796,17 +755,12 @@ _gate_policy_resolve() { violations_json="$(jq -nc \ --argjson tier_violation "$tier_violation" \ --arg requested_tier "$requested_tier" --arg minimum_tier "$minimum_tier" \ - --argjson missing_reviewers "$missing_json" \ - --argjson mode_violation "$mode_violation" \ - --arg requested_mode "$requested_mode" --arg required_mode "$required_mode" '[ + --argjson missing_reviewers "$missing_json" '[ if $tier_violation then { coordinate:"tier",requested:$requested_tier,required:$minimum_tier } else empty end, if ($missing_reviewers | length) > 0 then { coordinate:"coverage",requested:"explicit",required:$missing_reviewers - } else empty end, - if $mode_violation then { - coordinate:"mode",requested:$requested_mode,required:$required_mode } else empty end ]')" if [[ "$(jq -r 'length' <<<"$violations_json")" -gt 0 ]]; then @@ -831,9 +785,8 @@ _gate_policy_resolve() { .kind == "gate_policy_override_v1" and .schema_version == 1 and (.scope_fingerprint | type == "string" and test("^[a-f0-9]{64}$")) and (.allow | type == "object" and - (keys | sort) == (["mode","omit_reviewers","tier"] | sort)) and + (keys | sort) == (["omit_reviewers","tier"] | sort)) and (.allow.tier == null or (.allow.tier | IN("express","standard","full"))) and - (.allow.mode == null or (.allow.mode | IN("sequential","parallel"))) and (.allow.omit_reviewers | type == "array" and all(.[]; type == "string" and test("^[a-z0-9][a-z0-9-]*$")) and length == (unique | length)) and @@ -856,12 +809,9 @@ _gate_policy_resolve() { override_status=scope_mismatch else [[ "$tier_violation" == true ]] && expected_tier="$(jq -nc --arg value "$requested_tier" '$value')" - [[ "$mode_violation" == true ]] && expected_mode="$(jq -nc --arg value "$requested_mode" '$value')" if jq -e --argjson expected_tier "$expected_tier" \ - --argjson expected_mode "$expected_mode" \ --argjson expected_reviewers "$missing_json" ' .allow.tier == $expected_tier and - .allow.mode == $expected_mode and (.allow.omit_reviewers | sort) == ($expected_reviewers | sort) ' "$policy_override" >/dev/null; then override_status=applied @@ -894,7 +844,9 @@ _gate_policy_resolve() { --argjson classification "$classification_json" \ --arg minimum_tier "$minimum_tier" \ --argjson required_reviewers "$(_gate_policy_words_json "$required_reviewers")" \ - --arg recommended_mode "$recommended_mode" --arg required_mode "$required_mode" \ + --arg recommended_mode "$recommended_mode" \ + --arg mode_selection_source "$mode_selection_source" \ + --argjson mode_recommendation_overridden "$mode_recommendation_overridden" \ --argjson downgrade_requested "$downgrade_requested" \ --argjson downgrade_allowed "$downgrade_allowed" \ --argjson matched_signals "$signal_json" \ @@ -919,7 +871,8 @@ _gate_policy_resolve() { minimum_tier:$minimum_tier, required_reviewers:$required_reviewers, recommended_mode:$recommended_mode, - required_mode:(if $required_mode == "none" then null else $required_mode end), + mode_selection_source:$mode_selection_source, + mode_recommendation_overridden:$mode_recommendation_overridden, downgrade_requested:$downgrade_requested, downgrade_allowed:$downgrade_allowed }, @@ -1024,7 +977,7 @@ _kill_process_tree() { # pr-gate-help:start # pr-gate.sh -- PR-gate review via a dispatched session # -# DEFAULT (single-session / sequential): +# SINGLE-SESSION (--mode sequential; --sequential is compatible): # All reviewers run in order inside ONE combined dispatch session. # Lower token cost. All reviewer findings appear in a single output file. # Use this for most routine changes. @@ -1036,6 +989,9 @@ _kill_process_tree() { # Higher token cost. Use for auth/payment/migration paths or when reviewer # independence is worth the extra cost. # +# Explicit user mode always wins. When mode is omitted, policy automatically +# selects its recommendation from the consumer and matched risk signals. +# # Adjacent test files (not in the diff but directly paired to a changed source # file) are automatically added to every reviewer brief so coverage gaps in # unchanged test files are visible to the gate. @@ -1046,7 +1002,7 @@ _kill_process_tree() { # Options: # --cd working directory (required) # --tier express|standard|full -- overrides auto-detection -# --mode sequential|parallel execution topology (default: sequential) +# --mode sequential|parallel; omitted mode follows the policy recommendation # --brief dispatch brief; trusted architecture_impact contributes to policy resolution # --policy generic|maintainer consumer policy (default: generic) # --reviewers comma-separated requested coverage; does not change tier or pass kind @@ -1418,7 +1374,8 @@ else "binary_or_unknown_count","layer_roots"])) and (.policy.resolution | only_keys(["minimum_tier","required_reviewers","recommended_mode", - "required_mode","downgrade_requested","downgrade_allowed"])) and + "mode_selection_source","mode_recommendation_overridden", + "downgrade_requested","downgrade_allowed"])) and (.policy.resolved | only_keys(["tier","mode","reviewers"])) and (.policy.enforcement | only_keys(["status","violations"])) and (.policy.override | @@ -1454,25 +1411,33 @@ else $policy.resolution.downgrade_allowed)) and (.policy.resolution.recommended_mode | IN("sequential","parallel")) and - (.policy.resolution.required_mode == null or - (.policy.resolution.required_mode | - IN("sequential","parallel"))) and + (.policy.resolution.mode_selection_source | IN("user","policy")) and + (.policy.resolution.mode_recommendation_overridden | type == "boolean") and + (if .policy.request.mode == "default" + then + .policy.resolution.mode_selection_source == "policy" and + .policy.resolved.mode == .policy.resolution.recommended_mode and + .policy.resolution.mode_recommendation_overridden == false + else + .policy.resolution.mode_selection_source == "user" and + .policy.resolved.mode == .policy.request.mode and + .policy.resolution.mode_recommendation_overridden == + (.policy.request.mode != .policy.resolution.recommended_mode) + end) and (.policy.resolution.downgrade_requested | type == "boolean") and (.policy.resolution.downgrade_allowed | type == "boolean") and (.policy.matched_signals | type == "array" and length > 0) and ([.policy.matched_signals[].id] | strings_unique) and (all(.policy.matched_signals[]; only_keys(["id","source","matches","minimum_tier", - "required_reviewers","recommended_mode","required_mode"]) and + "required_reviewers","recommended_mode"]) and (.id | type == "string" and length > 0) and (.source | IN("consumer-policy","classification","path-regex","brief-value")) and (.matches | strings_unique and length > 0) and (.minimum_tier | IN("express","standard","full")) and (.required_reviewers | strings_unique) and - (.recommended_mode | IN("sequential","parallel")) and - (.required_mode == null or - (.required_mode | IN("sequential","parallel"))))) and + (.recommended_mode | IN("sequential","parallel")))) and .policy.resolved.tier == .coordinates.tier.resolved and .policy.resolved.mode == .coordinates.mode.resolved and same_set(.policy.resolved.reviewers; @@ -1481,7 +1446,7 @@ else (.policy.enforcement.violations | type == "array") and (all(.policy.enforcement.violations[]; only_keys(["coordinate","requested","required"]) and - (.coordinate | IN("tier","coverage","mode")))) and + (.coordinate | IN("tier","coverage")))) and (.policy.override.status | IN("not_provided","not_needed","applied","scope_mismatch", "allowance_mismatch")) and @@ -1700,18 +1665,10 @@ if ! GATE_PASS_KIND_VALUES="$(_gate_assurance_policy_values pass-kinds pass_kind exit 2 fi -if ! GATE_MODE_DEFAULT="$(_gate_assurance_policy_lookup modes is_default true mode)"; then - printf 'Error: gate mode policy must declare exactly one default\n' >&2 - exit 2 -fi if ! GATE_PASS_KIND_DEFAULT="$(_gate_assurance_policy_lookup pass-kinds is_default true pass_kind)"; then printf 'Error: gate pass-kind policy must declare exactly one default\n' >&2 exit 2 fi -if [[ "$GATE_MODE_DEFAULT" != "sequential" ]]; then - printf 'Error: gate mode policy default must remain sequential\n' >&2 - exit 2 -fi if [[ "$GATE_PASS_KIND_DEFAULT" != "initial" ]]; then printf 'Error: gate pass-kind policy default must remain initial\n' >&2 exit 2 @@ -3217,29 +3174,32 @@ INITIAL_RESULT_DISPLAY="${INITIAL_RESULT_RESOLVED:-none}" POLICY_REQUIRED_REVIEWERS_DISPLAY="$(jq -r \ '.resolution.required_reviewers | if length == 0 then "none" else join(",") end' \ <<<"$GATE_POLICY_RESOLUTION")" -POLICY_REQUIRED_MODE_DISPLAY="$(jq -r '.resolution.required_mode // "none"' \ +POLICY_MODE_SELECTION_SOURCE="$(jq -r '.resolution.mode_selection_source' \ <<<"$GATE_POLICY_RESOLUTION")" +POLICY_MODE_RECOMMENDATION_OVERRIDDEN="$(jq -r \ + '.resolution.mode_recommendation_overridden' <<<"$GATE_POLICY_RESOLUTION")" POLICY_ESCALATION_SIGNALS_DISPLAY="$(jq -c '[ .matched_signals[] | select(.source != "consumer-policy") - | select((.required_reviewers | length) > 0 or .required_mode != null) + | select((.required_reviewers | length) > 0 or .recommended_mode == "parallel") | { id, required_reviewers, - required_mode + recommended_mode } ]' <<<"$GATE_POLICY_RESOLUTION")" printf -v GATE_ASSURANCE_CONTEXT_BLOCK \ - ' Assurance coordinates (resolved by the gate shell; do not reinterpret):\n tier.requested: %s\n tier.resolved: %s\n tier.evidence_floor: %s\n mode.requested: %s\n mode.resolved: %s\n mode.topology: %s\n mode.synthesis: %s\n pass.requested: %s\n pass.resolved: %s\n pass.scope: %s\n pass.initial_result: %s\n coverage.requested: %s\n coverage.selected: %s\n coverage.skipped: %s\n policy.consumer: %s\n policy.minimum_tier: %s\n policy.required_reviewers: %s\n policy.recommended_mode: %s\n policy.required_mode: %s\n policy.escalation_signals: %s\n policy.scope_fingerprint: %s\n policy.source: %s\n' \ + ' Assurance coordinates (resolved by the gate shell; do not reinterpret):\n tier.requested: %s\n tier.resolved: %s\n tier.evidence_floor: %s\n mode.requested: %s\n mode.resolved: %s\n mode.topology: %s\n mode.synthesis: %s\n mode.selection_source: %s\n mode.recommendation_overridden: %s\n pass.requested: %s\n pass.resolved: %s\n pass.scope: %s\n pass.initial_result: %s\n coverage.requested: %s\n coverage.selected: %s\n coverage.skipped: %s\n policy.consumer: %s\n policy.minimum_tier: %s\n policy.required_reviewers: %s\n policy.recommended_mode: %s\n policy.escalation_signals: %s\n policy.scope_fingerprint: %s\n policy.source: %s\n' \ "$TIER_REQUESTED" "$TIER_RESOLVED" "$TIER_EVIDENCE_FLOOR" \ "$MODE_REQUESTED" "$MODE_RESOLVED" "$MODE_TOPOLOGY" "$MODE_SYNTHESIS" \ + "$POLICY_MODE_SELECTION_SOURCE" "$POLICY_MODE_RECOMMENDATION_OVERRIDDEN" \ "$PASS_KIND_REQUESTED" "$PASS_KIND_RESOLVED" "$PASS_SCOPE" "$INITIAL_RESULT_DISPLAY" \ "$COVERAGE_REQUESTED_DISPLAY" \ "$COVERAGE_SELECTED_DISPLAY" "$COVERAGE_SKIPPED_DISPLAY" \ "$POLICY_CONSUMER" "$(jq -r '.resolution.minimum_tier' <<<"$GATE_POLICY_RESOLUTION")" \ "$POLICY_REQUIRED_REVIEWERS_DISPLAY" \ "$(jq -r '.resolution.recommended_mode' <<<"$GATE_POLICY_RESOLUTION")" \ - "$POLICY_REQUIRED_MODE_DISPLAY" "$POLICY_ESCALATION_SIGNALS_DISPLAY" \ + "$POLICY_ESCALATION_SIGNALS_DISPLAY" \ "$POLICY_SCOPE_FINGERPRINT" \ "$GATE_ASSURANCE_POLICY_SOURCE" @@ -3249,11 +3209,12 @@ say 'pr-gate: tier %s -> %s; mode %s -> %s; pass %s -> %s\n' \ say 'pr-gate: coverage requested=%s selected=%s skipped=%s; policy=%s/%s\n' \ "$COVERAGE_REQUESTED_DISPLAY" "$COVERAGE_SELECTED_DISPLAY" \ "$COVERAGE_SKIPPED_DISPLAY" "$POLICY_CONSUMER" "$GATE_ASSURANCE_POLICY_SOURCE" -say 'pr-gate: policy minimum-tier=%s required-reviewers=%s recommended-mode=%s required-mode=%s scope=%s\n' \ +say 'pr-gate: policy minimum-tier=%s required-reviewers=%s recommended-mode=%s mode-source=%s recommendation-overridden=%s scope=%s\n' \ "$(jq -r '.resolution.minimum_tier' <<<"$GATE_POLICY_RESOLUTION")" \ "$POLICY_REQUIRED_REVIEWERS_DISPLAY" \ "$(jq -r '.resolution.recommended_mode' <<<"$GATE_POLICY_RESOLUTION")" \ - "$POLICY_REQUIRED_MODE_DISPLAY" "$POLICY_SCOPE_FINGERPRINT" + "$POLICY_MODE_SELECTION_SOURCE" "$POLICY_MODE_RECOMMENDATION_OVERRIDDEN" \ + "$POLICY_SCOPE_FINGERPRINT" [[ "${ADJ_COUNT:-0}" -gt 0 ]] && say ' adjacent test files added: %d\n' "$ADJ_COUNT" say 'result will be written to: %s\n\n' "$OUTPUT_FILE" diff --git a/runtime/lib/gate-result-verify.sh b/runtime/lib/gate-result-verify.sh index 3157d6d6..706500b4 100644 --- a/runtime/lib/gate-result-verify.sh +++ b/runtime/lib/gate-result-verify.sh @@ -127,7 +127,8 @@ gate_assurance_verify() { "binary_or_unknown_count","layer_roots"])) and (.policy.resolution | only_keys(["minimum_tier","required_reviewers","recommended_mode", - "required_mode","downgrade_requested","downgrade_allowed"])) and + "mode_selection_source","mode_recommendation_overridden", + "downgrade_requested","downgrade_allowed"])) and (.policy.resolved | only_keys(["tier","mode","reviewers"])) and (.policy.enforcement | only_keys(["status","violations"])) and (.policy.override | @@ -163,25 +164,33 @@ gate_assurance_verify() { $policy.resolution.downgrade_allowed)) and (.policy.resolution.recommended_mode | IN("sequential","parallel")) and - (.policy.resolution.required_mode == null or - (.policy.resolution.required_mode | - IN("sequential","parallel"))) and + (.policy.resolution.mode_selection_source | IN("user","policy")) and + (.policy.resolution.mode_recommendation_overridden | type == "boolean") and + (if .policy.request.mode == "default" + then + .policy.resolution.mode_selection_source == "policy" and + .policy.resolved.mode == .policy.resolution.recommended_mode and + .policy.resolution.mode_recommendation_overridden == false + else + .policy.resolution.mode_selection_source == "user" and + .policy.resolved.mode == .policy.request.mode and + .policy.resolution.mode_recommendation_overridden == + (.policy.request.mode != .policy.resolution.recommended_mode) + end) and (.policy.resolution.downgrade_requested | type == "boolean") and (.policy.resolution.downgrade_allowed | type == "boolean") and (.policy.matched_signals | type == "array" and length > 0) and ([.policy.matched_signals[].id] | strings_unique) and (all(.policy.matched_signals[]; only_keys(["id","source","matches","minimum_tier", - "required_reviewers","recommended_mode","required_mode"]) and + "required_reviewers","recommended_mode"]) and (.id | type == "string" and length > 0) and (.source | IN("consumer-policy","classification","path-regex","brief-value")) and (.matches | strings_unique and length > 0) and (.minimum_tier | IN("express","standard","full")) and (.required_reviewers | strings_unique) and - (.recommended_mode | IN("sequential","parallel")) and - (.required_mode == null or - (.required_mode | IN("sequential","parallel"))))) and + (.recommended_mode | IN("sequential","parallel")))) and .policy.resolved.tier == .coordinates.tier.resolved and .policy.resolved.mode == .coordinates.mode.resolved and same_set(.policy.resolved.reviewers; @@ -190,7 +199,7 @@ gate_assurance_verify() { (.policy.enforcement.violations | type == "array") and (all(.policy.enforcement.violations[]; only_keys(["coordinate","requested","required"]) and - (.coordinate | IN("tier","coverage","mode")))) and + (.coordinate | IN("tier","coverage")))) and (.policy.override.status | IN("not_provided","not_needed","applied","scope_mismatch", "allowance_mismatch")) and diff --git a/skills/pr-gate-review/SKILL.md b/skills/pr-gate-review/SKILL.md index 4c58b785..03c6815d 100644 --- a/skills/pr-gate-review/SKILL.md +++ b/skills/pr-gate-review/SKILL.md @@ -26,7 +26,12 @@ implement → pr-gate → fix NO-GO → push → PR). - `standard` — feature, `architecture_impact: minor`; adds architecture-reviewer with conceptual map. - `full` — large or architectural change, `architecture_impact: major`; defaults to all reviewer dimensions. - Sensitive paths add their security/risk/architecture reviewer without automatically forcing `full`. -- Default = **sequential**, low token cost. `--mode parallel` gives each reviewer an independent session; `--parallel` remains a compatibility spelling. +- With no mode flag, policy auto-selects its recommendation. An explicit + `--mode sequential` (lower token cost) or `--mode parallel` (independent + reviewer sessions) always wins; `--sequential` / `--parallel` remain + compatibility spellings. +- Preserve an explicit user mode on follow-up or targeted gates after NO-GO; + omit the flag only when the user left mode selection to policy. - Trusted `architecture_impact` from `--brief ` is enforced by the canonical policy resolver (`minor` → at least `standard`, `major` → `full`). - Request a tier with `--tier express|standard|full`; a request below the policy floor fails before dispatch. Re-gate a remediation subset with `--targeted r1,r2 --initial-result `. diff --git a/tests/shell/test-core-schemas.sh b/tests/shell/test-core-schemas.sh index 22cfbffb..092d7c2a 100755 --- a/tests/shell/test-core-schemas.sh +++ b/tests/shell/test-core-schemas.sh @@ -803,7 +803,8 @@ _gate_assurance_valid_instance() { minimum_tier:"standard", required_reviewers:["critic","qa-tester","architecture-reviewer"], recommended_mode:"parallel", - required_mode:null, + mode_selection_source:"user", + mode_recommendation_overridden:true, downgrade_requested:false, downgrade_allowed:false }, @@ -814,8 +815,7 @@ _gate_assurance_valid_instance() { matches:["generic:initial"], minimum_tier:"express", required_reviewers:["critic","qa-tester"], - recommended_mode:"sequential", - required_mode:null + recommended_mode:"sequential" }, { id:"medium-change", @@ -823,8 +823,7 @@ _gate_assurance_valid_instance() { matches:["changed-lines:120"], minimum_tier:"standard", required_reviewers:["architecture-reviewer"], - recommended_mode:"parallel", - required_mode:null + recommended_mode:"parallel" } ], resolved:{ @@ -908,8 +907,7 @@ _gate_policy_override_valid_instance() { scope_fingerprint:("a" * 64), allow:{ tier:"express", - omit_reviewers:["security-reviewer"], - mode:null + omit_reviewers:["security-reviewer"] }, reason:"User accepted this exact bounded downgrade.", approver:{ @@ -964,6 +962,21 @@ case_gate_policy_override_extra_key_rejected() { rm -f "$tmpf" } +case_gate_policy_override_mode_key_rejected() { + local name="gate-policy-override: mode is user choice, not a downgrade key" + should_run "$name" || return 0 + local schema_file="$CORE_DIR/schema/gate-policy-override.schema.json" tmpf + tmpf="$(mktemp /tmp/gate-policy-override-mode-key-XXXXXX.json)" + _gate_policy_override_valid_instance | + jq '.allow.mode = "sequential"' > "$tmpf" + if jsonschema -i "$tmpf" "$schema_file" >/dev/null 2>&1; then + fail "$name" "schema accepted mode as a policy downgrade allowance" + else + pass "$name" + fi + rm -f "$tmpf" +} + case_context_pack_v1_still_valid case_context_pack_v2_new_fields_valid case_context_pack_memory_source_domain_valid @@ -977,5 +990,6 @@ case_gate_assurance_non_user_policy_approver_rejected case_gate_policy_override_valid_instance case_gate_policy_override_non_user_approver_rejected case_gate_policy_override_extra_key_rejected +case_gate_policy_override_mode_key_rejected th_summary diff --git a/tests/shell/test-pmctl-gate.sh b/tests/shell/test-pmctl-gate.sh index dfd1f9ea..ffee5600 100755 --- a/tests/shell/test-pmctl-gate.sh +++ b/tests/shell/test-pmctl-gate.sh @@ -448,7 +448,8 @@ _mk_gate_result_v2() { minimum_tier:"express", required_reviewers:["critic","qa-tester"], recommended_mode:"sequential", - required_mode:null, + mode_selection_source:"policy", + mode_recommendation_overridden:false, downgrade_requested:false, downgrade_allowed:false }, @@ -459,8 +460,7 @@ _mk_gate_result_v2() { matches:["generic:initial"], minimum_tier:"express", required_reviewers:["critic","qa-tester"], - recommended_mode:"sequential", - required_mode:null + recommended_mode:"sequential" }, { id:"docs-only", @@ -468,8 +468,7 @@ _mk_gate_result_v2() { matches:["README.md"], minimum_tier:"express", required_reviewers:[], - recommended_mode:"sequential", - required_mode:null + recommended_mode:"sequential" } ], resolved:{ diff --git a/tests/shell/test-pr-gate.sh b/tests/shell/test-pr-gate.sh index b04ca11b..336c8e61 100755 --- a/tests/shell/test-pr-gate.sh +++ b/tests/shell/test-pr-gate.sh @@ -659,9 +659,9 @@ test_gate_policy_sources_control_default_coverage() { } # Behavior: a copied gate without canonical policy files resolves the same -# defaults from its bounded generated snapshot and reports the degraded source. +# coordinates from its bounded generated snapshot and reports the degraded source. # Steps: remove the copied TSV files, run a docs gate, and assert express / -# sequential / initial defaults plus generated-snapshot provenance. +# policy-selected sequential mode / initial pass plus generated-snapshot provenance. test_gate_assurance_policy_snapshot_is_copy_mode_fallback() { local name="gate-assurance-policy-snapshot-is-copy-mode-fallback" should_run "$name" || return 0 @@ -707,7 +707,7 @@ test_dormant_policy_signal_with_unknown_reviewer_fails_before_dispatch() { create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer create_repo "$repo" docs printf '%s\n' \ - $'dormant-signal\tpath-regex\tnever-match-this-fixture\tstandard\tunknown-reviewer\tparallel\tnone' \ + $'dormant-signal\tpath-regex\tnever-match-this-fixture\tstandard\tunknown-reviewer\tparallel' \ >> "$runner/core/policy/gate-policy-signals.tsv" set +e @@ -737,7 +737,7 @@ test_duplicate_policy_signal_id_fails_before_dispatch() { create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer create_repo "$repo" docs printf '%s\n' \ - $'docs-only\tpath-regex\tnever-match-this-fixture\texpress\tnone\tsequential\tnone' \ + $'docs-only\tpath-regex\tnever-match-this-fixture\texpress\tnone\tsequential' \ >> "$runner/core/policy/gate-policy-signals.tsv" set +e @@ -755,9 +755,10 @@ test_duplicate_policy_signal_id_fails_before_dispatch() { } # Behavior: the maintainer initial-pass policy fixes reviewer coverage at all -# five dimensions without rewriting express tier or requiring parallel mode. -test_maintainer_initial_policy_fixes_coverage_only() { - local name="maintainer-initial-policy-fixes-coverage-only" +# five dimensions and supplies parallel as the auto-selected mode while leaving +# an explicit user mode authoritative. +test_maintainer_initial_policy_sets_coverage_and_auto_mode() { + local name="maintainer-initial-policy-sets-coverage-and-auto-mode" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" @@ -777,17 +778,20 @@ test_maintainer_initial_policy_fixes_coverage_only() { return fi assert_file_contains "$name" "$brief" "tier.resolved: express" || return - assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return + assert_file_contains "$name" "$brief" "mode.resolved: parallel" || return + assert_file_contains "$name" "$brief" "mode.selection_source: policy" || return + assert_file_contains "$name" "$brief" "mode.recommendation_overridden: false" || return assert_file_contains "$name" "$brief" "policy.consumer: maintainer" || return assert_file_contains "$name" "$brief" "policy.recommended_mode: parallel" || return - assert_file_contains "$name" "$brief" "policy.required_mode: none" || return assert_file_contains "$name" "$brief" \ "coverage.selected: critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer" || return result_path="$(awk '/^result: / {sub(/^result: /, ""); print; exit}' "$out")" jq -e ' .policy.consumer_policy == "maintainer" and .policy.resolved.tier == "express" and - .policy.resolved.mode == "sequential" and + .policy.resolved.mode == "parallel" and + .policy.resolution.mode_selection_source == "policy" and + .policy.resolution.mode_recommendation_overridden == false and .policy.resolved.reviewers == ["critic","qa-tester","architecture-reviewer","security-reviewer","risk-reviewer"] ' "${result_path}.assurance.json" >/dev/null || { @@ -839,14 +843,17 @@ test_maintainer_targeted_policy_preserves_remediation_scope() { pass "$name" } -# Behavior: an explicit input/execution boundary signal requires parallel -# isolation independently of its standard tier and security coverage. -test_input_execution_signal_requires_parallel_mode() { - local name="input-execution-signal-requires-parallel-mode" +# Behavior: an input/execution boundary signal auto-selects parallel when mode +# is omitted, but an explicit sequential request remains authoritative and is +# recorded as an override of the recommendation rather than a policy downgrade. +test_input_execution_signal_auto_selects_parallel_but_respects_user_mode() { + local name="input-execution-signal-auto-selects-parallel-but-respects-user-mode" should_run "$name" || return 0 local dir="$TMP_ROOT/$name" local home="$dir/home" repo="$dir/repo" runner="$dir/runner" local out="$dir/out" err="$dir/err" brief="$dir/brief.md" result_path + local sequential_brief="$dir/sequential-brief.md" + local sequential_result="$dir/sequential-result.md" mkdir -p "$dir" create_runner "$runner" create_agents "$home" critic qa-tester architecture-reviewer security-reviewer risk-reviewer @@ -866,7 +873,8 @@ test_input_execution_signal_requires_parallel_mode() { fi assert_file_contains "$name" "$brief" "tier.resolved: standard" || return assert_file_contains "$name" "$brief" "mode.resolved: parallel" || return - assert_file_contains "$name" "$brief" "policy.required_mode: parallel" || return + assert_file_contains "$name" "$brief" "mode.selection_source: policy" || return + assert_file_contains "$name" "$brief" "mode.recommendation_overridden: false" || return assert_file_contains "$name" "$brief" "policy.escalation_signals:" || return assert_file_contains "$name" "$brief" '"id":"input-execution-path"' || return assert_not_contains "$name" "$brief" "any diff file matches (" || return @@ -883,15 +891,31 @@ test_input_execution_signal_requires_parallel_mode() { } set +e - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --mode sequential + CODEX_GATE_CAPTURE_BRIEF="$sequential_brief" \ + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --mode sequential --output "$sequential_result" code=$? set -e - if [[ "$code" -ne 3 ]]; then - fail "$name" "explicit sequential mode exited $code, expected policy rejection 3" + if [[ "$code" -ne 0 ]]; then + fail "$name" "explicit sequential mode exited $code, expected user choice to pass" return fi - assert_file_contains "$name" "$err" "requested=sequential required=parallel" || return - assert_not_contains "$name" "$err" "DISPATCH_STUB" || return + assert_file_contains "$name" "$sequential_brief" "mode.requested: sequential" || return + assert_file_contains "$name" "$sequential_brief" "mode.resolved: sequential" || return + assert_file_contains "$name" "$sequential_brief" "mode.selection_source: user" || return + assert_file_contains "$name" "$sequential_brief" \ + "mode.recommendation_overridden: true" || return + jq -e ' + .policy.resolution.recommended_mode == "parallel" and + .policy.resolution.mode_selection_source == "user" and + .policy.resolution.mode_recommendation_overridden == true and + .policy.resolution.downgrade_requested == false and + .policy.resolved.mode == "sequential" and + .policy.enforcement.status == "pass" + ' "${sequential_result}.assurance.json" >/dev/null || { + fail "$name" "assurance did not preserve the explicit sequential choice" + return + } pass "$name" } @@ -1461,7 +1485,7 @@ test_malformed_policy_override_contract_fails_before_dispatch() { kind:"gate_policy_override_v1", schema_version:1, scope_fingerprint:("a" * 64), - allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + allow:{tier:null,omit_reviewers:["qa-tester"]}, approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:test"} }' > "$policy_override" @@ -1514,7 +1538,7 @@ test_scope_bound_policy_override_authorizes_exact_coverage_downgrade() { kind:"gate_policy_override_v1", schema_version:1, scope_fingerprint:$scope, - allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + allow:{tier:null,omit_reviewers:["qa-tester"]}, reason:"User accepts critic-only coverage for this bounded fixture.", approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:test"} }' > "$policy_override" @@ -1565,7 +1589,7 @@ test_policy_override_scope_mismatch_fails_closed() { kind:"gate_policy_override_v1", schema_version:1, scope_fingerprint:("0" * 64), - allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + allow:{tier:null,omit_reviewers:["qa-tester"]}, reason:"Approval belongs to a different change scope.", approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:other"} }' > "$policy_override" @@ -1617,7 +1641,7 @@ test_policy_override_scope_binds_diff_content() { kind:"gate_policy_override_v1", schema_version:1, scope_fingerprint:$scope, - allow:{tier:null,omit_reviewers:["qa-tester"],mode:null}, + allow:{tier:null,omit_reviewers:["qa-tester"]}, reason:"Approval is intentionally bound to the original fixture content.", approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:content"} }' > "$policy_override" @@ -1677,7 +1701,7 @@ test_policy_override_allowance_mismatch_fails_closed() { kind:"gate_policy_override_v1", schema_version:1, scope_fingerprint:$scope, - allow:{tier:null,omit_reviewers:[],mode:null}, + allow:{tier:null,omit_reviewers:[]}, reason:"This allowance intentionally omits none of the missing reviewers.", approver:{kind:"user",identity:"fixture-user",approval_ref:"conversation:mismatch"} }' > "$policy_override" @@ -2464,7 +2488,7 @@ test_adjacent_test_not_duplicated_when_in_diff() { # 1. Create a minimal repo (express tier, docs change) # 2. CODEX_GATE_STUB_VERDICT=block: reviewers write Verdict: block → SHELL_FINAL=NO-GO # CODEX_GATE_STUB_SYNTHESIS_FINAL=GO: synthesis stub writes Final: GO -# 3. Run gate in parallel mode (default) +# 3. Run gate in explicit parallel mode # 4. Assert non-zero exit and "contradicts shell-computed" in stderr test_synthesis_verdict_mismatch_aborts_gate() { local name="synthesis-verdict-mismatch-aborts-gate" @@ -2496,7 +2520,7 @@ test_synthesis_verdict_mismatch_aborts_gate() { # Steps: # 1. Create a repo with a committed service.go (clean tracked file) # 2. CODEX_GATE_STUB_SYNTHESIS_INJECT_FILE=service.go: synthesis stub appends to service.go -# 3. Run gate in parallel mode (default) +# 3. Run gate in explicit parallel mode # 4. Assert non-zero exit, "synthesis session modified" in stderr test_post_synthesis_injection_detected() { local name="post-synthesis-injection-detected" @@ -2531,7 +2555,7 @@ test_post_synthesis_injection_detected() { # Steps: # 1. Create a minimal repo (express tier, docs change) # 2. CODEX_GATE_STUB_SYNTHESIS_MODE=no-output: reviewers write output; synthesis does not -# 3. Run gate in parallel mode (default) +# 3. Run gate in explicit parallel mode # 4. Assert non-zero exit and "synthesis did not produce" in stderr test_synthesis_no_output_aborts_gate() { local name="synthesis-no-output-aborts-gate" @@ -2561,7 +2585,7 @@ test_synthesis_no_output_aborts_gate() { # Steps: # 1. Create a minimal repo (express tier, docs change) # 2. CODEX_GATE_STUB_MODE=no-verdict: reviewer writes output but no Verdict line -# 3. Run gate in parallel mode (default) +# 3. Run gate in explicit parallel mode # 4. Assert non-zero exit and "exactly one valid Verdict line" in stderr test_reviewer_invalid_verdict_aborts_gate() { local name="reviewer-invalid-verdict-aborts-gate" @@ -2647,7 +2671,7 @@ test_reviewer_heading_and_explicit_verdict_must_agree() { # Steps: # 1. Create a minimal repo (express tier, docs change) # 2. CODEX_GATE_STUB_MODE=no-output: all dispatches (reviewers + synthesis) omit output -# 3. Run gate in parallel mode (default) +# 3. Run gate in explicit parallel mode # 4. Assert non-zero exit and "reviewer output missing or empty" in stderr test_reviewer_no_output_aborts_gate() { local name="reviewer-no-output-aborts-gate" @@ -2677,7 +2701,7 @@ test_reviewer_no_output_aborts_gate() { # Steps: # 1. Create a minimal repo (express tier, docs change) # 2. CODEX_GATE_STUB_MODE=no-output: dispatch exits 0 without output -# 3. Run gate in sequential mode (default) +# 3. Run gate in policy-selected sequential mode # 4. Assert non-zero exit and "sequential gate did not produce" in stderr test_sequential_no_output_aborts_gate() { local name="sequential-no-output-aborts-gate" @@ -2707,7 +2731,7 @@ test_sequential_no_output_aborts_gate() { # Steps: # 1. Create a minimal repo (express tier, docs change) # 2. CODEX_GATE_STUB_MODE=no-verdict: dispatch writes output but no Final line -# 3. Run gate in sequential mode (default) +# 3. Run gate in policy-selected sequential mode # 4. Assert non-zero exit and "must contain exactly one Final" in stderr test_sequential_no_final_line_aborts_gate() { local name="sequential-no-final-line-aborts-gate" @@ -2741,7 +2765,7 @@ test_sequential_no_final_line_aborts_gate() { # 1. Create a full-tier repo change (5 reviewers) # 2. CODEX_GATE_STUB_MODE=sequential-partial-timeout: dispatch writes 2 of # 5 reviewer sections then exits 124 (simulated timeout) -# 3. Run gate in sequential mode (default) +# 3. Run gate in policy-selected sequential mode # 4. Assert non-zero exit, stderr reports Timeout + partial completion # counts + the completed/incomplete reviewer names, and the output # file on disk still contains the 2 completed reviewer sections @@ -3395,7 +3419,7 @@ test_parallel_frontmatter_parity_mismatch_aborts_gate() { # Steps: # 1. Create a repo with a committed service.go (clean tracked file) # 2. CODEX_GATE_STUB_INJECT_FILE=service.go: reviewer stub appends to service.go -# 3. Run gate in parallel mode (default) +# 3. Run gate in explicit parallel mode # 4. Assert non-zero exit, "prompt injection" in stderr, and no "[synthesis]" in stdout test_prompt_injection_detected() { local name="prompt-injection-detected" @@ -4262,9 +4286,10 @@ test_full_tier_line_count() { } # Behavior: a bounded auth-path change adds the security reviewer without -# conflating sensitive coverage with full-tier or parallel topology. -# Steps: create a tiny auth diff and assert express intent plus the security -# dimension, while parallel remains only a recommendation. +# conflating sensitive coverage with full-tier intent; because mode is omitted, +# the parallel recommendation becomes the auto-selected topology. +# Steps: create a tiny auth diff and assert express intent, the security +# dimension, and policy-selected parallel mode. test_sensitive_file_adds_security_without_forcing_full() { local name="sensitive-file-adds-security" should_run "$name" || return 0 @@ -4288,11 +4313,11 @@ test_sensitive_file_adds_security_without_forcing_full() { assert_file_contains "$name" "$brief" "Reviewers: critic,qa-tester,security-reviewer" || return assert_file_contains "$name" "$brief" "policy.required_reviewers: critic,qa-tester,security-reviewer" || return assert_file_contains "$name" "$brief" "policy.recommended_mode: parallel" || return - assert_file_contains "$name" "$brief" "policy.required_mode: none" || return assert_file_contains "$name" "$brief" \ 'policy.escalation_signals: [{"id":"security-sensitive-path"' || return assert_not_contains "$name" "$brief" "any diff file matches (" || return - assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return + assert_file_contains "$name" "$brief" "mode.resolved: parallel" || return + assert_file_contains "$name" "$brief" "mode.selection_source: policy" || return pass "$name" } @@ -4325,7 +4350,8 @@ test_plural_signal_paths_add_required_dimensions() { assert_file_contains "$name" "$brief" "tier.resolved: standard" || return assert_file_contains "$name" "$brief" \ "coverage.selected: critic,qa-tester,architecture-reviewer,security-reviewer,risk-reviewer" || return - assert_file_contains "$name" "$brief" "policy.required_mode: none" || return + assert_file_contains "$name" "$brief" "mode.resolved: parallel" || return + assert_file_contains "$name" "$brief" "mode.selection_source: policy" || return result_path="$(awk -F'result: ' '/^result: /{path=$2} END{print path}' "$out")" jq -e ' any(.policy.matched_signals[]; @@ -4557,9 +4583,9 @@ run_test test_gate_policy_sources_control_default_coverage run_test test_gate_assurance_policy_snapshot_is_copy_mode_fallback run_test test_dormant_policy_signal_with_unknown_reviewer_fails_before_dispatch run_test test_duplicate_policy_signal_id_fails_before_dispatch -run_test test_maintainer_initial_policy_fixes_coverage_only +run_test test_maintainer_initial_policy_sets_coverage_and_auto_mode run_test test_maintainer_targeted_policy_preserves_remediation_scope -run_test test_input_execution_signal_requires_parallel_mode +run_test test_input_execution_signal_auto_selects_parallel_but_respects_user_mode run_test test_tier_detection run_test test_pr_gate_does_not_mutate_gitignore run_test test_artifact_filter_drops_gate_artifacts @@ -5343,7 +5369,7 @@ test_targeted_pass_references_initial_result() { } # Behavior: a targeted pass with auto mode and no input brief resolves all -# policy coordinates before constructing the default sequential reviewer brief. +# policy coordinates before constructing the policy-selected sequential reviewer brief. # Steps: run a critic-only targeted gate without --mode or --brief and assert # successful dispatch plus initialized sequential coordinates. This covers the # real runtime path that previously aborted on unbound MODE_RESOLVED/BRIEF_FILE. @@ -5371,6 +5397,8 @@ test_targeted_auto_mode_initializes_brief_coordinates() { fi assert_file_contains "$name" "$brief" "mode.requested: default" || return assert_file_contains "$name" "$brief" "mode.resolved: sequential" || return + assert_file_contains "$name" "$brief" "mode.selection_source: policy" || return + assert_file_contains "$name" "$brief" "mode.recommendation_overridden: false" || return assert_file_contains "$name" "$brief" "pass.resolved: targeted" || return assert_not_contains "$name" "$err" "unbound variable" || return pass "$name" @@ -5573,6 +5601,8 @@ test_equivalent_mode_spellings_are_accepted() { fi assert_file_contains "$name" "$brief" "mode.requested: parallel" || return assert_file_contains "$name" "$brief" "mode.resolved: parallel" || return + assert_file_contains "$name" "$brief" "mode.selection_source: user" || return + assert_file_contains "$name" "$brief" "mode.recommendation_overridden: true" || return assert_file_contains "$name" "$out" "launched critic" || return pass "$name" } @@ -5643,6 +5673,8 @@ test_seq_brief_ascii_separator() { fail "$name" "exit $code, expected 0" return fi + assert_file_contains "$name" "$brief" "mode.selection_source: user" || return + assert_file_contains "$name" "$brief" "mode.recommendation_overridden: false" || return # Template heading must use ASCII -- not em dash assert_file_contains "$name" "$brief" "PR-Gate Result --" || return # Reviewer heading format must use ASCII -- not em dash @@ -5886,7 +5918,9 @@ test_dirty_preflight_fails_on_committed_plus_dirty() { # Steps: create a repo with committed feature-branch changes, add an # untracked file (dirtysrc.go), run the gate against main with # --allow-dirty, and assert exit 0, dispatch succeeds, the brief lists -# dirtysrc.go, and stderr notes --allow-dirty was set. +# dirtysrc.go, and stderr notes --allow-dirty was set. The test explicitly +# selects sequential so CODEX_GATE_CAPTURE_BRIEF receives the combined reviewer +# brief rather than a policy-selected parallel synthesis brief. test_dirty_preflight_allow_dirty_includes_worktree() { local name="dirty-preflight-allow-dirty-includes-worktree" should_run "$name" || return 0 @@ -5900,7 +5934,8 @@ test_dirty_preflight_allow_dirty_includes_worktree() { (cd "$repo" && printf 'x\n' > dirtysrc.go) set +e - CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --allow-dirty + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --allow-dirty --mode sequential local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -5922,8 +5957,8 @@ test_dirty_preflight_allow_dirty_includes_worktree() { # Steps: commit tracked_base.go on main, branch to feature, commit app.go # (so BASE...HEAD covers app.go but NOT tracked_base.go), modify # tracked_base.go in the worktree without committing, run the gate against -# main with --allow-dirty, and assert exit 0 and the brief includes -# tracked_base.go. +# main with --allow-dirty and explicit sequential mode, then assert exit 0 and +# the combined reviewer brief includes tracked_base.go. test_allow_dirty_includes_uncommitted_tracked() { local name="allow-dirty-includes-uncommitted-tracked" should_run "$name" || return 0 @@ -5952,7 +5987,8 @@ test_allow_dirty_includes_uncommitted_tracked() { ) set +e - CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --allow-dirty + CODEX_GATE_CAPTURE_BRIEF="$brief" run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --allow-dirty --mode sequential local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -7372,7 +7408,8 @@ run_test test_gate_run_dir_parallel_failure_leaves_no_repo_artifacts # committed change. # 2. Check out main (NOT feature) so the working tree is not on # the reviewed ref. -# 3. Run the gate with --base main --head feature. +# 3. Run the gate with --base main --head feature and explicit sequential mode +# so the captured brief is the combined reviewer brief. # 4. Assert exit 0, the brief records "Head: feature", and the # feature-only file is in scope. test_head_override_diffs_fixed_ref() { @@ -7391,7 +7428,8 @@ test_head_override_diffs_fixed_ref() { set +e CODEX_GATE_CAPTURE_BRIEF="$brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --head feature + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --head feature --mode sequential local code=$? set -e if [[ "$code" -ne 0 ]]; then @@ -7475,7 +7513,8 @@ test_head_override_rejects_allow_dirty() { # Steps: # 1. Build a repo with main + a feature branch carrying a committed change (app.go). # 2. Check out main and commit an independent main-only file the feature branch never sees. -# 3. Run the gate with --base main --head feature (base and head now diverged both ways). +# 3. Run the gate with --base main --head feature and explicit sequential mode +# (base and head now diverged both ways). # 4. Assert exit 0, app.go is in scope, and main-only.txt is NOT in scope -- # a two-dot diff would additionally report main-only.txt as removed. test_head_override_merge_base_semantics() { @@ -7498,7 +7537,8 @@ test_head_override_merge_base_semantics() { set +e CODEX_GATE_CAPTURE_BRIEF="$brief" \ - run_gate "$home" "$runner" "$repo" "$out" "$err" --base main --head feature + run_gate "$home" "$runner" "$repo" "$out" "$err" \ + --base main --head feature --mode sequential local code=$? set -e if [[ "$code" -ne 0 ]]; then From 024c0297abb88dfff20ac8c203c868cb7694f217 Mon Sep 17 00:00:00 2001 From: screenleon Date: Tue, 28 Jul 2026 12:12:48 +0900 Subject: [PATCH 3/4] docs(backlog): clarify targeted gate coordinates --- BACKLOG.md | 69 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 69 insertions(+) diff --git a/BACKLOG.md b/BACKLOG.md index f72ff865..129cba32 100644 --- a/BACKLOG.md +++ b/BACKLOG.md @@ -46,6 +46,7 @@ CC-001/CC-002 were consumed by PR #24 fix bundle inline, with no standalone entr | CC-524 | 🔵 active | `pmctl artifacts show` 顯示 canonical absolute run root 並提供穩定 machine-readable locator | ux/ops | 2026-07-27 | feedback:2026-07-27 | P2 | hygiene | | CC-525 | 🔵 active | copy-mode verifier fallback 的 generated provenance 必須指向實際 generator,並由 parity ratchet 防止再次漂移 | ops/test | 2026-07-28 | feedback:2026-07-28 | P3 | hygiene | | CC-526 | 🔵 active | reviewer override file 的 symlink trust-boundary hardening 與相容性契約 | security/gate | 2026-07-28 | feedback:2026-07-28 | P2 | hygiene | +| CC-527 | 🔵 active | targeted gate CLI 拆分 pass、reviewer coverage 與 tier,避免 full targeted 語意重疊 | ux/gate | 2026-07-28 | feedback:2026-07-28 | P2 | design | | CC-465 | 🔵 active | memory/context 關鍵詞管線 CJK 支援:抽出共用零依賴斷詞 lib,取代三處各自 ASCII-only 抽詞;工作序列起點(465→467→468→466)(2026-07-07 記憶系統深入分析) | memory | 2026-07-07 | feedback:2026-07-07 | P2 | retrieval | | CC-466 | ⏸ deferred | 記憶卡片生命週期閉環:expires_at 執行 + 關窗式 supersede + usage sidecar 休眠偵測 + doctor→distill 接線;僅在 CC-467 證明 stale/dormant card 已形成實際問題時啟動 | memory | 2026-07-07 | feedback:2026-07-07 | P2 | retrieval | | CC-467 | 🔵 active | `pmctl memory stats`:注入效益可視化(唯讀聚合器)——注入 bytes/卡片命中分佈/從未命中卡/episode 填寫率,回答「記憶有跟沒有差在哪」;排在 CC-466 之前(2026-07-07;業界僅離線 recall 評測,無 per-injection 遙測) | DX/memory | 2026-07-07 | — | P2 | retrieval | @@ -2421,6 +2422,74 @@ policy downgrade;不宣稱防禦具有同一 OS 帳號寫入權限的攻擊者 --- +## CC-527 — targeted gate CLI coordinate separation 與 truthful labeling 🔵 active + +**Framing**: 本票只收斂既有 gate assurance coordinates 的 CLI 表達與 human +label,不新增 gate kind、review workflow、tier 或 reviewer。[[CC-512]] 已確立 +tier、pass kind 與 reviewer coverage 是正交座標;本票讓 public CLI 也能直接表達 +這三軸,而不是由一個 `--targeted ` 同時承擔 pass kind 與 coverage。 +既有 shorthand 必須保留 bounded compatibility,不能藉語意清理限制 generic gate +使用者選擇 reviewer 或 execution mode。 + +**Problem**: `--tier full --targeted qa-tester --initial-result ` 在 machine +contract 中可解析為 `tier=full`、`pass=targeted`、`coverage=[qa-tester]`,但 human +語意容易把 `full` 誤讀成完整五 reviewer comprehensive gate。`--targeted` 目前又 +同時選擇 remediation-delta pass 與 reviewer coverage,而 tier table 仍提供 default +reviewers;即使 resolver 有確定 precedence,CLI 表面仍讓 rigor、pass scope 與 +coverage 看似互相覆蓋。這可能導致 maintainer recipe 錯稱「full gate」、重啟不必要 +的 comprehensive discovery,或誤以為 targeted qa 已取得 full reviewer coverage。 + +**Requirement**: + +1. 定義 canonical explicit form,設計目標為 + `--pass targeted --reviewers qa-tester --initial-result `;pass kind、 + coverage 與 initial-result 必須各自驗證。既有 `--targeted ` 保留為 + compatibility shorthand,且必須機械展開為完全相同的 coordinates,不能形成 + 第二條 resolver path。 +2. Targeted tier resolution 必須有單一、可解釋的 basis:未明確指定 tier 時,優先 + 繼承 subject-applicable initial result 的 resolved tier;若 initial artifact + 無法提供可信 tier,必須使用 canonical policy resolution 或 fail closed,不得從 + targeted reviewer 數量反推 tier。這項 inheritance/applicability 接線依賴 + [[CC-515]],不可用未驗證 frontmatter prose 代替。 +3. 使用者若有獨立 rigor 理由仍可明確請求 tier,但 explicit tier 不得擴張、替代或 + 暗示 targeted coverage。CLI progress、brief、result 與 assurance 必須並列輸出 + `tier=`、`pass=targeted`、`coverage=[...]` 及各自 selection basis; + `tier=full` 不得被 human 文案命名為 `full gate` 或 comprehensive review。 +4. Canonical 與 compatibility spellings 混用時,同值可接受、不同 pass/coverage + 請求 fail closed;`targeted` 缺 `--initial-result`、initial pass 帶 initial result、 + duplicate/empty reviewer、tier inheritance 不可驗證都必須在 dispatch 前給出 + actionable error。 +5. 更新 `/pr-gate`、maintainer `/ship` 與 review model:follow-up confirmation + 必須描述為 targeted remediation pass,列出 tier 與 selected reviewers,不得以 + 「full」代稱 coverage。[[CC-517]] 的 conditional targeted confirmation 使用 + canonical explicit form,但本票不實作 remediation closure。 +6. Artifact/verifier 必須能機械回答 pass kind、tier basis、coverage basis、initial + result reference 與 shorthand provenance;copy-mode、repo-layout、 + sequential/parallel 必須 meaning-parity。若既有 `gate_assurance_v2` 已可完整 + 表達,優先重用而不新增 schema family。 +7. Deterministic fixtures 覆蓋 canonical targeted、legacy shorthand parity、 + explicit full-tier + QA-only coverage 的 truthful labeling、tier inheritance、 + stale/legacy initial result、conflicting spellings、缺 initial result,以及 + consumer 不得把 targeted artifact 當 comprehensive full-coverage evidence。 + +**Done-when**: 操作者看到任一 targeted command/result 都能分辨「審查 rigor、 +remediation pass scope、實際 reviewer coverage」;canonical form 不再由 +`--targeted` 一個參數承擔兩個座標,legacy shorthand 仍相容,且任何 +`tier=full + pass=targeted + coverage=[qa-tester]` artifact 都不會被 UI、文件或 +consumer 誤稱為 full/comprehensive gate。 + +**Non-goals**: 不移除 targeted confirmation;不強制 targeted 使用特定 tier、 +reviewer 或 mode;不新增 workflow engine/FSM/gate kind;不把本票擴張成 +[[CC-517]] remediation ledger、[[CC-515]] freshness verifier或新的 tier taxonomy。 + +**Dependencies**: CLI coordinate 分離延伸 [[CC-512]];可信 initial-tier +inheritance 依賴 [[CC-515]];maintainer consumer 接線由 [[CC-517]] 使用。P2, +可先交付 syntax/parity,再於 applicability verifier 完成後接 inheritance。 + +**Cross-link**: [[CC-512]]、[[CC-513]]、[[CC-514]]、[[CC-515]]、[[CC-517]]。 + +--- + ## CC-508 — 所有間接 dispatch 的 parent-operation control plane ✅ 2026-07-25 **Problem**: `pmctl gate run`、`pmctl ship --parallel`/adapter 路徑、`pmctl task dispatch` 與任何未來 producer 都可能以一個 parent operation 間接啟動一或多個 detached dispatch;但產品控制面主要只暴露個別 `pmctl dispatch cancel `。parent ID 與其子 run 沒有強制、可查的 ownership relation,也沒有一致的 producer-level cancel surface。當任一 producer 卡住、選錯 executor 或需中止時,操作者無法透過 pmctl 取消整個 operation;直接對 supervisor PID 操作會繞過 run state、sentinel 與 cancel-vs-complete 單一終態契約,並可能留下無法判定的 stale operation。 From 199d42fbd7f78a45d9481a82da3ef657aeb0f9df Mon Sep 17 00:00:00 2001 From: screenleon Date: Tue, 28 Jul 2026 13:52:23 +0900 Subject: [PATCH 4/4] fix(gate): honor readiness timeout under load --- runtime/lib/pmctl-gate.sh | 47 ++++++++--------- tests/shell/test-gate-lifecycle.sh | 85 ++++++++++++++++++++++++++++-- tests/shell/test-pmctl-gate.sh | 28 ++++++++-- 3 files changed, 128 insertions(+), 32 deletions(-) diff --git a/runtime/lib/pmctl-gate.sh b/runtime/lib/pmctl-gate.sh index fd29d51c..73e7c30e 100644 --- a/runtime/lib/pmctl-gate.sh +++ b/runtime/lib/pmctl-gate.sh @@ -323,6 +323,12 @@ pmctl_gate_run() { pmctl_gate_run_detached() { local repo_root="$1" effective_cd="$2"; shift 2 local -a forward=("$@") + local _ready_timeout="${PM_GATE_READY_TIMEOUT:-5}" + + if ! [[ "$_ready_timeout" =~ ^[1-9][0-9]*$ ]]; then + printf 'pmctl gate run: invalid PM_GATE_READY_TIMEOUT %q (expected positive seconds)\n' "$_ready_timeout" >&2 + return 2 + fi if [[ "$(type -t detached_launch_generate_nonce 2>/dev/null)" != function ]]; then local _dl_lib="$repo_root/runtime/lib/detached-launch.sh" @@ -411,11 +417,7 @@ pmctl_gate_run_detached() { # the supervisor has exited in the intervening scheduling window. detached_launch_capture_identity "$_sup_pid" "$_isolated" >"$supervisor_identity" 2>/dev/null || true - local _ready_timeout="${PM_GATE_READY_TIMEOUT:-5}" _ready_start _ready_state _ready_pid _ready_starttime _ready_rc _pre_ready_evidence_polls=0 - if ! [[ "$_ready_timeout" =~ ^[1-9][0-9]*$ ]]; then - printf 'pmctl gate run: invalid PM_GATE_READY_TIMEOUT %q (expected positive seconds)\n' "$_ready_timeout" >&2 - return 2 - fi + local _ready_start _ready_state _ready_pid _ready_starttime _ready_rc _ready_start=$SECONDS while true; do if [[ -f "$ready_sentinel" ]]; then @@ -459,24 +461,21 @@ pmctl_gate_run_detached() { _ready_rc=0 else # The child can publish ready + terminal after the parent's first - # identity snapshot, then exit before this liveness check. The ready - # record is the authoritative evidence, so give its atomic rename a - # bounded observation window for both a missing and a now-dead identity. - _pre_ready_evidence_polls=$((_pre_ready_evidence_polls + 1)) - if (( _pre_ready_evidence_polls <= 5 )); then - sleep 0.05 - continue - fi + # identity snapshot, then exit before this liveness check. A loaded host + # can also delay the separately scheduled supervisor after the launch + # PID stops being observable. The configured readiness timeout owns that + # whole evidence window; a fixed sub-second poll count would turn normal + # scheduler latency into a false early-death result. _ready_rc=1 fi - if [[ "$_ready_rc" -ne 0 ]]; then - printf 'pmctl gate run: detached supervisor exited before readiness for %s; inspect %s and retry with --lifecycle foreground (sandbox parent-death may prevent detached runs)\n' \ - "$gate_id" "$supervisor_log" >&2 - return 2 - fi if (( SECONDS - _ready_start >= _ready_timeout )); then - printf 'pmctl gate run: detached supervisor did not become ready within %ss for %s; inspect %s and retry with --lifecycle foreground\n' \ - "$_ready_timeout" "$gate_id" "$supervisor_log" >&2 + if [[ "$_ready_rc" -ne 0 ]]; then + printf 'pmctl gate run: detached supervisor exited before readiness for %s; inspect %s and retry with --lifecycle foreground (sandbox parent-death may prevent detached runs)\n' \ + "$gate_id" "$supervisor_log" >&2 + else + printf 'pmctl gate run: detached supervisor did not become ready within %ss for %s; inspect %s and retry with --lifecycle foreground\n' \ + "$_ready_timeout" "$gate_id" "$supervisor_log" >&2 + fi return 2 fi sleep 0.05 @@ -584,15 +583,16 @@ pmctl_gate_wait() { return 2 fi - local _sentinel + local _sentinel _ready_sentinel _sentinel="$(detached_launch_sentinel_path "pm-gate" "$gate_id" "$_key_nonce")" + _ready_sentinel="$(detached_launch_sentinel_path "pm-gate-ready" "$gate_id" "$_key_nonce")" if detached_launch_wait_for_sentinel "$_sentinel" "$timeout" "${PM_GATE_WAIT_POLL_INTERVAL:-2}"; then local _state _exit _result _state="$(grep -m1 '^final_state=' "$_sentinel" 2>/dev/null | cut -d= -f2-)" || true _exit="$(grep -m1 '^exit_code=' "$_sentinel" 2>/dev/null | cut -d= -f2-)" || true _result="$(grep -m1 '^result_file=' "$_sentinel" 2>/dev/null | cut -d= -f2-)" || true _operation="$(grep -m1 '^parent_operation=' "$_sentinel" 2>/dev/null | cut -d= -f2-)" || true - rm -f "$_sentinel" "$_key_file" 2>/dev/null || true + rm -f "$_sentinel" "$_ready_sentinel" "$_key_file" 2>/dev/null || true [[ "$_exit" =~ ^-?[0-9]+$ ]] || _exit="1" printf 'gate: %s state: %s exit: %s\n' "$gate_id" "${_state:-unknown}" "$_exit" if [[ -n "$_result" ]]; then @@ -670,8 +670,7 @@ pmctl_gate_wait() { # it the supervisor never became ready; with it but a dead recorded identity, # it died after launch. Keep those failures distinct from a live gate that # merely exceeded the caller's wait budget. - local _ready_sentinel _wait_run_dir _wait_identity _wait_liveness - _ready_sentinel="$(detached_launch_sentinel_path "pm-gate-ready" "$gate_id" "$_key_nonce")" + local _wait_run_dir _wait_identity _wait_liveness if [[ ! -f "$_ready_sentinel" ]]; then printf 'pmctl gate wait: indeterminate: %s never reached supervisor readiness; detached launch did not start a waitable gate (exit=3)\n' "$gate_id" >&2 return 3 diff --git a/tests/shell/test-gate-lifecycle.sh b/tests/shell/test-gate-lifecycle.sh index bc99e3f2..06274705 100755 --- a/tests/shell/test-gate-lifecycle.sh +++ b/tests/shell/test-gate-lifecycle.sh @@ -31,6 +31,27 @@ export XDG_RUNTIME_DIR="$_TEST_XDG_RUNTIME_DIR" export PM_GATE_WAIT_POLL_INTERVAL="${PM_GATE_WAIT_POLL_INTERVAL:-0.1}" _WAIT_OK="${PM_GATE_TEST_WAIT_TIMEOUT:-30}" +# Launch-only cases intentionally do not call `gate wait`, so consume their +# suite-owned nonce paths before the harness removes its isolated key dir. +# This keeps concurrent/full test runs from accumulating global /tmp sentinels. +_cleanup_unconsumed_gate_test_sentinels() { + local key_dir="$XDG_RUNTIME_DIR/pm-gate-dispatch" + local key_file gate_id nonce ready_sentinel terminal_sentinel + [[ -d "$key_dir" ]] || return 0 + for key_file in "$key_dir"/gate-*; do + [[ -f "$key_file" ]] || continue + gate_id="${key_file##*/}" + nonce="$(cat "$key_file" 2>/dev/null || true)" + [[ -n "$nonce" ]] || continue + ready_sentinel="$(detached_launch_sentinel_path "pm-gate-ready" "$gate_id" "$nonce")" + terminal_sentinel="$(detached_launch_sentinel_path "pm-gate" "$gate_id" "$nonce")" + if [[ -e "$ready_sentinel" && ! -e "$terminal_sentinel" ]]; then + detached_launch_wait_for_sentinel "$terminal_sentinel" 1 0.01 >/dev/null 2>&1 || true + fi + rm -f "$ready_sentinel" "$terminal_sentinel" "$key_file" 2>/dev/null || true + done +} + # Build a fixture repo root with the real pmctl-gate.sh + gate-supervisor.sh # + their dependencies (state-paths.sh/portable.sh for sw_project_run_dir, # gate-result-verify.sh for the post-wait structural check) so @@ -225,7 +246,7 @@ case_detached_launch_fails_loud_on_early_supervisor_death() { _run_gate_wrapper "$fixture" "$run_wrapper" local out code - set +e; out="$("$run_wrapper" --cd "$work" --lifecycle detached 2>&1)"; code=$?; set -e + set +e; out="$(PM_GATE_READY_TIMEOUT=1 "$run_wrapper" --cd "$work" --lifecycle detached 2>&1)"; code=$?; set -e if [[ "$code" -eq 2 ]] && [[ "$out" == *"exited before readiness"* ]] && [[ "$out" == *"--lifecycle foreground"* ]]; then pass "$name" else @@ -308,6 +329,53 @@ WRAPPER if [[ "$code" -eq 0 && "$out" == *"pmctl gate wait gate-"* && "$out" == *$'\ngate-'* ]]; then pass "$name"; else fail "$name" "code=$code out=$out"; fi } +# ---- 1g: the configured readiness timeout owns the full evidence window ----- +case_detached_launch_honors_timeout_during_extended_liveness_gap() { + # A loaded host can delay the supervisor after the launcher's PID identity + # probe stops succeeding. Drive that gap deterministically for more than the + # old five-poll grace, then release readiness while still inside the declared + # timeout. The launch must wait for authenticated evidence instead of + # reporting a false early death. + local name="gate-lifecycle/detached launch honors timeout during extended liveness gap" + should_run "$name" || return 0 + local fixture="$tmp_root/c1g/fixture" work="$tmp_root/c1g/work" + local release="$tmp_root/c1g/release" polls="$tmp_root/c1g/polls" + mkdir -p "$work"; _mk_fixture_repo "$fixture"; _mk_fake_gate "$fixture" 0 + # shellcheck disable=SC2016 # sed must preserve the supervisor's env expansion. + sed -i 's/_write_ready || _die "failed to publish supervisor readiness evidence"/while [[ ! -f "${PM_TEST_READY_RELEASE:-}" ]]; do sleep 0.01; done\n_write_ready || _die "failed to publish supervisor readiness evidence"/' "$fixture/runtime/bin/gate-supervisor.sh" + + local run_wrapper="$tmp_root/c1g/run" out code + cat > "$run_wrapper" <> "$polls" + if (( _test_parent_sleep_count == 6 )); then + : > "$release" + fi + command sleep "\$@" +} +export PM_TEST_READY_RELEASE="$release" +pmctl_gate_run "$fixture" "\$@" +WRAPPER + chmod +x "$run_wrapper" + set +e; out="$("$run_wrapper" --cd "$work" --lifecycle detached 2>&1)"; code=$?; set -e + # Always release the fixture child so a failing implementation leaves no + # blocked detached process behind when the test harness removes its tmp dir. + : > "$release" + if [[ "$code" -eq 0 && "$out" == *"pmctl gate wait gate-"* && "$out" == *$'\ngate-'* ]] \ + && [[ "$(wc -l < "$polls" 2>/dev/null || printf '0')" -ge 6 ]]; then + pass "$name" + else + fail "$name" "code=$code polls=$(wc -l < "$polls" 2>/dev/null || printf '0') out=$out" + fi +} + # ---- 2: gate wait resolves GO (exit 0) after supervisor completes ------------ case_wait_resolves_go() { local name="gate-lifecycle/gate wait resolves GO after supervisor completes" @@ -440,19 +508,24 @@ case_wait_indeterminate_on_consumed_sentinel() { _run_gate_wrapper "$fixture" "$run_wrapper" _wait_wrapper "$fixture" "$wait_wrapper" - local gate_id + local gate_id key_file nonce ready_sentinel terminal_sentinel gate_id="$("$run_wrapper" --cd "$work" --lifecycle detached)" + key_file="$XDG_RUNTIME_DIR/pm-gate-dispatch/$gate_id" + nonce="$(cat "$key_file")" + ready_sentinel="$(detached_launch_sentinel_path "pm-gate-ready" "$gate_id" "$nonce")" + terminal_sentinel="$(detached_launch_sentinel_path "pm-gate" "$gate_id" "$nonce")" - # First wait consumes the sentinel + key file (one-shot). + # First wait consumes terminal + readiness evidence and the key (one-shot). "$wait_wrapper" "$gate_id" --cd "$work" --timeout "$_WAIT_OK" >/dev/null 2>&1 || true local out code set +e; out="$("$wait_wrapper" "$gate_id" --cd "$work" --timeout 2 2>&1)"; code=$?; set -e - if [[ "$code" -eq 3 ]] && [[ "$out" == *"indeterminate"* ]]; then + if [[ "$code" -eq 3 ]] && [[ "$out" == *"indeterminate"* ]] \ + && [[ ! -e "$ready_sentinel" && ! -e "$terminal_sentinel" && ! -e "$key_file" ]]; then pass "$name" else - fail "$name" "code=$code out=$out" + fail "$name" "code=$code ready_exists=$([[ -e "$ready_sentinel" ]] && printf yes || printf no) terminal_exists=$([[ -e "$terminal_sentinel" ]] && printf yes || printf no) key_exists=$([[ -e "$key_file" ]] && printf yes || printf no) out=$out" fi } @@ -760,6 +833,7 @@ case_detached_launch_rejects_invalid_ready_identity case_detached_launch_rejects_invalid_ready_timeout case_detached_launch_accepts_terminal_evidence_after_capture_race case_detached_launch_accepts_terminal_evidence_after_liveness_race +case_detached_launch_honors_timeout_during_extended_liveness_gap case_wait_resolves_go case_wait_reloads_verifier_over_incomplete_export case_wait_resolves_nogo @@ -775,4 +849,5 @@ case_wait_fails_on_corrupt_result case_wait_fails_on_cd_partition_mismatch case_wait_usage_errors +_cleanup_unconsumed_gate_test_sentinels th_summary diff --git a/tests/shell/test-pmctl-gate.sh b/tests/shell/test-pmctl-gate.sh index ffee5600..b8632fea 100755 --- a/tests/shell/test-pmctl-gate.sh +++ b/tests/shell/test-pmctl-gate.sh @@ -29,6 +29,8 @@ git -C "$_GATE_VERIFY_REPO" init -q # shellcheck source=runtime/lib/state-paths.sh . "$REPO_ROOT/runtime/lib/state-paths.sh" +# shellcheck source=runtime/lib/detached-launch.sh +. "$REPO_ROOT/runtime/lib/detached-launch.sh" # --------------------------------------------------------------------------- # Helpers @@ -1061,13 +1063,33 @@ case_default_lifecycle_is_detached() { # stdout must stay a single bare gate_id line (callers capture it with # command substitution); the copy-paste wait hint goes to stderr only. - local err_out; err_out="$(cat "$err_file" 2>/dev/null)" + local err_out key_file nonce ready_sentinel terminal_sentinel cleanup_ok=true + err_out="$(cat "$err_file" 2>/dev/null)" + if [[ "$out" =~ ^gate-[0-9]{8}-[0-9]{6}-[A-Za-z0-9]{6,}$ ]]; then + key_file="$XDG_RUNTIME_DIR/pm-gate-dispatch/$out" + nonce="$(cat "$key_file" 2>/dev/null || true)" + if [[ -n "$nonce" ]]; then + ready_sentinel="$(detached_launch_sentinel_path "pm-gate-ready" "$out" "$nonce")" + terminal_sentinel="$(detached_launch_sentinel_path "pm-gate" "$out" "$nonce")" + PM_DISPATCH_STATE_ROOT="$PM_DISPATCH_STATE_ROOT" XDG_RUNTIME_DIR="$XDG_RUNTIME_DIR" \ + PM_GATE_WAIT_POLL_INTERVAL=0.01 \ + bash -c '. "$1/runtime/lib/pmctl-gate.sh"; pmctl_gate_wait "$1" "$2" --cd "$3" --timeout 5' \ + _ "$fixture" "$out" "$work" >/dev/null 2>&1 || true + else + cleanup_ok=false + fi + if [[ "$cleanup_ok" == true ]] \ + && [[ -e "$key_file" || -e "$ready_sentinel" || -e "$terminal_sentinel" ]]; then + cleanup_ok=false + fi + fi if [[ "$code" -eq 0 ]] \ && [[ "$out" =~ ^gate-[0-9]{8}-[0-9]{6}-[A-Za-z0-9]{6,}$ ]] \ - && [[ "$err_out" == *"pmctl gate wait $out --cd"* ]]; then + && [[ "$err_out" == *"pmctl gate wait $out --cd"* ]] \ + && [[ "$cleanup_ok" == true ]]; then pass "$name" else - fail "$name" "code=$code out=$out err=$err_out (expected bare gate_id on stdout + wait hint on stderr)" + fail "$name" "code=$code cleanup_ok=$cleanup_ok out=$out err=$err_out (expected bare gate_id on stdout + wait hint on stderr)" fi }