From 6d1b30211e139cff2dd789847612800e4bcd7421 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 17:05:53 +0100 Subject: [PATCH 01/15] docs(ci): plan change-impact backend verification --- .commitrail/INDEX.md | 1 + .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 80 +++++++++++ .../initiatives/WS-CI-006/WS-CI-006-01.md | 126 ++++++++++++++++++ 3 files changed, 207 insertions(+) create mode 100644 .commitrail/initiatives/WS-CI-006/OVERVIEW.md create mode 100644 .commitrail/initiatives/WS-CI-006/WS-CI-006-01.md diff --git a/.commitrail/INDEX.md b/.commitrail/INDEX.md index 10e56d87b..c1240d0ef 100644 --- a/.commitrail/INDEX.md +++ b/.commitrail/INDEX.md @@ -17,6 +17,7 @@ for current product capability. | [WS-REV-001](initiatives/WS-REV-001/OVERVIEW.md) | Planned | Shared acceptance/source and existing fence foundations; human hidden review work remains independently dependency-gated | | [WS-QUAL-002](initiatives/WS-QUAL-002/OVERVIEW.md) | Planned | Populate subsystem ownership before changed-line mutation work | | [WS-QUAL-003](initiatives/WS-QUAL-003/OVERVIEW.md) | Planned | Audit and prune test proof, add missing safety cases, decompose oversized test modules | +| [WS-CI-006](initiatives/WS-CI-006/OVERVIEW.md) | Planned | Deterministic change-impact test selection with full-suite fallback and periodic complete verification | | [WS-XINT-002](initiatives/WS-XINT-002/OVERVIEW.md) | Planned | Remaining ART/AUTH activation edges only | | [WS-XINT-003](initiatives/WS-XINT-003/OVERVIEW.md) | Planned | Resume activation only against exact merged REV behavior | | WS-POL-002 | Superseded | Future guide inference belongs to WS-POL-003; reframe remaining executor work against current specifications | diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md new file mode 100644 index 000000000..3dcbf51e6 --- /dev/null +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -0,0 +1,80 @@ +# WS-CI-006 — Change-impact backend verification + +- Disposition: Planned +- Next usable boundary: [WS-CI-006-01](WS-CI-006-01.md) + +## Intent + +Reduce routine backend pull-request feedback by running the tests affected by a +change, while keeping the required result trustworthy. A narrow change should +not wait for every unrelated subsystem; changes whose impact cannot be proven +must still run the full suite. + +## Current evidence + +- `backend/scripts/test_lane_catalogue.py` assigns every discovered test module + to the complete backend run. `backend/scripts/run_test_lanes.py` then + deterministically hash-partitions node IDs within shared, project, and task + groups; these partitions are not an impact map. +- `.github/workflows/backend.yml` always schedules nine backend lanes, a + separate authorization-boundary preflight, MinIO image construction, and a + final real-API/evidence aggregation job. +- Run [36724982896](https://github.com/Flow-Research/workstream/actions/runs/36724982896) + on 2026-09-30 completed 7,918 tests, zero skipped/deselected, in about 44 + minutes. Three lane jobs began about 21 minutes after the first six; the + longest lane then ran about 17 minutes. This is one observed run, not a + universal baseline. +- `CONTRIBUTING.md` currently requires hosted full suites. Selective PR checks + therefore require an explicit, reviewed policy update—not merely a workflow + optimization. + +## Design direction + +- Use a deterministic repository-owned changed-source-to-test ownership map; + do not add an external test-impact service or trust mutable historical + selector state. +- Bind selection to the exact PR base/head and a machine-validated manifest. + Changed tests always run. A path with no reviewed mapping, shared fixtures, + schema/migration/dependency changes, or CI-selection changes conservatively + selects the complete suite. +- Keep the required Backend workflow and final check present on every PR. Do + not use GitHub workflow path filters to suppress a required status. +- Make the selected impact set the PR gate. Keep full-suite execution available + for broad changes, manual runs, and a scheduled main-branch audit. Never + describe the scheduled full-suite result as proof for a different PR head. +- Preserve real PostgreSQL, S3-protocol, concurrency, migration, and public API + checks whenever their owner paths are selected. The five-minute target is + for routine narrow changes, not broad/security/schema changes or runner + outages; hosted wall time remains measured, not promised. +- Leave test bodies, assertions, coverage policy, product behavior, and + external review requirements unchanged. + +## Proposed boundary + +1. **WS-CI-006-01:** implement the exact-base impact manifest, conservative + fallback, selected-node evidence validation, and CI integration; change the + contributor policy in the same PR; prove selection decisions with adversarial + tests and the required full suite on the candidate. +2. Add source/test ownership mappings incrementally only when their consumer + coverage and shared-fixture edges are demonstrated. Unmapped code remains + full-suite. Do not create follow-on PRs solely to satisfy this overview. + +## Risks and controls + +- **False-negative selection:** exact ownership coverage, full-suite fallback + for unknown or cross-cutting paths, mapping mutation probes, plus scheduled + full-suite audits. +- **Stale or mismatched evidence:** exact base/head, tree and manifest digests; + no cached result may attest to a different source tree. +- **Broken branch protection:** workflow and final required status always run; + no path-filtered required workflow and no empty-selection success. +- **Misleading performance claim:** report selected test count and hosted wall + time separately; retain full execution for risky changes and report measured + results rather than claiming every PR is under five minutes. + +## Non-goals + +- No test deletion, skipping, weakened assertions, coverage threshold, arbitrary + shard expansion, runner-provider change, new service, or agent-controlled + selection authority. +- No changes to product code or test behavior. diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md new file mode 100644 index 000000000..51cf01dd5 --- /dev/null +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -0,0 +1,126 @@ +# WS-CI-006-01 — Gate PRs on proven change-impact tests + +- Initiative: `WS-CI-006` +- Durable disposition: `Planned` +- Intended merge outcome: Routine mapped changes run only their proven impacted backend tests, while uncertain or broad changes fail closed to the full suite. + +## Intent + +Make backend feedback proportional to the change without making test omission a +success path. The current required workflow executes the full 7,918-node suite +on every backend pull request; the measured 2026-09-30 run took about 44 +minutes. A five-minute goal is limited to routine narrow changes and must be +measured on hosted runs. + +## Current behavior + +- `backend/scripts/test_lane_catalogue.py` requires every discovered test module + in the full-run inventory and partitions selected nodes across nine jobs. +- `backend/scripts/run_test_lanes.py` binds collection/completion evidence to + the candidate SHA and rejects skipped/deselected nodes, but it has no + changed-source impact selector. +- `.github/workflows/backend.yml` runs all nine lanes plus preflight and the + final real-API/evidence job for each PR and main push. +- `CONTRIBUTING.md` says full hosted suites remain required; this PR must update + that policy explicitly and keep full-suite and real-integration checks + blocking under the documented impact policy. + +## Bounded change + +### Allowed + +- `.github/workflows/backend.yml` +- `backend/scripts/test_lane_catalogue.py` +- `backend/scripts/run_test_lanes.py` +- `backend/scripts/merge_test_lane_evidence.py` +- `backend/scripts/validate_test_lane_evidence.py` +- Their direct CI workflow/catalogue/evidence tests under `backend/tests/` +- A small reviewed impact map and its validator under `.ci/test-impact/` +- `AGENTS.md`, `CONTRIBUTING.md`, `docs/roadmap_status.md` +- This initiative record and `.commitrail/INDEX.md` + +### Not allowed + +- Product source, schemas, migrations, test bodies/assertions, public APIs, or + test deletion/skip/xfail/deselection behavior. +- GitHub workflow-level path filters for required checks. +- Third-party test-selection services, mutable selector databases, cache-only + proof, new runner infrastructure, or arbitrary shards. +- Marking a broad, unmapped, stale-base, or empty selection green. + +## Design and decisions + +- Select explicit test modules/nodes from a version-controlled source-to-test + ownership map; the present hash-based lane assignment cannot itself establish + impact. +- Compute the change against GitHub's exact base and head SHAs. Validate the + complete changed-path inventory, impact-map digest, selected node manifest, + test completion, and selected coverage custody at fan-in. +- Always include newly changed test modules. Unmapped paths, shared setup, + dependency/CI/test-inventory changes, schema/migrations, and cross-cutting + authority/storage/transaction changes select the full suite until their + narrower impact is explicitly proven. +- Keep the existing required workflow/final status present. PR and push runs + use the same deterministic selector. Run the complete suite on a schedule + and by manual dispatch; broad changes still run it before merge. +- Do not consume a scheduled run as evidence for a PR commit. Stale/missing + selector data or inability to prove selection falls back to the full suite. +- Preserve real Postgres/MinIO/API checks for selected owners. Only provision + services proven unnecessary for the selected set; any uncertain dependency + keeps the service and its tests. + +## Acceptance criteria + +- [ ] Exact base/head and every changed path are bound into selection evidence. +- [ ] Tests and changed files outside the reviewed impact map force full-suite + selection; the map cannot silently omit a required module. +- [ ] Selected nodes are enumerated before execution, all complete, and no + selected node is skipped or deselected; fan-in rejects omissions, duplicates, + foreign-head evidence, missing jobs, and empty selections. +- [ ] Adversarial tests inject an omitted module, an unknown path, a stale base, + a duplicate/missing result, and a broad/global path; each must force full + selection or fail closed. +- [ ] Real PostgreSQL/S3/API integration proofs remain blocking for the owners + that need them; no test body, assertion, or coverage policy is weakened. +- [ ] Required Backend check is always reported; the ordinary PR selector can + never bypass the final evidence aggregator. +- [ ] Full-suite scheduled/manual workflow remains available and its result is + bound to its own head; roadmap and contributor policy accurately distinguish + full-audit from per-PR impact evidence. +- [ ] Hosted timing evidence measures representative narrow, broad, and unknown + changes; only observed routine narrow changes may be reported against the + five-minute goal. + +## Risk and review routing + +- Risk class: `L1` +- Required reviewers: `ci_integrity`, `qa`, `test_delta`, `security`, + `architecture`, `documentation` +- Human review focus: whether the change-to-test map is complete enough for the + initially enabled paths, whether full fallback triggers are broad enough, + and whether scheduled full-suite proof is an acceptable complement to the + required per-PR impact gate. + +## Evidence + +| Claim | Command or proof | Result | Remaining uncertainty | +|---|---|---|---| +| Current required workflow executes the complete suite | Exact workflow and lane catalogue inspection | Baseline: 9 lanes and 7,918 nodes in run 36724982896 | One run is not a long-term distribution | +| Typical narrow changes can avoid unrelated suite work safely | Impact-map negative/positive fixtures plus exact-hosted PR runs | Pending implementation | Map coverage must grow conservatively | +| Full suite remains complete over time | Scheduled/manual exact-head run with existing evidence fan-in | Pending implementation | Scheduler delays do not prove freshness | +| Required status cannot be skipped | Workflow and fan-in probes for all change classifications | Pending implementation | Branch protection configuration must be inspected | + +## Review findings + +No implementation findings yet. + +## Reconciliation + +- Current-source reconciliation: backend CI has nine hash-partitioned lanes, + exact test-result custody, a required final aggregator, and hosted wall-time + evidence. CONTRIBUTING currently requires the full hosted suite on every PR. +- Next usable boundary: implement selector, evidence validation, and CI policy + together; do not activate partial selection until fail-closed tests and exact + hosted evidence pass. +- Remaining risks: test-to-source ownership is not yet recorded; broad paths + must remain full-suite until every required consumer is identified. From df15104ddbaeec5df8bd8f8eb408727255906925 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 17:32:40 +0100 Subject: [PATCH 02/15] docs(ci): bound change impact pilot and evidence --- .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 47 ++-- .../initiatives/WS-CI-006/WS-CI-006-01.md | 233 ++++++++++++------ 2 files changed, 188 insertions(+), 92 deletions(-) diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md index 3dcbf51e6..da5382bd1 100644 --- a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -33,15 +33,26 @@ must still run the full suite. - Use a deterministic repository-owned changed-source-to-test ownership map; do not add an external test-impact service or trust mutable historical selector state. -- Bind selection to the exact PR base/head and a machine-validated manifest. - Changed tests always run. A path with no reviewed mapping, shared fixtures, - schema/migration/dependency changes, or CI-selection changes conservatively - selects the complete suite. +- Start with one narrow reviewed mapping only: changes confined to + `backend/app/core/s3_validation.py` select its direct configuration contract + tests, provider-neutral namespace-conformance tests, and real S3/MinIO adapter + tests. Changes to its callers, shared test support, schemas/migrations, + dependencies, or CI-selection machinery select the complete suite. No other + application-source path is selective initially. +- Bind selection to the exact PR base, head, synthetic merge execution SHA/tree, + merge-base, changed-path digest, impact-map digest, test-inventory digest, + and selected test/job manifest. A mismatch, missing Git object, stale + candidate, changed test support, or unreviewed path selects the complete + suite. Changes within mapped test modules run the complete mapped closure; + new or unmapped tests and shared test fixtures select the full suite. - Keep the required Backend workflow and final check present on every PR. Do not use GitHub workflow path filters to suppress a required status. - Make the selected impact set the PR gate. Keep full-suite execution available - for broad changes, manual runs, and a scheduled main-branch audit. Never - describe the scheduled full-suite result as proof for a different PR head. + for broad changes and manual runs, and run the complete suite nightly on + `main`. A failed nightly audit remains failed and requires diagnosis of the + map or product/test defect; it does not silently green PR checks or substitute + for exact-PR evidence. Never describe a scheduled result as proof for another + PR head. - Preserve real PostgreSQL, S3-protocol, concurrency, migration, and public API checks whenever their owner paths are selected. The five-minute target is for routine narrow changes, not broad/security/schema changes or runner @@ -51,13 +62,15 @@ must still run the full suite. ## Proposed boundary -1. **WS-CI-006-01:** implement the exact-base impact manifest, conservative - fallback, selected-node evidence validation, and CI integration; change the - contributor policy in the same PR; prove selection decisions with adversarial - tests and the required full suite on the candidate. -2. Add source/test ownership mappings incrementally only when their consumer - coverage and shared-fixture edges are demonstrated. Unmapped code remains - full-suite. Do not create follow-on PRs solely to satisfy this overview. +1. **WS-CI-006-01:** implement the exact-target impact manifest, the initial + S3-validation mapping, conservative full-suite fallback, selected-node/job + evidence validation, and CI integration; change `AGENTS.md` and + `CONTRIBUTING.md` in the same PR; prove the selector adversarially and run + the complete suite on the candidate. +2. Add further source/test ownership mappings only when their complete + consumer-test closure and shared-fixture/infrastructure dependencies are + demonstrated. Unmapped application code remains full-suite. Do not add a + mapping just to claim a broader speedup. ## Risks and controls @@ -65,9 +78,13 @@ must still run the full suite. for unknown or cross-cutting paths, mapping mutation probes, plus scheduled full-suite audits. - **Stale or mismatched evidence:** exact base/head, tree and manifest digests; - no cached result may attest to a different source tree. + execution tree must be the exact GitHub PR merge candidate whose parents are + the event base and head; no cached result may attest to another source tree. - **Broken branch protection:** workflow and final required status always run; - no path-filtered required workflow and no empty-selection success. + preserve the existing required `test` context, distinguish expected + unselected jobs from missing selected jobs using the digest-bound + manifest, and reject an empty test selection. No path-filtered required + workflow. - **Misleading performance claim:** report selected test count and hosted wall time separately; retain full execution for risky changes and report measured results rather than claiming every PR is under five minutes. diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index 51cf01dd5..71dce2076 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -2,28 +2,37 @@ - Initiative: `WS-CI-006` - Durable disposition: `Planned` -- Intended merge outcome: Routine mapped changes run only their proven impacted backend tests, while uncertain or broad changes fail closed to the full suite. +- Intended merge outcome: The exact Backend `test` PR check blocks on the complete tests for one initially mapped low-level owner, while every other or uncertain change runs the complete suite. A nightly full suite audits for selector drift. ## Intent -Make backend feedback proportional to the change without making test omission a -success path. The current required workflow executes the full 7,918-node suite -on every backend pull request; the measured 2026-09-30 run took about 44 -minutes. A five-minute goal is limited to routine narrow changes and must be -measured on hosted runs. +Reduce routine backend feedback without making test omission a success path. +The current required workflow ran all 7,918 test nodes on every backend PR; +run 36724982896 took about 44 minutes wall-clock. The initial selective +boundary is deliberately one owner only: canonical S3 configuration and +namespace validation. All other application-source changes remain full-suite +until their complete consumer-test closure is explicitly mapped. The five- +minute target applies only to measured, routine narrow changes; it is not a +promise for broad changes or GitHub runner delays. -## Current behavior +## Current behavior and constraints -- `backend/scripts/test_lane_catalogue.py` requires every discovered test module - in the full-run inventory and partitions selected nodes across nine jobs. +- `backend/scripts/test_lane_catalogue.py` inventories every discovered test + module and hash-partitions nodes across nine jobs. A lane is not an impact + owner; omitting one drops arbitrary hashed nodes. - `backend/scripts/run_test_lanes.py` binds collection/completion evidence to - the candidate SHA and rejects skipped/deselected nodes, but it has no - changed-source impact selector. -- `.github/workflows/backend.yml` runs all nine lanes plus preflight and the - final real-API/evidence job for each PR and main push. -- `CONTRIBUTING.md` says full hosted suites remain required; this PR must update - that policy explicitly and keep full-suite and real-integration checks - blocking under the documented impact policy. + the checked-out candidate and rejects skipped/deselected nodes, but it has no + source-impact selector. +- `.github/workflows/backend.yml` runs nine lanes, authorization preflight, + MinIO build and final API/evidence aggregation on each PR and main push. +- Live branch protection requires the `test` and `agent-gates` checks, requires + up-to-date branches and one human approval, and enforces protection for + administrators. Keep the exact `test` required status present on every PR. +- `AGENTS.md`, `CONTRIBUTING.md` and the roadmap currently require full hosted + suites on every PR. This PR must change that policy consistently: the exact + selected set is blocking for a mapped PR; uncertain/broad changes require the + full suite; a nightly full suite audits the map but never substitutes for + evidence from a different PR head. ## Bounded change @@ -34,93 +43,163 @@ measured on hosted runs. - `backend/scripts/run_test_lanes.py` - `backend/scripts/merge_test_lane_evidence.py` - `backend/scripts/validate_test_lane_evidence.py` -- Their direct CI workflow/catalogue/evidence tests under `backend/tests/` -- A small reviewed impact map and its validator under `.ci/test-impact/` +- Direct CI workflow/catalogue/selector/evidence tests under `backend/tests/` +- The exact reviewed impact map and validator under `.ci/test-impact/` - `AGENTS.md`, `CONTRIBUTING.md`, `docs/roadmap_status.md` - This initiative record and `.commitrail/INDEX.md` ### Not allowed -- Product source, schemas, migrations, test bodies/assertions, public APIs, or - test deletion/skip/xfail/deselection behavior. -- GitHub workflow-level path filters for required checks. -- Third-party test-selection services, mutable selector databases, cache-only - proof, new runner infrastructure, or arbitrary shards. -- Marking a broad, unmapped, stale-base, or empty selection green. +- Product source other than `backend/app/core/s3_validation.py`, schemas, + migrations, public API changes, or changes to product-behavior assertions. +- Deleting tests or changing skip/xfail/deselection behavior; rewriting tests to + match broken behavior. +- GitHub path filters for the required workflow; third-party impact services, + mutable selector databases, cache-only proof, new runner infrastructure, or + arbitrary shards. +- A green result for a broad, unmapped, mismatched, empty, or incomplete test + selection. ## Design and decisions -- Select explicit test modules/nodes from a version-controlled source-to-test - ownership map; the present hash-based lane assignment cannot itself establish - impact. -- Compute the change against GitHub's exact base and head SHAs. Validate the - complete changed-path inventory, impact-map digest, selected node manifest, - test completion, and selected coverage custody at fan-in. -- Always include newly changed test modules. Unmapped paths, shared setup, - dependency/CI/test-inventory changes, schema/migrations, and cross-cutting - authority/storage/transaction changes select the full suite until their - narrower impact is explicitly proven. -- Keep the existing required workflow/final status present. PR and push runs - use the same deterministic selector. Run the complete suite on a schedule - and by manual dispatch; broad changes still run it before merge. -- Do not consume a scheduled run as evidence for a PR commit. Stale/missing - selector data or inability to prove selection falls back to the full suite. -- Preserve real Postgres/MinIO/API checks for selected owners. Only provision - services proven unnecessary for the selected set; any uncertain dependency - keeps the service and its tests. +### Initial selective boundary + +Only a PR whose application-source change is confined to +`backend/app/core/s3_validation.py` may use impact selection. Its required +closure is: + +| Owner behavior | Required tests | Infrastructure / shared dependencies | +|---|---|---| +| S3 region/bucket/prefix and MinIO endpoint canonicalization; provider namespace descriptor validation | `backend/tests/test_config.py`; `backend/tests/test_artifact_store_conformance.py` (including its standalone namespace-value cases); `backend/tests/test_s3_artifact_store.py` (including inherited real-adapter conformance vectors) | Source-pinned real MinIO is required by the S3 adapter tests. PostgreSQL is not required. Shared unchanged support includes `backend/tests/conftest.py` and `backend/tests/artifact_store_helpers.py`. | + +`test_config.py` directly proves helper behavior and Settings/namespace +validation. The S3 adapter tests prove the canonical configuration reaches the +real provider adapter correctly. Any change to the helper's consumers +(`backend/app/core/config.py`, `backend/app/interfaces/artifacts.py`, or +`backend/app/adapters/artifacts/s3_compatible.py`), to shared support, or to +any other application source selects the full suite. No other source mapping +is enabled by this change. + +Changes within any of the three mapped test modules run the complete three- +module closure, not just changed node IDs. Changes to any other test module, a +new test, shared fixture/helper, test inventory, dependency, schema/migration, +runner, workflow, impact map or selector run the full suite. This prevents test +edits from silently shrinking or redefining the proof used by the mapping. + +### Target and evidence custody + +- For a PR, compute changed paths from explicit event `base.sha` and + `head.sha`, using their Git merge-base; do not read PR descriptions as + authority. Run the selector/map from the trusted base revision, not from PR + code. A changed selector, map or workflow itself always selects full suite. +- Bind event base SHA, head SHA, merge-base, GitHub execution SHA and tree, + changed-path list/digest, selector/map digest, full test-inventory digest, + selected modules/node IDs, infrastructure profile, and expected job names in + a deterministic manifest. +- Require the test checkout to be the exact GitHub synthetic merge candidate + whose parents are the event base and head. Validate object presence, + ancestry, parent identities, clean checkout, and tree equality before + accepting selective mode. If candidate/base/head is stale, unavailable, + malformed, has unexpected parents, or any digest differs, run the full suite + against the checked-out execution tree. Never attest one tree with another + tree's test list or cached evidence. +- Changed tests are part of the selected set. The selected manifest must be + non-empty, enumerate expected nodes before execution, and bind expected job + inventory. Selected tests may not be skipped or deselected. + +### Workflow and branch protection + +- Preserve the always-created required `test` check and the separate required + `agent-gates` check. Do not use workflow-level path filters. +- The final `test` aggregator validates the digest-bound manifest and exact + expected job inventory. Full mode expects all current nine lanes; mapped mode + expects the explicit MinIO-backed impact job and always-required static and + authorization-boundary jobs. Expected-unselected lanes differ from missing + selected jobs. Missing, duplicate, foreign-target, skipped, deselected or + incomplete selected evidence fails. +- Keep full lint and docstring checks blocking but execute them once in a + shared job, not redundantly in every test lane. Keep authorization-boundary + preflight required for every PR and at fan-in. Impact mode retains real + MinIO tests; full mode retains PostgreSQL, MinIO, migration, transaction, + concurrency, boundary and real API proof. +- Add nightly full-suite execution on `main` and manual dispatch. The full run + must bind to its own head, reject skips/deselections and report failures + normally. A nightly result never satisfies a PR check. A failure requires + diagnosis of a product/test defect or selector drift; do not translate it to + green or claim it proves another commit. +- Remove the redundant full Backend run on each protected `main` push only + after retaining exact PR checks, strict up-to-date protection and nightly / + manual full runs. Protection currently requires review and both checks, and + is enforced for admins, so merged code has passed the exact candidate check. + Record this as removing duplicated execution, not removing required PR + verification. ## Acceptance criteria -- [ ] Exact base/head and every changed path are bound into selection evidence. -- [ ] Tests and changed files outside the reviewed impact map force full-suite - selection; the map cannot silently omit a required module. -- [ ] Selected nodes are enumerated before execution, all complete, and no - selected node is skipped or deselected; fan-in rejects omissions, duplicates, - foreign-head evidence, missing jobs, and empty selections. -- [ ] Adversarial tests inject an omitted module, an unknown path, a stale base, - a duplicate/missing result, and a broad/global path; each must force full - selection or fail closed. -- [ ] Real PostgreSQL/S3/API integration proofs remain blocking for the owners - that need them; no test body, assertion, or coverage policy is weakened. -- [ ] Required Backend check is always reported; the ordinary PR selector can - never bypass the final evidence aggregator. -- [ ] Full-suite scheduled/manual workflow remains available and its result is - bound to its own head; roadmap and contributor policy accurately distinguish - full-audit from per-PR impact evidence. -- [ ] Hosted timing evidence measures representative narrow, broad, and unknown - changes; only observed routine narrow changes may be reported against the - five-minute goal. +- [ ] The only initial product-source mapping is + `backend/app/core/s3_validation.py` to the exact full test closure above; + its MinIO requirement and lack of PostgreSQL dependency are executable and + tested. +- [ ] All other source paths, shared support changes, infrastructure changes, + stale/missing Git objects and unknown changes fail closed to full suite. +- [ ] Base/head/merge-base/execution SHA+tree, changed-path and map digests, + test inventory, selected nodes, infrastructure profile and expected jobs are + bound to evidence. A synthetic merge-parent/tree mismatch cannot pass + selective mode. +- [ ] Adversarial tests cover unknown path, omitted owner module, stale base, + altered execution tree, duplicate/missing results, missing expected job, + unexpected job, broad/global path, empty selection and changed selector/map. +- [ ] Every selected node completes; no selected node is skipped/deselected; + fan-in rejects missing, duplicate, foreign-target and incomplete evidence. +- [ ] Required `test` remains present and blocking; full mode expects all nine + current lanes and impact mode expects exactly the manifest-selected jobs plus + always-required checks. +- [ ] Full scheduled/manual run remains complete, is bound to its own head, + and is never cross-used as exact PR evidence. A full-suite failure stays red. +- [ ] `AGENTS.md`, `CONTRIBUTING.md` and the affected roadmap claims describe + exact-PR impact evidence, full fallback, nightly audit, and retained real + integration checks consistently. +- [ ] Full lint/docstring validation remains blocking and runs once per + workflow. Removing the main-push rerun is justified by protected exact-PR + verification and does not remove a required check. +- [ ] Hosted timing evidence includes mapped narrow and full-fallback runs; + report real elapsed time and do not promise under five minutes absent + representative hosted proof. ## Risk and review routing - Risk class: `L1` - Required reviewers: `ci_integrity`, `qa`, `test_delta`, `security`, - `architecture`, `documentation` -- Human review focus: whether the change-to-test map is complete enough for the - initially enabled paths, whether full fallback triggers are broad enough, - and whether scheduled full-suite proof is an acceptable complement to the - required per-PR impact gate. + `documentation` +- Human review focus: completeness of the only enabled consumer closure, + trustworthiness of target/job custody, full fallback, and preserving every + required full/integration check. ## Evidence | Claim | Command or proof | Result | Remaining uncertainty | |---|---|---|---| -| Current required workflow executes the complete suite | Exact workflow and lane catalogue inspection | Baseline: 9 lanes and 7,918 nodes in run 36724982896 | One run is not a long-term distribution | -| Typical narrow changes can avoid unrelated suite work safely | Impact-map negative/positive fixtures plus exact-hosted PR runs | Pending implementation | Map coverage must grow conservatively | -| Full suite remains complete over time | Scheduled/manual exact-head run with existing evidence fan-in | Pending implementation | Scheduler delays do not prove freshness | -| Required status cannot be skipped | Workflow and fan-in probes for all change classifications | Pending implementation | Branch protection configuration must be inspected | +| Current required workflow executes the complete suite | Workflow, lane catalogue and run 36724982896 | Baseline: 9 lanes and 7,918 tests; about 44 minutes | One run is not a long-term distribution | +| Initial S3 validation owner has a complete selective closure | Direct source-consumer trace, adversarial map tests and exact hosted PR | Pending implementation | Hosted time for real-MinIO closure not measured | +| Unmapped code and mismatched targets fail closed | Selector, synthetic-merge and fan-in adversarial tests | Pending implementation | Full fallback must be exercised on hosted CI | +| Full suite remains an independent drift audit | Nightly/manual exact-head run | Pending implementation | Audit never attests another PR head | +| Required `test` cannot be skipped or spoofed | Workflow and exact expected-job/fan-in probes | Pending implementation | Recheck branch protection and check name after workflow changes | ## Review findings -No implementation findings yet. +Initial plan review findings are being resolved before implementation. ## Reconciliation - Current-source reconciliation: backend CI has nine hash-partitioned lanes, exact test-result custody, a required final aggregator, and hosted wall-time - evidence. CONTRIBUTING currently requires the full hosted suite on every PR. -- Next usable boundary: implement selector, evidence validation, and CI policy - together; do not activate partial selection until fail-closed tests and exact - hosted evidence pass. -- Remaining risks: test-to-source ownership is not yet recorded; broad paths - must remain full-suite until every required consumer is identified. + evidence. Current protected-branch settings require exact PR statuses, + up-to-date branches and human approval; current docs still require full + suites on every PR. +- Next usable boundary: implement the single reviewed S3-validation mapping, + exact-target manifest, conservative fallback, evidence fan-in and current + policy updates together. Do not enable partial selection until its adversarial + tests and exact hosted evidence pass. +- Remaining risks: no other source owner is mapped. Broad product changes, + newly added tests, selector changes and unclassified dependencies remain + full-suite until their complete consumer closure is demonstrated. From f4426e1b1fc83ec304fe4eeaf574a07b1c0ee818 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 17:38:38 +0100 Subject: [PATCH 03/15] docs(ci): refine impact runner evidence and cache plan --- .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 11 +++- .../initiatives/WS-CI-006/WS-CI-006-01.md | 66 ++++++++++++++----- 2 files changed, 60 insertions(+), 17 deletions(-) diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md index da5382bd1..312e68937 100644 --- a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -47,6 +47,11 @@ must still run the full suite. new or unmapped tests and shared test fixtures select the full suite. - Keep the required Backend workflow and final check present on every PR. Do not use GitHub workflow path filters to suppress a required status. +- Run shared lint/docstring and authorization-boundary checks once, not once per + test lane. Keep full public-API E2E and service startup in full mode unless a + reviewed impact mapping specifically requires it. Reuse the pinned MinIO + build by exact source-input digest, not commit SHA; verify cache contents + before tests. - Make the selected impact set the PR gate. Keep full-suite execution available for broad changes and manual runs, and run the complete suite nightly on `main`. A failed nightly audit remains failed and requires diagnosis of the @@ -86,8 +91,10 @@ must still run the full suite. manifest, and reject an empty test selection. No path-filtered required workflow. - **Misleading performance claim:** report selected test count and hosted wall - time separately; retain full execution for risky changes and report measured - results rather than claiming every PR is under five minutes. + time separately; the selector-changing PR itself full-fallbacks. Wait for the + first later naturally eligible mapped PR before reporting narrow-mode hosted + timing; retain full execution for risky changes and do not claim every PR is + under five minutes. ## Non-goals diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index 71dce2076..f28879f77 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -25,6 +25,10 @@ promise for broad changes or GitHub runner delays. source-impact selector. - `.github/workflows/backend.yml` runs nine lanes, authorization preflight, MinIO build and final API/evidence aggregation on each PR and main push. + The current final aggregator also starts PostgreSQL and MinIO and runs the + full real-API E2E on every PR, even for a narrow owner. Its MinIO image cache + key includes the workflow SHA, so unchanged pinned source is rebuilt on new + PR heads. - Live branch protection requires the `test` and `agent-gates` checks, requires up-to-date branches and one human approval, and enforces protection for administrators. Keep the exact `test` required status present on every PR. @@ -39,19 +43,22 @@ promise for broad changes or GitHub runner delays. ### Allowed - `.github/workflows/backend.yml` +- `docker/minio/README.md` (cache policy description only) - `backend/scripts/test_lane_catalogue.py` - `backend/scripts/run_test_lanes.py` - `backend/scripts/merge_test_lane_evidence.py` - `backend/scripts/validate_test_lane_evidence.py` - Direct CI workflow/catalogue/selector/evidence tests under `backend/tests/` +- `scripts/test_lightweight_agent_gates.py` (its Backend workflow shape checks) - The exact reviewed impact map and validator under `.ci/test-impact/` - `AGENTS.md`, `CONTRIBUTING.md`, `docs/roadmap_status.md` - This initiative record and `.commitrail/INDEX.md` ### Not allowed -- Product source other than `backend/app/core/s3_validation.py`, schemas, - migrations, public API changes, or changes to product-behavior assertions. +- Product source other than `backend/app/core/s3_validation.py`, Docker/MinIO + build inputs, schemas, migrations, public API changes, or changes to + product-behavior assertions. - Deleting tests or changing skip/xfail/deselection behavior; rewriting tests to match broken behavior. - GitHub path filters for the required workflow; third-party impact services, @@ -81,10 +88,11 @@ any other application source selects the full suite. No other source mapping is enabled by this change. Changes within any of the three mapped test modules run the complete three- -module closure, not just changed node IDs. Changes to any other test module, a -new test, shared fixture/helper, test inventory, dependency, schema/migration, -runner, workflow, impact map or selector run the full suite. This prevents test -edits from silently shrinking or redefining the proof used by the mapping. + module closure, not just changed node IDs. Changes to any other test module, a + new test, shared fixture/helper, test inventory, dependency, schema/migration, + Docker/MinIO build input, runner, workflow, impact map or selector run the + full suite. This prevents test edits from silently shrinking or redefining + the proof used by the mapping. ### Target and evidence custody @@ -119,9 +127,21 @@ edits from silently shrinking or redefining the proof used by the mapping. incomplete selected evidence fails. - Keep full lint and docstring checks blocking but execute them once in a shared job, not redundantly in every test lane. Keep authorization-boundary - preflight required for every PR and at fan-in. Impact mode retains real - MinIO tests; full mode retains PostgreSQL, MinIO, migration, transaction, - concurrency, boundary and real API proof. + preflight required for every PR and at fan-in. Combine static checks with + that existing preflight job so they share one dependency installation. The + final evidence `test` job must not start database/storage services or rerun + product tests; it aggregates and verifies evidence from expected jobs. Move real-API E2E into + an explicit full-suite job. Impact mode does not run that E2E unless a future + reviewed mapping names it; the S3 mapping retains real MinIO adapter tests + but does not require PostgreSQL. Full mode and nightly audits retain + PostgreSQL, MinIO, migration, transaction, concurrency, boundary and real + API proof. This removes duplicate services only where the selected test + closure proves them unnecessary. +- Change the MinIO cache key from workflow commit SHA to a digest of every + source build input plus runner OS/architecture. The cache is only an + optimization, never evidence: verify the archive checksum and start the + cached image's version/health probe before publishing it to test jobs. An + input change or invalid cache triggers a fresh pinned-source build. - Add nightly full-suite execution on `main` and manual dispatch. The full run must bind to its own head, reject skips/deselections and report failures normally. A nightly result never satisfies a PR check. A failure requires @@ -162,9 +182,23 @@ edits from silently shrinking or redefining the proof used by the mapping. - [ ] Full lint/docstring validation remains blocking and runs once per workflow. Removing the main-push rerun is justified by protected exact-PR verification and does not remove a required check. -- [ ] Hosted timing evidence includes mapped narrow and full-fallback runs; - report real elapsed time and do not promise under five minutes absent - representative hosted proof. +- [ ] The final evidence aggregator starts no PostgreSQL/Redis/MinIO service + and runs no duplicate API E2E; those remain in full-suite mode or a future + specifically mapped closure. +- [ ] MinIO image cache keys include all source/platform inputs but not commit + SHA; warm reuse passes checksum and live version/health checks; invalid or + missing cache builds from pinned source. +- [ ] This selector/workflow implementation PR runs full fallback on its exact + candidate and proves selector/fan-in adversarial cases. Because this PR + changes the selector, map and workflow, it cannot use its own newly added + mapping to create mapped-mode evidence. +- [ ] The first later, naturally eligible PR confined to the mapped S3 owner + or its mapped tests supplies the exact hosted selected-job custody and timing + observation. Until then, the feature is implemented but the narrow hosted + performance result is unverified; do not claim the five-minute target. +- [ ] Report hosted elapsed time for mapped and full-fallback runs when each + naturally occurs; do not use a synthetic path list or manual selector + override as proof. ## Risk and review routing @@ -180,7 +214,7 @@ edits from silently shrinking or redefining the proof used by the mapping. | Claim | Command or proof | Result | Remaining uncertainty | |---|---|---|---| | Current required workflow executes the complete suite | Workflow, lane catalogue and run 36724982896 | Baseline: 9 lanes and 7,918 tests; about 44 minutes | One run is not a long-term distribution | -| Initial S3 validation owner has a complete selective closure | Direct source-consumer trace, adversarial map tests and exact hosted PR | Pending implementation | Hosted time for real-MinIO closure not measured | +| Initial S3 validation owner has a complete selective closure | Direct source-consumer trace, adversarial map tests and exact hosted PR | Pending implementation | Hosted mapped mode cannot be exercised by the selector-changing PR; observe first later naturally eligible S3 change | | Unmapped code and mismatched targets fail closed | Selector, synthetic-merge and fan-in adversarial tests | Pending implementation | Full fallback must be exercised on hosted CI | | Full suite remains an independent drift audit | Nightly/manual exact-head run | Pending implementation | Audit never attests another PR head | | Required `test` cannot be skipped or spoofed | Workflow and exact expected-job/fan-in probes | Pending implementation | Recheck branch protection and check name after workflow changes | @@ -198,8 +232,10 @@ Initial plan review findings are being resolved before implementation. suites on every PR. - Next usable boundary: implement the single reviewed S3-validation mapping, exact-target manifest, conservative fallback, evidence fan-in and current - policy updates together. Do not enable partial selection until its adversarial - tests and exact hosted evidence pass. + policy updates together. The implementation PR must full-fallback and prove + adversarial cases; the first later naturally eligible mapped PR supplies the + selected-mode hosted observation. Never report the performance target before + that observation. - Remaining risks: no other source owner is mapped. Broad product changes, newly added tests, selector changes and unclassified dependencies remain full-suite until their complete consumer closure is demonstrated. From c1c2dc84bbb0eb11b8ad578a15f6ebfbb669c5d3 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 17:43:23 +0100 Subject: [PATCH 04/15] docs(ci): define metadata-only impact handling --- .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 5 +++++ .../initiatives/WS-CI-006/WS-CI-006-01.md | 17 +++++++++++++++++ 2 files changed, 22 insertions(+) diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md index 312e68937..6f87d15ea 100644 --- a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -39,6 +39,11 @@ must still run the full suite. tests. Changes to its callers, shared test support, schemas/migrations, dependencies, or CI-selection machinery select the complete suite. No other application-source path is selective initially. +- Keep `.commitrail/**` in the exact change manifest, but exclude it only when + classifying backend source impact: every bounded product PR carries a durable + change record. Metadata-only PRs still run Markdown/stale-doc gates and a + non-empty authorization/static preflight; other documentation paths remain + full-suite unless separately reviewed. - Bind selection to the exact PR base, head, synthetic merge execution SHA/tree, merge-base, changed-path digest, impact-map digest, test-inventory digest, and selected test/job manifest. A mismatch, missing Git object, stale diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index f28879f77..dbbf500e3 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -100,6 +100,17 @@ Changes within any of the three mapped test modules run the complete three- `head.sha`, using their Git merge-base; do not read PR descriptions as authority. Run the selector/map from the trusted base revision, not from PR code. A changed selector, map or workflow itself always selects full suite. +- Preserve the complete changed-path list and digest in the evidence manifest. + For backend-impact classification only, remove paths under `.commitrail/` + from the source-impact input: every bounded PR carries its own durable + Commitrail record, so treating that metadata as backend source would force + all product PRs into full fallback. A PR containing only `.commitrail/` + changes uses an explicit metadata-only mode with a non-empty, named + authorization-boundary/static preflight; it does not select zero tests. + The required `agent-gates` workflow continues to validate Commitrail and + Markdown/stale wording. Mixed Commitrail plus mapped S3 changes use the S3 + closure; mixed Commitrail plus any unclassified path use full suite. Do not + exempt arbitrary documentation, skills, agent instructions, or other paths. - Bind event base SHA, head SHA, merge-base, GitHub execution SHA and tree, changed-path list/digest, selector/map digest, full test-inventory digest, selected modules/node IDs, infrastructure profile, and expected job names in @@ -162,6 +173,12 @@ Changes within any of the three mapped test modules run the complete three- tested. - [ ] All other source paths, shared support changes, infrastructure changes, stale/missing Git objects and unknown changes fail closed to full suite. +- [ ] Commitrail paths remain in the exact changed-path manifest but are + excluded only from backend source-impact classification. A Commitrail-only + PR has a named, non-empty authorization/static preflight and still runs + `agent-gates`; mixed changes follow the mapped-owner or conservative full + fallback rule. No other documentation or metadata path is implicitly + exempted. - [ ] Base/head/merge-base/execution SHA+tree, changed-path and map digests, test inventory, selected nodes, infrastructure profile and expected jobs are bound to evidence. A synthetic merge-parent/tree mismatch cannot pass From c2c443e4f225ea3ca87025144e624ebbeb3f7e19 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 17:47:43 +0100 Subject: [PATCH 05/15] docs(ci): preserve Commitrail policy test impact --- .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 4 +- .../initiatives/WS-CI-006/WS-CI-006-01.md | 38 +++++++++++++------ 2 files changed, 30 insertions(+), 12 deletions(-) diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md index 6f87d15ea..00be24bc5 100644 --- a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -41,7 +41,9 @@ must still run the full suite. application-source path is selective initially. - Keep `.commitrail/**` in the exact change manifest, but exclude it only when classifying backend source impact: every bounded product PR carries a durable - change record. Metadata-only PRs still run Markdown/stale-doc gates and a + change record. Every PR that changes Commitrail metadata also runs the + backend policy-semantics module that reads specific planning files. A + metadata-only PR still runs that module, Markdown/stale-doc gates, and the non-empty authorization/static preflight; other documentation paths remain full-suite unless separately reviewed. - Bind selection to the exact PR base, head, synthetic merge execution SHA/tree, diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index dbbf500e3..7015dbf1e 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -104,13 +104,23 @@ Changes within any of the three mapped test modules run the complete three- For backend-impact classification only, remove paths under `.commitrail/` from the source-impact input: every bounded PR carries its own durable Commitrail record, so treating that metadata as backend source would force - all product PRs into full fallback. A PR containing only `.commitrail/` - changes uses an explicit metadata-only mode with a non-empty, named - authorization-boundary/static preflight; it does not select zero tests. + all product PRs into full fallback. Each PR with any `.commitrail/` change + additionally selects the complete module + `backend/tests/projects/review_policy/test_semantics.py`, including its + `test_future_activation_owners_preserve_unavailable_mode_contract` case, + which reads and checks two Commitrail planning files. This test is run for + metadata-only changes and unioned with any mapped source closure. A PR + containing only Commitrail changes therefore has a named, non-empty backend + test set as well as the authorization/static preflight; it never selects + zero tests. The required `agent-gates` workflow continues to validate Commitrail and - Markdown/stale wording. Mixed Commitrail plus mapped S3 changes use the S3 - closure; mixed Commitrail plus any unclassified path use full suite. Do not - exempt arbitrary documentation, skills, agent instructions, or other paths. + Markdown/stale wording. Any Commitrail change adds the named policy-contract + test module; mixed Commitrail plus mapped S3 changes run both closures. Mixed + Commitrail plus any unclassified path use full suite. Changes that add or + alter backend tests, shared support, or their Commitrail dependencies already + trigger full fallback through their changed test/support/selector paths. + Do not exempt arbitrary documentation, skills, agent instructions, or other + paths. - Bind event base SHA, head SHA, merge-base, GitHub execution SHA and tree, changed-path list/digest, selector/map digest, full test-inventory digest, selected modules/node IDs, infrastructure profile, and expected job names in @@ -174,11 +184,17 @@ Changes within any of the three mapped test modules run the complete three- - [ ] All other source paths, shared support changes, infrastructure changes, stale/missing Git objects and unknown changes fail closed to full suite. - [ ] Commitrail paths remain in the exact changed-path manifest but are - excluded only from backend source-impact classification. A Commitrail-only - PR has a named, non-empty authorization/static preflight and still runs - `agent-gates`; mixed changes follow the mapped-owner or conservative full - fallback rule. No other documentation or metadata path is implicitly - exempted. + excluded only from backend source-impact classification. Any Commitrail + change selects the complete policy-semantics test module, including + `test_future_activation_owners_preserve_unavailable_mode_contract`; this is + unioned with the S3 closure for mixed mapped changes. Metadata-only mode also + runs the named non-empty authorization/static preflight and `agent-gates`. + Mixed changes with unclassified paths use full suite. No other documentation + or metadata path is implicitly exempted. +- [ ] Adversarial classification tests change each Commitrail file consumed by + the policy-contract test and prove that the module is selected, including + when combined with the mapped S3 owner; a Commitrail-only unrelated metadata + path still selects this non-empty module and preflight. - [ ] Base/head/merge-base/execution SHA+tree, changed-path and map digests, test inventory, selected nodes, infrastructure profile and expected jobs are bound to evidence. A synthetic merge-parent/tree mismatch cannot pass From 6a9d0ebcf51e25437b9cebcb9ed413785214b995 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 18:21:09 +0100 Subject: [PATCH 06/15] docs(ci): bound selector and testing operations files --- .commitrail/initiatives/WS-CI-006/WS-CI-006-01.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index 7015dbf1e..247c9938f 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -45,6 +45,7 @@ promise for broad changes or GitHub runner delays. - `.github/workflows/backend.yml` - `docker/minio/README.md` (cache policy description only) - `backend/scripts/test_lane_catalogue.py` +- `backend/scripts/test_impact_selection.py` - `backend/scripts/run_test_lanes.py` - `backend/scripts/merge_test_lane_evidence.py` - `backend/scripts/validate_test_lane_evidence.py` @@ -52,6 +53,7 @@ promise for broad changes or GitHub runner delays. - `scripts/test_lightweight_agent_gates.py` (its Backend workflow shape checks) - The exact reviewed impact map and validator under `.ci/test-impact/` - `AGENTS.md`, `CONTRIBUTING.md`, `docs/roadmap_status.md` +- `docs/operations_backend_testing.md` - This initiative record and `.commitrail/INDEX.md` ### Not allowed From 387547e780a867025553bd5ffe44b559b7c6550f Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Wed, 30 Sep 2026 18:40:00 +0100 Subject: [PATCH 07/15] ci(backend): add trusted change-impact selection --- .ci/test-impact/impact_map.json | 24 + .ci/test-impact/run_selected_tests.py | 343 ++++++++++++ .ci/test-impact/validate_workflow_jobs.py | 81 +++ .../initiatives/WS-CI-006/WS-CI-006-01.md | 5 +- .github/workflows/backend.yml | 513 +++++++++++++++--- AGENTS.md | 10 +- CONTRIBUTING.md | 10 +- backend/scripts/test_impact_selection.py | 309 +++++++++++ backend/scripts/test_lane_catalogue.py | 1 + backend/tests/test_ci_impact_selection.py | 299 ++++++++++ docker/minio/README.md | 17 +- docs/operations_backend_testing.md | 61 ++- docs/roadmap_status.md | 31 +- scripts/test_lightweight_agent_gates.py | 147 +++-- 14 files changed, 1710 insertions(+), 141 deletions(-) create mode 100644 .ci/test-impact/impact_map.json create mode 100644 .ci/test-impact/run_selected_tests.py create mode 100644 .ci/test-impact/validate_workflow_jobs.py create mode 100644 backend/scripts/test_impact_selection.py create mode 100644 backend/tests/test_ci_impact_selection.py diff --git a/.ci/test-impact/impact_map.json b/.ci/test-impact/impact_map.json new file mode 100644 index 000000000..48537d383 --- /dev/null +++ b/.ci/test-impact/impact_map.json @@ -0,0 +1,24 @@ +{ + "schema_version": 1, + "commitrail_test_modules": [ + "tests/projects/review_policy/test_semantics.py" + ], + "owners": [ + { + "source_paths": [ + "backend/app/core/s3_validation.py" + ], + "test_paths": [ + "backend/tests/test_config.py", + "backend/tests/test_artifact_store_conformance.py", + "backend/tests/test_s3_artifact_store.py" + ], + "test_modules": [ + "tests/test_config.py", + "tests/test_artifact_store_conformance.py", + "tests/test_s3_artifact_store.py" + ], + "infrastructure": "minio" + } + ] +} diff --git a/.ci/test-impact/run_selected_tests.py b/.ci/test-impact/run_selected_tests.py new file mode 100644 index 000000000..806bf97b7 --- /dev/null +++ b/.ci/test-impact/run_selected_tests.py @@ -0,0 +1,343 @@ +#!/usr/bin/env python3 +"""Execute and attest one exact impact selection without database services.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import time +from typing import Any + +BACKEND = Path(__file__).resolve().parents[2] / "backend" +ROOT = BACKEND.parent +sys.path.insert(0, str(BACKEND)) + +from scripts.test_impact_selection import ( # noqa: E402 + MAP_RELATIVE_PATH, + SelectionError, + _inventory, + classify_paths, +) + + +def _canonical_json(value: Any) -> bytes: + return (json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n").encode() + + +def _hash_bytes(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def _read_manifest(path: Path) -> dict[str, Any]: + if path.is_symlink() or not path.is_file(): + raise SelectionError("missing_selection_manifest") + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise SelectionError("invalid_selection_manifest") from exc + if not isinstance(value, dict): + raise SelectionError("invalid_selection_manifest") + return value + + +def _digest(path: Path) -> str: + if path.is_symlink() or not path.is_file(): + raise SelectionError("missing_bound_input") + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def verify_selection(manifest: dict[str, Any], *, expected_job: str) -> list[str]: + """Recompute every candidate-side binding before collecting tests.""" + current_sha = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip() + current_tree = subprocess.check_output( + ["git", "rev-parse", "HEAD^{tree}"], cwd=ROOT, text=True + ).strip() + status = subprocess.check_output( + ["git", "status", "--porcelain"], cwd=ROOT, text=True + ) + if ( + current_sha != manifest.get("execution_sha") + or current_tree != manifest.get("execution_tree") + or status + or os.environ.get("GITHUB_SHA") != current_sha + or os.environ.get("GITHUB_EVENT_PULL_REQUEST_BASE_SHA") != manifest.get("base_sha") + or os.environ.get("GITHUB_EVENT_PULL_REQUEST_HEAD_SHA") != manifest.get("head_sha") + ): + raise SelectionError("candidate_target_mismatch") + base_sha = str(manifest.get("base_sha", "")) + head_sha = str(manifest.get("head_sha", "")) + execution_parents = subprocess.check_output( + ["git", "show", "-s", "--format=%P", current_sha], cwd=ROOT, text=True + ).strip().split() + if execution_parents != [base_sha, head_sha]: + raise SelectionError("execution_parent_mismatch") + merge_base = subprocess.check_output( + ["git", "merge-base", base_sha, head_sha], cwd=ROOT, text=True + ).strip() + if merge_base != manifest.get("merge_base_sha"): + raise SelectionError("merge_base_mismatch") + raw_paths = subprocess.check_output( + ["git", "diff", "--name-only", "-z", "--no-renames", merge_base, head_sha, "--"], + cwd=ROOT, + ) + try: + actual_changed_paths = sorted( + path.decode("utf-8", errors="strict") for path in raw_paths.split(b"\0") if path + ) + except UnicodeDecodeError as exc: + raise SelectionError("invalid_changed_path_encoding") from exc + if actual_changed_paths != manifest.get("changed_paths"): + raise SelectionError("changed_path_set_mismatch") + if manifest.get("mode") not in {"impact", "pure"}: + raise SelectionError("nonselective_manifest") + expected_jobs = manifest.get("expected_jobs") + if not isinstance(expected_jobs, list) or expected_job not in expected_jobs: + raise SelectionError("unexpected_impact_job") + + map_path = ROOT / MAP_RELATIVE_PATH + selector_path = BACKEND / "scripts/test_impact_selection.py" + runner_path = Path(__file__).resolve() + if _digest(map_path) != manifest.get("impact_map_sha256"): + raise SelectionError("impact_map_drift") + if _digest(selector_path) != manifest.get("selector_sha256"): + raise SelectionError("selector_drift") + if _digest(runner_path) != manifest.get("impact_runner_sha256"): + raise SelectionError("runner_drift") + + changed_paths = manifest.get("changed_paths") + if ( + not isinstance(changed_paths, list) + or any(not isinstance(path, str) for path in changed_paths) + or hashlib.sha256(_canonical_json(changed_paths)).hexdigest() + != manifest.get("changed_paths_sha256") + ): + raise SelectionError("changed_path_digest_mismatch") + impact_map = json.loads(map_path.read_text(encoding="utf-8")) + mode, modules, jobs, profile = classify_paths(changed_paths, impact_map) + if ( + mode != manifest.get("mode") + or modules != manifest.get("selected_modules") + or jobs != manifest.get("expected_jobs") + or profile != manifest.get("infrastructure_profile") + ): + raise SelectionError("classification_drift") + + inventory, inventory_digest = _inventory(ROOT, current_sha) + if ( + inventory != manifest.get("test_inventory") + or inventory_digest != manifest.get("test_inventory_sha256") + ): + raise SelectionError("test_inventory_drift") + if _digest(Path(__file__)) != manifest.get("impact_runner_sha256"): + raise SelectionError("runner_drift") + return modules + + +def validate_evidence( + manifest_path: Path, + evidence_path: Path, + *, + expected_job: str, + expected_manifest_sha256: str, +) -> None: + """Verify the selected test artifact against the selection job output.""" + manifest = _read_manifest(manifest_path) + if _digest(manifest_path) != expected_manifest_sha256: + raise SelectionError("selection_artifact_digest_mismatch") + evidence = _read_manifest(evidence_path) + expected_path = evidence_path.parent / "expected.json" + expected = _read_manifest(expected_path) + nodes = evidence.get("selected_nodes") + completed = evidence.get("completed_nodes") + observed = evidence.get("observed_collected_nodes") + elapsed = evidence.get("elapsed_seconds") + if ( + evidence.get("job") != expected_job + or evidence.get("execution_sha") != manifest.get("execution_sha") + or evidence.get("execution_tree") != manifest.get("execution_tree") + or evidence.get("test_inventory_sha256") != manifest.get("test_inventory_sha256") + or evidence.get("selection_manifest_sha256") != _hash_bytes( + _canonical_json(manifest) + ) + or evidence.get("exit_code") != 0 + or not isinstance(nodes, list) + or not nodes + or any(not isinstance(node, str) for node in nodes) + or nodes != sorted(set(nodes)) + or completed != nodes + or observed != nodes + or evidence.get("skipped_nodes") != [] + or evidence.get("deselected_nodes") != [] + or evidence.get("selected_modules") != manifest.get("selected_modules") + or evidence.get("expected_payload_sha256") != _digest(expected_path) + or expected.get("execution_sha") != manifest.get("execution_sha") + or expected.get("execution_tree") != manifest.get("execution_tree") + or expected.get("selected_modules") != manifest.get("selected_modules") + or expected.get("selected_nodes") != nodes + or expected.get("selected_nodes_sha256") != _hash_bytes(_canonical_json(nodes)) + or expected.get("selection_manifest_sha256") + != _hash_bytes(_canonical_json(manifest)) + or evidence.get("expected_nodes_sha256") != _hash_bytes(_canonical_json(nodes)) + or evidence.get("completed_nodes_sha256") != _hash_bytes(_canonical_json(completed)) + or isinstance(elapsed, bool) + or not isinstance(elapsed, (int, float)) + or elapsed < 0 + ): + raise SelectionError("impact_evidence_incomplete") + + +def run_selected(manifest: dict[str, Any], output: Path, *, expected_job: str) -> int: + """Collect exact target nodes, execute them, and require complete custody.""" + from scripts.run_test_lanes import ( + COLLECTED_ENV, + COMPLETED_ENV, + DESELECTED_ENV, + HEAD_ENV, + SKIPPED_ENV, + _read_nodes, + collect_nodes, + ) + + modules = verify_selection(manifest, expected_job=expected_job) + if output.exists() or output.is_symlink(): + raise SelectionError("impact_output_exists") + output.mkdir(parents=True, mode=0o700) + collection_dir = output / "collection" + collection_dir.mkdir(mode=0o700) + started = time.monotonic() + tree_sha = str(manifest["execution_sha"]) + collection_code, nodes, deselected = collect_nodes( + tuple(modules), collection_dir, tree_sha + ) + if collection_code != 0 or deselected or not nodes: + raise SelectionError("impact_collection_failed") + + expected_bytes = _canonical_json( + { + "execution_sha": tree_sha, + "execution_tree": manifest["execution_tree"], + "selected_modules": modules, + "selected_nodes": nodes, + "selected_nodes_sha256": hashlib.sha256(_canonical_json(nodes)).hexdigest(), + "selection_manifest_sha256": hashlib.sha256( + _canonical_json(manifest) + ).hexdigest(), + } + ) + expected_path = output / "expected.json" + expected_path.write_bytes(expected_bytes) + + collected_path = output / "collected.jsonl" + completed_path = output / "completed.jsonl" + skipped_path = output / "skipped.jsonl" + deselected_path = output / "deselected.jsonl" + for path in (collected_path, completed_path, skipped_path, deselected_path): + path.touch(mode=0o600, exist_ok=False) + coverage_path = output / ".coverage" + env = os.environ.copy() + env.update( + { + "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1", + "PYTHONPATH": os.pathsep.join( + value for value in (str(BACKEND), env.get("PYTHONPATH", "")) if value + ), + "COVERAGE_FILE": str(coverage_path), + COLLECTED_ENV: str(collected_path), + COMPLETED_ENV: str(completed_path), + SKIPPED_ENV: str(skipped_path), + DESELECTED_ENV: str(deselected_path), + HEAD_ENV: tree_sha, + } + ) + command = [ + sys.executable, + "-m", + "pytest", + "-q", + "-p", + "pytest_asyncio.plugin", + "-p", + "pytest_cov.plugin", + "-p", + "scripts.run_test_lanes", + "--cov=app", + "--cov-report=", + *nodes, + ] + result = subprocess.run(command, cwd=BACKEND, env=env, check=False) + completed = _read_nodes(completed_path, allow_empty=True) + observed_collected = _read_nodes(collected_path, allow_empty=True) + skipped = _read_nodes(skipped_path, allow_empty=True) + run_deselected = _read_nodes(deselected_path, allow_empty=True) + elapsed = round(time.monotonic() - started, 3) + evidence = { + "completed_nodes": sorted(completed), + "completed_nodes_sha256": hashlib.sha256( + _canonical_json(sorted(completed)) + ).hexdigest(), + "expected_nodes_sha256": hashlib.sha256(_canonical_json(nodes)).hexdigest(), + "expected_payload_sha256": hashlib.sha256(expected_bytes).hexdigest(), + "execution_sha": tree_sha, + "execution_tree": manifest["execution_tree"], + "exit_code": result.returncode, + "job": expected_job, + "observed_collected_nodes": sorted(observed_collected), + "selected_modules": modules, + "selected_nodes": nodes, + "skipped_nodes": skipped, + "deselected_nodes": sorted(set(deselected + run_deselected)), + "elapsed_seconds": elapsed, + "selection_manifest_sha256": hashlib.sha256( + _canonical_json(manifest) + ).hexdigest(), + "test_inventory_sha256": manifest["test_inventory_sha256"], + } + evidence_path = output / "evidence.json" + evidence_path.write_bytes(_canonical_json(evidence)) + if ( + result.returncode != 0 + or sorted(observed_collected) != nodes + or sorted(completed) != nodes + or skipped + or run_deselected + ): + print("impact test custody incomplete", file=sys.stderr) + return 1 + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--manifest", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--expected-job", required=True, choices=("impact-s3", "impact-pure")) + parser.add_argument("--validate-only", action="store_true") + parser.add_argument("--evidence", type=Path) + parser.add_argument("--expected-manifest-sha256") + args = parser.parse_args() + try: + if args.validate_only: + if args.evidence is None or args.expected_manifest_sha256 is None: + raise SelectionError("missing_validation_input") + validate_evidence( + args.manifest, + args.evidence, + expected_job=args.expected_job, + expected_manifest_sha256=args.expected_manifest_sha256, + ) + return 0 + return run_selected( + _read_manifest(args.manifest), args.output, expected_job=args.expected_job + ) + except (OSError, subprocess.SubprocessError, RuntimeError, SelectionError) as exc: + print(f"impact test execution failed: {exc}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.ci/test-impact/validate_workflow_jobs.py b/.ci/test-impact/validate_workflow_jobs.py new file mode 100644 index 000000000..235a63260 --- /dev/null +++ b/.ci/test-impact/validate_workflow_jobs.py @@ -0,0 +1,81 @@ +#!/usr/bin/env python3 +"""Require the exact job inventory declared by the trusted selector.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path +import re +import sys +from typing import Any + +FULL = { + "impact-selection", + "auth-boundary-preflight", + "minio-image", + "lanes", + "full-api-e2e", +} +S3 = {"impact-selection", "auth-boundary-preflight", "minio-image", "impact-s3"} +PURE = {"impact-selection", "auth-boundary-preflight", "impact-pure"} +KNOWN = FULL | S3 | PURE +DIGEST_RE = re.compile(r"^[0-9a-f]{64}$") + + +def validate(manifest: dict[str, Any], results: dict[str, str]) -> None: + expected_by_mode = { + ("full", "full"): FULL, + ("impact", "minio"): S3, + ("pure", "none"): PURE, + } + mode_profile = (manifest.get("mode"), manifest.get("infrastructure_profile")) + expected = expected_by_mode.get(mode_profile) + manifest_jobs = manifest.get("expected_jobs") + if ( + manifest.get("schema_version") != 1 + or expected is None + or not isinstance(manifest_jobs, list) + or set(manifest_jobs) != expected + or len(manifest_jobs) != len(expected) + or set(results) != KNOWN + ): + raise ValueError("invalid_expected_job_inventory") + for job, result in results.items(): + if job in expected and result != "success": + raise ValueError(f"expected_job_not_success:{job}") + if job not in expected and result != "skipped": + raise ValueError(f"unexpected_job_ran:{job}") + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--manifest", required=True, type=Path) + parser.add_argument("--manifest-sha256", required=True) + parser.add_argument("--results-json", required=True) + args = parser.parse_args() + try: + if args.manifest.is_symlink() or not args.manifest.is_file(): + raise ValueError("missing_selection_manifest") + if DIGEST_RE.fullmatch(args.manifest_sha256) is None: + raise ValueError("invalid_selection_manifest_digest") + raw = args.manifest.read_bytes() + if hashlib.sha256(raw).hexdigest() != args.manifest_sha256: + raise ValueError("selection_manifest_digest_mismatch") + manifest = json.loads(raw) + results = json.loads(args.results_json) + if not isinstance(manifest, dict) or not isinstance(results, dict) or any( + not isinstance(job, str) or not isinstance(result, str) + for job, result in results.items() + ): + raise ValueError("invalid_job_result_manifest") + validate(manifest, results) + return 0 + except (OSError, UnicodeDecodeError, json.JSONDecodeError, ValueError) as exc: + print(f"backend job fan-in rejected: {exc}", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index 247c9938f..cb414d25e 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -101,7 +101,10 @@ Changes within any of the three mapped test modules run the complete three- - For a PR, compute changed paths from explicit event `base.sha` and `head.sha`, using their Git merge-base; do not read PR descriptions as authority. Run the selector/map from the trusted base revision, not from PR - code. A changed selector, map or workflow itself always selects full suite. + code. Because the first rollout candidate has no trusted selector on its base, + that candidate uses a fixed full-suite manifest and the complete required job + set; it never invokes its new selector to authorize selective execution. A + changed selector, map or workflow itself always selects full suite. - Preserve the complete changed-path list and digest in the evidence manifest. For backend-impact classification only, remove paths under `.commitrail/` from the source-impact input: every bounded PR carries its own durable diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index 9cd448bd3..9f4bb1d50 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -2,9 +2,9 @@ name: Backend on: pull_request: - push: - branches: - - main + schedule: + - cron: "19 3 * * *" + workflow_dispatch: concurrency: group: backend-${{ github.event.pull_request.number || github.ref }} @@ -18,7 +18,121 @@ env: MINIO_IMAGE: workstream-minio:source jobs: + impact-selection: + runs-on: ubuntu-latest + timeout-minutes: 5 + outputs: + mode: ${{ steps.select.outputs.mode }} + profile: ${{ steps.select.outputs.profile }} + manifest_sha256: ${{ steps.select.outputs.manifest_sha256 }} + started_at: ${{ steps.timing.outputs.started_at }} + steps: + - id: timing + name: Start Backend workflow timing + run: echo "started_at=$(date +%s)" >> "${GITHUB_OUTPUT}" + + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 + with: + ref: ${{ github.sha }} + path: candidate + fetch-depth: 0 + persist-credentials: false + + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 + with: + ref: ${{ github.event.pull_request.base.sha || github.sha }} + path: trusted + fetch-depth: 0 + persist-credentials: false + + - name: Select exact tests using the trusted base map + id: select + env: + EVENT_NAME: ${{ github.event_name }} + EVENT_BASE_SHA: ${{ github.event.pull_request.base.sha || github.sha }} + EVENT_HEAD_SHA: ${{ github.event.pull_request.head.sha || github.sha }} + EXECUTION_SHA: ${{ github.sha }} + shell: bash + run: | + set -euo pipefail + manifest="${RUNNER_TEMP}/backend-impact/selection.json" + mkdir -p "$(dirname "${manifest}")" + args=( + --trusted-root "${GITHUB_WORKSPACE}/trusted" + --candidate-root "${GITHUB_WORKSPACE}/candidate" + --base-sha "${EVENT_BASE_SHA}" + --head-sha "${EVENT_HEAD_SHA}" + --execution-sha "${EXECUTION_SHA}" + --output "${manifest}" + ) + if [[ "${EVENT_NAME}" != pull_request ]]; then + args+=(--force-full) + fi + if [[ -f trusted/backend/scripts/test_impact_selection.py \ + && -f trusted/.ci/test-impact/impact_map.json \ + && -f trusted/.ci/test-impact/run_selected_tests.py ]]; then + python3 trusted/backend/scripts/test_impact_selection.py "${args[@]}" + else + # Bootstrap this workflow change with a static full-suite manifest. + # The trusted base has no selector yet, so no selective mode is possible. + test "${EVENT_NAME}" = pull_request + test "$(git -C candidate rev-parse HEAD)" = "${EXECUTION_SHA}" + test -z "$(git -C candidate status --porcelain)" + python3 - "${manifest}" <<'PY' + import hashlib + import json + import os + from pathlib import Path + import subprocess + import sys + + candidate = Path("candidate") + execution = os.environ["EXECUTION_SHA"] + tree = subprocess.check_output( + ["git", "-C", str(candidate), "rev-parse", f"{execution}^{{tree}}"], + text=True, + ).strip() + payload = { + "schema_version": 1, + "mode": "full", + "infrastructure_profile": "full", + "base_sha": os.environ["EVENT_BASE_SHA"], + "head_sha": os.environ["EVENT_HEAD_SHA"], + "execution_sha": execution, + "execution_tree": tree, + "selected_modules": [], + "expected_jobs": [ + "impact-selection", + "auth-boundary-preflight", + "minio-image", + "lanes", + "full-api-e2e", + ], + } + encoded = (json.dumps(payload, sort_keys=True, separators=(",", ":")) + "\n").encode() + target = Path(sys.argv[1]) + target.write_bytes(encoded) + with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: + output.write("mode=full\n") + output.write("profile=full\n") + output.write(f"manifest_sha256={hashlib.sha256(encoded).hexdigest()}\n") + PY + fi + + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} + path: ${{ runner.temp }}/backend-impact/selection.json + if-no-files-found: error + retention-days: 7 + minio-image: + needs: impact-selection + if: >- + ${{ + needs.impact-selection.outputs.mode == 'full' || + (needs.impact-selection.outputs.mode == 'impact' && needs.impact-selection.outputs.profile == 'minio') + }} runs-on: ubuntu-latest timeout-minutes: 20 outputs: @@ -40,8 +154,8 @@ jobs: id: cache uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 with: - path: ${{ runner.temp }}/minio-image/minio.tar - key: minio-source-v1-${{ github.sha }}-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('docker/minio/**') }} + path: ${{ runner.temp }}/minio-image/ + key: minio-source-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('docker/minio/**') }} - name: Build source-pinned MinIO once if: steps.cache.outputs.cache-hit != 'true' @@ -53,26 +167,42 @@ jobs: docker save --output "${RUNNER_TEMP}/minio-image/minio.tar" "${MINIO_IMAGE}" - name: Verify the cached or freshly built provider + env: + MINIO_CACHE_HIT: ${{ steps.cache.outputs.cache-hit }} shell: bash run: | set -euo pipefail - docker load --input "${RUNNER_TEMP}/minio-image/minio.tar" - docker run --rm "${MINIO_IMAGE}" --version - docker run --detach --rm --name minio-build-probe \ - --publish 127.0.0.1:9000:9000 \ - --env MINIO_ROOT_USER=workstream-minio \ - --env MINIO_ROOT_PASSWORD=workstream-minio-secret-key \ - "${MINIO_IMAGE}" server /data --address :9000 - trap 'docker logs minio-build-probe; docker stop minio-build-probe' EXIT - for attempt in $(seq 1 60); do - if curl --fail --silent http://127.0.0.1:9000/minio/health/live >/dev/null; then - cd "${RUNNER_TEMP}/minio-image" - sha256sum minio.tar > minio.tar.sha256 - exit 0 + verify_provider() { + if [[ "${MINIO_CACHE_HIT}" == true ]]; then + (cd "${RUNNER_TEMP}/minio-image" && sha256sum --check minio.tar.sha256) || return 1 fi - sleep 1 - done - exit 1 + docker load --input "${RUNNER_TEMP}/minio-image/minio.tar" || return 1 + docker run --rm "${MINIO_IMAGE}" --version || return 1 + docker run --detach --rm --name minio-build-probe \ + --publish 127.0.0.1:9000:9000 \ + --env MINIO_ROOT_USER=workstream-minio \ + --env MINIO_ROOT_PASSWORD=workstream-minio-secret-key \ + "${MINIO_IMAGE}" server /data --address :9000 || return 1 + for attempt in $(seq 1 60); do + if curl --fail --silent http://127.0.0.1:9000/minio/health/live >/dev/null; then + docker stop minio-build-probe || return 1 + return 0 + fi + sleep 1 + done + docker logs minio-build-probe + docker stop minio-build-probe + return 1 + } + if ! verify_provider; then + rm -f "${RUNNER_TEMP}/minio-image/minio.tar" \ + "${RUNNER_TEMP}/minio-image/minio.tar.sha256" + docker build --tag "${MINIO_IMAGE}" docker/minio + docker save --output "${RUNNER_TEMP}/minio-image/minio.tar" "${MINIO_IMAGE}" + verify_provider + fi + cd "${RUNNER_TEMP}/minio-image" + sha256sum minio.tar > minio.tar.sha256 - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 with: @@ -118,16 +248,10 @@ jobs: shell: bash run: | set -euo pipefail - ruff check \ - app/modules/authorization/api \ - scripts/authorization_boundary.py \ - scripts/module_boundaries.py \ - scripts/test_structure_boundary.py \ - tests/architecture/test_authorization_boundary.py \ - tests/architecture/test_module_boundaries.py \ - tests/architecture/test_test_structure_boundary.py + ruff check app tests scripts + docstr-coverage --config .docstr.yaml python -m scripts.module_boundaries validate \ - --protected-base "${{ github.event.pull_request.base.sha || github.event.before }}" + --protected-base "${{ github.event.pull_request.base.sha || github.sha }}" PYTEST_DISABLE_PLUGIN_AUTOLOAD=1 python -m pytest -q \ -p pytest_asyncio.plugin \ tests/architecture/test_module_boundaries.py \ @@ -140,7 +264,8 @@ jobs: python -m scripts.behavior_ownership validate lanes: - needs: minio-image + needs: [impact-selection, minio-image] + if: ${{ needs.impact-selection.outputs.mode == 'full' }} runs-on: ubuntu-latest timeout-minutes: 45 strategy: @@ -214,14 +339,6 @@ jobs: python -m pip install ruff==0.15.22 test "$(ruff --version)" = "ruff 0.15.22" - - name: Lint - working-directory: backend - run: ruff check app tests scripts - - - name: Docstring coverage - working-directory: backend - run: docstr-coverage --config .docstr.yaml - - name: Download source-pinned MinIO image uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 with: @@ -315,9 +432,9 @@ jobs: if-no-files-found: error retention-days: 7 - test: - if: ${{ always() }} - needs: [auth-boundary-preflight, lanes, minio-image] + full-api-e2e: + needs: [impact-selection, minio-image] + if: ${{ needs.impact-selection.outputs.mode == 'full' }} runs-on: ubuntu-latest timeout-minutes: 30 @@ -355,17 +472,13 @@ jobs: with: python-version: "3.12" - - id: identity - name: Bind exact checked-out tree + - name: Bind exact API-test candidate shell: bash run: | set -euo pipefail - job_start_epoch="$(date +%s)" - tree_sha="$(git rev-parse HEAD)" - test "${tree_sha}" = "${GITHUB_SHA}" + test "$(git rev-parse HEAD)" = "${GITHUB_SHA}" test -z "$(git status --porcelain)" - echo "job_start_epoch=${job_start_epoch}" >> "${GITHUB_OUTPUT}" - echo "tree_sha=${tree_sha}" >> "${GITHUB_OUTPUT}" + test "$(git rev-parse HEAD^{tree})" = "$(git rev-parse "${GITHUB_SHA}^{tree}")" - name: Install backend working-directory: backend @@ -397,22 +510,215 @@ jobs: docker logs workstream-minio exit 1 - - name: Download separate lane attempts from this run + - name: API contract real API e2e + working-directory: backend + env: + WORKSTREAM_TEST_BROKER_URL: redis://localhost:6380/0 + WORKSTREAM_TEST_ADMIN_DATABASE_URL: postgresql+asyncpg://workstream:workstream@localhost:5433/postgres + WORKSTREAM_TEST_MINIO_ENDPOINT: http://127.0.0.1:9000 + run: >- + python scripts/run_isolated_tests.py + --metadata-json "${RUNNER_TEMP}/api-database.json" + --timeout-seconds 1500 + -- python scripts/api_contract_e2e.py + + impact-pure: + needs: impact-selection + if: ${{ needs.impact-selection.outputs.mode == 'pure' }} + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 + with: + persist-credentials: false + fetch-depth: 0 + + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 + with: + python-version: "3.12" + + - name: Install backend + working-directory: backend + run: python -m pip install -e ".[dev,agents]" + + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} + path: ${{ runner.temp }}/backend-impact + + - name: Run selected policy tests with exact custody + env: + GITHUB_EVENT_PULL_REQUEST_BASE_SHA: ${{ github.event.pull_request.base.sha }} + GITHUB_EVENT_PULL_REQUEST_HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: >- + python3 .ci/test-impact/run_selected_tests.py + --manifest "${RUNNER_TEMP}/backend-impact/selection.json" + --output "${RUNNER_TEMP}/backend-impact/evidence" + --expected-job impact-pure + + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 if: ${{ always() }} + with: + name: backend-impact-evidence-${{ github.sha }}-impact-pure-attempt-${{ github.run_attempt }} + path: ${{ runner.temp }}/backend-impact/evidence/ + include-hidden-files: true + if-no-files-found: warn + retention-days: 7 + + impact-s3: + needs: [impact-selection, minio-image] + if: ${{ needs.impact-selection.outputs.mode == 'impact' }} + runs-on: ubuntu-latest + timeout-minutes: 25 + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 + with: + persist-credentials: false + fetch-depth: 0 + + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 + with: + python-version: "3.12" + + - name: Bind exact S3-impact candidate + shell: bash + run: | + set -euo pipefail + test "$(git rev-parse HEAD)" = "${GITHUB_SHA}" + test -z "$(git status --porcelain)" + test "$(git rev-parse HEAD^{tree})" = "$(git rev-parse "${GITHUB_SHA}^{tree}")" + + - name: Install backend + working-directory: backend + run: python -m pip install -e ".[dev,agents]" + + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} + path: ${{ runner.temp }}/backend-impact + + - name: Download source-pinned MinIO image uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 with: - pattern: backend-lane-${{ github.sha }}-*-attempt-* - path: backend/.ci/download - merge-multiple: false + name: ${{ needs.minio-image.outputs.artifact }} + path: ${{ runner.temp }}/minio-image + + - name: Start real MinIO adapter provider + shell: bash + run: | + set -euo pipefail + (cd "${RUNNER_TEMP}/minio-image" && sha256sum --check minio.tar.sha256) + docker load --input "${RUNNER_TEMP}/minio-image/minio.tar" + docker run --detach --rm --name workstream-minio \ + --publish 127.0.0.1:9000:9000 \ + --env MINIO_ROOT_USER=workstream-minio \ + --env MINIO_ROOT_PASSWORD=workstream-minio-secret-key \ + "${MINIO_IMAGE}" server /data --address :9000 + for attempt in $(seq 1 60); do + if curl --fail --silent http://127.0.0.1:9000/minio/health/live >/dev/null; then + exit 0 + fi + sleep 1 + done + docker logs workstream-minio + exit 1 - - name: Require preflight and every semantic lane + - name: Run selected S3 and Commitrail tests with exact custody + working-directory: backend + env: + GITHUB_EVENT_PULL_REQUEST_BASE_SHA: ${{ github.event.pull_request.base.sha }} + GITHUB_EVENT_PULL_REQUEST_HEAD_SHA: ${{ github.event.pull_request.head.sha }} + WORKSTREAM_TEST_MINIO_ENDPOINT: http://127.0.0.1:9000 + run: >- + python3 ../.ci/test-impact/run_selected_tests.py + --manifest "${RUNNER_TEMP}/backend-impact/selection.json" + --output "${RUNNER_TEMP}/backend-impact/evidence" + --expected-job impact-s3 + + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 if: ${{ always() }} + with: + name: backend-impact-evidence-${{ github.sha }}-impact-s3-attempt-${{ github.run_attempt }} + path: ${{ runner.temp }}/backend-impact/evidence/ + include-hidden-files: true + if-no-files-found: warn + retention-days: 7 + + test: + if: ${{ always() }} + needs: [impact-selection, auth-boundary-preflight, minio-image, lanes, full-api-e2e, impact-pure, impact-s3] + runs-on: ubuntu-latest + timeout-minutes: 30 + + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 + with: + persist-credentials: false + fetch-depth: 0 + + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 + with: + python-version: "3.12" + + - id: identity + name: Bind exact checked-out tree + shell: bash + run: | + set -euo pipefail + job_start_epoch="$(date +%s)" + tree_sha="$(git rev-parse HEAD)" + test "${tree_sha}" = "${GITHUB_SHA}" + test -z "$(git status --porcelain)" + echo "job_start_epoch=${job_start_epoch}" >> "${GITHUB_OUTPUT}" + echo "tree_sha=${tree_sha}" >> "${GITHUB_OUTPUT}" + + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} + path: .ci/impact-selection + + - name: Validate the exact expected GitHub job inventory env: - PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }} + SELECTION_SHA256: ${{ needs.impact-selection.outputs.manifest_sha256 }} + IMPACT_SELECTION_RESULT: ${{ needs.impact-selection.result }} + AUTH_BOUNDARY_PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }} + MINIO_IMAGE_RESULT: ${{ needs.minio-image.result }} LANES_RESULT: ${{ needs.lanes.result }} - run: test "${PREFLIGHT_RESULT}" = success && test "${LANES_RESULT}" = success + FULL_API_E2E_RESULT: ${{ needs.full-api-e2e.result }} + IMPACT_PURE_RESULT: ${{ needs.impact-pure.result }} + IMPACT_S3_RESULT: ${{ needs.impact-s3.result }} + shell: bash + run: | + set -euo pipefail + results="$(python3 - <<'PY' + import json + import os + print(json.dumps({ + "impact-selection": os.environ["IMPACT_SELECTION_RESULT"], + "auth-boundary-preflight": os.environ["AUTH_BOUNDARY_PREFLIGHT_RESULT"], + "minio-image": os.environ["MINIO_IMAGE_RESULT"], + "lanes": os.environ["LANES_RESULT"], + "full-api-e2e": os.environ["FULL_API_E2E_RESULT"], + "impact-pure": os.environ["IMPACT_PURE_RESULT"], + "impact-s3": os.environ["IMPACT_S3_RESULT"], + })) + PY + )" + python3 .ci/test-impact/validate_workflow_jobs.py \ + --manifest .ci/impact-selection/selection.json \ + --manifest-sha256 "${SELECTION_SHA256}" \ + --results-json "${results}" + + - name: Download separate lane attempts from this run + if: ${{ needs.impact-selection.outputs.mode == 'full' }} + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + pattern: backend-lane-${{ github.sha }}-*-attempt-* + path: backend/.ci/download + merge-multiple: false - name: Merge and independently validate exact lane custody + if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend shell: bash run: | @@ -429,6 +735,7 @@ jobs: --summary-json .ci/test-lanes/run-summary.json - name: Combine semantic-lane coverage exactly once + if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend shell: bash run: | @@ -447,26 +754,22 @@ jobs: done coverage combine - - name: API contract real API e2e + - name: Install coverage tooling for the complete suite + if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend - env: - WORKSTREAM_TEST_BROKER_URL: redis://localhost:6380/0 - WORKSTREAM_TEST_ADMIN_DATABASE_URL: postgresql+asyncpg://workstream:workstream@localhost:5433/postgres - WORKSTREAM_TEST_MINIO_ENDPOINT: http://127.0.0.1:9000 - run: >- - python scripts/run_isolated_tests.py - --metadata-json "${RUNNER_TEMP}/api-database.json" - --timeout-seconds 1500 - -- python scripts/api_contract_e2e.py + run: python -m pip install -e ".[dev,agents]" - name: Backend coverage diagnostics (no percentage gate) + if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend run: coverage report --precision=2 - name: Record hosted timing, complete execution and diagnostic coverage + if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend env: EXPECTED_HEAD_SHA: ${{ steps.identity.outputs.tree_sha }} + BACKEND_STARTED_AT: ${{ needs.impact-selection.outputs.started_at }} shell: bash run: | set -euo pipefail @@ -593,10 +896,13 @@ jobs: if not timing_value.isdigit(): raise SystemExit("invalid lane start timing") start_epochs.append(int(timing_value)) - start_epoch = min(start_epochs) - total_wall = time.time() - start_epoch + backend_started_at = int(os.environ["BACKEND_STARTED_AT"]) + first_lane_start_delay = min(start_epochs) - backend_started_at + total_wall = time.time() - backend_started_at if not math.isfinite(total_wall) or total_wall < 0: raise SystemExit("invalid Backend hosted wall time") + if first_lane_start_delay < 0: + raise SystemExit("invalid first lane start timing") totals = coverage.get("totals") percent = totals.get("percent_covered") if isinstance(totals, dict) else None if ( @@ -615,6 +921,7 @@ jobs: "global_coverage_percent": float(percent), "global_coverage_sha256": digest(coverage_path), "head_sha": expected_head, + "first_lane_start_delay_seconds": first_lane_start_delay, "run_summary_sha256": digest(summary_path), "slowest_lane_seconds": float(slowest), "timing_target_met": total_wall <= 480, @@ -625,6 +932,74 @@ jobs: ) PY + - name: Download exact impact evidence + if: ${{ needs.impact-selection.outputs.mode == 'pure' }} + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: backend-impact-evidence-${{ github.sha }}-impact-pure-attempt-${{ github.run_attempt }} + path: .ci/impact-evidence + + - name: Validate pure impact evidence + if: ${{ needs.impact-selection.outputs.mode == 'pure' }} + env: + SELECTION_SHA256: ${{ needs.impact-selection.outputs.manifest_sha256 }} + run: >- + python3 .ci/test-impact/run_selected_tests.py + --validate-only + --manifest .ci/impact-selection/selection.json + --evidence .ci/impact-evidence/evidence.json + --expected-job impact-pure + --expected-manifest-sha256 "${SELECTION_SHA256}" + + - name: Download exact S3 impact evidence + if: ${{ needs.impact-selection.outputs.mode == 'impact' }} + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: backend-impact-evidence-${{ github.sha }}-impact-s3-attempt-${{ github.run_attempt }} + path: .ci/impact-evidence + + - name: Validate S3 impact evidence + if: ${{ needs.impact-selection.outputs.mode == 'impact' }} + env: + SELECTION_SHA256: ${{ needs.impact-selection.outputs.manifest_sha256 }} + run: >- + python3 .ci/test-impact/run_selected_tests.py + --validate-only + --manifest .ci/impact-selection/selection.json + --evidence .ci/impact-evidence/evidence.json + --expected-job impact-s3 + --expected-manifest-sha256 "${SELECTION_SHA256}" + + - name: Record selective workflow wall time + if: ${{ needs.impact-selection.outputs.mode != 'full' }} + env: + BACKEND_STARTED_AT: ${{ needs.impact-selection.outputs.started_at }} + IMPACT_MODE: ${{ needs.impact-selection.outputs.mode }} + run: | + set -euo pipefail + python3 - <<'PY' + import json + import os + from pathlib import Path + import time + + evidence_path = Path(".ci/impact-evidence/evidence.json") + evidence = json.loads(evidence_path.read_text(encoding="utf-8")) + elapsed = time.time() - int(os.environ["BACKEND_STARTED_AT"]) + if elapsed < 0: + raise SystemExit("invalid Backend workflow timing") + output = { + "head_sha": evidence["execution_sha"], + "mode": os.environ["IMPACT_MODE"], + "selected_node_count": len(evidence["selected_nodes"]), + "timing_target_met": elapsed <= 300, + "total_backend_wall_seconds": round(elapsed, 3), + } + evidence_path.with_name("hosted-evidence.json").write_text( + json.dumps(output, indent=2, sort_keys=True) + "\n", encoding="utf-8" + ) + PY + - name: Reassert exact tree custody if: ${{ always() }} shell: bash @@ -641,6 +1016,8 @@ jobs: path: | backend/.ci/download/** backend/.ci/test-lanes/** + .ci/impact-selection/** + .ci/impact-evidence/** include-hidden-files: true if-no-files-found: warn retention-days: 7 diff --git a/AGENTS.md b/AGENTS.md index d301736d0..369dd21c2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -183,8 +183,14 @@ definition or ownership boundary of Workstream. one, identify its behavior and retained proof, or explain why the requirement is obsolete. Remove duplicated assertions and implementation-only tests only when no distinct contract or failure mode is lost. Do not replace the suite - with end-to-end-only tests, skip failures, or set a deletion quota. Full-suite - completeness and real integration checks remain blocking. + with end-to-end-only tests, skip failures, or set a deletion quota. Backend + CI uses a trusted-base, reviewed source-to-test map: an eligible mapped change + must run its complete enumerated consumer closure with its required real + infrastructure; every unmapped, broad, uncertain, or CI-selection change + runs the complete suite. Full-suite and real integration proof remain + blocking for fallback PRs and scheduled/manual audits. Nightly results never + substitute for exact-PR evidence. See the current CI boundary in + `.commitrail/initiatives/WS-CI-006/OVERVIEW.md`. ## Done Criteria diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f237e24cc..cce8a90d0 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -97,8 +97,14 @@ active queue or approval gate. - Explain the goal, scope, non-goals, and important design decisions. - Keep the change small enough to review. -- Run the relevant behavior tests, lint and type checks; full hosted suites and - real API drills remain required. Coverage is diagnostic, not a merge threshold. +- Run the relevant behavior tests, lint and type checks. The required hosted + Backend check uses a reviewed source-to-test map: mapped changes run the + complete named consumer closure with its required real services; every + unmapped, broad, uncertain, or CI-selection change falls back to all nine + semantic lanes and the real API integration job. The complete suite also + runs on its scheduled audit and manual dispatch. A scheduled result never + substitutes for checks on your exact PR candidate. Coverage is diagnostic, + not a merge threshold. - Preserve security defaults and meaningful negative-path proof. Do not add tests to meet a percentage or count. For test removals/consolidations, name the retained behavioral proof or the retired requirement. Keep real database, diff --git a/backend/scripts/test_impact_selection.py b/backend/scripts/test_impact_selection.py new file mode 100644 index 000000000..c00f90d70 --- /dev/null +++ b/backend/scripts/test_impact_selection.py @@ -0,0 +1,309 @@ +#!/usr/bin/env python3 +"""Build a trusted, exact-target backend test-impact selection manifest.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess +import sys +from typing import Any + +SHA_RE = re.compile(r"^[0-9a-f]{40}$") +MAP_RELATIVE_PATH = ".ci/test-impact/impact_map.json" +RUNNER_RELATIVE_PATH = ".ci/test-impact/run_selected_tests.py" +FULL_LANES = [ + "shared_foundations_a", + "shared_foundations_b", + "schema_contracts", + "project_lifecycle_a", + "project_lifecycle_b", + "project_lifecycle_c", + "task_lifecycle_a", + "task_lifecycle_b", + "task_lifecycle_c", +] +POLICY_MODULE = "tests/projects/review_policy/test_semantics.py" +FULL_JOBS = ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"] + + +def _full() -> tuple[str, list[str], list[str], str]: + return "full", [], list(FULL_JOBS), "full" + + +class SelectionError(RuntimeError): + """The candidate cannot be safely classified for selective test execution.""" + + +def _canonical(value: Any) -> bytes: + return (json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n").encode() + + +def _sha256(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def _git(repository: Path, *args: str) -> bytes: + try: + return subprocess.run( + ["git", *args], + cwd=repository, + check=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ).stdout + except (OSError, subprocess.CalledProcessError) as exc: + raise SelectionError("git_object_or_command_unavailable") from exc + + +def classify_paths( + changed_paths: list[str], + impact_map: dict[str, Any], +) -> tuple[str, list[str], list[str], str]: + """Return mode, exact modules, expected jobs, and infrastructure profile.""" + if not changed_paths or len(set(changed_paths)) != len(changed_paths): + return _full() + if any( + not path + or path.startswith("/") + or "\\" in path + or any(part in {"", ".", ".."} for part in path.split("/")) + for path in changed_paths + ): + return _full() + + commitrail_paths = [path for path in changed_paths if path.startswith(".commitrail/")] + source_paths = [path for path in changed_paths if path not in commitrail_paths] + commitrail_modules = impact_map.get("commitrail_test_modules") + if ( + not isinstance(commitrail_modules, list) + or not commitrail_modules + or any(not isinstance(module, str) for module in commitrail_modules) + ): + return _full() + + if not source_paths: + if not commitrail_paths: + return _full() + return ( + "pure", + sorted(set(commitrail_modules)), + ["impact-selection", "auth-boundary-preflight", "impact-pure"], + "none", + ) + + owners = impact_map.get("owners") + if not isinstance(owners, list): + return _full() + for owner in owners: + if not isinstance(owner, dict): + continue + sources = owner.get("source_paths") + tests = owner.get("test_paths") + modules = owner.get("test_modules") + infrastructure = owner.get("infrastructure") + if ( + not isinstance(sources, list) + or not isinstance(tests, list) + or not isinstance(modules, list) + or not sources + or not modules + or any(not isinstance(path, str) for path in (*sources, *tests)) + or any(not isinstance(module, str) for module in modules) + or infrastructure not in {"minio", "none"} + or len(set(sources)) != len(sources) + or len(set(tests)) != len(tests) + or len(set(modules)) != len(modules) + or any(not path.startswith("backend/tests/") for path in tests) + or sorted(modules) + != sorted(path.removeprefix("backend/") for path in tests) + ): + continue + permitted_paths = set(sources) | set(tests) + if set(source_paths) <= permitted_paths and set(source_paths) & permitted_paths: + selected_modules = set(modules) + if commitrail_paths: + selected_modules.update(commitrail_modules) + profile = "minio" if infrastructure == "minio" else "none" + job = "impact-s3" if profile == "minio" else "impact-pure" + jobs = ["impact-selection", "auth-boundary-preflight"] + if profile == "minio": + jobs.append("minio-image") + jobs.append(job) + return "impact", sorted(selected_modules), jobs, profile + return _full() + + +def _inventory(repository: Path, execution_sha: str) -> tuple[list[dict[str, str]], str]: + raw = _git(repository, "ls-tree", "-r", "-z", execution_sha, "--", "backend/tests") + inventory: list[dict[str, str]] = [] + for entry in raw.split(b"\0"): + if not entry: + continue + try: + metadata, raw_path = entry.split(b"\t", 1) + mode, kind, blob = metadata.decode("ascii").split(" ") + path = raw_path.decode("utf-8", errors="strict") + except (ValueError, UnicodeDecodeError) as exc: + raise SelectionError("invalid_test_inventory") from exc + name = path.rsplit("/", 1)[-1] + if kind == "blob" and name.startswith("test_") and name.endswith(".py"): + inventory.append({"blob": blob, "path": path}) + inventory.sort(key=lambda row: row["path"]) + if not inventory: + raise SelectionError("empty_test_inventory") + return inventory, _sha256(_canonical(inventory)) + + +def build_manifest( + trusted_root: Path, + candidate_root: Path, + *, + base_sha: str, + head_sha: str, + execution_sha: str, + force_full: bool = False, +) -> dict[str, Any]: + """Bind selection to event commits, merge candidate, trusted map and tests.""" + if any(SHA_RE.fullmatch(value) is None for value in (base_sha, head_sha, execution_sha)): + raise SelectionError("invalid_event_sha") + candidate_head = _git(candidate_root, "rev-parse", "HEAD").decode().strip() + candidate_tree = _git(candidate_root, "rev-parse", "HEAD^{tree}").decode().strip() + candidate_status = _git(candidate_root, "status", "--porcelain").decode() + if candidate_head != execution_sha or candidate_status: + raise SelectionError("candidate_checkout_mismatch") + for value in (base_sha, head_sha, execution_sha): + _git(candidate_root, "cat-file", "-e", f"{value}^{{commit}}") + execution_tree = _git(candidate_root, "rev-parse", f"{execution_sha}^{{tree}}").decode().strip() + if candidate_tree != execution_tree: + raise SelectionError("candidate_tree_mismatch") + if force_full: + if not base_sha == head_sha == execution_sha: + raise SelectionError("invalid_forced_full_target") + merge_base = execution_sha + changed_raw = b"" + else: + parents = ( + _git(candidate_root, "show", "-s", "--format=%P", execution_sha) + .decode() + .strip() + .split() + ) + if parents != [base_sha, head_sha]: + raise SelectionError("execution_parent_mismatch") + merge_base = _git(candidate_root, "merge-base", base_sha, head_sha).decode().strip() + if SHA_RE.fullmatch(merge_base) is None: + raise SelectionError("invalid_merge_base") + changed_raw = _git( + candidate_root, + "diff", + "--name-only", + "-z", + "--no-renames", + merge_base, + head_sha, + "--", + ) + try: + changed_paths = sorted( + path.decode("utf-8", errors="strict") + for path in changed_raw.split(b"\0") + if path + ) + except UnicodeDecodeError as exc: + raise SelectionError("invalid_changed_path_encoding") from exc + + map_path = trusted_root / MAP_RELATIVE_PATH + if map_path.is_symlink() or not map_path.is_file(): + raise SelectionError("missing_trusted_impact_map") + map_bytes = map_path.read_bytes() + try: + impact_map = json.loads(map_bytes) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise SelectionError("invalid_trusted_impact_map") from exc + if not isinstance(impact_map, dict) or impact_map.get("schema_version") != 1: + raise SelectionError("invalid_trusted_impact_map") + + mode, modules, jobs, profile = ( + _full() if force_full else classify_paths(changed_paths, impact_map) + ) + inventory, inventory_digest = _inventory(candidate_root, execution_sha) + known_modules = {f"{Path(row['path']).relative_to('backend')}" for row in inventory} + expected_paths: set[str] = set() + for owner in impact_map.get("owners", []): + if isinstance(owner, dict): + expected_paths.update(owner.get("test_modules", [])) + expected_paths.update(impact_map.get("commitrail_test_modules", [])) + if any(module not in known_modules for module in modules): + mode, modules, jobs, profile = _full() + if mode != "full" and any(module not in expected_paths for module in modules): + mode, modules, jobs, profile = _full() + + trusted_script = trusted_root / "backend/scripts/test_impact_selection.py" + if trusted_script.is_symlink() or not trusted_script.is_file(): + raise SelectionError("missing_trusted_selector") + runner_path = trusted_root / RUNNER_RELATIVE_PATH + if runner_path.is_symlink() or not runner_path.is_file(): + raise SelectionError("missing_trusted_runner") + return { + "base_sha": base_sha, + "changed_paths": changed_paths, + "changed_paths_sha256": _sha256(_canonical(changed_paths)), + "execution_sha": execution_sha, + "execution_tree": execution_tree, + "expected_jobs": jobs, + "head_sha": head_sha, + "impact_map_sha256": _sha256(map_bytes), + "infrastructure_profile": profile, + "merge_base_sha": merge_base, + "mode": mode, + "schema_version": 1, + "selected_modules": modules, + "selector_sha256": _sha256(trusted_script.read_bytes()), + "impact_runner_sha256": _sha256(runner_path.read_bytes()), + "test_inventory": inventory, + "test_inventory_sha256": inventory_digest, + } + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--trusted-root", required=True, type=Path) + parser.add_argument("--candidate-root", required=True, type=Path) + parser.add_argument("--base-sha", required=True) + parser.add_argument("--head-sha", required=True) + parser.add_argument("--execution-sha", required=True) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--force-full", action="store_true") + args = parser.parse_args() + try: + manifest = build_manifest( + args.trusted_root.resolve(strict=True), + args.candidate_root.resolve(strict=True), + base_sha=args.base_sha, + head_sha=args.head_sha, + execution_sha=args.execution_sha, + force_full=args.force_full, + ) + args.output.parent.mkdir(parents=True, exist_ok=True) + manifest_bytes = _canonical(manifest) + args.output.write_bytes(manifest_bytes) + github_output = os.environ.get("GITHUB_OUTPUT") + if github_output: + with Path(github_output).open("a", encoding="utf-8") as output: + output.write(f"mode={manifest['mode']}\n") + output.write(f"profile={manifest['infrastructure_profile']}\n") + output.write(f"manifest_sha256={_sha256(manifest_bytes)}\n") + print(json.dumps({key: manifest[key] for key in ("mode", "expected_jobs", "infrastructure_profile")})) + return 0 + except (OSError, SelectionError, subprocess.SubprocessError) as exc: + print(f"impact selection failed closed: {exc}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/backend/scripts/test_lane_catalogue.py b/backend/scripts/test_lane_catalogue.py index 916d0b485..fe20de231 100644 --- a/backend/scripts/test_lane_catalogue.py +++ b/backend/scripts/test_lane_catalogue.py @@ -74,6 +74,7 @@ class TestLane: "tests/test_artifacts.py", "tests/test_assertion_helpers.py", "tests/test_aws_credential_isolation.py", + "tests/test_ci_impact_selection.py", "tests/test_ci_test_lanes.py", "tests/test_ci_lane_catalogue.py", "tests/test_config.py", diff --git a/backend/tests/test_ci_impact_selection.py b/backend/tests/test_ci_impact_selection.py new file mode 100644 index 000000000..70e2e39f7 --- /dev/null +++ b/backend/tests/test_ci_impact_selection.py @@ -0,0 +1,299 @@ +"""Fail-closed tests for the reviewed backend impact classifier.""" + +from __future__ import annotations + +import json +import importlib.util +import hashlib +import subprocess +from pathlib import Path + +import pytest + +from scripts.test_impact_selection import ( + MAP_RELATIVE_PATH, + SelectionError, + build_manifest, + classify_paths, +) + + +ROOT = Path(__file__).resolve().parents[2] +RUN_SELECTED_PATH = ROOT / ".ci/test-impact/run_selected_tests.py" +IMPACT_MAP = json.loads((ROOT / MAP_RELATIVE_PATH).read_text(encoding="utf-8")) +S3_TEST_PATHS = [ + "backend/tests/test_config.py", + "backend/tests/test_artifact_store_conformance.py", + "backend/tests/test_s3_artifact_store.py", +] +S3_TEST_MODULES = [path.removeprefix("backend/") for path in S3_TEST_PATHS] +POLICY_PATHS = [ + ".commitrail/initiatives/WS-ARCH-001/planning/chunks/WS-ARCH-001-CP07-project-guide-policy-binding.md", + ".commitrail/initiatives/WS-AUTH-001/planning/chunks/WS-AUTH-001-12H-guide-activation.md", +] +POLICY_MODULE = "tests/projects/review_policy/test_semantics.py" + + +_RUN_SELECTED_SPEC = importlib.util.spec_from_file_location( + "ci_run_selected_tests", RUN_SELECTED_PATH +) +assert _RUN_SELECTED_SPEC is not None and _RUN_SELECTED_SPEC.loader is not None +_RUN_SELECTED = importlib.util.module_from_spec(_RUN_SELECTED_SPEC) +_RUN_SELECTED_SPEC.loader.exec_module(_RUN_SELECTED) + + +@pytest.mark.parametrize("path", POLICY_PATHS) +def test_committrail_policy_inputs_select_the_exact_backend_consumer(path: str) -> None: + mode, modules, jobs, profile = classify_paths([path], IMPACT_MAP) + + assert mode == "pure" + assert modules == [POLICY_MODULE] + assert jobs == ["impact-selection", "auth-boundary-preflight", "impact-pure"] + assert profile == "none" + + +def test_unrelated_committrail_metadata_is_still_nonempty() -> None: + mode, modules, jobs, profile = classify_paths( + [".commitrail/changes/some-new-record.md"], IMPACT_MAP + ) + + assert (mode, modules, jobs, profile) == ( + "pure", + [POLICY_MODULE], + ["impact-selection", "auth-boundary-preflight", "impact-pure"], + "none", + ) + + +def test_committrail_policy_test_is_unioned_with_mapped_s3_closure() -> None: + mode, modules, jobs, profile = classify_paths( + ["backend/app/core/s3_validation.py", ".commitrail/changes/change.md"], + IMPACT_MAP, + ) + + assert mode == "impact" + assert modules == sorted([*S3_TEST_MODULES, POLICY_MODULE]) + assert jobs == ["impact-selection", "auth-boundary-preflight", "minio-image", "impact-s3"] + assert profile == "minio" + + +def test_s3_owner_or_mapped_test_edits_select_the_whole_owner_closure() -> None: + for path in ["backend/app/core/s3_validation.py", *S3_TEST_PATHS]: + mode, modules, jobs, profile = classify_paths([path], IMPACT_MAP) + assert mode == "impact" + assert modules == sorted(S3_TEST_MODULES) + assert jobs == ["impact-selection", "auth-boundary-preflight", "minio-image", "impact-s3"] + assert profile == "minio" + + +def test_incomplete_owner_test_closure_falls_back_to_full() -> None: + incomplete_map = json.loads(json.dumps(IMPACT_MAP)) + incomplete_map["owners"][0]["test_modules"] = incomplete_map["owners"][0][ + "test_modules" + ][:-1] + + assert classify_paths(["backend/app/core/s3_validation.py"], incomplete_map) == ( + "full", + [], + ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"], + "full", + ) + + +@pytest.mark.parametrize( + "path", + [ + "backend/app/core/config.py", + "backend/tests/conftest.py", + "backend/tests/test_unmapped.py", + "backend/scripts/test_lane_catalogue.py", + "backend/scripts/test_impact_selection.py", + ".ci/test-impact/impact_map.json", + ".ci/test-impact/run_selected_tests.py", + ".github/workflows/backend.yml", + "docs/roadmap_status.md", + "AGENTS.md", + ], +) +def test_unmapped_or_shared_changes_fail_closed_to_full_suite(path: str) -> None: + assert classify_paths(["backend/app/core/s3_validation.py", path], IMPACT_MAP) == ( + "full", + [], + ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"], + "full", + ) + + +@pytest.mark.parametrize("paths", [[], ["../escape"], ["/absolute"], ["a\\b"]]) +def test_empty_or_malformed_changes_never_produce_empty_green_selection( + paths: list[str], +) -> None: + mode, modules, jobs, profile = classify_paths(paths, IMPACT_MAP) + assert (mode, modules, jobs, profile) == ( + "full", + [], + ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"], + "full", + ) + + +def test_impact_fan_in_requires_predeclared_exact_nodes_and_completion(tmp_path: Path) -> None: + manifest = { + "execution_sha": "a" * 40, + "execution_tree": "b" * 40, + "selected_modules": ["tests/test_config.py"], + "test_inventory_sha256": "c" * 64, + } + nodes = ["tests/test_config.py::test_one"] + + def canonical(value: object) -> bytes: + return (json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n").encode() + + selection_path = tmp_path / "selection.json" + selection_bytes = canonical(manifest) + selection_path.write_bytes(selection_bytes) + payload = { + "execution_sha": manifest["execution_sha"], + "execution_tree": manifest["execution_tree"], + "selected_modules": manifest["selected_modules"], + "selected_nodes": nodes, + "selected_nodes_sha256": hashlib.sha256(canonical(nodes)).hexdigest(), + "selection_manifest_sha256": hashlib.sha256(selection_bytes).hexdigest(), + } + evidence = { + "job": "impact-pure", + "execution_sha": manifest["execution_sha"], + "execution_tree": manifest["execution_tree"], + "test_inventory_sha256": manifest["test_inventory_sha256"], + "selection_manifest_sha256": hashlib.sha256(selection_bytes).hexdigest(), + "exit_code": 0, + "selected_nodes": nodes, + "completed_nodes": nodes, + "observed_collected_nodes": nodes, + "skipped_nodes": [], + "deselected_nodes": [], + "selected_modules": manifest["selected_modules"], + "expected_nodes_sha256": hashlib.sha256(canonical(nodes)).hexdigest(), + "completed_nodes_sha256": hashlib.sha256(canonical(nodes)).hexdigest(), + "elapsed_seconds": 1.0, + "expected_payload_sha256": hashlib.sha256(canonical(payload)).hexdigest(), + } + expected_path = tmp_path / "expected.json" + expected_path.write_bytes(canonical(payload)) + evidence_path = tmp_path / "evidence.json" + evidence_path.write_bytes(canonical(evidence)) + manifest_digest = hashlib.sha256(selection_bytes).hexdigest() + + _RUN_SELECTED.validate_evidence( + selection_path, + evidence_path, + expected_job="impact-pure", + expected_manifest_sha256=manifest_digest, + ) + + evidence_path.write_bytes(canonical({**evidence, "completed_nodes": []})) + with pytest.raises(SelectionError, match="impact_evidence_incomplete"): + _RUN_SELECTED.validate_evidence( + selection_path, + evidence_path, + expected_job="impact-pure", + expected_manifest_sha256=manifest_digest, + ) + + +def test_execution_candidate_must_be_the_event_merge_commit(tmp_path: Path) -> None: + repository = tmp_path / "candidate" + repository.mkdir() + _git(repository, "init", "-q", "-b", "main") + _git(repository, "config", "user.email", "ci@example.invalid") + _git(repository, "config", "user.name", "CI test") + paths = [ + "backend/app/core/s3_validation.py", + *S3_TEST_PATHS, + "backend/tests/projects/review_policy/test_semantics.py", + ] + for path in paths: + target = repository / path + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text("# inventory fixture\n", encoding="utf-8") + _git(repository, "add", "backend") + _git(repository, "commit", "-q", "-m", "base") + base = _git(repository, "rev-parse", "HEAD") + source = repository / "backend/app/core/s3_validation.py" + source.write_text("# changed source\n", encoding="utf-8") + _git(repository, "add", "backend/app/core/s3_validation.py") + _git(repository, "commit", "-q", "-m", "change") + head = _git(repository, "rev-parse", "HEAD") + tree = _git(repository, "rev-parse", "HEAD^{tree}") + execution = subprocess.run( + ["git", "commit-tree", tree, "-p", base, "-p", head], + cwd=repository, + check=True, + text=True, + stdout=subprocess.PIPE, + ).stdout.strip() + _git(repository, "reset", "--hard", execution) + + manifest = build_manifest( + ROOT, + repository, + base_sha=base, + head_sha=head, + execution_sha=execution, + ) + assert manifest["merge_base_sha"] == base + assert manifest["execution_sha"] == execution + assert manifest["selected_modules"] == sorted(S3_TEST_MODULES) + assert manifest["expected_jobs"] == [ + "impact-selection", + "auth-boundary-preflight", + "minio-image", + "impact-s3", + ] + forced_full = build_manifest( + ROOT, + repository, + base_sha=execution, + head_sha=execution, + execution_sha=execution, + force_full=True, + ) + assert forced_full["mode"] == "full" + assert forced_full["selected_modules"] == [] + assert set(forced_full["expected_jobs"]) == { + "impact-selection", + "auth-boundary-preflight", + "minio-image", + "lanes", + "full-api-e2e", + } + _git(repository, "reset", "--hard", head) + with pytest.raises(SelectionError, match="candidate_checkout_mismatch"): + build_manifest( + ROOT, + repository, + base_sha=base, + head_sha=head, + execution_sha=execution, + ) + _git(repository, "reset", "--hard", head) + with pytest.raises(SelectionError, match="execution_parent_mismatch"): + build_manifest( + ROOT, + repository, + base_sha=base, + head_sha=head, + execution_sha=head, + ) + _git(repository, "reset", "--hard", execution) + + +def _git(repository: Path, *args: str) -> str: + return subprocess.run( + ["git", *args], + cwd=repository, + check=True, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ).stdout.strip() diff --git a/docker/minio/README.md b/docker/minio/README.md index fc2dad15e..acc4256d1 100644 --- a/docker/minio/README.md +++ b/docker/minio/README.md @@ -28,14 +28,15 @@ requires network access and Go compilation resources; do not launch it on an already memory-constrained workstation. Subsequent builds reuse Docker layers. -Backend CI builds or restores one image cache keyed by the exact Git commit, -this directory's contents and runner platform. Older PR commits cannot supply a -cached executable to a new commit; retries of the same commit can reuse its -image. CI verifies server startup, then supplies a checksummed image -artifact to the existing lanes and aggregate job. Jobs never substitute a mock -storage provider. A missing build, artifact or health check fails verification. -The source-image artifact is independent of test/coverage evidence and cannot -make a failed test lane pass. +Backend CI builds or restores one image cache keyed by the complete +`docker/minio/` source-input digest and runner platform, independent of Git +commit. Unchanged pinned source can therefore be reused by later commits. CI +verifies the archive checksum, server version and live health before supplying +a checksummed image artifact to the full-suite or selected S3 test job. An +invalid cache is rebuilt from pinned source and verified again. Jobs never +substitute a mock storage provider. A missing build, artifact or health check +fails verification. The source-image artifact is independent of test evidence +and cannot make a failed test pass. When updating upstream source, update the commit, archive checksum, provenance and relevant build pins together. Require a fresh image build, health check and diff --git a/docs/operations_backend_testing.md b/docs/operations_backend_testing.md index ab85af6f4..540dd15ef 100644 --- a/docs/operations_backend_testing.md +++ b/docs/operations_backend_testing.md @@ -108,19 +108,51 @@ creation must use the real admission-backed command, not this fixture. If provisioning fails, confirm the local PostgreSQL provisioning credential can create/drop databases and roles, terminate owned sessions, and reach the named admin database. Diagnostics omit credentials. -## Hosted semantic-lane full-suite proof - -The required GitHub check remains `Backend / test`. Nine matrix jobs each own a -digest-pinned PostgreSQL service container, a pinned-source MinIO image, -and exactly one dependency lane. A step-level curl health loop admits MinIO -before collection. This is semantic fan-out, not arbitrary test-count sharding: -lane ownership remains repository-defined and exact. - -The explicit inventory lives in `backend/scripts/test_lane_catalogue.py`. -Authorization preflight runs alongside the nine lanes. The final `test` job -requires both preflight and every lane to succeed before validating evidence and -coverage; failed, cancelled or skipped prerequisites remain blocking. This saves -serial waiting on valid changes at the cost of lane work when preflight fails. +## Hosted Backend checks and complete-suite proof + +The required GitHub check remains `Backend / test` on every pull request. The +trusted-base selector in `backend/scripts/test_impact_selection.py` reads the +reviewed map at `.ci/test-impact/impact_map.json` and binds the exact base, head, +merge candidate/tree, changed paths, test inventory, selected modules/nodes and +expected jobs. The map initially permits only +`backend/app/core/s3_validation.py` and its complete configuration, +provider-neutral namespace-conformance and real MinIO adapter test closure. +Every Commitrail change also runs the complete backend policy-semantics module +that reads Commitrail planning inputs. A Commitrail-only change runs that module +plus the always-required authorization/static preflight. + +The initial rollout PR predates the trusted selector on its base revision, so +that one candidate emits a fixed full-suite manifest and runs the complete +required job set; it does not use its new selector for selective execution. + +Any unknown, broad, unclassified, stale, malformed or CI-selection change falls +back to the complete suite. Changes to other documentation, fixtures, tests, +schemas, dependencies or product modules are not implicitly ignored. The full +mode uses all nine semantic lanes plus real PostgreSQL-backed API integration; +it runs on every fallback PR and as a scheduled and manual audit. Nightly/manual +evidence is tied to its own main head and never substitutes for exact-PR tests. +No workflow path filters may hide the required check. + +Full-suite semantic lanes are defined in `backend/scripts/test_lane_catalogue.py`. +Each lane owns its declared test inventory, a digest-pinned PostgreSQL service, +and the shared pinned-source MinIO image. Selected impact jobs bypass PostgreSQL +only for the explicitly mapped DB-free closure; S3 behavior still uses live +MinIO. A service-free final `test` job validates the selector's exact expected +job set and the complete node/job evidence. Missing, duplicate, skipped, +deselected, mismatched or incomplete evidence fails closed. + +Authorization-boundary preflight runs in every mode and includes repository-wide +lint, docstring and module/test-structure checks. The nine matrix jobs no longer +repeat lint or docstring work. Full mode combines the nine lane artifacts once +for diagnostic coverage and runs the real API integration separately with its +own PostgreSQL, Redis and MinIO services. A full Backend run is not repeated on +each protected `main` push: strict, up-to-date PR checks and a human approval +protect merges, while the scheduled and manual runs audit the complete suite. + +The selector, workflow, map, evidence validator or test-catalogue tooling are +not trusted to select themselves: changing any of them forces full mode. A new +source-to-test relation may be mapped only after its complete consumers and +required infrastructure are proven; otherwise it remains in the full fallback. Assertion-map validation analyzes each exact historical revision/module once per invocation, then checks every referenced node and assertion against that analysis. It does not cache current source or reuse analysis across validation calls. @@ -130,7 +162,8 @@ to reduce ephemeral reset I/O. A runtime guard verifies the mount, capacity, data directory and enabled `fsync`, `full_page_writes` and `synchronous_commit` before tests. Real SQL, transaction, lock, isolation, and full hosted behavior checks remain. -The schema-contract lane and aggregate job retain disk-backed databases. +The schema-contract lane retains disk-backed storage. The final evidence +aggregator is service-free. This is not a production configuration or proof of host-power-loss durability: [Docker tmpfs data disappears when the container stops](https://docs.docker.com/engine/storage/tmpfs/). An exhausted mount fails the job; it does not silently change storage or skip tests. diff --git a/docs/roadmap_status.md b/docs/roadmap_status.md index 1adeda6f5..0105090a0 100644 --- a/docs/roadmap_status.md +++ b/docs/roadmap_status.md @@ -189,17 +189,23 @@ cannot be reused as post-submission review-gate evidence. See the - Cross-module behavior is moving through explicit public ports under the modular-monolith boundary. New private edges are prohibited and touched debt is reduced incrementally. -- GitHub CI distributes the backend suite across semantic lanes, rejects - skipped/deselected tests and requires behavior, boundary and real API proof. - Coverage is diagnostic only, with no percentage gate or test-count target. - Redundant coverage-only reruns are removed; their tests remain in full-suite lanes. - Its nine-lane allocation uses three project lanes, three task lanes, two - shared-foundation lanes and one schema lane. Database resets batch trigger - commands within the existing transaction while retaining full schema checks; - authorization preflight runs alongside lanes and remains mandatory at fan-in. - Ordinary CI databases use bounded private RAM-backed storage with write - settings verified; schema-contract and aggregate databases stay disk-backed. - Hosted runtime remains measured rather than guaranteed. +- GitHub CI keeps the required Backend `test` check on every PR and uses a + trusted-base, reviewed source-to-test map. The only initial application + mapping is `backend/app/core/s3_validation.py` to its complete configuration, + provider-neutral namespace-conformance and real MinIO adapter test closure; + Commitrail changes also run the backend policy-semantics test that reads its + planning inputs. The exact candidate manifest binds base, head, merge tree, + changed paths, map, test inventory, selected nodes and expected jobs. Unknown, + broad, or CI-selection changes run the full suite. Full mode retains all nine + semantic lanes, real PostgreSQL and API integration proof; it also runs on + scheduled and manual audits, which never attest a different PR. Selected + tests reject skips, deselections and incomplete evidence. Coverage is + diagnostic only, with no percentage or test-count gate. Redundant + coverage-only reruns are removed without deleting their behavior tests. + Lint/docstring and authorization-boundary preflight run once and remain + mandatory at fan-in. Ordinary CI databases use bounded private RAM-backed + storage with write settings verified; schema-contract storage remains + disk-backed. Hosted runtime is measured, not guaranteed. ### Identity and authorization @@ -516,7 +522,8 @@ the remaining service-actor, profile/link and other AUTH families or the full suite audit. Remaining work includes those AUTH families and the TASK, CHECKER, ART, CON, REV, and tooling audit. The audit requires behavioral proof, not only file splitting or coverage percentages. Real PostgreSQL, concurrency, storage, -and full hosted behavior/integration checks remain required. Product +and complete fallback/nightly hosted behavior checks remain required; a mapped +PR must pass its entire exact-hosted owner closure instead. Product implementation is already progressing alongside this audit with separate file ownership. diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index 2a3d88d08..8fb530139 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -2,9 +2,12 @@ from __future__ import annotations +import json +import hashlib import os import re import subprocess +import sys import tempfile import textwrap import unittest @@ -121,6 +124,21 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertNotIn("pull_request_review:", workflow) self.assertIn("cancel-in-progress: true", workflow) + self.assertCountEqual( + re.findall( + r"(?m)^ ([a-z][a-z0-9-]*):\s*$", workflow.split("\njobs:\n", 1)[1] + ), + [ + "impact-selection", + "minio-image", + "auth-boundary-preflight", + "lanes", + "full-api-e2e", + "impact-pure", + "impact-s3", + "test", + ], + ) self.assertEqual(len(re.findall(r"(?m)^ matrix:$", workflow)), 1) matrix = re.search(r"(?m)^ matrix:\n((?: {8,}[^\n]*\n|\n)+)", workflow) self.assertIsNotNone(matrix) @@ -137,11 +155,8 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None " - task_lifecycle_b\n" " - task_lifecycle_c", ) - self.assertIn( - " test:\n if: ${{ always() }}\n" - " needs: [auth-boundary-preflight, lanes, minio-image]", workflow - ) - self.assertIn("Require preflight and every semantic lane", workflow) + self.assertIn(" needs: [impact-selection, auth-boundary-preflight, minio-image, lanes, full-api-e2e, impact-pure, impact-s3]", workflow) + self.assertIn("Validate the exact expected GitHub job inventory", workflow) self.assertIn("python -m scripts.merge_test_lane_evidence", workflow) self.assertIn("scripts/validate_test_lane_evidence.py", workflow) self.assertIn( @@ -151,6 +166,16 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertIn("include-hidden-files: true", workflow) self.assertIn("coverage report --precision=2", workflow) self.assertNotIn("fail-under", workflow) + self.assertIn("test_impact_selection.py", workflow) + self.assertIn("backend-impact-selection-", workflow) + self.assertIn("validate_workflow_jobs.py", workflow) + self.assertIn("Bootstrap this workflow change with a static full-suite manifest.", workflow) + self.assertIn('test "${EVENT_NAME}" = pull_request', workflow) + self.assertIn('"mode": "full"', workflow) + self.assertIn("full-api-e2e", workflow) + self.assertNotIn("push:\n branches:\n - main", workflow) + self.assertIn("schedule:", workflow) + self.assertIn("workflow_dispatch:", workflow) self.assertNotIn("pull_request_review:", agent_gates) self.assertNotIn("--require-pr-approval", agent_gates) self.assertNotIn("pull-requests:", agent_gates) @@ -185,13 +210,15 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None def test_minio_source_image_is_built_once_and_shared_without_bypassing_lanes(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") image_job = workflow.split("\n minio-image:\n", 1)[1].split("\n auth-boundary-preflight:\n", 1)[0] - self.assertEqual(workflow.count('docker build --tag "${MINIO_IMAGE}" docker/minio'), 1) + self.assertEqual(workflow.count('docker build --tag "${MINIO_IMAGE}" docker/minio'), 2) + self.assertIn("if ! verify_provider; then", image_job) self.assertNotIn("quay.io/minio", workflow) self.assertIn("hashFiles('docker/minio/**')", image_job) self.assertIn( - "key: minio-source-v1-${{ github.sha }}-${{ runner.os }}-${{ runner.arch }}-", + "key: minio-source-v2-${{ runner.os }}-${{ runner.arch }}-", image_job, ) + self.assertNotIn("${{ github.sha }}", image_job.split("key:", 1)[1].splitlines()[0]) self.assertNotIn("restore-keys:", image_job) self.assertIn("minio-source-${GITHUB_SHA}-${GITHUB_RUN_ATTEMPT}", image_job) self.assertIn("artifact: ${{ steps.identity.outputs.artifact }}", image_job) @@ -199,52 +226,104 @@ def test_minio_source_image_is_built_once_and_shared_without_bypassing_lanes(sel self.assertIn("/minio/health/live", image_job) self.assertIn("if-no-files-found: error", image_job) self.assertNotIn("continue-on-error", image_job) - for name, end in (("lanes", "test"), ("test", None)): + for name, end in (("lanes", "full-api-e2e"), ("full-api-e2e", "impact-pure"), ("impact-s3", "test")): job = workflow.split(f"\n {name}:\n", 1)[1] if end: job = job.split(f"\n {end}:\n", 1)[0] - self.assertIn("name: ${{ needs.minio-image.outputs.artifact }}", job) - self.assertIn("sha256sum --check minio.tar.sha256", job) - self.assertIn('docker load --input "${RUNNER_TEMP}/minio-image/minio.tar"', job) - self.assertIn('"${MINIO_IMAGE}" server /data --address :9000', job) + if name in {"lanes", "full-api-e2e", "impact-s3"}: + self.assertIn("name: ${{ needs.minio-image.outputs.artifact }}", job) + self.assertIn("sha256sum --check minio.tar.sha256", job) + self.assertIn('docker load --input "${RUNNER_TEMP}/minio-image/minio.tar"', job) + self.assertIn('"${MINIO_IMAGE}" server /data --address :9000', job) def test_parallel_preflight_and_lanes_fail_closed_at_fan_in(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") lanes = workflow.split("\n lanes:\n", 1)[1].split("\n test:\n", 1)[0] - self.assertRegex(lanes, r"(?m)^ needs: minio-image$") + self.assertRegex(lanes, r"(?m)^ needs: \[impact-selection, minio-image\]$") self.assertNotIn("needs: auth-boundary-preflight", lanes) - step = workflow.split( - " - name: Require preflight and every semantic lane\n", 1 - )[1].split("\n - name:", 1)[0] - self.assertIn("if: ${{ always() }}", step) - self.assertIn("PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }}", step) - self.assertIn("LANES_RESULT: ${{ needs.lanes.result }}", step) - guard = re.search(r"(?m)^ run: (.+)$", step) - self.assertIsNotNone(guard) - for preflight in ("success", "failure", "cancelled", "skipped", "", "unknown"): - for lanes_result in ("success", "failure", "cancelled", "skipped", "", "unknown"): - with self.subTest(preflight=preflight, lanes=lanes_result): - result = subprocess.run( - ["bash", "-e", "-c", guard[1]], - env={"PREFLIGHT_RESULT": preflight, "LANES_RESULT": lanes_result}, - capture_output=True, - check=False, - ) - self.assertEqual( - result.returncode == 0, - preflight == lanes_result == "success", - ) + self.assertIn("if: ${{ always() }}", workflow.split("\n test:\n", 1)[1]) + self.assertIn("needs.auth-boundary-preflight.result", workflow) + self.assertIn("needs.lanes.result", workflow) + self.assertIn("needs.impact-s3.result", workflow) + self.assertIn("needs.impact-pure.result", workflow) + + def test_impact_job_inventory_validator_rejects_missing_or_unexpected_jobs(self) -> None: + validator = Path(".ci/test-impact/validate_workflow_jobs.py") + manifest = { + "schema_version": 1, + "mode": "impact", + "infrastructure_profile": "minio", + "expected_jobs": [ + "impact-selection", + "auth-boundary-preflight", + "minio-image", + "impact-s3", + ], + } + statuses = { + "impact-selection": "success", + "auth-boundary-preflight": "success", + "minio-image": "success", + "lanes": "skipped", + "full-api-e2e": "skipped", + "impact-pure": "skipped", + "impact-s3": "success", + } + with tempfile.TemporaryDirectory() as directory: + manifest_path = Path(directory) / "selection.json" + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + digest = hashlib.sha256(manifest_path.read_bytes()).hexdigest() + results_json = json.dumps(statuses) + valid = subprocess.run( + [ + sys.executable, + str(validator), + "--manifest", + str(manifest_path), + "--manifest-sha256", + digest, + "--results-json", + results_json, + ], + check=False, + capture_output=True, + text=True, + ) + self.assertEqual(valid.returncode, 0, valid.stderr) + + for job, status in (("impact-s3", "skipped"), ("lanes", "success")): + invalid_statuses = {**statuses, job: status} + invalid = subprocess.run( + [ + sys.executable, + str(validator), + "--manifest", + str(manifest_path), + "--manifest-sha256", + digest, + "--results-json", + json.dumps(invalid_statuses), + ], + check=False, + capture_output=True, + text=True, + ) + self.assertNotEqual(invalid.returncode, 0) def test_postgres_storage_is_bounded_and_disk_contracts_remain(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") lane_service = workflow.split("\n lanes:\n", 1)[1].split("\n steps:", 1)[0] aggregate_service = workflow.split("\n test:\n", 1)[1].split("\n steps:", 1)[0] + api_service = workflow.split("\n full-api-e2e:\n", 1)[1].split("\n steps:", 1)[0] self.assertIn( "${{ matrix.lane != 'schema_contracts' && " "'--tmpfs /var/lib/postgresql/data:rw,nosuid,nodev,noexec,size=2147483648' || '' }}", lane_service, ) self.assertNotIn("--tmpfs", aggregate_service) + self.assertNotIn("services:", aggregate_service) + self.assertIn("postgres:", api_service) + self.assertIn("redis:", api_service) self.assertLess( workflow.index("- name: Verify PostgreSQL CI storage and write settings"), workflow.index("- name: Execute semantic lane"), From 846d62968e38e5bc4f3fe955719b3b00cd186937 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 08:18:39 +0100 Subject: [PATCH 08/15] ci(backend): report test impact in shadow mode --- .ci/test-impact/impact_map.json | 37 +- .ci/test-impact/run_selected_tests.py | 343 ----------- .ci/test-impact/validate_workflow_jobs.py | 81 --- .commitrail/INDEX.md | 2 +- .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 136 ++--- .../initiatives/WS-CI-006/WS-CI-006-01.md | 390 ++++-------- .github/workflows/backend.yml | 560 ++++-------------- AGENTS.md | 10 +- CONTRIBUTING.md | 10 +- backend/scripts/test_impact_selection.py | 494 +++++++-------- backend/scripts/test_lane_catalogue.py | 2 +- backend/tests/test_ci_impact_selection.py | 456 ++++++-------- docker/minio/README.md | 17 +- docs/operations_backend_testing.md | 77 +-- docs/roadmap_status.md | 33 +- scripts/test_lightweight_agent_gates.py | 168 ++---- 16 files changed, 821 insertions(+), 1995 deletions(-) delete mode 100644 .ci/test-impact/run_selected_tests.py delete mode 100644 .ci/test-impact/validate_workflow_jobs.py diff --git a/.ci/test-impact/impact_map.json b/.ci/test-impact/impact_map.json index 48537d383..5e483ab2e 100644 --- a/.ci/test-impact/impact_map.json +++ b/.ci/test-impact/impact_map.json @@ -1,24 +1,21 @@ { "schema_version": 1, - "commitrail_test_modules": [ - "tests/projects/review_policy/test_semantics.py" - ], - "owners": [ - { - "source_paths": [ - "backend/app/core/s3_validation.py" - ], - "test_paths": [ - "backend/tests/test_config.py", - "backend/tests/test_artifact_store_conformance.py", - "backend/tests/test_s3_artifact_store.py" - ], - "test_modules": [ - "tests/test_config.py", - "tests/test_artifact_store_conformance.py", - "tests/test_s3_artifact_store.py" - ], - "infrastructure": "minio" + "lane_groups": { + "shared_foundations": [ + "shared_foundations_a", + "shared_foundations_b" + ] + }, + "source_paths": { + "backend/app/core/s3_validation.py": { + "lane_group": "shared_foundations", + "reason": "S3 configuration validation is exercised in the shared-foundation partition." } - ] + }, + "path_prefixes": { + ".commitrail/": { + "lane_group": "shared_foundations", + "reason": "Commitrail policy semantics are tested in the shared-foundation partition." + } + } } diff --git a/.ci/test-impact/run_selected_tests.py b/.ci/test-impact/run_selected_tests.py deleted file mode 100644 index 806bf97b7..000000000 --- a/.ci/test-impact/run_selected_tests.py +++ /dev/null @@ -1,343 +0,0 @@ -#!/usr/bin/env python3 -"""Execute and attest one exact impact selection without database services.""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import os -from pathlib import Path -import subprocess -import sys -import time -from typing import Any - -BACKEND = Path(__file__).resolve().parents[2] / "backend" -ROOT = BACKEND.parent -sys.path.insert(0, str(BACKEND)) - -from scripts.test_impact_selection import ( # noqa: E402 - MAP_RELATIVE_PATH, - SelectionError, - _inventory, - classify_paths, -) - - -def _canonical_json(value: Any) -> bytes: - return (json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n").encode() - - -def _hash_bytes(value: bytes) -> str: - return hashlib.sha256(value).hexdigest() - - -def _read_manifest(path: Path) -> dict[str, Any]: - if path.is_symlink() or not path.is_file(): - raise SelectionError("missing_selection_manifest") - try: - value = json.loads(path.read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: - raise SelectionError("invalid_selection_manifest") from exc - if not isinstance(value, dict): - raise SelectionError("invalid_selection_manifest") - return value - - -def _digest(path: Path) -> str: - if path.is_symlink() or not path.is_file(): - raise SelectionError("missing_bound_input") - return hashlib.sha256(path.read_bytes()).hexdigest() - - -def verify_selection(manifest: dict[str, Any], *, expected_job: str) -> list[str]: - """Recompute every candidate-side binding before collecting tests.""" - current_sha = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip() - current_tree = subprocess.check_output( - ["git", "rev-parse", "HEAD^{tree}"], cwd=ROOT, text=True - ).strip() - status = subprocess.check_output( - ["git", "status", "--porcelain"], cwd=ROOT, text=True - ) - if ( - current_sha != manifest.get("execution_sha") - or current_tree != manifest.get("execution_tree") - or status - or os.environ.get("GITHUB_SHA") != current_sha - or os.environ.get("GITHUB_EVENT_PULL_REQUEST_BASE_SHA") != manifest.get("base_sha") - or os.environ.get("GITHUB_EVENT_PULL_REQUEST_HEAD_SHA") != manifest.get("head_sha") - ): - raise SelectionError("candidate_target_mismatch") - base_sha = str(manifest.get("base_sha", "")) - head_sha = str(manifest.get("head_sha", "")) - execution_parents = subprocess.check_output( - ["git", "show", "-s", "--format=%P", current_sha], cwd=ROOT, text=True - ).strip().split() - if execution_parents != [base_sha, head_sha]: - raise SelectionError("execution_parent_mismatch") - merge_base = subprocess.check_output( - ["git", "merge-base", base_sha, head_sha], cwd=ROOT, text=True - ).strip() - if merge_base != manifest.get("merge_base_sha"): - raise SelectionError("merge_base_mismatch") - raw_paths = subprocess.check_output( - ["git", "diff", "--name-only", "-z", "--no-renames", merge_base, head_sha, "--"], - cwd=ROOT, - ) - try: - actual_changed_paths = sorted( - path.decode("utf-8", errors="strict") for path in raw_paths.split(b"\0") if path - ) - except UnicodeDecodeError as exc: - raise SelectionError("invalid_changed_path_encoding") from exc - if actual_changed_paths != manifest.get("changed_paths"): - raise SelectionError("changed_path_set_mismatch") - if manifest.get("mode") not in {"impact", "pure"}: - raise SelectionError("nonselective_manifest") - expected_jobs = manifest.get("expected_jobs") - if not isinstance(expected_jobs, list) or expected_job not in expected_jobs: - raise SelectionError("unexpected_impact_job") - - map_path = ROOT / MAP_RELATIVE_PATH - selector_path = BACKEND / "scripts/test_impact_selection.py" - runner_path = Path(__file__).resolve() - if _digest(map_path) != manifest.get("impact_map_sha256"): - raise SelectionError("impact_map_drift") - if _digest(selector_path) != manifest.get("selector_sha256"): - raise SelectionError("selector_drift") - if _digest(runner_path) != manifest.get("impact_runner_sha256"): - raise SelectionError("runner_drift") - - changed_paths = manifest.get("changed_paths") - if ( - not isinstance(changed_paths, list) - or any(not isinstance(path, str) for path in changed_paths) - or hashlib.sha256(_canonical_json(changed_paths)).hexdigest() - != manifest.get("changed_paths_sha256") - ): - raise SelectionError("changed_path_digest_mismatch") - impact_map = json.loads(map_path.read_text(encoding="utf-8")) - mode, modules, jobs, profile = classify_paths(changed_paths, impact_map) - if ( - mode != manifest.get("mode") - or modules != manifest.get("selected_modules") - or jobs != manifest.get("expected_jobs") - or profile != manifest.get("infrastructure_profile") - ): - raise SelectionError("classification_drift") - - inventory, inventory_digest = _inventory(ROOT, current_sha) - if ( - inventory != manifest.get("test_inventory") - or inventory_digest != manifest.get("test_inventory_sha256") - ): - raise SelectionError("test_inventory_drift") - if _digest(Path(__file__)) != manifest.get("impact_runner_sha256"): - raise SelectionError("runner_drift") - return modules - - -def validate_evidence( - manifest_path: Path, - evidence_path: Path, - *, - expected_job: str, - expected_manifest_sha256: str, -) -> None: - """Verify the selected test artifact against the selection job output.""" - manifest = _read_manifest(manifest_path) - if _digest(manifest_path) != expected_manifest_sha256: - raise SelectionError("selection_artifact_digest_mismatch") - evidence = _read_manifest(evidence_path) - expected_path = evidence_path.parent / "expected.json" - expected = _read_manifest(expected_path) - nodes = evidence.get("selected_nodes") - completed = evidence.get("completed_nodes") - observed = evidence.get("observed_collected_nodes") - elapsed = evidence.get("elapsed_seconds") - if ( - evidence.get("job") != expected_job - or evidence.get("execution_sha") != manifest.get("execution_sha") - or evidence.get("execution_tree") != manifest.get("execution_tree") - or evidence.get("test_inventory_sha256") != manifest.get("test_inventory_sha256") - or evidence.get("selection_manifest_sha256") != _hash_bytes( - _canonical_json(manifest) - ) - or evidence.get("exit_code") != 0 - or not isinstance(nodes, list) - or not nodes - or any(not isinstance(node, str) for node in nodes) - or nodes != sorted(set(nodes)) - or completed != nodes - or observed != nodes - or evidence.get("skipped_nodes") != [] - or evidence.get("deselected_nodes") != [] - or evidence.get("selected_modules") != manifest.get("selected_modules") - or evidence.get("expected_payload_sha256") != _digest(expected_path) - or expected.get("execution_sha") != manifest.get("execution_sha") - or expected.get("execution_tree") != manifest.get("execution_tree") - or expected.get("selected_modules") != manifest.get("selected_modules") - or expected.get("selected_nodes") != nodes - or expected.get("selected_nodes_sha256") != _hash_bytes(_canonical_json(nodes)) - or expected.get("selection_manifest_sha256") - != _hash_bytes(_canonical_json(manifest)) - or evidence.get("expected_nodes_sha256") != _hash_bytes(_canonical_json(nodes)) - or evidence.get("completed_nodes_sha256") != _hash_bytes(_canonical_json(completed)) - or isinstance(elapsed, bool) - or not isinstance(elapsed, (int, float)) - or elapsed < 0 - ): - raise SelectionError("impact_evidence_incomplete") - - -def run_selected(manifest: dict[str, Any], output: Path, *, expected_job: str) -> int: - """Collect exact target nodes, execute them, and require complete custody.""" - from scripts.run_test_lanes import ( - COLLECTED_ENV, - COMPLETED_ENV, - DESELECTED_ENV, - HEAD_ENV, - SKIPPED_ENV, - _read_nodes, - collect_nodes, - ) - - modules = verify_selection(manifest, expected_job=expected_job) - if output.exists() or output.is_symlink(): - raise SelectionError("impact_output_exists") - output.mkdir(parents=True, mode=0o700) - collection_dir = output / "collection" - collection_dir.mkdir(mode=0o700) - started = time.monotonic() - tree_sha = str(manifest["execution_sha"]) - collection_code, nodes, deselected = collect_nodes( - tuple(modules), collection_dir, tree_sha - ) - if collection_code != 0 or deselected or not nodes: - raise SelectionError("impact_collection_failed") - - expected_bytes = _canonical_json( - { - "execution_sha": tree_sha, - "execution_tree": manifest["execution_tree"], - "selected_modules": modules, - "selected_nodes": nodes, - "selected_nodes_sha256": hashlib.sha256(_canonical_json(nodes)).hexdigest(), - "selection_manifest_sha256": hashlib.sha256( - _canonical_json(manifest) - ).hexdigest(), - } - ) - expected_path = output / "expected.json" - expected_path.write_bytes(expected_bytes) - - collected_path = output / "collected.jsonl" - completed_path = output / "completed.jsonl" - skipped_path = output / "skipped.jsonl" - deselected_path = output / "deselected.jsonl" - for path in (collected_path, completed_path, skipped_path, deselected_path): - path.touch(mode=0o600, exist_ok=False) - coverage_path = output / ".coverage" - env = os.environ.copy() - env.update( - { - "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1", - "PYTHONPATH": os.pathsep.join( - value for value in (str(BACKEND), env.get("PYTHONPATH", "")) if value - ), - "COVERAGE_FILE": str(coverage_path), - COLLECTED_ENV: str(collected_path), - COMPLETED_ENV: str(completed_path), - SKIPPED_ENV: str(skipped_path), - DESELECTED_ENV: str(deselected_path), - HEAD_ENV: tree_sha, - } - ) - command = [ - sys.executable, - "-m", - "pytest", - "-q", - "-p", - "pytest_asyncio.plugin", - "-p", - "pytest_cov.plugin", - "-p", - "scripts.run_test_lanes", - "--cov=app", - "--cov-report=", - *nodes, - ] - result = subprocess.run(command, cwd=BACKEND, env=env, check=False) - completed = _read_nodes(completed_path, allow_empty=True) - observed_collected = _read_nodes(collected_path, allow_empty=True) - skipped = _read_nodes(skipped_path, allow_empty=True) - run_deselected = _read_nodes(deselected_path, allow_empty=True) - elapsed = round(time.monotonic() - started, 3) - evidence = { - "completed_nodes": sorted(completed), - "completed_nodes_sha256": hashlib.sha256( - _canonical_json(sorted(completed)) - ).hexdigest(), - "expected_nodes_sha256": hashlib.sha256(_canonical_json(nodes)).hexdigest(), - "expected_payload_sha256": hashlib.sha256(expected_bytes).hexdigest(), - "execution_sha": tree_sha, - "execution_tree": manifest["execution_tree"], - "exit_code": result.returncode, - "job": expected_job, - "observed_collected_nodes": sorted(observed_collected), - "selected_modules": modules, - "selected_nodes": nodes, - "skipped_nodes": skipped, - "deselected_nodes": sorted(set(deselected + run_deselected)), - "elapsed_seconds": elapsed, - "selection_manifest_sha256": hashlib.sha256( - _canonical_json(manifest) - ).hexdigest(), - "test_inventory_sha256": manifest["test_inventory_sha256"], - } - evidence_path = output / "evidence.json" - evidence_path.write_bytes(_canonical_json(evidence)) - if ( - result.returncode != 0 - or sorted(observed_collected) != nodes - or sorted(completed) != nodes - or skipped - or run_deselected - ): - print("impact test custody incomplete", file=sys.stderr) - return 1 - return 0 - - -def main() -> int: - parser = argparse.ArgumentParser() - parser.add_argument("--manifest", required=True, type=Path) - parser.add_argument("--output", required=True, type=Path) - parser.add_argument("--expected-job", required=True, choices=("impact-s3", "impact-pure")) - parser.add_argument("--validate-only", action="store_true") - parser.add_argument("--evidence", type=Path) - parser.add_argument("--expected-manifest-sha256") - args = parser.parse_args() - try: - if args.validate_only: - if args.evidence is None or args.expected_manifest_sha256 is None: - raise SelectionError("missing_validation_input") - validate_evidence( - args.manifest, - args.evidence, - expected_job=args.expected_job, - expected_manifest_sha256=args.expected_manifest_sha256, - ) - return 0 - return run_selected( - _read_manifest(args.manifest), args.output, expected_job=args.expected_job - ) - except (OSError, subprocess.SubprocessError, RuntimeError, SelectionError) as exc: - print(f"impact test execution failed: {exc}", file=sys.stderr) - return 2 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/.ci/test-impact/validate_workflow_jobs.py b/.ci/test-impact/validate_workflow_jobs.py deleted file mode 100644 index 235a63260..000000000 --- a/.ci/test-impact/validate_workflow_jobs.py +++ /dev/null @@ -1,81 +0,0 @@ -#!/usr/bin/env python3 -"""Require the exact job inventory declared by the trusted selector.""" - -from __future__ import annotations - -import argparse -import hashlib -import json -from pathlib import Path -import re -import sys -from typing import Any - -FULL = { - "impact-selection", - "auth-boundary-preflight", - "minio-image", - "lanes", - "full-api-e2e", -} -S3 = {"impact-selection", "auth-boundary-preflight", "minio-image", "impact-s3"} -PURE = {"impact-selection", "auth-boundary-preflight", "impact-pure"} -KNOWN = FULL | S3 | PURE -DIGEST_RE = re.compile(r"^[0-9a-f]{64}$") - - -def validate(manifest: dict[str, Any], results: dict[str, str]) -> None: - expected_by_mode = { - ("full", "full"): FULL, - ("impact", "minio"): S3, - ("pure", "none"): PURE, - } - mode_profile = (manifest.get("mode"), manifest.get("infrastructure_profile")) - expected = expected_by_mode.get(mode_profile) - manifest_jobs = manifest.get("expected_jobs") - if ( - manifest.get("schema_version") != 1 - or expected is None - or not isinstance(manifest_jobs, list) - or set(manifest_jobs) != expected - or len(manifest_jobs) != len(expected) - or set(results) != KNOWN - ): - raise ValueError("invalid_expected_job_inventory") - for job, result in results.items(): - if job in expected and result != "success": - raise ValueError(f"expected_job_not_success:{job}") - if job not in expected and result != "skipped": - raise ValueError(f"unexpected_job_ran:{job}") - - -def main() -> int: - parser = argparse.ArgumentParser() - parser.add_argument("--manifest", required=True, type=Path) - parser.add_argument("--manifest-sha256", required=True) - parser.add_argument("--results-json", required=True) - args = parser.parse_args() - try: - if args.manifest.is_symlink() or not args.manifest.is_file(): - raise ValueError("missing_selection_manifest") - if DIGEST_RE.fullmatch(args.manifest_sha256) is None: - raise ValueError("invalid_selection_manifest_digest") - raw = args.manifest.read_bytes() - if hashlib.sha256(raw).hexdigest() != args.manifest_sha256: - raise ValueError("selection_manifest_digest_mismatch") - manifest = json.loads(raw) - results = json.loads(args.results_json) - if not isinstance(manifest, dict) or not isinstance(results, dict) or any( - not isinstance(job, str) or not isinstance(result, str) - for job, result in results.items() - ): - raise ValueError("invalid_job_result_manifest") - validate(manifest, results) - return 0 - except (OSError, UnicodeDecodeError, json.JSONDecodeError, ValueError) as exc: - print(f"backend job fan-in rejected: {exc}", file=sys.stderr) - return 1 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/.commitrail/INDEX.md b/.commitrail/INDEX.md index c1240d0ef..cf5c11813 100644 --- a/.commitrail/INDEX.md +++ b/.commitrail/INDEX.md @@ -17,7 +17,7 @@ for current product capability. | [WS-REV-001](initiatives/WS-REV-001/OVERVIEW.md) | Planned | Shared acceptance/source and existing fence foundations; human hidden review work remains independently dependency-gated | | [WS-QUAL-002](initiatives/WS-QUAL-002/OVERVIEW.md) | Planned | Populate subsystem ownership before changed-line mutation work | | [WS-QUAL-003](initiatives/WS-QUAL-003/OVERVIEW.md) | Planned | Audit and prune test proof, add missing safety cases, decompose oversized test modules | -| [WS-CI-006](initiatives/WS-CI-006/OVERVIEW.md) | Planned | Deterministic change-impact test selection with full-suite fallback and periodic complete verification | +| [WS-CI-006](initiatives/WS-CI-006/OVERVIEW.md) | Planned | Shadow-mode semantic-lane impact report beside the unchanged full required suite; gate changes require later evidence and review | | [WS-XINT-002](initiatives/WS-XINT-002/OVERVIEW.md) | Planned | Remaining ART/AUTH activation edges only | | [WS-XINT-003](initiatives/WS-XINT-003/OVERVIEW.md) | Planned | Resume activation only against exact merged REV behavior | | WS-POL-002 | Superseded | Future guide inference belongs to WS-POL-003; reframe remaining executor work against current specifications | diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md index 00be24bc5..ae18573e8 100644 --- a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -5,107 +5,43 @@ ## Intent -Reduce routine backend pull-request feedback by running the tests affected by a -change, while keeping the required result trustworthy. A narrow change should -not wait for every unrelated subsystem; changes whose impact cannot be proven -must still run the full suite. +Reduce routine backend PR feedback by selecting tests according to demonstrated +change impact, without making incomplete test evidence authoritative. Today +the full Backend suite remains required. The first usable step is a shadow +report whose recommendation is compared with the complete suite on the same +exact PR target. -## Current evidence +## Current evidence and limits - `backend/scripts/test_lane_catalogue.py` assigns every discovered test module - to the complete backend run. `backend/scripts/run_test_lanes.py` then - deterministically hash-partitions node IDs within shared, project, and task - groups; these partitions are not an impact map. -- `.github/workflows/backend.yml` always schedules nine backend lanes, a - separate authorization-boundary preflight, MinIO image construction, and a - final real-API/evidence aggregation job. + to the complete backend run. Shared, project and task node collections are + partitioned across two, three and three jobs respectively; a module in a + partitioned group requires every shard in its group. +- `.github/workflows/backend.yml` continues to run all nine lanes, authorization + preflight, MinIO, real API proof, and evidence aggregation on every PR. The + added `impact-report` job is informational and cannot control lane execution. - Run [36724982896](https://github.com/Flow-Research/workstream/actions/runs/36724982896) - on 2026-09-30 completed 7,918 tests, zero skipped/deselected, in about 44 - minutes. Three lane jobs began about 21 minutes after the first six; the - longest lane then ran about 17 minutes. This is one observed run, not a - universal baseline. -- `CONTRIBUTING.md` currently requires hosted full suites. Selective PR checks - therefore require an explicit, reviewed policy update—not merely a workflow - optimization. - -## Design direction - -- Use a deterministic repository-owned changed-source-to-test ownership map; - do not add an external test-impact service or trust mutable historical - selector state. -- Start with one narrow reviewed mapping only: changes confined to - `backend/app/core/s3_validation.py` select its direct configuration contract - tests, provider-neutral namespace-conformance tests, and real S3/MinIO adapter - tests. Changes to its callers, shared test support, schemas/migrations, - dependencies, or CI-selection machinery select the complete suite. No other - application-source path is selective initially. -- Keep `.commitrail/**` in the exact change manifest, but exclude it only when - classifying backend source impact: every bounded product PR carries a durable - change record. Every PR that changes Commitrail metadata also runs the - backend policy-semantics module that reads specific planning files. A - metadata-only PR still runs that module, Markdown/stale-doc gates, and the - non-empty authorization/static preflight; other documentation paths remain - full-suite unless separately reviewed. -- Bind selection to the exact PR base, head, synthetic merge execution SHA/tree, - merge-base, changed-path digest, impact-map digest, test-inventory digest, - and selected test/job manifest. A mismatch, missing Git object, stale - candidate, changed test support, or unreviewed path selects the complete - suite. Changes within mapped test modules run the complete mapped closure; - new or unmapped tests and shared test fixtures select the full suite. -- Keep the required Backend workflow and final check present on every PR. Do - not use GitHub workflow path filters to suppress a required status. -- Run shared lint/docstring and authorization-boundary checks once, not once per - test lane. Keep full public-API E2E and service startup in full mode unless a - reviewed impact mapping specifically requires it. Reuse the pinned MinIO - build by exact source-input digest, not commit SHA; verify cache contents - before tests. -- Make the selected impact set the PR gate. Keep full-suite execution available - for broad changes and manual runs, and run the complete suite nightly on - `main`. A failed nightly audit remains failed and requires diagnosis of the - map or product/test defect; it does not silently green PR checks or substitute - for exact-PR evidence. Never describe a scheduled result as proof for another - PR head. -- Preserve real PostgreSQL, S3-protocol, concurrency, migration, and public API - checks whenever their owner paths are selected. The five-minute target is - for routine narrow changes, not broad/security/schema changes or runner - outages; hosted wall time remains measured, not promised. -- Leave test bodies, assertions, coverage policy, product behavior, and - external review requirements unchanged. - -## Proposed boundary - -1. **WS-CI-006-01:** implement the exact-target impact manifest, the initial - S3-validation mapping, conservative full-suite fallback, selected-node/job - evidence validation, and CI integration; change `AGENTS.md` and - `CONTRIBUTING.md` in the same PR; prove the selector adversarially and run - the complete suite on the candidate. -2. Add further source/test ownership mappings only when their complete - consumer-test closure and shared-fixture/infrastructure dependencies are - demonstrated. Unmapped application code remains full-suite. Do not add a - mapping just to claim a broader speedup. - -## Risks and controls - -- **False-negative selection:** exact ownership coverage, full-suite fallback - for unknown or cross-cutting paths, mapping mutation probes, plus scheduled - full-suite audits. -- **Stale or mismatched evidence:** exact base/head, tree and manifest digests; - execution tree must be the exact GitHub PR merge candidate whose parents are - the event base and head; no cached result may attest to another source tree. -- **Broken branch protection:** workflow and final required status always run; - preserve the existing required `test` context, distinguish expected - unselected jobs from missing selected jobs using the digest-bound - manifest, and reject an empty test selection. No path-filtered required - workflow. -- **Misleading performance claim:** report selected test count and hosted wall - time separately; the selector-changing PR itself full-fallbacks. Wait for the - first later naturally eligible mapped PR before reporting narrow-mode hosted - timing; retain full execution for risky changes and do not claim every PR is - under five minutes. - -## Non-goals - -- No test deletion, skipping, weakened assertions, coverage threshold, arbitrary - shard expansion, runner-provider change, new service, or agent-controlled - selection authority. -- No changes to product code or test behavior. + completed 7,918 tests with zero skips/deselections in about 44 minutes. This + is one observed run, not a universal baseline. +- Initial explicit mappings are intentionally narrow: exact S3 validation + owner to both shared-foundation shards; Commitrail-only changes to those same + shards; changed test modules to every lane partition that owns them. Any + unmapped path recommends all nine lanes. +- The report is generated by the PR candidate and is not independent policy + evidence. PRs that change the selector, map, catalogue or workflow cannot + validate their own changes; no lane is omitted from actual CI. + +## Direction after shadow evidence + +1. Observe recommendations beside complete test results on the same exact + target across representative PRs. +2. Expand mappings only after tracing production owners, downstream consumers, + shared fixtures, integration services and the lane catalogue. Unknown impact + remains full-suite. +3. Evaluate whether the accumulated evidence supports a separate bounded change + to required test execution. That decision must preserve the full suite on + `main` and broad/cross-cutting changes; this initiative does not pre-approve + a CI gate reduction or promise every PR completes in five minutes. + +No external selector service, historical mutable test signal, test deletion, +coverage quota, arbitrary sharding, or product behavior change is in scope. diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index cb414d25e..f0151e7cf 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -1,279 +1,141 @@ -# WS-CI-006-01 — Gate PRs on proven change-impact tests +# WS-CI-006-01 — Shadow-mode backend test-impact report - Initiative: `WS-CI-006` - Durable disposition: `Planned` -- Intended merge outcome: The exact Backend `test` PR check blocks on the complete tests for one initially mapped low-level owner, while every other or uncertain change runs the complete suite. A nightly full suite audits for selector drift. +- Intended merge outcome: Backend CI reports a deterministic, exact-PR proposed semantic-lane selection and rationale while the existing complete Backend suite remains unchanged and blocking on every PR and `main` push. This change does not enable selective gating. ## Intent -Reduce routine backend feedback without making test omission a success path. -The current required workflow ran all 7,918 test nodes on every backend PR; -run 36724982896 took about 44 minutes wall-clock. The initial selective -boundary is deliberately one owner only: canonical S3 configuration and -namespace validation. All other application-source changes remain full-suite -until their complete consumer-test closure is explicitly mapped. The five- -minute target applies only to measured, routine narrow changes; it is not a -promise for broad changes or GitHub runner delays. - -## Current behavior and constraints - -- `backend/scripts/test_lane_catalogue.py` inventories every discovered test - module and hash-partitions nodes across nine jobs. A lane is not an impact - owner; omitting one drops arbitrary hashed nodes. -- `backend/scripts/run_test_lanes.py` binds collection/completion evidence to - the checked-out candidate and rejects skipped/deselected nodes, but it has no - source-impact selector. -- `.github/workflows/backend.yml` runs nine lanes, authorization preflight, - MinIO build and final API/evidence aggregation on each PR and main push. - The current final aggregator also starts PostgreSQL and MinIO and runs the - full real-API E2E on every PR, even for a narrow owner. Its MinIO image cache - key includes the workflow SHA, so unchanged pinned source is rebuilt on new - PR heads. -- Live branch protection requires the `test` and `agent-gates` checks, requires - up-to-date branches and one human approval, and enforces protection for - administrators. Keep the exact `test` required status present on every PR. -- `AGENTS.md`, `CONTRIBUTING.md` and the roadmap currently require full hosted - suites on every PR. This PR must change that policy consistently: the exact - selected set is blocking for a mapped PR; uncertain/broad changes require the - full suite; a nightly full suite audits the map but never substitutes for - evidence from a different PR head. +Backend PRs currently pay for the complete suite even when a change has a +bounded impact. Before changing that gate, collect exact-head evidence about +which semantic lanes a conservative change-impact map would select. Use the +existing lane catalogue as the first coarse unit; later work may select within +a lane only after the shadow reports establish a sound owner-to-test closure. + +The observed baseline is run +[36724982896](https://github.com/Flow-Research/workstream/actions/runs/36724982896): +7,918 tests, zero skipped/deselected, about 44 minutes wall time. This is one +run, not a universal timing estimate. The target is under five minutes for +small, well-understood changes after a later separately reviewed gating phase; +this shadow change makes no runtime reduction or timing claim. + +## Current behavior and lane semantics + +- `backend/scripts/test_lane_catalogue.py` assigns every discovered test module + to the complete run. Shared, project, and task groups are partitioned across + multiple jobs. A module appearing in a partitioned group requires every + shard in that group to cover all of its node IDs. +- `.github/workflows/backend.yml` runs all nine lanes, the authorization + preflight, MinIO-backed tests, real API proof and aggregate evidence on every + PR and `main` push. The required `test` result and full-suite policy remain + unchanged in this PR. +- A shadow report may recommend lanes but cannot control `if` conditions, + services, test commands, required jobs, fan-in, or merge status. Unknown, + broad, malformed, or unclassified input reports `all lanes`. ## Bounded change -### Allowed - -- `.github/workflows/backend.yml` -- `docker/minio/README.md` (cache policy description only) -- `backend/scripts/test_lane_catalogue.py` -- `backend/scripts/test_impact_selection.py` -- `backend/scripts/run_test_lanes.py` -- `backend/scripts/merge_test_lane_evidence.py` -- `backend/scripts/validate_test_lane_evidence.py` -- Direct CI workflow/catalogue/selector/evidence tests under `backend/tests/` -- `scripts/test_lightweight_agent_gates.py` (its Backend workflow shape checks) -- The exact reviewed impact map and validator under `.ci/test-impact/` -- `AGENTS.md`, `CONTRIBUTING.md`, `docs/roadmap_status.md` -- `docs/operations_backend_testing.md` -- This initiative record and `.commitrail/INDEX.md` - -### Not allowed - -- Product source other than `backend/app/core/s3_validation.py`, Docker/MinIO - build inputs, schemas, migrations, public API changes, or changes to - product-behavior assertions. -- Deleting tests or changing skip/xfail/deselection behavior; rewriting tests to - match broken behavior. -- GitHub path filters for the required workflow; third-party impact services, - mutable selector databases, cache-only proof, new runner infrastructure, or - arbitrary shards. -- A green result for a broad, unmapped, mismatched, empty, or incomplete test - selection. - -## Design and decisions - -### Initial selective boundary - -Only a PR whose application-source change is confined to -`backend/app/core/s3_validation.py` may use impact selection. Its required -closure is: - -| Owner behavior | Required tests | Infrastructure / shared dependencies | -|---|---|---| -| S3 region/bucket/prefix and MinIO endpoint canonicalization; provider namespace descriptor validation | `backend/tests/test_config.py`; `backend/tests/test_artifact_store_conformance.py` (including its standalone namespace-value cases); `backend/tests/test_s3_artifact_store.py` (including inherited real-adapter conformance vectors) | Source-pinned real MinIO is required by the S3 adapter tests. PostgreSQL is not required. Shared unchanged support includes `backend/tests/conftest.py` and `backend/tests/artifact_store_helpers.py`. | - -`test_config.py` directly proves helper behavior and Settings/namespace -validation. The S3 adapter tests prove the canonical configuration reaches the -real provider adapter correctly. Any change to the helper's consumers -(`backend/app/core/config.py`, `backend/app/interfaces/artifacts.py`, or -`backend/app/adapters/artifacts/s3_compatible.py`), to shared support, or to -any other application source selects the full suite. No other source mapping -is enabled by this change. - -Changes within any of the three mapped test modules run the complete three- - module closure, not just changed node IDs. Changes to any other test module, a - new test, shared fixture/helper, test inventory, dependency, schema/migration, - Docker/MinIO build input, runner, workflow, impact map or selector run the - full suite. This prevents test edits from silently shrinking or redefining - the proof used by the mapping. - -### Target and evidence custody - -- For a PR, compute changed paths from explicit event `base.sha` and - `head.sha`, using their Git merge-base; do not read PR descriptions as - authority. Run the selector/map from the trusted base revision, not from PR - code. Because the first rollout candidate has no trusted selector on its base, - that candidate uses a fixed full-suite manifest and the complete required job - set; it never invokes its new selector to authorize selective execution. A - changed selector, map or workflow itself always selects full suite. -- Preserve the complete changed-path list and digest in the evidence manifest. - For backend-impact classification only, remove paths under `.commitrail/` - from the source-impact input: every bounded PR carries its own durable - Commitrail record, so treating that metadata as backend source would force - all product PRs into full fallback. Each PR with any `.commitrail/` change - additionally selects the complete module - `backend/tests/projects/review_policy/test_semantics.py`, including its - `test_future_activation_owners_preserve_unavailable_mode_contract` case, - which reads and checks two Commitrail planning files. This test is run for - metadata-only changes and unioned with any mapped source closure. A PR - containing only Commitrail changes therefore has a named, non-empty backend - test set as well as the authorization/static preflight; it never selects - zero tests. - The required `agent-gates` workflow continues to validate Commitrail and - Markdown/stale wording. Any Commitrail change adds the named policy-contract - test module; mixed Commitrail plus mapped S3 changes run both closures. Mixed - Commitrail plus any unclassified path use full suite. Changes that add or - alter backend tests, shared support, or their Commitrail dependencies already - trigger full fallback through their changed test/support/selector paths. - Do not exempt arbitrary documentation, skills, agent instructions, or other - paths. -- Bind event base SHA, head SHA, merge-base, GitHub execution SHA and tree, - changed-path list/digest, selector/map digest, full test-inventory digest, - selected modules/node IDs, infrastructure profile, and expected job names in - a deterministic manifest. -- Require the test checkout to be the exact GitHub synthetic merge candidate - whose parents are the event base and head. Validate object presence, - ancestry, parent identities, clean checkout, and tree equality before - accepting selective mode. If candidate/base/head is stale, unavailable, - malformed, has unexpected parents, or any digest differs, run the full suite - against the checked-out execution tree. Never attest one tree with another - tree's test list or cached evidence. -- Changed tests are part of the selected set. The selected manifest must be - non-empty, enumerate expected nodes before execution, and bind expected job - inventory. Selected tests may not be skipped or deselected. - -### Workflow and branch protection - -- Preserve the always-created required `test` check and the separate required - `agent-gates` check. Do not use workflow-level path filters. -- The final `test` aggregator validates the digest-bound manifest and exact - expected job inventory. Full mode expects all current nine lanes; mapped mode - expects the explicit MinIO-backed impact job and always-required static and - authorization-boundary jobs. Expected-unselected lanes differ from missing - selected jobs. Missing, duplicate, foreign-target, skipped, deselected or - incomplete selected evidence fails. -- Keep full lint and docstring checks blocking but execute them once in a - shared job, not redundantly in every test lane. Keep authorization-boundary - preflight required for every PR and at fan-in. Combine static checks with - that existing preflight job so they share one dependency installation. The - final evidence `test` job must not start database/storage services or rerun - product tests; it aggregates and verifies evidence from expected jobs. Move real-API E2E into - an explicit full-suite job. Impact mode does not run that E2E unless a future - reviewed mapping names it; the S3 mapping retains real MinIO adapter tests - but does not require PostgreSQL. Full mode and nightly audits retain - PostgreSQL, MinIO, migration, transaction, concurrency, boundary and real - API proof. This removes duplicate services only where the selected test - closure proves them unnecessary. -- Change the MinIO cache key from workflow commit SHA to a digest of every - source build input plus runner OS/architecture. The cache is only an - optimization, never evidence: verify the archive checksum and start the - cached image's version/health probe before publishing it to test jobs. An - input change or invalid cache triggers a fresh pinned-source build. -- Add nightly full-suite execution on `main` and manual dispatch. The full run - must bind to its own head, reject skips/deselections and report failures - normally. A nightly result never satisfies a PR check. A failure requires - diagnosis of a product/test defect or selector drift; do not translate it to - green or claim it proves another commit. -- Remove the redundant full Backend run on each protected `main` push only - after retaining exact PR checks, strict up-to-date protection and nightly / - manual full runs. Protection currently requires review and both checks, and - is enforced for admins, so merged code has passed the exact candidate check. - Record this as removing duplicated execution, not removing required PR - verification. +### Allowed files + +- `.github/workflows/backend.yml` for an always-running PR classifier/report + job only; the existing full-suite job graph and commands stay blocking. +- `.ci/test-impact/impact_map.json` and + `backend/scripts/test_impact_selection.py` for deterministic shadow + classification and exact-target report generation. +- `backend/tests/test_ci_impact_selection.py` and + `scripts/test_lightweight_agent_gates.py` for selector and workflow-shape + regressions. +- `docs/operations_backend_testing.md`, the WS-CI-006 initiative overview, + this record, and the Commitrail index for the shadow-only operating contract. +- `docs/roadmap_status.md` only if its current CI capability statement needs + correction to describe this intended merged state. + +### Prohibited changes + +- No changes to test bodies, assertions, collection, skip/deselect behavior, + coverage policy, current lane partitioning, services, test commands, required + status checks, branch protection, or merge rules. +- No selector-driven workflow conditions, lane omissions, workflow-level path + filters, test execution service, third-party impact product, or mutable + historical selection authority. +- No test deletion or claim that shadow output reduces CI time. + +## Shadow selection contract + +- Resolve the changed paths from the exact PR base/head and merge base; bind the + report to the base SHA, head SHA, execution SHA/tree, changed-path digest, + selector/map version and digests, selected semantic lanes and rationale. +- Use the current semantic lane catalogue to map test modules to every lane + shard that owns their nodes. A changed test module is included in the proposed + impact closure. Shared fixtures, schema/migrations, dependencies, workflow, + lane catalogue, map/selector changes, unknown source paths, or unavailable + Git evidence conservatively recommend all nine lanes. +- Initial source mapping is deliberately limited to the reviewed S3 validation + owner `backend/app/core/s3_validation.py`, whose mapped tests are all in the + partitioned `shared_foundations` group. The report must therefore recommend + both `shared_foundations_a` and `shared_foundations_b`, not a hand-picked + test-module subset. Changes to other application source recommend all lanes + until additional consumer closures are demonstrated and explicitly mapped. +- `.commitrail/**` remains in the changed-path report and recommends the lane + group containing `tests/projects/review_policy/test_semantics.py`; mixed + changes union this with the source selection. If any other path is + unclassified, recommend all lanes. Documentation, skills, and agent-policy + paths are not implicitly exempted. +- The report explains every selected lane and every omitted lane. Omission is + allowed in the *recommendation only* when the exact mapping explains why; + missing evidence yields all lanes. The classifier is observational: CI still + executes all nine lanes and all existing integration/preflight jobs. +- The selector and map are part of the PR candidate in this shadow phase. Their + report is candidate-produced diagnostic evidence, not a trusted test policy + or independent audit receipt. A PR that changes the selector, map, catalogue, + or workflow cannot validate those changed inputs; any later gating change + must establish trusted selection policy separately. +- The report is uploaded and linked from the PR workflow summary. It is + descriptive evidence for comparing the proposed lane set with the complete + run from the same PR head; it never attests to another commit. ## Acceptance criteria -- [ ] The only initial product-source mapping is - `backend/app/core/s3_validation.py` to the exact full test closure above; - its MinIO requirement and lack of PostgreSQL dependency are executable and - tested. -- [ ] All other source paths, shared support changes, infrastructure changes, - stale/missing Git objects and unknown changes fail closed to full suite. -- [ ] Commitrail paths remain in the exact changed-path manifest but are - excluded only from backend source-impact classification. Any Commitrail - change selects the complete policy-semantics test module, including - `test_future_activation_owners_preserve_unavailable_mode_contract`; this is - unioned with the S3 closure for mixed mapped changes. Metadata-only mode also - runs the named non-empty authorization/static preflight and `agent-gates`. - Mixed changes with unclassified paths use full suite. No other documentation - or metadata path is implicitly exempted. -- [ ] Adversarial classification tests change each Commitrail file consumed by - the policy-contract test and prove that the module is selected, including - when combined with the mapped S3 owner; a Commitrail-only unrelated metadata - path still selects this non-empty module and preflight. -- [ ] Base/head/merge-base/execution SHA+tree, changed-path and map digests, - test inventory, selected nodes, infrastructure profile and expected jobs are - bound to evidence. A synthetic merge-parent/tree mismatch cannot pass - selective mode. -- [ ] Adversarial tests cover unknown path, omitted owner module, stale base, - altered execution tree, duplicate/missing results, missing expected job, - unexpected job, broad/global path, empty selection and changed selector/map. -- [ ] Every selected node completes; no selected node is skipped/deselected; - fan-in rejects missing, duplicate, foreign-target and incomplete evidence. -- [ ] Required `test` remains present and blocking; full mode expects all nine - current lanes and impact mode expects exactly the manifest-selected jobs plus - always-required checks. -- [ ] Full scheduled/manual run remains complete, is bound to its own head, - and is never cross-used as exact PR evidence. A full-suite failure stays red. -- [ ] `AGENTS.md`, `CONTRIBUTING.md` and the affected roadmap claims describe - exact-PR impact evidence, full fallback, nightly audit, and retained real - integration checks consistently. -- [ ] Full lint/docstring validation remains blocking and runs once per - workflow. Removing the main-push rerun is justified by protected exact-PR - verification and does not remove a required check. -- [ ] The final evidence aggregator starts no PostgreSQL/Redis/MinIO service - and runs no duplicate API E2E; those remain in full-suite mode or a future - specifically mapped closure. -- [ ] MinIO image cache keys include all source/platform inputs but not commit - SHA; warm reuse passes checksum and live version/health checks; invalid or - missing cache builds from pinned source. -- [ ] This selector/workflow implementation PR runs full fallback on its exact - candidate and proves selector/fan-in adversarial cases. Because this PR - changes the selector, map and workflow, it cannot use its own newly added - mapping to create mapped-mode evidence. -- [ ] The first later, naturally eligible PR confined to the mapped S3 owner - or its mapped tests supplies the exact hosted selected-job custody and timing - observation. Until then, the feature is implemented but the narrow hosted - performance result is unverified; do not claim the five-minute target. -- [ ] Report hosted elapsed time for mapped and full-fallback runs when each - naturally occurs; do not use a synthetic path list or manual selector - override as proof. - -## Risk and review routing - -- Risk class: `L1` -- Required reviewers: `ci_integrity`, `qa`, `test_delta`, `security`, - `documentation` -- Human review focus: completeness of the only enabled consumer closure, - trustworthiness of target/job custody, full fallback, and preserving every - required full/integration check. - -## Evidence - -| Claim | Command or proof | Result | Remaining uncertainty | -|---|---|---|---| -| Current required workflow executes the complete suite | Workflow, lane catalogue and run 36724982896 | Baseline: 9 lanes and 7,918 tests; about 44 minutes | One run is not a long-term distribution | -| Initial S3 validation owner has a complete selective closure | Direct source-consumer trace, adversarial map tests and exact hosted PR | Pending implementation | Hosted mapped mode cannot be exercised by the selector-changing PR; observe first later naturally eligible S3 change | -| Unmapped code and mismatched targets fail closed | Selector, synthetic-merge and fan-in adversarial tests | Pending implementation | Full fallback must be exercised on hosted CI | -| Full suite remains an independent drift audit | Nightly/manual exact-head run | Pending implementation | Audit never attests another PR head | -| Required `test` cannot be skipped or spoofed | Workflow and exact expected-job/fan-in probes | Pending implementation | Recheck branch protection and check name after workflow changes | - -## Review findings - -Initial plan review findings are being resolved before implementation. +- [ ] The current full Backend workflow still runs unchanged on every PR and + `main` push, including all nine lanes, preflight, API/integration proof, and + the required `test` aggregate. +- [ ] The shadow job always reports exact base/head/execution tree, changed + paths, selector/map identity, selected lanes, omitted lanes, and per-lane + reasons; its artifact and summary identify the tested PR head. +- [ ] The initial S3 source change recommends both shared-foundation shards; + each mapped test module resolves to every partition owning its nodes. +- [ ] Commitrail-only changes recommend the shared-foundation shards containing + the policy-semantics tests; mixed known changes union their closures. +- [ ] Unknown paths, shared test support, migrations/schema, dependency and CI + machinery changes, malformed input, stale or missing Git objects, and + selector errors recommend all nine lanes rather than a partial set. +- [ ] Adversarial tests cover malformed/empty path lists, duplicate paths, + unknown files, changed tests, shared fixtures, each protected map/selector + input, stale or mismatched PR targets, and missing lane ownership. +- [ ] Workflow regression tests prove the classifier output cannot condition, + skip, replace, or weaken any full-suite job or required check. +- [ ] A hosted PR run shows the shadow report beside complete passing test + evidence for the same head. Subsequent naturally occurring PRs provide the + representative comparison set; do not infer safety from synthetic paths or + a single S3-only example. +- [ ] Full-suite completeness, real integration checks, authorization, + concurrency, migration and rollback proof remain blocking. Coverage remains + diagnostic only. + +## Risk and review + +- Risk class: `L1` CI/workflow integrity. +- Required tracks: `ci_integrity`, `qa`, `test_delta`, `security`, and + `documentation`, selected through the reviewer matrix. +- Human review focus: proof the report is observational only, partition-aware + lane selection, exact-head binding, and unchanged full-suite enforcement. ## Reconciliation -- Current-source reconciliation: backend CI has nine hash-partitioned lanes, - exact test-result custody, a required final aggregator, and hosted wall-time - evidence. Current protected-branch settings require exact PR statuses, - up-to-date branches and human approval; current docs still require full - suites on every PR. -- Next usable boundary: implement the single reviewed S3-validation mapping, - exact-target manifest, conservative fallback, evidence fan-in and current - policy updates together. The implementation PR must full-fallback and prove - adversarial cases; the first later naturally eligible mapped PR supplies the - selected-mode hosted observation. Never report the performance target before - that observation. -- Remaining risks: no other source owner is mapped. Broad product changes, - newly added tests, selector changes and unclassified dependencies remain - full-suite until their complete consumer closure is demonstrated. +- Current source: nine complete semantic lanes and their integration/fan-in + remain required on PRs and `main`. +- Next boundary: collect shadow classifications against full runs across + representative changes, then review the map and timing evidence before + planning any separate selective-gating change. +- No contribution instructions or product capability claims are relaxed by + this change. No main-push check is removed. diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index 9f4bb1d50..5b3f0128f 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -2,9 +2,9 @@ name: Backend on: pull_request: - schedule: - - cron: "19 3 * * *" - workflow_dispatch: + push: + branches: + - main concurrency: group: backend-${{ github.event.pull_request.number || github.ref }} @@ -18,121 +18,7 @@ env: MINIO_IMAGE: workstream-minio:source jobs: - impact-selection: - runs-on: ubuntu-latest - timeout-minutes: 5 - outputs: - mode: ${{ steps.select.outputs.mode }} - profile: ${{ steps.select.outputs.profile }} - manifest_sha256: ${{ steps.select.outputs.manifest_sha256 }} - started_at: ${{ steps.timing.outputs.started_at }} - steps: - - id: timing - name: Start Backend workflow timing - run: echo "started_at=$(date +%s)" >> "${GITHUB_OUTPUT}" - - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 - with: - ref: ${{ github.sha }} - path: candidate - fetch-depth: 0 - persist-credentials: false - - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 - with: - ref: ${{ github.event.pull_request.base.sha || github.sha }} - path: trusted - fetch-depth: 0 - persist-credentials: false - - - name: Select exact tests using the trusted base map - id: select - env: - EVENT_NAME: ${{ github.event_name }} - EVENT_BASE_SHA: ${{ github.event.pull_request.base.sha || github.sha }} - EVENT_HEAD_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - EXECUTION_SHA: ${{ github.sha }} - shell: bash - run: | - set -euo pipefail - manifest="${RUNNER_TEMP}/backend-impact/selection.json" - mkdir -p "$(dirname "${manifest}")" - args=( - --trusted-root "${GITHUB_WORKSPACE}/trusted" - --candidate-root "${GITHUB_WORKSPACE}/candidate" - --base-sha "${EVENT_BASE_SHA}" - --head-sha "${EVENT_HEAD_SHA}" - --execution-sha "${EXECUTION_SHA}" - --output "${manifest}" - ) - if [[ "${EVENT_NAME}" != pull_request ]]; then - args+=(--force-full) - fi - if [[ -f trusted/backend/scripts/test_impact_selection.py \ - && -f trusted/.ci/test-impact/impact_map.json \ - && -f trusted/.ci/test-impact/run_selected_tests.py ]]; then - python3 trusted/backend/scripts/test_impact_selection.py "${args[@]}" - else - # Bootstrap this workflow change with a static full-suite manifest. - # The trusted base has no selector yet, so no selective mode is possible. - test "${EVENT_NAME}" = pull_request - test "$(git -C candidate rev-parse HEAD)" = "${EXECUTION_SHA}" - test -z "$(git -C candidate status --porcelain)" - python3 - "${manifest}" <<'PY' - import hashlib - import json - import os - from pathlib import Path - import subprocess - import sys - - candidate = Path("candidate") - execution = os.environ["EXECUTION_SHA"] - tree = subprocess.check_output( - ["git", "-C", str(candidate), "rev-parse", f"{execution}^{{tree}}"], - text=True, - ).strip() - payload = { - "schema_version": 1, - "mode": "full", - "infrastructure_profile": "full", - "base_sha": os.environ["EVENT_BASE_SHA"], - "head_sha": os.environ["EVENT_HEAD_SHA"], - "execution_sha": execution, - "execution_tree": tree, - "selected_modules": [], - "expected_jobs": [ - "impact-selection", - "auth-boundary-preflight", - "minio-image", - "lanes", - "full-api-e2e", - ], - } - encoded = (json.dumps(payload, sort_keys=True, separators=(",", ":")) + "\n").encode() - target = Path(sys.argv[1]) - target.write_bytes(encoded) - with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: - output.write("mode=full\n") - output.write("profile=full\n") - output.write(f"manifest_sha256={hashlib.sha256(encoded).hexdigest()}\n") - PY - fi - - - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 - with: - name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} - path: ${{ runner.temp }}/backend-impact/selection.json - if-no-files-found: error - retention-days: 7 - minio-image: - needs: impact-selection - if: >- - ${{ - needs.impact-selection.outputs.mode == 'full' || - (needs.impact-selection.outputs.mode == 'impact' && needs.impact-selection.outputs.profile == 'minio') - }} runs-on: ubuntu-latest timeout-minutes: 20 outputs: @@ -154,8 +40,8 @@ jobs: id: cache uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 with: - path: ${{ runner.temp }}/minio-image/ - key: minio-source-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('docker/minio/**') }} + path: ${{ runner.temp }}/minio-image/minio.tar + key: minio-source-v1-${{ github.sha }}-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('docker/minio/**') }} - name: Build source-pinned MinIO once if: steps.cache.outputs.cache-hit != 'true' @@ -167,42 +53,26 @@ jobs: docker save --output "${RUNNER_TEMP}/minio-image/minio.tar" "${MINIO_IMAGE}" - name: Verify the cached or freshly built provider - env: - MINIO_CACHE_HIT: ${{ steps.cache.outputs.cache-hit }} shell: bash run: | set -euo pipefail - verify_provider() { - if [[ "${MINIO_CACHE_HIT}" == true ]]; then - (cd "${RUNNER_TEMP}/minio-image" && sha256sum --check minio.tar.sha256) || return 1 + docker load --input "${RUNNER_TEMP}/minio-image/minio.tar" + docker run --rm "${MINIO_IMAGE}" --version + docker run --detach --rm --name minio-build-probe \ + --publish 127.0.0.1:9000:9000 \ + --env MINIO_ROOT_USER=workstream-minio \ + --env MINIO_ROOT_PASSWORD=workstream-minio-secret-key \ + "${MINIO_IMAGE}" server /data --address :9000 + trap 'docker logs minio-build-probe; docker stop minio-build-probe' EXIT + for attempt in $(seq 1 60); do + if curl --fail --silent http://127.0.0.1:9000/minio/health/live >/dev/null; then + cd "${RUNNER_TEMP}/minio-image" + sha256sum minio.tar > minio.tar.sha256 + exit 0 fi - docker load --input "${RUNNER_TEMP}/minio-image/minio.tar" || return 1 - docker run --rm "${MINIO_IMAGE}" --version || return 1 - docker run --detach --rm --name minio-build-probe \ - --publish 127.0.0.1:9000:9000 \ - --env MINIO_ROOT_USER=workstream-minio \ - --env MINIO_ROOT_PASSWORD=workstream-minio-secret-key \ - "${MINIO_IMAGE}" server /data --address :9000 || return 1 - for attempt in $(seq 1 60); do - if curl --fail --silent http://127.0.0.1:9000/minio/health/live >/dev/null; then - docker stop minio-build-probe || return 1 - return 0 - fi - sleep 1 - done - docker logs minio-build-probe - docker stop minio-build-probe - return 1 - } - if ! verify_provider; then - rm -f "${RUNNER_TEMP}/minio-image/minio.tar" \ - "${RUNNER_TEMP}/minio-image/minio.tar.sha256" - docker build --tag "${MINIO_IMAGE}" docker/minio - docker save --output "${RUNNER_TEMP}/minio-image/minio.tar" "${MINIO_IMAGE}" - verify_provider - fi - cd "${RUNNER_TEMP}/minio-image" - sha256sum minio.tar > minio.tar.sha256 + sleep 1 + done + exit 1 - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 with: @@ -248,10 +118,16 @@ jobs: shell: bash run: | set -euo pipefail - ruff check app tests scripts - docstr-coverage --config .docstr.yaml + ruff check \ + app/modules/authorization/api \ + scripts/authorization_boundary.py \ + scripts/module_boundaries.py \ + scripts/test_structure_boundary.py \ + tests/architecture/test_authorization_boundary.py \ + tests/architecture/test_module_boundaries.py \ + tests/architecture/test_test_structure_boundary.py python -m scripts.module_boundaries validate \ - --protected-base "${{ github.event.pull_request.base.sha || github.sha }}" + --protected-base "${{ github.event.pull_request.base.sha || github.event.before }}" PYTEST_DISABLE_PLUGIN_AUTOLOAD=1 python -m pytest -q \ -p pytest_asyncio.plugin \ tests/architecture/test_module_boundaries.py \ @@ -263,9 +139,50 @@ jobs: --ledger ../.ci/auth-boundaries/TEST_STRUCTURE_DEBT.json python -m scripts.behavior_ownership validate + impact-report: + if: ${{ github.event_name == 'pull_request' }} + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + steps: + - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 + with: + persist-credentials: false + fetch-depth: 0 + + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 + with: + python-version: "3.12" + + - name: Bind and classify the exact pull request target + env: + PR_BASE_SHA: ${{ github.event.pull_request.base.sha }} + PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: >- + python backend/scripts/test_impact_selection.py + --json .ci/test-impact/report.json + --markdown .ci/test-impact/report.md + + - name: Upload exact-target impact report + id: report + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: backend-test-impact-${{ github.sha }} + path: | + .ci/test-impact/report.json + .ci/test-impact/report.md + if-no-files-found: error + retention-days: 7 + + - name: Link report artifact in run summary + env: + REPORT_URL: ${{ steps.report.outputs.artifact-url }} + run: | + printf '\n[Download exact-head test-impact report](%s)\n' "${REPORT_URL}" >> "${GITHUB_STEP_SUMMARY}" + lanes: - needs: [impact-selection, minio-image] - if: ${{ needs.impact-selection.outputs.mode == 'full' }} + needs: minio-image runs-on: ubuntu-latest timeout-minutes: 45 strategy: @@ -339,6 +256,14 @@ jobs: python -m pip install ruff==0.15.22 test "$(ruff --version)" = "ruff 0.15.22" + - name: Lint + working-directory: backend + run: ruff check app tests scripts + + - name: Docstring coverage + working-directory: backend + run: docstr-coverage --config .docstr.yaml + - name: Download source-pinned MinIO image uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 with: @@ -432,9 +357,9 @@ jobs: if-no-files-found: error retention-days: 7 - full-api-e2e: - needs: [impact-selection, minio-image] - if: ${{ needs.impact-selection.outputs.mode == 'full' }} + test: + if: ${{ always() }} + needs: [auth-boundary-preflight, lanes, minio-image, impact-report] runs-on: ubuntu-latest timeout-minutes: 30 @@ -472,13 +397,17 @@ jobs: with: python-version: "3.12" - - name: Bind exact API-test candidate + - id: identity + name: Bind exact checked-out tree shell: bash run: | set -euo pipefail - test "$(git rev-parse HEAD)" = "${GITHUB_SHA}" + job_start_epoch="$(date +%s)" + tree_sha="$(git rev-parse HEAD)" + test "${tree_sha}" = "${GITHUB_SHA}" test -z "$(git status --porcelain)" - test "$(git rev-parse HEAD^{tree})" = "$(git rev-parse "${GITHUB_SHA}^{tree}")" + echo "job_start_epoch=${job_start_epoch}" >> "${GITHUB_OUTPUT}" + echo "tree_sha=${tree_sha}" >> "${GITHUB_OUTPUT}" - name: Install backend working-directory: backend @@ -510,215 +439,33 @@ jobs: docker logs workstream-minio exit 1 - - name: API contract real API e2e - working-directory: backend - env: - WORKSTREAM_TEST_BROKER_URL: redis://localhost:6380/0 - WORKSTREAM_TEST_ADMIN_DATABASE_URL: postgresql+asyncpg://workstream:workstream@localhost:5433/postgres - WORKSTREAM_TEST_MINIO_ENDPOINT: http://127.0.0.1:9000 - run: >- - python scripts/run_isolated_tests.py - --metadata-json "${RUNNER_TEMP}/api-database.json" - --timeout-seconds 1500 - -- python scripts/api_contract_e2e.py - - impact-pure: - needs: impact-selection - if: ${{ needs.impact-selection.outputs.mode == 'pure' }} - runs-on: ubuntu-latest - timeout-minutes: 20 - steps: - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 - with: - persist-credentials: false - fetch-depth: 0 - - - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 - with: - python-version: "3.12" - - - name: Install backend - working-directory: backend - run: python -m pip install -e ".[dev,agents]" - - - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 - with: - name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} - path: ${{ runner.temp }}/backend-impact - - - name: Run selected policy tests with exact custody - env: - GITHUB_EVENT_PULL_REQUEST_BASE_SHA: ${{ github.event.pull_request.base.sha }} - GITHUB_EVENT_PULL_REQUEST_HEAD_SHA: ${{ github.event.pull_request.head.sha }} - run: >- - python3 .ci/test-impact/run_selected_tests.py - --manifest "${RUNNER_TEMP}/backend-impact/selection.json" - --output "${RUNNER_TEMP}/backend-impact/evidence" - --expected-job impact-pure - - - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + - name: Download separate lane attempts from this run if: ${{ always() }} - with: - name: backend-impact-evidence-${{ github.sha }}-impact-pure-attempt-${{ github.run_attempt }} - path: ${{ runner.temp }}/backend-impact/evidence/ - include-hidden-files: true - if-no-files-found: warn - retention-days: 7 - - impact-s3: - needs: [impact-selection, minio-image] - if: ${{ needs.impact-selection.outputs.mode == 'impact' }} - runs-on: ubuntu-latest - timeout-minutes: 25 - steps: - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 - with: - persist-credentials: false - fetch-depth: 0 - - - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 - with: - python-version: "3.12" - - - name: Bind exact S3-impact candidate - shell: bash - run: | - set -euo pipefail - test "$(git rev-parse HEAD)" = "${GITHUB_SHA}" - test -z "$(git status --porcelain)" - test "$(git rev-parse HEAD^{tree})" = "$(git rev-parse "${GITHUB_SHA}^{tree}")" - - - name: Install backend - working-directory: backend - run: python -m pip install -e ".[dev,agents]" - - - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 - with: - name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} - path: ${{ runner.temp }}/backend-impact - - - name: Download source-pinned MinIO image uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 with: - name: ${{ needs.minio-image.outputs.artifact }} - path: ${{ runner.temp }}/minio-image - - - name: Start real MinIO adapter provider - shell: bash - run: | - set -euo pipefail - (cd "${RUNNER_TEMP}/minio-image" && sha256sum --check minio.tar.sha256) - docker load --input "${RUNNER_TEMP}/minio-image/minio.tar" - docker run --detach --rm --name workstream-minio \ - --publish 127.0.0.1:9000:9000 \ - --env MINIO_ROOT_USER=workstream-minio \ - --env MINIO_ROOT_PASSWORD=workstream-minio-secret-key \ - "${MINIO_IMAGE}" server /data --address :9000 - for attempt in $(seq 1 60); do - if curl --fail --silent http://127.0.0.1:9000/minio/health/live >/dev/null; then - exit 0 - fi - sleep 1 - done - docker logs workstream-minio - exit 1 - - - name: Run selected S3 and Commitrail tests with exact custody - working-directory: backend - env: - GITHUB_EVENT_PULL_REQUEST_BASE_SHA: ${{ github.event.pull_request.base.sha }} - GITHUB_EVENT_PULL_REQUEST_HEAD_SHA: ${{ github.event.pull_request.head.sha }} - WORKSTREAM_TEST_MINIO_ENDPOINT: http://127.0.0.1:9000 - run: >- - python3 ../.ci/test-impact/run_selected_tests.py - --manifest "${RUNNER_TEMP}/backend-impact/selection.json" - --output "${RUNNER_TEMP}/backend-impact/evidence" - --expected-job impact-s3 + pattern: backend-lane-${{ github.sha }}-*-attempt-* + path: backend/.ci/download + merge-multiple: false - - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + - name: Require preflight, impact report and every semantic lane if: ${{ always() }} - with: - name: backend-impact-evidence-${{ github.sha }}-impact-s3-attempt-${{ github.run_attempt }} - path: ${{ runner.temp }}/backend-impact/evidence/ - include-hidden-files: true - if-no-files-found: warn - retention-days: 7 - - test: - if: ${{ always() }} - needs: [impact-selection, auth-boundary-preflight, minio-image, lanes, full-api-e2e, impact-pure, impact-s3] - runs-on: ubuntu-latest - timeout-minutes: 30 - - steps: - - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 - with: - persist-credentials: false - fetch-depth: 0 - - - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 - with: - python-version: "3.12" - - - id: identity - name: Bind exact checked-out tree - shell: bash - run: | - set -euo pipefail - job_start_epoch="$(date +%s)" - tree_sha="$(git rev-parse HEAD)" - test "${tree_sha}" = "${GITHUB_SHA}" - test -z "$(git status --porcelain)" - echo "job_start_epoch=${job_start_epoch}" >> "${GITHUB_OUTPUT}" - echo "tree_sha=${tree_sha}" >> "${GITHUB_OUTPUT}" - - - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 - with: - name: backend-impact-selection-${{ github.sha }}-attempt-${{ github.run_attempt }} - path: .ci/impact-selection - - - name: Validate the exact expected GitHub job inventory env: - SELECTION_SHA256: ${{ needs.impact-selection.outputs.manifest_sha256 }} - IMPACT_SELECTION_RESULT: ${{ needs.impact-selection.result }} - AUTH_BOUNDARY_PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }} - MINIO_IMAGE_RESULT: ${{ needs.minio-image.result }} + PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }} LANES_RESULT: ${{ needs.lanes.result }} - FULL_API_E2E_RESULT: ${{ needs.full-api-e2e.result }} - IMPACT_PURE_RESULT: ${{ needs.impact-pure.result }} - IMPACT_S3_RESULT: ${{ needs.impact-s3.result }} + IMPACT_REPORT_RESULT: ${{ needs.impact-report.result }} + EVENT_NAME: ${{ github.event_name }} shell: bash run: | set -euo pipefail - results="$(python3 - <<'PY' - import json - import os - print(json.dumps({ - "impact-selection": os.environ["IMPACT_SELECTION_RESULT"], - "auth-boundary-preflight": os.environ["AUTH_BOUNDARY_PREFLIGHT_RESULT"], - "minio-image": os.environ["MINIO_IMAGE_RESULT"], - "lanes": os.environ["LANES_RESULT"], - "full-api-e2e": os.environ["FULL_API_E2E_RESULT"], - "impact-pure": os.environ["IMPACT_PURE_RESULT"], - "impact-s3": os.environ["IMPACT_S3_RESULT"], - })) - PY - )" - python3 .ci/test-impact/validate_workflow_jobs.py \ - --manifest .ci/impact-selection/selection.json \ - --manifest-sha256 "${SELECTION_SHA256}" \ - --results-json "${results}" - - - name: Download separate lane attempts from this run - if: ${{ needs.impact-selection.outputs.mode == 'full' }} - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 - with: - pattern: backend-lane-${{ github.sha }}-*-attempt-* - path: backend/.ci/download - merge-multiple: false + test "${PREFLIGHT_RESULT}" = success + test "${LANES_RESULT}" = success + if [[ "${EVENT_NAME}" == pull_request ]]; then + test "${IMPACT_REPORT_RESULT}" = success + else + test "${IMPACT_REPORT_RESULT}" = skipped + fi - name: Merge and independently validate exact lane custody - if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend shell: bash run: | @@ -735,7 +482,6 @@ jobs: --summary-json .ci/test-lanes/run-summary.json - name: Combine semantic-lane coverage exactly once - if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend shell: bash run: | @@ -754,22 +500,26 @@ jobs: done coverage combine - - name: Install coverage tooling for the complete suite - if: ${{ needs.impact-selection.outputs.mode == 'full' }} + - name: API contract real API e2e working-directory: backend - run: python -m pip install -e ".[dev,agents]" + env: + WORKSTREAM_TEST_BROKER_URL: redis://localhost:6380/0 + WORKSTREAM_TEST_ADMIN_DATABASE_URL: postgresql+asyncpg://workstream:workstream@localhost:5433/postgres + WORKSTREAM_TEST_MINIO_ENDPOINT: http://127.0.0.1:9000 + run: >- + python scripts/run_isolated_tests.py + --metadata-json "${RUNNER_TEMP}/api-database.json" + --timeout-seconds 1500 + -- python scripts/api_contract_e2e.py - name: Backend coverage diagnostics (no percentage gate) - if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend run: coverage report --precision=2 - name: Record hosted timing, complete execution and diagnostic coverage - if: ${{ needs.impact-selection.outputs.mode == 'full' }} working-directory: backend env: EXPECTED_HEAD_SHA: ${{ steps.identity.outputs.tree_sha }} - BACKEND_STARTED_AT: ${{ needs.impact-selection.outputs.started_at }} shell: bash run: | set -euo pipefail @@ -896,13 +646,10 @@ jobs: if not timing_value.isdigit(): raise SystemExit("invalid lane start timing") start_epochs.append(int(timing_value)) - backend_started_at = int(os.environ["BACKEND_STARTED_AT"]) - first_lane_start_delay = min(start_epochs) - backend_started_at - total_wall = time.time() - backend_started_at + start_epoch = min(start_epochs) + total_wall = time.time() - start_epoch if not math.isfinite(total_wall) or total_wall < 0: raise SystemExit("invalid Backend hosted wall time") - if first_lane_start_delay < 0: - raise SystemExit("invalid first lane start timing") totals = coverage.get("totals") percent = totals.get("percent_covered") if isinstance(totals, dict) else None if ( @@ -921,7 +668,6 @@ jobs: "global_coverage_percent": float(percent), "global_coverage_sha256": digest(coverage_path), "head_sha": expected_head, - "first_lane_start_delay_seconds": first_lane_start_delay, "run_summary_sha256": digest(summary_path), "slowest_lane_seconds": float(slowest), "timing_target_met": total_wall <= 480, @@ -932,74 +678,6 @@ jobs: ) PY - - name: Download exact impact evidence - if: ${{ needs.impact-selection.outputs.mode == 'pure' }} - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 - with: - name: backend-impact-evidence-${{ github.sha }}-impact-pure-attempt-${{ github.run_attempt }} - path: .ci/impact-evidence - - - name: Validate pure impact evidence - if: ${{ needs.impact-selection.outputs.mode == 'pure' }} - env: - SELECTION_SHA256: ${{ needs.impact-selection.outputs.manifest_sha256 }} - run: >- - python3 .ci/test-impact/run_selected_tests.py - --validate-only - --manifest .ci/impact-selection/selection.json - --evidence .ci/impact-evidence/evidence.json - --expected-job impact-pure - --expected-manifest-sha256 "${SELECTION_SHA256}" - - - name: Download exact S3 impact evidence - if: ${{ needs.impact-selection.outputs.mode == 'impact' }} - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 - with: - name: backend-impact-evidence-${{ github.sha }}-impact-s3-attempt-${{ github.run_attempt }} - path: .ci/impact-evidence - - - name: Validate S3 impact evidence - if: ${{ needs.impact-selection.outputs.mode == 'impact' }} - env: - SELECTION_SHA256: ${{ needs.impact-selection.outputs.manifest_sha256 }} - run: >- - python3 .ci/test-impact/run_selected_tests.py - --validate-only - --manifest .ci/impact-selection/selection.json - --evidence .ci/impact-evidence/evidence.json - --expected-job impact-s3 - --expected-manifest-sha256 "${SELECTION_SHA256}" - - - name: Record selective workflow wall time - if: ${{ needs.impact-selection.outputs.mode != 'full' }} - env: - BACKEND_STARTED_AT: ${{ needs.impact-selection.outputs.started_at }} - IMPACT_MODE: ${{ needs.impact-selection.outputs.mode }} - run: | - set -euo pipefail - python3 - <<'PY' - import json - import os - from pathlib import Path - import time - - evidence_path = Path(".ci/impact-evidence/evidence.json") - evidence = json.loads(evidence_path.read_text(encoding="utf-8")) - elapsed = time.time() - int(os.environ["BACKEND_STARTED_AT"]) - if elapsed < 0: - raise SystemExit("invalid Backend workflow timing") - output = { - "head_sha": evidence["execution_sha"], - "mode": os.environ["IMPACT_MODE"], - "selected_node_count": len(evidence["selected_nodes"]), - "timing_target_met": elapsed <= 300, - "total_backend_wall_seconds": round(elapsed, 3), - } - evidence_path.with_name("hosted-evidence.json").write_text( - json.dumps(output, indent=2, sort_keys=True) + "\n", encoding="utf-8" - ) - PY - - name: Reassert exact tree custody if: ${{ always() }} shell: bash @@ -1016,8 +694,6 @@ jobs: path: | backend/.ci/download/** backend/.ci/test-lanes/** - .ci/impact-selection/** - .ci/impact-evidence/** include-hidden-files: true if-no-files-found: warn retention-days: 7 diff --git a/AGENTS.md b/AGENTS.md index 369dd21c2..d301736d0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -183,14 +183,8 @@ definition or ownership boundary of Workstream. one, identify its behavior and retained proof, or explain why the requirement is obsolete. Remove duplicated assertions and implementation-only tests only when no distinct contract or failure mode is lost. Do not replace the suite - with end-to-end-only tests, skip failures, or set a deletion quota. Backend - CI uses a trusted-base, reviewed source-to-test map: an eligible mapped change - must run its complete enumerated consumer closure with its required real - infrastructure; every unmapped, broad, uncertain, or CI-selection change - runs the complete suite. Full-suite and real integration proof remain - blocking for fallback PRs and scheduled/manual audits. Nightly results never - substitute for exact-PR evidence. See the current CI boundary in - `.commitrail/initiatives/WS-CI-006/OVERVIEW.md`. + with end-to-end-only tests, skip failures, or set a deletion quota. Full-suite + completeness and real integration checks remain blocking. ## Done Criteria diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index cce8a90d0..f237e24cc 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -97,14 +97,8 @@ active queue or approval gate. - Explain the goal, scope, non-goals, and important design decisions. - Keep the change small enough to review. -- Run the relevant behavior tests, lint and type checks. The required hosted - Backend check uses a reviewed source-to-test map: mapped changes run the - complete named consumer closure with its required real services; every - unmapped, broad, uncertain, or CI-selection change falls back to all nine - semantic lanes and the real API integration job. The complete suite also - runs on its scheduled audit and manual dispatch. A scheduled result never - substitutes for checks on your exact PR candidate. Coverage is diagnostic, - not a merge threshold. +- Run the relevant behavior tests, lint and type checks; full hosted suites and + real API drills remain required. Coverage is diagnostic, not a merge threshold. - Preserve security defaults and meaningful negative-path proof. Do not add tests to meet a percentage or count. For test removals/consolidations, name the retained behavioral proof or the retired requirement. Keep real database, diff --git a/backend/scripts/test_impact_selection.py b/backend/scripts/test_impact_selection.py index c00f90d70..a7a7b42ec 100644 --- a/backend/scripts/test_impact_selection.py +++ b/backend/scripts/test_impact_selection.py @@ -1,5 +1,4 @@ -#!/usr/bin/env python3 -"""Build a trusted, exact-target backend test-impact selection manifest.""" +"""Produce a conservative, observational test-impact recommendation.""" from __future__ import annotations @@ -7,302 +6,257 @@ import hashlib import json import os -from pathlib import Path -import re +from pathlib import Path, PurePosixPath import subprocess import sys from typing import Any -SHA_RE = re.compile(r"^[0-9a-f]{40}$") -MAP_RELATIVE_PATH = ".ci/test-impact/impact_map.json" -RUNNER_RELATIVE_PATH = ".ci/test-impact/run_selected_tests.py" -FULL_LANES = [ - "shared_foundations_a", - "shared_foundations_b", - "schema_contracts", - "project_lifecycle_a", - "project_lifecycle_b", - "project_lifecycle_c", - "task_lifecycle_a", - "task_lifecycle_b", - "task_lifecycle_c", -] -POLICY_MODULE = "tests/projects/review_policy/test_semantics.py" -FULL_JOBS = ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"] +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "backend")) +from scripts.test_lane_catalogue import LANES # noqa: E402 -def _full() -> tuple[str, list[str], list[str], str]: - return "full", [], list(FULL_JOBS), "full" +MAP_PATH = ROOT / ".ci/test-impact/impact_map.json" +CATALOGUE_PATH = ROOT / "backend/scripts/test_lane_catalogue.py" +SCRIPT_PATH = Path(__file__).resolve() +ALL_LANES = tuple(lane.name for lane in LANES) +SELECTOR_VERSION = 1 +SHA_LENGTH = 40 class SelectionError(RuntimeError): - """The candidate cannot be safely classified for selective test execution.""" + """Raised when exact-target evidence cannot be established.""" -def _canonical(value: Any) -> bytes: - return (json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n").encode() +def _sha256(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() -def _sha256(value: bytes) -> str: - return hashlib.sha256(value).hexdigest() +def _git(*args: str) -> str: + result = subprocess.run( + ["git", *args], + cwd=ROOT, + check=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + return result.stdout.strip() -def _git(repository: Path, *args: str) -> bytes: - try: - return subprocess.run( - ["git", *args], - cwd=repository, - check=True, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - ).stdout - except (OSError, subprocess.CalledProcessError) as exc: - raise SelectionError("git_object_or_command_unavailable") from exc - - -def classify_paths( - changed_paths: list[str], - impact_map: dict[str, Any], -) -> tuple[str, list[str], list[str], str]: - """Return mode, exact modules, expected jobs, and infrastructure profile.""" - if not changed_paths or len(set(changed_paths)) != len(changed_paths): - return _full() - if any( - not path - or path.startswith("/") - or "\\" in path - or any(part in {"", ".", ".."} for part in path.split("/")) - for path in changed_paths - ): - return _full() - - commitrail_paths = [path for path in changed_paths if path.startswith(".commitrail/")] - source_paths = [path for path in changed_paths if path not in commitrail_paths] - commitrail_modules = impact_map.get("commitrail_test_modules") - if ( - not isinstance(commitrail_modules, list) - or not commitrail_modules - or any(not isinstance(module, str) for module in commitrail_modules) - ): - return _full() - - if not source_paths: - if not commitrail_paths: - return _full() - return ( - "pure", - sorted(set(commitrail_modules)), - ["impact-selection", "auth-boundary-preflight", "impact-pure"], - "none", - ) +def _read_map() -> dict[str, Any]: + raw = MAP_PATH.read_bytes() + value = json.loads(raw) + if not isinstance(value, dict) or value.get("schema_version") != 1: + raise SelectionError("impact map has an unsupported schema") + return value - owners = impact_map.get("owners") - if not isinstance(owners, list): - return _full() - for owner in owners: - if not isinstance(owner, dict): - continue - sources = owner.get("source_paths") - tests = owner.get("test_paths") - modules = owner.get("test_modules") - infrastructure = owner.get("infrastructure") - if ( - not isinstance(sources, list) - or not isinstance(tests, list) - or not isinstance(modules, list) - or not sources - or not modules - or any(not isinstance(path, str) for path in (*sources, *tests)) - or any(not isinstance(module, str) for module in modules) - or infrastructure not in {"minio", "none"} - or len(set(sources)) != len(sources) - or len(set(tests)) != len(tests) - or len(set(modules)) != len(modules) - or any(not path.startswith("backend/tests/") for path in tests) - or sorted(modules) - != sorted(path.removeprefix("backend/") for path in tests) - ): + +def _changed_paths(base: str, head: str) -> tuple[str, list[str]]: + if len(base) != SHA_LENGTH or len(head) != SHA_LENGTH: + raise SelectionError("PR base and head must be full commit SHAs") + merge_base = _git("merge-base", base, head) + raw = subprocess.run( + ["git", "diff", "--name-only", "-z", f"{merge_base}...{head}"], + cwd=ROOT, + check=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ).stdout + paths = [item.decode("utf-8", errors="strict") for item in raw.split(b"\0") if item] + if not paths: + raise SelectionError("target diff contains no changed paths") + if len(paths) != len(set(paths)): + raise SelectionError("target diff contains duplicate paths") + for path in paths: + parsed = PurePosixPath(path) + if parsed.is_absolute() or ".." in parsed.parts or "\\" in path: + raise SelectionError(f"unsafe changed path: {path!r}") + return merge_base, sorted(paths) + + +def _test_path_owner(path: str) -> tuple[str, ...] | None: + if not path.startswith("backend/tests/") or not path.endswith(".py"): + return None + module_path = path.removeprefix("backend/") + owners = tuple(lane.name for lane in LANES if module_path in lane.modules) + return owners or None + + +def classify(paths: list[str], impact_map: dict[str, Any]) -> tuple[list[str], dict[str, list[str]]]: + """Return lane recommendations and auditable per-lane causes.""" + groups = impact_map.get("lane_groups") + source_paths = impact_map.get("source_paths") + prefixes = impact_map.get("path_prefixes") + if not all(isinstance(item, dict) for item in (groups, source_paths, prefixes)): + raise SelectionError("impact map has an invalid structure") + + selected: set[str] = set() + reasons: dict[str, list[str]] = {lane: [] for lane in ALL_LANES} + full_run_reasons: list[str] = [] + + for path in paths: + mapping: dict[str, Any] | None = None + if path in source_paths: + mapping = source_paths[path] + else: + matching_prefixes = [prefix for prefix in prefixes if path.startswith(prefix)] + if matching_prefixes: + mapping = prefixes[max(matching_prefixes, key=len)] + + if mapping is not None: + group_name = mapping.get("lane_group") + lanes = groups.get(group_name) + if not isinstance(lanes, list) or not lanes: + raise SelectionError(f"invalid lane group for {path}") + reason = f"{path}: {mapping.get('reason', 'explicit impact mapping')}" + selected.update(lanes) + for lane in lanes: + reasons[lane].append(reason) continue - permitted_paths = set(sources) | set(tests) - if set(source_paths) <= permitted_paths and set(source_paths) & permitted_paths: - selected_modules = set(modules) - if commitrail_paths: - selected_modules.update(commitrail_modules) - profile = "minio" if infrastructure == "minio" else "none" - job = "impact-s3" if profile == "minio" else "impact-pure" - jobs = ["impact-selection", "auth-boundary-preflight"] - if profile == "minio": - jobs.append("minio-image") - jobs.append(job) - return "impact", sorted(selected_modules), jobs, profile - return _full() - - -def _inventory(repository: Path, execution_sha: str) -> tuple[list[dict[str, str]], str]: - raw = _git(repository, "ls-tree", "-r", "-z", execution_sha, "--", "backend/tests") - inventory: list[dict[str, str]] = [] - for entry in raw.split(b"\0"): - if not entry: + + test_owners = _test_path_owner(path) + if test_owners is not None: + selected.update(test_owners) + for lane in test_owners: + reasons[lane].append(f"{path}: changed test module belongs to this semantic lane") continue - try: - metadata, raw_path = entry.split(b"\t", 1) - mode, kind, blob = metadata.decode("ascii").split(" ") - path = raw_path.decode("utf-8", errors="strict") - except (ValueError, UnicodeDecodeError) as exc: - raise SelectionError("invalid_test_inventory") from exc - name = path.rsplit("/", 1)[-1] - if kind == "blob" and name.startswith("test_") and name.endswith(".py"): - inventory.append({"blob": blob, "path": path}) - inventory.sort(key=lambda row: row["path"]) - if not inventory: - raise SelectionError("empty_test_inventory") - return inventory, _sha256(_canonical(inventory)) - - -def build_manifest( - trusted_root: Path, - candidate_root: Path, - *, - base_sha: str, - head_sha: str, - execution_sha: str, - force_full: bool = False, -) -> dict[str, Any]: - """Bind selection to event commits, merge candidate, trusted map and tests.""" - if any(SHA_RE.fullmatch(value) is None for value in (base_sha, head_sha, execution_sha)): - raise SelectionError("invalid_event_sha") - candidate_head = _git(candidate_root, "rev-parse", "HEAD").decode().strip() - candidate_tree = _git(candidate_root, "rev-parse", "HEAD^{tree}").decode().strip() - candidate_status = _git(candidate_root, "status", "--porcelain").decode() - if candidate_head != execution_sha or candidate_status: - raise SelectionError("candidate_checkout_mismatch") - for value in (base_sha, head_sha, execution_sha): - _git(candidate_root, "cat-file", "-e", f"{value}^{{commit}}") - execution_tree = _git(candidate_root, "rev-parse", f"{execution_sha}^{{tree}}").decode().strip() - if candidate_tree != execution_tree: - raise SelectionError("candidate_tree_mismatch") - if force_full: - if not base_sha == head_sha == execution_sha: - raise SelectionError("invalid_forced_full_target") - merge_base = execution_sha - changed_raw = b"" - else: - parents = ( - _git(candidate_root, "show", "-s", "--format=%P", execution_sha) - .decode() - .strip() - .split() - ) - if parents != [base_sha, head_sha]: - raise SelectionError("execution_parent_mismatch") - merge_base = _git(candidate_root, "merge-base", base_sha, head_sha).decode().strip() - if SHA_RE.fullmatch(merge_base) is None: - raise SelectionError("invalid_merge_base") - changed_raw = _git( - candidate_root, - "diff", - "--name-only", - "-z", - "--no-renames", - merge_base, - head_sha, - "--", - ) - try: - changed_paths = sorted( - path.decode("utf-8", errors="strict") - for path in changed_raw.split(b"\0") - if path - ) - except UnicodeDecodeError as exc: - raise SelectionError("invalid_changed_path_encoding") from exc - map_path = trusted_root / MAP_RELATIVE_PATH - if map_path.is_symlink() or not map_path.is_file(): - raise SelectionError("missing_trusted_impact_map") - map_bytes = map_path.read_bytes() - try: - impact_map = json.loads(map_bytes) - except (UnicodeDecodeError, json.JSONDecodeError) as exc: - raise SelectionError("invalid_trusted_impact_map") from exc - if not isinstance(impact_map, dict) or impact_map.get("schema_version") != 1: - raise SelectionError("invalid_trusted_impact_map") - - mode, modules, jobs, profile = ( - _full() if force_full else classify_paths(changed_paths, impact_map) - ) - inventory, inventory_digest = _inventory(candidate_root, execution_sha) - known_modules = {f"{Path(row['path']).relative_to('backend')}" for row in inventory} - expected_paths: set[str] = set() - for owner in impact_map.get("owners", []): - if isinstance(owner, dict): - expected_paths.update(owner.get("test_modules", [])) - expected_paths.update(impact_map.get("commitrail_test_modules", [])) - if any(module not in known_modules for module in modules): - mode, modules, jobs, profile = _full() - if mode != "full" and any(module not in expected_paths for module in modules): - mode, modules, jobs, profile = _full() - - trusted_script = trusted_root / "backend/scripts/test_impact_selection.py" - if trusted_script.is_symlink() or not trusted_script.is_file(): - raise SelectionError("missing_trusted_selector") - runner_path = trusted_root / RUNNER_RELATIVE_PATH - if runner_path.is_symlink() or not runner_path.is_file(): - raise SelectionError("missing_trusted_runner") + full_run_reasons.append(f"{path}: no reviewed impact mapping; recommend all lanes") + + if full_run_reasons: + selected = set(ALL_LANES) + for lane in ALL_LANES: + reasons[lane].extend(full_run_reasons) + elif not selected: + selected = set(ALL_LANES) + for lane in ALL_LANES: + reasons[lane].append("no safely classifiable changed path; recommend all lanes") + + return [lane for lane in ALL_LANES if lane in selected], reasons + + +def _markdown(report: dict[str, Any]) -> str: + lines = [ + "## Backend test-impact shadow report", + "", + "This is an advisory recommendation only. The complete required backend suite still runs.", + "", + f"- PR base: `{report['base_sha']}`", + f"- PR head: `{report['head_sha']}`", + f"- Workflow execution SHA/tree: `{report['execution_sha']}` / `{report['execution_tree_sha']}`", + f"- Merge base: `{report['merge_base']}`", + f"- Selector version: `{report['selector_version']}`", + f"- Changed-path SHA-256: `{report['changed_paths_sha256']}`", + f"- Lane catalogue SHA-256: `{report['lane_catalogue_sha256']}`", + f"- Impact map SHA-256: `{report['impact_map_sha256']}`", + f"- Classification status: **{report['classification_status']}**", + "", + "### Recommended lanes", + "", + ] + for lane in report["lanes"]: + if lane["selected"]: + lines.append(f"- **{lane['name']}** — " + "; ".join(lane["reasons"])) + else: + lines.append(f"- {lane['name']} — omitted: {lane['omission_reason']}") + lines.extend(["", "### Changed paths", ""]) + lines.extend(f"- `{path}`" for path in report["changed_paths"]) + if report.get("classification_error"): + lines.extend(["", f"Fallback detail: `{report['classification_error']}`"]) + return "\n".join(lines) + "\n" + + +def build_report(base: str, head: str, execution_sha: str) -> dict[str, Any]: + """Bind the recommendation to exact Git targets and selector inputs.""" + if _git("rev-parse", "HEAD") != execution_sha: + raise SelectionError("checked-out execution SHA does not match the workflow target") + tree_sha = _git("rev-parse", f"{execution_sha}^{{tree}}") + execution_parents = _git("show", "-s", "--format=%P", execution_sha).split() + if execution_parents != [base, head]: + raise SelectionError("workflow execution commit is not the exact PR base/head merge") + merge_base, paths = _changed_paths(base, head) + if _git("rev-parse", f"{base}^{{commit}}") != base or _git("rev-parse", f"{head}^{{commit}}") != head: + raise SelectionError("base or head does not resolve to the requested commit") + impact_map = _read_map() + selected, reasons = classify(paths, impact_map) + lanes = [] + for name in ALL_LANES: + is_selected = name in selected + lanes.append( + { + "name": name, + "selected": is_selected, + "reasons": reasons[name] if is_selected else [], + "omission_reason": "all changed paths have reviewed owners outside this lane" if not is_selected else None, + } + ) return { - "base_sha": base_sha, - "changed_paths": changed_paths, - "changed_paths_sha256": _sha256(_canonical(changed_paths)), - "execution_sha": execution_sha, - "execution_tree": execution_tree, - "expected_jobs": jobs, - "head_sha": head_sha, - "impact_map_sha256": _sha256(map_bytes), - "infrastructure_profile": profile, - "merge_base_sha": merge_base, - "mode": mode, "schema_version": 1, - "selected_modules": modules, - "selector_sha256": _sha256(trusted_script.read_bytes()), - "impact_runner_sha256": _sha256(runner_path.read_bytes()), - "test_inventory": inventory, - "test_inventory_sha256": inventory_digest, + "selector_version": SELECTOR_VERSION, + "classification_status": "classified", + "base_sha": base, + "head_sha": head, + "execution_sha": execution_sha, + "execution_tree_sha": tree_sha, + "merge_base": merge_base, + "changed_paths": paths, + "changed_paths_sha256": _sha256("\0".join(paths).encode()), + "selector_sha256": _sha256(SCRIPT_PATH.read_bytes()), + "impact_map_sha256": _sha256(MAP_PATH.read_bytes()), + "lane_catalogue_sha256": _sha256(CATALOGUE_PATH.read_bytes()), + "selected_lanes": selected, + "lanes": lanes, } def main() -> int: parser = argparse.ArgumentParser() - parser.add_argument("--trusted-root", required=True, type=Path) - parser.add_argument("--candidate-root", required=True, type=Path) - parser.add_argument("--base-sha", required=True) - parser.add_argument("--head-sha", required=True) - parser.add_argument("--execution-sha", required=True) - parser.add_argument("--output", required=True, type=Path) - parser.add_argument("--force-full", action="store_true") + parser.add_argument("--base", default=os.environ.get("PR_BASE_SHA", "")) + parser.add_argument("--head", default=os.environ.get("PR_HEAD_SHA", "")) + parser.add_argument("--execution-sha", default=os.environ.get("GITHUB_SHA", "")) + parser.add_argument("--json", type=Path, required=True) + parser.add_argument("--markdown", type=Path, required=True) args = parser.parse_args() + + error: str | None = None try: - manifest = build_manifest( - args.trusted_root.resolve(strict=True), - args.candidate_root.resolve(strict=True), - base_sha=args.base_sha, - head_sha=args.head_sha, - execution_sha=args.execution_sha, - force_full=args.force_full, - ) - args.output.parent.mkdir(parents=True, exist_ok=True) - manifest_bytes = _canonical(manifest) - args.output.write_bytes(manifest_bytes) - github_output = os.environ.get("GITHUB_OUTPUT") - if github_output: - with Path(github_output).open("a", encoding="utf-8") as output: - output.write(f"mode={manifest['mode']}\n") - output.write(f"profile={manifest['infrastructure_profile']}\n") - output.write(f"manifest_sha256={_sha256(manifest_bytes)}\n") - print(json.dumps({key: manifest[key] for key in ("mode", "expected_jobs", "infrastructure_profile")})) - return 0 - except (OSError, SelectionError, subprocess.SubprocessError) as exc: - print(f"impact selection failed closed: {exc}", file=sys.stderr) - return 2 + report = build_report(args.base, args.head, args.execution_sha) + except Exception as exc: # noqa: BLE001 - any classifier fault falls back to all lanes. + error = f"{type(exc).__name__}: {exc}" + report = { + "schema_version": 1, + "selector_version": SELECTOR_VERSION, + "classification_status": "fallback_all_lanes", + "base_sha": args.base, + "head_sha": args.head, + "execution_sha": args.execution_sha, + "execution_tree_sha": None, + "merge_base": None, + "changed_paths": [], + "changed_paths_sha256": None, + "selector_sha256": _sha256(SCRIPT_PATH.read_bytes()), + "impact_map_sha256": _sha256(MAP_PATH.read_bytes()) if MAP_PATH.is_file() else None, + "lane_catalogue_sha256": _sha256(CATALOGUE_PATH.read_bytes()), + "selected_lanes": list(ALL_LANES), + "lanes": [ + {"name": lane, "selected": True, "reasons": ["selection evidence unavailable; fail safe to all lanes"], "omission_reason": None} + for lane in ALL_LANES + ], + "classification_error": error, + } + args.json.parent.mkdir(parents=True, exist_ok=True) + args.markdown.parent.mkdir(parents=True, exist_ok=True) + args.json.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + args.markdown.write_text(_markdown(report), encoding="utf-8") + summary_path = os.environ.get("GITHUB_STEP_SUMMARY") + if summary_path: + with Path(summary_path).open("a", encoding="utf-8") as summary: + summary.write(args.markdown.read_text(encoding="utf-8")) + print(f"Impact report: {args.json}") + if error: + print(f"Impact classification fell back safely to all lanes: {error}", file=sys.stderr) + return 0 if __name__ == "__main__": diff --git a/backend/scripts/test_lane_catalogue.py b/backend/scripts/test_lane_catalogue.py index fe20de231..dd3f1e455 100644 --- a/backend/scripts/test_lane_catalogue.py +++ b/backend/scripts/test_lane_catalogue.py @@ -74,9 +74,9 @@ class TestLane: "tests/test_artifacts.py", "tests/test_assertion_helpers.py", "tests/test_aws_credential_isolation.py", - "tests/test_ci_impact_selection.py", "tests/test_ci_test_lanes.py", "tests/test_ci_lane_catalogue.py", + "tests/test_ci_impact_selection.py", "tests/test_config.py", "tests/test_compensation.py", "tests/compensation/test_adapter_binding_api.py", diff --git a/backend/tests/test_ci_impact_selection.py b/backend/tests/test_ci_impact_selection.py index 70e2e39f7..f320beae8 100644 --- a/backend/tests/test_ci_impact_selection.py +++ b/backend/tests/test_ci_impact_selection.py @@ -1,299 +1,217 @@ -"""Fail-closed tests for the reviewed backend impact classifier.""" +"""Regressions for the observational backend impact recommendation.""" from __future__ import annotations import json -import importlib.util -import hashlib -import subprocess from pathlib import Path +import subprocess +import sys +from unittest.mock import patch -import pytest +import scripts.test_impact_selection as selector +from scripts.test_impact_selection import ALL_LANES, classify +from scripts.test_lane_catalogue import LANES, PARTITIONED_SHARED_LANES -from scripts.test_impact_selection import ( - MAP_RELATIVE_PATH, - SelectionError, - build_manifest, - classify_paths, -) +ROOT = Path(__file__).resolve().parents[2] +IMPACT_MAP = json.loads((ROOT / ".ci/test-impact/impact_map.json").read_text()) -ROOT = Path(__file__).resolve().parents[2] -RUN_SELECTED_PATH = ROOT / ".ci/test-impact/run_selected_tests.py" -IMPACT_MAP = json.loads((ROOT / MAP_RELATIVE_PATH).read_text(encoding="utf-8")) -S3_TEST_PATHS = [ - "backend/tests/test_config.py", - "backend/tests/test_artifact_store_conformance.py", - "backend/tests/test_s3_artifact_store.py", -] -S3_TEST_MODULES = [path.removeprefix("backend/") for path in S3_TEST_PATHS] -POLICY_PATHS = [ - ".commitrail/initiatives/WS-ARCH-001/planning/chunks/WS-ARCH-001-CP07-project-guide-policy-binding.md", - ".commitrail/initiatives/WS-AUTH-001/planning/chunks/WS-AUTH-001-12H-guide-activation.md", -] -POLICY_MODULE = "tests/projects/review_policy/test_semantics.py" - - -_RUN_SELECTED_SPEC = importlib.util.spec_from_file_location( - "ci_run_selected_tests", RUN_SELECTED_PATH -) -assert _RUN_SELECTED_SPEC is not None and _RUN_SELECTED_SPEC.loader is not None -_RUN_SELECTED = importlib.util.module_from_spec(_RUN_SELECTED_SPEC) -_RUN_SELECTED_SPEC.loader.exec_module(_RUN_SELECTED) - - -@pytest.mark.parametrize("path", POLICY_PATHS) -def test_committrail_policy_inputs_select_the_exact_backend_consumer(path: str) -> None: - mode, modules, jobs, profile = classify_paths([path], IMPACT_MAP) - - assert mode == "pure" - assert modules == [POLICY_MODULE] - assert jobs == ["impact-selection", "auth-boundary-preflight", "impact-pure"] - assert profile == "none" - - -def test_unrelated_committrail_metadata_is_still_nonempty() -> None: - mode, modules, jobs, profile = classify_paths( - [".commitrail/changes/some-new-record.md"], IMPACT_MAP - ) +def test_exact_s3_owner_recommends_both_shared_partitions() -> None: + selected, _ = classify(["backend/app/core/s3_validation.py"], IMPACT_MAP) - assert (mode, modules, jobs, profile) == ( - "pure", - [POLICY_MODULE], - ["impact-selection", "auth-boundary-preflight", "impact-pure"], - "none", - ) + assert selected == list(PARTITIONED_SHARED_LANES) -def test_committrail_policy_test_is_unioned_with_mapped_s3_closure() -> None: - mode, modules, jobs, profile = classify_paths( - ["backend/app/core/s3_validation.py", ".commitrail/changes/change.md"], - IMPACT_MAP, - ) +def test_committrail_only_recommends_shared_semantics_partitions() -> None: + selected, _ = classify([".commitrail/changes/ci-example.md"], IMPACT_MAP) - assert mode == "impact" - assert modules == sorted([*S3_TEST_MODULES, POLICY_MODULE]) - assert jobs == ["impact-selection", "auth-boundary-preflight", "minio-image", "impact-s3"] - assert profile == "minio" + assert selected == list(PARTITIONED_SHARED_LANES) -def test_s3_owner_or_mapped_test_edits_select_the_whole_owner_closure() -> None: - for path in ["backend/app/core/s3_validation.py", *S3_TEST_PATHS]: - mode, modules, jobs, profile = classify_paths([path], IMPACT_MAP) - assert mode == "impact" - assert modules == sorted(S3_TEST_MODULES) - assert jobs == ["impact-selection", "auth-boundary-preflight", "minio-image", "impact-s3"] - assert profile == "minio" +def test_mapped_source_and_test_changes_union_their_lane_closures() -> None: + selected, _ = classify( + [ + "backend/app/core/s3_validation.py", + "backend/tests/test_projects.py", + ], + IMPACT_MAP, + ) + assert set(PARTITIONED_SHARED_LANES) <= set(selected) + assert {"project_lifecycle_a", "project_lifecycle_b", "project_lifecycle_c"} <= set(selected) -def test_incomplete_owner_test_closure_falls_back_to_full() -> None: - incomplete_map = json.loads(json.dumps(IMPACT_MAP)) - incomplete_map["owners"][0]["test_modules"] = incomplete_map["owners"][0][ - "test_modules" - ][:-1] - assert classify_paths(["backend/app/core/s3_validation.py"], incomplete_map) == ( - "full", - [], - ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"], - "full", +def test_changed_test_module_selects_every_partition_that_owns_it() -> None: + owners = tuple( + lane.name for lane in LANES if "tests/test_s3_artifact_store.py" in lane.modules ) + selected, _ = classify(["backend/tests/test_s3_artifact_store.py"], IMPACT_MAP) + + assert owners == PARTITIONED_SHARED_LANES + assert selected == list(owners) + -@pytest.mark.parametrize( - "path", - [ - "backend/app/core/config.py", +def test_unknown_source_fixture_and_unmapped_test_fail_safe_to_all_lanes() -> None: + for path in ( + "backend/app/modules/tasks/service.py", "backend/tests/conftest.py", - "backend/tests/test_unmapped.py", - "backend/scripts/test_lane_catalogue.py", - "backend/scripts/test_impact_selection.py", + "backend/tests/test_not_in_catalogue.py", + "docs/operations_backend_testing.md", ".ci/test-impact/impact_map.json", - ".ci/test-impact/run_selected_tests.py", + "backend/scripts/test_impact_selection.py", + "backend/scripts/test_lane_catalogue.py", ".github/workflows/backend.yml", - "docs/roadmap_status.md", - "AGENTS.md", - ], -) -def test_unmapped_or_shared_changes_fail_closed_to_full_suite(path: str) -> None: - assert classify_paths(["backend/app/core/s3_validation.py", path], IMPACT_MAP) == ( - "full", - [], - ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"], - "full", - ) + ): + selected, _ = classify([path], IMPACT_MAP) + assert selected == list(ALL_LANES), path -@pytest.mark.parametrize("paths", [[], ["../escape"], ["/absolute"], ["a\\b"]]) -def test_empty_or_malformed_changes_never_produce_empty_green_selection( - paths: list[str], -) -> None: - mode, modules, jobs, profile = classify_paths(paths, IMPACT_MAP) - assert (mode, modules, jobs, profile) == ( - "full", - [], - ["impact-selection", "auth-boundary-preflight", "minio-image", "lanes", "full-api-e2e"], - "full", +def test_mixed_known_and_unknown_paths_fail_safe_to_all_lanes() -> None: + selected, reasons = classify( + ["backend/app/core/s3_validation.py", "backend/requirements.lock"], IMPACT_MAP ) - -def test_impact_fan_in_requires_predeclared_exact_nodes_and_completion(tmp_path: Path) -> None: - manifest = { - "execution_sha": "a" * 40, - "execution_tree": "b" * 40, - "selected_modules": ["tests/test_config.py"], - "test_inventory_sha256": "c" * 64, - } - nodes = ["tests/test_config.py::test_one"] - - def canonical(value: object) -> bytes: - return (json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n").encode() - - selection_path = tmp_path / "selection.json" - selection_bytes = canonical(manifest) - selection_path.write_bytes(selection_bytes) - payload = { - "execution_sha": manifest["execution_sha"], - "execution_tree": manifest["execution_tree"], - "selected_modules": manifest["selected_modules"], - "selected_nodes": nodes, - "selected_nodes_sha256": hashlib.sha256(canonical(nodes)).hexdigest(), - "selection_manifest_sha256": hashlib.sha256(selection_bytes).hexdigest(), - } - evidence = { - "job": "impact-pure", - "execution_sha": manifest["execution_sha"], - "execution_tree": manifest["execution_tree"], - "test_inventory_sha256": manifest["test_inventory_sha256"], - "selection_manifest_sha256": hashlib.sha256(selection_bytes).hexdigest(), - "exit_code": 0, - "selected_nodes": nodes, - "completed_nodes": nodes, - "observed_collected_nodes": nodes, - "skipped_nodes": [], - "deselected_nodes": [], - "selected_modules": manifest["selected_modules"], - "expected_nodes_sha256": hashlib.sha256(canonical(nodes)).hexdigest(), - "completed_nodes_sha256": hashlib.sha256(canonical(nodes)).hexdigest(), - "elapsed_seconds": 1.0, - "expected_payload_sha256": hashlib.sha256(canonical(payload)).hexdigest(), - } - expected_path = tmp_path / "expected.json" - expected_path.write_bytes(canonical(payload)) - evidence_path = tmp_path / "evidence.json" - evidence_path.write_bytes(canonical(evidence)) - manifest_digest = hashlib.sha256(selection_bytes).hexdigest() - - _RUN_SELECTED.validate_evidence( - selection_path, - evidence_path, - expected_job="impact-pure", - expected_manifest_sha256=manifest_digest, - ) - - evidence_path.write_bytes(canonical({**evidence, "completed_nodes": []})) - with pytest.raises(SelectionError, match="impact_evidence_incomplete"): - _RUN_SELECTED.validate_evidence( - selection_path, - evidence_path, - expected_job="impact-pure", - expected_manifest_sha256=manifest_digest, - ) - - -def test_execution_candidate_must_be_the_event_merge_commit(tmp_path: Path) -> None: - repository = tmp_path / "candidate" - repository.mkdir() - _git(repository, "init", "-q", "-b", "main") - _git(repository, "config", "user.email", "ci@example.invalid") - _git(repository, "config", "user.name", "CI test") - paths = [ - "backend/app/core/s3_validation.py", - *S3_TEST_PATHS, - "backend/tests/projects/review_policy/test_semantics.py", - ] - for path in paths: - target = repository / path - target.parent.mkdir(parents=True, exist_ok=True) - target.write_text("# inventory fixture\n", encoding="utf-8") - _git(repository, "add", "backend") - _git(repository, "commit", "-q", "-m", "base") - base = _git(repository, "rev-parse", "HEAD") - source = repository / "backend/app/core/s3_validation.py" - source.write_text("# changed source\n", encoding="utf-8") - _git(repository, "add", "backend/app/core/s3_validation.py") - _git(repository, "commit", "-q", "-m", "change") - head = _git(repository, "rev-parse", "HEAD") - tree = _git(repository, "rev-parse", "HEAD^{tree}") - execution = subprocess.run( - ["git", "commit-tree", tree, "-p", base, "-p", head], - cwd=repository, - check=True, + assert selected == list(ALL_LANES) + assert all("no reviewed impact mapping" in ";".join(reasons[lane]) for lane in ALL_LANES) + + +def test_empty_path_list_recommends_every_lane() -> None: + selected, _ = classify([], IMPACT_MAP) + + assert selected == list(ALL_LANES) + + +def test_report_binds_execution_tree_and_exact_pr_merge_parents() -> None: + base = "a" * 40 + head = "b" * 40 + execution = "c" * 40 + tree = "d" * 40 + + def git(*args: str) -> str: + if args == ("rev-parse", "HEAD"): + return execution + if args == ("rev-parse", f"{execution}^{{tree}}"): + return tree + if args == ("show", "-s", "--format=%P", execution): + return f"{base} {head}" + if args == ("rev-parse", f"{base}^{{commit}}"): + return base + if args == ("rev-parse", f"{head}^{{commit}}"): + return head + raise AssertionError(args) + + with ( + patch.object(selector, "_git", side_effect=git), + patch.object(selector, "_changed_paths", return_value=(base, [".commitrail/change.md"])), + ): + report = selector.build_report(base, head, execution) + + assert report["base_sha"] == base + assert report["head_sha"] == head + assert report["execution_sha"] == execution + assert report["execution_tree_sha"] == tree + assert report["merge_base"] == base + + +def test_report_rejects_execution_commit_with_stale_pr_parents() -> None: + base = "a" * 40 + head = "b" * 40 + execution = "c" * 40 + + def git(*args: str) -> str: + if args == ("rev-parse", "HEAD"): + return execution + if args == ("rev-parse", f"{execution}^{{tree}}"): + return "d" * 40 + if args == ("show", "-s", "--format=%P", execution): + return f"{base} {'e' * 40}" + raise AssertionError(args) + + with patch.object(selector, "_git", side_effect=git): + try: + selector.build_report(base, head, execution) + except selector.SelectionError as exc: + assert "not the exact PR base/head merge" in str(exc) + else: + raise AssertionError("stale target accepted") + + +def test_duplicate_or_unsafe_git_paths_fail_classification() -> None: + with ( + patch.object(selector, "_git", return_value="a" * 40), + patch.object( + selector.subprocess, + "run", + return_value=type("Result", (), {"stdout": b"backend/app.py\0backend/app.py\0"})(), + ), + ): + try: + selector._changed_paths("a" * 40, "b" * 40) + except selector.SelectionError as exc: + assert "duplicate paths" in str(exc) + else: + raise AssertionError("duplicate Git paths accepted") + + with ( + patch.object(selector, "_git", return_value="a" * 40), + patch.object( + selector.subprocess, + "run", + return_value=type("Result", (), {"stdout": b"../outside\0"})(), + ), + ): + try: + selector._changed_paths("a" * 40, "b" * 40) + except selector.SelectionError as exc: + assert "unsafe changed path" in str(exc) + else: + raise AssertionError("unsafe Git path accepted") + + +def test_cli_reports_all_lane_fallback_when_target_evidence_is_unavailable(tmp_path: Path) -> None: + report_path = tmp_path / "report.json" + markdown_path = tmp_path / "report.md" + result = subprocess.run( + [ + sys.executable, + str(ROOT / "backend/scripts/test_impact_selection.py"), + "--base", + "a" * 40, + "--head", + "b" * 40, + "--execution-sha", + "c" * 40, + "--json", + str(report_path), + "--markdown", + str(markdown_path), + ], + cwd=ROOT, + capture_output=True, + check=False, text=True, - stdout=subprocess.PIPE, - ).stdout.strip() - _git(repository, "reset", "--hard", execution) - - manifest = build_manifest( - ROOT, - repository, - base_sha=base, - head_sha=head, - execution_sha=execution, - ) - assert manifest["merge_base_sha"] == base - assert manifest["execution_sha"] == execution - assert manifest["selected_modules"] == sorted(S3_TEST_MODULES) - assert manifest["expected_jobs"] == [ - "impact-selection", - "auth-boundary-preflight", - "minio-image", - "impact-s3", - ] - forced_full = build_manifest( - ROOT, - repository, - base_sha=execution, - head_sha=execution, - execution_sha=execution, - force_full=True, ) - assert forced_full["mode"] == "full" - assert forced_full["selected_modules"] == [] - assert set(forced_full["expected_jobs"]) == { - "impact-selection", - "auth-boundary-preflight", - "minio-image", - "lanes", - "full-api-e2e", - } - _git(repository, "reset", "--hard", head) - with pytest.raises(SelectionError, match="candidate_checkout_mismatch"): - build_manifest( - ROOT, - repository, - base_sha=base, - head_sha=head, - execution_sha=execution, - ) - _git(repository, "reset", "--hard", head) - with pytest.raises(SelectionError, match="execution_parent_mismatch"): - build_manifest( - ROOT, - repository, - base_sha=base, - head_sha=head, - execution_sha=head, - ) - _git(repository, "reset", "--hard", execution) - - -def _git(repository: Path, *args: str) -> str: - return subprocess.run( - ["git", *args], - cwd=repository, - check=True, - text=True, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - ).stdout.strip() + + assert result.returncode == 0 + report = json.loads(report_path.read_text(encoding="utf-8")) + assert report["classification_status"] == "fallback_all_lanes" + assert report["selected_lanes"] == list(ALL_LANES) + assert "selection evidence unavailable" in markdown_path.read_text(encoding="utf-8") + + +def test_backend_workflow_keeps_report_out_of_lane_execution_control() -> None: + workflow = (ROOT / ".github/workflows/backend.yml").read_text(encoding="utf-8") + impact_job = workflow.split("\n impact-report:\n", 1)[1].split("\n lanes:\n", 1)[0] + lane_job = workflow.split("\n lanes:\n", 1)[1].split("\n test:\n", 1)[0] + aggregate_job = workflow.split("\n test:\n", 1)[1] + + lane_header = lane_job.split(" services:", 1)[0] + assert "\n if:" not in lane_header + assert "needs: impact-report" not in lane_header + assert "selected_lanes" not in lane_job + assert "if: ${{ github.event_name == 'pull_request' }}" in impact_job + assert "backend-test-impact-${{ github.sha }}" in impact_job + assert "needs: [auth-boundary-preflight, lanes, minio-image, impact-report]" in aggregate_job + assert "IMPACT_REPORT_RESULT" in aggregate_job + assert "Require preflight, impact report and every semantic lane" in aggregate_job diff --git a/docker/minio/README.md b/docker/minio/README.md index acc4256d1..fc2dad15e 100644 --- a/docker/minio/README.md +++ b/docker/minio/README.md @@ -28,15 +28,14 @@ requires network access and Go compilation resources; do not launch it on an already memory-constrained workstation. Subsequent builds reuse Docker layers. -Backend CI builds or restores one image cache keyed by the complete -`docker/minio/` source-input digest and runner platform, independent of Git -commit. Unchanged pinned source can therefore be reused by later commits. CI -verifies the archive checksum, server version and live health before supplying -a checksummed image artifact to the full-suite or selected S3 test job. An -invalid cache is rebuilt from pinned source and verified again. Jobs never -substitute a mock storage provider. A missing build, artifact or health check -fails verification. The source-image artifact is independent of test evidence -and cannot make a failed test pass. +Backend CI builds or restores one image cache keyed by the exact Git commit, +this directory's contents and runner platform. Older PR commits cannot supply a +cached executable to a new commit; retries of the same commit can reuse its +image. CI verifies server startup, then supplies a checksummed image +artifact to the existing lanes and aggregate job. Jobs never substitute a mock +storage provider. A missing build, artifact or health check fails verification. +The source-image artifact is independent of test/coverage evidence and cannot +make a failed test lane pass. When updating upstream source, update the commit, archive checksum, provenance and relevant build pins together. Require a fresh image build, health check and diff --git a/docs/operations_backend_testing.md b/docs/operations_backend_testing.md index 540dd15ef..3b0769eaf 100644 --- a/docs/operations_backend_testing.md +++ b/docs/operations_backend_testing.md @@ -108,51 +108,35 @@ creation must use the real admission-backed command, not this fixture. If provisioning fails, confirm the local PostgreSQL provisioning credential can create/drop databases and roles, terminate owned sessions, and reach the named admin database. Diagnostics omit credentials. -## Hosted Backend checks and complete-suite proof - -The required GitHub check remains `Backend / test` on every pull request. The -trusted-base selector in `backend/scripts/test_impact_selection.py` reads the -reviewed map at `.ci/test-impact/impact_map.json` and binds the exact base, head, -merge candidate/tree, changed paths, test inventory, selected modules/nodes and -expected jobs. The map initially permits only -`backend/app/core/s3_validation.py` and its complete configuration, -provider-neutral namespace-conformance and real MinIO adapter test closure. -Every Commitrail change also runs the complete backend policy-semantics module -that reads Commitrail planning inputs. A Commitrail-only change runs that module -plus the always-required authorization/static preflight. - -The initial rollout PR predates the trusted selector on its base revision, so -that one candidate emits a fixed full-suite manifest and runs the complete -required job set; it does not use its new selector for selective execution. - -Any unknown, broad, unclassified, stale, malformed or CI-selection change falls -back to the complete suite. Changes to other documentation, fixtures, tests, -schemas, dependencies or product modules are not implicitly ignored. The full -mode uses all nine semantic lanes plus real PostgreSQL-backed API integration; -it runs on every fallback PR and as a scheduled and manual audit. Nightly/manual -evidence is tied to its own main head and never substitutes for exact-PR tests. -No workflow path filters may hide the required check. - -Full-suite semantic lanes are defined in `backend/scripts/test_lane_catalogue.py`. -Each lane owns its declared test inventory, a digest-pinned PostgreSQL service, -and the shared pinned-source MinIO image. Selected impact jobs bypass PostgreSQL -only for the explicitly mapped DB-free closure; S3 behavior still uses live -MinIO. A service-free final `test` job validates the selector's exact expected -job set and the complete node/job evidence. Missing, duplicate, skipped, -deselected, mismatched or incomplete evidence fails closed. - -Authorization-boundary preflight runs in every mode and includes repository-wide -lint, docstring and module/test-structure checks. The nine matrix jobs no longer -repeat lint or docstring work. Full mode combines the nine lane artifacts once -for diagnostic coverage and runs the real API integration separately with its -own PostgreSQL, Redis and MinIO services. A full Backend run is not repeated on -each protected `main` push: strict, up-to-date PR checks and a human approval -protect merges, while the scheduled and manual runs audit the complete suite. - -The selector, workflow, map, evidence validator or test-catalogue tooling are -not trusted to select themselves: changing any of them forces full mode. A new -source-to-test relation may be mapped only after its complete consumers and -required infrastructure are proven; otherwise it remains in the full fallback. +## Hosted semantic-lane full-suite proof + +For pull requests, the workflow also publishes an exact-target test-impact +shadow report in the Backend run summary and as a seven-day artifact. It records +the PR base/head, merge base, checked-out execution SHA/tree, changed paths, +selector/map/catalogue digests, and lane-level selection reasons. The initial +map is deliberately narrow; changed test modules use the existing lane +catalogue, while shared fixtures, migrations/schema, dependencies, workflow or +catalogue changes and any unmapped path recommend all nine lanes. A +classification error records an all-lanes fallback. The report is observational: +all nine matrix lanes, authorization preflight, API/integration proof and +evidence fan-in remain required and run independently of its recommendation. +Do not use a shadow report as evidence that omitted lanes passed. A later +change to CI selection policy requires representative same-head comparisons, +trusted selection policy, and its own review. Because the report is generated +from the PR candidate, a PR changing its selector, map, catalogue, or workflow +does not validate that changed input. + +The required GitHub check remains `Backend / test`. Nine matrix jobs each own a +digest-pinned PostgreSQL service container, a pinned-source MinIO image, +and exactly one dependency lane. A step-level curl health loop admits MinIO +before collection. This is semantic fan-out, not arbitrary test-count sharding: +lane ownership remains repository-defined and exact. + +The explicit inventory lives in `backend/scripts/test_lane_catalogue.py`. +Authorization preflight runs alongside the nine lanes. The final `test` job +requires both preflight and every lane to succeed before validating evidence and +coverage; failed, cancelled or skipped prerequisites remain blocking. This saves +serial waiting on valid changes at the cost of lane work when preflight fails. Assertion-map validation analyzes each exact historical revision/module once per invocation, then checks every referenced node and assertion against that analysis. It does not cache current source or reuse analysis across validation calls. @@ -162,8 +146,7 @@ to reduce ephemeral reset I/O. A runtime guard verifies the mount, capacity, data directory and enabled `fsync`, `full_page_writes` and `synchronous_commit` before tests. Real SQL, transaction, lock, isolation, and full hosted behavior checks remain. -The schema-contract lane retains disk-backed storage. The final evidence -aggregator is service-free. +The schema-contract lane and aggregate job retain disk-backed databases. This is not a production configuration or proof of host-power-loss durability: [Docker tmpfs data disappears when the container stops](https://docs.docker.com/engine/storage/tmpfs/). An exhausted mount fails the job; it does not silently change storage or skip tests. diff --git a/docs/roadmap_status.md b/docs/roadmap_status.md index 0105090a0..ca8c935f1 100644 --- a/docs/roadmap_status.md +++ b/docs/roadmap_status.md @@ -189,23 +189,19 @@ cannot be reused as post-submission review-gate evidence. See the - Cross-module behavior is moving through explicit public ports under the modular-monolith boundary. New private edges are prohibited and touched debt is reduced incrementally. -- GitHub CI keeps the required Backend `test` check on every PR and uses a - trusted-base, reviewed source-to-test map. The only initial application - mapping is `backend/app/core/s3_validation.py` to its complete configuration, - provider-neutral namespace-conformance and real MinIO adapter test closure; - Commitrail changes also run the backend policy-semantics test that reads its - planning inputs. The exact candidate manifest binds base, head, merge tree, - changed paths, map, test inventory, selected nodes and expected jobs. Unknown, - broad, or CI-selection changes run the full suite. Full mode retains all nine - semantic lanes, real PostgreSQL and API integration proof; it also runs on - scheduled and manual audits, which never attest a different PR. Selected - tests reject skips, deselections and incomplete evidence. Coverage is - diagnostic only, with no percentage or test-count gate. Redundant - coverage-only reruns are removed without deleting their behavior tests. - Lint/docstring and authorization-boundary preflight run once and remain - mandatory at fan-in. Ordinary CI databases use bounded private RAM-backed - storage with write settings verified; schema-contract storage remains - disk-backed. Hosted runtime is measured, not guaranteed. +- GitHub CI distributes the backend suite across semantic lanes and reports a + PR-only shadow impact recommendation; all nine full-suite lanes remain + required. It rejects skipped/deselected tests and requires behavior, boundary + and real API proof. + Coverage is diagnostic only, with no percentage gate or test-count target. + Redundant coverage-only reruns are removed; their tests remain in full-suite lanes. + Its nine-lane allocation uses three project lanes, three task lanes, two + shared-foundation lanes and one schema lane. Database resets batch trigger + commands within the existing transaction while retaining full schema checks; + authorization preflight runs alongside lanes and remains mandatory at fan-in. + Ordinary CI databases use bounded private RAM-backed storage with write + settings verified; schema-contract and aggregate databases stay disk-backed. + Hosted runtime remains measured rather than guaranteed. ### Identity and authorization @@ -522,8 +518,7 @@ the remaining service-actor, profile/link and other AUTH families or the full suite audit. Remaining work includes those AUTH families and the TASK, CHECKER, ART, CON, REV, and tooling audit. The audit requires behavioral proof, not only file splitting or coverage percentages. Real PostgreSQL, concurrency, storage, -and complete fallback/nightly hosted behavior checks remain required; a mapped -PR must pass its entire exact-hosted owner closure instead. Product +and full hosted behavior/integration checks remain required. Product implementation is already progressing alongside this audit with separate file ownership. diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index 8fb530139..397f200b5 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -2,12 +2,9 @@ from __future__ import annotations -import json -import hashlib import os import re import subprocess -import sys import tempfile import textwrap import unittest @@ -124,21 +121,6 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertNotIn("pull_request_review:", workflow) self.assertIn("cancel-in-progress: true", workflow) - self.assertCountEqual( - re.findall( - r"(?m)^ ([a-z][a-z0-9-]*):\s*$", workflow.split("\njobs:\n", 1)[1] - ), - [ - "impact-selection", - "minio-image", - "auth-boundary-preflight", - "lanes", - "full-api-e2e", - "impact-pure", - "impact-s3", - "test", - ], - ) self.assertEqual(len(re.findall(r"(?m)^ matrix:$", workflow)), 1) matrix = re.search(r"(?m)^ matrix:\n((?: {8,}[^\n]*\n|\n)+)", workflow) self.assertIsNotNone(matrix) @@ -155,8 +137,14 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None " - task_lifecycle_b\n" " - task_lifecycle_c", ) - self.assertIn(" needs: [impact-selection, auth-boundary-preflight, minio-image, lanes, full-api-e2e, impact-pure, impact-s3]", workflow) - self.assertIn("Validate the exact expected GitHub job inventory", workflow) + self.assertIn( + " test:\n if: ${{ always() }}\n" + " needs: [auth-boundary-preflight, lanes, minio-image, impact-report]", workflow + ) + self.assertIn("Require preflight, impact report and every semantic lane", workflow) + lanes = workflow.split("\n lanes:\n", 1)[1].split("\n test:\n", 1)[0] + self.assertNotIn("needs: impact-report", lanes) + self.assertNotIn("selected_lanes", lanes) self.assertIn("python -m scripts.merge_test_lane_evidence", workflow) self.assertIn("scripts/validate_test_lane_evidence.py", workflow) self.assertIn( @@ -166,16 +154,6 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertIn("include-hidden-files: true", workflow) self.assertIn("coverage report --precision=2", workflow) self.assertNotIn("fail-under", workflow) - self.assertIn("test_impact_selection.py", workflow) - self.assertIn("backend-impact-selection-", workflow) - self.assertIn("validate_workflow_jobs.py", workflow) - self.assertIn("Bootstrap this workflow change with a static full-suite manifest.", workflow) - self.assertIn('test "${EVENT_NAME}" = pull_request', workflow) - self.assertIn('"mode": "full"', workflow) - self.assertIn("full-api-e2e", workflow) - self.assertNotIn("push:\n branches:\n - main", workflow) - self.assertIn("schedule:", workflow) - self.assertIn("workflow_dispatch:", workflow) self.assertNotIn("pull_request_review:", agent_gates) self.assertNotIn("--require-pr-approval", agent_gates) self.assertNotIn("pull-requests:", agent_gates) @@ -210,15 +188,13 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None def test_minio_source_image_is_built_once_and_shared_without_bypassing_lanes(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") image_job = workflow.split("\n minio-image:\n", 1)[1].split("\n auth-boundary-preflight:\n", 1)[0] - self.assertEqual(workflow.count('docker build --tag "${MINIO_IMAGE}" docker/minio'), 2) - self.assertIn("if ! verify_provider; then", image_job) + self.assertEqual(workflow.count('docker build --tag "${MINIO_IMAGE}" docker/minio'), 1) self.assertNotIn("quay.io/minio", workflow) self.assertIn("hashFiles('docker/minio/**')", image_job) self.assertIn( - "key: minio-source-v2-${{ runner.os }}-${{ runner.arch }}-", + "key: minio-source-v1-${{ github.sha }}-${{ runner.os }}-${{ runner.arch }}-", image_job, ) - self.assertNotIn("${{ github.sha }}", image_job.split("key:", 1)[1].splitlines()[0]) self.assertNotIn("restore-keys:", image_job) self.assertIn("minio-source-${GITHUB_SHA}-${GITHUB_RUN_ATTEMPT}", image_job) self.assertIn("artifact: ${{ steps.identity.outputs.artifact }}", image_job) @@ -226,104 +202,70 @@ def test_minio_source_image_is_built_once_and_shared_without_bypassing_lanes(sel self.assertIn("/minio/health/live", image_job) self.assertIn("if-no-files-found: error", image_job) self.assertNotIn("continue-on-error", image_job) - for name, end in (("lanes", "full-api-e2e"), ("full-api-e2e", "impact-pure"), ("impact-s3", "test")): + for name, end in (("lanes", "test"), ("test", None)): job = workflow.split(f"\n {name}:\n", 1)[1] if end: job = job.split(f"\n {end}:\n", 1)[0] - if name in {"lanes", "full-api-e2e", "impact-s3"}: - self.assertIn("name: ${{ needs.minio-image.outputs.artifact }}", job) - self.assertIn("sha256sum --check minio.tar.sha256", job) - self.assertIn('docker load --input "${RUNNER_TEMP}/minio-image/minio.tar"', job) - self.assertIn('"${MINIO_IMAGE}" server /data --address :9000', job) + self.assertIn("name: ${{ needs.minio-image.outputs.artifact }}", job) + self.assertIn("sha256sum --check minio.tar.sha256", job) + self.assertIn('docker load --input "${RUNNER_TEMP}/minio-image/minio.tar"', job) + self.assertIn('"${MINIO_IMAGE}" server /data --address :9000', job) def test_parallel_preflight_and_lanes_fail_closed_at_fan_in(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") lanes = workflow.split("\n lanes:\n", 1)[1].split("\n test:\n", 1)[0] - self.assertRegex(lanes, r"(?m)^ needs: \[impact-selection, minio-image\]$") + self.assertRegex(lanes, r"(?m)^ needs: minio-image$") self.assertNotIn("needs: auth-boundary-preflight", lanes) - self.assertIn("if: ${{ always() }}", workflow.split("\n test:\n", 1)[1]) - self.assertIn("needs.auth-boundary-preflight.result", workflow) - self.assertIn("needs.lanes.result", workflow) - self.assertIn("needs.impact-s3.result", workflow) - self.assertIn("needs.impact-pure.result", workflow) - - def test_impact_job_inventory_validator_rejects_missing_or_unexpected_jobs(self) -> None: - validator = Path(".ci/test-impact/validate_workflow_jobs.py") - manifest = { - "schema_version": 1, - "mode": "impact", - "infrastructure_profile": "minio", - "expected_jobs": [ - "impact-selection", - "auth-boundary-preflight", - "minio-image", - "impact-s3", - ], - } - statuses = { - "impact-selection": "success", - "auth-boundary-preflight": "success", - "minio-image": "success", - "lanes": "skipped", - "full-api-e2e": "skipped", - "impact-pure": "skipped", - "impact-s3": "success", - } - with tempfile.TemporaryDirectory() as directory: - manifest_path = Path(directory) / "selection.json" - manifest_path.write_text(json.dumps(manifest), encoding="utf-8") - digest = hashlib.sha256(manifest_path.read_bytes()).hexdigest() - results_json = json.dumps(statuses) - valid = subprocess.run( - [ - sys.executable, - str(validator), - "--manifest", - str(manifest_path), - "--manifest-sha256", - digest, - "--results-json", - results_json, - ], - check=False, - capture_output=True, - text=True, - ) - self.assertEqual(valid.returncode, 0, valid.stderr) - - for job, status in (("impact-s3", "skipped"), ("lanes", "success")): - invalid_statuses = {**statuses, job: status} - invalid = subprocess.run( - [ - sys.executable, - str(validator), - "--manifest", - str(manifest_path), - "--manifest-sha256", - digest, - "--results-json", - json.dumps(invalid_statuses), - ], - check=False, - capture_output=True, - text=True, - ) - self.assertNotEqual(invalid.returncode, 0) + step = workflow.split( + " - name: Require preflight, impact report and every semantic lane\n", 1 + )[1].split("\n - name:", 1)[0] + self.assertIn("if: ${{ always() }}", step) + self.assertIn("PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }}", step) + self.assertIn("LANES_RESULT: ${{ needs.lanes.result }}", step) + self.assertIn("IMPACT_REPORT_RESULT: ${{ needs.impact-report.result }}", step) + guard = textwrap.dedent(step.split(" run: |\n", 1)[1]) + for preflight in ("success", "failure", "cancelled", "skipped", "", "unknown"): + for lanes_result in ("success", "failure", "cancelled", "skipped", "", "unknown"): + for event, impact in ( + ("pull_request", "success"), + ("pull_request", "failure"), + ("push", "skipped"), + ("push", "success"), + ): + with self.subTest( + preflight=preflight, + lanes=lanes_result, + event=event, + impact=impact, + ): + result = subprocess.run( + ["bash", "-e", "-c", guard], + env={ + "PREFLIGHT_RESULT": preflight, + "LANES_RESULT": lanes_result, + "IMPACT_REPORT_RESULT": impact, + "EVENT_NAME": event, + }, + capture_output=True, + check=False, + ) + expected = ( + preflight == lanes_result == "success" + and (event == "pull_request" and impact == "success" + or event == "push" and impact == "skipped") + ) + self.assertEqual(result.returncode == 0, expected) def test_postgres_storage_is_bounded_and_disk_contracts_remain(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") lane_service = workflow.split("\n lanes:\n", 1)[1].split("\n steps:", 1)[0] aggregate_service = workflow.split("\n test:\n", 1)[1].split("\n steps:", 1)[0] - api_service = workflow.split("\n full-api-e2e:\n", 1)[1].split("\n steps:", 1)[0] self.assertIn( "${{ matrix.lane != 'schema_contracts' && " "'--tmpfs /var/lib/postgresql/data:rw,nosuid,nodev,noexec,size=2147483648' || '' }}", lane_service, ) self.assertNotIn("--tmpfs", aggregate_service) - self.assertNotIn("services:", aggregate_service) - self.assertIn("postgres:", api_service) - self.assertIn("redis:", api_service) self.assertLess( workflow.index("- name: Verify PostgreSQL CI storage and write settings"), workflow.index("- name: Execute semantic lane"), From 1ec19f235ad2761ac94f87bc7100e3102b37994d Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 08:38:07 +0100 Subject: [PATCH 09/15] fix(ci): keep impact report observational --- .ci/test-impact/impact_map.json | 16 +- .commitrail/initiatives/WS-CI-006/OVERVIEW.md | 12 +- .../initiatives/WS-CI-006/WS-CI-006-01.md | 58 +++--- .github/workflows/backend.yml | 17 +- backend/scripts/test_impact_selection.py | 146 +++++++++------ backend/tests/test_ci_impact_selection.py | 170 ++++++++++++++---- docs/operations_backend_testing.md | 14 +- scripts/git_delta.py | 25 +++ scripts/test_git_delta.py | 22 +++ scripts/test_lightweight_agent_gates.py | 61 +++---- 10 files changed, 365 insertions(+), 176 deletions(-) diff --git a/.ci/test-impact/impact_map.json b/.ci/test-impact/impact_map.json index 5e483ab2e..ac07bc329 100644 --- a/.ci/test-impact/impact_map.json +++ b/.ci/test-impact/impact_map.json @@ -1,20 +1,20 @@ { "schema_version": 1, - "lane_groups": { - "shared_foundations": [ - "shared_foundations_a", - "shared_foundations_b" - ] - }, "source_paths": { "backend/app/core/s3_validation.py": { - "lane_group": "shared_foundations", + "test_modules": [ + "tests/test_config.py", + "tests/test_artifact_store_conformance.py", + "tests/test_s3_artifact_store.py" + ], "reason": "S3 configuration validation is exercised in the shared-foundation partition." } }, "path_prefixes": { ".commitrail/": { - "lane_group": "shared_foundations", + "test_modules": [ + "tests/projects/review_policy/test_semantics.py" + ], "reason": "Commitrail policy semantics are tested in the shared-foundation partition." } } diff --git a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md index ae18573e8..6c0dfc531 100644 --- a/.commitrail/initiatives/WS-CI-006/OVERVIEW.md +++ b/.commitrail/initiatives/WS-CI-006/OVERVIEW.md @@ -23,13 +23,17 @@ exact PR target. - Run [36724982896](https://github.com/Flow-Research/workstream/actions/runs/36724982896) completed 7,918 tests with zero skips/deselections in about 44 minutes. This is one observed run, not a universal baseline. -- Initial explicit mappings are intentionally narrow: exact S3 validation - owner to both shared-foundation shards; Commitrail-only changes to those same - shards; changed test modules to every lane partition that owns them. Any +- Initial explicit mappings are intentionally narrow: S3 validation maps to + three reviewed test modules, and Commitrail-only changes map to the policy + semantics module. Their lane owners are derived from the canonical catalogue; + changed test modules also select every lane partition that owns them. Any unmapped path recommends all nine lanes. - The report is generated by the PR candidate and is not independent policy evidence. PRs that change the selector, map, catalogue or workflow cannot - validate their own changes; no lane is omitted from actual CI. + validate their own changes; no lane is omitted from actual CI. The report job + is not a dependency of the required aggregate, lane fan-in or API end-to-end + proof, so its failure cannot suppress those checks. Branch protection remains + the authority over which standalone job statuses are merge requirements. ## Direction after shadow evidence diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index f0151e7cf..2b2c5ef38 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -42,6 +42,9 @@ this shadow change makes no runtime reduction or timing claim. - `.ci/test-impact/impact_map.json` and `backend/scripts/test_impact_selection.py` for deterministic shadow classification and exact-target report generation. +- `backend/scripts/test_lane_catalogue.py` only as the authoritative inventory + read by the selector; `scripts/git_delta.py` and + `scripts/test_git_delta.py` only for shared byte-safe Git delta primitives. - `backend/tests/test_ci_impact_selection.py` and `scripts/test_lightweight_agent_gates.py` for selector and workflow-shape regressions. @@ -62,25 +65,28 @@ this shadow change makes no runtime reduction or timing claim. ## Shadow selection contract -- Resolve the changed paths from the exact PR base/head and merge base; bind the - report to the base SHA, head SHA, execution SHA/tree, changed-path digest, - selector/map version and digests, selected semantic lanes and rationale. +- Resolve changed paths from the exact PR base/head and merge base; bind a + successfully resolved report to the base SHA, head SHA, execution SHA/tree, + changed-path digest, selector/map version and digests, selected semantic + lanes and rationale. If target identity or changed paths cannot be resolved, + mark unavailable facts as unavailable and recommend all lanes. - Use the current semantic lane catalogue to map test modules to every lane shard that owns their nodes. A changed test module is included in the proposed impact closure. Shared fixtures, schema/migrations, dependencies, workflow, lane catalogue, map/selector changes, unknown source paths, or unavailable Git evidence conservatively recommend all nine lanes. - Initial source mapping is deliberately limited to the reviewed S3 validation - owner `backend/app/core/s3_validation.py`, whose mapped tests are all in the - partitioned `shared_foundations` group. The report must therefore recommend - both `shared_foundations_a` and `shared_foundations_b`, not a hand-picked - test-module subset. Changes to other application source recommend all lanes - until additional consumer closures are demonstrated and explicitly mapped. -- `.commitrail/**` remains in the changed-path report and recommends the lane - group containing `tests/projects/review_policy/test_semantics.py`; mixed - changes union this with the source selection. If any other path is - unclassified, recommend all lanes. Documentation, skills, and agent-policy - paths are not implicitly exempted. + owner `backend/app/core/s3_validation.py`, mapped to + `tests/test_config.py`, `tests/test_artifact_store_conformance.py`, and + `tests/test_s3_artifact_store.py`. Resolve their owning lanes from the + canonical catalogue; do not copy shard membership into the impact map. + Changes to other application source recommend all lanes until additional + consumer closures are demonstrated and explicitly mapped. +- `.commitrail/**` remains in the changed-path report and maps to + `tests/projects/review_policy/test_semantics.py`; derive its owner lanes from + the canonical catalogue. Mixed changes union this with source selection. If + any other path is unclassified, recommend all lanes. Documentation, skills, + and agent-policy paths are not implicitly exempted. - The report explains every selected lane and every omitted lane. Omission is allowed in the *recommendation only* when the exact mapping explains why; missing evidence yields all lanes. The classifier is observational: CI still @@ -99,9 +105,11 @@ this shadow change makes no runtime reduction or timing claim. - [ ] The current full Backend workflow still runs unchanged on every PR and `main` push, including all nine lanes, preflight, API/integration proof, and the required `test` aggregate. -- [ ] The shadow job always reports exact base/head/execution tree, changed - paths, selector/map identity, selected lanes, omitted lanes, and per-lane - reasons; its artifact and summary identify the tested PR head. +- [ ] A successfully resolved target report includes exact base/head/execution + tree, changed paths, selector/map identity, selected lanes, omitted lanes and + per-lane reasons. If target facts cannot be established, it clearly marks + them unavailable and recommends all lanes. The artifact and summary identify + the supplied PR head without asserting unverified facts. - [ ] The initial S3 source change recommends both shared-foundation shards; each mapped test module resolves to every partition owning its nodes. - [ ] Commitrail-only changes recommend the shared-foundation shards containing @@ -113,7 +121,9 @@ this shadow change makes no runtime reduction or timing claim. unknown files, changed tests, shared fixtures, each protected map/selector input, stale or mismatched PR targets, and missing lane ownership. - [ ] Workflow regression tests prove the classifier output cannot condition, - skip, replace, or weaken any full-suite job or required check. + skip, replace, or weaken any full-suite job or required check; in particular, + the report job is not a dependency of the required aggregate, lane fan-in or + API end-to-end step. - [ ] A hosted PR run shows the shadow report beside complete passing test evidence for the same head. Subsequent naturally occurring PRs provide the representative comparison set; do not infer safety from synthetic paths or @@ -122,14 +132,22 @@ this shadow change makes no runtime reduction or timing claim. concurrency, migration and rollback proof remain blocking. Coverage remains diagnostic only. -## Risk and review +## Risk and review routing - Risk class: `L1` CI/workflow integrity. -- Required tracks: `ci_integrity`, `qa`, `test_delta`, `security`, and - `documentation`, selected through the reviewer matrix. +- Required tracks: `ci_integrity`, `qa`, `test_delta`, `security`, + `documentation`, and `reuse_dedup`, selected through the reviewer matrix. - Human review focus: proof the report is observational only, partition-aware lane selection, exact-head binding, and unchanged full-suite enforcement. +## Evidence + +Local focused verification uses the selector behavior tests, the lightweight +workflow-shape and fan-in tests, Ruff on changed Python files, Commitrail +validation, Markdown-link validation and the stale-wording scan. The PR must +also complete the existing hosted Backend workflow on the exact candidate; its +nine lanes, API drill, infrastructure and evidence checks remain unchanged. + ## Reconciliation - Current source: nine complete semantic lanes and their integration/fan-in diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index 5b3f0128f..dab1d6eb6 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -359,7 +359,7 @@ jobs: test: if: ${{ always() }} - needs: [auth-boundary-preflight, lanes, minio-image, impact-report] + needs: [auth-boundary-preflight, lanes, minio-image] runs-on: ubuntu-latest timeout-minutes: 30 @@ -447,23 +447,12 @@ jobs: path: backend/.ci/download merge-multiple: false - - name: Require preflight, impact report and every semantic lane + - name: Require preflight and every semantic lane if: ${{ always() }} env: PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }} LANES_RESULT: ${{ needs.lanes.result }} - IMPACT_REPORT_RESULT: ${{ needs.impact-report.result }} - EVENT_NAME: ${{ github.event_name }} - shell: bash - run: | - set -euo pipefail - test "${PREFLIGHT_RESULT}" = success - test "${LANES_RESULT}" = success - if [[ "${EVENT_NAME}" == pull_request ]]; then - test "${IMPACT_REPORT_RESULT}" = success - else - test "${IMPACT_REPORT_RESULT}" = skipped - fi + run: test "${PREFLIGHT_RESULT}" = success && test "${LANES_RESULT}" = success - name: Merge and independently validate exact lane custody working-directory: backend diff --git a/backend/scripts/test_impact_selection.py b/backend/scripts/test_impact_selection.py index a7a7b42ec..c25f30fe1 100644 --- a/backend/scripts/test_impact_selection.py +++ b/backend/scripts/test_impact_selection.py @@ -7,13 +7,14 @@ import json import os from pathlib import Path, PurePosixPath -import subprocess import sys from typing import Any ROOT = Path(__file__).resolve().parents[2] -sys.path.insert(0, str(ROOT / "backend")) +sys.path.insert(0, str(ROOT)) +sys.path.insert(1, str(ROOT / "backend")) +from scripts.git_delta import resolve_commit, resolve_merge_base, run_checked, run_checked_bytes # noqa: E402 from scripts.test_lane_catalogue import LANES # noqa: E402 MAP_PATH = ROOT / ".ci/test-impact/impact_map.json" @@ -32,37 +33,14 @@ def _sha256(data: bytes) -> str: return hashlib.sha256(data).hexdigest() -def _git(*args: str) -> str: - result = subprocess.run( - ["git", *args], - cwd=ROOT, - check=True, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - ) - return result.stdout.strip() - - -def _read_map() -> dict[str, Any]: - raw = MAP_PATH.read_bytes() - value = json.loads(raw) - if not isinstance(value, dict) or value.get("schema_version") != 1: - raise SelectionError("impact map has an unsupported schema") - return value - - def _changed_paths(base: str, head: str) -> tuple[str, list[str]]: if len(base) != SHA_LENGTH or len(head) != SHA_LENGTH: raise SelectionError("PR base and head must be full commit SHAs") - merge_base = _git("merge-base", base, head) - raw = subprocess.run( + merge_base = resolve_merge_base(base, head, repository_root=ROOT) + raw = run_checked_bytes( ["git", "diff", "--name-only", "-z", f"{merge_base}...{head}"], - cwd=ROOT, - check=True, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - ).stdout + repository_root=ROOT, + ) paths = [item.decode("utf-8", errors="strict") for item in raw.split(b"\0") if item] if not paths: raise SelectionError("target diff contains no changed paths") @@ -85,10 +63,9 @@ def _test_path_owner(path: str) -> tuple[str, ...] | None: def classify(paths: list[str], impact_map: dict[str, Any]) -> tuple[list[str], dict[str, list[str]]]: """Return lane recommendations and auditable per-lane causes.""" - groups = impact_map.get("lane_groups") source_paths = impact_map.get("source_paths") prefixes = impact_map.get("path_prefixes") - if not all(isinstance(item, dict) for item in (groups, source_paths, prefixes)): + if not all(isinstance(item, dict) for item in (source_paths, prefixes)): raise SelectionError("impact map has an invalid structure") selected: set[str] = set() @@ -105,12 +82,20 @@ def classify(paths: list[str], impact_map: dict[str, Any]) -> tuple[list[str], d mapping = prefixes[max(matching_prefixes, key=len)] if mapping is not None: - group_name = mapping.get("lane_group") - lanes = groups.get(group_name) - if not isinstance(lanes, list) or not lanes: - raise SelectionError(f"invalid lane group for {path}") + modules = mapping.get("test_modules") + if not isinstance(modules, list) or not modules: + raise SelectionError(f"invalid test-module mapping for {path}") + owners: set[str] = set() + for module in modules: + if not isinstance(module, str): + raise SelectionError(f"invalid mapped test module for {path}") + module_owners = {lane.name for lane in LANES if module in lane.modules} + if not module_owners: + raise SelectionError(f"mapped test module has no lane owner: {module}") + owners.update(module_owners) + lanes = [lane for lane in ALL_LANES if lane in owners] reason = f"{path}: {mapping.get('reason', 'explicit impact mapping')}" - selected.update(lanes) + selected.update(owners) for lane in lanes: reasons[lane].append(reason) continue @@ -136,50 +121,96 @@ def classify(paths: list[str], impact_map: dict[str, Any]) -> tuple[list[str], d return [lane for lane in ALL_LANES if lane in selected], reasons +def _inline_code(value: object) -> str: + """Render untrusted values as one-line code without Markdown delimiters.""" + visible: list[str] = [] + for character in str(value): + codepoint = ord(character) + if character == "`": + visible.append(r"\x60") + elif character == "\n": + visible.append(r"\n") + elif character == "\r": + visible.append(r"\r") + elif codepoint < 32 or codepoint == 127: + visible.append(rf"\x{codepoint:02x}") + else: + visible.append(character) + return f"`{''.join(visible)}`" + + def _markdown(report: dict[str, Any]) -> str: lines = [ "## Backend test-impact shadow report", "", "This is an advisory recommendation only. The complete required backend suite still runs.", "", - f"- PR base: `{report['base_sha']}`", - f"- PR head: `{report['head_sha']}`", - f"- Workflow execution SHA/tree: `{report['execution_sha']}` / `{report['execution_tree_sha']}`", - f"- Merge base: `{report['merge_base']}`", - f"- Selector version: `{report['selector_version']}`", - f"- Changed-path SHA-256: `{report['changed_paths_sha256']}`", - f"- Lane catalogue SHA-256: `{report['lane_catalogue_sha256']}`", - f"- Impact map SHA-256: `{report['impact_map_sha256']}`", - f"- Classification status: **{report['classification_status']}**", + f"- PR base: {_inline_code(report['base_sha'])}", + f"- PR head: {_inline_code(report['head_sha'])}", + "- Workflow execution SHA/tree: " + f"{_inline_code(report['execution_sha'])} / {_inline_code(report['execution_tree_sha'])}", + f"- Merge base: {_inline_code(report['merge_base'])}", + f"- Selector version: {_inline_code(report['selector_version'])}", + f"- Changed-path SHA-256: {_inline_code(report['changed_paths_sha256'])}", + f"- Lane catalogue SHA-256: {_inline_code(report['lane_catalogue_sha256'])}", + f"- Impact map SHA-256: {_inline_code(report['impact_map_sha256'])}", + f"- Classification status: {_inline_code(report['classification_status'])}", "", "### Recommended lanes", "", ] for lane in report["lanes"]: if lane["selected"]: - lines.append(f"- **{lane['name']}** — " + "; ".join(lane["reasons"])) + reason_text = "; ".join(_inline_code(reason) for reason in lane["reasons"]) + lines.append(f"- {_inline_code(lane['name'])} — {reason_text}") else: - lines.append(f"- {lane['name']} — omitted: {lane['omission_reason']}") + lines.append( + f"- {_inline_code(lane['name'])} — omitted: " + f"{_inline_code(lane['omission_reason'])}" + ) lines.extend(["", "### Changed paths", ""]) - lines.extend(f"- `{path}`" for path in report["changed_paths"]) + lines.extend(f"- {_inline_code(path)}" for path in report["changed_paths"]) if report.get("classification_error"): - lines.extend(["", f"Fallback detail: `{report['classification_error']}`"]) + lines.extend(["", f"Fallback detail: {_inline_code(report['classification_error'])}"]) return "\n".join(lines) + "\n" def build_report(base: str, head: str, execution_sha: str) -> dict[str, Any]: """Bind the recommendation to exact Git targets and selector inputs.""" - if _git("rev-parse", "HEAD") != execution_sha: + if resolve_commit("HEAD", repository_root=ROOT) != execution_sha: raise SelectionError("checked-out execution SHA does not match the workflow target") - tree_sha = _git("rev-parse", f"{execution_sha}^{{tree}}") - execution_parents = _git("show", "-s", "--format=%P", execution_sha).split() + tree_sha = run_checked( + ["git", "rev-parse", f"{execution_sha}^{{tree}}"], repository_root=ROOT + ).strip() + execution_parents = run_checked( + ["git", "show", "-s", "--format=%P", execution_sha], repository_root=ROOT + ).split() if execution_parents != [base, head]: raise SelectionError("workflow execution commit is not the exact PR base/head merge") merge_base, paths = _changed_paths(base, head) - if _git("rev-parse", f"{base}^{{commit}}") != base or _git("rev-parse", f"{head}^{{commit}}") != head: + if ( + resolve_commit(base, repository_root=ROOT) != base + or resolve_commit(head, repository_root=ROOT) != head + ): raise SelectionError("base or head does not resolve to the requested commit") - impact_map = _read_map() - selected, reasons = classify(paths, impact_map) + map_digest: str | None = None + classification_error: str | None = None + try: + map_raw = MAP_PATH.read_bytes() + map_digest = _sha256(map_raw) + impact_map = json.loads(map_raw) + if not isinstance(impact_map, dict) or impact_map.get("schema_version") != 1: + raise SelectionError("impact map has an unsupported schema") + selected, reasons = classify(paths, impact_map) + classification_status = "classified" + except Exception as exc: # noqa: BLE001 - any classifier fault recommends all lanes. + classification_error = f"{type(exc).__name__}: {exc}" + selected = list(ALL_LANES) + reasons = { + lane: ["classification failed after exact target resolution; recommend all lanes"] + for lane in ALL_LANES + } + classification_status = "fallback_all_lanes" lanes = [] for name in ALL_LANES: is_selected = name in selected @@ -194,7 +225,7 @@ def build_report(base: str, head: str, execution_sha: str) -> dict[str, Any]: return { "schema_version": 1, "selector_version": SELECTOR_VERSION, - "classification_status": "classified", + "classification_status": classification_status, "base_sha": base, "head_sha": head, "execution_sha": execution_sha, @@ -203,10 +234,11 @@ def build_report(base: str, head: str, execution_sha: str) -> dict[str, Any]: "changed_paths": paths, "changed_paths_sha256": _sha256("\0".join(paths).encode()), "selector_sha256": _sha256(SCRIPT_PATH.read_bytes()), - "impact_map_sha256": _sha256(MAP_PATH.read_bytes()), + "impact_map_sha256": map_digest, "lane_catalogue_sha256": _sha256(CATALOGUE_PATH.read_bytes()), "selected_lanes": selected, "lanes": lanes, + "classification_error": classification_error, } diff --git a/backend/tests/test_ci_impact_selection.py b/backend/tests/test_ci_impact_selection.py index f320beae8..4f16f7559 100644 --- a/backend/tests/test_ci_impact_selection.py +++ b/backend/tests/test_ci_impact_selection.py @@ -88,22 +88,29 @@ def test_report_binds_execution_tree_and_exact_pr_merge_parents() -> None: execution = "c" * 40 tree = "d" * 40 - def git(*args: str) -> str: - if args == ("rev-parse", "HEAD"): + def resolve_commit(ref: str, *, repository_root: Path) -> str: + assert repository_root == ROOT + if ref == "HEAD": return execution - if args == ("rev-parse", f"{execution}^{{tree}}"): - return tree - if args == ("show", "-s", "--format=%P", execution): - return f"{base} {head}" - if args == ("rev-parse", f"{base}^{{commit}}"): + if ref == base: return base - if args == ("rev-parse", f"{head}^{{commit}}"): + if ref == head: return head - raise AssertionError(args) + raise AssertionError(ref) + + def run_checked(command: list[str], *, repository_root: Path) -> str: + assert repository_root == ROOT + if command == ["git", "rev-parse", f"{execution}^{{tree}}"]: + return tree + if command == ["git", "show", "-s", "--format=%P", execution]: + return f"{base} {head}" + raise AssertionError(command) with ( - patch.object(selector, "_git", side_effect=git), - patch.object(selector, "_changed_paths", return_value=(base, [".commitrail/change.md"])), + patch.object(selector, "resolve_commit", side_effect=resolve_commit), + patch.object(selector, "run_checked", side_effect=run_checked), + patch.object(selector, "resolve_merge_base", return_value=base), + patch.object(selector, "run_checked_bytes", return_value=b".commitrail/change.md\0"), ): report = selector.build_report(base, head, execution) @@ -112,6 +119,27 @@ def git(*args: str) -> str: assert report["execution_sha"] == execution assert report["execution_tree_sha"] == tree assert report["merge_base"] == base + assert report["changed_paths"] == [".commitrail/change.md"] + assert report["changed_paths_sha256"] == selector._sha256(b".commitrail/change.md") + assert report["selector_sha256"] == selector._sha256( + selector.SCRIPT_PATH.read_bytes() + ) + assert report["impact_map_sha256"] == selector._sha256( + selector.MAP_PATH.read_bytes() + ) + assert report["lane_catalogue_sha256"] == selector._sha256( + selector.CATALOGUE_PATH.read_bytes() + ) + assert report["selected_lanes"] == list(PARTITIONED_SHARED_LANES) + assert report["lanes"][0]["reasons"] == [ + ".commitrail/change.md: Commitrail policy semantics are tested in the shared-foundation partition." + ] + assert report["lanes"][1]["reasons"] == report["lanes"][0]["reasons"] + assert all( + lane["omission_reason"] + for lane in report["lanes"] + if not lane["selected"] + ) def test_report_rejects_execution_commit_with_stale_pr_parents() -> None: @@ -119,16 +147,22 @@ def test_report_rejects_execution_commit_with_stale_pr_parents() -> None: head = "b" * 40 execution = "c" * 40 - def git(*args: str) -> str: - if args == ("rev-parse", "HEAD"): + def resolve_commit(ref: str, *, repository_root: Path) -> str: + if ref == "HEAD": return execution - if args == ("rev-parse", f"{execution}^{{tree}}"): + raise AssertionError(ref) + + def run_checked(command: list[str], *, repository_root: Path) -> str: + if command == ["git", "rev-parse", f"{execution}^{{tree}}"]: return "d" * 40 - if args == ("show", "-s", "--format=%P", execution): + if command == ["git", "show", "-s", "--format=%P", execution]: return f"{base} {'e' * 40}" - raise AssertionError(args) + raise AssertionError(command) - with patch.object(selector, "_git", side_effect=git): + with ( + patch.object(selector, "resolve_commit", side_effect=resolve_commit), + patch.object(selector, "run_checked", side_effect=run_checked), + ): try: selector.build_report(base, head, execution) except selector.SelectionError as exc: @@ -137,14 +171,78 @@ def git(*args: str) -> str: raise AssertionError("stale target accepted") +def test_classifier_failure_preserves_exact_target_and_changed_path_evidence() -> None: + base = "a" * 40 + head = "b" * 40 + execution = "c" * 40 + tree = "d" * 40 + path = "backend/app/core/s3_validation.py" + + def resolve_commit(ref: str, *, repository_root: Path) -> str: + del repository_root + return {"HEAD": execution, base: base, head: head}[ref] + + def run_checked(command: list[str], *, repository_root: Path) -> str: + del repository_root + if command[1] == "rev-parse": + return tree + if command[1] == "show": + return f"{base} {head}" + raise AssertionError(command) + + with ( + patch.object(selector, "resolve_commit", side_effect=resolve_commit), + patch.object(selector, "run_checked", side_effect=run_checked), + patch.object(selector, "resolve_merge_base", return_value=base), + patch.object(selector, "run_checked_bytes", return_value=f"{path}\0".encode()), + patch.object(selector, "classify", side_effect=selector.SelectionError("bad map")), + ): + report = selector.build_report(base, head, execution) + + assert report["classification_status"] == "fallback_all_lanes" + assert report["base_sha"] == base + assert report["head_sha"] == head + assert report["execution_sha"] == execution + assert report["execution_tree_sha"] == tree + assert report["merge_base"] == base + assert report["changed_paths"] == [path] + assert report["changed_paths_sha256"] == selector._sha256(path.encode()) + assert report["selector_sha256"] + assert report["impact_map_sha256"] + assert report["lane_catalogue_sha256"] + assert report["selected_lanes"] == list(ALL_LANES) + assert report["classification_error"] == "SelectionError: bad map" + + +def test_markdown_report_escapes_untrusted_paths_and_reasons() -> None: + injected = "evil`\n\n## Forged status" + rendered = selector._markdown( + { + "base_sha": "base", + "head_sha": "head", + "execution_sha": "execution", + "execution_tree_sha": "tree", + "merge_base": "merge", + "selector_version": 1, + "changed_paths_sha256": "digest", + "lane_catalogue_sha256": "catalogue", + "impact_map_sha256": "map", + "classification_status": "classified", + "lanes": [ + {"name": "lane", "selected": True, "reasons": [injected]}, + ], + "changed_paths": [injected], + } + ) + + assert "\n## Forged status" not in rendered + assert r"\x60\n\n## Forged status" in rendered + + def test_duplicate_or_unsafe_git_paths_fail_classification() -> None: with ( - patch.object(selector, "_git", return_value="a" * 40), - patch.object( - selector.subprocess, - "run", - return_value=type("Result", (), {"stdout": b"backend/app.py\0backend/app.py\0"})(), - ), + patch.object(selector, "resolve_merge_base", return_value="a" * 40), + patch.object(selector, "run_checked_bytes", return_value=b"backend/app.py\0backend/app.py\0"), ): try: selector._changed_paths("a" * 40, "b" * 40) @@ -154,12 +252,8 @@ def test_duplicate_or_unsafe_git_paths_fail_classification() -> None: raise AssertionError("duplicate Git paths accepted") with ( - patch.object(selector, "_git", return_value="a" * 40), - patch.object( - selector.subprocess, - "run", - return_value=type("Result", (), {"stdout": b"../outside\0"})(), - ), + patch.object(selector, "resolve_merge_base", return_value="a" * 40), + patch.object(selector, "run_checked_bytes", return_value=b"../outside\0"), ): try: selector._changed_paths("a" * 40, "b" * 40) @@ -197,6 +291,10 @@ def test_cli_reports_all_lane_fallback_when_target_evidence_is_unavailable(tmp_p report = json.loads(report_path.read_text(encoding="utf-8")) assert report["classification_status"] == "fallback_all_lanes" assert report["selected_lanes"] == list(ALL_LANES) + assert report["execution_tree_sha"] is None + assert report["merge_base"] is None + assert report["changed_paths"] == [] + assert report["changed_paths_sha256"] is None assert "selection evidence unavailable" in markdown_path.read_text(encoding="utf-8") @@ -208,10 +306,16 @@ def test_backend_workflow_keeps_report_out_of_lane_execution_control() -> None: lane_header = lane_job.split(" services:", 1)[0] assert "\n if:" not in lane_header - assert "needs: impact-report" not in lane_header + assert "impact-report" not in lane_job assert "selected_lanes" not in lane_job assert "if: ${{ github.event_name == 'pull_request' }}" in impact_job assert "backend-test-impact-${{ github.sha }}" in impact_job - assert "needs: [auth-boundary-preflight, lanes, minio-image, impact-report]" in aggregate_job - assert "IMPACT_REPORT_RESULT" in aggregate_job - assert "Require preflight, impact report and every semantic lane" in aggregate_job + assert "needs: [auth-boundary-preflight, lanes, minio-image]" in aggregate_job + assert "impact-report" not in aggregate_job + assert "Require preflight and every semantic lane" in aggregate_job + assert "API contract real API e2e" in aggregate_job + api_step = aggregate_job.split(" - name: API contract real API e2e\n", 1)[1].split( + "\n - name:", 1 + )[0] + assert "\n if:" not in api_step + assert "scripts/run_isolated_tests.py" in api_step diff --git a/docs/operations_backend_testing.md b/docs/operations_backend_testing.md index 3b0769eaf..5c7e6c9a3 100644 --- a/docs/operations_backend_testing.md +++ b/docs/operations_backend_testing.md @@ -110,16 +110,22 @@ If provisioning fails, confirm the local PostgreSQL provisioning credential can ## Hosted semantic-lane full-suite proof -For pull requests, the workflow also publishes an exact-target test-impact -shadow report in the Backend run summary and as a seven-day artifact. It records -the PR base/head, merge base, checked-out execution SHA/tree, changed paths, -selector/map/catalogue digests, and lane-level selection reasons. The initial +For pull requests, the workflow also publishes a test-impact shadow report in +the Backend run summary and as a seven-day artifact. When exact Git target and +changed-path evidence resolves, it records the PR base/head, merge base, +checked-out execution SHA/tree, changed paths, selector/map/catalogue digests, +and lane-level selection reasons. If that evidence cannot be established, it +marks unavailable fields and recommends all lanes rather than presenting an +unverified exact-target classification. The initial map is deliberately narrow; changed test modules use the existing lane catalogue, while shared fixtures, migrations/schema, dependencies, workflow or catalogue changes and any unmapped path recommend all nine lanes. A classification error records an all-lanes fallback. The report is observational: all nine matrix lanes, authorization preflight, API/integration proof and evidence fan-in remain required and run independently of its recommendation. +The report job is not a dependency of the required aggregate, lane fan-in or API +end-to-end proof; report failure cannot suppress those checks. Branch protection +remains the authority over which standalone job statuses are merge requirements. Do not use a shadow report as evidence that omitted lanes passed. A later change to CI selection policy requires representative same-head comparisons, trusted selection policy, and its own review. Because the report is generated diff --git a/scripts/git_delta.py b/scripts/git_delta.py index 2045fa221..449a79091 100644 --- a/scripts/git_delta.py +++ b/scripts/git_delta.py @@ -41,6 +41,31 @@ def run_checked( return result.stdout +def run_checked_bytes( + command: list[str], + *, + repository_root: Path | None = None, + timeout_seconds: float = 10, +) -> bytes: + """Return byte-exact stdout for Git, preserving NUL-delimited path data.""" + try: + result = subprocess.run( + command, + cwd=repository_root, + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=timeout_seconds, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + code = "GIT_TIMEOUT" if isinstance(exc, subprocess.TimeoutExpired) else "GIT_EXEC_ERROR" + raise GitCommandError(code, command, str(exc)) from exc + if result.returncode != 0: + detail = result.stderr.decode("utf-8", errors="replace").strip() + raise GitCommandError("GIT_COMMAND_FAILED", command, detail) + return result.stdout + + def resolve_commit(ref: str, *, repository_root: Path | None = None) -> str: """Resolve a ref to one full commit SHA or fail closed.""" command = ["git", "rev-parse", "--verify", f"{ref}^{{commit}}"] diff --git a/scripts/test_git_delta.py b/scripts/test_git_delta.py index c0453f699..d05a632ae 100644 --- a/scripts/test_git_delta.py +++ b/scripts/test_git_delta.py @@ -11,6 +11,7 @@ from scripts.git_delta import committed_changed_files from scripts.git_delta import diff_text from scripts.git_delta import numstat +from scripts.git_delta import run_checked_bytes class GitDeltaTests(unittest.TestCase): @@ -48,6 +49,27 @@ def test_committed_delta_is_sorted_and_local_changes_are_optional(self) -> None: self.assertEqual(numstat(base, head, repository_root=root, include_local=False)[:2], (2, 0)) self.assertIn("+++ b/a.txt", diff_text(base, head, repository_root=root, include_local=False)) + def test_checked_byte_output_preserves_nul_delimited_paths(self) -> None: + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + self._git(root, "init") + self._git(root, "config", "user.email", "test@example.com") + self._git(root, "config", "user.name", "Test") + (root / "base.txt").write_text("base\n", encoding="utf-8") + self._git(root, "add", "base.txt") + self._git(root, "commit", "-m", "base") + unusual_path = "line\nbreak.txt" + (root / unusual_path).write_text("content\n", encoding="utf-8") + self._git(root, "add", unusual_path) + self._git(root, "commit", "-m", "add unusual path") + + result = run_checked_bytes( + ["git", "diff", "--name-only", "-z", "HEAD^", "HEAD"], + repository_root=root, + ) + + self.assertEqual(result, b"line\nbreak.txt\0") + @staticmethod def _git(root: Path, *arguments: str) -> str: return subprocess.check_output( diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index 397f200b5..973d69c93 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -139,12 +139,19 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None ) self.assertIn( " test:\n if: ${{ always() }}\n" - " needs: [auth-boundary-preflight, lanes, minio-image, impact-report]", workflow + " needs: [auth-boundary-preflight, lanes, minio-image]", workflow ) - self.assertIn("Require preflight, impact report and every semantic lane", workflow) + self.assertIn("Require preflight and every semantic lane", workflow) lanes = workflow.split("\n lanes:\n", 1)[1].split("\n test:\n", 1)[0] - self.assertNotIn("needs: impact-report", lanes) - self.assertNotIn("selected_lanes", lanes) + aggregate = workflow.split("\n test:\n", 1)[1] + self.assertNotIn("impact-report", lanes) + self.assertNotIn("impact-report", aggregate) + self.assertIn("API contract real API e2e", aggregate) + api_e2e = aggregate.split(" - name: API contract real API e2e\n", 1)[1].split( + "\n - name:", 1 + )[0] + self.assertNotIn("\n if:", api_e2e) + self.assertIn("scripts/run_isolated_tests.py", api_e2e) self.assertIn("python -m scripts.merge_test_lane_evidence", workflow) self.assertIn("scripts/validate_test_lane_evidence.py", workflow) self.assertIn( @@ -217,44 +224,26 @@ def test_parallel_preflight_and_lanes_fail_closed_at_fan_in(self) -> None: self.assertRegex(lanes, r"(?m)^ needs: minio-image$") self.assertNotIn("needs: auth-boundary-preflight", lanes) step = workflow.split( - " - name: Require preflight, impact report and every semantic lane\n", 1 + " - name: Require preflight and every semantic lane\n", 1 )[1].split("\n - name:", 1)[0] self.assertIn("if: ${{ always() }}", step) self.assertIn("PREFLIGHT_RESULT: ${{ needs.auth-boundary-preflight.result }}", step) self.assertIn("LANES_RESULT: ${{ needs.lanes.result }}", step) - self.assertIn("IMPACT_REPORT_RESULT: ${{ needs.impact-report.result }}", step) - guard = textwrap.dedent(step.split(" run: |\n", 1)[1]) + guard = re.search(r"(?m)^ run: (.+)$", step) + self.assertIsNotNone(guard) for preflight in ("success", "failure", "cancelled", "skipped", "", "unknown"): for lanes_result in ("success", "failure", "cancelled", "skipped", "", "unknown"): - for event, impact in ( - ("pull_request", "success"), - ("pull_request", "failure"), - ("push", "skipped"), - ("push", "success"), - ): - with self.subTest( - preflight=preflight, - lanes=lanes_result, - event=event, - impact=impact, - ): - result = subprocess.run( - ["bash", "-e", "-c", guard], - env={ - "PREFLIGHT_RESULT": preflight, - "LANES_RESULT": lanes_result, - "IMPACT_REPORT_RESULT": impact, - "EVENT_NAME": event, - }, - capture_output=True, - check=False, - ) - expected = ( - preflight == lanes_result == "success" - and (event == "pull_request" and impact == "success" - or event == "push" and impact == "skipped") - ) - self.assertEqual(result.returncode == 0, expected) + with self.subTest(preflight=preflight, lanes=lanes_result): + result = subprocess.run( + ["bash", "-e", "-c", guard[1]], + env={"PREFLIGHT_RESULT": preflight, "LANES_RESULT": lanes_result}, + capture_output=True, + check=False, + ) + self.assertEqual( + result.returncode == 0, + preflight == lanes_result == "success", + ) def test_postgres_storage_is_bounded_and_disk_contracts_remain(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") From 75ef4c4871c9c0200bada2c4932737f748b42705 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 08:42:30 +0100 Subject: [PATCH 10/15] fix(ci): classify renames conservatively --- .../initiatives/WS-CI-006/WS-CI-006-01.md | 5 +- .github/workflows/backend.yml | 2 +- backend/scripts/test_impact_selection.py | 8 ++-- backend/tests/test_ci_impact_selection.py | 47 ++++++++++++++++++- 4 files changed, 54 insertions(+), 8 deletions(-) diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index 2b2c5ef38..820808056 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -118,8 +118,9 @@ this shadow change makes no runtime reduction or timing claim. machinery changes, malformed input, stale or missing Git objects, and selector errors recommend all nine lanes rather than a partial set. - [ ] Adversarial tests cover malformed/empty path lists, duplicate paths, - unknown files, changed tests, shared fixtures, each protected map/selector - input, stale or mismatched PR targets, and missing lane ownership. + unknown files, renames from unknown source paths, changed tests, shared + fixtures, each protected map/selector input, stale or mismatched PR targets, + and missing lane ownership. - [ ] Workflow regression tests prove the classifier output cannot condition, skip, replace, or weaken any full-suite job or required check; in particular, the report job is not a dependency of the required aggregate, lane fan-in or diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index dab1d6eb6..fa6dcd5a2 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -168,7 +168,7 @@ jobs: id: report uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 with: - name: backend-test-impact-${{ github.sha }} + name: backend-test-impact-${{ github.sha }}-${{ github.run_attempt }} path: | .ci/test-impact/report.json .ci/test-impact/report.md diff --git a/backend/scripts/test_impact_selection.py b/backend/scripts/test_impact_selection.py index c25f30fe1..552da9a7b 100644 --- a/backend/scripts/test_impact_selection.py +++ b/backend/scripts/test_impact_selection.py @@ -33,13 +33,13 @@ def _sha256(data: bytes) -> str: return hashlib.sha256(data).hexdigest() -def _changed_paths(base: str, head: str) -> tuple[str, list[str]]: +def _changed_paths(base: str, head: str, *, repository_root: Path = ROOT) -> tuple[str, list[str]]: if len(base) != SHA_LENGTH or len(head) != SHA_LENGTH: raise SelectionError("PR base and head must be full commit SHAs") - merge_base = resolve_merge_base(base, head, repository_root=ROOT) + merge_base = resolve_merge_base(base, head, repository_root=repository_root) raw = run_checked_bytes( - ["git", "diff", "--name-only", "-z", f"{merge_base}...{head}"], - repository_root=ROOT, + ["git", "diff", "--no-renames", "--name-only", "-z", f"{merge_base}...{head}"], + repository_root=repository_root, ) paths = [item.decode("utf-8", errors="strict") for item in raw.split(b"\0") if item] if not paths: diff --git a/backend/tests/test_ci_impact_selection.py b/backend/tests/test_ci_impact_selection.py index 4f16f7559..cb64ab371 100644 --- a/backend/tests/test_ci_impact_selection.py +++ b/backend/tests/test_ci_impact_selection.py @@ -67,6 +67,34 @@ def test_unknown_source_fixture_and_unmapped_test_fail_safe_to_all_lanes() -> No assert selected == list(ALL_LANES), path +def test_rename_from_unmapped_source_keeps_deleted_path_and_fails_safe(tmp_path: Path) -> None: + _git(tmp_path, "init") + _git(tmp_path, "config", "user.email", "test@example.com") + _git(tmp_path, "config", "user.name", "Test") + old_path = tmp_path / "backend/app/unknown.py" + old_path.parent.mkdir(parents=True) + old_path.write_text("same content\n", encoding="utf-8") + _git(tmp_path, "add", "backend/app/unknown.py") + _git(tmp_path, "commit", "-m", "add unmapped source") + base = _git(tmp_path, "rev-parse", "HEAD") + + mapped_path = tmp_path / "backend/app/core/s3_validation.py" + mapped_path.parent.mkdir(parents=True) + old_path.rename(mapped_path) + _git(tmp_path, "add", "--all") + _git(tmp_path, "commit", "-m", "rename into mapped source") + head = _git(tmp_path, "rev-parse", "HEAD") + + _, paths = selector._changed_paths(base, head, repository_root=tmp_path) + selected, _ = classify(paths, IMPACT_MAP) + + assert paths == [ + "backend/app/core/s3_validation.py", + "backend/app/unknown.py", + ] + assert selected == list(ALL_LANES) + + def test_mixed_known_and_unknown_paths_fail_safe_to_all_lanes() -> None: selected, reasons = classify( ["backend/app/core/s3_validation.py", "backend/requirements.lock"], IMPACT_MAP @@ -290,11 +318,22 @@ def test_cli_reports_all_lane_fallback_when_target_evidence_is_unavailable(tmp_p assert result.returncode == 0 report = json.loads(report_path.read_text(encoding="utf-8")) assert report["classification_status"] == "fallback_all_lanes" + assert report["base_sha"] == "a" * 40 + assert report["head_sha"] == "b" * 40 + assert report["execution_sha"] == "c" * 40 assert report["selected_lanes"] == list(ALL_LANES) assert report["execution_tree_sha"] is None assert report["merge_base"] is None assert report["changed_paths"] == [] assert report["changed_paths_sha256"] is None + assert report["selector_sha256"] == selector._sha256(selector.SCRIPT_PATH.read_bytes()) + assert report["impact_map_sha256"] == selector._sha256(selector.MAP_PATH.read_bytes()) + assert report["lane_catalogue_sha256"] == selector._sha256( + selector.CATALOGUE_PATH.read_bytes() + ) + markdown = markdown_path.read_text(encoding="utf-8") + assert f"`{'a' * 40}`" in markdown + assert f"`{'b' * 40}`" in markdown assert "selection evidence unavailable" in markdown_path.read_text(encoding="utf-8") @@ -309,7 +348,7 @@ def test_backend_workflow_keeps_report_out_of_lane_execution_control() -> None: assert "impact-report" not in lane_job assert "selected_lanes" not in lane_job assert "if: ${{ github.event_name == 'pull_request' }}" in impact_job - assert "backend-test-impact-${{ github.sha }}" in impact_job + assert "backend-test-impact-${{ github.sha }}-${{ github.run_attempt }}" in impact_job assert "needs: [auth-boundary-preflight, lanes, minio-image]" in aggregate_job assert "impact-report" not in aggregate_job assert "Require preflight and every semantic lane" in aggregate_job @@ -319,3 +358,9 @@ def test_backend_workflow_keeps_report_out_of_lane_execution_control() -> None: )[0] assert "\n if:" not in api_step assert "scripts/run_isolated_tests.py" in api_step + + +def _git(repository: Path, *arguments: str) -> str: + return subprocess.check_output( + ["git", *arguments], cwd=repository, text=True, stderr=subprocess.STDOUT + ).strip() From 5946015eb663cd964c39c59f685e0bc242af2b5c Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 08:56:26 +0100 Subject: [PATCH 11/15] fix(ci): keep selector outside backend ownership inventory --- .commitrail/initiatives/WS-CI-006/WS-CI-006-01.md | 4 ++-- .github/workflows/backend.yml | 2 +- backend/tests/test_ci_impact_selection.py | 8 ++++---- .../backend_test_impact.py | 4 ++-- 4 files changed, 9 insertions(+), 9 deletions(-) rename backend/scripts/test_impact_selection.py => scripts/backend_test_impact.py (99%) diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index 820808056..fd6543c03 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -40,8 +40,8 @@ this shadow change makes no runtime reduction or timing claim. - `.github/workflows/backend.yml` for an always-running PR classifier/report job only; the existing full-suite job graph and commands stay blocking. - `.ci/test-impact/impact_map.json` and - `backend/scripts/test_impact_selection.py` for deterministic shadow - classification and exact-target report generation. + `scripts/backend_test_impact.py` for deterministic shadow classification and + exact-target report generation. - `backend/scripts/test_lane_catalogue.py` only as the authoritative inventory read by the selector; `scripts/git_delta.py` and `scripts/test_git_delta.py` only for shared byte-safe Git delta primitives. diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index fa6dcd5a2..8a436e7c7 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -160,7 +160,7 @@ jobs: PR_BASE_SHA: ${{ github.event.pull_request.base.sha }} PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} run: >- - python backend/scripts/test_impact_selection.py + python scripts/backend_test_impact.py --json .ci/test-impact/report.json --markdown .ci/test-impact/report.md diff --git a/backend/tests/test_ci_impact_selection.py b/backend/tests/test_ci_impact_selection.py index cb64ab371..3954b7133 100644 --- a/backend/tests/test_ci_impact_selection.py +++ b/backend/tests/test_ci_impact_selection.py @@ -8,8 +8,8 @@ import sys from unittest.mock import patch -import scripts.test_impact_selection as selector -from scripts.test_impact_selection import ALL_LANES, classify +import scripts.backend_test_impact as selector +from scripts.backend_test_impact import ALL_LANES, classify from scripts.test_lane_catalogue import LANES, PARTITIONED_SHARED_LANES ROOT = Path(__file__).resolve().parents[2] @@ -59,7 +59,7 @@ def test_unknown_source_fixture_and_unmapped_test_fail_safe_to_all_lanes() -> No "backend/tests/test_not_in_catalogue.py", "docs/operations_backend_testing.md", ".ci/test-impact/impact_map.json", - "backend/scripts/test_impact_selection.py", + "scripts/backend_test_impact.py", "backend/scripts/test_lane_catalogue.py", ".github/workflows/backend.yml", ): @@ -297,7 +297,7 @@ def test_cli_reports_all_lane_fallback_when_target_evidence_is_unavailable(tmp_p result = subprocess.run( [ sys.executable, - str(ROOT / "backend/scripts/test_impact_selection.py"), + str(ROOT / "scripts/backend_test_impact.py"), "--base", "a" * 40, "--head", diff --git a/backend/scripts/test_impact_selection.py b/scripts/backend_test_impact.py similarity index 99% rename from backend/scripts/test_impact_selection.py rename to scripts/backend_test_impact.py index 552da9a7b..5134f0863 100644 --- a/backend/scripts/test_impact_selection.py +++ b/scripts/backend_test_impact.py @@ -1,4 +1,4 @@ -"""Produce a conservative, observational test-impact recommendation.""" +"""Produce a conservative, observational backend test-impact recommendation.""" from __future__ import annotations @@ -10,7 +10,7 @@ import sys from typing import Any -ROOT = Path(__file__).resolve().parents[2] +ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) sys.path.insert(1, str(ROOT / "backend")) From 8fbbca1024a18352a3ee8e1b173611b164b37c35 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 09:09:52 +0100 Subject: [PATCH 12/15] fix(ci): keep impact tests in repository tooling --- .../initiatives/WS-CI-006/WS-CI-006-01.md | 20 +++--- .github/workflows/agent-gates.yml | 1 + backend/scripts/test_lane_catalogue.py | 1 - .../test_backend_test_impact.py | 69 ++++++++++++++++++- scripts/test_lightweight_agent_gates.py | 8 ++- 5 files changed, 86 insertions(+), 13 deletions(-) rename backend/tests/test_ci_impact_selection.py => scripts/test_backend_test_impact.py (83%) diff --git a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md index fd6543c03..e45758b51 100644 --- a/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md +++ b/.commitrail/initiatives/WS-CI-006/WS-CI-006-01.md @@ -37,17 +37,19 @@ this shadow change makes no runtime reduction or timing claim. ### Allowed files -- `.github/workflows/backend.yml` for an always-running PR classifier/report - job only; the existing full-suite job graph and commands stay blocking. +- `.github/workflows/backend.yml` for a PR-only classifier/report job, and + `.github/workflows/agent-gates.yml` to run repository-level selector tests; + the existing backend full-suite graph and commands stay blocking. - `.ci/test-impact/impact_map.json` and `scripts/backend_test_impact.py` for deterministic shadow classification and exact-target report generation. +- `scripts/test_backend_test_impact.py` for selector behavior regressions in + the repository-level lightweight gate suite. - `backend/scripts/test_lane_catalogue.py` only as the authoritative inventory read by the selector; `scripts/git_delta.py` and `scripts/test_git_delta.py` only for shared byte-safe Git delta primitives. -- `backend/tests/test_ci_impact_selection.py` and - `scripts/test_lightweight_agent_gates.py` for selector and workflow-shape - regressions. +- `scripts/test_lightweight_agent_gates.py` for backend workflow-shape and + fan-in regressions. - `docs/operations_backend_testing.md`, the WS-CI-006 initiative overview, this record, and the Commitrail index for the shadow-only operating contract. - `docs/roadmap_status.md` only if its current CI capability statement needs @@ -55,9 +57,11 @@ this shadow change makes no runtime reduction or timing claim. ### Prohibited changes -- No changes to test bodies, assertions, collection, skip/deselect behavior, - coverage policy, current lane partitioning, services, test commands, required - status checks, branch protection, or merge rules. +- No changes to Backend test bodies, assertions, collection, + skip/deselect behavior, coverage policy, current lane partitioning, services, + test commands, required status checks, branch protection, or merge rules. +- The selector regression suite may be added to the existing repository-level + Agent Gates test command; this does not change Backend test collection. - No selector-driven workflow conditions, lane omissions, workflow-level path filters, test execution service, third-party impact product, or mutable historical selection authority. diff --git a/.github/workflows/agent-gates.yml b/.github/workflows/agent-gates.yml index 66237449b..d209b3180 100644 --- a/.github/workflows/agent-gates.yml +++ b/.github/workflows/agent-gates.yml @@ -73,4 +73,5 @@ jobs: scripts.test_commitrail_contribution_paths scripts.test_commitrail_archive_batch scripts.test_commitrail_markdown_structure + scripts.test_backend_test_impact scripts.test_lightweight_agent_gates diff --git a/backend/scripts/test_lane_catalogue.py b/backend/scripts/test_lane_catalogue.py index dd3f1e455..916d0b485 100644 --- a/backend/scripts/test_lane_catalogue.py +++ b/backend/scripts/test_lane_catalogue.py @@ -76,7 +76,6 @@ class TestLane: "tests/test_aws_credential_isolation.py", "tests/test_ci_test_lanes.py", "tests/test_ci_lane_catalogue.py", - "tests/test_ci_impact_selection.py", "tests/test_config.py", "tests/test_compensation.py", "tests/compensation/test_adapter_binding_api.py", diff --git a/backend/tests/test_ci_impact_selection.py b/scripts/test_backend_test_impact.py similarity index 83% rename from backend/tests/test_ci_impact_selection.py rename to scripts/test_backend_test_impact.py index 3954b7133..14a918cc6 100644 --- a/backend/tests/test_ci_impact_selection.py +++ b/scripts/test_backend_test_impact.py @@ -4,15 +4,18 @@ import json from pathlib import Path +import re import subprocess import sys +import tempfile +import unittest from unittest.mock import patch import scripts.backend_test_impact as selector from scripts.backend_test_impact import ALL_LANES, classify from scripts.test_lane_catalogue import LANES, PARTITIONED_SHARED_LANES -ROOT = Path(__file__).resolve().parents[2] +ROOT = Path(__file__).resolve().parents[1] IMPACT_MAP = json.loads((ROOT / ".ci/test-impact/impact_map.json").read_text()) @@ -344,10 +347,15 @@ def test_backend_workflow_keeps_report_out_of_lane_execution_control() -> None: aggregate_job = workflow.split("\n test:\n", 1)[1] lane_header = lane_job.split(" services:", 1)[0] - assert "\n if:" not in lane_header + assert not re.search(r"(?m)^\s*if\s*:", lane_header) assert "impact-report" not in lane_job assert "selected_lanes" not in lane_job assert "if: ${{ github.event_name == 'pull_request' }}" in impact_job + assert "python scripts/backend_test_impact.py" in impact_job + impact_step = impact_job.split( + " - name: Bind and classify the exact pull request target\n", 1 + )[1] + assert "run: >-\n python scripts/backend_test_impact.py" in impact_step assert "backend-test-impact-${{ github.sha }}-${{ github.run_attempt }}" in impact_job assert "needs: [auth-boundary-preflight, lanes, minio-image]" in aggregate_job assert "impact-report" not in aggregate_job @@ -356,7 +364,7 @@ def test_backend_workflow_keeps_report_out_of_lane_execution_control() -> None: api_step = aggregate_job.split(" - name: API contract real API e2e\n", 1)[1].split( "\n - name:", 1 )[0] - assert "\n if:" not in api_step + assert not re.search(r"(?m)^\s*if\s*:", api_step) assert "scripts/run_isolated_tests.py" in api_step @@ -364,3 +372,58 @@ def _git(repository: Path, *arguments: str) -> str: return subprocess.check_output( ["git", *arguments], cwd=repository, text=True, stderr=subprocess.STDOUT ).strip() + + +class BackendTestImpactTests(unittest.TestCase): + """Run repository CI selector checks in the lightweight standard suite.""" + + def test_exact_s3_owner(self) -> None: + test_exact_s3_owner_recommends_both_shared_partitions() + + def test_committrail_only(self) -> None: + test_committrail_only_recommends_shared_semantics_partitions() + + def test_mapped_source_and_test_union(self) -> None: + test_mapped_source_and_test_changes_union_their_lane_closures() + + def test_changed_test_partition_owners(self) -> None: + test_changed_test_module_selects_every_partition_that_owns_it() + + def test_unknown_source_and_test_fallback(self) -> None: + test_unknown_source_fixture_and_unmapped_test_fail_safe_to_all_lanes() + + def test_rename_fallback(self) -> None: + with tempfile.TemporaryDirectory() as temporary: + test_rename_from_unmapped_source_keeps_deleted_path_and_fails_safe( + Path(temporary) + ) + + def test_mixed_unknown_fallback(self) -> None: + test_mixed_known_and_unknown_paths_fail_safe_to_all_lanes() + + def test_empty_path_fallback(self) -> None: + test_empty_path_list_recommends_every_lane() + + def test_exact_report_binding(self) -> None: + test_report_binds_execution_tree_and_exact_pr_merge_parents() + + def test_stale_execution_parent_rejected(self) -> None: + test_report_rejects_execution_commit_with_stale_pr_parents() + + def test_classification_failure_keeps_target_evidence(self) -> None: + test_classifier_failure_preserves_exact_target_and_changed_path_evidence() + + def test_markdown_values_are_escaped(self) -> None: + test_markdown_report_escapes_untrusted_paths_and_reasons() + + def test_duplicate_and_unsafe_paths_rejected(self) -> None: + test_duplicate_or_unsafe_git_paths_fail_classification() + + def test_unavailable_target_report(self) -> None: + with tempfile.TemporaryDirectory() as temporary: + test_cli_reports_all_lane_fallback_when_target_evidence_is_unavailable( + Path(temporary) + ) + + def test_workflow_keeps_report_advisory(self) -> None: + test_backend_workflow_keeps_report_out_of_lane_execution_control() diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index 973d69c93..f3c88f710 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -150,7 +150,13 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None api_e2e = aggregate.split(" - name: API contract real API e2e\n", 1)[1].split( "\n - name:", 1 )[0] - self.assertNotIn("\n if:", api_e2e) + lane_header = lanes.split(" services:", 1)[0] + self.assertNotRegex(lane_header, r"(?m)^\s*if\s*:") + self.assertNotRegex(api_e2e, r"(?m)^\s*if\s*:") + immediate_lane_guard = " lanes:\n if: ${{ false }}\n runs-on: ubuntu-latest\n" + immediate_api_guard = " - name: API contract real API e2e\n if: ${{ false }}\n" + self.assertRegex(immediate_lane_guard, r"(?m)^\s*if\s*:") + self.assertRegex(immediate_api_guard, r"(?m)^\s*if\s*:") self.assertIn("scripts/run_isolated_tests.py", api_e2e) self.assertIn("python -m scripts.merge_test_lane_evidence", workflow) self.assertIn("scripts/validate_test_lane_evidence.py", workflow) From be17f3e87ab0b57fd943a15cdbde2169a3ff602a Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 09:16:37 +0100 Subject: [PATCH 13/15] test(ci): assert impact suite gate registration --- scripts/test_lightweight_agent_gates.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index f3c88f710..1265c05df 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -182,6 +182,7 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertIn("scripts.test_commitrail_contribution_paths", agent_gates) self.assertIn('WORKSTREAM_BASE_SHA: ${{ github.event.pull_request.base.sha }}', agent_gates) self.assertNotIn("scripts.test_chunk_state_sync", agent_gates) + self.assertIn("scripts.test_backend_test_impact", agent_gates) self.assertIn("--require-hashes", agent_gates) self.assertIn("-r .github/requirements/agent-gates.txt", agent_gates) for package in ( From f66e36c8e6ec0a13e27d12dbfabfeab2baf7aa34 Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 09:21:18 +0100 Subject: [PATCH 14/15] test(ci): require active impact test invocation --- scripts/test_lightweight_agent_gates.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index 1265c05df..f1f51348d 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -115,6 +115,9 @@ def test_stale_artifact_rejects_unknown_phase(self) -> None: def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None: workflow = Path(".github/workflows/backend.yml").read_text(encoding="utf-8") agent_gates = Path(".github/workflows/agent-gates.yml").read_text(encoding="utf-8") + lightweight_gate_step = agent_gates.split( + " - name: Lightweight gate regression tests\n", 1 + )[1].split("\n - name:", 1)[0] gate_requirements = Path(".github/requirements/agent-gates.txt").read_text( encoding="utf-8" ) @@ -182,7 +185,13 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertIn("scripts.test_commitrail_contribution_paths", agent_gates) self.assertIn('WORKSTREAM_BASE_SHA: ${{ github.event.pull_request.base.sha }}', agent_gates) self.assertNotIn("scripts.test_chunk_state_sync", agent_gates) - self.assertIn("scripts.test_backend_test_impact", agent_gates) + self.assertIn( + " run: >-\n python3 -m unittest -v\n", lightweight_gate_step + ) + self.assertRegex( + lightweight_gate_step, + r"(?m)^ scripts\.test_backend_test_impact\s*$", + ) self.assertIn("--require-hashes", agent_gates) self.assertIn("-r .github/requirements/agent-gates.txt", agent_gates) for package in ( From 539ea183b0029389802f3abb0c0923028327884f Mon Sep 17 00:00:00 2001 From: Commitrail Probe Date: Thu, 1 Oct 2026 09:24:42 +0100 Subject: [PATCH 15/15] test(ci): parse active impact test custody --- scripts/test_lightweight_agent_gates.py | 32 +++++++++++++++++++++---- 1 file changed, 27 insertions(+), 5 deletions(-) diff --git a/scripts/test_lightweight_agent_gates.py b/scripts/test_lightweight_agent_gates.py index f1f51348d..b13e41f41 100644 --- a/scripts/test_lightweight_agent_gates.py +++ b/scripts/test_lightweight_agent_gates.py @@ -4,6 +4,7 @@ import os import re +import shlex import subprocess import tempfile import textwrap @@ -19,6 +20,20 @@ from scripts.check_stale_workstream_wording import forbidden_path_failures +def _run_command_tokens(step: str) -> list[str]: + """Read active argv from one GitHub Actions folded run block.""" + run_block = re.search(r"(?ms)^ run: >-\n((?: {10,}[^\n]*\n)+)", step) + if run_block is None: + return [] + folded_command = " ".join( + line[10:].strip() for line in run_block[1].splitlines() if line.strip() + ) + shell = shlex.shlex(folded_command, posix=True) + shell.whitespace_split = True + shell.commenters = "#" + return list(shell) + + class LightweightAgentGateTests(unittest.TestCase): """Keep the retained checks executable and cover their core parsing rules.""" @@ -185,12 +200,19 @@ def test_backend_uses_distributed_semantic_lanes_and_stable_fan_in(self) -> None self.assertIn("scripts.test_commitrail_contribution_paths", agent_gates) self.assertIn('WORKSTREAM_BASE_SHA: ${{ github.event.pull_request.base.sha }}', agent_gates) self.assertNotIn("scripts.test_chunk_state_sync", agent_gates) - self.assertIn( - " run: >-\n python3 -m unittest -v\n", lightweight_gate_step + command = _run_command_tokens(lightweight_gate_step) + self.assertEqual(command[:4], ["python3", "-m", "unittest", "-v"]) + modules = command[4:] + self.assertTrue(modules) + self.assertTrue(all(re.fullmatch(r"scripts\.[a-z][a-z0-9_]*", item) for item in modules)) + self.assertIn("scripts.test_backend_test_impact", modules) + commented_selector = lightweight_gate_step.replace( + " scripts.test_backend_test_impact\n", + " # remaining text is intentionally shell-commented\n" + " scripts.test_backend_test_impact\n", ) - self.assertRegex( - lightweight_gate_step, - r"(?m)^ scripts\.test_backend_test_impact\s*$", + self.assertNotIn( + "scripts.test_backend_test_impact", _run_command_tokens(commented_selector) ) self.assertIn("--require-hashes", agent_gates) self.assertIn("-r .github/requirements/agent-gates.txt", agent_gates)