From 71439735b8e1e17d8bfa5c0ee91aa539c949c6bc Mon Sep 17 00:00:00 2001 From: "Jonathan D.A. Jewell" <6759885+hyperpolymath@users.noreply.github.com> Date: Mon, 14 Sep 2026 22:20:13 +0100 Subject: [PATCH] feat(settings): add estate-wide settings drift detector MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Repository settings are not in git. Nothing reviews them, nothing diffs them, and a change leaves no artefact — no commit, no PR, no run. The estate measures 97-100% compliant with config/settings/repo.json at any given moment, so re-applying the canon gains almost nothing; what was missing is anything that NOTICES a setting moving. Measured 2026-09-14: the Actions allowlist on 239 repositories moved from `selected` + EMPTY patterns_allowed + verified_allowed=false to a 92-pattern allowlist with verified_allowed=true, BETWEEN a census and the write meant to fix it. Nobody could say when or by whom, because no instrument was watching. Separately, ZERO of the 301 repos calling mirror.yml had all seven *_MIRROR_ENABLED variables set and roughly a third had NONE, so their mirror runs reported GREEN while backing up nowhere — the forge jobs are gated on `if: vars._MIRROR_ENABLED` and a skip is not a red. Reports only. Opens/updates one tracking issue and never mutates a setting or another repository. Two traps handled explicitly in the code: - The scan loop uses process substitution, NOT `gh repo list | while read`. A pipe runs the loop body in a subshell, so every drift flag is discarded on exit and the detector always reports success — a drift detector structurally incapable of reporting drift. Proved with a control before shipping: process-substitution 1, pipe 0. - The workflow fails loudly when its token cannot read actions/permissions. That read is admin-scoped and security_and_analysis is absent from LIST endpoints entirely, so a weak token yields a FALSE CLEAN — the guard would answer a different question than its consumer. Verified: both `uses:` refs are present verbatim in actions.lock (the lockfile matches literal ref strings, not resolved commits, so an unlisted SHA startup-kills the run); positive control over 6 repos returned 31 drift rows and exit 1. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01QMTyDv9CoJo5PfeNzyp519 --- .github/workflows/settings-drift-detect.yml | 122 ++++++++++++++ scripts/check-settings-drift.sh | 178 ++++++++++++++++++++ 2 files changed, 300 insertions(+) create mode 100644 .github/workflows/settings-drift-detect.yml create mode 100755 scripts/check-settings-drift.sh diff --git a/.github/workflows/settings-drift-detect.yml b/.github/workflows/settings-drift-detect.yml new file mode 100644 index 000000000..4f2d8d47a --- /dev/null +++ b/.github/workflows/settings-drift-detect.yml @@ -0,0 +1,122 @@ +# SPDX-License-Identifier: MPL-2.0 +# Copyright (c) 2026 Jonathan D.A. Jewell +# +# settings-drift-detect.yml — estate-wide sweep for settings that moved unseen. +# +# THE FAULT: repository settings are not in git. Nothing reviews them, nothing +# diffs them, and a change leaves no artefact — no commit, no PR, no run. The +# estate sits at 97–100% compliance with config/settings/repo.json at any given +# moment, so re-applying the canon gains almost nothing. What was missing is +# anything that NOTICES a setting moving. +# +# MEASURED 2026-09-14, the day this was written: the Actions allowlist on 239 +# repositories went from allowed_actions=selected with an EMPTY patterns_allowed +# and verified_allowed=false, to a 92-pattern allowlist with verified_allowed +# =true — between a census and the authorised write meant to fix it. Nobody +# could say when, by whom, or whether it was deliberate, because no instrument +# was watching. Separately, ZERO repos in the 301-repo mirror fleet had all +# seven *_MIRROR_ENABLED variables set, and roughly a third had NONE — so their +# mirror runs reported GREEN while backing up nowhere. +# +# WHY IT MATTERS MORE THAN IT LOOKS: an empty patterns_allowed refuses every +# third-party action, including transitively through a reusable workflow, +# because actions are judged against the CALLER repo's allow-list. Runs then die +# at startup_failure with jobs.total_count == 0 and emit NO check run — a +# required context never reports and the repo looks GREENER than a healthy one. +# A settings change can disarm CI estate-wide while every dashboard stays clean. +# +# This workflow REPORTS ONLY. It opens/updates one tracking issue and never +# mutates a setting or another repository — consistent with the estate guardrail +# that unattended cross-repo mutation is a human decision. Settings are exactly +# where that guardrail matters most. +name: Settings Drift Detect + +on: + schedule: + # Monday 06:40 UTC — before the working week, so a weekend change is seen + # on Monday rather than discovered mid-campaign. + - cron: '40 6 * * 1' + workflow_dispatch: + inputs: + limit: + description: 'Max repos to scan (0 = all)' + type: string + default: '0' + +concurrency: + group: ${{ github.workflow }} + cancel-in-progress: false + +permissions: + contents: read + issues: write + +jobs: + detect: + name: Sweep the estate for settings drift + runs-on: ubuntu-latest + timeout-minutes: 45 + steps: + - name: Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 + + - name: Scan estate for settings drift + id: scan + env: + # A settings read needs more than the default GITHUB_TOKEN can see: + # actions/permissions and actions/variables are admin-scoped, and + # security_and_analysis is ABSENT from LIST endpoints entirely, so it + # needs a per-repo GET. Without an admin-scoped token this reports a + # FALSE CLEAN — it cannot read the settings it is checking. That is + # the "guard asks a different question than its consumer" trap, so the + # step fails loudly below rather than reporting an empty result. + GH_TOKEN: ${{ secrets.ESTATE_SETTINGS_READ_TOKEN || github.token }} + run: | + set -o pipefail + if ! gh api repos/${{ github.repository }}/actions/permissions >/dev/null 2>&1; then + echo "::error::Token cannot read actions/permissions. A clean result would be FALSE." + echo "::error::Set secret ESTATE_SETTINGS_READ_TOKEN to a PAT with admin:repo scope." + exit 2 + fi + chmod +x scripts/check-settings-drift.sh + set +e + scripts/check-settings-drift.sh --limit "${{ inputs.limit || '0' }}" > drift.tsv + rc=$? + set -e + echo "rows=$(( $(wc -l < drift.tsv) - 1 ))" >> "$GITHUB_OUTPUT" + echo "rc=$rc" >> "$GITHUB_OUTPUT" + cat drift.tsv + + - name: Upload report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a + with: + name: settings-drift + path: drift.tsv + + - name: Open or update the tracking issue + if: steps.scan.outputs.rc == '1' + env: + GH_TOKEN: ${{ github.token }} + run: | + { + echo "Settings drift detected: ${{ steps.scan.outputs.rows }} row(s)." + echo + echo "A row means the live setting differs from \`config/settings/repo.json\`." + echo "It does NOT by itself mean anything is broken — read the calibration" + echo "section at the top of \`scripts/check-settings-drift.sh\` before acting." + echo "In particular a CLEAN result is not proof CI is healthy: a startup" + echo "failure can be event-specific and invisible in every setting and file." + echo + echo '```' + head -200 drift.tsv + echo '```' + } > body.md + n=$(gh issue list --repo "${{ github.repository }}" --state open \ + --search "Settings drift in:title" --json number --jq '.[0].number') + if [ -n "$n" ]; then + gh issue comment "$n" --repo "${{ github.repository }}" --body-file body.md + else + gh issue create --repo "${{ github.repository }}" \ + --title "Settings drift detected" --body-file body.md + fi diff --git a/scripts/check-settings-drift.sh b/scripts/check-settings-drift.sh new file mode 100755 index 000000000..c16de792c --- /dev/null +++ b/scripts/check-settings-drift.sh @@ -0,0 +1,178 @@ +#!/bin/bash +# SPDX-License-Identifier: MPL-2.0 +# Copyright (c) 2026 Jonathan D.A. Jewell +set -eo pipefail + +# check-settings-drift.sh — detect estate settings moving without anyone noticing. +# +# ── The fault this catches ───────────────────────────────────────────────── +# Repository SETTINGS are not in git. Nothing reviews them, nothing diffs them, +# and a change leaves no artefact — no commit, no PR, no run. The estate is +# 97–100% compliant with `config/settings/repo.json` at any given moment, so +# re-applying the canon gains almost nothing. What is missing is anything that +# NOTICES when a setting moves. +# +# Measured 2026-09-14, the day this was written: the Actions allowlist on 239 +# repositories changed from `allowed_actions=selected` with an EMPTY +# `patterns_allowed` and `verified_allowed=false`, to a 92-pattern allowlist +# with `verified_allowed=true` — between a census and the authorised write that +# was meant to fix it. The fix had already happened. Nobody could say when, by +# whom, or whether it was deliberate, because no instrument was watching. +# That is the whole argument for this script. +# +# ── Why an EMPTY allowlist is the highest-value check here ───────────────── +# An empty `patterns_allowed` refuses every third-party action, INCLUDING +# transitively through a reusable workflow, because actions are judged against +# the CALLER repo's allow-list. The run then dies at `startup_failure` with +# `jobs.total_count == 0` and emits NO check run — so a required context never +# reports and the repo looks GREENER than a healthy one. A settings change can +# therefore disarm CI estate-wide while every dashboard stays clean. +# +# ── Calibration: what a hit does and does not prove ──────────────────────── +# A reported line means the live setting differs from canon. It does NOT prove +# anything is broken: +# * `per_repo_deltas_allowed` keys are legitimately per-repo and are skipped. +# * `security_and_analysis` is unavailable on private repos of a Free account; +# canon itself carries `repo_private_overrides` for this. Skipped on private. +# * An empty allowlist harms only repos that actually CALL a third-party +# action. Measured 2026-09-14: of 61 such repos, 48 were genuinely broken +# and 13 were CLEAR. This script reports the setting; it does not claim the +# repo is dead. +# Conversely a CLEAN result is not proof CI is healthy — a startup failure can +# be EVENT-SPECIFIC and invisible in every setting and every file. Proven the +# same day: `sanctify-php` run 34768365915 (`push`, head a25c8edb) died with +# jobs=0, while run 34895780522 (`workflow_dispatch`, the SAME head a25c8edb) +# started 7 jobs and mirrored successfully. Identical bytes, identical settings. +# +# ── The subset rule (do not "fix" this into an equality check) ───────────── +# GitHub ECHOES BACK defaults it was never sent. Comparing canon to live by +# equality produces permanent false drift on keys nobody ever set. Every +# comparison here is CANON ⊆ LIVE, one key at a time. The same trap already +# cost two hard 422s when `config/rulesets/base.json` was PUT verbatim. +# +# ── What this is NOT ─────────────────────────────────────────────────────── +# It REPORTS ONLY. It never writes a setting. Unattended cross-repo mutation is +# a human decision, and settings are exactly where that matters most. +# +# Usage: check-settings-drift.sh [--owner OWNER] [--canon PATH] [--limit N] +# Output: TSV — repo key expected actual +# Exit: 0 = no drift 1 = drift found 2 = usage / environment error + +OWNER="hyperpolymath" +CANON="config/settings/repo.json" +LIMIT=0 + +while [ $# -gt 0 ]; do + case "$1" in + --owner) OWNER="$2"; shift 2 ;; + --canon) CANON="$2"; shift 2 ;; + --limit) LIMIT="$2"; shift 2 ;; + -h|--help) sed -n '5,70p' "$0"; exit 0 ;; + *) echo "unknown argument: $1" >&2; exit 2 ;; + esac +done + +command -v gh >/dev/null || { echo "gh not found" >&2; exit 2; } +command -v jq >/dev/null || { echo "jq not found" >&2; exit 2; } +[ -f "$CANON" ] || { echo "canon not found: $CANON" >&2; exit 2; } + +DELTAS=$(jq -r '(.per_repo_deltas_allowed // [])[]' "$CANON" | tr '\n' ' ') +drift=0 + +emit() { printf '%s\t%s\t%s\t%s\n' "$1" "$2" "$3" "$4"; drift=1; } + +# CANON ⊆ LIVE for the flat repo block, skipping allowed per-repo deltas. +check_repo_block() { + local repo="$1" live="$2" private="$3" + local key exp act + while IFS=$'\t' read -r key exp; do + [ -z "$key" ] && continue + case " $DELTAS " in *" $key "*) continue ;; esac + act=$(echo "$live" | jq -r --arg k "$key" '.[$k] // "ABSENT" | tostring') + [ "$act" = "$exp" ] || emit "$repo" "$key" "$exp" "$act" + done < <(jq -r '.repo | to_entries[] | select(.value|type != "object") + | "\(.key)\t\(.value|tostring)"' "$CANON") + + # security_and_analysis is ABSENT from LIST endpoints — it needs the per-repo + # GET, which $live already is. Skipped on private: canon's own + # repo_private_overrides records that secret scanning is unavailable there. + if [ "$private" != "true" ]; then + while IFS=$'\t' read -r key exp; do + [ -z "$key" ] && continue + act=$(echo "$live" | jq -r --arg k "$key" '.security_and_analysis[$k].status // "ABSENT"') + [ "$act" = "$exp" ] || emit "$repo" "security_and_analysis.$key" "$exp" "$act" + done < <(jq -r '.repo.security_and_analysis | to_entries[] + | "\(.key)\t\(.value.status)"' "$CANON") + fi +} + +check_actions() { + local repo="$1" perms sel exp_allowed n verified pinning + perms=$(gh api "repos/$OWNER/$repo/actions/permissions" 2>/dev/null) || return 0 + + exp_allowed=$(jq -r '.actions_permissions.allowed_actions' "$CANON") + act=$(echo "$perms" | jq -r '.allowed_actions // "ABSENT"') + # `all` is MORE permissive than canon, not drift toward breakage — report it + # distinctly so a reader is never told a working repo is broken. + if [ "$act" != "$exp_allowed" ]; then + emit "$repo" "actions.allowed_actions" "$exp_allowed" "$act" + fi + + pinning=$(echo "$perms" | jq -r '.sha_pinning_required // "ABSENT"') + exp=$(jq -r '.actions_permissions.sha_pinning_required|tostring' "$CANON") + [ "$pinning" = "$exp" ] || emit "$repo" "actions.sha_pinning_required" "$exp" "$pinning" + + # THE HIGH-VALUE CHECK: selected + empty patterns disarms CI silently. + if [ "$act" = "selected" ]; then + sel=$(gh api "repos/$OWNER/$repo/actions/permissions/selected-actions" 2>/dev/null) || return 0 + n=$(echo "$sel" | jq -r '(.patterns_allowed // []) | length') + [ "$n" = "0" ] && emit "$repo" "actions.patterns_allowed" "non-empty" "EMPTY (refuses all third-party actions, incl. via reusables)" + verified=$(echo "$sel" | jq -r '.verified_allowed // "ABSENT"') + [ "$verified" = "false" ] && emit "$repo" "actions.verified_allowed" "true" "false (refuses even Marketplace-verified actions)" + fi + + wf=$(gh api "repos/$OWNER/$repo/actions/permissions/workflow" 2>/dev/null) || return 0 + for k in default_workflow_permissions can_approve_pull_request_reviews; do + exp=$(jq -r --arg k "$k" '.actions_workflow_permissions[$k]|tostring' "$CANON") + act=$(echo "$wf" | jq -r --arg k "$k" '.[$k] // "ABSENT" | tostring') + [ "$act" = "$exp" ] || emit "$repo" "actions_workflow.$k" "$exp" "$act" + done +} + +# Mirror forge enablement. Each forge job in mirror-reusable.yml is gated on +# `if: vars._MIRROR_ENABLED == 'true'`, so a false variable makes the job +# SKIP — and a run whose every forge job skipped reports GREEN while backing up +# nowhere. Measured 2026-09-14 across the mirror fleet: ZERO repos had all seven +# enabled, and roughly a third had NONE. This is checked only where the repo +# actually calls mirror.yml, because the variable is meaningless otherwise. +FORGES="GITLAB BITBUCKET CODEBERG SOURCEHUT DISROOT GITEA RADICLE" +check_mirror_vars() { + local repo="$1" vars f v + gh api "repos/$OWNER/$repo/contents/.github/workflows/mirror.yml" >/dev/null 2>&1 || return 0 + vars=$(gh api "repos/$OWNER/$repo/actions/variables?per_page=100" 2>/dev/null) || return 0 + for f in $FORGES; do + v=$(echo "$vars" | jq -r --arg n "${f}_MIRROR_ENABLED" '.variables[]|select(.name==$n)|.value' 2>/dev/null) + [ "$v" = "true" ] || emit "$repo" "vars.${f}_MIRROR_ENABLED" "true" "${v:-ABSENT} (forge job SKIPS; run can report GREEN while mirroring nowhere)" + done +} + +printf 'repo\tkey\texpected\tactual\n' + +# NOTE: process substitution, NOT a pipe. `gh repo list | while read` runs the +# loop in a SUBSHELL, so every `drift=1` set by emit() is discarded when the +# subshell exits and the script always reports success — a drift detector that +# can never report drift. That is the same shape as the estate's recurring +# "guard asks a different question than its consumer" fault, so it is spelled +# out here rather than left to be reintroduced by a later tidy-up. +i=0 +while read -r repo; do + i=$((i+1)) + [ "$LIMIT" != "0" ] && [ "$i" -gt "$LIMIT" ] && break + live=$(gh api "repos/$OWNER/$repo" 2>/dev/null) || continue + private=$(echo "$live" | jq -r '.private') + check_repo_block "$repo" "$live" "$private" + check_actions "$repo" + check_mirror_vars "$repo" +done < <(gh repo list "$OWNER" --no-archived --limit 1000 --json name --jq '.[].name') + +exit $drift