From 699e4db33a475f614a18d4c5359f57d35f5e3843 Mon Sep 17 00:00:00 2001 From: Ankush Kapoor Date: Mon, 28 Sep 2026 12:31:24 -0500 Subject: [PATCH] Port the post-v0.230.0 development train Runtime-render migration (litclock-dev#871 Stage B), the bootcheck rollback fixes (litclock-dev#894, litclock-dev#896), and two pre-port fixes found reviewing this port (litclock-dev#897, litclock-dev#898). Dev range 1e81722b..abbd54b9. --- .gitignore | 13 + CHANGELOG.md | 8 + CLAUDE.md | 211 ++++- env.sh.sample | 21 +- scripts/first-boot.sh | 7 +- scripts/lib/state.sh | 55 +- scripts/prepare-for-cloning.sh | 30 + scripts/reset-setup.sh | 17 + scripts/update.sh | 526 +++++++++-- src/control_server/routes/status.py | 20 +- tests/test_first_boot_flow.py | 19 +- tests/test_runtime_render_autostamp.py | 1179 +++++++++++++++++++++++- tests/test_update_sh.py | 503 ++++++++-- 13 files changed, 2396 insertions(+), 213 deletions(-) diff --git a/.gitignore b/.gitignore index 84620aa..b6b666f 100644 --- a/.gitignore +++ b/.gitignore @@ -10,6 +10,19 @@ env.sh # (litclock-dev#721 bench pass). env.sh.lock env.sh.bak +# mktemp staging for the atomic env.sh writers (`mktemp "${dest}.XXXXXX"` in +# lib/state.sh and update.sh). Normally renamed over env.sh within milliseconds, +# but a SIGKILL or power loss inside that window leaves a full, unredacted copy +# of env.sh — API key, coordinates, city — sitting beside it. The siblings above +# are enumerated, so this shape was not covered and `git reset --hard` never +# removes it (litclock-dev#871 /review, adversarial). Exactly six X's, matching +# mktemp's template, so nothing else that shares the prefix is swept up. +env.sh.?????? +# `sample` is also six characters, so the template pattern matches the TRACKED +# env.sh.sample. Harmless while it stays tracked (ignore rules do not apply to +# tracked files) and a trap the moment it is removed and re-added, so exclude it +# explicitly rather than relying on that. +!env.sh.sample *.log .vscode *.pickle diff --git a/CHANGELOG.md b/CHANGELOG.md index ad87a79..ee43cb0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,14 @@ All notable changes to LitClock are documented here. Format loosely follows [Kee ## [Unreleased] +### Changed + +- Your clock now draws each quote from text instead of using a pre-made picture, once it has checked on itself that it can do so correctly; the pictures stay on the clock as a backup. + +### Fixed + +- If an update ever leaves your clock unable to start, it now reliably goes back to the version that last worked, and draws its quotes from the backup pictures until a newer version is installed. + ## [v0.230.0] - 2026-09-22 ### Fixed diff --git a/CLAUDE.md b/CLAUDE.md index fdd5db0..dd70ad9 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -426,10 +426,12 @@ margin.) Not in the checklist above because it is not a first-boot flow — but it is the highest-blast-radius surface in the tree, and **the failure direction is -asymmetric**. A false GREEN ships a broken update once. A false RED reverts -every update, on every device, every weekly tick, forever: nothing writes a -blocked-sha on the smoke-revert path, so the next tick resolves the same target -and reverts again. Test on a device you can re-flash. +asymmetric**. A false GREEN ships a broken update once. A false RED reverts a +healthy update on every affected device and, since litclock-dev#865, BLOCKS it: +the revert records the release in `blocked-sha`, so those devices stay off it +until a newer release ships (before litclock-dev#865 it was worse — every weekly tick +re-applied and re-reverted the same target, forever). Test on a device you can +re-flash. Force a run rather than waiting for the Sunday timer: @@ -500,7 +502,7 @@ sudo systemctl start litclock-update.service && journalctl -fu litclock-update ticks verified on hardware 2026-09-18. The pip hash stays deleted across the blocked ticks (the revert removes `HASH_FILE`) so the recovering release re-runs pip once; that is expected, not a second fault. -- **The runtime-render self-test runs after the marker, and its verdict lands in the memo (litclock-dev#871 Stage A — INERT this release).** It runs only on an APPLIED release: a tick with nothing new exits at the no-op early-out before Phase 4.5, so publish one from a local bare remote — and build that release so it touches NONE of the six proof inputs (`fonts/`, the measurement dump, `requirements.txt`, `quote_renderer.py`, `gd_measure.py`, `validate_measurement.py`), or Phase 2b's revoke block removes the marker you are about to tamper with, the KEEP arm re-stamps a fresh one, and the self-test passes; a `runtime-render validation marker removed:` line in the journal means the staging was undone and the run proves nothing. A normal run on a device with a marker must log `Running the runtime-render self-test`, then `[selftest] dry-run: rendered 800x480 image, render_mode=runtime`, then `runtime-render self-test PASSED in N.Ns`, and `/var/lib/litclock/runtime-render-selftest.json` holding `"result":"passed"` with `duration_s` and the release `sha` — the durable pass record Stage B will gate on (the journal is capped at 7 days, so the log line alone would be gone before the next release); read `duration_s`: Stage B will need devices that render inside the 4s lead — and `LITCLOCK_RUNTIME_RENDER` in `env.sh` must be UNCHANGED afterwards, because Stage A changes nothing on the device; that is the point. A failure is silent by design, so stage one to prove the harness can fail — **in the marker, not in the tree**: `sudo sed -i 's/digest=[0-9a-f]*/digest=deadbeef/' /home/pi/litclock/.runtime-render-validated`. The marker is gitignored, so Phase 2's `git reset --hard` leaves it alone; it is still PRESENT, so the KEEP arm does not re-stamp it and clears the memo; and the painter's digest check declines the text tier, so the self-test paints a PNG and exits 3. Expect `[selftest] ... render_mode=image`, then `self-test did not pass: the painter fell back`, then `jq .result,.rc /var/lib/litclock/runtime-render-validation.json` printing `"selftest-failed"` and `3` (the file is compact JSON — do not grep for a spaced literal), `/var/lib/litclock/runtime-render-selftest.json` GONE (a fail retires an earlier pass), and `/api/status` showing the memo as `runtime_render_validation`. **The update must still end in `Update Complete`** and the panel must keep painting — a failed self-test is not an update failure. Two stagings that CANNOT fail, recorded so nobody re-derives them: moving the marker aside (the KEEP arm re-stamps an absent marker before the self-test runs, so that run passes), and `sudo mv fonts{,.bak}` (four tracked files — Phase 2 restores them before anything looks, and even if it did not, the PNG tier needs the same fonts for the masthead, so the smoke gate's own dry-run would revert the release before reaching the self-test). Then restore the marker — `sudo -u pi /home/pi/litclock/venv/bin/python3 tools/validate_measurement.py check --stamp` re-earns it — and apply another release: the present marker clears the memo at the top of the KEEP arm, the self-test passes, and the memo file must be gone. Three more things to confirm while there: with a non-English `LITCLOCK_LANGUAGE` in `env.sh` the `[selftest]` lines render THAT corpus (the catalog probes above are pinned to English; this one deliberately is not); `/run/litclock/current-quote.png` keeps its pre-tick mtime (the self-test renders into a throwaway directory); and no weather request leaves the device during the self-test (`WEATHER_ENABLED` is forced off for it — a capability probe has no business on the network). +- **The runtime-render self-test runs after the marker, and its verdict lands in the memo (litclock-dev#871 Stage A — shipped inert one release ahead of Stage B).** **It runs on EVERY tick, not only on an applied release** — this line used to claim the opposite and it is wrong (measured 2026-09-23 during the Stage B bench run, and pre-existing since Stage A shipped). There is no same-SHA early-out: an up-to-date tick logs `Already up to date ()` as a plain message and runs straight on through Phase 4.5, self-test and all — verified at 3.9s on a tick that applied nothing. So a plain `sudo systemctl start litclock-update.service` is enough to exercise the self-test, and the harness is only needed when you must control what the release CONTAINS. It also means the ~3.5s self-test is paid weekly on every marker-bearing device forever, and it is what makes Stage B's "retry next cycle" real. When you do publish one, build that release so it touches NONE of the six proof inputs (`fonts/`, the measurement dump, `requirements.txt`, `quote_renderer.py`, `gd_measure.py`, `validate_measurement.py`), or Phase 2b's revoke block removes the marker you are about to tamper with, the KEEP arm re-stamps a fresh one, and the self-test passes; a `runtime-render validation marker removed:` line in the journal means the staging was undone and the run proves nothing. A normal run on a device with a marker must log `Running the runtime-render self-test`, then `[selftest] dry-run: rendered 800x480 image, render_mode=runtime`, then `runtime-render self-test PASSED in N.Ns`, and `/var/lib/litclock/runtime-render-selftest.json` holding `"result":"passed"` with `duration_s` and the release `sha` — the durable pass record Stage B gates on (the journal is capped at 7 days, so the log line alone would be gone before the next release). `duration_s` is recorded for diagnosis, not as a gate: Stage B reads the record's `result` and `sha` only (why, below). The self-test itself changes nothing on the device. What this step checks is narrow: `LITCLOCK_RUNTIME_RENDER` may END the tick as `true` when it did not start that way ONLY with a `MIGRATED to runtime text rendering` line in THIS run's journal — bound the read with `--since` the run's start, because journald is persistent and an older MIGRATED line would vouch for a flip nobody can explain. That covers both starting points: `false`, and a key that was MISSING, which Phase 3 backfills from `env.sh.sample` as `false` before Stage B runs, so an old device can go missing → `true` in one tick with a MIGRATED line (correct), or missing → `false` with none (also correct: Stage B refused, silently or with a `Not migrating …` line). Any other change is the failure. Whether a given device SHOULD have flipped is not this step's question — Stage B's guards decide that, and the Stage B section below tests them. A failure is silent by design, so stage one to prove the harness can fail — **in the marker, not in the tree**: `sudo sed -i 's/digest=[0-9a-f]*/digest=deadbeef/' /home/pi/litclock/.runtime-render-validated`. The marker is gitignored, so Phase 2's `git reset --hard` leaves it alone; it is still PRESENT, so the KEEP arm does not re-stamp it and clears the memo; and the painter's digest check declines the text tier, so the self-test paints a PNG and exits 3. Expect `[selftest] ... render_mode=image`, then `self-test did not pass: the painter fell back`, then `jq .result,.rc /var/lib/litclock/runtime-render-validation.json` printing `"selftest-failed"` and `3` (the file is compact JSON — do not grep for a spaced literal), `/var/lib/litclock/runtime-render-selftest.json` GONE (a fail retires an earlier pass), and `/api/status` showing the memo as `runtime_render_validation`. **The update must still end in `Update Complete`** and the panel must keep painting — a failed self-test is not an update failure. Two stagings that CANNOT fail, recorded so nobody re-derives them: moving the marker aside (the KEEP arm re-stamps an absent marker before the self-test runs, so that run passes), and `sudo mv fonts{,.bak}` (four tracked files — Phase 2 restores them before anything looks, and even if it did not, the PNG tier needs the same fonts for the masthead, so the smoke gate's own dry-run would revert the release before reaching the self-test). Then restore the marker — `sudo -u pi /home/pi/litclock/venv/bin/python3 tools/validate_measurement.py check --stamp` re-earns it — and apply another release: the present marker clears the memo at the top of the KEEP arm, the self-test passes, and the memo file must be gone. Three more things to confirm while there: with a non-English `LITCLOCK_LANGUAGE` in `env.sh` the `[selftest]` lines render THAT corpus (the catalog probes above are pinned to English; this one deliberately is not); `/run/litclock/current-quote.png` keeps its pre-tick mtime (the self-test renders into a throwaway directory); and no weather request leaves the device during the self-test (`WEATHER_ENABLED` is forced off for it — a capability probe has no business on the network). **First field measurement, 2026-09-20 (`dev-20260920-512c291`, Pi Zero 2 W):** the self-test recorded **3.8s** during the update; five hand-run repeats on an idle system gave **3.4 / 3.5 / 3.5 / 3.5 / 3.7s**. The gap is update-time load, not instrument bias, and it errs the safe way. @@ -597,6 +599,205 @@ sudo systemctl start litclock-update.service && journalctl -fu litclock-update 0 — a non-zero exit with empty stdout is indistinguishable from a dead interpreter, and the gate would have to guess. Restore the bundle afterwards. +### Stage B — the runtime-render migration (litclock-dev#871) + +The flip: `update.sh` rewrites `export LITCLOCK_RUNTIME_RENDER=false` to `true` +under `env.sh.lock` once the Stage A self-test passes on that run. Both calls +sit in the smoke-gate KEEP arm behind the same `-f "$RUNTIME_MARKER"` and +`ROLLBACK_MODE` guard, self-test first, migration second. + +**Almost none of this belongs on hardware.** `TestStageBMigrationExecutes` +already EXECUTES every guard arm — missing/unparseable/sha-mismatched record, +absent or empty `images/metadata`, two active assignments, a bare assignment +beside the export, a held lock, a transformation that loses lines, idempotence. +Re-running those on a Pi tests nothing the suite does not already pin. What the +suite cannot reach is the glass and the clock: it proves `env.sh` changed, never +that the painter read it, that the panel shows a text frame, or that the frame +still lands on time. That is the whole of this section. + +**Run it on the bench, not the fielded clock.** The bench is re-flashable and +already in the exact pre-migration state — flashed from the PUBLIC v0.230.0 +image, so `LITCLOCK_RUNTIME_RENDER=false`, marker present from the build-time +stamp, `images/` on disk, and **no `/var/lib/litclock/runtime-render-selftest.json` +at all**, because an image flash never runs `update.sh`. That absence is not a +fault; it is the majority-fleet path (most devices are internet flashers on +images), and it means the first Stage B tick must write the record AND migrate +in the same run, with `rec_sha` and `git rev-parse HEAD` resolving to the same +new release. The fielded clock is already on runtime render and is production — +do not stage a downgrade on it to create a test subject. + +Stage the release from a local bare remote and a tag-list API stub +(a synthetic release avoiding the six proof inputs, or Phase 2b revokes the +marker and the KEEP arm re-stamps it — see the Stage A bullet above). Build the +synthetic release on **public's** HEAD when the device is a public flash: the +two histories diverge, so a dev commit served to a public-tracking device breaks +ancestry. Check it (`git merge-base --is-ancestor `) rather +than assuming. **And make the API stub match the tag LIST endpoint exactly** — +`"/tags" in path` also matches `/repos/{o}/{r}/releases/tags/{tag}`, which +`download_images.sh` calls; handing that the tags ARRAY throws `AttributeError: +'list' object has no attribute 'get'` and quarantines `images/`, which silently +converts the next run into the control case (measured 2026-09-23). + +- **REUSE A TAG NAME ACROSS SESSIONS AND YOU SILENTLY TEST THE OLD CODE.** Hit + on 2026-09-23 and it produced a clean-looking PASS on a tree that did not + contain the fix under test. `update.sh`'s tag fetch is deliberately + non-forcing (`refs/tags/:refs/tags/`, no `+`), so a device that + already carries `v0.231.0` from an earlier harness run REFUSES the new + sibling commit and the resolver keeps pointing at the old sha. Nothing warns: + the tick logs `Target: Release SHA ` and `Already up to date`, both of + which read as normal. **Before every harness run, on the device:** + `git -C ~/litclock tag -d `, then confirm + `git rev-parse --short ` matches the bare repo's + (`git -C rev-parse --short `). Cheaper still, + check the applied sha afterwards — `Updated: → ` naming the sha you + built is the only line that proves the release under test actually landed, and + a grep for the fix in `~/litclock/scripts/update.sh` settles it outright. +- **The CONTROL first, and it must REFUSE.** Before staging a passing run, empty + the fallback rung: `sudo mv /home/pi/litclock/images/metadata{,.qabak}`. Run + the tick. It must log `Not migrating to runtime render: images/metadata is + missing or empty`, leave `env.sh` reading `false`, write the memo as + `migration-skipped`, and still finish `Update Complete`. **Grep the refusal + line, not the word `migrat`** — `MIGRATED to runtime text rendering` and `Not + migrating` both match a prefix search, so a sloppy grep passes on either + outcome, which is the shape this repo keeps getting caught by. Bound the read + with `--since ""` too: journald here is persistent, so an + unbounded grep on the second run happily matches the FIRST run's verdict — + which on this check is the opposite verdict. If the control migrates anyway, + the staging is wrong and nothing below means anything. + - **The control is only meaningful because the smoke gate still passes without + images — verified, not assumed** (2026-09-23): a dry-run with + `images/metadata` absent degrades to `render_mode=time-only` and exits **0**, + so the run reaches the migration. Had it exited non-zero, the gate would have + reverted first and the refusal would have proved nothing about the guard. + Re-check this if the painter's empty-corpus behaviour ever changes. + - **Restoring is not just undoing the `mv`.** The tick's image sync verifies + the PNGs against the bundled byte manifest, fails with `images/metadata` + gone, and quarantines the WHOLE tree to `images.failed.`. Recover by + moving that directory back to `images/` and renaming `metadata.qabak` inside + it, then confirm `cd ~/litclock/images && sha256sum --quiet --strict -c + files.sha256` returns **rc 0** before the real run. Skip that and the next + tick re-quarantines, the migration refuses again, and you read a false + negative as a Stage B failure. +- **Then the real tick: the flag flips and the PANEL follows.** `env.sh` must + read `true`, the journal must carry the `MIGRATED to runtime text rendering` + line, and `/var/lib/litclock/runtime-render-validation.json` must be GONE (a + successful migration clears the memo). None of that is the payoff. Wait for + the next minute tick and read `jq -r .render_mode + /run/litclock/current-quote.json` — it must say **`runtime`**. Until that + field flips, the device has a rewritten `env.sh` and is still painting PNGs. +- **Look at the glass once.** The unit tests never see a frame. Compare a + text-rendered frame against the PNG it replaced for the same minute: masthead, + margins, the long-quote sizes, and the attribution line. Runtime render picks + its own font size, so a quote near the wrap boundary is where the two tiers + diverge; pick one deliberately rather than whatever minute you happen to be + standing there for. +- **The frame must still settle by `:00`, measured over SSH with `picked_at`.** + The second-of-minute of `picked_at` in `/run/litclock/current-quote.json` is + **when the frame settled on glass**, which is what makes this checkable without + standing at the panel. **Do not re-derive that from the source and conclude + otherwise** — `literary_clock.py` carries two different `picked_at` values, and + the two in the quote builders (~806, ~856) are SELECTION stamps that are never + published: `_write_status_file` builds its own payload with a fresh + `_time.time()` and is called immediately after `epd.display()` returns. Grepping + the field name lands on the selection stamps first and invites exactly the wrong + reading. Sample five consecutive minutes before and after the migration: + + ```bash + for i in $(seq 5); do + sleep 60 + jq -r '"\(.time) picked_at=\(.picked_at|strftime("%M:%S")) mode=\(.render_mode)"' \ + /run/litclock/current-quote.json + done + ``` + + Take these from SCHEDULED `:56` runs — a `systemctl restart litclock.timer` + fires an immediate off-schedule paint whose stamp means nothing, and **an + update ends by restarting the timer**, so the first stamp you see after a + migration is always one of those (measured `settled=:51` on 2026-09-23). + Discard it and start counting from the next minute. Baselines measured + 2026-09-13 at lead 4.0s: bench **:02.8** on images, bench **:03.3** on runtime, + fielded clock **:04.3** on runtime. So the flip is expected to cost ~0.5-1s of + settle and still land a correct frame. What fails the check is the + quote's timestring disagreeing with the wall clock once settled, not the + offset itself. If it does disagree, raise `LITCLOCK_RENDER_LEAD_S` on that + device — do NOT edit the default, and do NOT revert the migration for it. +- **The fallback rung still works AFTER the migration.** This is the layer the + whole brick argument rests on, and migrating is exactly when it stops being + hypothetical. With the device now on `true`, corrupt the marker the litclock-dev#886 + way (`printf 'freetype=2.13.2 digest=\xff\xfe\xfd broken\n' > + .runtime-render-validated`) and confirm the next painted frame comes back with + `render_mode` **`image`** and the panel keeps painting — not a traceback, not a + frozen panel. Restore the marker with `validate_measurement.py check --stamp`. + A migrated device that cannot fall back is the one outcome Stage B is not + allowed to produce. +- **The flip survives a reboot and does not re-fire.** Reboot, confirm `env.sh` + still says `true` and `render_mode` is still `runtime`, then run one more tick + with nothing new: `env.sh` must stay `true`, no `MIGRATED` line may appear in + THAT tick's journal (`--since` its start — the migration tick's own line is + still in the persistent journal), and it should write NO memo. An already-`true` device is the goal state, not + a negative result. If a `migration-skipped` memo DOES appear, read its reason + before calling it a regression: the record guards run before the flag is + read, so a pass record that is missing, unreadable or not for this release + (the new one failed to write and an older one is still on disk) produces + that memo on an already-migrated device too. Any OTHER reason (the images tier, the + lock, the rewrite) on a device with a single `true` assignment means the + idempotence guard regressed: those guards sit after the silent already-`true` + return. (A hand-edited env.sh holding an exported `=false` and a later + `=true` is the one exception: effectively `true`, but it has an exported + `=false` for Stage B to find, so it can get an images or duplicate-assignment + memo legitimately.) + +**Measured end to end on the bench, 2026-09-23 — control failed first, real run +migrated.** Control (images/metadata aside): self-test PASSED 3.5s, `Not +migrating … images/metadata is missing or empty`, `env.sh` unchanged, memo +`migration-skipped`, `Update Complete`. Real run: self-test PASSED 3.6s, +`MIGRATED to runtime text rendering`, memo cleared, record sha == HEAD, +`images/` retained, and `render_mode` on the next scheduled paint **`runtime`**. +Settle moved `:01` -> `:02` (five samples each way) — quote the ~1s DELTA, not +the absolute, since this build's image baseline is already faster than the +2026-09-13 `:02.8` reference. Corrupt-marker fallback on the MIGRATED device +degraded to `image` and kept painting on time, then returned to `runtime` on +re-stamp. Reboot held; a further no-op tick left `env.sh` alone and wrote no +memo. Full record is kept with the maintainer's QA notes, off-repo. + +**Re-QAed the same day after the /review fixes, and the smoke-gate half is the +part worth re-reading.** The gate now sources `env.sh`, so it renders the tier +the device is actually on. The evidence is one log line on one device across two +ticks: `[smoke] dry-run: … render_mode=image` before the migration (env.sh still +`false` at smoke time) and `render_mode=runtime` on the next. Before the fix the +second said `image` forever, which meant the only thing that can revert a +release had stopped exercising the tier the fleet runs. Also confirmed end to +end: a staged refusal (images/metadata aside, flag reset to `false` so the +device is eligible) now reaches `/api/status` as +`runtime_render_validation.result = "migration-skipped"` with its reason — it +was `null` before the allowlist fix, i.e. identical to a healthy clock. + +**Old images take the jump in one tick — measured 2026-09-27.** Fresh flashes +of public v0.219.0 and v0.226.0 (the OLD 600s unit, no baked marker), then one +tick to a synthetic v0.231.0 via a pass-through tag stub (it fakes only the +tag list, so the 124MB image download is real): 379s / 373s, `MIGRATED` on +that FIRST tick, then `runtime` settling `:03`, and a no-op second tick. The +budget guard was the only thing between that and the old unit's SIGKILL, with +~220s to spare — a slow link defers the validator to tick 2 by design. Stop +`litclock-update.timer` before anything else on a fresh old-image flash, or a real GitHub tick spoils it. +**Stream the journal to the laptop during any run that reboots** (`journalctl -f` +over SSH, restarted after each boot): the 2026-09-24 rollback runs' `update.sh` +lines were missing from the card's own journal three days later, cause unknown. + +**The rollback marker revoke, re-run after the pre-landing hardening moved it +before the reset (2026-09-27): PASS**, without a reflash — a release built on +the device's own LKG plus a README line serves as the fleeing release, since +the revoke lives in the FLEEING release's `update.sh`. Evidence line is +`marker removed: bootcheck rollback`, count exactly 1 in `--since` the reboot, +ordered before `Updated:`. That staging never re-execs (both `update.sh` copies +are identical), so it cannot test the snapshot path; the 2026-09-24 run did. + +- **What this does NOT cover.** The other three fleet devices are unreachable + gifts and will migrate unobserved — that asymmetry is why Stage A shipped inert + for a release first, and it does not change here. Nothing on the bench + exercises a device whose self-test passes and whose panel then fails, because + that device does not exist to hand. + ### litclock-dev#337 — Location/Weather/Temperature IA (post-design-review, A9-A18) After litclock-dev#337 lands, the Settings tab IA changes substantially. Replace the pre-litclock-dev#337 expectations with these: diff --git a/env.sh.sample b/env.sh.sample index 57e84fe..3ff899b 100755 --- a/env.sh.sample +++ b/env.sh.sample @@ -81,8 +81,25 @@ export GIFT_MODE_MESSAGE= # pre-rendered PNG (falls back to the PNG, then a plain time display, on # any render failure). The flag is only honored after this device passes # `venv/bin/python3 tools/validate_measurement.py check --stamp` (writes -# the .runtime-render-validated marker the clock requires). Default off -# until the Stage-2 soak validates it. +# the .runtime-render-validated marker the clock requires). +# +# THIS FILE IS THE BACKFILL SOURCE FOR EXISTING DEVICES, which is why the +# value here is `false` while a fresh flash is seeded `true`. The divergence +# is deliberate and load-bearing — see `env_sh_defaults()` in +# scripts/lib/state.sh, which is the fresh-flash seeder. +# +# update.sh Phase 3 copies the line below VERBATIM onto any device missing +# this key. A `true` here would therefore switch on an old device with none +# of litclock-dev#871 Stage B's guards run — no on-device self-test, no +# images/ fallback check — before the smoke gate, and with no revert path +# that undoes it. litclock-dev#783 records devices born missing up to ten of +# these knobs, so the oldest and least reachable clocks are the exposed ones. +# +# An existing device is migrated by update.sh instead, only after its own +# self-test has rendered a quote from text on THIS release and only while +# images/ is still there to fall back to. Set it to false by hand and the +# next weekly tick sets it back; that is deliberate (owner call 2026-09-19), +# not a bug. export LITCLOCK_RUNTIME_RENDER=false # --- Advanced Settings (uncomment to override defaults) --- diff --git a/scripts/first-boot.sh b/scripts/first-boot.sh index f9f6abc..fc5f3b5 100755 --- a/scripts/first-boot.sh +++ b/scripts/first-boot.sh @@ -607,7 +607,10 @@ main() { local _defaults # NO language argument — a fresh device keeps Accept-Language # negotiation alive on this boot (the litclock-dev#743 empty-seed contract). - _defaults=$(env_sh_defaults)$'\n' + # `true`: a FRESH flash renders text from its first paint. The + # helper defaults to false for the reset and cloning callers, + # which run on existing devices (litclock-dev#871 Stage B). + _defaults=$(env_sh_defaults "" true)$'\n' if ! atomic_write_env_sh "$ENV_FILE" "$_defaults"; then local _rc=$? if [[ "$_rc" == "75" ]]; then @@ -646,7 +649,7 @@ export ALLOW_NSFW_QUOTES=false export LITCLOCK_LANGUAGE= export SHOW_DIAGNOSTICS_SHORTCUT=false export GIFT_MODE_MESSAGE= -export LITCLOCK_RUNTIME_RENDER=false +export LITCLOCK_RUNTIME_RENDER=true # export DISPLAY_CLEAR_HOUR=2 # export LITCLOCK_RENDER_LEAD_S=4 # export WEATHER_API_TIMEOUT=15 diff --git a/scripts/lib/state.sh b/scripts/lib/state.sh index a83197e..f6ba35b 100644 --- a/scripts/lib/state.sh +++ b/scripts/lib/state.sh @@ -234,12 +234,22 @@ ENV_FILE_DEFAULT="${LITCLOCK_ENV_FILE:-/home/pi/litclock/env.sh}" # substitution STRIPS trailing newlines, so every caller re-adds one: # `DEFAULTS=$(env_sh_defaults)$'\n'`. Do not "simplify" that away. # -# VALUES may differ from the sample and two deliberately do: the sample +# VALUES may differ from the sample and three deliberately do: the sample # documents WEATHER_LATITUDE/LONGITUDE with real Austin coordinates as an # example, while a seeded device must leave them EMPTY. With # WEATHER_LOCATION_MODE=auto the IP-geo resolver fills them on a good boot; # on the ip-api.com-blocked path a seeded coordinate would render Austin # weather on a device that is not in Austin — worse than an honest empty. +# The third is LITCLOCK_RUNTIME_RENDER, which is a PARAMETER here rather than +# a constant: `true` when first-boot.sh asks for it (a fresh flash renders text +# from its first paint, against a marker stamped at image-build time), `false` +# for every other caller and for the sample. The sample is what update.sh +# Phase 3 BACKFILLS onto existing devices, so a `true` there would switch an +# old clock on with none of litclock-dev#871 Stage B's guards run. Fresh-flash +# default and existing-device backfill are different jobs and this is the only +# key where they disagree; do not "fix" them into agreement. +# tests/test_first_boot_flow.py compares the KEY SET and comment status, not +# values, and tests/test_runtime_render_autostamp.py pins the divergence. # # LANGUAGE ($1, optional) seeds LITCLOCK_LANGUAGE. Empty (the default, used by # first-boot and prepare-for-cloning) keeps Accept-Language negotiation alive @@ -254,6 +264,23 @@ ENV_FILE_DEFAULT="${LITCLOCK_ENV_FILE:-/home/pi/litclock/env.sh}" # that moving the interpolation into this file did not leave the belt behind. env_sh_defaults() { local language="${1-}" + # RUNTIME_RENDER ($2, optional) — litclock-dev#871 Stage B. Defaults to + # `false`, and the default is the fail-safe direction ON PURPOSE: this + # helper is NOT a fresh-flash-only seeder. reset-setup.sh and + # prepare-for-cloning.sh both call it on an EXISTING device, and neither + # removes `.runtime-render-validated` (they clear the self-test record, not + # the marker). So a `true` default would hand a reset clock runtime render + # with none of Stage B's five guards run — including a clock whose + # self-test had FAILED, which is precisely the device the guards exist to + # hold back. Those callers take the default; a device that can render text + # is migrated properly by update.sh on its next tick, within a week. + # `first-boot.sh` passes `true` explicitly, because a fresh flash carries a + # marker stamped at image-build time against the shipped renderer. + local runtime_render="${2-false}" + if [[ "$runtime_render" != "true" && "$runtime_render" != "false" ]]; then + echo "[state] env_sh_defaults: runtime_render '$runtime_render' is not true/false; seeding false" >&2 + runtime_render="false" + fi if [[ -n "$language" && ! "$language" =~ ^[a-z][a-z0-9-]{0,16}$ ]]; then echo "[state] env_sh_defaults: language '$language' failed the shape check; seeding empty" >&2 language="" @@ -273,7 +300,7 @@ export ALLOW_NSFW_QUOTES=false export LITCLOCK_LANGUAGE=$language export SHOW_DIAGNOSTICS_SHORTCUT=false export GIFT_MODE_MESSAGE= -export LITCLOCK_RUNTIME_RENDER=false +export LITCLOCK_RUNTIME_RENDER=$runtime_render # export DISPLAY_CLEAR_HOUR=2 # export LITCLOCK_RENDER_LEAD_S=4 # export WEATHER_API_TIMEOUT=15 @@ -367,14 +394,32 @@ _atomic_write_env_sh_finalize() { if [[ -e "$dest" ]]; then owner=$(stat -c '%U:%G' "$dest" 2>/dev/null) || owner="" mode=$(stat -c '%a' "$dest" 2>/dev/null) || mode="" + # MODE FIRST, then ownership. The other order installs an unreadable + # env.sh whenever the destination is root-owned and this runs as pi: the + # mktemp staging file is pi-owned 0600, `sudo chown root:root` succeeds, + # the subsequent unprivileged `chmod 644` then FAILS on a file pi no + # longer owns, and the rename replaces a world-readable config with a + # root-owned 0600 one that every pi-user service — the painter, the + # control server — can no longer read. Both failures are swallowed, so + # the caller reports success (review of litclock-dev#871 Stage B, which + # is the first path to exercise this helper from update.sh). + # + # Setting the mode while pi still owns the temp file is the case that + # matters and it succeeds there; it can still fail (a read-only mount, + # an exotic ACL), which is why the sudo fallback and the `|| true` stay. + # If the chown then fails the file lands pi:pi at whatever mode was set + # — readable, degraded, not broken. That is the safe direction, and it + # is why these stay best-effort rather than becoming a hard abort. + # `$mode` is the DESTINATION's mode, not a constant 0644: a device whose + # env.sh is 0600 keeps 0600. + if [[ -n "$mode" ]]; then + chmod "$mode" "$tmp" 2>/dev/null || sudo chmod "$mode" "$tmp" 2>/dev/null || true + fi if [[ -n "$owner" ]]; then chown "$owner" "$tmp" 2>/dev/null \ || sudo chown "$owner" "$tmp" 2>/dev/null \ || true fi - if [[ -n "$mode" ]]; then - chmod "$mode" "$tmp" 2>/dev/null || true - fi else # First-boot path: mktemp staged the file at 0600. env.sh must be # world-readable so the pi-user `source env.sh` in runtheclock.sh diff --git a/scripts/prepare-for-cloning.sh b/scripts/prepare-for-cloning.sh index f2f1b56..8e30607 100755 --- a/scripts/prepare-for-cloning.sh +++ b/scripts/prepare-for-cloning.sh @@ -834,6 +834,36 @@ if [[ -n "$_ENV_LEAKS" ]]; then _abort_env_credentials "$_ENV_LEAKS" fi unset _ENV_LEAKS + +# litclock-dev#871 /review (adversarial) — the gate above reads ONE path, and +# the atomic env.sh writers stage through `mktemp "${dest}.XXXXXX"` beside it. +# Every in-script failure arm removes that file; a SIGKILL or power loss inside +# the printf -> chmod -> chown -> mv window does not, and `with_env_lock` runs +# the writer in a subshell where bash has reset this script's traps to default, +# so the signal handler does not cover it either. What is left is a full, +# UNREDACTED copy of the previous owner's env.sh — API key, coordinates, city — +# which Step 2's wipe never touches because it targets `env.sh`, which +# `git reset --hard` never removes because it is untracked, and which this gate +# printed `done` over because it never looked. It then ships on every clone. +# +# Refuse rather than delete: a staging file here means a writer died mid-write, +# so the card's state is not what the operator thinks, and silently removing the +# evidence is the wrong answer on a path whose whole job is to certify the card. +shopt -s nullglob +_ENV_STAGING=("$INSTALL_DIR"/env.sh.??????) +shopt -u nullglob +# env.sh.sample is six characters too; it is tracked and is not a staging file. +_ENV_STAGING_REAL=() +for _s in "${_ENV_STAGING[@]}"; do + [[ "$(basename "$_s")" == "env.sh.sample" ]] && continue + _ENV_STAGING_REAL+=("$_s") +done +if [[ ${#_ENV_STAGING_REAL[@]} -gt 0 ]]; then + _abort_env_credentials \ + "abandoned atomic-write staging file(s) beside it: ${_ENV_STAGING_REAL[*]}" \ + "each is a full copy of the previous owner's env.sh; remove them and re-run." +fi +unset _ENV_STAGING _ENV_STAGING_REAL _s echo -e "${GREEN}done${NC}" echo "" diff --git a/scripts/reset-setup.sh b/scripts/reset-setup.sh index f992de1..cf2a203 100755 --- a/scripts/reset-setup.sh +++ b/scripts/reset-setup.sh @@ -872,6 +872,23 @@ if [[ -f "$INSTALL_DIR/env.sh" ]]; then # trailing newlines, and update.sh Phase 3 appends missing sample keys with # `>>`, which would otherwise splice the first one onto the last line. DEFAULTS=$(env_sh_defaults "$GIFT_LANGUAGE_CODE")$'\n' + # litclock-dev#871 /review (adversarial) — sweep abandoned atomic-write + # staging files FIRST. The writers stage through `mktemp "${dest}.XXXXXX"` + # beside env.sh and remove it on every in-script failure arm; a SIGKILL or + # power loss inside the write window does not, and `with_env_lock` runs the + # writer in a subshell where bash has reset this script's traps to default. + # What survives is a full copy of the gifter's env.sh — API key, home + # coordinates, city — which this wipe would leave untouched because it + # writes `env.sh` only, and which then travels to the recipient. + # env.sh.sample is six characters too and is a tracked repo file, so it is + # excluded by name rather than by glob. + shopt -s nullglob + for _stale in "$INSTALL_DIR"/env.sh.??????; do + [[ "$(basename "$_stale")" == "env.sh.sample" ]] && continue + rm -f "$_stale" 2>/dev/null || sudo rm -f "$_stale" 2>/dev/null || true + done + shopt -u nullglob + unset _stale if atomic_write_env_sh "$INSTALL_DIR/env.sh" "$DEFAULTS"; then echo -e "${GREEN}done${NC}" else diff --git a/scripts/update.sh b/scripts/update.sh index 586701c..83385f3 100755 --- a/scripts/update.sh +++ b/scripts/update.sh @@ -19,8 +19,10 @@ # Phase 4 venv hash-gate → pip install if hash changed # │ # Phase 4.5 smoke: $PYTHON src/literary_clock.py --dry-run (60s hard timeout) -# (pass) then: runtime-render marker (re-)stamp, and the -# litclock-dev#871 self-test -> memo (both inert to the update's verdict) +# (pass) then: runtime-render marker (re-)stamp, the litclock-dev#871 +# self-test -> memo, and the litclock-dev#871 Stage B migration, which +# may rewrite LITCLOCK_RUNTIME_RENDER=true in env.sh (all three +# inert to the update's verdict) # │ ╲ # (pass) (fail) # ▼ ▼ @@ -245,8 +247,11 @@ RUNTIME_VALIDATION_MEMO_FILE="$STATE_DIR/runtime-render-validation.json" # litclock-dev#871 Stage A — the POSITIVE record the memo above cannot carry: # {result: passed, duration_s, sha, at_unix}, written by the self-test on a # pass and removed on a fail. The journal is capped at 7 days (the weekly tick -# cadence) and a tick with nothing new exits before Phase 4.5, so a pass logged -# only there is gone before Stage B can read it; and "no memo" is not "passed" +# cadence), so a pass logged only there is gone before Stage B can read it — +# the journal is the reason this file exists, NOT a no-op early-out: there is +# none, and a tick with nothing new runs Phase 4.5 like any other (measured +# 2026-09-23; the claim that it exits first was wrong and is corrected in +# CLAUDE.md too); and "no memo" is not "passed" # (a reset or a rollback tick removes the memo without re-asking). Stage B # gates the flag flip on THIS file, sha-matched (litclock-dev#875 red team). RUNTIME_SELFTEST_RECORD_FILE="$STATE_DIR/runtime-render-selftest.json" @@ -989,20 +994,76 @@ fi SELF_SCRIPT="$(readlink -f "${BASH_SOURCE[0]}")" OLD_SELF_HASH=$(md5sum "$SELF_SCRIPT" | cut -d' ' -f1) +# The runtime-render validation marker (see ANCHOR: runtime-marker-revoke for +# what it proves). Resolved HERE, before the reset, because a bootcheck rollback +# must revoke it before anything can re-exec — see below. The reader honors +# LITCLOCK_RUNTIME_VALIDATED_MARKER (set via env.sh, which runtheclock.sh +# sources), so resolve the SAME path. Subshell so env.sh can't mutate this script. +RUNTIME_MARKER=$( + [[ -f "$INSTALL_DIR/env.sh" ]] && source "$INSTALL_DIR/env.sh" 2>/dev/null + echo "${LITCLOCK_RUNTIME_VALIDATED_MARKER:-$INSTALL_DIR/.runtime-render-validated}" +) +# A bootcheck rollback revokes it UNCONDITIONALLY (litclock-dev#894). Stage B +# flips LITCLOCK_RUNTIME_RENDER to true and nothing writes it back, so without +# this a rollback whose proof inputs happen to match would put the LKG's +# painter on the text tier — a combination that LKG never validated, and whose +# src/literary_clock.py is not a proof input (an LKG predating the litclock-dev#886 +# corrupt-marker fix dies on a bad marker instead of falling back). Dropping +# the marker instead of un-flipping the flag keeps the migration stateless — +# no previous-value record, no one-shot marker, nothing for the weekly +# migration to fight — and the painter declines text without one, so the +# recovered clock paints the pre-rendered tier. The next release it APPLIES +# re-stamps (the stamp is skipped in ROLLBACK_MODE, see Phase 4.5) — not the +# next weekly tick: bootcheck blocks the release it fled, and a tick that skips +# a blocked target exits before Phase 2, so the LKG stays on PNGs until a newer +# release lands. That is the conservative direction on purpose. +# +# BEFORE the reset and the re-exec (litclock-dev#896 review): if the snapshot below cannot +# be made, the run falls back to exec'ing the LKG's OWN update.sh, which lands +# on the right SHA but has no litclock-dev#894 arm. Revoking here holds on every path. +# The rm is VERIFIED: a marker the updater cannot unlink must not be reported +# as removed, because the LKG would then paint a tier it never validated. +if [[ -f "$RUNTIME_MARKER" && "$ROLLBACK_MODE" -eq 1 ]]; then + rm -f "$RUNTIME_MARKER" 2>/dev/null + if [[ -e "$RUNTIME_MARKER" ]]; then + log_error "could not remove the runtime-render validation marker ($RUNTIME_MARKER) during a bootcheck rollback — the last-known-good may paint runtime text it never validated (litclock-dev#894)" + else + log_info "runtime-render validation marker removed: bootcheck rollback — the last-known-good paints pre-rendered images until the next applied release re-earns it (litclock-dev#894)" + fi +fi + # In rollback mode, snapshot THIS script (which has the rollback logic) BEFORE -# the reset. If the LKG target carries a different update.sh — very likely, -# since the LKG usually predates this rollback feature — the self-modification -# guard below would otherwise re-exec the LKG's update.sh, which has no -# rollback logic and would re-resolve the latest (bad) Release, resetting -# straight back to the brick. Re-execing the snapshot instead completes the -# pinned LKG install (it re-reads rollback-target from disk). +# the reset, so the self-modification guard below re-execs THIS release's +# rollback logic rather than the LKG's update.sh. An LKG predating the rollback +# feature (litclock-dev#209 follow-up, first shipped in v0.209.0) would re-resolve the +# latest (bad) Release and reset straight back to the brick; every PUBLIC +# release from v0.219.0 carries it, so the fallback below lands on the right +# SHA, but loses whatever this release added (the litclock-dev#894 revoke is therefore done +# above, before the reset). The snapshot re-reads rollback-target from disk. +# +# The snapshot is a DIRECTORY holding the script AND its lib/ (litclock-dev#896). +# The script sources lib/*.sh relative to its own location, so a lone copy in +# /tmp looked for /tmp/lib/state.sh, found nothing, and ran with no helpers: +# read_sha_file was missing, so it judged rollback-target "malformed", dropped +# out of rollback mode and installed origin/master — the bad release, on a real +# device. Every public release through v0.230.0 shipped that; the lib copy must +# come from THIS tree, before the reset swaps in the LKG's. ROLLBACK_SELF_SNAPSHOT="" +ROLLBACK_SELF_SNAPSHOT_DIR="" if [[ "$ROLLBACK_MODE" -eq 1 ]]; then - ROLLBACK_SELF_SNAPSHOT="$(mktemp /tmp/litclock-update-rollback.XXXXXX 2>/dev/null || echo "")" - if [[ -n "$ROLLBACK_SELF_SNAPSHOT" ]] && cp "$SELF_SCRIPT" "$ROLLBACK_SELF_SNAPSHOT" 2>/dev/null; then + ROLLBACK_SELF_SNAPSHOT_DIR="$(mktemp -d -t litclock-update-rollback.XXXXXX 2>/dev/null || echo "")" + if [[ -n "$ROLLBACK_SELF_SNAPSHOT_DIR" ]] \ + && mkdir -p "$ROLLBACK_SELF_SNAPSHOT_DIR/lib" 2>/dev/null \ + && cp "$SELF_SCRIPT" "$ROLLBACK_SELF_SNAPSHOT_DIR/update.sh" 2>/dev/null \ + && cp "$_THIS_SCRIPT_DIR"/lib/*.sh "$ROLLBACK_SELF_SNAPSHOT_DIR/lib/" 2>/dev/null \ + && [[ -f "$ROLLBACK_SELF_SNAPSHOT_DIR/lib/state.sh" ]]; then + ROLLBACK_SELF_SNAPSHOT="$ROLLBACK_SELF_SNAPSHOT_DIR/update.sh" chmod +x "$ROLLBACK_SELF_SNAPSHOT" 2>/dev/null || true else + [[ -n "$ROLLBACK_SELF_SNAPSHOT_DIR" ]] && rm -rf "$ROLLBACK_SELF_SNAPSHOT_DIR" 2>/dev/null ROLLBACK_SELF_SNAPSHOT="" + ROLLBACK_SELF_SNAPSHOT_DIR="" + log_warn "could not snapshot update.sh + lib/ for the rollback (disk full or /tmp unwritable?) — if update.sh differs, the last-known-good's own update.sh finishes the rollback (litclock-dev#896)" fi fi @@ -1022,17 +1083,31 @@ git submodule update --init --recursive NEW_SELF_HASH=$(md5sum "$SELF_SCRIPT" | cut -d' ' -f1) if [[ "$OLD_SELF_HASH" != "$NEW_SELF_HASH" ]]; then log_info "update.sh changed — re-executing with new version..." - if [[ "$ROLLBACK_MODE" -eq 1 && -n "$ROLLBACK_SELF_SNAPSHOT" && -x "$ROLLBACK_SELF_SNAPSHOT" ]]; then + if [[ "$ROLLBACK_MODE" -eq 1 && -n "$ROLLBACK_SELF_SNAPSHOT" && -f "$ROLLBACK_SELF_SNAPSHOT" ]]; then # Rollback: re-exec the pre-reset snapshot (has rollback logic), NOT the - # on-disk LKG update.sh (which would re-resolve the latest bad Release). - # rollback-target persists on disk, so the snapshot re-detects rollback. - exec bash "$ROLLBACK_SELF_SNAPSHOT" "$OLD_SHA" + # on-disk LKG update.sh. rollback-target persists on disk, so the + # snapshot re-detects rollback. `exec bash`, so the mode bit is moot + # (-f, not -x: a failed chmod must not discard a good snapshot). The + # env var hands the directory to the snapshot run, which removes it + # below once it has sourced its lib/. + LITCLOCK_ROLLBACK_SNAPSHOT_RUNNING="$ROLLBACK_SELF_SNAPSHOT_DIR" exec bash "$ROLLBACK_SELF_SNAPSHOT" "$OLD_SHA" fi chmod +x "$SELF_SCRIPT" exec "$SELF_SCRIPT" "$OLD_SHA" fi # Snapshot no longer needed once we're past the re-exec point (same bytes). -[[ -n "$ROLLBACK_SELF_SNAPSHOT" ]] && rm -f "$ROLLBACK_SELF_SNAPSHOT" 2>/dev/null || true +[[ -n "$ROLLBACK_SELF_SNAPSHOT_DIR" ]] && rm -rf "$ROLLBACK_SELF_SNAPSHOT_DIR" 2>/dev/null || true +# ...nor the one THIS run is executing from, if it is a snapshot (litclock-dev#896 review: +# it leaked one directory per rollback onto the SD card). Its lib/ is already +# sourced and bash reads this script through an open fd, so removing it is +# safe. Only a directory that IS this script's own, and has the snapshot's +# name, is ever removed: the env var alone is not trusted. +if [[ -n "${LITCLOCK_ROLLBACK_SNAPSHOT_RUNNING:-}" \ + && "$LITCLOCK_ROLLBACK_SNAPSHOT_RUNNING" == "$_THIS_SCRIPT_DIR" \ + && "$(basename "$_THIS_SCRIPT_DIR")" == litclock-update-rollback.* ]]; then + rm -rf "$_THIS_SCRIPT_DIR" 2>/dev/null || true +fi +unset LITCLOCK_ROLLBACK_SNAPSHOT_RUNNING NEW_SHA=$(git rev-parse --short HEAD) @@ -1083,11 +1158,9 @@ fi # which runtheclock.sh sources) — resolve the SAME path here or a relocated # marker silently survives the one invalidation layer that covers # semantics-only proof changes, where the digest recompute can't help -# (litclock-dev#611 review). Subshell so env.sh can't mutate this script. -RUNTIME_MARKER=$( - [[ -f "$INSTALL_DIR/env.sh" ]] && source "$INSTALL_DIR/env.sh" 2>/dev/null - echo "${LITCLOCK_RUNTIME_VALIDATED_MARKER:-$INSTALL_DIR/.runtime-render-validated}" -) +# (litclock-dev#611 review). RUNTIME_MARKER is resolved before the reset, +# beside the litclock-dev#894 rollback revoke; env.sh is gitignored, so the +# reset cannot change the answer. if [[ -f "$RUNTIME_MARKER" ]]; then git diff --quiet "$OLD_SHA" "$NEW_SHA" -- \ fonts/ \ @@ -1185,7 +1258,15 @@ _phase3_merge_sample() { # Match both active and commented-out export lines if [[ "$line" =~ ^[#\ ]*(export[[:space:]]+)([A-Za-z_][A-Za-z0-9_]*)= ]]; then varname="${BASH_REMATCH[2]}" - if ! grep -q "^[# ]*export[[:space:]]\+${varname}=" "$INSTALL_DIR/env.sh"; then + # `[[:space:]#]*`, not `[# ]*`: a TAB-indented existing assignment + # went undetected and the sample's line was appended beside it, so + # the sample's value silently became the effective one (the last + # assignment wins when the file is sourced). NOT harmless before + # now — any key whose device value differs from the sample's could + # be silently reverted this way, and the sample's example latitude + # and longitude already differ from what a device is seeded with. + # litclock-dev#871 Stage B is what made it worth finding (review). + if ! grep -q "^[[:space:]#]*export[[:space:]]\+${varname}=" "$INSTALL_DIR/env.sh"; then if [[ "$needs_newline" == "true" ]]; then echo "" >> "$INSTALL_DIR/env.sh" needs_newline=false @@ -1587,8 +1668,50 @@ if [[ -x "$PYTHON" ]]; then # the smoke test reverts every successful update on every device. The # smoke test must mirror the production invocation path or it tests # nothing useful. - timeout 60 "$PYTHON" src/literary_clock.py --dry-run 2>&1 | sed 's/^/[smoke] /' + # + # SOURCE env.sh, for the same reason. `runtheclock.sh` sources it before + # the painter, so without it this gate renders whatever the DEFAULTS say — + # and since litclock-dev#871 Stage B those diverge: a migrated device runs + # the TEXT tier while `LITCLOCK_RUNTIME_RENDER` unset here made the gate + # render a PNG. The gate is the only thing in the system that can revert a + # release, so on a migrated fleet it was the one path that could not catch + # a runtime-render regression — and the marker-revoke block does not cover + # that gap either, because `src/literary_clock.py` is not one of the six + # proof inputs (/review 2026-09-23, adversarial). + # + # In a SUBSHELL, so env.sh cannot leak into this script (the isolation the + # self-test below already uses), and `|| true` because a syntactically + # broken env.sh must not decide the gate: a false RED reverts every update + # on every device — and since litclock-dev#865 the smoke-revert path + # records the release in blocked-sha, so a false RED holds a healthy + # release off the device until a newer one ships. + # + # Sourcing env.sh changed two other things, and both are undone here, the + # way the self-test below already does (litclock-dev#871 port /review): + # - WEATHER is forced off. Before the source, the gate had no location + # and never fetched weather; with env.sh it fetches live, and a slow + # provider (WEATHER_API_TIMEOUT is owner-tunable) can push the dry-run + # past the 60s bound and revert a healthy release. Weather is not what + # this gate tests; it only ever passed without it. + # - The runtime frame goes to a THROWAWAY directory. On a device on the + # text tier the dry-run writes current-quote.png, the file that claims + # to be the panel's current frame on tmpfs; a frame the panel never + # painted does not belong there. If no scratch directory can be made + # the gate still runs, into the live path — a misleading file is a far + # smaller cost than skipping the only revert path. + # Both are passed with env(1), not `export`: env.sh is sourced first, and + # a `readonly` there would make an export fail (bash logs it, but the gate + # carries on), handing the painter the owner's values anyway (port + # /review, reproduced). + smoke_render_dir="$(mktemp -d 2>/dev/null)" || smoke_render_dir="" + [[ -n "$smoke_render_dir" ]] || log_warn "smoke test: could not create a scratch directory; a text-tier dry-run will overwrite the live current-quote.png" + ( + [[ -f "$INSTALL_DIR/env.sh" ]] && { source "$INSTALL_DIR/env.sh" 2>/dev/null || true; } + env WEATHER_ENABLED=false ${smoke_render_dir:+"LITCLOCK_RUNTIME_RENDER_DIR=$smoke_render_dir"} \ + timeout 60 "$PYTHON" src/literary_clock.py --dry-run 2>&1 + ) | sed 's/^/[smoke] /' smoke_rc="${PIPESTATUS[0]}" + [[ -n "${smoke_render_dir:-}" && -d "$smoke_render_dir" ]] && rm -rf -- "$smoke_render_dir" 2>/dev/null # litclock-dev#532 (/review litclock-dev#738): the dry-run never imports the string # catalog, so a half-applied OTA missing languages/ passed smoke and # shipped raw keys to /api/status and every catalog-get script. Probe a @@ -1613,15 +1736,15 @@ if [[ -x "$PYTHON" ]]; then # # Latent only while the registry lists one active language. Once a release # activates a second and an owner selects it, the FIRST probe mismatches and - # short-circuits the rest, smoke_rc=1 takes the revert arm, and nothing - # writes a blocked-sha on that path — only bootcheck does — so the next - # weekly tick resolves the same target and reverts again. It is not - # unrecoverable: the re-exec above (update.sh changed => exec the new copy) - # means the first release that modifies THIS file with the pin runs the - # fixed gate and the device heals itself. Until such a release lands, an - # affected device stays on its old SHA, shows "Update FAILED (reverted)" in - # the PWA, and re-runs the whole pip install every week (the revert deletes - # HASH_FILE, which sets NEED_PIP — it does not recreate the venv). + # short-circuits the rest and smoke_rc=1 takes the revert arm. (When this + # was written nothing wrote a blocked-sha on that path, so every weekly + # tick re-applied and re-reverted the same target; since litclock-dev#865 + # the revert blocks the release instead, so the device stays on its old + # SHA until a newer release ships.) It is not unrecoverable: the re-exec + # above (update.sh changed => exec the new copy) means the first release + # that modifies THIS file with the pin runs the fixed gate and the device + # heals itself. Until such a release lands, an affected device stays on its + # old SHA and shows "Update FAILED (reverted)" in the PWA. # # The pin deliberately checks the CANONICAL catalog rather than the device's # own bundle. A missing translation degrades zz -> en inside get(), which is @@ -1629,7 +1752,22 @@ if [[ -x "$PYTHON" ]]; then # failure that is not, and that is the en -> key branch this still catches. # The cost is that the gate no longer exercises the owner's language at all # (litclock-dev#772). - if [[ "$smoke_rc" -eq 0 ]]; then + # + # NOT in bootcheck rollback mode. There this script is the release being + # FLED, running against the last-known-good's tree, and the probes below + # call subcommands an older LKG does not have: `catalog-get` arrived in + # v0.226.0 and `catalog-count` in v0.227.0. Against such a tree every probe + # fails, the rollback takes the smoke-failure exit instead of finishing, + # and `rollback-target` survives until the LKG's own update.sh re-runs the + # rollback a week later (litclock-dev#871 port /review). The LKG needs no + # catalog check: it is the code that last painted on this device, and the + # dry-run above has just rendered with it. + catalog_probes=1 + if [[ "${ROLLBACK_MODE:-0}" -eq 1 && "$smoke_rc" -eq 0 ]]; then + log_info "Catalog smoke skipped in rollback mode: the last-known-good may predate the probes, and it already painted on this device" + catalog_probes=0 + fi + if [[ "$smoke_rc" -eq 0 && "$catalog_probes" -eq 1 ]]; then catalog_probe="$(LITCLOCK_LANGUAGE=en timeout 30 "$PYTHON" src/eink_display.py catalog-get status.relative.just_now 2>/dev/null)" if [[ "$catalog_probe" = "just now" ]]; then log_info "Catalog smoke passed" @@ -1644,7 +1782,7 @@ if [[ -x "$PYTHON" ]]; then # the recovery screens degrade. Probe the two prefixes whose loss hurts # most: the boot splash (every boot) and the strict recovery screen # (the copy a stuck user is left staring at). - if [[ "$smoke_rc" -eq 0 ]]; then + if [[ "$smoke_rc" -eq 0 && "$catalog_probes" -eq 1 ]]; then for probe in "boot.splash.starting.title=LitClock" "firstboot.splash.setup_incomplete.title=Setup Incomplete"; do probe_key="${probe%%=*}" probe_want="${probe#*=}" @@ -1676,7 +1814,7 @@ if [[ -x "$PYTHON" ]]; then # 75% of it, so a growing catalog forces a deliberate bump rather than # letting the gate quietly go slack. CATALOG_MIN_KEYS=400 - if [[ "$smoke_rc" -eq 0 ]]; then + if [[ "$smoke_rc" -eq 0 && "$catalog_probes" -eq 1 ]]; then catalog_count="$(LITCLOCK_LANGUAGE=en timeout 30 "$PYTHON" src/eink_display.py catalog-count 2>/dev/null)" if [[ "$catalog_count" =~ ^[0-9]+$ ]] && [[ "$catalog_count" -ge "$CATALOG_MIN_KEYS" ]]; then log_info "Catalog size smoke passed ($catalog_count keys)" @@ -1900,8 +2038,11 @@ _defer_runtime_validation() { # calls this with its own bound and token. The validator's default deliberately # does NOT reserve for the self-test that follows it (litclock-dev#875 red team): on the # transition tick (a 600s unit) that would defer the one step that changes -# device behaviour — earning the marker — to protect an INERT probe that has -# its own guard and loses nothing by waiting. Stamp now, defer the self-test. +# device behaviour — earning the marker — to protect a probe that has its own +# guard. Deferring the self-test does defer Stage B's migration by a tick too +# (Stage B acts only on a pass in the same run), but the marker is what every +# later tick needs first, and it takes the longest. Stamp now, defer the +# self-test. _validation_fits_remaining_budget() { local needed="${1:-$VALIDATOR_TIMEOUT_S}" token="${2:-deferred}" local label="${3:-runtime-render validation}" @@ -1930,7 +2071,265 @@ _validation_fits_remaining_budget() { } # --- litclock-dev#835 budget helpers END --- -# litclock-dev#871 Stage A — the runtime-render SELF-TEST, shipped INERT. +# litclock-dev#871 Stage B — the FLIP, and the only part of this feature that +# changes a device's behaviour. +# +# Rewrites `export LITCLOCK_RUNTIME_RENDER=false` to `true` in env.sh, so the +# device paints quotes from TEXT instead of the pre-rendered PNG set. The point +# is not the 148MB (litclock-dev#592 is closed; the image pipeline stays as plan B) — it +# is that a new language becomes a CSV plus a strings bundle instead of a +# ~124MB image set with its own release tag, integrity gate and version pin. +# litclock-dev#532 Stage 4 is what this unblocks. +# +# Why flipping is safe to do unattended, stated once: runtime render DEGRADES, +# it does not blank. Every failure path in `literary_clock.py` returns False +# with "using pre-rendered images" — missing marker, unusable freetype, digest +# mismatch, empty corpus row, any other exception — so a device that cannot +# render text keeps painting PNGs — and below THAT is the plain 144pt time +# display, so losing the PNG tier costs the quote, not the clock. The +# realistic worst case of this flip is "the device stays on images", which is +# where it already is. +# +# What is NOT a backstop here, despite being the obvious one to reach for: +# `litclock-bootcheck`. It asks whether `/run/litclock/heartbeat` is at or +# after BOOT_EPOCH, not whether it is RECENT. A weekly update does not reboot, +# so a clock that painted before the update and stops after it still satisfies +# that check with its PRE-update heartbeat and is never recovered. Bootcheck +# covers a device that fails to paint across a BOOT; it does not cover one +# that stops mid-boot-session, which is the shape an update would produce. +# The safety argument above rests on the degradation ladder alone, and must +# not be written as though it had two layers (review). +# +# FIVE GUARDS, all required, checked in this order because that is the order +# that yields the most useful memo. This list is the single enumeration: the +# comments that count the guards elsewhere (state.sh, the tests) say five, and +# an off-by-one reads as a missing guard to a reviewer auditing against it. +# +# 1. The self-test PASSED on this run (Stage A, `_runtime_render_selftest`). Not the +# marker — the marker says freetype reproduces the GD measurement dump, +# which is a long way short of "this device COMPOSED a quote frame from +# text". Composed, not painted: the self-test is a --dry-run into a +# throwaway directory and never touches the panel, so it proves the +# render path, not the SPI write. +# 2. Its DURABLE pass record exists, parses as JSON, says `passed`, and +# carries a sha equal to HEAD. Not redundant with guard 1: if the record +# could not be written, Stage A's own durability claim is false on this +# device and a later reader would see a migrated clock with no evidence +# for why. Evaluated in `_runtime_render_migrate`. +# 3. The PNG fallback rung is present on disk. `images/metadata/` is checked, +# not `images/`, because that is the directory the painter's fallback +# actually globs — though a non-empty directory is not proof the PNGs in +# it are usable, only that the tier has not been removed. The tier below +# it is the plain 144pt time display, NOT a blank panel: losing images/ +# costs the quote, not the clock. It is still the rung worth keeping — +# a clock showing the time and no quote is the product failing quietly — +# but the earlier wording here overstated it. +# 4. env.sh currently says `false`. Self-limiting: after the flip the pattern +# no longer matches, so this is idempotent with no extra state to keep. It +# is also what makes the DELIBERATE absence of a one-shot marker workable +# (owner call, 2026-09-19) — an owner who sets `false` by hand is re-flipped +# on the next weekly tick, and that is accepted. THE ONE THING THAT WOULD +# CHANGE IT: if anything ever auto-writes `false` — a future auto-revert on +# repeated fallback, say — this would re-flip weekly and oscillate. Nothing +# does today. Add the marker at the same time as any such auto-revert. +# 5. Exactly one active assignment to rewrite, and it is an EXPORTED `=false`. +# A device whose env.sh somehow carries two is not one to guess about, and +# a lone BARE assignment never reached the painter in the first place. +# +# It never fails the update, for the same reason the self-test does not: a +# device that does not migrate is a device that keeps working. +# +# NOT every refusal writes a memo, and that is deliberate. A device already on +# `true`, one with no active assignment, and one whose self-test did not pass +# all return SILENTLY: the first two are goal states rather than negative +# results, and the third already has the self-test's own memo, which a second +# one would contradict. The memo is meant for "this device could have +# migrated and did not": no images tier, or the write itself failed. +# +# Two exceptions. The first is a matter of ORDER: the record guards (guard 2) +# run before the flag is read, so on a device already on `true` or with no +# active assignment, a pass record that is missing, unreadable or not for this +# release still logs `Not migrating …` and writes a `migration-skipped` memo. +# Harmless — nothing is changed, and the next tick with a good record is +# silent — so a memo on a long-migrated clock is usually a failed record +# write, not a regression. The second is a hand-edited env.sh holding an +# exported `=false` AND a later `=true`: effectively `true`, but the outer +# guard only asks whether an exported `=false` exists, so it proceeds and the +# images or duplicate-assignment guard refuses with a memo — on EVERY tick, +# until someone removes one of the two lines. +# +# The guards are numbered above in the order that yields the most useful memo; +# the code checks the cheap ones first. Guards 1-3 are evaluated in +# `_runtime_render_migrate`, guards 4 and 5 inside `_runtime_render_migrate_locked` +# once the lock is held, since both read the file. +_runtime_render_migrate_locked() { + local env_file="$1" body new + body=$(cat "$env_file" 2>/dev/null) || { + log_warn "Stage B: could not read $env_file — not migrating (litclock-dev#871)" + return 1 + } + # Only ACTIVE assignments. A commented `# export ...=false` is documentation + # (`env.sh.sample` documents this key in comments above the live line); + # rewriting one would edit prose. + # Count EVERY active assignment to this variable, whatever its value and + # whether or not it carries `export`. Two earlier versions were too narrow + # and both were defeated (review). Counting only `=false` let an env.sh + # carrying `=false` and a later `=0` migrate "successfully" while the + # effective value stayed off — the last assignment wins when the file is + # sourced. Requiring `export` then let a BARE `LITCLOCK_RUNTIME_RENDER=0` + # do the same, and a bare assignment is not an invented shape: the PWA's + # own writer treats `export` as optional (`_KV_PATTERN` in src/config.py), + # and re-assigning a variable does not clear its export attribute, so a + # bare line genuinely overrides an exported one. + # + # LIMIT, stated rather than implied: this is a line matcher, not a shell + # parser. An assignment inside a heredoc or an `if false` block is counted + # as active, and a `readonly` form is not counted at all. env.sh is a + # generated file of flat assignments — that is the shape config.py writes + # and every seeder emits — so those forms should not occur; if one does, + # the failure is toward refusing to migrate, not toward a bad flip. + # + # Counted both ways, but only an EXPORT form is rewritten. `runtheclock.sh` + # does `source ./env.sh` with no `set -a`, so a bare assignment never + # reaches the painter's environment at all — migrating a lone bare `=false` + # to a bare `=true` changes the file and nothing else, and reports success + # for a device that is still painting PNGs (measured: a child process sees + # the variable UNSET). Such a device was never honouring the flag either + # way; it is refused with a reason rather than silently "migrated". + local total false_count + total=$(printf '%s\n' "$body" | grep -cE '^[[:space:]]*(export[[:space:]]+)?LITCLOCK_RUNTIME_RENDER=' || true) + false_count=$(printf '%s\n' "$body" | grep -cE '^[[:space:]]*export[[:space:]]+LITCLOCK_RUNTIME_RENDER=false[[:space:]]*$' || true) + if [[ "$total" -ne 1 || "$false_count" -ne 1 ]]; then + log_warn "Stage B: expected exactly one active LITCLOCK_RUNTIME_RENDER assignment and for it to be an exported =false in $env_file, found $total active ($false_count exported =false) — not migrating (litclock-dev#871)" + return 1 + fi + # PIPESTATUS, and then a line count. `new=$(... | sed ...)` reports only the + # subshell's status, so a `sed` killed mid-stream emits a PARTIAL body that + # differs from the original — which sailed straight through the + # "did it change anything" check below and got renamed over env.sh, reported + # as a successful migration (review reproduced it with a sed exiting 137). + # An atomic rename protects against a torn write, not against committing a + # complete-looking but truncated body. + local sed_rc before_lines after_lines + # The capture keeps the line's own indentation. Only an export form reaches + # here (see the count above), so the shape is preserved rather than changed. + # + # NO PIPELINE. A herestring, because the fix above only covered the CONSUMER: + # `$?` on a pipeline is its LAST element, so a `sed` killed mid-stream was + # caught and a killed `printf` was not. A producer that dies after emitting a + # partial final line is invisible three times over — sed exits 0 on the short + # input, sed SUPPLIES the missing terminating newline so the line-count belt + # below still matches, and the body differs from the original so the + # "changed nothing" check passes too. Reproduced 2026-09-23: a producer + # exiting 137 committed `OPENWEATHERMAP_APIKEY=original-` over + # `=original-secret` and reported a successful migration. `<<<` removes the + # second process entirely, so there is no unchecked status left to miss; it + # appends the same trailing newline `printf '%s\n'` did, which is what + # `before_lines` below is still computed with. + new=$(sed -E 's/^([[:space:]]*export[[:space:]]+)LITCLOCK_RUNTIME_RENDER=false[[:space:]]*$/\1LITCLOCK_RUNTIME_RENDER=true/' <<<"$body") + sed_rc=$? + if [[ "$sed_rc" -ne 0 ]]; then + log_warn "Stage B: the env.sh rewrite failed (sed exited $sed_rc) — not migrating, env.sh is untouched (litclock-dev#871)" + return 1 + fi + before_lines=$(printf '%s\n' "$body" | wc -l) + after_lines=$(printf '%s\n' "$new" | wc -l) + if [[ "$before_lines" -ne "$after_lines" ]]; then + log_warn "Stage B: the rewrite changed the line count ($before_lines -> $after_lines) — not migrating, env.sh is untouched (litclock-dev#871)" + return 1 + fi + if [[ "$new" == "$body" ]]; then + log_warn "Stage B: the rewrite changed nothing — not migrating (litclock-dev#871)" + return 1 + fi + # The whole body, atomically, preserving owner and mode — NOT an in-place + # sed. A half-written env.sh is sourced by every surface on the device. + _atomic_write_env_sh_finalize_from_body "$env_file" "$new" || return 1 + return 0 +} + +# Stage the new body through mktemp and hand it to state.sh's finalizer, which +# preserves owner and mode. Separate from `atomic_write_env_sh` because we are +# already inside `with_env_lock` — taking the sidecar lock again from here +# would deadlock on a non-reentrant flock. +_atomic_write_env_sh_finalize_from_body() { + local dest="$1" content="$2" tmp + tmp=$(mktemp "${dest}.XXXXXX" 2>/dev/null) || return 1 + printf '%s\n' "$content" > "$tmp" || { rm -f "$tmp"; return 1; } + _atomic_write_env_sh_finalize "$tmp" "$dest" || { rm -f "$tmp" 2>/dev/null; return 1; } + return 0 +} + +_runtime_render_migrate() { + local env_file="$INSTALL_DIR/env.sh" + + if [[ "${RUNTIME_SELFTEST_RESULT:-}" != "pass" ]]; then + # The self-test already wrote its own memo saying why; do not write a + # second one contradicting it. + return 0 + fi + # And the DURABLE record, sha-matched — which is what the Stage A comment + # above RUNTIME_SELFTEST_RECORD_FILE says Stage B gates on, so gate on it. + # It is not redundant with the in-run result: if the record could not be + # written (no jq, an unwritable state dir) then Stage A's own durability + # claim is false on this device, and a later reader — a rollback tick, a + # support bundle, Stage B's own re-run — would see a migrated device with + # no evidence for why. Refusing keeps the device and the record consistent, + # and it retries next week. + local rec_result rec_sha head_sha + if [[ ! -f "$RUNTIME_SELFTEST_RECORD_FILE" ]] || ! command -v jq >/dev/null 2>&1; then + log_info "Not migrating to runtime render: the self-test passed but left no durable record (litclock-dev#871 Stage B)" + _runtime_validation_memo_write migration-skipped "" "the self-test passed but its pass record is missing, so the migration has no durable evidence" + return 0 + fi + # `|| echo ""` is NOT enough: on a record that is a valid object followed by + # truncated garbage, jq PRINTS the matching value and then exits non-zero, + # so the substitution keeps the value and the fallback never runs (review). + # Take the status explicitly and treat any parse failure as no record. + if ! rec_result=$(jq -er '.result // ""' "$RUNTIME_SELFTEST_RECORD_FILE" 2>/dev/null) \ + || ! rec_sha=$(jq -er '.sha // ""' "$RUNTIME_SELFTEST_RECORD_FILE" 2>/dev/null); then + log_info "Not migrating to runtime render: the self-test pass record could not be parsed (litclock-dev#871 Stage B)" + _runtime_validation_memo_write migration-skipped "" "the self-test pass record is not readable JSON, so the migration has no durable evidence" + return 0 + fi + head_sha=$(git rev-parse HEAD 2>/dev/null || echo "") + if [[ "$rec_result" != "passed" || -z "$rec_sha" || "$rec_sha" != "$head_sha" ]]; then + log_info "Not migrating to runtime render: the pass record is not for this release (result='$rec_result', sha='${rec_sha:0:12}', head='${head_sha:0:12}') (litclock-dev#871 Stage B)" + _runtime_validation_memo_write migration-skipped "" "the self-test pass record does not match this release" + return 0 + fi + if [[ ! -f "$env_file" ]]; then + return 0 + fi + # Already migrated, or never eligible — silent, and NOT a memo. A device + # running text render is the goal state, not a negative result. + if ! grep -qE '^[[:space:]]*export[[:space:]]+LITCLOCK_RUNTIME_RENDER=false[[:space:]]*$' "$env_file" 2>/dev/null; then + return 0 + fi + # Guard 3: the fallback rung. `images/metadata/` is what the painter globs. + if [[ ! -d "$INSTALL_DIR/images/metadata" ]] || [[ -z "$(ls -A "$INSTALL_DIR/images/metadata" 2>/dev/null)" ]]; then + log_info "Not migrating to runtime render: images/metadata is missing or empty, so there would be no fallback if a render ever failed (litclock-dev#871 Stage B)" + _runtime_validation_memo_write migration-skipped "" "the PNG fallback tier (images/metadata) is absent, so migrating would drop straight from the text renderer to the plain time display" + return 0 + fi + + ENV_FILE_DEFAULT="$env_file" with_env_lock _runtime_render_migrate_locked "$env_file" + local rc=$? + if [[ "$rc" -eq 0 ]]; then + log_info "MIGRATED to runtime text rendering: LITCLOCK_RUNTIME_RENDER is now true in env.sh (litclock-dev#871 Stage B). The self-test composed a quote frame from text on this device this run; images/ stays on disk as the fallback." + atomic_remove_file "$RUNTIME_VALIDATION_MEMO_FILE" + elif [[ "$rc" -eq 75 ]]; then + log_warn "Not migrating to runtime render: env.sh is locked by another writer — will retry next tick (litclock-dev#871 Stage B)" + _runtime_validation_memo_write migration-skipped "$rc" "env.sh was locked by another writer for the whole 30s wait" + else + log_warn "Not migrating to runtime render: the env.sh rewrite failed (rc=$rc) — will retry next tick (litclock-dev#871 Stage B)" + _runtime_validation_memo_write migration-skipped "$rc" "the env.sh rewrite did not complete; env.sh is unchanged" + fi + return 0 +} + +# litclock-dev#871 Stage A — the runtime-render SELF-TEST, shipped INERT one +# release ahead of Stage B. # # The marker says this device's freetype reproduces the GD measurement dump; # it does not say this device can paint a quote from text. Between the two sit @@ -1938,8 +2337,9 @@ _validation_fits_remaining_budget() { # own guards, and every reason `_runtime_render_enabled()` has to decline — # and today every one of them degrades SILENTLY to the PNG tier, so a fleet # device could carry a valid marker and still never render a line of text. -# Stage B (release N+1) will flip LITCLOCK_RUNTIME_RENDER on devices whose -# self-test passes; this release only asks the question and records the answer. +# Stage B flips LITCLOCK_RUNTIME_RENDER on devices whose self-test passes and +# whose other four guards hold — see `_runtime_render_migrate` above. This +# function only asks the question and records the answer; it changes nothing. # # The same dry-run the smoke gate runs, with the renderer FORCED on (env.sh's # flag says what the owner chose, not what the device can do) and @@ -1961,12 +2361,20 @@ _validation_fits_remaining_budget() { # anything else = the painter died), or `selftest-deferred` when the budget or # the scratch directory is gone. A pass writes the POSITIVE record # (RUNTIME_SELFTEST_RECORD_FILE) with its DURATION: Stage B needs a durable -# on-device pass, sha-matched, and needs to know the device renders inside -# the 4s lead, not merely that it renders (litclock-dev#875 review + red team). +# on-device pass, sha-matched — and Stage B does gate on this file, not just +# on the in-run verdict. The DURATION is recorded for diagnosis, NOT as a +# gate: litclock-dev#883 established that it is whole-process wall time +# (env.sh sourcing, interpreter start, the PIL import, then the render), so +# it is not comparable to the 4s render lead, and the painter pays that same +# startup BEFORE it computes its target instant, so the lead does not cover +# it either. A "renders inside the lead" threshold built on this number +# would be measuring the wrong thing; deciding what such a gate should +# actually measure is its own piece of work (litclock-dev#875 review + red team). # One sample, one minute, one random row: a corpus gap at that minute reads as # a fallback too, and the painter now says so in its own [selftest] line, so # a `selftest-failed` is "did not render text this time", never proof of # incapacity — Stage B acts only on a PASS. + # The pass record (RUNTIME_SELFTEST_RECORD_FILE). $1 duration in seconds, one # decimal. Same jq/atomic/0644 shape as the memo writer; best-effort. _runtime_selftest_record_write() { @@ -1989,15 +2397,19 @@ _runtime_selftest_record_write() { _runtime_render_selftest() { local rc dir t0 dur_ms duration + # Stage B reads this. Set on EVERY arm, including the early returns, so + # no stale or unset value can be mistaken for a pass. + RUNTIME_SELFTEST_RESULT=deferred _validation_fits_remaining_budget "$SELFTEST_TIMEOUT_S" selftest-deferred "the runtime-render self-test" || return 0 if ! dir=$(mktemp -d 2>/dev/null) || [[ -z "$dir" ]]; then log_info "deferring the runtime-render self-test to the next update: could not create a scratch directory (litclock-dev#871)" _runtime_validation_memo_write selftest-deferred "" "self-test: could not create a scratch directory (mktemp -d failed)" return 0 fi - log_info "Running the runtime-render self-test (litclock-dev#871 Stage A; inert — records a verdict, changes nothing)..." - # Microseconds via EPOCHREALTIME (bash 5): the figure is compared against a - # 4s render lead, and whole-second $SECONDS is ±1s on it (litclock-dev#875 red team). + log_info "Running the runtime-render self-test (litclock-dev#871 Stage A: records a verdict; a PASS lets Stage B migrate an eligible device)..." + # Microseconds via EPOCHREALTIME (bash 5): whole-second $SECONDS is ±1s on a + # figure of ~3.5s (litclock-dev#875 red team). It is recorded for diagnosis, NOT + # compared against the 4s render lead: litclock-dev#883, above. # # Strip EVERY non-digit (litclock-dev#879, widened in litclock-dev#881). EPOCHREALTIME's decimal # separator follows LC_NUMERIC, so under a comma locale the original @@ -2045,10 +2457,10 @@ _runtime_render_selftest() { # exit — `if cmd | sed; then` tests sed (the smoke gate's own lesson). ( [[ -f "${INSTALL_DIR:-}/env.sh" ]] && source "${INSTALL_DIR:-}/env.sh" 2>/dev/null - export LITCLOCK_RUNTIME_RENDER=true - export LITCLOCK_RUNTIME_RENDER_DIR="$dir" - export WEATHER_ENABLED=false - timeout "$SELFTEST_TIMEOUT_S" "$PYTHON" src/literary_clock.py --dry-run --require-runtime-render 2>&1 + # env(1), not `export`, for the reason the smoke gate gives: a + # `readonly` in the sourced env.sh would defeat an export. + env LITCLOCK_RUNTIME_RENDER=true LITCLOCK_RUNTIME_RENDER_DIR="$dir" WEATHER_ENABLED=false \ + timeout "$SELFTEST_TIMEOUT_S" "$PYTHON" src/literary_clock.py --dry-run --require-runtime-render 2>&1 ) | sed 's/^/[selftest] /' rc="${PIPESTATUS[0]}" dur_ms=$(( (${EPOCHREALTIME//[^0-9]/} - t0) / 1000 )) @@ -2056,11 +2468,13 @@ _runtime_render_selftest() { # Only ever the fresh mktemp directory: -d, and never a fallback path. [[ -n "$dir" && -d "$dir" ]] && rm -rf -- "$dir" 2>/dev/null if [[ "$rc" -eq 0 ]]; then - log_info "runtime-render self-test PASSED in ${duration}s — this device renders text (litclock-dev#871 Stage A; LITCLOCK_RUNTIME_RENDER is unchanged this release)" + log_info "runtime-render self-test PASSED in ${duration}s — this device composed a quote frame from text (litclock-dev#871; Stage B may now migrate LITCLOCK_RUNTIME_RENDER on an eligible device, and says so if it does; a device already on true normally logs nothing further)" _runtime_selftest_record_write "$duration" + RUNTIME_SELFTEST_RESULT=pass return 0 fi # A fail retires any earlier pass: Stage B must never flip on a stale one. + RUNTIME_SELFTEST_RESULT=fail atomic_remove_file "$RUNTIME_SELFTEST_RECORD_FILE" if [[ "$rc" -eq 124 ]]; then log_info "runtime-render self-test did not finish within ${SELFTEST_TIMEOUT_S}s (${duration}s elapsed) — staying as is (this is not an update failure)" @@ -2108,9 +2522,9 @@ if [[ "$smoke_rc" -eq 0 ]]; then # Pi Zero 2W — on every weekly tick would be pure cost. # # Not in ROLLBACK_MODE (litclock-dev#835): that run exists to get the - # last-known-good clock painting again as fast as possible, and the LKG's - # proof inputs almost always differ from HEAD's, so the revoke block above - # has just removed the marker. Spending minutes re-earning it here, with + # last-known-good clock painting again as fast as possible, and the revoke + # block above has just removed the marker — unconditionally in this mode + # since litclock-dev#894. Spending minutes re-earning it here, with # litclock.timer still stopped, is the opposite of recovery. The next # normal update re-stamps. # @@ -2170,6 +2584,10 @@ if [[ "$smoke_rc" -eq 0 ]]; then # declines before it tries. Not in ROLLBACK_MODE, for the reason above. if [[ -f "$RUNTIME_MARKER" && -x "$PYTHON" && "${ROLLBACK_MODE:-0}" -ne 1 ]]; then _runtime_render_selftest + # litclock-dev#871 Stage B — the flip, gated on the verdict just recorded. + # Inside the same marker/rollback gate on purpose: without a marker the + # painter declines before it tries, so there is nothing to migrate to. + _runtime_render_migrate fi else log_error "Smoke test failed (exit $smoke_rc) — reverting to $REVERT_SHA" diff --git a/src/control_server/routes/status.py b/src/control_server/routes/status.py index 1c295f9..d703b1d 100644 --- a/src/control_server/routes/status.py +++ b/src/control_server/routes/status.py @@ -89,7 +89,25 @@ # its deferral (budget or scratch directory gone). Distinct tokens, so a # marker-bearing device whose validation passed is never reported as # "validation deferred" (litclock-dev#875 review). -RUNTIME_VALIDATION_RESULTS = frozenset({"deferred", "timeout", "failed", "selftest-failed", "selftest-deferred"}) +# `migration-skipped` since litclock-dev#871 Stage B: the same memo file carries +# the reason a device that COULD have migrated to runtime render did not — no +# PNG fallback tier, a pass record that is missing or not for this release, a +# locked env.sh, or a rewrite that did not complete. It was missing from this +# set when Stage B first wrote it, which meant every refusal was parsed, +# rejected here and reported as `null` — the same payload a healthy device +# sends, on the one feature whose safety argument is that a refusal is visible +# (/review 2026-09-23, adversarial). `tests/test_runtime_render_autostamp.py` +# now extracts every token `update.sh` writes and asserts membership, so the +# shell and this allowlist cannot drift apart again. +# +# `deferred` is LIVE: it is the marker-stamp validator's budget deferral, +# written through the default `$2` of `_validation_fits_remaining_budget` / +# `_defer_runtime_validation`. `selftest-deferred` is the self-test's own. An +# earlier version of this comment called `deferred` dead, which invited +# dropping it — and that device's memo would then reach the PWA as `null`. +RUNTIME_VALIDATION_RESULTS = frozenset( + {"deferred", "timeout", "failed", "selftest-failed", "selftest-deferred", "migration-skipped"} +) MAX_RUNTIME_VALIDATION_MEMO_BYTES = 8 * 1024 # litclock-dev#274 follow-up — adversarial-review P1: budget for treating a diff --git a/tests/test_first_boot_flow.py b/tests/test_first_boot_flow.py index c125da5..700ce21 100644 --- a/tests/test_first_boot_flow.py +++ b/tests/test_first_boot_flow.py @@ -1445,13 +1445,19 @@ def _block_keys(body): return out -def env_sh_defaults(language=None): +def _call(language, runtime_render): + if runtime_render is None: + return "env_sh_defaults" if language is None else f'env_sh_defaults "{language}"' + return f'env_sh_defaults "{language or ""}" "{runtime_render}"' + + +def env_sh_defaults(language=None, runtime_render=None): """Run the REAL helper out of scripts/lib/state.sh and return its stdout. Executed, never parsed: the whole point of litclock-dev#840 is that there is one body, so every expectation in this module derives from running it. """ - call = "env_sh_defaults" if language is None else f'env_sh_defaults "{language}"' + call = _call(language, runtime_render) r = subprocess.run( ["bash", "-c", f'. "{STATE_SH}"\n{call}'], capture_output=True, @@ -1674,7 +1680,10 @@ def test_first_boot_fallback_heredoc_matches_the_helper(): heredoc = next(body for _, kind, body in writes if kind == "heredoc") # The heredoc body is the file's lines up to the terminator; the helper # emits the same lines plus a trailing newline. - assert "\n".join(heredoc) + "\n" == env_sh_defaults(), ( + # The heredoc is first-boot's OWN fallback, so it must match the call + # first-boot makes — `env_sh_defaults "" true` (litclock-dev#871 Stage B + # made the render default a parameter, and a fresh flash asks for text). + assert "\n".join(heredoc) + "\n" == env_sh_defaults("", "true"), ( "first-boot.sh's state.sh-missing fallback heredoc has drifted from env_sh_defaults(). " "It is the one copy that cannot call the helper, so it must be kept in step by hand — " "paste the helper's exact body (litclock-dev#840)." @@ -1749,7 +1758,7 @@ def test_first_boot_falls_back_when_state_sh_is_too_old(tmp_path): r = subprocess.run(["bash", "-c", harness], timeout=10, capture_output=True, text=True) assert r.returncode == 0, r.stderr written = env_file.read_text() - assert written == env_sh_defaults(), ( + assert written == env_sh_defaults("", "true"), ( # first-boot seeds a FRESH device "with an older lib/state.sh (atomic_write_env_sh but no env_sh_defaults) first-boot did " f"not fall back to the inline heredoc — it seeded:\n{written!r}" ) @@ -1760,7 +1769,7 @@ def test_first_boot_actually_writes_every_env_sample_key(tmp_path, with_state_li """Both first-boot arms must WRITE the full body, not merely contain it.""" written = _run_first_boot_default_env(tmp_path, with_state_lib) arm = "flock writer" if with_state_lib else "state.sh-missing heredoc fallback" - assert written == env_sh_defaults(), ( + assert written == env_sh_defaults("", "true"), ( # first-boot seeds a FRESH device f"first-boot's {arm} WROTE an env.sh that is not env_sh_defaults(). The source block may " f"look complete while the value handed to the writer is not — assert on the file, not the " f"literal (litclock-dev#783/litclock-dev#840).\n--- written ---\n{written}" diff --git a/tests/test_runtime_render_autostamp.py b/tests/test_runtime_render_autostamp.py index b2f0996..8c5a969 100644 --- a/tests/test_runtime_render_autostamp.py +++ b/tests/test_runtime_render_autostamp.py @@ -18,6 +18,7 @@ import json import os import re +import shlex import subprocess import sys import tempfile @@ -62,8 +63,10 @@ SELFTEST_TIMEOUT_DEFAULT_S = 2 SELFTEST_BUDGET_RESERVE_S = 120 + def render_lead_s(): - """The render lead Stage B will gate `duration_s` against, read from the + """The render lead, the sanity bound an honestly recorded `duration_s` is + checked against (NOT a Stage B gate: Stage B reads `result` and `sha`), read from the source of truth so a change to the lead moves the test with it (litclock-dev#883). Read by AST, not by regex, and called from the test rather than at import. @@ -111,8 +114,12 @@ def test_the_validator_and_its_dump_ship(self): """Both are needed ON the device — the check runs there, not on a build host.""" assert VALIDATOR.is_file(), "the validator must ship; the check runs on the device" assert DUMP.is_file(), "the expected-measurement dump must ship alongside it" - out = subprocess.run(["git", "ls-files", "--error-unmatch", str(DUMP.relative_to(REPO_ROOT))], - cwd=REPO_ROOT, capture_output=True, text=True) + out = subprocess.run( + ["git", "ls-files", "--error-unmatch", str(DUMP.relative_to(REPO_ROOT))], + cwd=REPO_ROOT, + capture_output=True, + text=True, + ) assert out.returncode == 0, "the dump must be tracked, or a fresh clone cannot validate" @@ -181,7 +188,7 @@ def test_it_is_bounded_and_the_bound_fits_the_unit(self): assert vm, "update.sh must define VALIDATOR_TIMEOUT_S as a plain integer" validator_s = int(vm.group(1)) stamp_at = STAMP_CALL.search(body).start() - window = body[max(0, stamp_at - 400):stamp_at + 200] + window = body[max(0, stamp_at - 400) : stamp_at + 200] assert 'timeout "$VALIDATOR_TIMEOUT_S" "$PYTHON" tools/validate_measurement.py' in window, ( "the validation call must be bounded by VALIDATOR_TIMEOUT_S — one value, used by " "both the bound and the remaining-budget guard" @@ -192,9 +199,7 @@ def test_it_is_bounded_and_the_bound_fits_the_unit(self): assert um, "litclock-update.service must set an integer TimeoutStartSec" unit_s = int(um.group(1)) - executed = "\n".join( - line for line in body.splitlines() if not line.lstrip().startswith("#") - ) + executed = "\n".join(line for line in body.splitlines() if not line.lstrip().startswith("#")) inner = [int(n) for n in re.findall(r"\btimeout (\d+)\b", executed)] + [validator_s] assert MEASURED_VALIDATOR_S * VALIDATOR_MARGIN <= validator_s <= 2 * MEASURED_VALIDATOR_S, ( f"validator bound {validator_s}s must cover the measured {MEASURED_VALIDATOR_S}s " @@ -215,7 +220,7 @@ def test_it_is_skipped_in_rollback_mode(self): body = self._body() stamp_at = STAMP_CALL.search(body).start() guard_at = body.rindex('if [[ ! -f "$RUNTIME_MARKER"', 0, stamp_at) - guard = body[guard_at:body.index("\n", guard_at)] + guard = body[guard_at : body.index("\n", guard_at)] assert '"${ROLLBACK_MODE:-0}" -ne 1' in guard, guard def test_a_failed_validation_does_not_fail_the_update(self): @@ -223,7 +228,7 @@ def test_a_failed_validation_does_not_fail_the_update(self): fallback is impossible; a wrong render is not.""" body = self._body() stamp_at = STAMP_CALL.search(body).start() - window = body[stamp_at:stamp_at + 1200] + window = body[stamp_at : stamp_at + 1200] assert "not an update failure" in window, ( "the failure arm must say, on the code, that it does not fail the update" ) @@ -315,13 +320,17 @@ def _run( # `set -e`, so an unstubbed call would print `command not found` and # carry on green); the real function is driven on its own in # TestRuntimeRenderSelftestExecutes below. - '_runtime_render_selftest() { echo STUB_SELFTEST; }\n' + "_runtime_render_selftest() { echo STUB_SELFTEST; }\n" + # ...and Stage B's migration, called straight after it. It was left + # unstubbed and printed `command not found` on every marker-present + # run while the tests stayed green (litclock-dev#871 port /review). + "_runtime_render_migrate() { echo STUB_MIGRATE; }\n" f"{self._budget_helpers()}" # Stub AFTER the lifted helpers so the stub wins; the guard itself # is the real one. f"{elapsed_stub}" f"{self._keep_arm()}" - 'echo REACHED_END\n' + "echo REACHED_END\n" ) return subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60) @@ -330,6 +339,25 @@ def _memo(tmp_path): path = tmp_path / "memo.json" return json.loads(path.read_text()) if path.exists() else None + def test_the_migration_runs_after_the_selftest_only_with_a_marker_and_never_in_rollback(self, tmp_path): + """Executed, not positional: the flip is gated exactly like the self-test.""" + # One directory per run: the marker is a file in tmp_path, so a shared + # one carries the first run's marker into the "no marker" run. + dirs = [tmp_path / n for n in ("present", "absent", "rollback")] + for d in dirs: + d.mkdir() + present = self._run(dirs[0], marker_exists=True, validator_rc=0) + out = present.stdout + assert "STUB_SELFTEST" in out and "STUB_MIGRATE" in out, out + present.stderr + assert out.index("STUB_SELFTEST") < out.index("STUB_MIGRATE"), "the flip must follow the verdict" + absent = self._run(dirs[1], marker_exists=False, validator_rc=1) + assert "STUB_MIGRATE" not in absent.stdout, "no marker, nothing to migrate to" + rollback = self._run(dirs[2], marker_exists=True, validator_rc=0, rollback_mode=True) + assert "STUB_MIGRATE" not in rollback.stdout, "a rollback must never flip the flag" + for r in (present, absent, rollback): + assert "command not found" not in r.stderr, f"an unstubbed call in the KEEP arm:\n{r.stderr}" + assert "REACHED_END" in r.stdout + def test_rollback_mode_skips_the_validation(self, tmp_path): """litclock-dev#835, executed: the source pin on the guard line is one substring; this proves the arm is actually not entered.""" @@ -393,7 +421,6 @@ def test_an_unlimited_budget_runs_the_validation(self, tmp_path): r = self._run(tmp_path, marker_exists=False, validator_rc=0, elapsed_s=5000, installed_budget_s=0) assert "[validate] FAKE_VALIDATOR_RAN" in r.stdout, r.stdout - def test_a_failing_validation_lets_the_update_continue(self, tmp_path): r = self._run(tmp_path, marker_exists=False, validator_rc=1) # The block pipes the validator through `sed 's/^/[validate] /'` with @@ -444,7 +471,6 @@ def test_the_failure_arm_removes_the_validators_litter_and_nothing_else(self, tm for k in keep: assert k.exists(), f"{k.name} is not validator litter and must not be removed" - # ── litclock-dev#847 item 1: the negative-result memo ─────────────────── def test_a_deferral_is_memoed_with_its_reason(self, tmp_path): @@ -460,10 +486,14 @@ def test_a_deferral_is_memoed_with_its_reason(self, tmp_path): assert isinstance(memo["at_unix"], int) and memo["at_unix"] > 1_600_000_000, memo assert re.fullmatch(r"[0-9a-f]{40}", memo["sha"] or ""), memo - @pytest.mark.parametrize("kw", [ - {"elapsed_s": "unknown", "installed_budget_s": 1800}, - {"elapsed_s": 100, "installed_budget_s": None}, - ], ids=["unknown-elapsed", "unknown-budget"]) + @pytest.mark.parametrize( + "kw", + [ + {"elapsed_s": "unknown", "installed_budget_s": 1800}, + {"elapsed_s": 100, "installed_budget_s": None}, + ], + ids=["unknown-elapsed", "unknown-budget"], + ) def test_every_deferral_branch_writes_the_memo(self, tmp_path, kw): r = self._run(tmp_path, marker_exists=False, validator_rc=0, **kw) assert "deferring" in r.stdout, r.stdout @@ -524,7 +554,6 @@ def test_the_memo_is_the_writer_helper_not_an_inline_write(self, tmp_path): 'atomic_remove_file "$RUNTIME_VALIDATION_MEMO_FILE"', "" ), "the arm must not write the memo file directly" - def test_the_memo_is_left_readable_by_the_control_server(self, tmp_path): """litclock-dev#854 review — atomic_write_file stages with mktemp (0600, owned by whoever runs the script), so a maintainer's `sudo ./scripts/update.sh` @@ -623,9 +652,9 @@ def test_a_manual_run_is_waived_not_unknown(self, tmp_path): "kw", [ dict(invocation_id="abc", owner="other", start_us=1_000_000, budget_us="t 1800000000"), # not ours - dict(invocation_id="abc", owner="abc", start_us=0, budget_us="t 1800000000"), # zero stamp - dict(invocation_id="abc", owner="abc", start_us="", budget_us="t 1800000000"), # empty - dict(invocation_id="abc", owner="abc", start_us="abc", budget_us="t 1800000000"), # garbage + dict(invocation_id="abc", owner="abc", start_us=0, budget_us="t 1800000000"), # zero stamp + dict(invocation_id="abc", owner="abc", start_us="", budget_us="t 1800000000"), # empty + dict(invocation_id="abc", owner="abc", start_us="abc", budget_us="t 1800000000"), # garbage dict(invocation_id="abc", owner="abc", start_us=1_000_000, budget_us="t 1800000000", systemctl_rc=1), ], ) @@ -748,9 +777,7 @@ def _block(timeout_s=None): # keeps a zero zero and makes a real 30s grace test-sized. grace_re = r"^PIGEN_VALIDATOR_KILL_GRACE_S=(\d+)$" grace = min(int(re.search(grace_re, block, re.MULTILINE).group(1)), timeout_s) - block, n = re.subn( - grace_re, f"PIGEN_VALIDATOR_KILL_GRACE_S={grace}", block, flags=re.MULTILINE - ) + block, n = re.subn(grace_re, f"PIGEN_VALIDATOR_KILL_GRACE_S={grace}", block, flags=re.MULTILINE) assert n == 1, "PIGEN_VALIDATOR_KILL_GRACE_S must be one plain assignment the test can scale" return block @@ -784,7 +811,10 @@ def _run(self, tmp_path, *, rc=0, sleep_s=0, timeout_s=None, ignore_term=False, try: proc = subprocess.run( ["bash", "-e", "-c", self._block(timeout_s)], - cwd=tmp_path, capture_output=True, text=True, timeout=20, + cwd=tmp_path, + capture_output=True, + text=True, + timeout=20, ) except subprocess.TimeoutExpired: pytest.fail("the block did not return within 20s: the bound is not enforced against this interpreter") @@ -864,8 +894,7 @@ def test_a_marker_stamped_after_the_bound_expired_is_removed(self, tmp_path): warnings = self._annotations(proc) assert len(warnings) == 1 and "timed out" in warnings[0], proc.stdout assert not (tmp_path / ".runtime-render-validated").exists(), ( - "a marker written after the bound expired shipped in the image while the annotation " - "said there was none" + "a marker written after the bound expired shipped in the image while the annotation said there was none" ) @@ -881,12 +910,15 @@ def test_check_passes_and_stamps_here(self, tmp_path): pytest.importorskip( "freetype", reason="freetype-py is not installed: `pip install --user --break-system-packages freetype-py` " - "to run the real validator here (litclock-dev#840)", + "to run the real validator here (litclock-dev#840)", ) marker = tmp_path / ".runtime-render-validated" r = subprocess.run( [sys.executable, str(VALIDATOR), "check", "--stamp", "--marker", str(marker)], - cwd=REPO_ROOT, capture_output=True, text=True, timeout=600, + cwd=REPO_ROOT, + capture_output=True, + text=True, + timeout=600, ) assert r.returncode == 0, f"validation failed on the dev box\n{r.stdout}\n{r.stderr}" assert marker.is_file(), "PASS must write the marker at the requested path" @@ -906,9 +938,11 @@ def test_stamp_refuses_a_dump_that_is_not_the_committed_one(self, tmp_path): bogus = tmp_path / "bogus.json.gz" bogus.write_bytes(b"not a gzip") r = subprocess.run( - [sys.executable, str(VALIDATOR), "check", "--stamp", - "--dump", str(bogus), "--marker", str(marker)], - cwd=REPO_ROOT, capture_output=True, text=True, timeout=120, + [sys.executable, str(VALIDATOR), "check", "--stamp", "--dump", str(bogus), "--marker", str(marker)], + cwd=REPO_ROOT, + capture_output=True, + text=True, + timeout=120, ) assert r.returncode == 2, f"expected the refusal exit (2), got {r.returncode}\n{r.stderr}" assert "refusing --stamp" in (r.stdout + r.stderr) @@ -973,13 +1007,13 @@ def _selftest_fn(self): end = body.index("\n}\n", start) + len("\n}\n") fn = body[start:end] for required, why in ( - ("export LITCLOCK_RUNTIME_RENDER=true", "the renderer forced on"), + ("env LITCLOCK_RUNTIME_RENDER=true", "the renderer forced on, via env(1) so a readonly cannot defeat it"), ("--require-runtime-render", "the flag that makes a fallback a failure"), ("LITCLOCK_RUNTIME_RENDER_DIR", "the throwaway frame directory"), - ("export WEATHER_ENABLED=false", "weather forced off — a capability probe stays off the network"), - ("source \"${INSTALL_DIR:-}/env.sh\"", "env.sh sourced for the device's language"), + ("WEATHER_ENABLED=false \\", "weather forced off — a capability probe stays off the network"), + ('source "${INSTALL_DIR:-}/env.sh"', "env.sh sourced for the device's language"), ("selftest-deferred", "the deferral token, distinct from the validator's"), - ("_validation_fits_remaining_budget \"$SELFTEST_TIMEOUT_S\"", "the shared budget guard, own bound"), + ('_validation_fits_remaining_budget "$SELFTEST_TIMEOUT_S"', "the shared budget guard, own bound"), ('rc="${PIPESTATUS[0]}"', "the painter's status, not sed's"), ('[[ -n "$dir" && -d "$dir" ]] && rm -rf -- "$dir"', "cleanup of the fresh directory only"), ("_runtime_selftest_record_write", "the durable pass record"), @@ -1005,6 +1039,8 @@ def _run( selftest_timeout_s=SELFTEST_TIMEOUT_DEFAULT_S, mktemp_fails=False, pass_record_exists=False, + preseed_result=None, + env_extra="", ): log = tmp_path / "painter.log" fake_py = tmp_path / "python3" @@ -1033,6 +1069,7 @@ def _run( install.mkdir(exist_ok=True) (install / "env.sh").write_text( f"export LITCLOCK_LANGUAGE={language}\nexport LITCLOCK_RUNTIME_RENDER=false\nexport WEATHER_ENABLED=true\n" + + env_extra ) memo = tmp_path / "memo.json" # A test may call this twice in one tmp_path: start each run clean, or @@ -1067,10 +1104,13 @@ def _run( f"SELFTEST_TIMEOUT_S={selftest_timeout_s}\nVALIDATOR_BUDGET_RESERVE_S={SELFTEST_BUDGET_RESERVE_S}\n" f"{stub}" + ("mktemp() { return 1; }\n" if mktemp_fails else "") + + (f"RUNTIME_SELFTEST_RESULT={preseed_result}\n" if preseed_result else "") + f"{self._record_fn()}" + f"{self._selftest_fn()}" "_runtime_render_selftest\n" - 'echo "REACHED_END rc=$?"\n' + # RESULT is what Stage B reads (`!= pass` returns silently). Printed + # on the same line so `rc=$?` is still the function's status. + 'echo "REACHED_END rc=$? RESULT=${RUNTIME_SELFTEST_RESULT:-}"\n' ) record = tmp_path / "selftest.json" record.unlink(missing_ok=True) @@ -1094,6 +1134,42 @@ def _record(tmp_path): path = tmp_path / "selftest.json" return json.loads(path.read_text()) if path.exists() else None + @pytest.mark.parametrize( + "kwargs, want", + [ + (dict(painter_rc=0), "pass"), + (dict(painter_rc=3), "fail"), + (dict(painter_rc=124), "fail"), + # The early returns, with a stale `pass` pre-seeded: the reset at the + # top is what stops Stage B acting on a verdict nobody reached. + (dict(painter_rc=0, mktemp_fails=True, preseed_result="pass"), "deferred"), + (dict(painter_rc=0, elapsed_s=5000, installed_budget_s=600, preseed_result="pass"), "deferred"), + ], + ) + def test_the_verdict_stage_b_reads_matches_the_outcome(self, tmp_path, kwargs, want): + """The handoff between the two stages: Stage B acts only on + RUNTIME_SELFTEST_RESULT == pass, and every Stage B test injects that + value by hand, so nothing pinned the producer (port /review).""" + r, _painter, _memo = self._run(tmp_path, **kwargs) + # Whole token, end of line: a substring match let `passed` satisfy + # `pass`, which is exactly the rename this test exists to catch. + assert f"RESULT={want}\n" in r.stdout, r.stdout + r.stderr + + def test_a_readonly_env_sh_cannot_defeat_the_selftests_overrides(self, tmp_path): + """Same hole as the smoke gate's (port /review round 2): with `export`, a + readonly in env.sh put the probe on the network and its frame in the + live directory.""" + r, painter, _memo = self._run( + tmp_path, + painter_rc=0, + env_extra=( + "readonly WEATHER_ENABLED LITCLOCK_RUNTIME_RENDER\n" + "export LITCLOCK_RUNTIME_RENDER_DIR=/run/litclock\nreadonly LITCLOCK_RUNTIME_RENDER_DIR\n" + ), + ) + assert "render=true" in painter and "weather=false" in painter, painter + r.stderr + assert "dir=/run/litclock" not in painter and "exists=yes" in painter, painter + def test_a_pass_forces_the_renderer_on_with_the_devices_language_and_writes_no_memo(self, tmp_path): r, painter, memo = self._run(tmp_path, painter_rc=0) assert "REACHED_END rc=0" in r.stdout, r.stdout + r.stderr @@ -1104,7 +1180,7 @@ def test_a_pass_forces_the_renderer_on_with_the_devices_language_and_writes_no_m assert "lang=xx" in painter, "env.sh must be sourced so the device's language is the one rendered" assert "weather=false" in painter, "weather must be forced OFF, whatever env.sh says" assert "exists=yes" in painter, "the frame directory must EXIST while the painter runs" - assert "self-test PASSED in " in r.stdout, "the duration is the evidence Stage B needs" + assert "self-test PASSED in " in r.stdout, "a pass must be logged with its duration" assert memo is None rec = self._record(tmp_path) assert rec is not None and rec["result"] == "passed", "a pass must leave a DURABLE record for Stage B" @@ -1289,12 +1365,12 @@ class TestSelfTestDurationReachesTheRecord: # 1.4s — so subtracting a quantum "for safety" admits values the shipped # formatter cannot honestly produce. It cost a real mutation: `d * 0.98` # passed all three tests, recording a 4.05s paint as 3.92s — inside the lead. - SLEEP_S = 1.5 # deliberately mid-second: whole-second arithmetic + SLEEP_S = 1.5 # deliberately mid-second: whole-second arithmetic # lands 0.5s away whichever way it rounds CONTROL_SHORT_S = 0.3 CONTROL_LONG_S = 2.5 - MIN_MOVE_S = 1.5 # true movement 2.2s; ~0.7s of slack - SLOW_MARGIN_S = 0.5 # the slow painter runs this far ABOVE the lead + MIN_MOVE_S = 1.5 # true movement 2.2s; ~0.7s of slack + SLOW_MARGIN_S = 0.5 # the slow painter runs this far ABOVE the lead H = TestRuntimeRenderSelftestExecutes @@ -1364,8 +1440,10 @@ def test_a_slower_painter_records_a_longer_duration(self, tmp_path): def test_a_paint_slower_than_the_lead_is_recorded_as_slower(self, tmp_path): """The gate boundary — the shape the other two cannot see. - Stage B's question is "did this device render inside the lead", so the - value has to be honest AT that boundary, not merely somewhere near 1.5s. + Anyone who ever asks "did this device render inside the lead" needs the + value honest AT that boundary, not merely somewhere near 1.5s. Stage B + as built does not ask it (it gates on `result` and `sha`), which is why + this is a property of the record, not a Stage B threshold. Measured: a writer that CLAMPS the duration (`min(d, 3)`) passes both tests above and would report every slow device as comfortably inside the lead — the exact "falsely SMALL value sailing through a threshold" @@ -1384,6 +1462,1019 @@ def test_a_paint_slower_than_the_lead_is_recorded_as_slower(self, tmp_path): rec, wall = self._timed_run(tmp_path, painter_rc=0, sleep=slept) assert rec["duration_s"] >= lead, ( f"a paint that took {slept}s was recorded as {rec['duration_s']}s, inside " - f"the {lead}s lead — Stage B would pass a device that cannot make it" + f"the {lead}s lead — the record would call a slow device fast" ) self._assert_honest(rec, slept, wall) + + +class TestStageBMigrationExecutes: + """litclock-dev#871 Stage B — the FLIP, run with the real functions. + + The one part of this feature that changes a device's behaviour, so every + guard arm is EXECUTED here and each asserts what env.sh looks like + AFTERWARDS. A guard that is only grepped for is a guard nobody has seen + refuse anything (litclock-dev#782's whole finding), and this one decides + whether a fielded clock stops using its pre-rendered images. + """ + + SAMPLE_ENV = ( + "export WEATHER_ENABLED=true\n" + "export LITCLOCK_LANGUAGE=en\n" + "export LITCLOCK_RUNTIME_RENDER=false\n" + "# export LOG_LEVEL=WARNING\n" + ) + + @staticmethod + def _pass_record(tmp_path): + """A durable sha-matched pass record, which Stage B gates on.""" + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, capture_output=True, text=True, timeout=30 + ).stdout.strip() + path = tmp_path / "selftest.json" + path.write_text(json.dumps({"result": "passed", "duration_s": 1.2, "sha": head, "at_unix": 1})) + return path + + @staticmethod + def _fn(name): + body = UPDATE_SH.read_text() + assert body.count(f"{name}() {{") == 1, name + start = body.index(f"{name}() {{") + return body[start : body.index("\n}\n", start) + 3] + + def _run(self, tmp_path, *, selftest="pass", env_body=None, images=True, env_file=True, record="match"): + install = tmp_path / "install" + install.mkdir(exist_ok=True) + if env_file: + (install / "env.sh").write_text(self.SAMPLE_ENV if env_body is None else env_body) + if images: + meta = install / "images" / "metadata" + meta.mkdir(parents=True, exist_ok=True) + (meta / "quote_0000_0_credits.png").write_bytes(b"x") + memo = tmp_path / "memo.json" + memo.unlink(missing_ok=True) + # The durable sha-matched pass record Stage B gates on alongside the + # in-run verdict (litclock-dev#871; the Stage A comment above + # RUNTIME_SELFTEST_RECORD_FILE promises exactly this). + rec = tmp_path / "selftest.json" + rec.unlink(missing_ok=True) + head = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, capture_output=True, text=True, timeout=30 + ).stdout.strip() + if record == "match": + rec.write_text(json.dumps({"result": "passed", "duration_s": 1.2, "sha": head, "at_unix": 1})) + elif record == "stale": + rec.write_text(json.dumps({"result": "passed", "duration_s": 1.2, "sha": "0" * 40, "at_unix": 1})) + elif record == "failed": + rec.write_text(json.dumps({"result": "selftest-failed", "sha": head, "at_unix": 1})) + elif record == "corrupt": + # A VALID object followed by garbage. jq prints the matching value + # and THEN exits non-zero, so a `|| echo ""` fallback never runs and + # the substitution keeps the value — measured: the old form captured + # "passed" from exactly this file and the gate accepted it. + rec.write_text(json.dumps({"result": "passed", "sha": head, "at_unix": 1}) + "\ngarbage{\n") + # record == "absent": leave it unwritten + program = ( + "set -u\n" + 'log_info() { echo "[INFO] $1"; }\n' + 'log_warn() { echo "[WARN] $1"; }\n' + 'log_error() { echo "[ERROR] $1"; }\n' + 'atomic_write_file() { printf "%s" "$2" > "$1"; }\n' + 'atomic_remove_file() { rm -f "$1"; }\n' + # The REAL lock helper and the REAL env.sh finalizer, lifted from + # state.sh — the flip's atomicity and its owner/mode preservation + # are the point, so a stub here would test a different program. + f'source "{REPO_ROOT}/scripts/lib/state.sh"\n' + f"INSTALL_DIR={install}\nRUNTIME_VALIDATION_MEMO_FILE={memo}\n" + f"RUNTIME_SELFTEST_RECORD_FILE={rec}\n" + f"RUNTIME_SELFTEST_RESULT={selftest}\n" + f"{self._fn('_atomic_write_env_sh_finalize_from_body')}" + f"{self._fn('_runtime_render_migrate_locked')}" + f"{self._fn('_runtime_validation_memo_write')}" + f"{self._fn('_runtime_render_migrate')}" + "_runtime_render_migrate\n" + 'echo "REACHED_END rc=$?"\n' + ) + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60) + after = (install / "env.sh").read_text() if (install / "env.sh").exists() else None + return r, after, (json.loads(memo.read_text()) if memo.exists() else None) + + # ── the one arm that migrates ──────────────────────────────────────── + + def test_a_passing_selftest_with_images_flips_the_flag(self, tmp_path): + r, after, memo = self._run(tmp_path) + assert "REACHED_END rc=0" in r.stdout, r.stdout + r.stderr + assert "export LITCLOCK_RUNTIME_RENDER=true" in after + assert "=false" not in after + assert "MIGRATED to runtime text rendering" in r.stdout + assert memo is None, "a successful migration is not a negative result" + + def test_it_changes_nothing_else_in_env_sh(self, tmp_path): + """The flip rewrites the whole file, so 'it only touched that line' is a + property worth asserting rather than assuming.""" + _, after, _ = self._run(tmp_path) + assert after == self.SAMPLE_ENV.replace("RUNTIME_RENDER=false", "RUNTIME_RENDER=true") + + def test_it_is_idempotent(self, tmp_path): + """Self-limiting by design: after the flip the pattern no longer matches, + which is what lets Stage B ship with no one-shot marker.""" + self._run(tmp_path) + r, after, memo = self._run( + tmp_path, env_body=self.SAMPLE_ENV.replace("RUNTIME_RENDER=false", "RUNTIME_RENDER=true") + ) + assert "MIGRATED" not in r.stdout + assert "export LITCLOCK_RUNTIME_RENDER=true" in after + assert memo is None, "an already-migrated device is the goal state, not a negative result" + + # ── every arm that must NOT migrate ────────────────────────────────── + + @pytest.mark.parametrize("selftest", ["fail", "deferred", ""]) + def test_a_selftest_that_did_not_pass_leaves_env_sh_alone(self, tmp_path, selftest): + """The marker is not enough. It says freetype reproduces the measurement + dump; only the self-test says this device painted a quote from text.""" + r, after, memo = self._run(tmp_path, selftest=selftest) + assert after == self.SAMPLE_ENV, "env.sh must be byte-identical" + assert "MIGRATED" not in r.stdout + assert memo is None, "the self-test already wrote its own memo; do not contradict it" + + @pytest.mark.parametrize( + "record,why,reason_fragment", + [ + # The REASON, not just the token: an absent record and a mismatched + # one both refuse, so asserting only "migration-skipped" leaves the + # two arms indistinguishable and the absent-record branch could be + # deleted with the tests still green (mutation-checked). + ("absent", "the record could not be written at all", "pass record is missing"), + ("stale", "the record is from a different release", "does not match this release"), + ("failed", "the record says the self-test did not pass", "does not match this release"), + ("corrupt", "the record is a valid object followed by garbage", "not readable JSON"), + ], + ) + def test_the_durable_record_must_match_this_release(self, tmp_path, record, why, reason_fragment): + """Stage A's comment above RUNTIME_SELFTEST_RECORD_FILE says Stage B + gates on that file, sha-matched. It now does, and not redundantly with + the in-run verdict: if the record could not be written — no jq, an + unwritable state dir — then Stage A's own durability claim is false on + this device, and a later reader would find a migrated clock with no + evidence for why. Refusing keeps the device and its record consistent, + and it retries next week. + """ + r, after, memo = self._run(tmp_path, record=record) + assert after == self.SAMPLE_ENV, f"env.sh must be byte-identical when {why}" + assert "MIGRATED" not in r.stdout + assert memo is not None and memo["result"] == "migration-skipped", r.stdout + assert reason_fragment in memo["reason"], ( + f"the memo must say WHICH way the record failed, got: {memo['reason']!r}" + ) + + def test_no_images_leaves_env_sh_alone_and_memos_why(self, tmp_path): + """Migrating a device with no PNG tier would remove the only rung below + the text renderer — the one change that could turn a render fault into a + blank panel.""" + r, after, memo = self._run(tmp_path, images=False) + assert after == self.SAMPLE_ENV, "env.sh must be byte-identical" + assert "MIGRATED" not in r.stdout + assert memo is not None and memo["result"] == "migration-skipped" + assert "fallback" in memo["reason"] + + def test_an_empty_images_metadata_counts_as_absent(self, tmp_path): + """`images/metadata/` present but empty is what a half-finished cleanup + leaves, and the painter's fallback globs that directory.""" + install = tmp_path / "install" + (install / "images" / "metadata").mkdir(parents=True, exist_ok=True) + r, after, memo = self._run(tmp_path, images=False) + assert after == self.SAMPLE_ENV + assert memo is not None and memo["result"] == "migration-skipped" + + def test_a_commented_line_beside_the_active_one_still_migrates(self, tmp_path): + """The realistic env.sh: `env.sh.sample` documents the flag in comments + above the live assignment, and update.sh Phase 3 merges sample lines in. + The count guard must see ONE active assignment there, not two — a + loosened match refuses to migrate any device with the documentation + still attached, which is all of them. + """ + body = ( + "# export LITCLOCK_RUNTIME_RENDER=false <- what this used to default to\n" + "export LITCLOCK_RUNTIME_RENDER=false\n" + ) + r, after, memo = self._run(tmp_path, env_body=body) + assert "MIGRATED to runtime text rendering" in r.stdout, r.stdout + assert after == body.replace("\nexport LITCLOCK_RUNTIME_RENDER=false", "\nexport LITCLOCK_RUNTIME_RENDER=true") + assert after.startswith("# export LITCLOCK_RUNTIME_RENDER=false"), "the comment is untouched" + + def test_a_commented_assignment_is_not_rewritten(self, tmp_path): + """`env.sh.sample` ships commented examples and this runs over a real + device's env.sh; rewriting a comment would edit documentation.""" + body = "# export LITCLOCK_RUNTIME_RENDER=false\nexport WEATHER_ENABLED=true\n" + r, after, memo = self._run(tmp_path, env_body=body) + assert after == body, "a commented line is prose" + assert "MIGRATED" not in r.stdout + assert memo is None, "nothing to migrate is not a negative result" + + def test_two_active_assignments_are_refused(self, tmp_path): + """A device whose env.sh carries two is not one to guess about. + + Also pins the outer function's GENERIC failure branch — the `else` arm + of `_runtime_render_migrate`, reached whenever the locked helper returns + anything but 0 or 75. Asserting only "env.sh unchanged" would not: every + other guard leaves it unchanged too, so that assertion passes when a + DIFFERENT guard refuses, and deleting the branch outright left the whole + suite green (/review 2026-09-23). The memo is the only operator-visible + evidence this path produces, so the memo is what the test asserts. + """ + body = "export LITCLOCK_RUNTIME_RENDER=false\nexport LITCLOCK_RUNTIME_RENDER=false\n" + r, after, memo = self._run(tmp_path, env_body=body) + assert after == body, "env.sh must be byte-identical" + assert "found 2" in r.stdout + assert memo is not None, "the generic failure branch must write a memo" + assert memo["result"] == "migration-skipped", memo + assert "rewrite did not complete" in memo["reason"], memo + assert "the env.sh rewrite failed" in r.stdout, r.stdout + + def test_a_lock_timeout_is_memoed_as_such_and_not_as_a_rewrite_failure(self, tmp_path): + """rc=75 from `with_env_lock` is its own arm, and says something different. + + A real `flock` is held on the sidecar for the whole wait, which is what + the PWA's config.py writing env.sh at the same moment looks like. The + distinction matters to whoever reads the memo: "another writer had it" + is a retry-next-week condition, while "the rewrite did not complete" + points at the rewrite itself. Deleting both arms left 307 tests green + (/review 2026-09-23), so this asserts the rc and the reason, not merely + that env.sh survived. + """ + install = tmp_path / "install" + install.mkdir(exist_ok=True) + (install / "env.sh").write_text(self.SAMPLE_ENV) + meta = install / "images" / "metadata" + meta.mkdir(parents=True, exist_ok=True) + (meta / "q.png").write_bytes(b"x") + memo = tmp_path / "memo.json" + lockfile = install / "env.sh.lock" + held = tmp_path / "held" + + if subprocess.run(["bash", "-c", "command -v flock"], capture_output=True).returncode != 0: + pytest.skip("flock(1) unavailable — with_env_lock takes its no-lock fallback") + + # Hold the sidecar, and signal only once it is actually held: starting + # the holder is not the same as it owning the lock, and racing that + # would make this test pass for the wrong reason. + holder = subprocess.Popen( + ["bash", "-c", f'exec 200>"{lockfile}"; flock 200; : > "{held}"; sleep 30'], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + try: + deadline = time.time() + 10 + while not held.exists() and time.time() < deadline: + time.sleep(0.05) + assert held.exists(), "the lock holder never acquired the sidecar" + program = ( + "set -u\n" + 'log_info() { echo "[INFO] $1"; }\n' + 'log_warn() { echo "[WARN] $1"; }\n' + 'atomic_write_file() { printf "%s" "$2" > "$1"; }\n' + 'atomic_remove_file() { rm -f "$1"; }\n' + f'source "{REPO_ROOT}/scripts/lib/state.sh"\n' + f"INSTALL_DIR={install}\nRUNTIME_VALIDATION_MEMO_FILE={memo}\n" + f"RUNTIME_SELFTEST_RECORD_FILE={self._pass_record(tmp_path)}\n" + "RUNTIME_SELFTEST_RESULT=pass\nLITCLOCK_ENV_LOCK_WAIT=1\n" + f"{self._fn('_atomic_write_env_sh_finalize_from_body')}" + f"{self._fn('_runtime_render_migrate_locked')}" + f"{self._fn('_runtime_validation_memo_write')}" + f"{self._fn('_runtime_render_migrate')}" + "_runtime_render_migrate\n" + 'echo "REACHED_END rc=$?"\n' + ) + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60) + finally: + holder.kill() + holder.wait(timeout=10) + + assert "REACHED_END rc=0" in r.stdout, r.stdout + r.stderr + assert (install / "env.sh").read_text() == self.SAMPLE_ENV, "env.sh must be byte-identical" + assert "MIGRATED" not in r.stdout, r.stdout + loaded = json.loads(memo.read_text()) + assert loaded["result"] == "migration-skipped", loaded + assert str(loaded["rc"]) == "75", loaded + assert "locked by another writer" in loaded["reason"], loaded + # The two arms must stay distinguishable — this is the whole point. + assert "rewrite did not complete" not in loaded["reason"], loaded + + def test_the_under_lock_recheck_catches_a_writer_that_raced_the_outer_guard(self, tmp_path): + """`_runtime_render_migrate_locked` re-reads and re-counts, and that is + not redundant with the outer guard in `_runtime_render_migrate`. + + The outer grep runs BEFORE `with_env_lock`. Another writer — the PWA's + `config.py`, a `reset-setup.sh` run — can change env.sh in that window, + which is the whole reason the lock exists. The re-check is what stops + the migration acting on what the file said a moment ago. + + Unreachable with a static file, so the race is staged: `with_env_lock` + is replaced by a stub that rewrites env.sh into a bare-only form and + THEN calls the target, which is what losing that race looks like. + """ + install = tmp_path / "install" + install.mkdir(exist_ok=True) + (install / "env.sh").write_text(self.SAMPLE_ENV) + meta = install / "images" / "metadata" + meta.mkdir(parents=True, exist_ok=True) + (meta / "q.png").write_bytes(b"x") + memo = tmp_path / "memo.json" + raced = "LITCLOCK_RUNTIME_RENDER=false\n" # bare: never reaches the painter + program = ( + "set -u\n" + 'log_info() { echo "[INFO] $1"; }\n' + 'log_warn() { echo "[WARN] $1"; }\n' + 'atomic_write_file() { printf "%s" "$2" > "$1"; }\n' + 'atomic_remove_file() { rm -f "$1"; }\n' + f'source "{REPO_ROOT}/scripts/lib/state.sh"\n' + # the racing writer, standing in for the lock helper + f'with_env_lock() {{ printf %s {shlex.quote(raced)} > "{install}/env.sh"; "$@"; }}\n' + f"INSTALL_DIR={install}\nRUNTIME_VALIDATION_MEMO_FILE={memo}\n" + f"RUNTIME_SELFTEST_RECORD_FILE={self._pass_record(tmp_path)}\n" + "RUNTIME_SELFTEST_RESULT=pass\n" + f"{self._fn('_atomic_write_env_sh_finalize_from_body')}" + f"{self._fn('_runtime_render_migrate_locked')}" + f"{self._fn('_runtime_validation_memo_write')}" + f"{self._fn('_runtime_render_migrate')}" + "_runtime_render_migrate\n" + ) + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60) + after = (install / "env.sh").read_text() + assert after == raced, f"the raced file must be left as the other writer wrote it:\n{after}" + assert "MIGRATED" not in r.stdout, r.stdout + assert "0 exported =false" in r.stdout, r.stdout + + def test_a_missing_env_sh_is_a_silent_no_op(self, tmp_path): + r, after, memo = self._run(tmp_path, env_file=False) + assert after is None and "MIGRATED" not in r.stdout + assert memo is None + + def test_an_already_true_device_writes_no_memo(self, tmp_path): + body = "export LITCLOCK_RUNTIME_RENDER=true\n" + r, after, memo = self._run(tmp_path, env_body=body) + assert after == body and memo is None + + # ── the call site ──────────────────────────────────────────────────── + + def test_the_flip_is_called_after_the_selftest_inside_the_keep_arm(self): + """Position is the safety property, as it is for the validator above: a + smoke failure git-resets the tree, so a flip outside the KEEP arm could + migrate a device onto code that was just reverted.""" + body = UPDATE_SH.read_text() + keep = body.index('if [[ "$smoke_rc" -eq 0 ]]; then\n log_info "Smoke test passed"') + else_at = body.index('else\n log_error "Smoke test failed', keep) + selftest_call = body.index(" _runtime_render_selftest\n", keep) + migrate_call = body.index(" _runtime_render_migrate\n", keep) + assert keep < selftest_call < migrate_call < else_at + assert "ROLLBACK_MODE" in body[keep:selftest_call].rsplit("if [[", 1)[-1] + + # ── the review's findings, each with an executed test ──────────────── + + def test_another_assignment_with_a_different_value_is_refused(self, tmp_path): + """`=false` followed by `=0` used to migrate "successfully" while the + effective value stayed OFF — the last assignment wins when the file is + sourced, so the flip was cosmetic and the memo was cleared as if it had + worked. The count now looks at every ACTIVE assignment, not just the + `=false` ones.""" + body = "export LITCLOCK_RUNTIME_RENDER=false\nexport LITCLOCK_RUNTIME_RENDER=0\n" + r, after, _ = self._run(tmp_path, env_body=body) + assert after == body, "env.sh must be byte-identical" + assert "MIGRATED" not in r.stdout + assert "found 2 active" in r.stdout, r.stdout + + def test_a_true_assignment_beside_the_false_one_is_refused(self, tmp_path): + body = "export LITCLOCK_RUNTIME_RENDER=false\nexport LITCLOCK_RUNTIME_RENDER=true\n" + r, after, _ = self._run(tmp_path, env_body=body) + assert after == body + assert "MIGRATED" not in r.stdout + + def test_a_failed_transformation_is_not_committed(self, tmp_path): + """A `sed` killed mid-stream emits a PARTIAL body. It differs from the + original, so the "did it change anything" check accepted it and the + finalizer renamed a truncated env.sh into place, reported as a + successful migration. An atomic rename protects against a torn write, + not against committing a complete-looking truncated body. + + `sed` is shadowed by a function that prints one line and exits 137, the + shape a SIGKILL leaves. + """ + install = tmp_path / "install" + install.mkdir(exist_ok=True) + (install / "env.sh").write_text(self.SAMPLE_ENV) + meta = install / "images" / "metadata" + meta.mkdir(parents=True, exist_ok=True) + (meta / "q.png").write_bytes(b"x") + memo = tmp_path / "memo.json" + program = ( + "set -u\n" + 'log_info() { echo "[INFO] $1"; }\n' + 'log_warn() { echo "[WARN] $1"; }\n' + 'atomic_write_file() { printf "%s" "$2" > "$1"; }\n' + 'atomic_remove_file() { rm -f "$1"; }\n' + f'source "{REPO_ROOT}/scripts/lib/state.sh"\n' + f"INSTALL_DIR={install}\nRUNTIME_VALIDATION_MEMO_FILE={memo}\n" + f"RUNTIME_SELFTEST_RECORD_FILE={self._pass_record(tmp_path)}\n" + "RUNTIME_SELFTEST_RESULT=pass\n" + # the mid-stream kill + 'sed() { echo "export WEATHER_ENABLED=true"; return 137; }\n' + f"{self._fn('_atomic_write_env_sh_finalize_from_body')}" + f"{self._fn('_runtime_render_migrate_locked')}" + f"{self._fn('_runtime_validation_memo_write')}" + f"{self._fn('_runtime_render_migrate')}" + "_runtime_render_migrate\n" + 'echo "REACHED_END rc=$?"\n' + ) + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60) + after = (install / "env.sh").read_text() + assert after == self.SAMPLE_ENV, f"a partial body was committed:\n{after}" + assert "MIGRATED" not in r.stdout + assert "sed exited 137" in r.stdout, r.stdout + + def test_a_transformation_that_loses_lines_is_not_committed(self, tmp_path): + """The line-count belt, for a `sed` that truncates and still exits 0.""" + install = tmp_path / "install" + install.mkdir(exist_ok=True) + (install / "env.sh").write_text(self.SAMPLE_ENV) + meta = install / "images" / "metadata" + meta.mkdir(parents=True, exist_ok=True) + (meta / "q.png").write_bytes(b"x") + program = ( + "set -u\n" + 'log_info() { echo "[INFO] $1"; }\n' + 'log_warn() { echo "[WARN] $1"; }\n' + 'atomic_write_file() { printf "%s" "$2" > "$1"; }\n' + 'atomic_remove_file() { rm -f "$1"; }\n' + f'source "{REPO_ROOT}/scripts/lib/state.sh"\n' + f"INSTALL_DIR={install}\nRUNTIME_VALIDATION_MEMO_FILE={tmp_path / 'memo.json'}\n" + f"RUNTIME_SELFTEST_RECORD_FILE={self._pass_record(tmp_path)}\n" + "RUNTIME_SELFTEST_RESULT=pass\n" + 'sed() { echo "export LITCLOCK_RUNTIME_RENDER=true"; return 0; }\n' + f"{self._fn('_atomic_write_env_sh_finalize_from_body')}" + f"{self._fn('_runtime_render_migrate_locked')}" + f"{self._fn('_runtime_validation_memo_write')}" + f"{self._fn('_runtime_render_migrate')}" + "_runtime_render_migrate\n" + ) + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60) + assert (install / "env.sh").read_text() == self.SAMPLE_ENV + assert "changed the line count" in r.stdout, r.stdout + + def test_the_sample_stays_false_because_phase_3_backfills_it(self): + """THE finding that held Stage B back a release. + + `env.sh.sample` is not the fresh-flash default — it is what Phase 3 + copies VERBATIM onto any existing device missing the key. A `true` here + switches on an old clock with none of the five guards run, before the + smoke gate, with no revert path. litclock-dev#783 records devices born + missing up to ten of these knobs. + """ + sample = (REPO_ROOT / "env.sh.sample").read_text() + assert re.search(r"^export LITCLOCK_RUNTIME_RENDER=false$", sample, re.M), ( + "the sample is the BACKFILL source for existing devices and must stay false" + ) + + def test_the_seeder_defaults_to_false_and_only_a_fresh_flash_asks_for_true(self, tmp_path): + """`env_sh_defaults()` is NOT a fresh-flash-only seeder, which is what + the first fix for the above assumed. + + `reset-setup.sh` and `prepare-for-cloning.sh` both call it on an + EXISTING device, and neither removes `.runtime-render-validated` — they + clear the self-test record, not the marker. So a `true` default handed a + reset clock runtime render with none of Stage B's guards run, including + a clock whose self-test had FAILED, which is exactly the device the + guards exist to hold back (review, round 4). + + Executed against the real helper rather than grepped, because the value + is a parameter now and a source-text check would pin the wrong thing. + """ + state_sh = REPO_ROOT / "scripts" / "lib" / "state.sh" + + def seed(*args): + call = "env_sh_defaults " + " ".join(f'"{a}"' for a in args) if args else "env_sh_defaults" + r = subprocess.run( + ["bash", "-c", f'. "{state_sh}"\n{call}'], + capture_output=True, + text=True, + timeout=30, + ) + assert r.returncode == 0, r.stderr + return r.stdout + + assert "export LITCLOCK_RUNTIME_RENDER=false" in seed(), "the bare default must be conservative" + assert "export LITCLOCK_RUNTIME_RENDER=false" in seed("en"), "a language-only call is a reset caller" + assert "export LITCLOCK_RUNTIME_RENDER=true" in seed("", "true"), "a fresh flash asks for text" + # A caller that passes nonsense gets the safe value, not the nonsense. + assert "export LITCLOCK_RUNTIME_RENDER=false" in seed("", "yes") + + def test_the_reset_and_cloning_callers_take_the_conservative_default(self): + """Neither may start asking for `true`: both run on an existing device + whose self-test may have failed, and update.sh migrates it properly on + the next tick if it can render.""" + for name in ("reset-setup.sh", "prepare-for-cloning.sh"): + body = (REPO_ROOT / "scripts" / name).read_text() + for call in re.findall(r"env_sh_defaults[^\n)]*", body): + assert "true" not in call, f"{name} must not seed runtime render on an existing device: {call!r}" + + def test_first_boot_asks_for_true_in_both_of_its_seeders(self): + """first-boot has two: the helper call and the state.sh-missing heredoc. + litclock-dev#783 found the second one missed by a guard that only saw the first.""" + body = (REPO_ROOT / "scripts" / "first-boot.sh").read_text() + assert re.search(r'env_sh_defaults\s+""\s+true', body), "the helper call must ask for text render" + assert re.search(r"^export LITCLOCK_RUNTIME_RENDER=true$", body, re.M), ( + "the fallback heredoc seeds a fresh device too" + ) + + def test_a_bare_assignment_beside_the_export_is_refused(self, tmp_path): + """`export ...=false` plus a later bare `...=0` used to migrate + "successfully" while the effective value stayed off — re-assigning does + not clear the export attribute, so the bare line wins. A bare assignment + is not an invented shape: the PWA's own writer treats `export` as + optional (`_KV_PATTERN`, src/config.py).""" + body = "export LITCLOCK_RUNTIME_RENDER=false\nLITCLOCK_RUNTIME_RENDER=0\n" + r, after, _ = self._run(tmp_path, env_body=body) + assert after == body, "env.sh must be byte-identical" + assert "found 2 active" in r.stdout, r.stdout + + def test_a_lone_bare_assignment_is_refused_because_it_never_reached_the_painter(self, tmp_path): + """Migrating a bare assignment would be a no-op reported as success. + + `runtheclock.sh` does `source ./env.sh` with NO `set -a`, so a bare + assignment never enters the painter's environment — measured: a child + process sees the variable UNSET. A device with a lone bare `=false` was + therefore not honouring the flag either way, and rewriting it to a bare + `=true` changes the file and nothing else while clearing the memo and + logging a migration. Refused with a reason instead. + + Bare assignments are still COUNTED, because a bare one beside an + exported one does change the exported value — that case is covered by + `test_a_bare_assignment_beside_the_export_is_refused`. + """ + body = "LITCLOCK_RUNTIME_RENDER=false\nexport WEATHER_ENABLED=true\n" + r, after, memo = self._run(tmp_path, env_body=body) + assert "MIGRATED" not in r.stdout, r.stdout + assert after == body, "env.sh must be byte-identical" + # SILENT, and that is what pins the OUTER guard rather than the inner + # count: a device with no exported assignment was never eligible, so it + # is not a "could have migrated and did not" case. Loosening the outer + # guard to accept bare forms pushes this into the locked helper, which + # refuses with a warning and a memo — same file, different diagnosis. + assert memo is None, f"a device that was never eligible must not be memoed: {memo}" + assert "Stage B:" not in r.stdout, r.stdout + + def test_a_bare_assignment_really_does_not_reach_a_child(self, tmp_path): + """The measurement the refusal above rests on, executed rather than + asserted in prose — and the control beside it, since `export` must.""" + env = tmp_path / "env.sh" + results = {} + for label, line in (("bare", "FOO_X=1\n"), ("export", "export FOO_X=1\n")): + env.write_text(line) + r = subprocess.run( + ["bash", "-c", f". \"{env}\"; python3 -c \"import os;print(os.environ.get('FOO_X',''))\""], + capture_output=True, + text=True, + timeout=30, + ) + results[label] = r.stdout.strip() + assert results["bare"] == "", results + assert results["export"] == "1", results + + def test_phase_3_sees_an_indented_existing_assignment(self): + """A tab-indented assignment went undetected, so Phase 3 appended the + sample's line beside it and the sample's value silently became the + effective one. Not harmless before Stage B either: the sample's example + WEATHER_LATITUDE/LONGITUDE already differed from what a device is seeded + with. Stage B is what surfaced it, not the first divergence.""" + body = UPDATE_SH.read_text() + assert 'grep -q "^[[:space:]#]*export[[:space:]]\\+${varname}="' in body, ( + "Phase 3's existing-key detection must allow leading whitespace" + ) + + def test_the_finalizer_sets_mode_before_ownership(self): + """Ownership first installs an unreadable env.sh: the mktemp file is + pi-owned 0600, `sudo chown root:root` succeeds, the unprivileged `chmod + 644` then fails on a file pi no longer owns, and the rename replaces a + world-readable config with a root-owned 0600 one. Both failures are + swallowed, so the caller reports success.""" + state = (REPO_ROOT / "scripts" / "lib" / "state.sh").read_text() + fn = state[state.index("_atomic_write_env_sh_finalize() {") :] + fn = fn[: fn.index("\n}\n")] + assert fn.index('chmod "$mode"') < fn.index('chown "$owner"'), ( + "mode must be set while pi still owns the staging file" + ) + + +class TestMemoTokensRoundTripToThePWA: + """The shell writes the memo; `routes/status.py` decides whether to show it. + + Those two vocabularies live in different languages and nothing connected + them, so Stage B shipped `migration-skipped` that `RUNTIME_VALIDATION_RESULTS` + did not accept: every refusal was parsed, rejected and surfaced as `null` — + byte-identical to what a healthy device sends, on the one feature whose + whole safety argument is that a refusal would be visible (/review + 2026-09-23, adversarial). Grepping either side alone cannot catch that; the + test has to compare them. + """ + + # Every position a memo token can enter from. Two of these are forwarding + # wrappers (`_defer_runtime_validation`, `_validation_fits_remaining_budget`) + # that take the token as $2 and hand it to the writer as `"$token"`, so + # scanning only the writer's call sites would miss `selftest-deferred` + # entirely and make this test weaker than it looks. + TOKEN_SINKS = ( + r"_runtime_validation_memo_write\s+(\S+)", + r"_defer_runtime_validation\s+(?:\"[^\"]*\"|\S+)\s+(\S+)", + r"_validation_fits_remaining_budget\s+(?:\"[^\"]*\"|\S+)\s+(\S+)", + ) + # The wrappers' own `${2:-}` fallbacks are live tokens too: a caller + # that omits $2 writes the default. + DEFAULT_TOKEN_RE = r"local\s+(?:\w+=\S+\s+)?token=\"\$\{2:-([a-z-]+)\}\"" + + @classmethod + def _tokens_written_by_the_shell(cls): + """Every literal memo token reachable in update.sh. + + A forwarded `"$token"` is expected at the two wrapper sites and is + resolved through their callers and defaults instead. Any OTHER + non-literal is refused rather than skipped quietly, because a token the + extraction cannot see is a token this test silently stops guarding. + """ + body = UPDATE_SH.read_text() + tokens = set() + for pattern in cls.TOKEN_SINKS: + for raw in re.findall(pattern, body): + if raw in ('"$token"', "$token"): + continue # forwarded; resolved via callers + defaults below + assert not raw.startswith(("$", '"', "'")), ( + f"non-literal memo token {raw!r} — the round-trip check cannot see it; " + "keep the token a bare literal at the call site, or teach TOKEN_SINKS " + "how to resolve the new wrapper" + ) + tokens.add(raw) + defaults = re.findall(cls.DEFAULT_TOKEN_RE, body) + assert defaults, "no `${2:-}` wrapper default found — did update.sh refactor?" + tokens.update(defaults) + return tokens + + def test_every_token_the_shell_writes_is_renderable_by_the_pwa(self): + from control_server.routes.status import RUNTIME_VALIDATION_RESULTS + + written = self._tokens_written_by_the_shell() + assert written, "found no memo-write call sites — did update.sh refactor?" + missing = sorted(written - set(RUNTIME_VALIDATION_RESULTS)) + assert not missing, ( + f"update.sh writes {missing} but routes/status.py's " + f"RUNTIME_VALIDATION_RESULTS does not accept them, so those memos reach " + f"the PWA as null — indistinguishable from a healthy device. " + f"Add them to RUNTIME_VALIDATION_RESULTS." + ) + + def test_the_stage_b_token_specifically_round_trips(self): + """Named on its own so a regression points at the right release.""" + from control_server.routes.status import RUNTIME_VALIDATION_RESULTS + + assert "migration-skipped" in self._tokens_written_by_the_shell() + assert "migration-skipped" in RUNTIME_VALIDATION_RESULTS + + def test_a_refusal_memo_actually_renders(self, tmp_path): + """End to end through the real reader, not just the allowlist. + + Membership is necessary and not sufficient: the reader also gates on + `at_unix`, so a token in the set whose memo the reader still drops would + pass the check above and fail the owner. + """ + from control_server.routes.status import _resolve_runtime_validation_memo + + memo = tmp_path / "runtime-render-validation.json" + memo.write_text( + json.dumps( + { + "result": "migration-skipped", + "rc": None, + "reason": "the PNG fallback tier (images/metadata) is absent", + "sha": "0" * 40, + "at_unix": 1790000000, + } + ) + ) + resolved = _resolve_runtime_validation_memo(memo_file=memo) + assert resolved is not None, "a Stage B refusal must reach /api/status, not vanish" + assert resolved.get("result") == "migration-skipped", resolved + + +class TestTheSmokeGateRendersTheTierTheDeviceIsOn: + """The smoke gate is the only thing that can revert a release. + + Before litclock-dev#871 Stage B it ran the painter with no `env.sh`, so it + always exercised the PNG tier. That was harmless while the whole fleet was + on PNGs and became a hole the moment Stage B migrated them: a regression in + the text renderer would render a PNG here, pass, and never be reverted — + and `src/literary_clock.py` is not one of the six proof inputs, so the + marker-revoke block does not cover it either (/review 2026-09-23). + """ + + SMOKE_START = ' [[ -f "$INSTALL_DIR/env.sh" ]] && { source "$INSTALL_DIR/env.sh" 2>/dev/null || true; }' + SMOKE_END = " ) | sed 's/^/[smoke] /'" + + def test_the_gate_is_the_one_that_reverts(self): + """Anchor: the block we lift is the one whose status becomes smoke_rc.""" + body = UPDATE_SH.read_text() + end = body.index(self.SMOKE_END) + after = body[end : end + 200] + assert 'smoke_rc="${PIPESTATUS[0]}"' in after, ( + "the sourced block must be the one feeding smoke_rc, or this test guards a different gate" + ) + + @classmethod + def _smoke_sequence(cls): + """The scratch-dir setup, the subshell, smoke_rc, and the cleanup. + + From the `mktemp` line through the `rm` line, so the isolation added in + the port /review (weather off, a throwaway frame directory) is executed + as written, not re-typed here. + """ + body = UPDATE_SH.read_text() + start = body.index(' smoke_render_dir="$(mktemp -d 2>/dev/null)"') + rm = body.index('rm -rf -- "$smoke_render_dir"', start) + end = body.index("\n", rm) + seq = body[start:end] + assert cls.SMOKE_START in seq and 'smoke_rc="${PIPESTATUS[0]}"' in seq, seq + return seq + + def _run_block(self, tmp_path, env_body, mktemp_fails=False, painter_rc=0): + install = tmp_path / "install" + install.mkdir(exist_ok=True) + (install / "env.sh").write_text(env_body) + # A stub painter that reports what it was actually handed. + stub = tmp_path / "python3" + stub.write_text( + "#!/bin/bash\n" + "echo \"SAW=${LITCLOCK_RUNTIME_RENDER:-}\"\n" + "echo \"WEATHER=${WEATHER_ENABLED:-}\"\n" + "echo \"DIR=${LITCLOCK_RUNTIME_RENDER_DIR:-}\"\n" + "[[ -n \"${LITCLOCK_RUNTIME_RENDER_DIR:-}\" && -d \"$LITCLOCK_RUNTIME_RENDER_DIR\" ]]" + " && echo DIR_EXISTS_DURING\n" + f"exit {painter_rc}\n" + ) + stub.chmod(0o755) + program = ( + f"INSTALL_DIR={install}\nPYTHON={stub}\n" + 'log_warn() { echo "[WARN] $1"; }\n' + + ("mktemp() { return 1; }\n" if mktemp_fails else "") + # Prove the subshell isolation at the same time. + + "LITCLOCK_RUNTIME_RENDER=outer\n" + f"{self._smoke_sequence()}\n" + 'echo "rc=$smoke_rc"\n' + 'echo "PARENT_AFTER=$LITCLOCK_RUNTIME_RENDER"\n' + 'echo "SCRATCH=${smoke_render_dir:-}"\n' + '[[ -n "${smoke_render_dir:-}" && -e "$smoke_render_dir" ]] && echo SCRATCH_LEFT_BEHIND\n' + ) + # A scrubbed environment: inherited from the runner, WEATHER_ENABLED or + # LITCLOCK_RUNTIME_RENDER_DIR would satisfy (or break) the assertions + # below for reasons that have nothing to do with the gate (port /review). + env = { + k: v + for k, v in os.environ.items() + if not k.startswith(("LITCLOCK_", "WEATHER_")) and k != "ROLLBACK_MODE" + } + return subprocess.run( + ["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=60, env=env + ) + + def test_a_migrated_device_smoke_tests_the_text_tier(self, tmp_path): + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\n") + assert "[smoke] SAW=true" in r.stdout, r.stdout + r.stderr + assert "rc=0" in r.stdout, r.stdout + + def test_an_unmigrated_device_still_smoke_tests_the_png_tier(self, tmp_path): + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=false\n") + assert "[smoke] SAW=false" in r.stdout, r.stdout + r.stderr + + def test_env_sh_does_not_leak_into_the_updater(self, tmp_path): + """A subshell, not a bare source: update.sh must not inherit env.sh.""" + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\n") + assert "PARENT_AFTER=outer" in r.stdout, ( + "env.sh leaked into the updater's own shell:\n" + r.stdout + ) + + def test_a_broken_env_sh_does_not_fail_the_gate(self, tmp_path): + """The asymmetry that matters: a false RED reverts a healthy release. + + Since litclock-dev#865 the smoke-revert path blocks the release it fled, + so a gate that a corrupt env.sh could fail would hold every release off + every such device until the next one ships, and fail that one too. + """ + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\nthis is ( not valid shell\n") + assert "rc=0" in r.stdout, ( + "a syntactically broken env.sh must not make the smoke gate revert the release:\n" + + r.stdout + r.stderr + ) + + + @pytest.mark.parametrize("painter_rc", [0, 3, 124]) + def test_the_painters_verdict_reaches_smoke_rc_and_the_scratch_dir_goes(self, tmp_path, painter_rc): + """A failing painter must reach smoke_rc THROUGH the new lines around the + pipeline — a mutant that overwrote PIPESTATUS survived every test while + the stub always exited 0 — and the scratch directory must go either way.""" + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\n", painter_rc=painter_rc) + assert f"rc={painter_rc}\n" in r.stdout, r.stdout + r.stderr + assert "SCRATCH_LEFT_BEHIND" not in r.stdout, r.stdout + + def test_weather_is_off_even_when_env_sh_never_mentions_it(self, tmp_path): + """The case that needs the override itself: nothing in env.sh to undo.""" + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\n") + assert "[smoke] WEATHER=false" in r.stdout, r.stdout + r.stderr + + def test_a_readonly_env_sh_cannot_defeat_the_overrides(self, tmp_path): + """env.sh is sourced first, so a `readonly` there made `export` fail + and the painter got the owner's values (reproduced in review).""" + r = self._run_block( + tmp_path, + "export LITCLOCK_RUNTIME_RENDER=true\n" + "export WEATHER_ENABLED=true\nreadonly WEATHER_ENABLED\n" + "export LITCLOCK_RUNTIME_RENDER_DIR=/run/litclock\nreadonly LITCLOCK_RUNTIME_RENDER_DIR\n", + ) + assert "[smoke] WEATHER=false" in r.stdout, r.stdout + r.stderr + assert "[smoke] DIR=/run/litclock" not in r.stdout, r.stdout + assert "[smoke] DIR_EXISTS_DURING" in r.stdout, r.stdout + assert "rc=0" in r.stdout, r.stdout + + def test_weather_is_forced_off_whatever_env_sh_says(self, tmp_path): + """litclock-dev#871 port /review: sourcing env.sh gave the gate a + location, so it started fetching live weather, and a slow provider could + push the dry-run past its 60s bound and revert a healthy release.""" + r = self._run_block( + tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\nexport WEATHER_ENABLED=true\n" + ) + assert "[smoke] WEATHER=false" in r.stdout, r.stdout + r.stderr + assert "rc=0" in r.stdout, r.stdout + + def test_the_frame_goes_to_a_throwaway_directory_that_is_removed(self, tmp_path): + """On the text tier the dry-run writes current-quote.png, the live frame + the support bundle reports. The gate's frame must not land there.""" + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\n") + dirs = [ln.split("DIR=", 1)[1] for ln in r.stdout.splitlines() if "[smoke] DIR=" in ln] + assert dirs and dirs[0] not in ("", "/run/litclock"), r.stdout + r.stderr + assert "[smoke] DIR_EXISTS_DURING" in r.stdout, "the scratch directory must exist while the painter runs" + assert f"SCRATCH={dirs[0]}" in r.stdout, r.stdout + assert "SCRATCH_LEFT_BEHIND" not in r.stdout, "the scratch directory must be removed after the gate" + + def test_no_scratch_directory_still_runs_the_gate(self, tmp_path): + """The gate is the only revert path, so it must not be skipped for want + of a scratch directory: it runs into the live path and says so.""" + r = self._run_block(tmp_path, "export LITCLOCK_RUNTIME_RENDER=true\n", mktemp_fails=True) + assert "[smoke] SAW=true" in r.stdout, r.stdout + r.stderr + assert "[smoke] DIR=" in r.stdout, r.stdout + assert "[smoke] WEATHER=false" in r.stdout, r.stdout + assert "could not create a scratch directory" in r.stdout, r.stdout + assert "rc=0" in r.stdout, r.stdout + + +class TestTheRewriteHasNoUncheckedProducer: + """`$?` on a pipeline is its LAST element, so only the consumer was checked. + + The fix for a killed `sed` (take `$?`, not `PIPESTATUS[1]`) left the + `printf` feeding it unguarded. A producer dying after a partial final line + is invisible three times over: sed exits 0 on the short input, sed supplies + the missing newline so the line-count belt matches, and the body differs + from the original so the "changed nothing" check passes. Reproduced + 2026-09-23 committing a truncated `OPENWEATHERMAP_APIKEY`. + """ + + @staticmethod + def _rewrite_line(): + for line in UPDATE_SH.read_text().splitlines(): + s = line.strip() + if s.startswith("new=$(") and "LITCLOCK_RUNTIME_RENDER=false" in s: + return s + raise AssertionError("the env.sh rewrite line is gone — did update.sh refactor?") + + def test_the_transformation_is_not_fed_by_a_pipeline(self): + line = self._rewrite_line() + assert "|" not in line, ( + "the rewrite is fed by a pipeline again; $? cannot see the producer, so a " + "producer killed mid-stream commits a truncated env.sh and reports success:\n" + f" {line}" + ) + assert "<<<" in line, f"expected a herestring feed, got:\n {line}" + + def test_a_killed_producer_is_what_this_prevents(self): + """The failure the shape change removes, demonstrated on the OLD shape. + + Kept executable so the reason survives: without it, a future reader + sees only a style preference for `<<<` and may pipe it back. + """ + old_shape = r""" +body=$'export A=1\nexport LITCLOCK_RUNTIME_RENDER=false\nexport SECRET=original-secret' +flip='s/^([[:space:]]*export[[:space:]]+)LITCLOCK_RUNTIME_RENDER=false[[:space:]]*$/\1LITCLOCK_RUNTIME_RENDER=true/' +new=$( { printf '%s' $'export A=1\nexport LITCLOCK_RUNTIME_RENDER=false\nexport SECRET=original-'; kill -9 $BASHPID; } \ + | sed -E "$flip") +rc=$? +b=$(printf '%s\n' "$body" | wc -l); a=$(printf '%s\n' "$new" | wc -l) +echo "rc=$rc lines=$b/$a changed=$([[ "$new" == "$body" ]] && echo no || echo yes)" +printf '%s\n' "$new" | tail -1 +""" + r = subprocess.run(["bash", "-c", old_shape], capture_output=True, text=True, timeout=60) + assert "rc=0" in r.stdout, r.stdout + r.stderr + assert "lines=3/3" in r.stdout, f"the line-count belt caught it after all:\n{r.stdout}" + assert "changed=yes" in r.stdout, r.stdout + assert "SECRET=original-\n" in r.stdout, ( + f"expected the truncated secret the old shape committed:\n{r.stdout}" + ) + + def test_the_herestring_feeds_sed_the_same_bytes_printf_did(self): + """`<<<` must append exactly the newline `printf '%s\\n'` did. + + `before_lines` is still computed with `printf '%s\\n' "$body"`, so a + feed that differed by a trailing newline would make the belt compare + mismatched shapes and refuse every migration. + """ + for body in ("a\nb\nc", "a\nb\nc\n", "single", ""): + prog = ( + f"body={shlex.quote(body)}\n" + 'p=$(printf "%s\\n" "$body" | cat | od -c | md5sum)\n' + 'h=$(cat <<<"$body" | od -c | md5sum)\n' + '[[ "$p" == "$h" ]] && echo SAME || echo DIFF\n' + 'printf "%s\\n" "$body" | od -c\n' + 'cat <<<"$body" | od -c\n' + ) + r = subprocess.run(["bash", "-c", prog], capture_output=True, text=True, timeout=30) + assert "SAME" in r.stdout, f"feed differs for {body!r}:\n{r.stdout}" + + +class TestAbandonedEnvStagingFilesAreNotShipped: + """`mktemp "${dest}.XXXXXX"` beside env.sh is a full unredacted copy. + + Every in-script failure arm removes it. A SIGKILL or power loss inside the + printf -> chmod -> chown -> mv window does not, and `with_env_lock` runs the + writer in a subshell where bash has reset the caller's traps to default, so + the signal handler does not cover it either. It is untracked (so + `git reset --hard` never removes it), was unignored (so it could be + committed), and neither the clone-prep credential gate nor the gift wipe + looked at it — so the previous owner's API key and home coordinates shipped + on every clone (/review 2026-09-23, adversarial). + """ + + def test_the_staging_shape_is_gitignored(self): + r = subprocess.run( + ["git", "check-ignore", "--no-index", "env.sh.A1b2C3"], + cwd=REPO_ROOT, capture_output=True, text=True, timeout=30, + ) + assert r.returncode == 0, "env.sh. must be gitignored; it holds the owner's credentials" + + def test_the_tracked_sample_is_not_swept_up_by_that_pattern(self): + """`sample` is also six characters. + + Harmless while env.sh.sample stays tracked, because ignore rules do not + apply to tracked files — and a silent trap the moment it is removed and + re-added, which is exactly the kind of latency this repo keeps finding. + """ + r = subprocess.run( + ["git", "check-ignore", "--no-index", "-v", "env.sh.sample"], + cwd=REPO_ROOT, capture_output=True, text=True, timeout=30, + ) + assert "!env.sh.sample" in r.stdout, ( + "env.sh.sample must be explicitly re-included, not left to tracked-status luck:\n" + r.stdout + ) + + def test_clone_prep_refuses_rather_than_certifying_the_card(self): + body = (REPO_ROOT / "scripts" / "prepare-for-cloning.sh").read_text() + assert "env.sh.??????" in body, "the clone-prep credential gate does not look for staging files" + idx = body.index("env.sh.??????") + after = body[idx : idx + 1200] + assert "_abort_env_credentials" in after, ( + "clone prep must ABORT on a staging file, not delete it quietly: its job is to " + "certify the card, and a staging file means a writer died mid-write" + ) + assert "env.sh.sample" in after, "the six-character tracked sample must be excluded by name" + + def test_gift_mode_sweeps_them_before_writing_defaults(self): + body = (REPO_ROOT / "scripts" / "reset-setup.sh").read_text() + assert "env.sh.??????" in body, "the gift wipe does not sweep staging files" + idx = body.index("env.sh.??????") + wipe = body.index('atomic_write_env_sh "$INSTALL_DIR/env.sh" "$DEFAULTS"') + assert idx < wipe, "the sweep must run BEFORE the wipe, or the writer can stage a fresh one after it" + + def test_the_gate_actually_fires_on_a_planted_staging_file(self, tmp_path): + """Executed, because a grep for the glob proves only that it was typed.""" + install = tmp_path / "litclock" + install.mkdir() + (install / "env.sh").write_text("export WEATHER_ENABLED=true\n") + (install / "env.sh.A1b2C3").write_text("export OPENWEATHERMAP_APIKEY=leaked-key\n") + (install / "env.sh.sample").write_text("export WEATHER_ENABLED=true\n") + prog = f''' +INSTALL_DIR={install} +shopt -s nullglob +_ENV_STAGING=("$INSTALL_DIR"/env.sh.??????) +shopt -u nullglob +_ENV_STAGING_REAL=() +for _s in "${{_ENV_STAGING[@]}}"; do + [[ "$(basename "$_s")" == "env.sh.sample" ]] && continue + _ENV_STAGING_REAL+=("$_s") +done +echo "COUNT=${{#_ENV_STAGING_REAL[@]}}" +printf 'FOUND=%s\\n' "${{_ENV_STAGING_REAL[@]}}" +''' + r = subprocess.run(["bash", "-c", prog], capture_output=True, text=True, timeout=30) + assert "COUNT=1" in r.stdout, f"expected exactly the planted staging file:\n{r.stdout}" + assert "env.sh.A1b2C3" in r.stdout, r.stdout + assert "env.sh.sample" not in r.stdout, f"the tracked sample must not be flagged:\n{r.stdout}" diff --git a/tests/test_update_sh.py b/tests/test_update_sh.py index 3aae082..9fce6a5 100644 --- a/tests/test_update_sh.py +++ b/tests/test_update_sh.py @@ -286,7 +286,59 @@ def test_env_merge_preserves_user_values(self, update_sh_content): # litclock-dev#782 — EXECUTED lines (3 executed lines could vanish green). executed = _executed_lines(update_sh_content) assert "env.sh.sample" in executed - assert 'grep -q "^[# ]*export[[:space:]]\\+${varname}=" "$INSTALL_DIR/env.sh"' in executed + assert 'grep -q "^[[:space:]#]*export[[:space:]]\\+${varname}=" "$INSTALL_DIR/env.sh"' in executed + + def test_env_merge_detects_an_indented_existing_assignment(self, tmp_path): + """The detection must allow LEADING WHITESPACE, executed rather than + matched as source text. + + It was `[# ]*`, which sees a space-indented or commented assignment but + not a TAB-indented one. Phase 3 then appended the sample's line beside + the existing one, and since the last assignment wins when the file is + sourced, the SAMPLE's value silently became the effective value. That + was NOT inert before now — any key whose device value differs from the + sample's could be silently reverted this way, and the sample's example + coordinates already differ from what a device is seeded with. + litclock-dev#871 Stage B is what made it worth finding (review). + """ + env = tmp_path / "env.sh" + sample = tmp_path / "env.sh.sample" + sample.write_text("export FOO=sample-value\n") + body = UPDATE_SH.read_text() + fn = body[body.index("_phase3_merge_sample() {") :] + fn = fn[: fn.index("\n}\n") + 3] + results = {} + for label, existing in ( + ("tab", "\texport FOO=mine\n"), + ("space", " export FOO=mine\n"), + ("commented", "# export FOO=mine\n"), + ("plain", "export FOO=mine\n"), + ): + env.write_text(existing) + program = f"set -u\nINSTALL_DIR={tmp_path}\n_PHASE3_ADDED_FILE={tmp_path}/added\n{fn}_phase3_merge_sample\n" + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=30) + # Status asserted: without it the whole test passes when the merge + # never runs at all. Forcing every call to exit 127 left the first + # version green, because "the sample's value was not appended" is + # also what a merge that never executed looks like (review). + assert r.returncode == 0, f"{label}: the merge did not run: {r.stderr}" + results[label] = env.read_text() + for label, after in results.items(): + assert "sample-value" not in after, ( + f"an existing {label}-form assignment was not detected, so the sample's " + f"value was appended and now wins when env.sh is sourced:\n{after}" + ) + assert "FOO=mine" in after, f"{label}: the existing assignment was lost" + # POSITIVE CONTROL: a genuinely missing key MUST be appended, otherwise + # a detection that matches everything would satisfy the checks above. + env.write_text("export BAR=other\n") + program = f"set -u\nINSTALL_DIR={tmp_path}\n_PHASE3_ADDED_FILE={tmp_path}/added\n{fn}_phase3_merge_sample\n" + r = subprocess.run(["bash", "-c", program], cwd=REPO_ROOT, capture_output=True, text=True, timeout=30) + assert r.returncode == 0, r.stderr + assert "sample-value" in env.read_text(), ( + "a key the device is missing must still be backfilled — without this, a " + "detection that matched everything would pass every assertion above" + ) def test_stale_symlinks_removed(self, update_sh_content): """Post-litclock-dev#79 reorg: root-level script symlinks (boot-splash.sh etc.) @@ -371,9 +423,7 @@ def test_sources_lib_state(self, update_sh_content): # neighbouring comments naming lib/state.sh, so the raw form stayed green # with all three executed `source` lines deleted (measured). This is the # material one: every atomic write in update.sh goes through those helpers. - assert "lib/state.sh" in _executed_lines(update_sh_content), ( - "update.sh must source scripts/lib/state.sh" - ) + assert "lib/state.sh" in _executed_lines(update_sh_content), "update.sh must source scripts/lib/state.sh" # And the helpers must NOT be redefined inside update.sh itself. # `atomic_write_file()` definition would look like the function header. import re @@ -569,7 +619,6 @@ def _setup_fake_install(sandbox, on_master: bool = True, dirty: bool = False): (scripts / "placeholder.sh").write_text("#!/bin/bash\n") - class TestBlockedShaOnSmokeRevert: """litclock-dev#865 — a smoke-gate revert must be remembered. @@ -586,11 +635,10 @@ class TestBlockedShaOnSmokeRevert: reads identically in the diff. """ - BAD = "c5c4a135064f7f482b43d2e6a9c5d42d1bbfe4c5" # the release that failed smoke + BAD = "c5c4a135064f7f482b43d2e6a9c5d42d1bbfe4c5" # the release that failed smoke GOOD = "787d3079163aa94a1936e4b57535e87ec34df8cc" # what we reverted TO - def _run(self, tmp_path, update_sh_content, *, target, rollback=0, state_writable=True, - no_interpreter=0): + def _run(self, tmp_path, update_sh_content, *, target, rollback=0, state_writable=True, no_interpreter=0): """Execute the lifted helper with controlled globals; return (blocked, log).""" import subprocess @@ -682,7 +730,7 @@ def test_the_smoke_revert_arm_calls_it_with_the_failing_sha(self, tmp_path, upda code = "\n".join(ln.split("#", 1)[0] for ln in update_sh_content.splitlines()) smoke = code.index("Smoke test failed (exit $smoke_rc)") - window = code[smoke:smoke + 3000] + window = code[smoke : smoke + 3000] m = re.search(r"^\s*_block_reverted_release\s+(.+)$", window, re.M) assert m, "the smoke-revert arm does not call _block_reverted_release" call = m.group(0).strip() @@ -715,7 +763,7 @@ def test_pip_failure_arm_does_not_block(self, update_sh_content): transient wheel fetch would be worse than the bug this fixes.""" code = "\n".join(ln.split("#", 1)[0] for ln in update_sh_content.splitlines()) pip = code.index("pip install failed") - window = code[pip:pip + 2000] + window = code[pip : pip + 2000] assert "_block_reverted_release" not in window @@ -794,6 +842,7 @@ def test_no_block_file_at_all_falls_through(self, tmp_path, update_sh_content): out = self._run(tmp_path, update_sh_content, target=self.BLOCKED, file_contents=None) assert "REACHED_INSTALL" in out + class TestUpdateScriptExecution: """Run update.sh in a sandbox and verify subprocess orchestration.""" @@ -1208,9 +1257,7 @@ def test_atomic_write_uses_tmp_rename(self, update_sh_content): from pathlib import Path # litclock-dev#782 — EXECUTED lines (see test_sources_lib_state). - assert "lib/state.sh" in _executed_lines(update_sh_content), ( - "update.sh must source the shared atomic helpers" - ) + assert "lib/state.sh" in _executed_lines(update_sh_content), "update.sh must source the shared atomic helpers" state_lib = Path(__file__).resolve().parent.parent / "scripts" / "lib" / "state.sh" assert state_lib.exists(), "scripts/lib/state.sh must exist" lib_text = state_lib.read_text() @@ -1349,9 +1396,7 @@ def test_validate_then_cp_uses_jq_with_to_version_and_freshness(self, update_sh_ assert ".finished_at_unix" in block, "validate-then-cp must check .finished_at_unix freshness" # litclock-dev#782 — EXECUTED lines: this block's docstring and comments # both say "jq -e". - assert "jq -e" in _executed_lines(block), ( - "validate-then-cp must use jq -e for the boolean exit code" - ) + assert "jq -e" in _executed_lines(block), "validate-then-cp must use jq -e for the boolean exit code" # And it must reference $NEW_SHA — that's the just-installed target. assert "$NEW_SHA" in block or "expected" in block, ( "validate-then-cp must compare to_version against the just-installed SHA ($NEW_SHA)" @@ -1617,7 +1662,9 @@ def test_marker_removal_runs_after_the_git_reset(self): reset it would compare the wrong states and the marker would survive the exact updates that invalidate it.""" src = UPDATE_SH.read_text() - assert src.index('NEW_SHA=$(git rev-parse --short HEAD)') < src.index("RUNTIME_MARKER=") + # The PATH is resolved before the reset (litclock-dev#894 revokes there); + # the DIFF is what needs the new tree. + assert src.index("NEW_SHA=$(git rev-parse --short HEAD)") < src.index('git diff --quiet "$OLD_SHA" "$NEW_SHA"') def test_marker_removal_diffs_old_to_new(self): src = UPDATE_SH.read_text() @@ -1645,10 +1692,239 @@ def test_git_diff_failure_also_removes_the_marker_with_an_honest_log(self): block = src[start : src.index("Remove old systemd units", start)] assert "_marker_diff_rc" in block assert "-gt 1" in block - assert block.count('rm -f "$RUNTIME_MARKER"') == 2 + # Three: rollback (litclock-dev#894), inputs changed, diff failed. + assert block.count('rm -f "$RUNTIME_MARKER"') == 3 assert "could not verify" in block +class TestRollbackRevokesTheRuntimeMarker: + """litclock-dev#894 — Stage B flips LITCLOCK_RUNTIME_RENDER to true and + nothing writes it back, so a bootcheck rollback must drop the device to the + pre-rendered tier by revoking the marker, whatever the proof-input diff says. + + EXECUTES the revoke block with `git diff` stubbed. The case that matters is + rollback + UNCHANGED proof inputs: the old code kept the marker there, and + the LKG's painter then ran a tier that LKG never validated. + """ + + @staticmethod + def _slice(src, start, end): + i = src.index(start) + return src[i : src.index(end, i)] + + def _run(self, tmp_path, update_sh_content, *, rollback, diff_rc, writable=True): + # The rollback revoke runs BEFORE the reset (so it holds even when the + # run falls back to the LKG's own update.sh); the proof-input diff runs + # after it. Execute both, in order, exactly as the script does. + early = self._slice(update_sh_content, "RUNTIME_MARKER=$(", "# In rollback mode, snapshot THIS script") + late = self._slice(update_sh_content, "# ANCHOR: runtime-marker-revoke", "# Remove old systemd units") + install = tmp_path / "litclock" + install.mkdir(exist_ok=True) + marker = install / ".runtime-render-validated" + marker.write_text("freetype=2.13.2 digest=abc\n") + if not writable: + install.chmod(0o555) + script = f""" + log_info() {{ echo "INFO $*"; }} + log_error() {{ echo "ERROR $*"; }} + git() {{ return {diff_rc}; }} + INSTALL_DIR="{install}" + OLD_SHA=aaaaaaa NEW_SHA=bbbbbbb + ROLLBACK_MODE={rollback} + {early} + {late} + """ + try: + r = subprocess.run(["bash", "-c", script], capture_output=True, text=True, timeout=30) + finally: + install.chmod(0o755) + assert r.returncode == 0, r.stderr + return marker.exists(), r.stdout + r.stderr + + def test_the_revoke_runs_before_the_reset_and_the_reexec(self, update_sh_content): + """litclock-dev#896 review: if no snapshot can be made, the run execs the LKG's OWN + update.sh, which has no litclock-dev#894 arm. Only a revoke that runs before the + reset holds on that path.""" + src = update_sh_content + revoke = src.index('if [[ -f "$RUNTIME_MARKER" && "$ROLLBACK_MODE" -eq 1 ]]; then') + # Code lines, not the comments near line 500 that quote them. + assert revoke < src.index('\n git reset --hard "$TARGET_SHA"\n') + assert revoke < src.index('\n exec "$SELF_SCRIPT" "$OLD_SHA"\n') + + def test_a_marker_that_cannot_be_removed_is_reported_not_claimed(self, tmp_path, update_sh_content): + """Codex, litclock-dev#896 review: `rm -f` fails silently and the old block logged + "removed" regardless — the LKG would paint text it never validated while + the journal said otherwise. Skipped as root, which unlinks regardless.""" + if os.geteuid() == 0: + pytest.skip("root bypasses the read-only directory") + kept, log = self._run(tmp_path, update_sh_content, rollback=1, diff_rc=0, writable=False) + assert kept + assert "ERROR could not remove" in log + assert "marker removed" not in log + + def test_rollback_revokes_even_when_proof_inputs_are_unchanged(self, tmp_path, update_sh_content): + kept, log = self._run(tmp_path, update_sh_content, rollback=1, diff_rc=0) + assert not kept, "a bootcheck rollback left the marker, so the LKG would paint the text tier" + assert "bootcheck rollback" in log + + def test_control_a_normal_update_with_unchanged_inputs_keeps_the_marker(self, tmp_path, update_sh_content): + """Without this the test above passes on a block that deletes the marker + on every run — which would re-validate for ~3 min every weekly tick.""" + kept, log = self._run(tmp_path, update_sh_content, rollback=0, diff_rc=0) + assert kept + assert "removed" not in log + + def test_a_normal_update_with_changed_inputs_still_revokes(self, tmp_path, update_sh_content): + kept, log = self._run(tmp_path, update_sh_content, rollback=0, diff_rc=1) + assert not kept + assert "proof inputs changed" in log + assert "bootcheck rollback" not in log + + +class TestRollbackSnapshotCarriesItsLib: + """litclock-dev#896 — the rollback re-execs a pre-reset snapshot of update.sh, + and that snapshot sources lib/*.sh relative to ITS OWN location. A lone copy + in /tmp found no lib, ran without read_sha_file, judged rollback-target + "malformed" and installed origin/master: measured on the bench 2026-09-24, + and shipped in every public release through v0.230.0. + + EXECUTES the real snapshot block, deletes the source tree's lib/ (what the + reset to the LKG does to the helpers this release needs), then runs the + snapshot's own lib-loading prelude FROM the snapshot and asserts every helper + resolved. The control runs the same probe on a lone-file copy, the shape that + shipped, and must find the helpers missing. + """ + + # One function that exists ONLY in each lib (not in update.sh's no-op stubs). + HELPERS = ( + "read_sha_file", + "with_env_lock", + "atomic_remove_file", + "github_api_latest_release_tag", + "_write_status_json", + ) + + @staticmethod + def _slice(src, start, end): + i = src.index(start) + return src[i : src.index(end, i)] + + def _tree(self, tmp_path, *, with_lib=True): + scripts = tmp_path / "tree" / "scripts" + scripts.mkdir(parents=True) + shutil.copy(UPDATE_SH, scripts / "update.sh") + if with_lib: + shutil.copytree(REPO_ROOT / "scripts" / "lib", scripts / "lib") + tmpdir = tmp_path / "tmp" + tmpdir.mkdir() + return scripts, tmpdir + + def _snapshot(self, update_sh_content, scripts, tmpdir, *, rollback=1): + block = self._slice(update_sh_content, 'ROLLBACK_SELF_SNAPSHOT=""\n', 'if [[ -n "$TARGET_SHA" ]]; then') + script = f""" + log_warn() {{ echo "WARN $*" >&2; }} + ROLLBACK_MODE={rollback} + SELF_SCRIPT="{scripts / "update.sh"}" + _THIS_SCRIPT_DIR="{scripts}" + export TMPDIR="{tmpdir}" + {block} + echo "SNAP=$ROLLBACK_SELF_SNAPSHOT" + echo "DIR=$ROLLBACK_SELF_SNAPSHOT_DIR" + """ + r = subprocess.run(["bash", "-c", script], capture_output=True, text=True, timeout=30) + assert r.returncode == 0, r.stderr + out = dict(line.split("=", 1) for line in r.stdout.splitlines() if "=" in line) + self.last_stderr = r.stderr + return out["SNAP"], out["DIR"] + + def _probe(self, update_sh_content, script_path): + """Run update.sh's lib-loading prelude as if it WERE `script_path`, and + report which helpers it resolved.""" + prelude = self._slice(update_sh_content, "_THIS_SCRIPT_DIR=$(", "# LitClock state dir.") + probe = Path(script_path).parent / "probe.sh" + probe.write_text( + prelude + "\n" + "".join(f"declare -F {h} >/dev/null && echo HAVE {h}\n" for h in self.HELPERS) + ) + r = subprocess.run( + ["bash", str(probe)], + capture_output=True, + text=True, + timeout=30, + env={**os.environ, "HOME": str(probe.parent)}, + ) + return {line.split()[1] for line in r.stdout.splitlines() if line.startswith("HAVE ")} + + def test_the_snapshot_resolves_every_helper_after_the_tree_loses_its_lib(self, tmp_path, update_sh_content): + scripts, tmpdir = self._tree(tmp_path) + snap, snap_dir = self._snapshot(update_sh_content, scripts, tmpdir) + assert snap, "no snapshot was taken" + assert Path(snap_dir).parent == tmpdir, "snapshot must live outside the tree the reset replaces" + # Codex, litclock-dev#896 review: build the probe from the COPIED script, not the + # repo's, or a snapshot whose update.sh copy were empty would still pass. + assert Path(snap).read_bytes() == UPDATE_SH.read_bytes() + shutil.rmtree(scripts / "lib") # the reset to the LKG + assert self._probe(Path(snap).read_text(), snap) == set(self.HELPERS) + + def test_control_a_lone_copy_is_missing_the_helpers(self, tmp_path, update_sh_content): + """The shape every public release through v0.230.0 shipped. Without this + the probe above could pass on a prelude that resolves nothing at all.""" + lone = tmp_path / "lone" + lone.mkdir() + shutil.copy(UPDATE_SH, lone / "litclock-update-rollback.XXXXXX") + assert self._probe(update_sh_content, lone / "litclock-update-rollback.XXXXXX") == set() + + def test_a_snapshot_without_a_lib_is_refused_and_leaves_nothing_behind(self, tmp_path, update_sh_content): + """A copy that cannot carry state.sh is the broken snapshot again. Refusing + it falls back to the pre-existing no-snapshot path rather than re-execing + something known not to work.""" + scripts, tmpdir = self._tree(tmp_path, with_lib=False) + snap, snap_dir = self._snapshot(update_sh_content, scripts, tmpdir) + assert (snap, snap_dir) == ("", "") + assert list(tmpdir.iterdir()) == [] + # litclock-dev#896 review: the fallback used to be silent. + assert "could not snapshot" in self.last_stderr + + def test_no_snapshot_outside_rollback_mode(self, tmp_path, update_sh_content): + scripts, tmpdir = self._tree(tmp_path) + assert self._snapshot(update_sh_content, scripts, tmpdir, rollback=0) == ("", "") + assert list(tmpdir.iterdir()) == [] + + def _cleanup(self, update_sh_content, *, this_dir, running, nested=""): + tail = self._slice(update_sh_content, "# Snapshot no longer needed", "NEW_SHA=$(git rev-parse") + env = {k: v for k, v in os.environ.items() if k != "LITCLOCK_ROLLBACK_SNAPSHOT_RUNNING"} + if running is not None: + env["LITCLOCK_ROLLBACK_SNAPSHOT_RUNNING"] = str(running) + script = f""" + _THIS_SCRIPT_DIR="{this_dir}" + ROLLBACK_SELF_SNAPSHOT_DIR="{nested}" + {tail} + """ + r = subprocess.run(["bash", "-c", script], capture_output=True, text=True, timeout=30, env=env) + assert r.returncode == 0, r.stderr + + def test_the_snapshot_run_removes_its_own_directory(self, tmp_path, update_sh_content): + """Both litclock-dev#896 adversarial passes: the directory the snapshot runs FROM + was never removed — one leaked per rollback, on the SD card.""" + running = tmp_path / "litclock-update-rollback.AbC123" + nested = tmp_path / "litclock-update-rollback.XyZ789" + running.mkdir() + nested.mkdir() + self._cleanup(update_sh_content, this_dir=running, running=running, nested=nested) + assert not running.exists() and not nested.exists() + + def test_control_the_env_var_alone_never_deletes_anything(self, tmp_path, update_sh_content): + """It removes only its OWN directory, and only one with the snapshot's + name — a stray or hostile value must not turn into an rm -rf.""" + tree = tmp_path / "litclock" / "scripts" + tree.mkdir(parents=True) + self._cleanup(update_sh_content, this_dir=tree, running=tree) # wrong name + other = tmp_path / "litclock-update-rollback.Other1" + other.mkdir() + self._cleanup(update_sh_content, this_dir=tree, running=other) # not its own + self._cleanup(update_sh_content, this_dir=other, running=None) # not a snapshot run + assert tree.exists() and other.exists() + + # --- litclock-dev#682 ------------------------------------------------------ # Every `chmod` in update.sh must be accounted for here. Anything this file @@ -2269,8 +2545,7 @@ def test_credentialed_https_resolves_its_own_pair(self): PAT produces must resolve to ITS pair — the fallback would silently reproduce the litclock-dev#721 pinning. DEV-shaped so the fallback can't fake it.""" assert ( - self._run("https://oauth2:ghp_tok@github.com/kapoorankush/litclock-dev.git") - == "kapoorankush litclock-dev" + self._run("https://oauth2:ghp_tok@github.com/kapoorankush/litclock-dev.git") == "kapoorankush litclock-dev" ) def test_charset_invalid_component_falls_back(self): @@ -2315,9 +2590,7 @@ def test_resolver_actually_uses_the_derived_pair(self): body = UPDATE_SH.read_text() start = body.index("resolve_target_sha() {") end = body.index("\n}\n", start) - span = "\n".join( - ln for ln in body[start:end].splitlines() if not ln.lstrip().startswith("#") - ) + span = "\n".join(ln for ln in body[start:end].splitlines() if not ln.lstrip().startswith("#")) assert 'github_api_latest_release_tag "$_owner" "$_repo"' in span, ( "resolve_target_sha no longer queries the origin-derived pair (litclock-dev#721)" ) @@ -2335,7 +2608,10 @@ def test_env_sh_lock_is_gitignored(): not grepped against .gitignore prose.""" r = subprocess.run( ["git", "check-ignore", "env.sh.lock"], - cwd=REPO_ROOT, capture_output=True, text=True, timeout=30, + cwd=REPO_ROOT, + capture_output=True, + text=True, + timeout=30, ) assert r.returncode == 0, "env.sh.lock is not gitignored — every OTA warns about it forever" @@ -2363,7 +2639,7 @@ def _run(self, branch: str, at_release_tag: bool = True) -> str: assert "git status" not in span, "span overgrew into the porcelain block" describe = "printf 'v9.9.9\\n'" if at_release_tag else "return 1" script = ( - 'git() {\n' + "git() {\n" ' case "$1 $2" in\n' ' "rev-parse --abbrev-ref") printf %s\\\\n ' + shlex.quote(branch) + " ;;\n" ' "describe --tags") ' + describe + " ;;\n" @@ -2371,7 +2647,7 @@ def _run(self, branch: str, at_release_tag: bool = True) -> str: " *) return 1 ;;\n" " esac\n" "}\n" - 'log_warn() { printf \'WARN: %s\\n\' "$1"; }\n' + span + "log_warn() { printf 'WARN: %s\\n' \"$1\"; }\n" + span ) result = subprocess.run(["bash", "-c", script], capture_output=True, text=True, timeout=30) assert result.returncode == 0, result.stderr @@ -2535,13 +2811,16 @@ def _gate_span(content: str) -> str: dispatch too — it is the branch the probes exist to drive. """ probe_at = content.index("catalog_probe=") - start = content.rindex('if [[ "$smoke_rc" -eq 0 ]]; then', 0, probe_at) + # From the rollback switch, not the first probe's `if`: the switch is a + # branch condition of every probe, so by litclock-dev#662's rule it belongs inside + # the executed span (litclock-dev#871 port /review). + start = content.rindex("catalog_probes=1", 0, probe_at) # LIFT the initialisers rather than inject them (/review). Injecting # `smoke_rc=0` made deleting it from the script an equivalent mutant — # the harness supplied its own. Lifted, a deleted initialiser is an # IndexError here, i.e. red. - init = content[content.index(_SMOKE_INIT):][: len(_SMOKE_INIT)] + init = content[content.index(_SMOKE_INIT) :][: len(_SMOKE_INIT)] # litclock-dev#773 moved the KEEP/revert dispatch OUT of the # `if [[ -x "$PYTHON" ]]` block, so the text between the probes and the @@ -2555,9 +2834,7 @@ def _gate_span(content: str) -> str: end = content.index("fi\n", content.index("exit 1", end)) + len("fi\n") span = init + content[start:probes_end] + "\n" + content[dispatch_start:end] - invocations = [ - ln for ln in span.splitlines() if '"$PYTHON" src/eink_display.py catalog-get' in ln - ] + invocations = [ln for ln in span.splitlines() if '"$PYTHON" src/eink_display.py catalog-get' in ln] assert len(invocations) == 2, ( f"expected exactly 2 probe sites inside the lifted span; found {len(invocations)}. " "Counting the bare word 'catalog-get' would count the comments too. This is `==`, " @@ -2570,9 +2847,7 @@ def _gate_span(content: str) -> str: # a timeout), so it joins `invocations` rather than getting a weaker # check of its own — an unpinned or unbounded probe is the same hazard # whichever subcommand it calls. - count_invocations = [ - ln for ln in span.splitlines() if '"$PYTHON" src/eink_display.py catalog-count' in ln - ] + count_invocations = [ln for ln in span.splitlines() if '"$PYTHON" src/eink_display.py catalog-count' in ln] assert len(count_invocations) == 1, ( f"expected exactly 1 catalog-count probe inside the lifted span; found " f"{len(count_invocations)}. Same `==` reasoning as above." @@ -2582,9 +2857,7 @@ def _gate_span(content: str) -> str: # away is invisible to every test in this class, so nothing that the # class exists to check may live there. discarded = content[probes_end:dispatch_start] - discarded_code = "\n".join( - ln for ln in discarded.splitlines() if not ln.lstrip().startswith("#") - ) + discarded_code = "\n".join(ln for ln in discarded.splitlines() if not ln.lstrip().startswith("#")) for subcommand in ("catalog-get", "catalog-count"): assert subcommand not in discarded_code, ( f"a {subcommand} probe sits in the region the stitch discards — it would be " @@ -2710,7 +2983,7 @@ def _catalog_get(self, root, key, **overrides): env=self._clean_env(**overrides), ).stdout.strip() - def _run_gate(self, root, span): + def _run_gate(self, root, span, prelude=""): """Execute the lifted span. Returns (result, probes_in_order).""" probe_log = root / "probes.log" wrapper = root / "python-probe-wrapper" @@ -2755,6 +3028,10 @@ def _run_gate(self, root, span): # dies as `command not found`; driven for real in # tests/test_runtime_render_autostamp.py. '_runtime_render_selftest() { echo "STUB_SELFTEST"; }\n' + # ...and Stage B's migration, called straight after it. Unstubbed it + # printed `command not found` and the run stayed green (port /review). + '_runtime_render_migrate() { echo "STUB_MIGRATE"; }\n' + f"{prelude}" f"PYTHON={shlex.quote(str(wrapper))}\n" "REVERT_SHA=deadbeef\nUPDATE_FAILED_FILE=/dev/null\nHASH_FILE=/dev/null\n" # litclock-dev#531 — the KEEP arm now re-stamps the runtime-render @@ -2792,6 +3069,7 @@ def _run_gate(self, root, span): def _assert_kept(self, r): """The gate KEPT the update — the whole dispatch ran, not just the probes.""" assert "REACHED_END rc=0" in r.stdout, f"the gate did not keep the update\n{r.stdout}\n{r.stderr}" + assert "command not found" not in r.stderr, f"an unstubbed call in the lifted span:\n{r.stderr}" assert r.returncode == 0, r.stderr assert "Smoke test passed" in r.stdout assert "STUB_GIT reset --hard" not in r.stdout, "a passing gate must not revert" @@ -2825,6 +3103,41 @@ def test_the_fixture_really_does_create_the_failure_condition(self, tmp_path): # ...and the pin is what pulls it back to English. assert self._catalog_get(root, self.STATUS_KEY, LITCLOCK_LANGUAGE="en") == "just now" + @staticmethod + def _make_pre_catalog_lkg(root): + """Turn the fixture into a last-known-good from before the probes existed. + + Public v0.219.0-v0.225.3 have no `catalog-get` and v0.226.0 no + `catalog-count` (checked against the tags): argparse rejects the + subcommand with exit 2 and prints nothing on stdout. That is all the + probes can see of an old tree, so that is what this stands in for. + """ + (root / "src" / "eink_display.py").write_text( + "import sys\nprint('usage: eink_display.py ...', file=sys.stderr)\nsys.exit(2)\n", + encoding="utf-8", + ) + + def test_a_rollback_to_a_pre_catalog_lkg_is_kept(self, update_sh_content, tmp_path): + """litclock-dev#871 port /review: the rollback runs THIS release's gate + against the LKG's tree. An LKG older than the probes failed every one of + them, so the rollback took the smoke-failure exit instead of finishing.""" + root = self._fake_checkout(tmp_path) + self._make_pre_catalog_lkg(root) + r, probes = self._run_gate(root, self._gate_span(update_sh_content), prelude="ROLLBACK_MODE=1\n") + self._assert_kept(r) + assert "Catalog smoke skipped in rollback mode" in r.stdout, r.stdout + assert probes == [], f"no catalog probe may run in rollback mode; ran {probes}" + + def test_the_same_tree_outside_rollback_mode_still_reverts(self, update_sh_content, tmp_path): + """The control: the skip is rollback-only. A NEW release that shipped a + broken catalog must still be reverted, or the gate lost its purpose.""" + root = self._fake_checkout(tmp_path) + self._make_pre_catalog_lkg(root) + r, probes = self._run_gate(root, self._gate_span(update_sh_content), prelude="ROLLBACK_MODE=0\n") + self._assert_reverted(r) + assert "Catalog smoke skipped" not in r.stdout, r.stdout + assert probes, "the probes must have run outside rollback mode" + def test_the_gate_passes_on_a_non_english_device(self, update_sh_content, tmp_path): root = self._fake_checkout(tmp_path) r, probes = self._run_gate(root, self._gate_span(update_sh_content)) @@ -2995,9 +3308,7 @@ def test_a_bundle_truncated_to_the_probed_keys_is_now_caught(self, update_sh_con f"a bundle truncated to the 3 probed keys must fail the COUNT probe\n{r.stdout}" ) self._assert_reverted(r) - assert probes == [*probed, "catalog-count"], ( - f"the count probe must run after the value probes; got {probes}" - ) + assert probes == [*probed, "catalog-count"], f"the count probe must run after the value probes; got {probes}" def test_the_count_probe_reports_zero_rather_than_dying_when_the_catalog_is_broken(self, tmp_path): """Same `always exit 0 with a value` contract as catalog-get, and for @@ -3252,9 +3563,7 @@ def _full_span(content: str) -> str: def test_the_dispatch_is_outside_the_interpreter_check(self, update_sh_content): """Structural, and the reason: `smoke_rc` is not read anywhere after the gate, so an `else` arm that only sets it changes nothing.""" - executed = "\n".join( - ln for ln in update_sh_content.splitlines() if not ln.lstrip().startswith("#") - ) + executed = "\n".join(ln for ln in update_sh_content.splitlines() if not ln.lstrip().startswith("#")) gate = executed.index('if [[ -x "$PYTHON" ]]; then') dispatch = executed.index('if [[ "$smoke_rc" -eq 0 ]]; then\n log_info "Smoke test passed"') # Column 0 == top level == outside the -x block. @@ -3333,7 +3642,11 @@ def _run_missing_interpreter(self, update_sh_content, root, no_interpreter=True) 'echo "REACHED_END rc=$smoke_rc"\n' ) return subprocess.run( - ["bash", "-c", program], cwd=root, capture_output=True, text=True, timeout=120, + ["bash", "-c", program], + cwd=root, + capture_output=True, + text=True, + timeout=120, env={k: v for k, v in os.environ.items() if not k.startswith("LITCLOCK_")}, ) @@ -3344,9 +3657,7 @@ def test_a_missing_interpreter_reverts(self, update_sh_content, tmp_path): root = f._fake_checkout(tmp_path) r = self._run_missing_interpreter(update_sh_content, root) - assert "Smoke test SKIPPED" in r.stdout, ( - f"the missing interpreter was not reported\n{r.stdout}\n{r.stderr}" - ) + assert "Smoke test SKIPPED" in r.stdout, f"the missing interpreter was not reported\n{r.stdout}\n{r.stderr}" assert "STUB_GIT reset --hard deadbeef" in r.stdout, ( "a missing interpreter did not revert. Every check was skipped, so the update " f"would install and report SUCCESS having verified nothing.\n{r.stdout}" @@ -3376,12 +3687,8 @@ def test_a_missing_interpreter_reports_unrecovered_not_reverted(self, update_sh_ "the missing-interpreter path did not report failed_unrecovered. If it reported " f"failed_reverted the owner is told the clock is running while it is dead.\n{r.stdout}" ) - assert "STUB_STATUS_REVERTED" not in r.stdout, ( - f"both terminal statuses fired; exactly one must.\n{r.stdout}" - ) - assert "clock NOT restored" in r.stdout, ( - f"the console banner still claims a clean rollback.\n{r.stdout}" - ) + assert "STUB_STATUS_REVERTED" not in r.stdout, f"both terminal statuses fired; exactly one must.\n{r.stdout}" + assert "clock NOT restored" in r.stdout, f"the console banner still claims a clean rollback.\n{r.stdout}" def test_a_genuine_probe_failure_still_reports_reverted(self, update_sh_content, tmp_path): """The control. Without it, `failed_unrecovered` unconditionally — which @@ -3400,7 +3707,6 @@ def test_a_genuine_probe_failure_still_reports_reverted(self, update_sh_content, ) - class TestTheExitTrapRearmsTheClock: """litclock-dev#835, EXECUTED — the first version of this class scanned the trap's non-comment lines and passed with the re-arm inside `if false` @@ -3425,8 +3731,8 @@ def _run(self, update_sh_content, *, finalized: int, signal: bool, sudo_fails: b "set -u\n" 'sudo() { [ "$SUDO_FAILS" = 1 ] && return 1; echo "STUB_SUDO $*"; }\n' 'timeout() { [ "$1" = -k ] && shift 2; shift; "$@"; }\n' - 'update_status_failed_unrecovered() { echo STUB_STAMP; }\n' - 'atomic_write_file() { echo STUB_GRACE; }\n' + "update_status_failed_unrecovered() { echo STUB_STAMP; }\n" + "atomic_write_file() { echo STUB_GRACE; }\n" "_PHASE3_ADDED_FILE=\nPOST_UPDATE_GRACE_FILE=/dev/null\n" f"SUDO_FAILS={1 if sudo_fails else 0}\n" f"{self._lifted(update_sh_content)}" @@ -3477,8 +3783,8 @@ def test_a_signal_during_cleanup_does_not_abort_it(self, update_sh_content): # The re-arm itself delivers a TERM to the shell mid-cleanup. 'sudo() { echo "STUB_SUDO $*"; kill -TERM $$; sleep 0.2; echo REARM_FINISHED; }\n' 'timeout() { [ "$1" = -k ] && shift 2; shift; "$@"; }\n' - 'update_status_failed_unrecovered() { echo STUB_STAMP; }\n' - 'atomic_write_file() { echo STUB_GRACE; }\n' + "update_status_failed_unrecovered() { echo STUB_STAMP; }\n" + "atomic_write_file() { echo STUB_GRACE; }\n" "_PHASE3_ADDED_FILE=\nPOST_UPDATE_GRACE_FILE=/dev/null\n" f"{self._lifted(update_sh_content)}" "_LITCLOCK_UPDATE_FINALIZED=0\n_LITCLOCK_UPDATE_PHASE_INDEX=4\n" @@ -3566,7 +3872,7 @@ def test_a_hung_ls_remote_is_cut_at_the_bound(self, update_sh_content, tmp_path) program = ( f"set -u\nexport PATH={bindir}:$PATH\nexport LITCLOCK_LS_REMOTE_TIMEOUT_S=1\n" + self._gate_block(update_sh_content) - + "t0=$SECONDS; _remote_reachable; rc=$?; echo \"rc=$rc elapsed=$((SECONDS-t0))\"\n" + + 't0=$SECONDS; _remote_reachable; rc=$?; echo "rc=$rc elapsed=$((SECONDS-t0))"\n' f"printf '#!/bin/bash\\nexit 0\\n' > {bindir}/git\n_remote_reachable; echo \"ok_rc=$?\"\n" ) r = subprocess.run(["bash", "-c", program], capture_output=True, text=True, timeout=30) @@ -3589,7 +3895,7 @@ def test_a_term_immune_probe_tree_is_killed_by_the_escalation(self, update_sh_co program = ( f"set -u\nexport PATH={bindir}:$PATH\nexport LITCLOCK_LS_REMOTE_TIMEOUT_S=1\n" + self._gate_block(update_sh_content) - + "_remote_reachable; echo \"rc=$?\"\n" + + '_remote_reachable; echo "rc=$?"\n' "sleep 0.5\n" # Anchored: an unanchored -f matches this very harness's command line. f'pgrep -f "^sleep {marker}$" >/dev/null; echo "pgrep_rc=$?"\n' @@ -3642,14 +3948,14 @@ def test_a_term_injected_inside_cleanup_never_loses_the_recovery(self, update_sh "set -u\nset -T\n" 'sudo() { echo "STUB_SUDO $*"; }\n' 'timeout() { [ "$1" = -k ] && shift 2; shift; "$@"; }\n' - 'update_status_failed_unrecovered() { echo STUB_STAMP; }\n' - 'atomic_write_file() { echo STUB_GRACE; }\n' + "update_status_failed_unrecovered() { echo STUB_STAMP; }\n" + "atomic_write_file() { echo STUB_GRACE; }\n" "_PHASE3_ADDED_FILE=\nPOST_UPDATE_GRACE_FILE=/dev/null\n" f"{self._lifted(update_sh_content)}" "_LITCLOCK_UPDATE_FINALIZED=0\n_LITCLOCK_UPDATE_PHASE_INDEX=4\n" "_N=0\n" - "trap 'if [ \"${FUNCNAME[0]:-}\" = _litclock_update_cleanup ]; then _N=$((_N+1)); " - f"if [ \"$_N\" = {nth_statement} ]; then echo INJECTED; kill -TERM $$; fi; fi' DEBUG\n" + 'trap \'if [ "${FUNCNAME[0]:-}" = _litclock_update_cleanup ]; then _N=$((_N+1)); ' + f'if [ "$_N" = {nth_statement} ]; then echo INJECTED; kill -TERM $$; fi; fi\' DEBUG\n' "exit 7\n" ) r = subprocess.run(["bash", "-c", program], capture_output=True, text=True, timeout=30) @@ -3668,7 +3974,7 @@ def test_the_signal_handler_cleans_up_itself_and_cleanup_is_idempotent(self, upd assert r.returncode == 143 assert r.stdout.count("STUB_STAMP") == 1 and r.stdout.count("systemctl start") == 1, r.stdout lifted = self._lifted(update_sh_content) - handler = lifted[lifted.index("_litclock_update_on_signal() {"):] + handler = lifted[lifted.index("_litclock_update_on_signal() {") :] assert "_litclock_update_cleanup" in handler.split("exit 143")[0], ( "the signal handler must run the cleanup BEFORE exiting, not leave it to the EXIT trap" ) @@ -3725,8 +4031,7 @@ def _smoke_arm(content: str) -> str: end = content.index("exit 1", end) + len("exit 1") return content[start:end] - def _run(self, update_sh_content, tmp_path, *, arm: str, already_matching: bool = False, - cp_fails: bool = False): + def _run(self, update_sh_content, tmp_path, *, arm: str, already_matching: bool = False, cp_fails: bool = False): root = tmp_path / "tree" (root / "systemd").mkdir(parents=True) (root / "lkg").mkdir() @@ -3757,7 +4062,7 @@ def _run(self, update_sh_content, tmp_path, *, arm: str, already_matching: bool # cp really copies; every systemctl start of a clock unit snapshots # what is installed at that instant. f"CP_FAILS={1 if cp_fails else 0}\n" - 'sudo() {\n' + "sudo() {\n" ' case "$1" in\n' ' cp) [ "$CP_FAILS" = 1 ] && { echo "STUB_SUDO_CP_REFUSED $*"; return 1; };\n' ' cp "$2" "$3" && echo "STUB_SUDO $*";;\n' @@ -3765,8 +4070,8 @@ def _run(self, update_sh_content, tmp_path, *, arm: str, already_matching: bool ' if [ "$2" = start ]; then\n' ' echo "INSTALLED_AT_START $3 $(tr "\\n" "|" < "$SYSTEMD_UNIT_DIR/$3")"; fi;;\n' ' *) echo "STUB_SUDO $*";;\n' - ' esac\n' - '}\n' + " esac\n" + "}\n" 'atomic_write_file() { echo "STUB_ATOMIC_WRITE $1"; }\n' 'update_status_failed_reverted() { echo "STUB_STATUS_REVERTED $1"; }\n' 'update_status_failed_unrecovered() { echo "STUB_STATUS_UNRECOVERED $1"; }\n' @@ -3783,7 +4088,11 @@ def _run(self, update_sh_content, tmp_path, *, arm: str, already_matching: bool 'echo "REACHED_END"\n' ) r = subprocess.run( - ["bash", "-c", program], cwd=root, capture_output=True, text=True, timeout=60, + ["bash", "-c", program], + cwd=root, + capture_output=True, + text=True, + timeout=60, env={k: v for k, v in os.environ.items() if not k.startswith("LITCLOCK_")}, ) return r, units @@ -3837,7 +4146,8 @@ def test_each_copy_is_reloaded_before_the_next_one_starts(self, update_sh_conten successful copy removes it.""" r, _ = self._run(update_sh_content, tmp_path, arm=arm) events = [ - ln for ln in r.stdout.splitlines() + ln + for ln in r.stdout.splitlines() if ln.startswith("STUB_SUDO cp ") or ln == "STUB_SUDO systemctl daemon-reload" ] assert len(events) == 4, f"expected copy,reload,copy,reload; got {events}" @@ -3938,7 +4248,7 @@ def _run(self, update_sh_content, tmp_path, *, lib_sourced: bool = True, start_f ' case "$*" in *"--no-block"*) return 0;; esac\n' ' [ "$START_FAILS" = 1 ] && return 1; return 0; }\n' 'timeout() { [ "$1" = -k ] && shift 2; shift; "$@"; }\n' - 'atomic_write_file() { :; }\n' + "atomic_write_file() { :; }\n" "_PHASE3_ADDED_FILE=\nPOST_UPDATE_GRACE_FILE=/dev/null\n" + ("github_api_latest_release_tag() { :; }\n" if lib_sourced else "") + "update_status_init abc1234\nupdate_status_set_phase 1\n" @@ -4116,15 +4426,17 @@ def _run(cls, lines, sep, frac=None): several shapes and which one you get turns on the digits. """ frac = frac or cls._DEFAULT_FRAC - script = "\n".join([ - "unset EPOCHREALTIME", - f'EPOCHREALTIME="{cls._T0_SEC}{sep}{cls._T0_FRAC}"', - lines["t0"], - f'EPOCHREALTIME="{cls._T1_SEC}{sep}{frac}"', - lines["dur_ms"], - lines["duration"], - 'echo "$duration"', - ]) + script = "\n".join( + [ + "unset EPOCHREALTIME", + f'EPOCHREALTIME="{cls._T0_SEC}{sep}{cls._T0_FRAC}"', + lines["t0"], + f'EPOCHREALTIME="{cls._T1_SEC}{sep}{frac}"', + lines["dur_ms"], + lines["duration"], + 'echo "$duration"', + ] + ) r = subprocess.run(["bash", "-c", script], capture_output=True, text=True) return r.returncode, r.stdout.strip(), r.stderr.strip() @@ -4181,9 +4493,7 @@ def test_the_mutant_literal_dot_is_silently_wrong_with_a_comma(self, update_sh_c rc, out, err = self._run(old, ",") assert rc == 0, f"expected the old pattern to fail SILENTLY, got rc={rc}" assert err == "", f"expected no diagnostic, got: {err!r}" - assert out != self._expected(self._DEFAULT_FRAC), ( - "the old pattern parsed a comma correctly — premise gone" - ) + assert out != self._expected(self._DEFAULT_FRAC), "the old pattern parsed a comma correctly — premise gone" # …and it is RECORDED. Silence at the bash layer is only half the # hazard: `_runtime_selftest_record_write` builds the JSON through # `jq … tonumber`, so a value jq refuses never reaches the file and @@ -4191,8 +4501,9 @@ def test_the_mutant_literal_dot_is_silently_wrong_with_a_comma(self, update_sh_c if shutil.which("jq") is None: pytest.skip("the record is written through jq") r = subprocess.run( - ["jq", "-nc", "--arg", "d", out, '{duration_s: ($d | tonumber)}'], - capture_output=True, text=True, + ["jq", "-nc", "--arg", "d", out, "{duration_s: ($d | tonumber)}"], + capture_output=True, + text=True, ) assert r.returncode == 0, ( f"jq refused {out!r}, so this fraction exercises the LOUD band, not the " @@ -4214,8 +4525,7 @@ def test_the_mutant_literal_dot_can_also_land_in_the_loud_band(self, update_sh_c old = self._as_literal_dot(self._shipped_lines(update_sh_content)) rc, out, err = self._run(old, ",", frac=self._JQ_REFUSED_FRAC) assert (rc, err) == (0, ""), (rc, err) - r = subprocess.run(["jq", "-nc", "--arg", "d", out, '($d | tonumber)'], - capture_output=True, text=True) + r = subprocess.run(["jq", "-nc", "--arg", "d", out, "($d | tonumber)"], capture_output=True, text=True) assert r.returncode != 0, f"expected jq to refuse {out!r}, it accepted it" def test_the_mutant_literal_dot_is_noisy_on_a_leading_zero_fraction(self, update_sh_content): @@ -4255,13 +4565,13 @@ class TestSelfTestDurationSurvivesToTheRecord: """ SLEEP_S = 1.2 - TOLERANCE_S = 1.0 # generous: a loaded box may add most of a second + TOLERANCE_S = 1.0 # generous: a loaded box may add most of a second @staticmethod def _function_text(): sh = (REPO_ROOT / "scripts" / "update.sh").read_text() start = sh.index("_runtime_render_selftest() {") - return sh[start:sh.index("\n}\n", start) + 3] + return sh[start : sh.index("\n}\n", start) + 3] @classmethod def _run_real_function(cls, tmp_path, sleep_s=None): @@ -4269,7 +4579,8 @@ def _run_real_function(cls, tmp_path, sleep_s=None): painter = tmp_path / "painter" painter.write_text(f"#!/bin/bash\nsleep {sleep_s or cls.SLEEP_S}\n") painter.chmod(0o755) - harness = textwrap.dedent(f""" + harness = ( + textwrap.dedent(f""" set -o pipefail SELFTEST_TIMEOUT_S=60 INSTALL_DIR={tmp_path}/nonexistent @@ -4281,7 +4592,10 @@ def _run_real_function(cls, tmp_path, sleep_s=None): atomic_remove_file() {{ :; }} _runtime_validation_memo_write() {{ echo "MEMO rc=$2"; }} _runtime_selftest_record_write() {{ echo "RECORDED=$1"; }} - """) + cls._function_text() + "\n_runtime_render_selftest\n" + """) + + cls._function_text() + + "\n_runtime_render_selftest\n" + ) r = subprocess.run(["bash", "-c", harness], capture_output=True, text=True, cwd=REPO_ROOT) return r @@ -4328,4 +4642,3 @@ def test_a_longer_paint_records_a_longer_duration(self, tmp_path): f"a painter that ran 2.2s longer moved the recorded duration by only " f"{ln - s}s ({s} -> {ln}) — the value is not tracking the clock" ) -