diff --git a/.images-version b/.images-version index cc40603..882cecb 100644 --- a/.images-version +++ b/.images-version @@ -1 +1 @@ -v9 +v11 diff --git a/CHANGELOG.md b/CHANGELOG.md index afcd84a..5ea9cab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,24 @@ All notable changes to LitClock are documented here. Format loosely follows [Kee ## [Unreleased] +### Changed + +- Two more quotes are classified as mature, so they now appear only on a clock with mature content switched on. + +### Fixed + +- Rebooting while the startup screen is still on the panel now paints the restart farewell. Before, the panel could stay frozen on the startup screen through the power-off, or show the shutdown farewell for a restart. +- After a settings reset, the reboot it asks you to perform paints the farewell screen again instead of leaving the previous quote on the panel. +- A quote refresh that lands while the clock is still starting up no longer fails outright — it waits for the startup screen to release the display. +- A blank or invalid nightly display-cleaning hour no longer freezes the panel; the clock falls back to the default hour and keeps painting. +- A clock with a location you typed in yourself no longer shows a permanent stale-location warning in Diagnostics. +- On the setup page, the box for typing a hidden network's name stays out of the way until you pick "My network isn't listed", so the form can no longer show two conflicting answers at once, and a name typed on an earlier attempt can no longer come back already filled in. +- Preparing an SD card for cloning now stops before it erases anything if your settings could not be wiped, so a refused run leaves the card untouched and the clock still working. +- Preparing an SD card for cloning now really clears the shell history, instead of having it written back when you close the terminal you ran it from. +- An update that fails its safety check is now remembered, so the clock stops downloading, failing and rolling back the same broken release every week. +- A weekly update check that could not reach the internet is no longer reported as a failed update needing manual recovery, and no longer moves the "Last update" date for an update that never happened. +- A rolled-back update no longer leaves the clock a minute behind, showing the previous minute's quote on every tick until the next update succeeded. + ## [v0.227.0] - 2026-09-13 ### Changed diff --git a/CLAUDE.md b/CLAUDE.md index 71af191..ebcf79e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -49,7 +49,7 @@ The first-boot flow (`scripts/first-boot.sh`) provisions WiFi via a web UI; ever **Critical scenarios to test:** -- **WiFi-only hotspot form**: Verify the setup page shows ONLY the WiFi network picker + password field + Submit button (plus, on a multi-language fleet only, the Language select — dormant while English is the sole active registry language, litclock-dev#532). No Location, Timezone, Temperature, or Mature-content sections — those are PWA-only post-handoff. +- **WiFi-only hotspot form**: Verify the setup page shows ONLY the WiFi network picker + password field + Submit button (plus, on a multi-language fleet only, the Language select — dormant while English is the sole active registry language, litclock-dev#532). No Location, Timezone, Temperature, or Mature-content sections — those are PWA-only post-handoff. The hidden-network "Network name" box is NOT visible at load: since litclock-dev#848 it appears only after picking "My network isn't listed" in the dropdown (and is focused then — not at load), and picking a real network afterwards hides it again AND blanks anything typed. Two exceptions arrive with the option pre-selected and the box showing: a retry that echoes a hand-typed name, and an empty scan. Also check **Refresh**: with the manual option selected and a name typed, tap Refresh — the option must still be selected and the name still there when the list lands; and separately, tap Refresh from the placeholder and pick "My network isn't listed" DURING the scan (it runs 2-20s) — the box must NOT vanish when the response arrives. With JavaScript OFF the disclosure is present in every state, as it was before — but on a clean render it is CLOSED, so only its summary line shows and the box needs one tap to reveal; the retry-echo and empty-scan renders arrive open. That is the no-JS fallback, and the server's rule (a picked network wins over typed text) is the only guard there. - **Hotspot creation**: Power on with no known WiFi networks. Verify the Pi creates a hotspot and displays credentials + QR code on the e-ink screen. - **litclock-dev#620 hotspot-password block — RUN IN ORDER, and only on a device you can re-flash.** Checks 1-4 are sequential and destructive: check 2 needs the phone state check 1 leaves behind, check 3 destroys the password both depend on, and check 4 needs its own fresh flash. Out of order they need a re-flash to redo. Check 5 is order-independent — it only reads a support bundle, so run it any time after provisioning. (These steps describe the litclock-dev#620 feature: if that PR has not merged, none of the paths below exist yet.) 1. **The password is STABLE across cycles** — the core invariant, invisible without hardware. Provision, note the password on the e-ink, then re-enter setup with `sudo systemctl start --no-block litclock-wifi-reset.service`. **Do NOT run `litclock-wifi-reset.sh` foregrounded over SSH**: it deletes every WiFi profile before it clears `.setup-complete`, so your own connection dies mid-script and SIGHUP kills it before it finishes — leaving a device with no WiFi, no hotspot, and the look of a brick. Verify the panel shows **the same password**, then `sudo stat -c '%a %U:%G' /var/lib/litclock/hotspot-password` returns exactly `600 pi:pi`. Ownership matters as much as the mode: `litclock-firstboot.service` is `User=pi`, so a `600 root:root` file (what a maintainer running the CLI under sudo leaves behind) is unreadable to the real writer and silently rotates the password every cycle — the exact bug this feature removes. Both verification commands need a shell, and the reset you just triggered deleted the WiFi profile your SSH session was riding on — **reconnect over the `LitClock-Setup` hotspot and SSH to its gateway IP** (or run this step on an Ethernet-attached rig) before expecting them to work. If cycle 2 differs, get the cause for free with `journalctl -u litclock-firstboot | grep -iE 'hotspot.password'`: the code emits distinguishable lines for unreadable ("minting a REPLACEMENT"), invalid ("regenerating"), unwritable ("this cycle only") and race ("adopting the stored value"). Use that regex, not `'hotspot password'` — the race line is the one message that spells it `hotspot-password` with a hyphen, so a literal-space grep silently hides the single case this list exists to diagnose. @@ -61,15 +61,36 @@ The first-boot flow (`scripts/first-boot.sh`) provisions WiFi via a web UI; ever |---|---|---|---| | ~2026-08-10 (litclock-dev#620) | OnePlus 6T | *not recorded* | "No Internet Access"; no password field offered; **a QR scan did not override the saved entry** (the phone did not register an attempt) | | 2026-08-15 (bench, `dev-20260815-b0c0590`) | *not recorded* | *not recorded* | "Connection failed — Wrong password for `LitClock-Setup`"; **offered "Change password"**; **a QR scan connected on the first try**, captive portal followed, provisioning completed normally | + | 2026-09-18 (bench, `dev-20260918-787d307`) | iPhone | **iOS 27** | **QR scan joined first try** — camera scan raised a "join LitClock-Setup?" prompt, tapped Join, ~10s, captive portal opened by itself. No password prompt, no failure banner. First iOS measurement; the stale credential was two rotations old (a gift-mode prep and a PWA factory reset had each re-minted the key that same afternoon). **Do not record key values in this file** — it is published; the rotation COUNT is the measurement, the keys are a credential | The 2026-08-15 run was on a device whose hotspot password had just been rotated by `--gift-mode` — functionally the same condition the original describes. All three of the original's Android claims failed to reproduce. That the device and OS columns are half empty is itself the finding: fill them in next time. + + **The 2026-09-18 row is the first with both columns filled, and the first + iOS one.** It matters that the platform is recorded: the two Android rows + disagree with each other, so a third undated row would have been + unattributable to either a platform difference or OS drift. iOS 27 simply + joined — no prompt, no banner, no recovery needed. It was a QR scan, so it + settles the QR-override sub-check below on iOS. The inverse is now the + unrecorded one: whether a plain tap on `LitClock-Setup` (no QR) also joins + past a stale credential on iOS. + + **The owner reports iOS has behaved this way on earlier devices too** + (2026-09-18, recollection rather than a dated run — deliberately NOT given + a table row, since undated claims are the thing this table exists to + replace). It does change where re-test effort belongs: the two Android + rows contradict each other on whether recovery is discoverable at all, + while iOS appears consistent and always has. Treat the uncertainty as + **Android-side** — vary the device and the OS version there — and treat + iOS as settled unless it starts failing. - **Severity, corrected (litclock-dev#648).** The original block argued Android left *"no user-discoverable recovery short of Forget This Network, which the intended recipient will not find."* On the 2026-08-15 device recovery was one tap, and a QR scan bypassed the saved entry entirely. Treat the strong version as unproven rather than as established fact. **This does not weaken litclock-dev#620 itself**, which stays: a stable per-device password means a normal owner never reaches any of these screens, and that is the right outcome however gracefully a given Android build degrades. Only the severity narrative was stale. - - **New sub-check: a QR scan overrides a stale saved entry.** Now the interesting case, and previously untested anywhere. Every LitClock broadcasts the same SSID with a per-device password, so a phone that set up clock A meets clock B with the wrong key — the recovery path a real recipient is most likely to stumble into. Scan the panel QR on a phone holding a stale credential for `LitClock-Setup` and confirm it joins. It worked on 2026-08-15; it did not on the original measurement, so this is worth re-running per device rather than assuming. - 3. **Which resets keep the setup password and which rotate it — since litclock-dev#666 the DEFAULT rotates.** The old rule (wipe AND power-off) is gone; erasing both passwords is now what a bare reset does, and `--keep-wifi` is the only way to preserve either. Run the sub-checks in this order, because each destroys state the next would need. First **`--keep-wifi`** (the "same owner, moved house" case): `sudo cat /var/lib/litclock/hotspot-password`, keep the value, `sudo ./scripts/reset-setup.sh --keep-wifi --poweroff`, power back on, `sudo cat` again and confirm it is **UNCHANGED**. Read it from the file, not the panel — with the WiFi kept, the device boots straight onto its saved network and never raises a setup network, so there is no panel password on that path. Then a **bare `sudo ./scripts/reset-setup.sh --yes`** over SSH: expect the session to DROP when the WiFi goes (that is the documented behaviour, not a fault), power-cycle, and confirm the panel password is **DIFFERENT**. Then the **PWA Factory reset** (`litclock-reset.service` runs `--wipe-wifi --strict-env-wipe --poweroff --yes`): record the password, trigger it from PWA → System, power on, confirm the panel password is **DIFFERENT** — this is the litclock-dev#660 path and the only one that proves it end to end. Also confirm the PWA's confirm modal and the in-progress screen both say the password will be new and that a phone holding the old one must forget the network. Finally `sudo ./scripts/reset-setup.sh --gift-mode`, power on, confirm the panel password is **DIFFERENT**, and that `--gift-mode --keep-wifi` is **REFUSED** in both orderings. Gift prep must abort loudly ("do NOT ship this device") rather than print "done" if the file cannot be removed; an ordinary failing reset must say "Reset FAILED" instead, not tell you to stop passing the device on. After this sub-check the device has no saved WiFi, so check 4 needs a re-provision or a fresh flash regardless. **Every sub-step here now ends your SSH session** (litclock-dev#657): `--keep-wifi --poweroff` is permitted and still takes the poweroff arm, so it disables SSH too — and that sub-step explicitly rules out the panel and requires a shell, with no hotspot to fall back on because the WiFi was kept. Check 6's advice ("run anything needing SSH before a reset check") cannot rescue this one, because the check IS a before/after comparison across a reset. Either run the whole block from the console, or restore access between sub-steps by putting a blank `ssh` file in the SD card's boot partition. + - **New sub-check: a QR scan overrides a stale saved entry.** Now the interesting case, and previously untested anywhere. Every LitClock broadcasts the same SSID with a per-device password, so a phone that set up clock A meets clock B with the wrong key — the recovery path a real recipient is most likely to stumble into. Scan the panel QR on a phone holding a stale credential for `LitClock-Setup` and confirm it joins. It worked on 2026-08-15; it did not on the original measurement, so this is worth re-running per device rather than assuming. **iOS 27, 2026-09-18: worked.** Camera scan raised a "join LitClock-Setup?" prompt; Join, ~10s, captive portal auto-opened — against a credential two rotations stale. The sub-check now stands at one PASS on iOS, one PASS and one FAIL on Android — the same Android-side split the rest of this block shows. Incidentally the panel says "wait about 20 seconds"; 10 sufficed, so that copy errs the right way. + 3. **Which resets keep the setup password and which rotate it — since litclock-dev#666 the DEFAULT rotates.** The old rule (wipe AND power-off) is gone; erasing both passwords is now what a bare reset does, and `--keep-wifi` is the only way to preserve either. Run the sub-checks in this order, because each destroys state the next would need. First **`--keep-wifi`** (the "same owner, moved house" case): `sudo cat /var/lib/litclock/hotspot-password`, keep the value, `sudo ./scripts/reset-setup.sh --keep-wifi --poweroff`, power back on, `sudo cat` again and confirm it is **UNCHANGED**. Read it from the file, not the panel — with the WiFi kept, the device boots straight onto its saved network and never raises a setup network, so there is no panel password on that path. Then a **bare `sudo ./scripts/reset-setup.sh --yes`** over SSH: expect the session to DROP when the WiFi goes (that is the documented behaviour, not a fault), power-cycle, and confirm the panel password is **DIFFERENT**. Since litclock-dev#833 there is a second thing to check on this path, and it needs a console, not a power-cycle: run the same bare reset from the console (or Ethernet), then `sudo reboot` — the panel must paint "Restarting…" (a `sudo poweroff` would paint "Powered Off") instead of carrying the stale quote across. A power-cycle fires no stop edge, so it cannot test this, and the SSH session a WiFi-wipe drops leaves you no shell to type the reboot from. The plain arm re-arms `litclock-shutdown.service` inside Step 1, right after its own stop consumed the edge, so this holds even when the reset aborts later or the WiFi wipe SIGHUPs it. Then the **PWA Factory reset** (`litclock-reset.service` runs `--wipe-wifi --strict-env-wipe --poweroff --yes`): record the password, trigger it from PWA → System, power on, confirm the panel password is **DIFFERENT** — this is the litclock-dev#660 path and the only one that proves it end to end. Also confirm the PWA's confirm modal and the in-progress screen both say the password will be new and that a phone holding the old one must forget the network. Finally `sudo ./scripts/reset-setup.sh --gift-mode`, power on, confirm the panel password is **DIFFERENT**, and that `--gift-mode --keep-wifi` is **REFUSED** in both orderings. Gift prep must abort loudly ("do NOT ship this device") rather than print "done" if the file cannot be removed; an ordinary failing reset must say "Reset FAILED" instead, not tell you to stop passing the device on. After this sub-check the device has no saved WiFi, so check 4 needs a re-provision or a fresh flash regardless. **Every sub-step here now ends your SSH session** (litclock-dev#657): `--keep-wifi --poweroff` is permitted and still takes the poweroff arm, so it disables SSH too — and that sub-step explicitly rules out the panel and requires a shell, with no hotspot to fall back on because the WiFi was kept. Check 6's advice ("run anything needing SSH before a reset check") cannot rescue this one, because the check IS a before/after comparison across a reset. Either run the whole block from the console, or restore access between sub-steps by putting a blank `ssh` file in the SD card's boot partition. 4. **The SD-cloning path rotates it too** — the highest-fanout distribution channel, and the one gift mode does NOT cover. `docs/sd-card-cloning.md` is the "SD Cards for Friends & Family" flow: without this step every clone broadcasts `LitClock-Setup` with the SAME key, known to whoever made the cards and never rotated on any recipient. On a provisioned clock run `sudo ./scripts/prepare-for-cloning.sh --no-poweroff` — **the `--no-poweroff` matters**: since litclock-dev#660 the script powers the Pi off when it finishes, so without the flag the device is already down before you can inspect anything, and booting it to look is the exact action that re-mints the key. Confirm `/var/lib/litclock/hotspot-password` is gone along with any `.hotspot-password.*` staging files, and that the script aborts rather than reporting success if it cannot remove them. Separately, run it WITHOUT the flag once and confirm the Pi powers itself off. - **After the `--no-poweroff` run the clock looks bricked. It is not — do not debug it, and do not try to restart it back to life.** The panel freezes on whatever quote was last painted and port 80 refuses connections, indefinitely, with `/` still `rw`, load idle and nothing failed. The script stops `litclock-control.service` (Step 1) and `litclock.timer` (Step 4), but the stops are the transient half: it also clears `/etc/litclock/.setup-complete` and `.handoff-complete`, and `litclock-control.service` and `litclock.service` are `ConditionPathExists`-gated on those, so `systemctl start` on either exits 0 and changes nothing. Each half of the symptom has a familiar fault behind it — a frozen panel is what the litclock-dev#531 lgpio wedge looks like, a refused `:80` is what a bind failure looks like — so the pair reads as two faults at once rather than one intended state. (The pair is in fact a specific signature: `litclock-control.service` has no dependency on `litclock.service`, so neither lookalike produces both. That precision is no help at 1am.) It cost 20 minutes on 2026-08-15 (`dev-20260815-b0c0590`), hours after the fact, when the terminal holding the closing banner was long gone. **Shut the Pi down** (`sudo shutdown -h now`) — the card is a clone master and there is nothing left to test on it. - **The default (no-flag) run reaches the same state only if its power-off fails.** It normally halts, so there is no panel to misread; on the `poweroff || …` recovery path it prints "Power-off FAILED", exits 1, and leaves the identical frozen panel. Both arms of the banner say so. - **Answering `y` to the WiFi prompt over SSH-on-WiFi kills the run.** Step 3 deletes the profile you are connected over, the session drops, and `SIGHUP` takes the script with it — before Step 8 removes the hotspot key, so the card is NOT prepared even though nothing said otherwise. Pre-existing, not introduced by any litclock-dev#659 change. The delete loop walks **every** NM connection, wired included, so ethernet is no refuge — run the `y` path from a local console, and if a session ever drops mid-run, re-run the script rather than assuming it finished. + - **The bash history is a directory afterwards, and that is the lock, not damage (litclock-dev#834).** `sudo ls -ld /home/pi/.bash_history /root/.bash_history` must show two EMPTY directories after the `--no-poweroff` run. `history -c` reaches only the script's own shell; the console or SSH shell you ran it from writes its history back on exit, during the power-off — on the bench the file was back eight seconds after "Clearing bash history... done", holding the operator's last command. A directory at the path fails that write with EISDIR for root and pi alike (a mode-0 file does not: root bypasses it, and `history -w` renames over it). Exit your shell and confirm both paths are STILL directories, then boot a CLONE (never the master) and confirm `first-boot.sh` removed both (`ls -ld` shows nothing, and the recipient's first `exit` creates a normal file). Run the script as the only open session regardless — but note the rule cannot cover the shell you run it FROM, which is why the lock exists at all. **The lock is now a hard gate**: stage a failure (`sudo mkdir /home/pi/.bash_history && sudo touch /home/pi/.bash_history/x` before a run, which is what an earlier aborted run plus a stray file looks like) and confirm the step prints `FAILED` + `Do NOT clone this card` and exits 1 rather than a yellow note. Remove the stray file and re-run; it must complete. + - **A failed env.sh wipe stops BEFORE the WiFi question (litclock-dev#839).** Stage it by holding the sidecar lock from a second shell — `sudo flock /home/pi/litclock/env.sh.lock sleep 120` — then run the script. It must print `Clearing configuration (env.sh)... FAILED` and the `Do NOT clone this card` banner and exit 1 **without** asking `Clear saved WiFi networks?`; afterwards `/var/lib/litclock/hotspot-password` must still exist, `nmcli connection show` must still list your network, and `env.sh` must be byte-identical. Pre-fix the same run printed "Clearing setup-hotspot password... done" BEFORE the red env.sh line, having already wiped everything else. **The device is NOT in the looks-bricked state the `--no-poweroff` bullet below describes**, and that is the point of the fix: the setup-state markers are removed only after the wipe succeeds, so an aborted run leaves a fully provisioned clock that still paints quotes and still boots normally. What IS true until the next boot: Step 1 already stopped `litclock-control.service` (the PWA) and the updater, so port 80 refuses connections in the meantime — reboot, or `sudo systemctl start litclock-control.service`, restores it. Release the lock and re-run; it must complete normally. Also confirm the aborted run did NOT leave `/var/lib/litclock/clone-prep-unfinished` behind: it changed nothing, so the next run must not open with the "previous run did not finish" warning. - **Do not reboot it to "check" it — that contaminates the master.** `first-boot.sh` runs again on that card, and whether the boot also re-mints the setup-WiFi key depends on which branch it takes. `is_wifi_connected()` is literally `ip addr show wlan0 | grep -q 'inet '`, and `litclock-firstboot.service` is ordered `After=NetworkManager.service`, **not** `network-online.target` — so the branch turns on whether `wlan0` happens to hold an address at that instant, not on whether a profile is saved. With **no** address it raises the setup hotspot, and `create_hotspot()` mints a fresh permanent key (litclock-dev#660) that every clone taken afterwards would share. With an address it completes setup inline — no page, no hotspot (litclock-dev#647), **no new key** — and runs straight through to the handoff splash. The script's own closing banner states the key hazard unconditionally, which is the safe direction; do not read the second branch as permission to boot the master. 5. **The support bundle no longer carries the password.** litclock-dev#620 turns a transient leak into a durable one, so the redaction fix ships with it. After provisioning, note the password, then PWA → Diagnostics → copy the support payload (and the deep logs) and grep for that exact string. It must not appear. Pre-fix, a real `sudo` audit line came back from `redact_text()` with the password intact. 6. **A factory reset now turns SSH OFF on dev images too, and that is new.** litclock-dev#657 removed the @@ -159,6 +180,191 @@ them together, but the payoff is only observable on hardware. That is why the check is "read the journal", and why it is worth three restarts. +### Shutdown splash vs. boot splash (litclock-dev#856) + +The only part of litclock-dev#856 unit files cannot prove. `litclock-shutdown.service` is +now `Before=litclock-splash.service`, which in the stop direction means its +`ExecStop` (`shutdown-splash.sh`) waits for the boot splash's stop job — and +that stop is what releases GPIO17/SPI, because it SIGTERMs a control group +rather than running an `ExecStop`. Run on a device you can re-flash. + +**VERIFIED END TO END on the bench 2026-09-17 23:43–23:49 CDT** +(bench device, address and build deliberately not recorded here — this file +is public), control-fails-then-fixed-passes, +after this recipe had failed to test anything twice (litclock-dev#860). The numbers below +are that run; a follow-up pass on 2026-09-18 07:42–07:46 added the panel +corroboration from the journal and found litclock-dev#862 doing it (last two bullets). +Three things the earlier versions got wrong are corrected here — +**read all three before touching the device**, because each one on its own is +enough to make the check pass on broken code. + +**Where the delay has to go: inside the painter, after `epd.init()`.** The +paint is a synchronous `timeout 20 "$PYTHON" src/eink_display.py status ...` in +`scripts/boot-splash.sh`. A `sleep` placed *before* that line runs when Python +has not yet acquired GPIO; placed *after* it, Python has already exited and +released it. **Neither creates contention, and neither can fail** — a delay +around the paint tests nothing. GPIO17/SPI is held only between `epd.init()` +and `epd.sleep()` inside `display_image()` (`src/eink_display.py`, ~line 476). + +**Correction 1 — a plain `sleep` hold cannot reproduce the race, in either +configuration.** Measured: with both directives REVERTED and a 20s +`time.sleep()` hold, the shutdown splash **painted cleanly**. The splash's +python dies on SIGTERM in under a second (`time.sleep` is an interruptible +point, and SIGTERM's default action needs no Python at all), while +`shutdown-splash.sh` spends ~1.5s in Python startup and image generation before +it reaches `epd.init()`. It loses the race by default. **The hold must survive +SIGTERM** — which is also the honest model of the hazard, since litclock-dev#856's own +unit comment names "a python wedged in uninterruptible kernel I/O on the SPI +transfer" as the residual. Use this, not the old one-liner: + +```python +# QA HARNESS (litclock-dev#860) — remove after testing. +import os as _os, time as _t, signal as _sig +_h = float(_os.environ.get("LITCLOCK_QA_PANEL_HOLD_S") or "0") # `or`: an EMPTY value is this repo's unset idiom +if _h: + _sig.signal(_sig.SIGTERM, lambda *a: logging.warning("QA hold: SIGTERM ignored, GPIO still held")) + logging.warning("QA hold %ss (post-init, SIGTERM-resistant)", _h) + _end = _t.monotonic() + _h + while _t.monotonic() < _end: + _t.sleep(0.2) + logging.warning("QA hold over") +``` + +Off by default, so a forgotten line does not change a normal boot. Two more +device edits, all three reverted by the next `update.sh` — undo them yourself +with `git checkout -- src/eink_display.py scripts/boot-splash.sh +scripts/shutdown-splash.sh` (the third file only if you also add the +journal-corroboration probe in the second-to-last bullet): + +1. The block above, in `display_image()`, immediately after `epd.init()`. +2. `sudo systemctl edit litclock-splash.service` → `[Service]` → + `Environment=LITCLOCK_QA_PANEL_HOLD_S=20`. Confirm it lands: + `systemctl show litclock-splash.service -p Environment`. +3. In `scripts/boot-splash.sh`, raise the wrapper to `timeout 40` — at + `timeout 20` the hold is killed before it does anything. + +**Correction 2 — use `systemctl restart litclock-splash.service`, and fire it +clear of the minute tick.** Do not race a boot window; the unit has no +`Condition*=` and restarts by hand cleanly into the same start state holding +GPIO. `restart`, not `start`: it is a `RemainAfterExit=yes` oneshot, so `start` +on an already-active unit is a silent no-op. **And the painter it spawns loses +the panel to `litclock.service` if it lands on the `:56` tick** — measured at +23:46:58, `boot-splash.sh` exited in 1s with `Could not initialize display: +'GPIO busy'` and the run proved nothing. This is almost certainly what happened +on 2026-09-17 (litclock-dev#860): an ExecStart that finishes in ~1s with the painter +already gone is this, not a misplaced hold. Fire between `:10` and `:45` of a +minute, and **guard the reboot on the hold actually being live**: + +```bash +S=$(date +%S); sleep $(( (75 - 10#$S) % 60 )) # land at ~:15 +sudo systemctl restart --no-block litclock-splash.service +sleep 12 +N=$(pgrep -fc "eink_display.py statu[s]") # bracket: do not self-match +if [ "$N" -ge 1 ]; then sudo systemctl reboot; else echo "ABORT: hold not live"; fi +``` + +(No `exit` in that block on purpose — it is meant to be pasted into an +interactive SSH session, and an `exit` on the guard path closes the session you +are about to need.) + +**Budget — use 20s, and do NOT exceed it.** The hold is squeezed from both +ends. Above it: `~7–11s paint + hold` must stay inside `TimeoutStartSec=45`, +and the paint is 7s warm but **10s cold** (measured on a first boot), so a 30s +hold leaves only ~4s of margin — trip it and systemd kills `ExecStart`, the unit +fails, and the harness quietly becomes a start-timeout test instead. That is the +same "cannot fail" class this whole section exists to prevent, so take the +margin: **20s**. Below it: the hold must outlast `TimeoutStopSec=10s` measured +from the reboot at `t+12`, i.e. it must still be running at `t+22`; a 20s hold +started at `t+7..11` runs to `t+27..31`. 8s would not. Do not raise +`TimeoutStartSec` to buy room — 45s is the bound the whole thing lives in. +(The 2026-09-17/18 runs used 30s and did not trip it; 20s is the same test with +margin.) + +- **Run the CONTROL first, on the unfixed units, or you have not tested + anything.** Revert both directives on the device — drop + `litclock-splash.service` from `Before=` in + `/etc/systemd/system/litclock-shutdown.service` and the `TimeoutStopSec=10s` + from `/etc/systemd/system/litclock-splash.service` — `daemon-reload`, then run + the block above. **Correction 3 — the expected failure string is NOT + `KeyError: PinInfo(... 'GPIO17' ...)`.** That shape was inferred in litclock-dev#856 + ("no hardware reproduction yet") and never observed; `get_display()` catches + the lgpio error at `epd7in5.EPD()` construction, so what + `journalctl -b -1 -u litclock-shutdown` actually carries is: + ``` + WARNING: Could not initialize display: 'GPIO busy' + ERROR: No display available + ``` + A tester grepping for `KeyError` finds nothing and calls the control clean — + a third way this check could not fail. **Expected FAILURE, measured:** the + shutdown unit's `Stopping` and the splash's stop run CONCURRENTLY + (both 23:45:15), `'GPIO busy'` 1.5s later at 23:45:16.7, no shutdown paint, + and the panel powers off still showing **"Starting…"**. `shutdown-splash.sh` + swallows it on its `|| true` tail, so the journal is the only place it is + visible. If the control does not fail, fix the harness before reading + anything into the fixed run. +- **Then the fixed units: the payoff.** Restore both directives, + `daemon-reload`, re-run the block. **Measured PASS:** SIGKILL at exactly + `TimeoutStopSec` after the reboot (23:48:55.3 → 23:49:05), splash `Stopped`, + and only THEN `Stopping litclock-shutdown.service` — the ordering, in + reverse, visible as a timestamp gap the control does not have. ExecStop then + paints with no error and finishes 23:49:12. The e-ink must end on the + **reboot splash** ("To sleep, perchance to dream." and friends) while the Pi + is off. +- **The bound.** Time that same reboot. Measured: **~17s** with + `TimeoutStopSec=10s`, versus **26s** in the control (and up to 90s + 90s + unbounded, if the hold outlives `DefaultTimeoutStopSec`). A + reboot-during-splash that suddenly takes over a minute means the directive is + gone — and a minute is long enough that a real owner pulls the power and gets + no splash at all, which is the whole reason it is capped. +- **The start-direction half.** With the hold running, from a second SSH + session: `systemctl is-active litclock-shutdown.service` must read `active` + while `systemctl is-active litclock-splash.service` still reads `activating`. + **PASS 2026-09-17.** If the shutdown unit reads `activating` or `inactive`, + the `Before=` edge is not being honoured and a reboot in that window paints + **nothing** — a `Type=oneshot` stopped out of its start state never reaches + `ExecStop`. +- **The ordinary case still works.** Undo all three edits and the drop-in, + reboot normally, confirm the reboot splash paints as always. This is the + boot-direction regression check: a cycle would have had systemd drop an edge + at random. `systemd-analyze verify /etc/systemd/system/litclock-*.service` on + the device must report no ordering cycle. +- **What this run did NOT show, and it matters for how you read litclock-dev#856.** The + ordinary reboot-during-splash does not race at all (Correction 1): the fix + only changes the outcome once the splash needs longer than ~1.5s to die. So + the window litclock-dev#856 describes is real but narrower than the issue implies, and + the directive's day-to-day value is the 10s bound as much as the ordering. + Both are still right; neither is load-bearing on a healthy panel. +- **The fixed run paints the WRONG splash — litclock-dev#862, found 2026-09-18.** The + ordering works, and the paint it enables resolves `action=poweroff` on a + `sudo reboot`: the panel gets a final-state farewell ("So we beat on, boats + against the current") instead of "Restarting…". `shutdown-splash.sh` tier 3 + is `systemctl list-jobs | grep -q reboot.target`, and by the time the DELAYED + `ExecStop` runs the system bus is gone — captured at resolution time, the + command returns `Failed to connect to bus: Connection refused`, so the grep + fails and tier 4 falls through to poweroff. Deterministic, n=2; the same + reboot with the splash idle resolves `action=reboot` correctly. **Do not read + this as a reason to revert litclock-dev#856** — pre-fix, that window painted NOTHING + and carried "Starting…" across the power-off. When QAing this section, expect + the wrong variant until litclock-dev#862 lands, and judge litclock-dev#856 on the ordering and + the 10s bound, which are the two things it claims. +- **Corroborate the panel from the journal, not the glass.** The variant is not + logged today (litclock-dev#861), so the 2026-09-18 pass added two temporary `echo`s to + `scripts/shutdown-splash.sh` — one for `$SHUTDOWN_ACTION` before the `case`, + one for `${SPLASH_ARGS[*]}` before the paint — and dumped `systemctl + list-jobs` to a NON-tmpfs path (`/run/litclock` is tmpfs and does not survive + the reboot you are about to take). That turns a 10s eyes-on window into a + `journalctl -b -1` grep, and is how litclock-dev#862 was found at all. +- **Observe, do not fix here (Codex, out of scope for litclock-dev#856).** A SIGTERM + landing inside `display_image()` skips `epd.sleep()`, so the panel is left + out of its sleep state; whether it recovers cleanly from an interrupted SPI + transfer versus an interrupted BUSY waveform is unproven either way. This + predates litclock-dev#856 and the control run above deliberately provokes it. Note + what the panel does — ghosting, a partial frame, a refusal to take the next + paint — rather than treating it as a litclock-dev#856 regression. On the 2026-09-17 + run the next scheduled paint succeeded normally (`picked_at_age_s` 0.27s + after the following tick), so at least one SIGKILL mid-hold left no lasting + damage. + ### OTA smoke gate (litclock-dev#763, litclock-dev#773) Not in the checklist above because it is not a first-boot flow — but it is the @@ -187,27 +393,56 @@ sudo systemctl start litclock-update.service && journalctl -fu litclock-update activates a second and an owner selects it, so this needs a fleet with a second active language to be a real test rather than a shape check. - **A missing venv interpreter FAILS the gate, it does not skip it (litclock-dev#773 - item 1).** `sudo mv /home/pi/litclock/venv/bin/python3{,.bak}` then run an - update. Expect `Smoke test SKIPPED: ... is missing or not executable` followed - by `Nothing was verified — treating this as a smoke FAILURE`, and a revert. - Before this, the whole gate was skipped and Phase 5/7 reported SUCCESS. - Restore the interpreter afterwards. -- **A truncated catalog FAILS the gate (litclock-dev#773 item 2).** Truncate - `languages/en/strings.json` to a handful of keys — keeping - `status.relative.just_now`, `boot.splash.starting.title` and - `firstboot.splash.setup_incomplete.title`, which are the three the VALUE - probes check — then run an update. The value probes will pass and the COUNT - probe must fail with `Catalog smoke failed: catalog-count returned 'N' - (want >= 400)`. That + item 1) — but do NOT try to stage it with `mv`. That recipe was wrong + (litclock-dev#864).** `sudo mv /home/pi/litclock/venv/bin/python3{,.bak}` and + run an update and you get **`Update Complete`**, not a revert: the Phase-4 + venv guard (`update.sh:1325`, `! "$PYTHON" -c "import PIL, requests"`) fails + on a *missing* interpreter exactly as on a broken one, rebuilds the venv on + the spot, and the gate then runs against a healthy interpreter. Measured + 2026-09-18. Every other hand-staged variant lands somewhere else too: if you + also stop the rebuild from succeeding, `NEED_PIP` is already true and pip — + whose shebang points at the interpreter you removed — fails first, so the + **pip-failure arm** reverts, not this one. The `update.sh:1636` arm is + reachable only when Phase 4 gets far enough to skip or finish pip and still + leaves no interpreter (a pip run that breaks its own venv, or an interrupted + Phase 4) — which you cannot stage from a shell in one line. + **So do not hardware-test this one.** It is covered where it belongs, by + `tests/test_update_sh.py` (the executed arm assertion at ~line 2371 pins + `smoke_rc=1` AND `smoke_no_interpreter=1` together). Spend the bench time on + the catalog case below, which is both reachable and the one that shipped + green with 435 strings gone. +- **A truncated catalog FAILS the gate (litclock-dev#773 item 2) — and the truncation + must be in the TARGET tree, not on the device.** Phase 2 does + `git reset --hard ` before the gate runs, so a file you truncate on + the device is restored before anything looks at it; the old wording here said + "truncate … then run an update", which tests nothing (litclock-dev#864). + Publish a deliberately bad release instead — see the local-remote harness in + a local bare remote — with `languages/en/strings.json` cut to a handful + of keys, **keeping** `status.relative.just_now`, `boot.splash.starting.title` + and `firstboot.splash.setup_incomplete.title`, the three the VALUE probes + read. Build it on top of the release the device is already running, because + releases are cumulative and the gate that runs is the *target's* gate, not the + device's. The value probes must pass and the COUNT probe must fail with + `Catalog smoke failed: catalog-count returned 'N' (want >= 400)`. That combination is the whole point: before the count probe, exactly this bundle - passed green with 435 strings gone. + passed green with 435 strings gone. Verified 2026-09-18 (`returned '8'`). - **A revert leaves a working clock, not a brick.** After any of the failing cases above, confirm the panel is still painting quotes on the old SHA and the PWA Updates view shows the terminal banner **"Update failed verification — rolled back. Your clock is running normally."** (state `failed_reverted`). - Also note the device - re-runs the full pip install on the next tick (the revert deletes - `HASH_FILE`, which sets `NEED_PIP`); that is expected, not a second fault. + **Then run the tick again — this is the half that used to be missing.** Since + litclock-dev#865 the revert records the failing SHA in + `/var/lib/litclock/blocked-sha`, so the second tick must log + `Latest Release SHA … is blocked (a previous run reverted from it) — skipping + update`, leave `HEAD` untouched, finish `inactive` rather than `failed`, and + report `update_state: complete` (a deliberate no-op is not a failure). Before + that fix it re-fetched, re-applied, re-failed and reverted again — every week, + forever, with the clock stopped and a full pip install each time. Then publish + a NEWER healthy release and confirm it installs and CLEARS the block: a device + that can never be updated again would be a worse bug than the loop. All three + ticks verified on hardware 2026-09-18. The pip hash stays deleted across the + blocked ticks (the revert removes `HASH_FILE`) so the recovering release + re-runs pip once; that is expected, not a second fault. - **`catalog-count` is stdout-compared, so check its contract directly.** On the device: `sudo -u pi /home/pi/litclock/venv/bin/python3 src/eink_display.py catalog-count` must print an integer and **exit 0**. Break the bundle diff --git a/TODOS.md b/TODOS.md index b53308c..8ad8e7c 100644 --- a/TODOS.md +++ b/TODOS.md @@ -24,7 +24,7 @@ ### LKG auto-revert (bootcheck.service) — follow-up to litclock-dev#209 — SHIPPED -**Filed as litclock-dev#493; merged to master 2026-07-10 (bundles into v0.217.0); hardware-QA-validated on the test Pi.** `litclock-bootcheck.service` consumes the litclock-dev#209 `lkg-sha` writer: per-boot "did the clock paint a frame since boot?" via the tmpfs render heartbeat (network-independent). Fail 1-2 auto-reboot to retry; fail 3 pins the last-known-good SHA and routes recovery back through `update.sh` rollback mode (full install — git + submodules + venv + units + sudoers + smoke — not a code-only `git reset`), then a terminal `bootcheck-gave-up` marker + re-flash splash bounds the sequence to ~4 reboots. Runs as `pi` on the existing `020_litclock-control` grants, so it works after litclock-dev#387 drops `010`. The end-to-end auto-reboot self-heal was proven unattended on real hardware (2 reboots → rollback → quotes returned). Closes the litclock-dev#82 brick-recovery prerequisite. See https://github.com/kapoorankush/litclock/issues/209. +**Filed as litclock-dev#493; merged to master 2026-07-10 (bundles into v0.217.0); hardware-QA-validated on the test Pi.** `litclock-bootcheck.service` consumes the litclock-dev#209 `lkg-sha` writer: per-boot "did the clock paint a frame since boot?" via the tmpfs render heartbeat (network-independent). Fail 1-2 auto-reboot to retry; fail 3 pins the last-known-good SHA and routes recovery back through `update.sh` rollback mode (full install — git + submodules + venv + units + sudoers + smoke — not a code-only `git reset`), then a terminal `bootcheck-gave-up` marker + re-flash splash bounds the sequence to ~4 reboots. Runs as `pi` on the existing `020_litclock-control` grants, so it works after litclock-dev#387 drops `010`. (Written when litclock-dev#387 planned that drop; it was **reversed 2026-07-12** and `010` is kept. Note the original claim was always narrower than it reads: bootcheck's own `sudo mkdir -p "$STATE_DIR"` is not in `020` and degrades via a fallback, and the `update.sh` rollback it routes into copies unit files on the blanket grant.) The end-to-end auto-reboot self-heal was proven unattended on real hardware (2 reboots → rollback → quotes returned). Closes the litclock-dev#82 brick-recovery prerequisite. See https://github.com/kapoorankush/litclock/issues/209. ### Text-fit row triage (P3 under litclock-dev#211) — shipped diff --git a/docs/plans/lkg-bootcheck-plan.md b/docs/plans/lkg-bootcheck-plan.md index 2348e5c..6fcfa60 100644 --- a/docs/plans/lkg-bootcheck-plan.md +++ b/docs/plans/lkg-bootcheck-plan.md @@ -50,7 +50,7 @@ Per the TODO wording ("reverts … after 3 consecutive failed boots, then reboot - When it arms a fresh `lkg-sha` (all gates pass, write succeeds): also clear `rollback-sha` (new code is now the known-good; old rollback target retired). ### Reboot authorization -`020_litclock-control` already authorizes `/usr/bin/systemctl reboot`. bootcheck runs as pi and uses the same scoped grant — no new sudoers entry, works after litclock-dev#387 drops `010`. Verify the exact authorized form (`reboot` vs `reboot --no-block`). +`020_litclock-control` already authorizes `/usr/bin/systemctl reboot`. bootcheck runs as pi and uses the same scoped grant — no new sudoers entry, and no dependence on the blanket `010` grant. (As planned, litclock-dev#387 was going to drop `010`; that was reversed 2026-07-12 and `010` is kept. The scoped-grant property is what matters and is unaffected.) Verify the exact authorized form (`reboot` vs `reboot --no-block`). ## Reboot-loop bound (safety) Worst case: bad update → 3 natural power-cycles → revert + 1 reboot → reverted code also bad → 3 more cycles → give-up splash, no further reboots. Bounded. The give-up splash tells the user to reflash (last resort). diff --git a/docs/sd-card-cloning.md b/docs/sd-card-cloning.md index c1b9bff..051d377 100644 --- a/docs/sd-card-cloning.md +++ b/docs/sd-card-cloning.md @@ -10,6 +10,20 @@ Complete the full installation on one Pi by flashing the released image (see [Fl ## 2. Prepare for Cloning +Run this as the **only open session** on the Pi: log out of every other shell +first (SSH sessions, tmux panes, a `sudo -i` root shell). The script clears the +bash history, but any interactive shell still holding history when the Pi +powers off writes it back to disk on exit, and that is what gets imaged onto +every card. + +That rule cannot cover **the shell you run it from** — it is alive for the +whole run and writes its own history as it exits. The script covers that one +itself, by locking both history files against write-back until the clone's +first boot, and it now stops rather than continue if it cannot apply the lock. + +Running over SSH is otherwise supported; only the WiFi question below requires +a local console. + ```bash sudo ./scripts/prepare-for-cloning.sh ``` @@ -21,12 +35,17 @@ This script will (and then power the Pi off): - Optionally clear WiFi credentials - Re-enable the first-boot setup service - Clear logs and caches +- Clear the bash history, and lock both history files so no shell still open at power-off — including the one you are typing in — can write it back (the clone's first boot unlocks them). The run stops if either path cannot be cleared or locked. - Clear the SSL certificates - Delete the persisted setup-hotspot password, so no clone carries your key - Disable SSH, so clones ship in the same posture as a fresh flash (to get back into a clone: put a blank file named `ssh` in the SD card's boot partition) If any of those steps cannot finish, the script says so in red and stops rather -than reporting success. Do not clone a card it refused. The one refusal you may +than reporting success. Do not clone a card it refused. If the API-key and +location wipe itself fails, it stops right there, before asking about WiFi, and +nothing on the card has been changed at all — the clock still works and still +boots normally. Fix the cause (usually the control PWA still holding +`env.sh.lock`) and run it again; the PWA comes back on the next boot. The one refusal you may not SEE is the final SSH-disable check when running over the network (output is cut before it, deliberately) — its tell is a Pi that has not powered itself off within a minute of your session dropping; do not image that card either, and diff --git a/image-gen/litclock_annotated.csv b/image-gen/litclock_annotated.csv index ab341da..b63e1ce 100644 --- a/image-gen/litclock_annotated.csv +++ b/image-gen/litclock_annotated.csv @@ -1573,7 +1573,7 @@ 08:29|8:29|Mr. Trent himself followed a rigid routine. Each day, he arose at 7 a.m., breakfasted at 7:30, and departed for work at 8:10, arriving at 8:29.|The Great Train Robbery|Michael Crichton|NO 08:30|half-past eight|"""What nonsense!"" thought Vronsky, and glanced at his watch. It was half-past eight already."|Anna Karenina|Leo Tolstoy|NO 08:30|8:30 A.M.|At 8:30 A.M. Filomina walked into the Ale House white, round, and heaving from the extra hundred and forty-six pounds she carried on her small frame.|Pomegranate Soup|Marsha Mehran|NO -08:30|eight thirty|"At eight thirty, I called and said, ""I've been thinking I might get a boob job, just take them clean off. What do you think? Could I pull off the flat-chested look?"""|My Year of Rest and Relaxation|Ottessa Moshfegh|NO +08:30|eight thirty|"At eight thirty, I called and said, ""I've been thinking I might get a boob job, just take them clean off. What do you think? Could I pull off the flat-chested look?"""|My Year of Rest and Relaxation|Ottessa Moshfegh|YES 08:30|half past eight|At half past eight Millicent Hammitt barged in, without a preliminary knock, to say goodbye.|The Black Tower|P.D. James|NO 08:30|half past eight|At half past eight, Mr. Dursley picked up his briefcase, pecked Mrs. Dursley on the cheek, and tried to kiss Dudley good-bye but missed, because Dudley was now having a tantrum and throwing his cereal at the walls.|Harry Potter and the Philosopher's Stone|JK Rowling|NO 08:30|Eight-thirty|Eight-thirty the ward door opens and two technicians trot in, smelling like grape wine; technicians always move at a fast walk or a trot because they're always leaning so far forward they have to move fast to keep standing.|One Flew Over the Cuckoo's Nest|Ken Kesey|NO @@ -1980,7 +1980,7 @@ 10:00|Ten|. . .the clock ticked louder and louder until there was a terrific explosion right in her ear. Orlando leapt as if she had been violently struck on the head. Ten times she was struck|Orlando|Virginia Woolf|NO 10:00|ten o'clock|A new life opening up for me at the Villa Borghese. Only ten o'clock and we have already had breakfast and been out for a walk.|Tropic of Cancer|Henry Miller|NO 10:00|10 am|According to military records no US bombers or any other kind of aircraft were flying over that region at the time, that is around 10 am on November 7,1944.|Kafka on the shore|Haruki Murakami|NO -10:00|ten o'clock|Anyway, when he looked at his watch again it was ten o'clock. At ten o'clock she was lying on the divan with her boobies in her hands. That's the way he gives it to me-in driblets.|Tropic of Cancer|Henry Miller|NO +10:00|ten o'clock|Anyway, when he looked at his watch again it was ten o'clock. At ten o'clock she was lying on the divan with her boobies in her hands. That's the way he gives it to me-in driblets.|Tropic of Cancer|Henry Miller|YES 10:00|ten o'clock|At about ten o'clock in the morning the sun threw a bright dust-laden bar through one of the side windows, and in and out of the beam flies shot like rushing stars.|Of Mice And Men|John Steinbeck|NO 10:00|ten o'clock|At about ten o'clock the first tiny snowflakes came loitering down and settled on Jill's arm. Ten minutes later they were falling quite thickly. In twenty minutes the ground was noticeably white.|The Silver Chair|C.S. Lewis|NO 10:00|ten|At around ten every morning, my wife and I would take a cooler down to the beach. We'd lather up with sunblock, then sprawl out on mats on the sand. I'd listen to the Stones or Marvin Gaye on a Walkman, while my wife plowed through a paperback of Gone With the Wind. She claimed that she'd learned a lot about life from that book. I'd never read it, so I had no idea what she meant.|Blind Willow, Sleeping Woman|Haruki Murakami|NO @@ -4127,7 +4127,7 @@ 20:30|half-past eight|"""What nonsense!"" thought Vronsky, and glanced at his watch. It was half-past eight already."|Anna Karenina|Leo Tolstoy|NO 20:30|half-past eight|... she said, 'at exactly half-past eight you may be watching the middle upper window of the top floor. If I decide to forgive I will hang out of that window a white silk scarf. You will know by that that all is as was before, and you may come to me...|The Four Million|O. Henry|NO 20:30|half-past eight|"Alix took up a piece of needlework and began to stitch. Gerald read a few pages of his book. Then he glanced up at the clock and tossed the book away. ""Half-past eight. Time to go down to the cellar and start work."""|The Listerdale Mystery|Agatha Christie|NO -20:30|eight thirty|"At eight thirty, I called and said, ""I've been thinking I might get a boob job, just take them clean off. What do you think? Could I pull off the flat-chested look?"""|My Year of Rest and Relaxation|Ottessa Moshfegh|NO +20:30|eight thirty|"At eight thirty, I called and said, ""I've been thinking I might get a boob job, just take them clean off. What do you think? Could I pull off the flat-chested look?"""|My Year of Rest and Relaxation|Ottessa Moshfegh|YES 20:30|half past eight|At half past eight Millicent Hammitt barged in, without a preliminary knock, to say goodbye.|The Black Tower|P.D. James|NO 20:30|eight-thirty|By eight-thirty, with dusk growing too thick to be much different from night, the five searchers had grown to a dozen.|The Tommyknockers|Stephen King|NO 20:30|Eight-thirty|Eight-thirty the ward door opens and two technicians trot in, smelling like grape wine; technicians always move at a fast walk or a trot because they're always leaning so far forward they have to move fast to keep standing.|One Flew Over the Cuckoo's Nest|Ken Kesey|NO @@ -4404,7 +4404,7 @@ 22:00|ten|"""A considerable crime is in contemplation. I have every reason to believe that we shall be in time to stop it. But to-day being Saturday rather complicates matters. I shall want your help to-night."" ""At what time?"" ""Ten will be early enough."" ""I shall be at Baker Street at ten."""|The Adventures of Sherlock Holmes|Sir Arthur Conan Doyle|NO 22:00|ten o'clock|Alone at the too-small child's desk in her room, Eva finished the last line of the most pointless assignment ever, and though she wanted to start on the evening's true mission immediately, she kept the box of churro bites closed under her bed. Her parents went to sleep at ten o'clock on weeknights. Only three and a half more hours to wait out.|Kitchens of the Great Midwest|J. Ryan Stradal|NO 22:00|ten o'clock in the evening|And now sometimes, in the very midst of things, sometimes when I feel that I am absolutely free of it all, suddenly, in rounding a corner perhaps, there will bob up a little square, a few trees and a bench, a deserted spot where we stood and had it out, where we drove each other crazy with bitter, jealous scenes. Always some deserted spot, like the Place de l'Estrapade, for example, or those dingy, mournful streets off the Mosque or along that open tomb of an Avenue de Breteuil which at ten o'clock in the evening is so silent, so dead, that it makes one think of murder or suicide, anything that might create a vestige of human drama.|Tropic of Cancer|Henry Miller|NO -22:00|ten o'clock|Anyway, when he looked at his watch again it was ten o'clock. At ten o'clock she was lying on the divan with her boobies in her hands. That's the way he gives it to me-in driblets.|Tropic of Cancer|Henry Miller|NO +22:00|ten o'clock|Anyway, when he looked at his watch again it was ten o'clock. At ten o'clock she was lying on the divan with her boobies in her hands. That's the way he gives it to me-in driblets.|Tropic of Cancer|Henry Miller|YES 22:00|ten o'clock|At night, a quiet so vast it seemed it seemed almost to reverberate. Few cars passed on Route 114, and after ten o'clock or so there were none at all. After ten, the part of the world where I had come to rest belonged only to the loons and the wind in the fir trees.|11/22/63|Stephen King|NO 22:00|ten o'clock|At ten o'clock at night I hadn't returned yet, and then I went to bed; at about one o'clock in the morning they woke my steps in the room|The Seven Madmen|Roberto Arlt|NO 22:00|ten o'clock|At ten o'clock she went into her double cell, where the woman she shared with was already asleep, and there was the reassuring clang of the door and click of the lock. It felt safe to be caged in, now that she knew she had this other person inside her who was capable of escapades and contortions she'd never known about before.|The Heart Goes Last|Margaret Atwood|NO diff --git a/pi-gen/stage3/01-setup-app/00-run.sh b/pi-gen/stage3/01-setup-app/00-run.sh index b3fc09a..70db13d 100755 --- a/pi-gen/stage3/01-setup-app/00-run.sh +++ b/pi-gen/stage3/01-setup-app/00-run.sh @@ -74,13 +74,66 @@ rm -f /tmp/requirements-pigen.txt # fallback is impossible; a wrong render is not. The failure is logged loudly # so a build that quietly lost the runtime tier shows up in the CI log rather # than being discovered on a device. +# +# BOUNDED (litclock-dev#840). Unbounded, a validator that hangs under qemu is +# killed by the job's 180-minute limit, which fails the whole image build +# instead of degrading to "no marker". Measured: 179s in the build chroot on +# the 2026-09-13 master build (176s of it inside the check; the bench Pi Zero +# 2W measures 182s natively). The bound is ~5x that -- qemu on a shared runner +# is the least stable clock in this repo -- and a twelfth of the job budget, +# so a hung check costs the build minutes, not the build. update.sh has its +# own bound (VALIDATOR_TIMEOUT_S) sized against a systemd budget; this one is +# not it and must not be copied there. +# +# `-k`: coreutils timeout sends ONE SIGTERM and then waits for the child +# forever. The case this bound exists for -- a validator wedged under +# qemu-user -- is exactly the one where SIGTERM is not acted on promptly (qemu +# delivers host signals at a translation-block exit or syscall return, and a +# wedged guest reaches neither), so without a kill the build still ran to the +# 180-minute job kill (PR litclock-dev#852 review, reproduced with a TERM-ignoring child). +# The grace is generous: the validator has no cleanup worth waiting for. +# +# The FAIL and timeout arms also emit a GitHub Actions `::warning::` +# annotation, so a build that lost the runtime tier shows on the run summary +# and the PR checks rather than as one echo in a 6000-line log. The annotation +# is a workflow command and only works on stdout at line start -- keep it a +# bare echo, unprefixed. +PIGEN_VALIDATOR_TIMEOUT_S=900 +PIGEN_VALIDATOR_KILL_GRACE_S=30 echo "Validating GD-exact measurement (litclock-dev#531)..." -if ./venv/bin/python3 tools/validate_measurement.py check --stamp; then +if timeout -k "$PIGEN_VALIDATOR_KILL_GRACE_S" "$PIGEN_VALIDATOR_TIMEOUT_S" ./venv/bin/python3 tools/validate_measurement.py check --stamp; then echo " runtime-render validation PASSED — marker stamped into the image" else - echo " WARNING: runtime-render validation FAILED in the build chroot." + # $? at the top of an else arm is the condition's status (bash; the + # executed test in tests/test_runtime_render_autostamp.py pins it). No + # pipe: `if timeout ... | sed ...; then` would test sed's exit, not the + # validator's -- the trap update.sh already fell into once. + validate_rc=$? + # 124 AND 137. coreutils reports 124 when the child dies to its SIGTERM, + # but 128+9 = 137 when `-k` has to escalate to SIGKILL -- which is the + # qemu-wedge case this bound exists for, so testing 124 alone sent exactly + # that run down the FAIL arm, mislabelled "exited 137" and skipping the + # marker removal below (found by the executed TERM-ignoring test, not by + # reading). 137 from an OOM kill reads the same way and wants the same + # treatment: the check did not complete, so there is no proof. + if [ "$validate_rc" -eq 124 ] || [ "$validate_rc" -eq 137 ]; then + echo "::warning title=runtime-render validation timed out::validate_measurement.py check --stamp exceeded ${PIGEN_VALIDATOR_TIMEOUT_S}s in the build chroot and was killed (rc=${validate_rc}); the image ships WITHOUT the runtime-render marker (litclock-dev#531)" + echo " WARNING: runtime-render validation TIMED OUT after ${PIGEN_VALIDATOR_TIMEOUT_S}s in the build chroot (rc=${validate_rc})." + # A run that overran its bound is not a trusted proof, and a validator + # that survived the TERM long enough to finish would have stamped one + # -- the annotation above would then be a lie (PR litclock-dev#852 review, F2). + rm -f .runtime-render-validated + else + echo "::warning title=runtime-render validation failed::validate_measurement.py check --stamp exited ${validate_rc} in the build chroot; the image ships WITHOUT the runtime-render marker (litclock-dev#531)" + echo " WARNING: runtime-render validation FAILED in the build chroot (rc=${validate_rc})." + fi echo " The image is usable: it will serve pre-rendered images and ignore" echo " LITCLOCK_RUNTIME_RENDER. Investigate before relying on the runtime tier." + # A validator killed mid-write leaves its mkstemp litter next to the + # marker, and cp -a would ship it. Same glob as update.sh: `.` + # plus EIGHT [a-z0-9_] characters, which is Python's tempfile scheme + # exactly, so a `.backup` sibling cannot match. + rm -f .runtime-render-validated.[a-z0-9_][a-z0-9_][a-z0-9_][a-z0-9_][a-z0-9_][a-z0-9_][a-z0-9_][a-z0-9_] fi # Clean pip cache to reduce image size (litclock-dev#112) diff --git a/scripts/first-boot.sh b/scripts/first-boot.sh index 397d40f..d6afcb8 100755 --- a/scripts/first-boot.sh +++ b/scripts/first-boot.sh @@ -501,6 +501,47 @@ disable_first_boot() { fi } +# litclock-dev#834 — prepare-for-cloning.sh Step 6 replaces each bash history +# path with an empty DIRECTORY, so that shells still open when the master +# powers off cannot write their history back (bash's exit-time append, an +# explicit `history -w` and the HISTFILESIZE truncate all fail with EISDIR, +# root included). That directory rides every clone; this puts the paths back +# to "absent" so bash creates a normal file on the recipient's first login. +# +# `rmdir`, unconditionally, through sudo: it removes an EMPTY DIRECTORY and +# nothing else, so a real history file (ENOTDIR) or an absent path (ENOENT) is +# a no-op, and sudo is needed for /root, which pi cannot even stat into (/root +# is 0700). Like the `sudo tee`/`sudo cp`/`sudo systemctl` calls around it, +# this rides the `010_pi-nopasswd` grant rather than `020_litclock-control`, +# which covers only the PWA's own command list — consistent with its +# neighbours, not a new privilege. +# +# A directory that survives is reported, not fatal — shell history not being +# saved is an annoyance, not a setup failure. +# +# The verdict is a PRINTED TOKEN, not `sudo test -d`'s exit status +# (litclock-dev#855 review). sudo returns 1 both for "the command said no" and +# for its own configuration/authorization failures, so `if sudo test -d` +# reads a refused sudo as "the directory is gone" — and then reports success +# while the lock, and a permanently disabled history, ride on. An unusable +# probe must say "could not look", never "confirmed absent". +restore_bash_history_after_clone_prep() { + local _p _verdict + for _p in "$@"; do + sudo rmdir "$_p" 2>/dev/null || true + _verdict=$(sudo sh -c 'if [ -d "$1" ]; then echo LOCKED; else echo CLEAR; fi' sh "$_p" 2>/dev/null) || _verdict="" + case "$_verdict" in + CLEAR) ;; + LOCKED) + log "WARN clone-prep history lock at $_p could not be removed; shell history will not be saved there (litclock-dev#834)" + ;; + *) + log "WARN could not check the clone-prep history lock at $_p (the privileged probe did not run); if shell history is not saved there, remove it with: sudo rmdir $_p (litclock-dev#834)" + ;; + esac + done +} + # Main orchestration flow main() { log "======================================" @@ -516,6 +557,11 @@ main() { exit 0 fi + # litclock-dev#834 — undo prepare-for-cloning.sh's history lock before + # anything else on the not-yet-set-up path, which is the only path a + # cloned card takes, so the recipient's first login gets a normal history. + restore_bash_history_after_clone_prep /home/pi/.bash_history /root/.bash_history + # Stop the clock timer — if re-running first-boot (e.g. after removing # .setup-complete for testing), the timer may still be enabled from a # previous setup cycle and would show quotes during hotspot setup. @@ -534,54 +580,31 @@ main() { # where setup-complete didn't land before reboot. if [[ ! -f "$ENV_FILE" ]]; then log "Creating default env.sh..." - # litclock-dev#337 A3: WEATHER_LOCATION_MODE + WEATHER_IP_COUNTRY shipped from - # the very first boot. MODE=auto means the on-boot reresolve service - # will populate the rest once WiFi connects + IP-geo succeeds. - # litclock-dev#783 — this list must cover EVERY key in env.sh.sample. - # update.sh Phase 3 merges missing sample keys into env.sh, but that - # only runs on an OTA: a freshly flashed device that never updates was - # born missing nine documented knobs, so env.sh was not the knob - # surface the docs describe. Pinned by - # tests/test_first_boot_flow.py::test_every_env_write_matches_env_sample, and - # ::test_first_boot_actually_writes_every_env_sample_key which EXECUTES - # both arms and asserts on the env.sh actually written. + # litclock-dev#840 — the body comes from env_sh_defaults() in + # lib/state.sh, the single source shared with reset-setup.sh and + # prepare-for-cloning.sh. The key set, the litclock-dev#337 A3 MODE=auto default, + # the deliberately-empty coordinates and the load-bearing comment + # status all live there; read its contract before changing this. # - # VALUES may differ from the sample and two deliberately do: the sample - # documents WEATHER_LATITUDE/LONGITUDE with real Austin coordinates as - # an example, while first boot must leave them EMPTY. With - # WEATHER_LOCATION_MODE=auto the IP-geo resolver fills them on a good - # boot, but on the ip-api.com-blocked path (a QA scenario we test) a - # seeded coordinate would render Austin weather on a device that is not - # in Austin — worse than the honest empty state. So this seeds NAMES - # from the sample, not values, and the guard checks names only. + # The GATE tests BOTH helpers, not just the writer: the fallback below + # is the arm for "lib/state.sh is not usable", and a state.sh present + # but too old to define env_sh_defaults is exactly that case. Gating on + # the writer alone would call an undefined function and seed an EMPTY + # env.sh instead of taking the fallback. # - # COMMENT STATUS is copied from the sample and matters: an ACTIVE - # LITCLOCK_RENDER_LEAD_S would hard-pin 4 into the field and make a - # later retune leave every device rendering the wrong minute (litclock-dev#762), - # and an empty active value is parsed at import above the litclock-dev#531 - # BaseException guard, killing the painter every minute. - local _defaults - _defaults='# export OPENWEATHERMAP_APIKEY= -export WEATHER_ENABLED=true -export WEATHER_LATITUDE= -export WEATHER_LONGITUDE= -export WEATHER_LOCATION_NAME= -export WEATHER_UNITS=imperial -export WEATHER_LOCATION_MODE=auto -export WEATHER_IP_COUNTRY= -export WEATHER_LAST_IP_GEO_AT= -export WEATHER_TTL=3600 -export ALLOW_NSFW_QUOTES=false -export LITCLOCK_LANGUAGE= -export SHOW_DIAGNOSTICS_SHORTCUT=false -export GIFT_MODE_MESSAGE= -export LITCLOCK_RUNTIME_RENDER=false -# export DISPLAY_CLEAR_HOUR=2 -# export LITCLOCK_RENDER_LEAD_S=4 -# export WEATHER_API_TIMEOUT=15 -# export LOG_LEVEL=WARNING -' - if declare -F atomic_write_env_sh >/dev/null 2>&1; then + # The `$'\n'` is REQUIRED: command substitution strips trailing + # newlines, and update.sh Phase 3 appends missing sample keys with `>>`. + # + # Pinned by tests/test_first_boot_flow.py — + # ::test_env_sh_defaults_helper_matches_sample (the single source vs + # the sample) and ::test_first_boot_actually_writes_every_env_sample_key, + # which EXECUTES both arms and asserts on the env.sh actually written. + if declare -F atomic_write_env_sh >/dev/null 2>&1 \ + && declare -F env_sh_defaults >/dev/null 2>&1; then + local _defaults + # NO language argument — a fresh device keeps Accept-Language + # negotiation alive on this boot (the litclock-dev#743 empty-seed contract). + _defaults=$(env_sh_defaults)$'\n' if ! atomic_write_env_sh "$ENV_FILE" "$_defaults"; then local _rc=$? if [[ "$_rc" == "75" ]]; then @@ -595,11 +618,16 @@ export LITCLOCK_RUNTIME_RENDER=false # to the legacy heredoc; production Pis always have state.sh # because it ships in the same release as first-boot.sh. log "WARN scripts/lib/state.sh missing — falling back to unlocked default-env write" - # litclock-dev#783 — kept in step with the _defaults block above and - # with env.sh.sample. This degraded path was MISSED by the first - # version of that fix (found by /review): it is a second seeder in - # the same file, and a guard that named `_defaults=` could not see - # it. The guard now DISCOVERS seed blocks by shape instead. + # litclock-dev#840 — this heredoc STAYS INLINE, deliberately. It is + # the arm for "lib/state.sh could not be sourced", so it is the one + # copy that CANNOT call env_sh_defaults without reintroducing the + # dependency it exists to survive. That makes it the only remaining + # duplicate of the block, and + # tests/test_first_boot_flow.py::test_first_boot_fallback_heredoc_matches_the_helper + # pins it byte-for-byte against env_sh_defaults so it cannot drift. + # (litclock-dev#783 found this arm MISSED by the first version of that guard: + # a guard that named `_defaults=` could not see a second seeder in + # the same file.) cat > "$ENV_FILE" << 'ENVEOF' # export OPENWEATHERMAP_APIKEY= export WEATHER_ENABLED=true @@ -861,23 +889,23 @@ ENVEOF # # NOTE the asymmetry, and it is not an oversight: `systemctl poweroff` # IS in the scoped 020 allowlist, this `touch` is NOT (020 grants - # `touch` for /etc/litclock/.handoff-complete only), so today this line - # works solely via the broad 010 passwordless grant. If 010 is ever - # dropped the touch fails silently and the shutdown splash repaints - # over the recovery copy — the device still powers off, it just loses - # the message. Granting it in 020 is NOT the fix: shutdown-splash.sh - # justifies the root-owned path on the opposite ground (a pi-level - # process must not be able to plant it and mute the gift welcome), so - # closing this needs a root-owned wrapper like - # /usr/local/lib/litclock/litclock-set-timezone, not a wider allowlist. - # tests/test_sudoers_install.py pins both halves of that statement. + # `touch` for /etc/litclock/.handoff-complete only), so this line rides + # the broad 010 grant — kept deliberately, the litclock-dev#387/litclock-dev#82 drop having been + # reversed 2026-07-12 (see sudoers/020's header). Without it the touch + # fails silently and the splash repaints over the recovery copy; the + # device still powers off, it just loses the message. Granting it in 020 + # is NOT the fix: shutdown-splash.sh justifies the root-owned path on + # the opposite ground (a pi-level process must not be able to plant it + # and mute the gift welcome), so closing that needs a root-owned wrapper + # like /usr/local/lib/litclock/litclock-set-timezone, not a wider + # allowlist. tests/test_sudoers_install.py pins both halves. if [[ "$_incomplete_painted" == true ]]; then sudo touch /run/litclock-splash-suppress 2>/dev/null || true fi # `sudo systemctl poweroff` (not bare `sudo poweroff`) — matches the # sudo-systemctl form used everywhere else in this script and the - # scoped 020 sudoers allowlist, so it survives a future drop of the 010 - # passwordless-sudo grant. If the marker touch above failed, we still + # scoped 020 sudoers allowlist, so unlike the touch above it does not + # depend on the blanket 010 grant. If the marker touch above failed, we still # power off (a device stranded ON is worse than the splash getting # repainted) — the poweroff is deliberately not gated on it. # Not in #22: dev already decided at the sibling call site diff --git a/scripts/lib/state.sh b/scripts/lib/state.sh index f100c8b..63a6bdb 100644 --- a/scripts/lib/state.sh +++ b/scripts/lib/state.sh @@ -108,6 +108,80 @@ atomic_remove_file() { ENV_FILE_DEFAULT="${LITCLOCK_ENV_FILE:-/home/pi/litclock/env.sh}" +# env_sh_defaults [LANGUAGE] — print the canonical env.sh body every +# whole-file seeder writes (litclock-dev#840). +# +# THE SINGLE SOURCE. This block was hand-copied into first-boot.sh, +# reset-setup.sh and prepare-for-cloning.sh, and the copies drifted: measured +# at litclock-dev#783, they seeded 9, 10 and 8 of the sample's keys, so a +# device was born missing knobs depending on which path created it. The +# highest-fanout copy (prepare-for-cloning, the "SD Cards for Friends & +# Family" flow) was the worst. One function means a new env.sh.sample key is +# added in two places — the sample and here — not five. +# +# CONTRACT, all three parts load-bearing: +# +# 1. KEY SET. Must cover EVERY key in env.sh.sample. +# tests/test_first_boot_flow.py::test_env_sh_defaults_helper_matches_sample +# executes this function and compares against the sample. +# 2. COMMENT STATUS is copied from the sample and is NOT cosmetic. An +# ACTIVE `LITCLOCK_RENDER_LEAD_S` hard-pins 4 into the field and makes a +# later retune leave every device rendering the wrong minute (litclock-dev#762); +# an active-but-EMPTY value is parsed at import above the litclock-dev#531 +# `except BaseException` guard and kills the painter every minute. +# 3. TRAILING NEWLINE. The body ends with one. update.sh Phase 3 APPENDS +# missing sample keys with `>>`, so a body without it would splice the +# first appended key onto `# export LOG_LEVEL=WARNING`. Command +# substitution STRIPS trailing newlines, so every caller re-adds one: +# `DEFAULTS=$(env_sh_defaults)$'\n'`. Do not "simplify" that away. +# +# VALUES may differ from the sample and two deliberately do: the sample +# documents WEATHER_LATITUDE/LONGITUDE with real Austin coordinates as an +# example, while a seeded device must leave them EMPTY. With +# WEATHER_LOCATION_MODE=auto the IP-geo resolver fills them on a good boot; +# on the ip-api.com-blocked path a seeded coordinate would render Austin +# weather on a device that is not in Austin — worse than an honest empty. +# +# LANGUAGE ($1, optional) seeds LITCLOCK_LANGUAGE. Empty (the default, used by +# first-boot and prepare-for-cloning) keeps Accept-Language negotiation alive +# on the next first boot — the litclock-dev#743 empty-seed contract. reset-setup.sh passes +# the gift language so every env-reading surface on the recipient's device +# boots in the gifter's chosen language (litclock-dev#532). +# +# The shape gate travels WITH the interpolation point (Codex 5b /review): this +# value is interpolated into a root-written env.sh, so anything outside the +# language-tag alphabet is dropped rather than shipped. reset-setup.sh gates +# and WARNS before calling, so in practice this never fires — it is here so +# that moving the interpolation into this file did not leave the belt behind. +env_sh_defaults() { + local language="${1-}" + if [[ -n "$language" && ! "$language" =~ ^[a-z][a-z0-9-]{0,16}$ ]]; then + echo "[state] env_sh_defaults: language '$language' failed the shape check; seeding empty" >&2 + language="" + fi + cat </dev/null || true fi diff --git a/scripts/prepare-for-cloning.sh b/scripts/prepare-for-cloning.sh index f3a857a..dcdd087 100755 --- a/scripts/prepare-for-cloning.sh +++ b/scripts/prepare-for-cloning.sh @@ -8,6 +8,20 @@ # # Usage: sudo ./scripts/prepare-for-cloning.sh # +# Run it as the ONLY open session, from the local console, and log out of +# every other shell first (SSH sessions, tmux panes, a `sudo -i` root shell). +# Step 6 clears the bash history, but `history -c` reaches only this script's +# own non-interactive shell: any interactive shell still holding history when +# the Pi powers off writes it back to disk on exit -- `nmcli ... password ...` +# lines included -- and that is what gets imaged (litclock-dev#834). +# +# Logging out of the OTHER shells is not the whole answer, and the rule should +# not be read as if it were: the shell you launch this from cannot be one of +# them. It is alive for the whole run and writes its own history as it exits, +# during the power-off. That one is covered by Step 6 locking both history +# paths against write-back -- which is why the lock exists and why Step 6 now +# refuses to continue if it cannot apply it. +# set -e @@ -80,6 +94,12 @@ _NM_PROFILE_DIR="/etc/NetworkManager/system-connections" # branch, and a hardcoded /etc path inside an executed span would overwrite the # host's real wpa_supplicant.conf when the suite runs as root (/review litclock-dev#710). _WPA_SUPPLICANT_CONF="/etc/wpa_supplicant/wpa_supplicant.conf" +# litclock-dev#834 — same convention as the two above: fixed paths, not +# env-overridable, injected into the lifted span by the tests. Step 6 replaces +# each with an empty DIRECTORY (see there for why a directory), and +# scripts/first-boot.sh removes that directory on the clone's first boot. +_PI_BASH_HISTORY="/home/pi/.bash_history" +_ROOT_BASH_HISTORY="/root/.bash_history" # Source shared state-file helpers for atomic_write_env_sh (litclock-dev#274) — the # env.sh writer-lock that interoperates with src/config.py's fcntl.flock @@ -89,6 +109,29 @@ _WPA_SUPPLICANT_CONF="/etc/wpa_supplicant/wpa_supplicant.conf" _THIS_SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) # shellcheck source=/dev/null . "$_THIS_SCRIPT_DIR/lib/state.sh" +# +# litclock-dev#840 — VERIFY the helpers this script calls are actually defined, +# not merely that state.sh was sourced. `update.sh` installs the root-owned +# copy of this script in its privilege-helper loop and the root-owned +# lib/state.sh AFTERWARDS, so an interrupted or half-failed update leaves a NEW +# script beside an OLD state.sh: sourcing succeeds, `atomic_write_env_sh` +# exists, and `env_sh_defaults` does not. +# +# `set -e` (line 26) would abort this script on the resulting "command not +# found" anyway, so the gate is not what makes it safe — but an abrupt abort +# mid-Step-2 is the wrong report for a card-prep tool whose whole contract is +# "say DO NOT CLONE loudly". Check explicitly, before any step has run, so the +# operator gets a reason instead of a bash error. Same check as +# reset-setup.sh's, which needs it for correctness rather than for the message. +for _fn in atomic_write_env_sh env_sh_defaults; do + if ! declare -F "$_fn" >/dev/null 2>&1; then + echo -e "${RED}ERROR: $_fn is not defined after sourcing lib/state.sh.${NC}" >&2 + echo " $_THIS_SCRIPT_DIR/lib/state.sh is missing or too old for this script." >&2 + echo -e "${RED} Do NOT clone this card — nothing has been wiped.${NC}" >&2 + echo " Re-run the updater, or reinstall from a matching release." >&2 + exit 1 + fi +done # litclock-dev#701 — the opt-in WiFi wipe in Step 3 deletes EVERY NetworkManager # connection, wired included, so an operator running this over any NM-managed @@ -270,17 +313,78 @@ if ! mkdir -p "$STATE_DIR" 2>/dev/null || ! touch "$_UNFINISHED_MARKER" 2>/dev/n fi -# litclock-dev#821 — Step 2's env.sh wipe used to be best-effort with no abort: -# a flock timeout or a failed write printed one YELLOW line and the script ran -# on to "SD Card Ready for Cloning!" and powered off, so every card cut from the -# master carried the owner's API key and home location. Declared HERE rather -# than at Step 2 because the gate that reads it runs ~230 lines later and this -# script has no `set -u`. -ENV_WIPE_FAILED=false +# The shared "this card must not be cloned" abort. Every step that can leave +# owner data on the card routes through it, so the operator sees the same +# shape — FAILED on the step's own line, what is wrong, what a clone would +# carry, then what state the card is in — wherever the run stops. +# $1 what is wrong (ends with a full stop) +# $2 what a clone would therefore carry (completes "Do NOT clone this card — ") +# $3… extra RED lines, normally the state the card is left in +_abort_do_not_clone() { + local _what="$1" _carries="$2" + shift 2 + echo -e "${RED}FAILED${NC}" + echo -e "${RED}${_what}${NC}" + echo -e "${RED}Do NOT clone this card — ${_carries}${NC}" + local _line + for _line in "$@"; do + echo -e "${RED}${_line}${NC}" + done + exit 1 +} + +# litclock-dev#821 / litclock-dev#839 — the env.sh arm of the above, shared by +# Step 2 (a rewrite that REPORTED failure) and the end-of-run gate (a rewrite +# that returned 0 and left a bad file). litclock-dev#821 caught the failure with a flag +# the gate read after Step 8; litclock-dev#839 found that between the two the script +# still deleted every WiFi profile, stopped the clock, vacuumed the journal and +# removed the setup-WiFi key — on a card it then refused to clone — so the flag +# is gone and Step 2 aborts on the spot. $1 is the reason; any further +# arguments are extra RED lines (Step 2 uses them to say what state the card is +# in; the gate, running after every step, does not). +_abort_env_credentials() { + local _reason="$1" + shift + _abort_do_not_clone "env.sh still holds this device's owner data: ${_reason}." \ + "every copy would carry the API key and home location." "$@" +} + +# The env.sh Step 2 writes. Declared HERE, not at Step 2, because the gate +# ~350 lines below derives its key list from it and this script has no `set -u`. +# +# litclock-dev#840 — the body comes from env_sh_defaults() in lib/state.sh, +# the single source shared with first-boot.sh and reset-setup.sh. This copy +# was hand-maintained and was the SHORTEST of the four (litclock-dev#783): it missed +# WEATHER_LOCATION_NAME and LITCLOCK_LANGUAGE, so every SD card cut from this +# flow — the highest-fanout distribution path there is — produced devices +# lacking the language knob litclock-dev#532 depends on. The key set, the comment +# status and the litclock-dev#337 A3 MODE=auto default all live in that helper now; see +# its contract comment before changing anything about this line. +# +# NO language argument: a clone must not inherit the cloner's LITCLOCK_LANGUAGE +# (empty keeps Accept-Language negotiation alive on the recipient's first boot). +# +# The `$'\n'` is REQUIRED, not decoration: command substitution strips trailing +# newlines, and update.sh Phase 3 appends missing sample keys with `>>`, which +# would otherwise splice the first one onto `# export LOG_LEVEL=WARNING`. +DEFAULTS=$(env_sh_defaults)$'\n' +# litclock-dev#839 — the keys the gate re-reads are DERIVED from the wipe, not +# listed a second time: litclock-dev#821's hand-written list (five keys) was narrower +# than the wipe (eight), so WEATHER_IP_COUNTRY, WEATHER_LAST_IP_GEO_AT and +# LITCLOCK_LANGUAGE would have survived the "returned 0, bad file" path +# unflagged. Every key DEFAULTS writes EMPTY — live or commented out — is +# owner data (API key, coordinates, place name, IP country, geo stamp, +# language, gift message); a key the wipe gives a VALUE is a preference. +# One name per line. +_ENV_OWNER_KEYS=$(printf '%s\n' "$DEFAULTS" \ + | sed -nE 's/^#?[[:space:]]*export[[:space:]]+([A-Za-z_][A-Za-z0-9_]*)=[[:space:]]*$/\1/p') -# Step 1: Stop the setup-state writers, then remove the setup-state markers. +# Step 1: Stop the setup-state writers. The markers they can re-create are +# removed in Step 2b, once the env.sh wipe has succeeded (litclock-dev#855 +# review: removing them before an abortable step left a device that booted +# into neither setup nor the clock). # -# The markers below are RE-CREATABLE, so they must be removed with their +# The markers are RE-CREATABLE, so they must be removed with their # writers already down. reset-setup.sh has always done it in this order (it # stops six units before its own marker removal); this script did not, and # litclock-dev#673 /review found two live paths that put .handoff-complete @@ -301,7 +405,7 @@ ENV_WIPE_FAILED=false # # litclock-dev#274: stopping litclock-control.service also keeps the PWA from landing a # Settings save concurrent with the env.sh overwrite in Step 2. Best-effort -# (`|| true`) under the `set -e` at line 12 — a missing or already-stopped unit +# (`|| true`) under the `set -e` at the top of the file — a missing or already-stopped unit # must not abort the prep flow. litclock.timer is stopped in Step 4, which is # late but harmless: litclock.service only READS these markers. echo -n "Stopping setup-state writers... " @@ -314,6 +418,73 @@ systemctl stop litclock-handoff-fallback.timer 2>/dev/null || true systemctl stop litclock-handoff-fallback.service 2>/dev/null || true echo -e "${GREEN}done${NC}" +# Step 2: Clear env.sh credentials. +# +# The PWA writer was stopped in Step 1 (litclock-dev#274). Write $DEFAULTS (declared above) +# via atomic_write_env_sh, which holds the shared sidecar flock against the +# Python writer. An ABSENT env.sh is left absent, not created: first-boot.sh +# seeds one on the clone, and a card with no env.sh carries no owner data. +# +# litclock-dev#839 — a rewrite that REPORTS failure aborts HERE, before the +# Step 3 WiFi prompt. The realistic causes are a flock timeout (rc=75: the PWA +# still holding env.sh.lock) and a failed mktemp/printf/mv (a card gone +# read-only); on every one of them env.sh is untouched and still holds the +# owner's API key and location. litclock-dev#821 aborted at the gate after Step 8, which +# meant the script first deleted every WiFi profile, stopped the clock, +# vacuumed the journal and removed the setup-WiFi key — on a card it then +# refused to clone; over SSH-on-WiFi with `y` the session dropped at Step 3 and +# the red banner never printed at all. The gate after Step 8 stays, for the +# other failure: a write that returned 0 and left a bad file. +echo -n "Clearing configuration (env.sh)... " +if [[ -f "$INSTALL_DIR/env.sh" ]]; then + if atomic_write_env_sh "$INSTALL_DIR/env.sh" "$DEFAULTS"; then + echo -e "${GREEN}done${NC}" + else + _rc=$? + if [[ "$_rc" == "75" ]]; then + _ENV_WHY="the env.sh rewrite did not run — env.sh is locked by another writer (rc=75)" + else + _ENV_WHY="the env.sh rewrite failed (rc=$_rc) and env.sh is untouched" + fi + # litclock-dev#855 review F1 — retire the unfinished-run marker on THIS + # abort only. It means "a run died half-way and the card may carry the + # setup-WiFi key"; here nothing was mutated, so leaving it would open + # the re-run this banner prescribes with a warning that contradicts the + # state list below. The gate after Step 8 shares this abort and must + # NOT do this — by then the card really is part-way prepared — so it + # lives at the call site, not in the helper. + rm -f "$_UNFINISHED_MARKER" 2>/dev/null || true + _abort_env_credentials "$_ENV_WHY" \ + "Stopped before the WiFi prompt. Nothing on this card has been changed: the" \ + "setup-state markers, env.sh, the saved WiFi and the setup-WiFi key are all" \ + "intact, so this device still boots into its normal clock (the services this" \ + "script stopped come back on the next boot). Fix the cause — a stuck env.sh.lock" \ + "usually means litclock-control.service is still writing — then run again. The" \ + "PWA and the updater stay stopped until the next boot; the clock keeps painting." \ + "This run retired its own unfinished-run marker, so if the next run still warns" \ + "that a previous one did not finish, that warning is about an EARLIER run." + fi +else + echo -e "${GREEN}done${NC}" +fi + +# Step 2b (was the second half of Step 1 until the litclock-dev#855 review): +# remove the setup-state markers, AFTER the env.sh wipe has succeeded. +# +# The order is a recovery property, not tidiness. `litclock-firstboot.service` +# is DISABLED on a provisioned clock (first-boot.sh disables it once setup +# completes) and is only re-enabled at Step 4 — while litclock.service and +# litclock-control.service are ConditionPathExists-gated on the two markers +# below. So a run that removed the markers and then aborted at Step 2 left a +# device that boots into NOTHING: no setup, no clock, no PWA, which is the +# state litclock-dev#659 documents as indistinguishable from a brick, now +# reachable from an ordinary flock timeout. With the removal here, the Step 2 +# abort leaves a working, fully provisioned clock. +# +# Nothing is lost by the move: the writers were stopped in Step 1 (that is the +# primary defence and it has not moved), and DELAYING the removal can only +# SHRINK the window in which a concurrent updater could re-touch a marker we +# have already deleted. # litclock-dev#673: clear the handoff marker too, exactly as reset-setup.sh # does. Both scripts return the device to a fresh-setup state, so both must # clear every marker a systemd unit gates on. (Non-gate markers such as @@ -329,7 +500,7 @@ echo -e "${GREEN}done${NC}" # thing standing between a clone and a literary quote painted over the WiFi # setup instructions the recipient is trying to read. echo -n "Removing setup-state markers... " -# `|| true` under the `set -e` at line 12, with the existence check below as the +# `|| true` under the `set -e` at the top of the file, with the existence check below as the # real gate -- the same shape as Step 8 (litclock-dev#649), which this step lacked. # Without it a failing `rm` terminates the script ON THIS LINE: `done` is never # printed (the terminal is left mid-line), no diagnostic appears, and the run @@ -352,6 +523,11 @@ _SURVIVORS=() # litclock-dev#665: a clone must not ship carrying the master's reset-failure # marker — the recipient would be told not to pass on a card that is fine. rm -f "$STATE_DIR/reset-failed" 2>/dev/null || true +# litclock-dev#847 item 1 (litclock-dev#854 review): same argument for the runtime-render +# validation memo. It records that THIS device's freetype could not reproduce +# the expected measurement (or that a tick never had the budget to try); every +# clone would inherit the master's verdict about hardware it has never run on. +rm -f "$STATE_DIR/runtime-render-validation.json" 2>/dev/null || true for _m in .setup-complete .handoff-complete; do # `-L` alongside `-e` because `-e` follows symlinks and is false for a @@ -373,69 +549,6 @@ fi unset _MARKER_ERR _SURVIVORS _m echo -e "${GREEN}done${NC}" -# Step 2: Clear env.sh credentials. -# -# The PWA writer was stopped in Step 1 (litclock-dev#274). Write defaults via -# atomic_write_env_sh, which holds the shared sidecar flock against the Python -# writer; the explicit `|| true` on the helper call is required because `set -e` -# would otherwise treat a lock timeout (rc=75) as fatal and kill the whole prep -# flow halfway through. - -echo -n "Clearing configuration (env.sh)... " -if [[ -f "$INSTALL_DIR/env.sh" ]]; then - # litclock-dev#337 A3: defensive MODE + IP_COUNTRY defaults so a cloned image's - # first boot lands on MODE=auto (on-boot reresolve will populate the - # rest). Without these, a cloned env.sh would inherit whatever MODE - # the cloner had — could be "specific" with stale coords for a - # location 1000 miles from the cloned device's actual WiFi. - # litclock-dev#783 — must cover EVERY env.sh.sample key; comment status is - # copied from the sample and is load-bearing (an active - # LITCLOCK_RENDER_LEAD_S would hard-pin 4 into the field, litclock-dev#762). This - # list was the SHORTEST of the four and was missing WEATHER_LOCATION_NAME - # and LITCLOCK_LANGUAGE, so every SD card cut from this flow produced - # devices lacking the language knob litclock-dev#532 depends on. - DEFAULTS='# export OPENWEATHERMAP_APIKEY= -export WEATHER_ENABLED=true -export WEATHER_LATITUDE= -export WEATHER_LONGITUDE= -export WEATHER_LOCATION_NAME= -export WEATHER_UNITS=imperial -export WEATHER_LOCATION_MODE=auto -export WEATHER_IP_COUNTRY= -export WEATHER_LAST_IP_GEO_AT= -export WEATHER_TTL=3600 -export ALLOW_NSFW_QUOTES=false -export LITCLOCK_LANGUAGE= -export SHOW_DIAGNOSTICS_SHORTCUT=false -export GIFT_MODE_MESSAGE= -export LITCLOCK_RUNTIME_RENDER=false -# export DISPLAY_CLEAR_HOUR=2 -# export LITCLOCK_RENDER_LEAD_S=4 -# export WEATHER_API_TIMEOUT=15 -# export LOG_LEVEL=WARNING -' - # `|| true` not needed: every code path inside the if/else below - # ends with a 0-exit statement, so `set -e` won't trip. - if atomic_write_env_sh "$INSTALL_DIR/env.sh" "$DEFAULTS"; then - echo -e "${GREEN}done${NC}" - else - _rc=$? - if [[ "$_rc" == "75" ]]; then - echo -e "${YELLOW}skipped (env.sh locked by another writer)${NC}" - else - echo -e "${YELLOW}failed (rc=$_rc) — env.sh untouched${NC}" - fi - unset _rc - # litclock-dev#821 — was `true # explicit success for set -e`, which is - # what let a failed wipe reach the success banner. The gate before the - # banner reads this; `set -e` is still satisfied because assigning to a - # variable exits 0. - ENV_WIPE_FAILED=true - fi -else - echo -e "${GREEN}done${NC}" -fi - # Step 3: Clear WiFi credentials (optional - ask user) echo "" read -p "Clear saved WiFi networks? (y/N) " -n 1 -r @@ -551,11 +664,95 @@ journalctl --rotate 2>/dev/null || true journalctl --vacuum-time=1s 2>/dev/null || true echo -e "${GREEN}done${NC}" -# Step 6: Clear bash history +# Step 6: Clear bash history — and lock it against write-back (litclock-dev#834). +# +# `history -c` clears only THIS script's non-interactive shell. Any interactive +# shell still open when the Pi powers off (the console login the operator ran +# this from, an SSH session, a `sudo -i` root shell) writes its in-memory +# history back on exit — bash's save_history() APPENDS the session's lines, +# then truncates to HISTFILESIZE — so on the bench the file `rm` had removed was +# back eight seconds later holding the operator's last commands, and the card +# was imaged with them. On a master for strangers that can include +# `nmcli ... password ...` lines. +# +# The lock is an empty DIRECTORY at each history path. The shape is deliberate, +# measured against bash 5.2 / readline 8.2: +# - an empty mode-0 root-owned FILE blocks the exit-time append for `pi` +# (EACCES) but not for a root shell (DAC override), and not `history -w`, +# which readline writes to a sibling temp file and rename()s over the +# target — a rename needs only the parent directory; +# - a `/dev/null` symlink loses to the same rename; +# - `chattr +i` blocks everything but needs ext4 and root to undo, and a +# failed restore leaves the recipient an unremovable file; +# - a directory fails open(O_WRONLY|O_APPEND), rename() over it and the +# truncate's read with EISDIR, which no capability bypasses, and one +# `rmdir` removes it. +# scripts/first-boot.sh removes both directories on the clone's first boot +# (restore_bash_history_after_clone_prep), so the recipient gets an ordinary +# history. A HISTFILE drop-in under /etc/profile.d was rejected: it reaches +# only shells started AFTER it, never the open one doing the write-back. +# +# FATAL if either path cannot be emptied or locked (litclock-dev#855 review). +# A YELLOW note was the first cut and is not enough here: this script's whole +# contract is that the card carries nothing about this device, and the two +# realistic failures both leave real credentials on it — a history file that +# will not unlink (immutable, or a read-only parent) and a NON-EMPTY directory +# left by an earlier aborted run. The `-d` check alone read the second as +# "locked" and printed green over the contents, which first-boot's rmdir then +# leaves in place on every clone. echo -n "Clearing bash history... " -rm -f /home/pi/.bash_history 2>/dev/null || true -rm -f /root/.bash_history 2>/dev/null || true history -c 2>/dev/null || true +_HIST_DIRTY=() +_HIST_UNLOCKED=() +for _h in "$_PI_BASH_HISTORY" "$_ROOT_BASH_HISTORY"; do + # Empty the path whatever shape it has: rmdir takes the lock a previous run + # left (and refuses a non-empty one), rm -f takes a real history file. An + # absent path satisfies both harmlessly. + # + # The rmdir is NOT redundant with the mkdir below (litclock-dev#855 review + # G): it was, while a surviving entry only warned, but now a surviving + # entry ABORTS — so without it a second run over the lock this script + # itself left would meet a directory `rm -f` cannot remove and refuse the + # card. Pinned by test_a_second_run_over_the_lock_is_a_noop. + rmdir "$_h" 2>/dev/null || rm -f "$_h" 2>/dev/null || true + if [[ -e "$_h" || -L "$_h" ]]; then + # Something we could not remove survives — and on this path that means + # its CONTENTS survive too. `-L` as well as `-e`, which follows + # symlinks and is false for a dangling one. + _HIST_DIRTY+=("$_h") + continue + fi + mkdir "$_h" 2>/dev/null || true + # Just "is it a directory". Emptiness and non-symlink-ness need no test + # HERE and deliberately have none (litclock-dev#855 review E3, and a mutant + # that proved the point): nothing reaches this line unless the path was + # verified gone two lines up, so what `mkdir` leaves is a real, empty + # directory or nothing at all. A non-empty directory or a symlink — the + # shapes that matter — survive removal and are caught by the DIRTY arm + # above, which is where `-L` earns its place, because `-e` follows symlinks + # and is blind to a dangling one. An arm no case can reach is an arm no + # mutant can kill, so it is better not written. + if [[ ! -d "$_h" ]]; then + _HIST_UNLOCKED+=("$_h") + fi +done +if (( ${#_HIST_DIRTY[@]} )); then + _abort_do_not_clone "Could not clear the shell history at ${_HIST_DIRTY[*]}." \ + "every copy would carry the commands typed on this device, WiFi passwords included." \ + "A file that will not unlink (chattr +i, a read-only card) or a non-empty directory" \ + "left by an earlier aborted run. This card is already part-way prepared: markers" \ + "cleared, env.sh scrubbed, setup-WiFi key NOT yet removed. Fix the cause, then run" \ + "this script again from the start." +fi +if (( ${#_HIST_UNLOCKED[@]} )); then + _abort_do_not_clone "Could not lock ${_HIST_UNLOCKED[*]} against write-back." \ + "any shell still open at power-off would write its history onto the card." \ + "The history itself is cleared, but nothing stops the shell you are typing in from" \ + "writing it back as it exits — which is litclock-dev#834, the reason this lock" \ + "exists. This card is already part-way prepared: markers cleared, env.sh scrubbed," \ + "setup-WiFi key NOT yet removed. Fix the cause, then run this script again." +fi +unset _HIST_DIRTY _HIST_UNLOCKED _h echo -e "${GREEN}done${NC}" # Step 7: Clear legacy SSL certificates (nothing regenerates these since litclock-dev#715) @@ -578,7 +775,7 @@ echo -e "${GREEN}done${NC}" # the image. The glob catches staging files orphaned by a power cut between # mkstemp and os.replace, each holding a real past password. echo -n "Clearing setup-hotspot password... " -# `|| true` under the `set -e` at line 12 (litclock-dev#649). Without it, a genuinely +# `|| true` under the `set -e` at the top of the file (litclock-dev#649). Without it, a genuinely # failing `rm` terminates the script ON THIS LINE, so the `if` below never # runs and none of its three RED lines ever print — the warning written # specifically to stop someone cloning a compromised card was unreachable in @@ -657,32 +854,34 @@ echo -e "${GREEN}done${NC}" # wipe invisible: the operator is not watching a scrolled-past yellow line on a # Pi that then shuts itself down. # -# TWO checks, because they fail differently. The flag catches a write that -# REPORTED failure (flock timeout rc=75, or a failed mktemp/printf/mv). The -# re-read catches a write that returned 0 and produced a bad file anyway — -# Step 8's idiom, and the reason this is a verify rather than a trusted return -# code. Either one is disqualifying. +# This is the RE-READ half. Step 2 already aborts on a write that REPORTED +# failure (litclock-dev#839), so what is left to catch is a write that returned +# 0 and produced a bad file anyway — Step 8's idiom, and the reason this is a +# verify rather than a trusted return code. An ABSENT env.sh passes: Step 2 +# leaves an absent file absent, nothing between the two touches it, so absent +# means no owner data, and first-boot.sh seeds a fresh one on the clone. echo -n "Verifying env.sh carries no owner credentials... " _ENV_LEAKS="" -if [[ "$ENV_WIPE_FAILED" == "true" ]]; then - _ENV_LEAKS="the env.sh rewrite did not complete" -elif [[ -f "$INSTALL_DIR/env.sh" ]]; then - # Non-empty value on any secret-bearing key. A commented line is fine (the - # defaults comment OPENWEATHERMAP_APIKEY out entirely), so anchor on a live +if [[ -f "$INSTALL_DIR/env.sh" ]]; then + # Fail CLOSED on an empty key list — litclock-dev#773's rule: nothing verified is a + # failure, not a pass. The list is derived from $DEFAULTS at the top; if + # that derivation ever broke, this loop would check nothing and print `done`. + if [[ -z "$_ENV_OWNER_KEYS" ]]; then + _abort_env_credentials "no keys were verified (the owner-key list derived from the defaults is empty)" + fi + # Non-empty value on any owner-data key. A commented line is fine (the + # defaults comment the API key out entirely), so anchor on a live # `export KEY=`. - for _k in OPENWEATHERMAP_APIKEY WEATHER_LATITUDE WEATHER_LONGITUDE \ - WEATHER_LOCATION_NAME GIFT_MODE_MESSAGE; do + while read -r _k; do + [[ -n "$_k" ]] || continue if grep -qE "^[[:space:]]*export[[:space:]]+${_k}=[^[:space:]]" "$INSTALL_DIR/env.sh" 2>/dev/null; then _ENV_LEAKS="${_ENV_LEAKS}${_ENV_LEAKS:+, }${_k}" fi - done + done <<< "$_ENV_OWNER_KEYS" unset _k fi if [[ -n "$_ENV_LEAKS" ]]; then - echo -e "${RED}FAILED${NC}" - echo -e "${RED}env.sh still holds this device's owner data: ${_ENV_LEAKS}.${NC}" - echo -e "${RED}Do NOT clone this card — every copy would carry the API key and home location.${NC}" - exit 1 + _abort_env_credentials "$_ENV_LEAKS" fi unset _ENV_LEAKS echo -e "${GREEN}done${NC}" diff --git a/scripts/reset-setup.sh b/scripts/reset-setup.sh index 5ce89d7..9b797ec 100755 --- a/scripts/reset-setup.sh +++ b/scripts/reset-setup.sh @@ -29,6 +29,10 @@ CONFIG_DIR="/etc/litclock" # Same override convention as the other scripts (wifi-watchdog, bootcheck, # lkg-record, update) and as src/wifi_provision.py's STATE_DIR. STATE_DIR="${LITCLOCK_STATE_DIR:-/var/lib/litclock}" +# Bound on each `systemctl start litclock-shutdown.service` re-arm (litclock-dev#727, +# litclock-dev#833). Must stay well inside litclock-reset.service's TimeoutStartSec=60: +# on the PWA arm this script runs INSIDE that unit. +SHUTDOWN_REARM_TIMEOUT_S=10 # Source shared state-file helpers for atomic_write_env_sh (litclock-dev#274) — the # env.sh writer-lock that interoperates with src/config.py's fcntl.flock @@ -39,6 +43,30 @@ STATE_DIR="${LITCLOCK_STATE_DIR:-/var/lib/litclock}" _THIS_SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) # shellcheck source=/dev/null . "$_THIS_SCRIPT_DIR/lib/state.sh" +# +# litclock-dev#840 — VERIFY the helpers this script calls are actually defined, +# not merely that state.sh was sourced. `update.sh` installs the root-owned +# copy of this script in its privilege-helper loop and the root-owned +# lib/state.sh AFTERWARDS, so an interrupted or half-failed update leaves a NEW +# script beside an OLD state.sh: sourcing succeeds, `atomic_write_env_sh` +# exists, and `env_sh_defaults` does not. +# +# This script has NO `set -u` and NO `set -e`, so an undefined `env_sh_defaults` +# is not an error here — it prints "command not found" to stderr, the command +# substitution yields the EMPTY STRING, `atomic_write_env_sh` writes that +# successfully, and Step 3 prints a green "done" over a ONE-BYTE env.sh. Every +# knob and the gift language would be gone with nothing saying so, on a device +# that is usually about to be shipped to someone. Fail loudly instead: nothing +# destructive has run at this point, so exiting here is free. +for _fn in atomic_write_env_sh env_sh_defaults; do + if ! declare -F "$_fn" >/dev/null 2>&1; then + echo -e "${RED}ERROR: $_fn is not defined after sourcing lib/state.sh.${NC}" >&2 + echo " $_THIS_SCRIPT_DIR/lib/state.sh is missing or too old for this script." >&2 + echo " Refusing to reset: a partial reset would leave env.sh empty." >&2 + echo " Re-run the updater, or reinstall from a matching release." >&2 + exit 1 + fi +done # ── Function definitions — ALL of them, hoisted above every caller ────────── # litclock-dev#719: bash resolves function names at execution time, and the @@ -200,6 +228,31 @@ rotate_hotspot_password_for_handoff() { echo -e "${GREEN}done${NC}" } +# Re-arm litclock-shutdown.service so its next STOP edge paints a splash. +# The unit is a RemainAfterExit=yes oneshot whose ExecStop is the splash: +# `start` on the inactive unit runs ExecStart (/bin/true) and marks it +# active again; on an already-active unit it is a no-op. Two callers: +# Step 1 (litclock-dev#727, before its own stop so a same-boot retry paints) +# and the plain-arm re-arm right after that stop (litclock-dev#833, so the +# operator's reboot paints). DELIBERATELY BLOCKING — the usual "--no-block +# from inside a service" rule does not apply and would break both callers: +# a queued (unstarted) start job is simply REPLACED by the next stop, or +# never runs before the script exits. There is no job cycle to deadlock on +# (litclock-shutdown orders only against shutdown targets, litclock.service +# and — since litclock-dev#856 — litclock-splash.service; every one of those +# edges makes IT the predecessor, so this blocking `start` never waits on +# them, and a single-unit transaction ignores ordering deps to units with no +# job in it); the timeout is the belt if that analysis is ever wrong. A swallowed failure would recreate the no-paint outcome each caller +# exists to fix, so it WARNS instead of hiding — and systemctl's own stderr +# is left alone: it is the only diagnostic the WARNING can carry, and it +# lands on the console or in the journal, both fine. +# +# $1 the consequence, finishing the WARNING sentence for this caller. +rearm_shutdown_splash() { + timeout "$SHUTDOWN_REARM_TIMEOUT_S" systemctl start litclock-shutdown.service \ + || echo "WARNING: could not re-arm the shutdown splash; $1" +} + AUTO_YES=false DO_REBOOT=false # litclock-dev#627 — power OFF after the reset instead of rebooting. The PWA @@ -531,8 +584,19 @@ echo "" # are catalog-routed, the ExecStop splash consults the marker — a plain # reset's splash must not paint in the abandoned gift's language). Gift # resets manage the marker in their own arm above (overwrite-or-remove). +# +# litclock-dev#833 (/review F2): the same for .welcome-mode. Gift mode +# touches it and only first-boot.sh clears it, so a gift prep that aborted +# at the litclock-dev#393 env-wipe gate leaves it behind — and shutdown-splash.sh's +# ladder ranks it ABOVE reboot detection. Before litclock-dev#833 a later non-gift +# run's stop edge was either suppressed (plain) or the box was leaving its +# owner anyway (poweroff); now the plain arm leaves a live edge for the +# operator's reboot, which would paint "Welcome to LitClock" on a device +# nobody is gifting. (.welcome-message is inert without the mode marker; +# the gift arm manages it.) if [[ "$GIFT_MODE" != "true" ]]; then rm -f "$CONFIG_DIR/.gift-language" + rm -f "$CONFIG_DIR/.welcome-mode" fi # Issue litclock-dev#282: tell shutdown-splash.sh we're rebooting, not powering off. @@ -609,6 +673,13 @@ fi # the retry would keep warning its owner not to pass it on, forever. The # OnFailure unit writes it again if THIS attempt fails. rm -f "$STATE_DIR/reset-failed" 2>/dev/null || true +# litclock-dev#847 item 1 (litclock-dev#854 review): the runtime-render validation memo is +# per-DEVICE state — "this freetype could not reproduce that measurement" — so +# it must not survive a reset into the next owner's hands, where it would have +# the PWA report a failure that was never theirs. Best-effort, like the marker +# above: a surviving memo is cosmetic, and aborting a reset over it would be +# the wrong trade. +rm -f "$STATE_DIR/runtime-render-validation.json" 2>/dev/null || true # Verify, don't assume (litclock-dev#673's lesson). A directory, a symlink, or a # read-only remount leaves the marker in place and `rm -f` still returns 0. This # WARNS rather than aborting: a stale warning marker is the fail-safe direction @@ -636,25 +707,38 @@ systemctl stop litclock-reset-failed.service 2>/dev/null || true # earlier failed reset already spent it, so a same-boot retry painted # nothing and e-ink persistence carried the failure splash ("Do NOT pass # it on") through a SUCCESSFUL reset's poweroff — the gift-retry case -# ships the box with its scariest message. Re-arm before stopping: -# `start` on the inactive unit runs ExecStart (/bin/true) and marks it -# active again; on an already-active unit it is a no-op, so the first run -# is unchanged. DELIBERATELY BLOCKING — the usual "--no-block from inside -# a service" rule does not apply and would break the fix: a queued -# (unstarted) start job is simply REPLACED by the stop on the next line, -# so the unit never goes active and the retry paints nothing again. -# There is no job cycle to deadlock on (litclock-shutdown orders only -# against shutdown targets and litclock.service); `timeout 10` is the -# belt if that analysis is ever wrong, and a swallowed failure would -# recreate the no-paint retry — so it WARNS instead of hiding. -timeout 10 systemctl start litclock-shutdown.service 2>/dev/null \ - || echo "WARNING: could not re-arm the shutdown splash; this run may not paint its final screen." +# ships the box with its scariest message. Re-arm before stopping +# (rearm_shutdown_splash above says why it is blocking and bounded): on +# the first run of a boot the start is a no-op, so that run is unchanged. +rearm_shutdown_splash "this run may not paint its final screen." systemctl stop litclock-shutdown.service 2>/dev/null || true # The stop edge above has consumed the suppress decision — clear the # marker NOW so it cannot mute anything else this boot (/review litclock-dev#731; # the first-boot.sh principle: the marker protects the ONE stop just # requested, never the rest of the boot). No-op on arms that never wrote it. rm -f /run/litclock-splash-suppress 2>/dev/null || true +# litclock-dev#833: the stop edge above CONSUMED the unit's once-per-boot +# stop edge, and on the plain arm nothing else this run will start it. The +# three terminal arms are fine — the script's own reboot/poweroff is the +# next thing to happen, and their edge already painted the right screen +# under the right marker (gift: the welcome splash; a second armed edge +# there would paint "Powered Off" OVER it). The plain arm exits and hands +# the box to an operator whose `sudo reboot` — the very thing the closing +# banner asks for — is the next stop edge: with the unit inactive, +# shutdown-splash.sh never runs and e-ink persistence carries the stale +# quote across the power-off. Re-arm HERE, not at the end of the script: +# every abort past this point (the fail-closed rotation `exit 1`s, the +# survivor check, a Step 7 WiFi wipe that drops the SSH session and SIGHUPs +# the script — the common bare-reset-over-WiFi case) would otherwise leave +# the unit consumed and the operator's reboot silent, the same symptom +# through the abort door. From here the only unarmed window is the gap +# between the stop above and this start. The marker is already gone, so +# this edge paints normally; the gate matches the one that wrote it +# (default finish and --keep-wifi alike). Ordering pinned by +# tests/test_reset_setup_sh.py. +if [[ "$DO_REBOOT" != "true" && "$DO_POWEROFF" != "true" && "$GIFT_MODE" != "true" ]]; then + rearm_shutdown_splash "the reboot this run asks for may not paint one." +fi # litclock-dev#676 made the handoff fallback RECURRING, so it is now a live # writer of .handoff-complete rather than a one-shot that fired long ago. # The marker removal order below (.setup-complete first) already closes the @@ -703,35 +787,22 @@ if [[ -f "$INSTALL_DIR/env.sh" ]]; then echo -e "${YELLOW}gift language code failed the final shape check; seeding empty${NC}" GIFT_LANGUAGE_CODE="" fi - # In --gift-mode the validated language - # code (shape-gated [a-z0-9-] in the gift arm above — safe to - # interpolate) seeds LITCLOCK_LANGUAGE so EVERY env-reading surface on - # the recipient's device boots in the gifter's chosen language. Plain - # resets leave it empty, which keeps Accept-Language negotiation alive - # on the next first-boot (the litclock-dev#743 empty-seed contract). - # litclock-dev#783 — must cover EVERY env.sh.sample key; comment status is - # copied from the sample and is load-bearing (litclock-dev#762). LITCLOCK_LANGUAGE - # keeps its gift-mode interpolation. - DEFAULTS="# export OPENWEATHERMAP_APIKEY= -export WEATHER_ENABLED=true -export WEATHER_LATITUDE= -export WEATHER_LONGITUDE= -export WEATHER_LOCATION_NAME= -export WEATHER_UNITS=imperial -export WEATHER_LOCATION_MODE=auto -export WEATHER_IP_COUNTRY= -export WEATHER_LAST_IP_GEO_AT= -export WEATHER_TTL=3600 -export ALLOW_NSFW_QUOTES=false -export LITCLOCK_LANGUAGE=$GIFT_LANGUAGE_CODE -export SHOW_DIAGNOSTICS_SHORTCUT=false -export GIFT_MODE_MESSAGE= -export LITCLOCK_RUNTIME_RENDER=false -# export DISPLAY_CLEAR_HOUR=2 -# export LITCLOCK_RENDER_LEAD_S=4 -# export WEATHER_API_TIMEOUT=15 -# export LOG_LEVEL=WARNING -" + # litclock-dev#840 — the body comes from env_sh_defaults() in lib/state.sh, + # the single source shared with first-boot.sh and prepare-for-cloning.sh. + # The key set, the comment status and the litclock-dev#337 A3 MODE=auto default all + # live there; see its contract comment before changing this line. + # + # In --gift-mode the validated language code (shape-gated above, and gated + # AGAIN inside the helper at the interpolation point) seeds + # LITCLOCK_LANGUAGE so EVERY env-reading surface on the recipient's device + # boots in the gifter's chosen language. Plain resets pass it empty, which + # keeps Accept-Language negotiation alive on the next first-boot (the litclock-dev#743 + # empty-seed contract). + # + # The `$'\n'` is REQUIRED, not decoration: command substitution strips + # trailing newlines, and update.sh Phase 3 appends missing sample keys with + # `>>`, which would otherwise splice the first one onto the last line. + DEFAULTS=$(env_sh_defaults "$GIFT_LANGUAGE_CODE")$'\n' if atomic_write_env_sh "$INSTALL_DIR/env.sh" "$DEFAULTS"; then echo -e "${GREEN}done${NC}" else @@ -907,6 +978,20 @@ echo "" if [[ "$GIFT_MODE" == "true" ]]; then if [[ "$ENV_WIPE_FAILED" == "true" ]]; then + # litclock-dev#839 removed exactly this shape — a flag raised at the + # env.sh wipe and read after the WiFi wipe and the hotspot-password + # removal — from prepare-for-cloning.sh, so the divergence is recorded + # here rather than left to be rediscovered as an inconsistency. It is + # acceptable HERE and was not there, for two reasons. (1) Nothing + # downstream of the wipe is undone by reaching this late: gift mode's + # terminal act is the POWER-OFF, which this arm refuses, whereas the + # cloning script's late gate fired only after it had already deleted + # the very things a failed wipe means you must keep (litclock-dev#839). (2) The + # operator is present and reading: this arm leaves the device ON with + # the banner on screen, while the cloning script powers off under an + # operator who has walked away. A re-run is also cheap here — the gift + # unit re-reads the staged files — and the failure is not silent. + # # litclock-dev#393: the env.sh wipe is the load-bearing privacy step for a gift — # it clears the gifter's WEATHER_LATITUDE/LONGITUDE/LOCATION_NAME. It # failed (lock timeout rc=75 or a write error), so stale coordinates may @@ -989,6 +1074,8 @@ elif [[ "$DO_REBOOT" == "true" ]]; then # on Bookworm anyway). Cleaner systemd integration; not a race fix. systemctl reboot else + # litclock-dev#833: litclock-shutdown.service was re-armed in Step 1, + # right after its stop edge was consumed, so the reboot below paints. echo "Reboot to enter setup mode:" echo " sudo reboot" fi diff --git a/scripts/shutdown-splash.sh b/scripts/shutdown-splash.sh index 3fa8076..7c9c8a2 100755 --- a/scripts/shutdown-splash.sh +++ b/scripts/shutdown-splash.sh @@ -32,33 +32,84 @@ PYTHON="$INSTALL_DIR/venv/bin/python3" # PERSIST on the bistable e-ink through shutdown — so it touches this marker # (via sudo) and we exit without painting anything. The marker lives directly # in root-owned /run, NOT in pi-owned /run/litclock/ next to the action hint, -# so that creating it requires root. Note: on images carrying the 010 -# passwordless-sudo grant (all current images), a pi-level process can still -# `sudo touch` it — but such a process already has full root, so the marker -# adds no new exposure there; the root-owned path only becomes a real boundary -# once 010 is dropped. tmpfs, so it self-clears on the next boot; no stale +# so that creating it requires root. Note: every shipped image carries the 010 +# passwordless-sudo grant — kept deliberately, since the drop planned under +# litclock-dev#387/litclock-dev#82 was reversed 2026-07-12 — so a pi-level process can `sudo touch` it +# today. Such a process already has full root, so the marker adds no new +# exposure there; the root-owned path is the correct shape regardless (it is +# what keeps this out of the scoped 020 allowlist, where granting it WOULD hand +# pi the gift-welcome mute), and it would become a hard boundary if the posture +# were ever revisited. tmpfs, so it self-clears on the next boot; no stale # suppression can survive. +# litclock-dev#861 — the echo below: without it a suppressed shutdown and a +# script that died on its first command leave the same (empty) journal. if [[ -f /run/litclock-splash-suppress ]] && [[ ! -L /run/litclock-splash-suppress ]]; then + echo "shutdown splash: suppressed by /run/litclock-splash-suppress" exit 0 fi SHUTDOWN_ACTION="" +ACTION_SOURCE="" if [[ -f /etc/litclock/.welcome-mode ]]; then SHUTDOWN_ACTION="welcome" + ACTION_SOURCE="welcome-marker" elif [[ -f /run/litclock/shutdown-action ]] && [[ ! -L /run/litclock/shutdown-action ]]; then raw="$(timeout 1 head -c 32 /run/litclock/shutdown-action 2>/dev/null | tr -d '[:space:]')" case "$raw" in - reboot|poweroff) SHUTDOWN_ACTION="$raw" ;; + reboot|poweroff) SHUTDOWN_ACTION="$raw"; ACTION_SOURCE="hint-file" ;; esac fi if [[ -z "$SHUTDOWN_ACTION" ]]; then - if systemctl list-jobs 2>/dev/null | grep -q "reboot.target"; then + # litclock-dev#862 — ASK AS ROOT, and say so when the question fails. + # + # `systemctl` talks to PID 1 over the D-Bus system bus, and falls back to + # PID 1's private socket when the bus is gone. That socket is + # `srwx------ root root`, so the fallback is root-only — and this script + # runs `User=pi`. Measured on the bench 2026-09-18: ~10s into a shutdown + # `dbus.service` is already stopped, so as `pi` every systemctl call + # returns `Failed to connect to bus: Connection refused`, while + # `sudo -n systemctl list-jobs` answers correctly and shows + # `reboot.target start waiting`. + # + # That 10s is not hypothetical: litclock-dev#856 made this ExecStop wait + # for the boot splash's stop job (up to its TimeoutStopSec=10s), so a + # reboot taken during the ~11s boot-splash window landed here EVERY time + # with no answer and painted the poweroff splash on a device that was + # coming straight back up. Before litclock-dev#856 this ran at T+0 and worked only + # because dbus happened to still be alive — a race, not a mechanism. + # + # Root first, then the unprivileged call (which still answers early in a + # shutdown, and is the only one that works if 010_pi-nopasswd is ever + # withdrawn), then give up LOUDLY. Both are bounded: sudo and systemctl + # each have to be able to fail fast inside a 30s TimeoutStopSec that also + # has to fit a ~7s paint. + JOBS="" + if JOBS=$(timeout 5 sudo -n systemctl list-jobs 2>/dev/null) && [[ -n "$JOBS" ]]; then + ACTION_SOURCE="list-jobs(root)" + elif JOBS=$(timeout 5 systemctl list-jobs 2>/dev/null) && [[ -n "$JOBS" ]]; then + ACTION_SOURCE="list-jobs(unprivileged)" + else + JOBS="" + ACTION_SOURCE="list-jobs(UNAVAILABLE)" + fi + # litclock-dev#861 — an empty/failed probe and a genuine poweroff used to + # be the same `poweroff`, which is exactly why litclock-dev#862 sat unnoticed. The + # ANSWER still defaults to poweroff (painting "Restarting…" on a device + # that is really powering off is the worse error, and the PWA factory + # reset relies on this arm), but the SOURCE now says which one happened. + if [[ -n "$JOBS" ]] && grep -q "reboot\.target" <<<"$JOBS"; then SHUTDOWN_ACTION="reboot" else SHUTDOWN_ACTION="poweroff" fi fi +# litclock-dev#861 — journald is the only diagnostic channel on this device, +# and the painted variant is otherwise unrecoverable: the next boot repaints +# over it. Name the tier too, because a wrong answer and an unasked question +# look identical on the glass. +echo "shutdown splash: action=${SHUTDOWN_ACTION} source=${ACTION_SOURCE}" + case "$SHUTDOWN_ACTION" in welcome) # Gift mode: device was prepped via `reset-setup.sh --gift-mode` and is diff --git a/scripts/update.sh b/scripts/update.sh index 61bdbac..b247d02 100755 --- a/scripts/update.sh +++ b/scripts/update.sh @@ -107,7 +107,7 @@ if [[ "${LITCLOCK_UPDATE_LOCK_HELD:-0}" != "1" ]] && command -v flock >/dev/null # cleanup race is the worse of the two, so it stays. The leak side is # bounded elsewhere: the one gate that could leak a helper is under # coreutils `timeout` now, and the lock design itself is the follow-up - # on litclock-dev#835. + # on litclock-dev#847 (item 2; it began on litclock-dev#835). flock -n -E 75 "$LITCLOCK_UPDATE_LOCK_FILE" "$0" "$@" _rc=$? if [[ "$_rc" == "75" ]]; then @@ -203,6 +203,25 @@ LEGACY_UPDATE_CHECK_CACHE_FILE="$STATE_DIR/update-check.json" # offline-graceful-exit window where Phase 1 already cleared lkg-sha but # no new LKG was recorded yet. LAST_UPDATE_FILE="$STATE_DIR/last-update.json" +# litclock-dev#847 item 1 — the NEGATIVE-RESULT MEMO for the litclock-dev#531 +# runtime-render validation. Phase 4.5's KEEP arm writes it when the check is +# DEFERRED (not enough of this run's systemd budget left), TIMES OUT or FAILS, +# and removes it when the check passes or the marker is already present. +# JSON, one object: {result, rc, reason, sha, at_unix}. Persistent (SD-backed, +# like update-failed) because "why is this clock on the PNG tier" gets asked +# weeks after the journal line rotated. control_server's /api/status mirrors +# it as `runtime_render_validation` — the same marker-to-payload path +# PHASE3_SKIPPED_FILE takes. It records; it does not gate. A memo never skips +# the next attempt: rc=1 covers both a measurement mismatch and an import +# error from a half-built wheel, and skipping on the second would strand a +# device on the PNG tier until the next release. +RUNTIME_VALIDATION_MEMO_FILE="$STATE_DIR/runtime-render-validation.json" +# litclock-dev#845 — where the revert arms re-install the clock units from +# (_reinstall_clock_units_from_tree). Overridable only so the executed tests +# can point the arms at a fake directory; Phase 5's loop names +# /etc/systemd/system literally, and tests/test_update_sh.py pins its +# pre-existence discriminator on that literal. +SYSTEMD_UNIT_DIR="${LITCLOCK_SYSTEMD_UNIT_DIR:-/etc/systemd/system}" # Resolve the target SHA for this update cycle. # Path: /repos/.../tags → highest semver vX.Y.Z → git fetch → git rev-list -n 1 @@ -427,7 +446,8 @@ fi # pre-NTP system time can read EARLIER than the marker's mtime; the age goes # negative, compares as "recent", and the migration is refused on exactly # the long-provisioned device the condition was meant to protect. This file -# already documents that hazard 1100 lines below (the pre-1970 stamp note). +# already documents that hazard in _litclock_persist_last_update (the litclock-dev#342 +# I2 pre-1970 stamp note). # # What actually identifies a pre-PR2 device is that its INSTALLED unit predates # the gate. This block runs long before Phase 5 reinstalls units, so the old @@ -566,7 +586,8 @@ _LS_REMOTE_TIMEOUT_S="${LITCLOCK_LS_REMOTE_TIMEOUT_S:-30}" # timeout's to kill. git waits for its helpers and git-remote-https honours # TERM, so both shapes are synthetic here; the consequence that would # matter (a leaked descendant holding the update lock) is the lock-design -# follow-up on litclock-dev#835, see the flock comment at the top. +# follow-up on litclock-dev#847 (item 2; this residual is its item 3), see +# the flock comment at the top. _remote_reachable() { timeout -k 5 "$_LS_REMOTE_TIMEOUT_S" git ls-remote --exit-code origin &>/dev/null } @@ -635,7 +656,7 @@ update_status_init "$OLD_SHA" # than a stopped timer, and the status file already says manual # recovery is needed. It is NOT covered by the bootcheck/LKG chain — # that chain asks "did this BOOT paint once", and a mid-uptime update -# has always painted before it ran. +# has always painted before it ran (litclock-dev#847 item 6). _LITCLOCK_UPDATE_FINALIZED=0 _LITCLOCK_UPDATE_CLEANED=0 # The one cleanup, IDEMPOTENT, reachable from both the EXIT trap and the @@ -680,7 +701,8 @@ _litclock_update_cleanup() { 2>/dev/null || true fi # litclock-dev#274 cleanup: Phase 3 side-channel tempfile that mapfile reads - # ADDED_VARS from. Cleared on normal exit at line ~556, but a SIGKILL + # ADDED_VARS from. Cleared at the end of Phase 3 on a normal run (the + # `rm -f "$_PHASE3_ADDED_FILE"` after the skip-marker clear), but a SIGKILL # mid-Phase-3 (power loss, OOM) would otherwise leak it across runs. [ -n "${_PHASE3_ADDED_FILE:-}" ] && rm -f "$_PHASE3_ADDED_FILE" 2>/dev/null return 0 @@ -823,24 +845,46 @@ elif declare -F github_api_latest_release_tag >/dev/null 2>&1; then log_warn "Resolver returned a malformed value; treating as offline" TARGET_SHA="" fi - # blocked-sha suppression (litclock-bootcheck): refuse to re-install a - # Release SHA that bootcheck reverted away from, until a NEWER release - # (a different SHA) supersedes it. Without this the weekly timer would - # re-brick the device on the same bad release the day after a recovery. + # blocked-sha suppression: refuse to re-install a Release SHA a previous + # run reverted away from, until a NEWER release (a different SHA) + # supersedes it. Two writers, two failure modes: litclock-bootcheck blocks + # a release that bricked the boot, and the smoke-revert arm blocks one that + # failed the gate (litclock-dev#865). Without this the weekly timer would + # re-brick — or re-revert — on the same bad release, every week. if [[ -n "$TARGET_SHA" && -f "$BLOCKED_SHA_FILE" ]]; then _blocked=$(read_sha_file "$BLOCKED_SHA_FILE") if [[ -n "$_blocked" && "$TARGET_SHA" == "$_blocked" ]]; then - log_warn "Latest Release SHA $TARGET_SHA is blocked (bootcheck reverted from it) — skipping update" + log_warn "Latest Release SHA $TARGET_SHA is blocked (a previous run reverted from it) — skipping update" log_info "Clock continues on the recovered SHA. A newer Release will clear the block." # Terminal CLEAN status: this is a deliberate no-op, not a failure. # Without finalizing, the EXIT trap would stamp failed_unrecovered # (state was set to running by Phase 1) and the PWA would raise a # false "manual recovery needed" alarm every single weekly tick until # a newer release ships. The clock is running fine on the recovered - # SHA, so "complete" is the honest state. - update_status_complete 2>/dev/null || true - _LITCLOCK_UPDATE_FINALIZED=1 - sudo systemctl start litclock.timer 2>/dev/null || true + # SHA, so "complete" is the honest state for THE RUN. + # + # ANCHOR: no-op-early-out + # RE-ARM FIRST, FINALIZE ONLY IF IT WORKED (litclock-dev#854 review). Finalizing + # ahead of the re-arm disarms the EXIT trap while the clock is still + # stopped: a signal in that window — or the start below failing — + # would leave a stopped timer with neither the trap's fallback + # re-arm nor its stamp. A failed start falls through unfinalized on + # purpose: the trap retries the re-arm (bounded, non-interactive) + # and stamps failed_unrecovered, which is the honest state for a + # clock that is not ticking. + # + # "complete" NEVER means "installed something" — `to_version` is + # what answers that, and this run exits before Phase 2 ever sets + # it. The Status hero's "Last update" row requires a non-null + # to_version for exactly this reason (see _payload_to_last_update + # in src/control_server/routes/status.py), so a no-op tick cannot + # present itself as a fresh install. + if sudo systemctl start litclock.timer 2>/dev/null; then + update_status_complete 2>/dev/null || true + _LITCLOCK_UPDATE_FINALIZED=1 + else + log_warn "could not re-arm litclock.timer — leaving this run unfinalized so the EXIT trap retries the re-arm and stamps" + fi exit 0 fi fi @@ -850,8 +894,21 @@ if [[ -z "$TARGET_SHA" ]]; then if declare -F github_api_latest_release_tag >/dev/null 2>&1; then log_warn "Could not resolve a blessed Release SHA — exiting cleanly (offline or no releases yet)" log_info "Clock continues on its pinned SHA. Next timer fire will retry." + # Terminal CLEAN status (litclock-dev#847 item 4). This `exit 0` used + # to leave the run unfinalized, so the EXIT trap stamped + # failed_unrecovered — "manual recovery needed" in the PWA — for an + # ordinary offline tick, on every tick the device spent offline. The + # blocked-sha arm above is the same shape and the same reasoning; see + # ANCHOR: no-op-early-out there for why the timer is re-armed BEFORE + # the finalize, and why "complete" here cannot be read as "installed" + # (nothing sets to_version before Phase 2). # Restart the timer so the clock keeps ticking — we stopped it in Phase 1. - sudo systemctl start litclock.timer 2>/dev/null || true + if sudo systemctl start litclock.timer 2>/dev/null; then + update_status_complete 2>/dev/null || true + _LITCLOCK_UPDATE_FINALIZED=1 + else + log_warn "could not re-arm litclock.timer — leaving this run unfinalized so the EXIT trap retries the re-arm and stamps" + fi exit 0 fi # No lib sourced (fresh-image run before github_api.sh lands) — fall back to @@ -900,6 +957,7 @@ else fi git submodule update --init --recursive +# ANCHOR: reexec-guard # Re-exec if update.sh itself changed — bash holds a stale fd after # git replaces the file, which can cause it to read garbled content # from the new file at the old byte offset. @@ -953,6 +1011,7 @@ if [[ ${#REMOVED[@]} -gt 0 ]]; then log_info "Removed stale files: ${REMOVED[*]}" fi +# ANCHOR: runtime-marker-revoke # litclock-dev#604 — invalidate the runtime-render validation marker when # any of its proof inputs changed in this update. The marker records that # `validate_measurement.py check --stamp` proved THIS freetype reproduces @@ -1047,10 +1106,11 @@ ADDED_VARS=() # with_env_lock) so any bash array it builds is local to that subshell. # Write added var names to a temp file inside the helper, then read # them back into ADDED_VARS in the parent shell after the lock releases -# so the end-of-update summary still lists them. Cleanup runs at line -# ~556 on normal exit; the existing _litclock_update_trap (line 298) -# also `rm -f`s this on SIGKILL/SIGTERM/power loss so the tempfile -# doesn't accumulate in /tmp across failed weekly updates. +# so the end-of-update summary still lists them. Cleanup is the +# `rm -f "$_PHASE3_ADDED_FILE"` at the end of this phase on a normal run; +# _litclock_update_cleanup (the EXIT/signal trap) also `rm -f`s it on +# SIGTERM/power loss so the tempfile doesn't accumulate in /tmp across +# failed weekly updates. _PHASE3_ADDED_FILE=$(mktemp 2>/dev/null) || _PHASE3_ADDED_FILE="/tmp/litclock-update-added.$$" _phase3_merge_sample() { @@ -1136,6 +1196,101 @@ unset _phase3_marker_just_written rm -f "$_PHASE3_ADDED_FILE" 2>/dev/null unset _PHASE3_ADDED_FILE +# --- litclock-dev#845 clock-unit reinstall BEGIN --- +# ANCHOR: reinstall-clock-units +# Re-install litclock.service and litclock.timer from the tree at HEAD when +# they differ from the installed copies, then daemon-reload. Both revert +# arms call this after their `git reset --hard "$REVERT_SHA"` and BEFORE +# their `systemctl start`s (Phase 4 pip failure, Phase 4.5 smoke failure). +# +# Why: litclock-dev#762 coupled systemd/litclock.timer (OnCalendar=*:*:56) to painter +# code that renders 4s ahead — either half without the other paints the +# wrong minute. The unit copy loop is Phase 5, AFTER both revert arms' +# `exit 1`, so a revert arm used to restart whatever timer was installed +# against whatever code it had just reset to. On a NORMAL failed update the +# two match: the installed units came from the last successful Phase 5, +# which installed OLD_SHA == REVERT_SHA. In bootcheck ROLLBACK_MODE they do +# not: the bricking tick already installed the :56 timer, REVERT_SHA is the +# LKG, and if the rollback run then fails pip (offline, piwheels down) or +# smoke, the device was left with the :56 timer driving un-offset LKG code — +# the previous minute's quote on every tick until a later update succeeded +# (v0.227.0 port review, adversarial pass). +# +# ALWAYS, not only in ROLLBACK_MODE. The `cmp -s` gate makes the normal arm +# a no-op by construction, and a conditional would leave the invariant +# ("the installed clock units match the checked-out code") unenforced on the +# arm that runs a hundred times more often. The one way a normal arm can +# see a difference is an operator's hand edit to the installed main unit +# file, and Phase 5 already overwrites those on every successful update — +# drop-ins (`.d/*.conf`) are the supported override and survive here +# exactly as they survive Phase 5. +# +# ONLY these two units: they are the pair the revert arm restarts. The +# service travels with the timer because the same coupling can land there +# next (an ExecStart argument, an Environment= for the lead). Every other +# unit stays as Phase 5 last installed it — in particular this run's own +# litclock-update.service, whose TimeoutStartSec a pre-litclock-dev#835 LKG carries +# at 600: re-installing and reloading it from inside the run it governs is +# not a shape a revert arm should reason about. +# +# Not factored out of Phase 5's loop: that body is enable-policy (new-unit +# detection, `systemctl enable`) the revert arms must not run, and the one +# line they share is a `cp`. +# +# PRIVILEGE, stated because it is not granted by sudoers/020 (litclock-dev#854 review). +# `sudo cp` and `sudo systemctl daemon-reload` work through +# /etc/sudoers.d/010_pi-nopasswd (NOPASSWD: ALL), exactly as Phase 5's unit +# installs do. 010 is KEPT deliberately: dropping it was planned under +# litclock-dev#548 / litclock-dev#387 / litclock-dev#82 and REVERSED by the owner on 2026-07-12, so +# this is a settled arrangement, not a residual waiting on that drop. +# It is still NOT a gap to close by adding the lines to 020: this service is +# User=pi and pi OWNS $INSTALL_DIR, so `cp +# /etc/systemd/system/...` granted to pi IS root, which is why 020 has never +# carried it and must not start. If the posture is ever revisited (it reopens +# only behind a non-default per-device password), BOTH this and Phase 5 need a +# root-owned installer helper — the +# /usr/local/lib/litclock/litclock-set-timezone shape, where root-owned code +# re-validates what pi passes — not a broader sudoers line. (The timer-start +# grants 020 DOES carry name a fixed unit with no path argument, so they +# hand pi nothing it can redirect; that is the scoped default for anything +# that does not need the blanket grant.) Tracked on litclock-dev#847. +# +# Returns non-zero when a unit still differs after the attempt, so the arms +# can say so loudly. Never fatal: the arm's own `exit 1` and status stamp are +# the operator signal, and a revert that restored the CODE is still a revert. +_reinstall_clock_units_from_tree() { + local name unit installed rc=0 + for name in litclock.service litclock.timer; do + unit="$INSTALL_DIR/systemd/$name" + installed="$SYSTEMD_UNIT_DIR/$name" + # Regular files only — the tree is pi-writable and `cp` under sudo + # dereferences a planted symlink (the tmpfiles install has the same + # guard). + [[ -f "$unit" && ! -L "$unit" ]] || continue + cmp -s "$unit" "$installed" && continue + if sudo cp "$unit" "$installed" 2>/dev/null; then + log_info "[revert] re-installed $name from the reverted tree so the unit matches the code it drives (litclock-dev#845)" + # RELOAD PER COPY, not once after the loop (litclock-dev#854 review). A TERM + # between the copy and a trailing reload would send the EXIT trap's + # re-arm into systemd's PREVIOUSLY loaded definition — the window + # this whole helper exists to close, reopened a few lines wide. A + # reload is idempotent and cheap, so paying for it twice removes + # the window entirely. + sudo systemctl daemon-reload 2>/dev/null \ + || log_warn "[revert] daemon-reload failed after re-installing $name; systemd still holds the previous definition (litclock-dev#845)" + else + log_warn "[revert] could not re-install $name from the reverted tree (litclock-dev#845)" + fi + # VERIFY, do not assume: `sudo cp` can report success under a + # read-only /etc remount or a full disk on some paths, and it fails + # outright wherever the blanket grant is absent (a hand-built dev box, + # say). Compare again rather than trusting rc. + cmp -s "$unit" "$installed" || rc=1 + done + return "$rc" +} +# --- litclock-dev#845 clock-unit reinstall END --- + # ─── Phase 4: Update Python packages ───────────────────────────────── REQUIREMENTS="$INSTALL_DIR/requirements.txt" @@ -1231,6 +1386,7 @@ if [[ "$NEED_PIP" == "true" ]]; then # The PWA's failed_unrecovered copy is "manual recovery needed", # not "clock is dead" — it's the correct user-facing string for # "venv state uncertain". + # ANCHOR: pip-failure-arm log_error "pip install failed — reverting code to $REVERT_SHA (venv state uncertain)" rm -f "$REQUIREMENTS_FILTERED" # Capture revert exit codes so we can distinguish "code reverted, @@ -1240,8 +1396,10 @@ if [[ "$NEED_PIP" == "true" ]]; then # REVERT_SHA == OLD_SHA in a normal update (unchanged behavior); in # bootcheck rollback mode it is the LKG target, so a failure here # stays on the last-known-good rather than falling back to the bad code. + # ANCHOR: revert-is-self-modifying-and-safe # SELF-MODIFYING, and safe — litclock-dev#773 item 3 asked why this has - # no re-exec guard when the pull path at :758 does. Measured with strace + # no re-exec guard when the pull path does (ANCHOR: reexec-guard, Phase + # 2). The smoke-failure arm points here: ONE copy. Measured with strace # against git 2.43.0: `git reset --hard` does `unlink()` then # `openat(O_WRONLY|O_CREAT|O_EXCL)`. O_EXCL means git NEVER writes into # the existing inode, so the bash executing this file keeps reading the @@ -1269,6 +1427,7 @@ if [[ "$NEED_PIP" == "true" ]]; then if [ "${PIPESTATUS[0]:-0}" -ne 0 ]; then REVERT_OK=0 fi + # ANCHOR: hash-delete-sets-need-pip # Delete the pip hash so the next timer fire re-attempts pip install # (without this, the next run sees an unchanged requirements.txt # under OLD_SHA but the indeterminate venv claims it matches NEW_SHA's @@ -1281,6 +1440,14 @@ if [[ "$NEED_PIP" == "true" ]]; then if ! atomic_write_file "$UPDATE_FAILED_FILE" ""; then log_warn "Could not write $UPDATE_FAILED_FILE — corner glyph will not render" fi + # litclock-dev#845 — the installed clock units must match the code we + # just reset to, BEFORE the `systemctl start`s below re-arm them + # (ANCHOR: reinstall-clock-units). A failure here does not change the + # verdict — this branch already reports failed_unrecovered — but it + # must not be silent: the clock is about to be restarted against a + # unit that does not match its code. + _reinstall_clock_units_from_tree \ + || log_error "[revert] the installed clock units do NOT match the reverted tree; the timer may drive code it was not paired with (litclock-dev#845)" # Still attempt to start the clock — even with an uncertain venv, # most python packages are forward-compatible and the clock has a # fighting chance of running. We do NOT use the result of this @@ -1544,24 +1711,127 @@ _update_elapsed_seconds() { echo $(( now_s - start_us / 1000000 )) } +# litclock-dev#865 — remember the release a SMOKE failure reverted away from, +# so the next weekly tick skips it instead of re-installing, re-failing and +# reverting again, forever. +# +# The suppression already exists and is already correct: bootcheck writes +# $BLOCKED_SHA_FILE, Phase 1 skips a TARGET_SHA that matches it (ANCHOR: +# no-op-early-out), and a successful apply in normal mode clears it because a +# different SHA means a genuinely new release. Only the WRITE was missing on +# this path, which update.sh's own litclock-dev#763 comment has said in prose since +# v0.227.0 ("nothing writes a blocked-sha on that path — only bootcheck does"). +# Confirmed on hardware 2026-09-18: two consecutive ticks against the same bad +# release each re-fetched it, re-applied it, failed the same probe and reverted, +# with the clock stopped for the duration and a full pip install each time +# (the revert deletes HASH_FILE). +# +# SMOKE FAILURES ONLY — deliberately not wired to the pip-failure or +# git-revert arms. A failed smoke is a property of the release (a truncated +# catalog is truncated on every device, forever); a failed pip is usually the +# network, and blocking a good release on a transient wheel fetch would be a +# far worse bug than the one this fixes. +# +# NOT IN ROLLBACK MODE. There TARGET_SHA is the LKG target bootcheck is +# recovering TO, and blocking it would pin the device off its own last-known- +# good. bootcheck owns the file in that mode and deliberately keeps it (see +# the Phase 7 clear, which makes the same distinction). +_block_reverted_release() { + # $1 — the SHA that failed smoke (the target we installed, then reverted + # away from), NOT the SHA we reverted TO. Blocking the latter would pin the + # device off the only build known to work on it. + local bad="${1-}" + if [[ "${ROLLBACK_MODE:-0}" -eq 1 ]]; then + log_info "[revert] rollback mode — leaving blocked-sha to bootcheck (litclock-dev#865)" + return 0 + fi + # NOT when the gate could not RUN. smoke_no_interpreter means $PYTHON was + # absent after Phase 4 — "nothing was verified" (litclock-dev#773), which is a fault + # in THIS DEVICE's venv, not evidence about the release. Blocking on it + # would pin a device off a release that is very likely fine, and would + # outlive the fault: the Phase-4 guard rebuilds a broken venv on the next + # tick, so the device heals itself and would then refuse the release it + # could now install. Same reasoning as the pip-failure arm, one step later. + if [[ "${smoke_no_interpreter:-0}" -eq 1 ]]; then + log_info "[revert] gate could not run (no interpreter) — not blaming the release (litclock-dev#865)" + return 0 + fi + if [[ ! "$bad" =~ ^[0-9a-f]{40}$ ]]; then + log_warn "[revert] no usable target SHA to block; next tick will retry this release (litclock-dev#865)" + return 0 + fi + if atomic_write_file "$BLOCKED_SHA_FILE" "$bad"; then + log_info "[revert] blocked $bad — the next tick skips it until a newer Release supersedes it (litclock-dev#865)" + else + log_warn "[revert] could not write $BLOCKED_SHA_FILE; next tick will re-install and re-revert this release (litclock-dev#865)" + fi +} + +# litclock-dev#847 item 1 — write the negative-result memo (see +# RUNTIME_VALIDATION_MEMO_FILE). $1 result: deferred|timeout|failed; +# $2 the validator's rc, or empty when it never ran; $3 the reason, prose. +# jq builds the object so the reason cannot break the JSON, and the file +# lands via the same atomic writer as every other state marker. Best-effort: +# a failed memo is a warning, never an update failure. +_runtime_validation_memo_write() { + local result="$1" rc="${2:-}" reason="${3:-}" sha json + if ! command -v jq >/dev/null 2>&1; then + log_warn "jq missing — not recording the runtime-render validation result (litclock-dev#847)" + return 0 + fi + sha=$(git rev-parse HEAD 2>/dev/null || echo "") + json=$(jq -nc --arg result "$result" --arg rc "$rc" --arg reason "$reason" \ + --arg sha "$sha" --arg at "$(date +%s 2>/dev/null || echo 0)" \ + '{result: $result, + rc: ($rc | if . == "" then null else tonumber end), + reason: $reason, + sha: ($sha | if . == "" then null else . end), + at_unix: ($at | tonumber)}' 2>/dev/null) || json="" + if [[ -z "$json" ]] || ! atomic_write_file "$RUNTIME_VALIDATION_MEMO_FILE" "$json"; then + log_warn "Could not write $RUNTIME_VALIDATION_MEMO_FILE — the PWA will not surface this validation result (litclock-dev#847)" + return 0 + fi + # READABLE BY THE PWA (litclock-dev#854 review). atomic_write_file stages with mktemp, + # which is 0600 and owned by whoever runs the script — so a maintainer's + # `sudo ./scripts/update.sh` leaves a 0600 root:root memo that the pi-user + # control_server cannot open, and /api/status silently reports null. Same + # best-effort fix _litclock_persist_last_update already applies to + # last-update.json, and the same shape: the memo carries no secret (a + # result token, an rc, a reason and a SHA), so 0644 is right for a file + # whose whole purpose is to be read by another service. + # One line each, not a `\`-continuation: tests/test_update_sh.py's chmod + # inventory parses this file line by line and fails loudly on a shape it + # cannot classify (it caught the continuation form). + chmod 0644 "$RUNTIME_VALIDATION_MEMO_FILE" 2>/dev/null || sudo chmod 0644 "$RUNTIME_VALIDATION_MEMO_FILE" 2>/dev/null || true + chown pi:pi "$RUNTIME_VALIDATION_MEMO_FILE" 2>/dev/null || sudo chown pi:pi "$RUNTIME_VALIDATION_MEMO_FILE" 2>/dev/null || true + return 0 +} +# One deferral: the log line the operator reads, then the memo. The +# reason is the log text minus the fixed prefix, so the journal and the +# memo cannot disagree. Returns 1, the guard's own "does not fit" answer. +_defer_runtime_validation() { + log_info "deferring runtime-render validation to the next update: $1 (litclock-dev#835)" + _runtime_validation_memo_write deferred "" "$1" + return 1 +} + # 120s of margin covers Phase 5-7 (unit copies, daemon-reload, the first # paint) after a validator that used its whole bound. It is a reserve, not -# an enforced bound on that work. Deferral is logged but never persisted: a -# device that NEVER has budget never validates, stays on the PNG tier, and -# the marker's absence is the only trace — the safe direction, and the -# negative-result memo still open on litclock-dev#835 is where a durable -# reason belongs. +# an enforced bound on that work. A deferral is logged AND memoed +# (litclock-dev#847 item 1): a device that NEVER has budget never validates +# and stays on the PNG tier — the safe direction — and the memo is how that +# stops being invisible. _validation_fits_remaining_budget() { local elapsed budget rc elapsed=$(_update_elapsed_seconds); rc=$? if [[ $rc -eq 2 ]]; then return 0 elif [[ $rc -ne 0 ]]; then - log_info "deferring runtime-render validation to the next update: cannot determine how much of this run's systemd budget is left (litclock-dev#835)" + _defer_runtime_validation "cannot determine how much of this run's systemd budget is left" return 1 fi if ! budget=$(_update_budget_seconds); then - log_info "deferring runtime-render validation to the next update: cannot read the unit's effective TimeoutStartSec (litclock-dev#835)" + _defer_runtime_validation "cannot read the unit's effective TimeoutStartSec" return 1 fi # 0 here means UNLIMITED (the helper maps infinity to 0 and rounds a @@ -1570,7 +1840,7 @@ _validation_fits_remaining_budget() { # so treating the two alike is harmless in that direction only). [[ "$budget" -eq 0 ]] && return 0 if (( elapsed + VALIDATOR_TIMEOUT_S + VALIDATOR_BUDGET_RESERVE_S > budget )); then - log_info "deferring runtime-render validation to the next update: ${elapsed}s of this run's ${budget}s budget are gone and the check needs up to ${VALIDATOR_TIMEOUT_S}s plus a ${VALIDATOR_BUDGET_RESERVE_S}s reserve for the phases after it, with litclock.timer stopped (litclock-dev#835)" + _defer_runtime_validation "${elapsed}s of this run's ${budget}s budget are gone and the check needs up to ${VALIDATOR_TIMEOUT_S}s plus a ${VALIDATOR_BUDGET_RESERVE_S}s reserve for the phases after it, with litclock.timer stopped" return 1 fi return 0 @@ -1587,9 +1857,10 @@ if [[ "$smoke_rc" -eq 0 ]]; then # script, and there is no install.sh. `_runtime_render_enabled()` requires # the flag AND the marker, so on every fielded device the flag was inert # and the runtime renderer was unreachable — shipped, soaked, and dead. - # The block ~500 lines above only REVOKES it (litclock-dev#604) when an update - # changes the proof inputs, which without a re-stamp is a one-way ratchet - # down to the pre-rendered tier forever. + # The revoke block in Phase 2b (ANCHOR: runtime-marker-revoke) only + # REVOKES it (litclock-dev#604) when an update changes the proof inputs, which + # without a re-stamp is a one-way ratchet down to the pre-rendered tier + # forever. # # POSITION IS LOAD-BEARING, three ways: # * AFTER Phase 4, so the venv — and the freetype-py wheel whose bundled @@ -1613,6 +1884,14 @@ if [[ "$smoke_rc" -eq 0 ]]; then # has just removed the marker. Spending minutes re-earning it here, with # litclock.timer still stopped, is the opposite of recovery. The next # normal update re-stamps. + # + # A PRESENT marker supersedes any negative-result memo (litclock-dev#847 + # item 1): the device validated since — by a later tick, or by an + # operator running `check --stamp` by hand — and a memo left behind would + # have the PWA report a failure the marker contradicts. + if [[ -f "$RUNTIME_MARKER" ]]; then + atomic_remove_file "$RUNTIME_VALIDATION_MEMO_FILE" + fi if [[ ! -f "$RUNTIME_MARKER" && -x "$PYTHON" && "${ROLLBACK_MODE:-0}" -ne 1 ]] && _validation_fits_remaining_budget; then log_info "Validating GD-exact measurement for the runtime renderer..." # Re-arm the LKG writer's grace window first: it is 900s from the @@ -1631,8 +1910,17 @@ if [[ "$smoke_rc" -eq 0 ]]; then _validate_rc="${PIPESTATUS[0]}" if [[ "$_validate_rc" -eq 0 && -f "$RUNTIME_MARKER" ]]; then log_info "runtime-render validation PASSED — marker stamped; LITCLOCK_RUNTIME_RENDER will be honored" + atomic_remove_file "$RUNTIME_VALIDATION_MEMO_FILE" else log_info "runtime-render validation did not pass on this device (rc=$_validate_rc) — staying on pre-rendered images (this is not an update failure)" + # The memo (litclock-dev#847 item 1). 124 is coreutils timeout's + # own status, so "timeout" is told apart from a validator that + # ran to a verdict; everything else is "failed" with its rc. + if [[ "$_validate_rc" -eq 124 ]]; then + _runtime_validation_memo_write timeout "$_validate_rc" "the validator did not finish within ${VALIDATOR_TIMEOUT_S}s" + else + _runtime_validation_memo_write failed "$_validate_rc" "tools/validate_measurement.py check --stamp exited $_validate_rc" + fi # A validator killed by `timeout` leaves its mkstemp litter next # to the marker; only the exact marker name is gitignored. The # glob is Python's tempfile scheme exactly — `.` plus @@ -1649,24 +1937,14 @@ else log_error "Smoke test failed (exit $smoke_rc) — reverting to $REVERT_SHA" # REVERT_SHA == OLD_SHA normally; in bootcheck rollback mode it is the # LKG target so a failed LKG smoke stays on LKG (never the bad code). - # SELF-MODIFYING, and safe — litclock-dev#773 item 3 asked why this has - # no re-exec guard when the pull path at :758 does. Measured with strace - # against git 2.43.0: `git reset --hard` does `unlink()` then - # `openat(O_WRONLY|O_CREAT|O_EXCL)`. O_EXCL means git NEVER writes into - # the existing inode, so the bash executing this file keeps reading the - # PRE-reset content through its already-open fd (verified end to end: a - # script that git-resets itself mid-block, forks, and falls through to - # top level ran every remaining statement correctly, inode 4906177 -> - # 4906189). The stale-fd hazard the pull path guards needs an IN-PLACE - # rewrite; git never does one. - # - # The pull path re-execs for a different reason anyway: it CONTINUES and - # must run the new code against the new tree. A revert arm terminates, - # so it never needs the new code. There is no asymmetry to defend. + # SELF-MODIFYING, and safe, for exactly the reason the pip-failure arm + # gives — see ANCHOR: revert-is-self-modifying-and-safe (Phase 4) for the + # strace measurement. One copy, on purpose, so the two cannot drift. git reset --hard "$REVERT_SHA" 2>&1 | sed 's/^/[revert] /' || true git submodule update --init --recursive 2>&1 | sed 's/^/[revert] /' || true # Delete the pip hash so the next run re-runs `pip install` (it sets - # NEED_PIP; it does NOT by itself recreate the venv — see :1236 and the + # NEED_PIP; it does NOT by itself recreate the venv — see ANCHOR: + # hash-delete-sets-need-pip in the pip-failure arm and the # missing-interpreter arm below). Otherwise a revert could leave the venv # half-upgraded while the hash claims it matches. rm -f "$HASH_FILE" @@ -1674,6 +1952,25 @@ else if ! atomic_write_file "$UPDATE_FAILED_FILE" ""; then log_warn "Could not write $UPDATE_FAILED_FILE — corner glyph will not render" fi + # litclock-dev#865 — block the release we just reverted FROM, not the one + # we reverted TO. See _block_reverted_release. + # + # `${TARGET_SHA:-}`, not `$TARGET_SHA`: this arm is lifted and executed + # under `set -u` by the smoke-gate harnesses in tests/test_update_sh.py, + # which define the gate's own variables and not Phase 1's. An unbound + # expansion there aborts the fragment mid-revert — silently turning twelve + # tests that assert the clock is restarted into failures about the wrong + # thing. The helper already treats an empty value as "nothing to block" and + # says so, which is also the right degradation if a future path ever + # reaches here without a resolved target. + _block_reverted_release "${TARGET_SHA:-}" + # litclock-dev#845 — the installed clock units must match the code we just + # reset to, BEFORE the `systemctl start`s below re-arm them (ANCHOR: + # reinstall-clock-units). Loud on failure, but the verdict stays + # failed_reverted: the code really was restored, and turning a revert into + # a harder failure helps nobody. + _reinstall_clock_units_from_tree \ + || log_error "[revert] the installed clock units do NOT match the reverted tree; the timer may drive code it was not paired with (litclock-dev#845)" # Bring the clock back up on the OLD SHA before exiting. log_info "Restoring clock on previous SHA..." sudo systemctl start litclock.service 2>/dev/null || true @@ -1686,8 +1983,9 @@ else # and leaves the interpreter still missing, and runtheclock.sh sources # ./venv/bin/activate. The two `systemctl start`s just above therefore # cannot bring the clock back, and "rolled back, clock is fine" would - # be a lie. The pip-failure branch ~250 lines up reasons exactly this - # way for a strictly WEAKER condition (an INDETERMINATE venv): "Lying + # be a lie. The pip-failure arm (ANCHOR: pip-failure-arm) reasons + # exactly this way for a strictly WEAKER condition (an INDETERMINATE + # venv): "Lying # about state is worse than admitting we don't know." A missing # interpreter is not indeterminate — it is known broken. # @@ -1854,7 +2152,8 @@ for conf in "$INSTALL_DIR"/systemd/tmpfiles.d/*.conf; do # a symlink between the test and the `sudo install` below; treating it as a # privilege boundary would be wrong. It is not one here: the attacker would # be the pi user, who already runs this script and (per the shipped - # 010_pi-nopasswd) already has full sudo. See the O_NOFOLLOW/O_NONBLOCK + # 010_pi-nopasswd, kept deliberately — see sudoers/020's header) already + # has full sudo. See the O_NOFOLLOW/O_NONBLOCK # note in the project learnings for the shape a real boundary needs. if [[ ! -f "$conf" || -L "$conf" ]]; then log_warn "skipping $(basename "$conf") — not a regular file" diff --git a/src/clear.py b/src/clear.py index 01a0e86..5ea81cd 100644 --- a/src/clear.py +++ b/src/clear.py @@ -1,9 +1,5 @@ -import logging -import os -import sys -import traceback - from display_driver import epd7in5 +from hard_exit import run_and_exit def main(): @@ -17,49 +13,18 @@ def main(): if __name__ == "__main__": - # litclock-dev#815 — terminate WITHOUT interpreter finalization, for the - # same reason src/literary_clock.py does. "litclock-dev#531" there is SHORTHAND for - # the lgpio teardown crash, not a real issue reference (litclock-dev#531 is - # the runtime-render epic); the NOTE in src/literary_clock.py and the - # CHANGELOG carry the full story. - # - # The mechanism: the module-level `from display_driver import epd7in5` above - # pulls in lgpio via gpiozero and spawns the daemon thread `Thread-1`. It - # parks in a blocking read and nothing removes it, so at Py_FinalizeEx it can - # unwind against half-cleared globals and raise. `os._exit` skips - # finalization entirely. litclock-dev#556 fixed this in the per-minute painter only, - # so this file ran the race on every invocation until litclock-dev#815. - # - # The exit shim below is deliberately IDENTICAL to the one in - # src/eink_display.py — tests/test_splash_renderers_skip_finalization.py - # asserts the two are AST-equivalent, so a fix to one cannot silently miss - # the other. That is the gap that let litclock-dev#556 leave BOTH files behind for a - # whole release. This comment is the part that differs, because the two - # files differ: eink_display.py's flush is load-bearing for the OTA smoke - # gate (it compares `catalog-get` stdout through a pipe); here the only - # stdout write is `main()`'s `print(e)` on an OSError, so the flush is - # ordinary correctness rather than a gate dependency. + # litclock-dev#815 — terminate WITHOUT interpreter finalization: the + # module-level `from display_driver import epd7in5` above pulls in lgpio + # via gpiozero and spawns the daemon thread `Thread-1`, which parks in a + # blocking read and can raise at Py_FinalizeEx (the crash tracked as + # litclock-dev#531 — see src/literary_clock.py's NOTE and the CHANGELOG). litclock-dev#556 + # fixed that in the per-minute painter only, so this file ran the race on + # every invocation until litclock-dev#815. # - # Every step is individually guarded anyway: a flush can itself raise - # (BrokenPipeError on a closed reader), and litclock-dev#813 is the lesson - # that anything unguarded ahead of the exit can cost you the exit. The - # `except SystemExit` arm is defensive — nothing in this file raises it — - # and is kept so the shim stays identical to its twin. - _exit_code = 0 - try: - main() - except SystemExit as _e: # defensive here; load-bearing in the eink_display.py twin - _exit_code = 0 if _e.code is None else (_e.code if isinstance(_e.code, int) else 1) - except BaseException: - traceback.print_exc() - _exit_code = 1 - for _stream in (sys.stdout, sys.stderr): - try: - _stream.flush() - except BaseException: - pass - try: - logging.shutdown() - except BaseException: - pass - os._exit(_exit_code) + # The shim lives in src/hard_exit.py, shared with src/eink_display.py + # (litclock-dev#837 / litclock-dev#840): the two used to carry AST-identical copies + # held together by an AST test, and the BrokenPipe escape found in the + # v0.227.0 port review had to be fixed in both. Here the only stdout write + # is main()'s `print(e)` on an OSError, so the helper's flush is ordinary + # correctness rather than the OTA-gate dependency it is in the twin. + run_and_exit(main) diff --git a/src/config.py b/src/config.py index 6db1c1f..8bdc0df 100644 --- a/src/config.py +++ b/src/config.py @@ -136,6 +136,24 @@ def _validate_weather_location_mode(value: str) -> tuple[bool, str | None]: return True, None +def weather_location_mode(raw: object) -> str: + """Normalise a raw ``WEATHER_LOCATION_MODE`` value the way every READER does. + + Absent, ``None``, empty or whitespace-only is ``"auto"`` — legacy / pre-litclock-dev#337 + env.sh files never set the key, and the collector maps an empty env value + to ``None``. Anything else is returned stripped but otherwise verbatim, so + a value the writer would have rejected (``"autp"``, ``"AUTO"``) comes back + as itself and the caller can decide what an INVALID mode means for it: + the resolver skips IP-geo for any non-``auto`` value; the diagnostics + staleness check exempts only ``"specific"`` (litclock-dev#836). This is the + ONE copy of the expression — ``location_resolver.main()``, the settings + writer and the diagnostics anomaly all call it, and + ``tests/test_control_server_diagnostics_anomalies.py`` scans ``src/`` for + a re-inlined copy. + """ + return (str(raw) if raw else "auto").strip() or "auto" + + def _validate_weather_ip_country(value: str) -> tuple[bool, str | None]: # litclock-dev#337 A6.1: ISO 3166-1 alpha-2 country code (uppercase) OR empty # (pre-S2 envs + first-resolve cases where IP-geo hasn't run yet). diff --git a/src/control_server/__init__.py b/src/control_server/__init__.py index 1a75766..b1de2eb 100644 --- a/src/control_server/__init__.py +++ b/src/control_server/__init__.py @@ -81,6 +81,8 @@ def create_app(test_config: dict | None = None) -> Flask: LAST_UPDATE_FILE=os.environ.get("LITCLOCK_LAST_UPDATE_FILE"), LKG_SHA_FILE=os.environ.get("LITCLOCK_LKG_SHA_FILE"), PHASE3_SKIPPED_FILE=os.environ.get("LITCLOCK_PHASE3_SKIPPED_FILE"), + # litclock-dev#847 item 1 — the runtime-render validation memo. + RUNTIME_VALIDATION_MEMO_FILE=os.environ.get("LITCLOCK_RUNTIME_VALIDATION_MEMO_FILE"), # EPIC litclock-dev#383 PR2 handoff markers. Same env-override pattern as above so # tests point these at a tmp dir (and a direct marker write succeeds # there without sudo). See control_server/handoff.py for the lifecycle. @@ -132,7 +134,12 @@ def create_app(test_config: dict | None = None) -> Flask: # not, so a hand-edited env.sh could paint weather on the panel and render # the toggle OFF. Registered as the callable itself so the template asks the # same function the panel does. - import config as _config # noqa: PLC0415 — lazy, matches _env.py's pattern + # Function-local like every other import in create_app (strings_catalog + # above, the blueprints below): the module scope holds only Flask, so the + # package imports without the src/ chain. NOT _env.py's pattern, which + # the old comment cited — that one lazy-loads for its stubbed callers. + # (litclock-dev#840; _anomalies.py took the module-scope form instead.) + import config as _config # noqa: PLC0415 app.jinja_env.globals["weather_enabled"] = _config.weather_enabled diff --git a/src/control_server/routes/diagnostics/_anomalies.py b/src/control_server/routes/diagnostics/_anomalies.py index f254d95..cb3c22a 100644 --- a/src/control_server/routes/diagnostics/_anomalies.py +++ b/src/control_server/routes/diagnostics/_anomalies.py @@ -21,6 +21,13 @@ from flask import current_app +# Module scope, like the sibling _collectors (litclock-dev#840). It used to be a +# function-local import "matching _env.py's pattern", but that pattern is +# _env.py's own (it lazy-loads so stubbed callers skip the cost). Here the +# module-scope `from ._collectors import ...` below loads config at import +# time regardless, so the old function-local form deferred nothing. +import config as _config # src/ on sys.path; same hard dep as _collectors + from ._collectors import ( DEFAULT_COLLECTED_MARKER_PATH, DEFAULT_LAST_RENDERED_IP_PATH, @@ -108,11 +115,33 @@ def _weather_is_enabled(values: dict[str, Any]) -> bool: return raw if not raw: return False - import config as _config # noqa: PLC0415 — lazy, matches _env.py's pattern - return _config.weather_enabled({"WEATHER_ENABLED": str(raw)}) +def _location_mode(values: dict[str, Any]) -> str: + """The normalised ``WEATHER_LOCATION_MODE`` for this payload. + + One call into :func:`config.weather_location_mode`, the same normaliser + ``location_resolver.main()`` and the settings writer use, so the three + readers cannot drift (litclock-dev#836; the maintainability pass of its + review found this module had grown a THIRD inline copy). Absent, ``None``, + empty or whitespace-only is ``"auto"``; anything else comes back stripped. + + The two callers here ask different questions of it, deliberately: + + - the staleness check in :func:`_compute_anomalies` is suppressed only for + ``"specific"`` — the one non-auto value the writer accepts. An INVALID + value (``"autp"``, ``"AUTO"`` from a hand-edited env.sh) makes the + resolver stop refreshing the stamp exactly as ``specific`` does, but + that is a broken configuration, not a choice, and this anomaly is the + only surface that shows it (Codex, litclock-dev#836 review: the first cut + exempted every non-auto value and hid the fault); + - the grey tier in :func:`_compute_uncollected` requires ``"auto"``, so an + invalid mode stays orange there too, as it did before. + """ + return _config.weather_location_mode(values.get("weather_location_mode")) + + def _compute_anomalies(values: dict[str, Any]) -> list[str]: """Return the list of section IDs whose data tripped an anomaly. @@ -139,8 +168,17 @@ def _compute_anomalies(values: dict[str, Any]) -> list[str]: nothing works" — ``gateway`` is collected but never consulted, and there is no reachability probe. Accepted gap: the DHCP heuristic never caught that state either. - - ``time-location`` — weather enabled AND (city empty OR mode=specific - with empty place OR last IP-geo > 7 days). + - ``time-location`` — weather enabled AND (city empty OR (mode is not + ``specific``, per :func:`_location_mode`, AND last IP-geo > 7 days)). + The mode gate is litclock-dev#836: since litclock-dev#791 the resolver writes + ``WEATHER_LAST_IP_GEO_AT``, and an owner who then picks a Specific + location keeps that stamp while the resolver deliberately stops + refreshing it — so after seven days a correctly configured clock + showed a permanent "Location stale" that no reboot or save could + clear. Only the valid ``specific`` is exempt: an invalid mode also + freezes the stamp, but as a fault, and this is where it shows. The + stamp is still DISPLAYED in both modes (the ``Last IP-geo`` row); + only the age check is gated. - ``services`` — ANY non-oneshot unit non-active. ``DIAG_ONESHOT_UNITS`` is the explicit allowlist of post-boot-inactive-by-design services; members also get a pass on the transient ``activating``/``deactivating`` @@ -209,7 +247,10 @@ def _compute_anomalies(values: dict[str, Any]) -> list[str]: if not values.get("weather_location_name"): tl_anomaly = True ipgeo_iso = values.get("last_ip_geo_at") - if isinstance(ipgeo_iso, str) and ipgeo_iso: + # litclock-dev#836 — in Specific mode the stamp is a frozen record of the last + # auto resolve, not a fault, and the resolver will never refresh it. + # Only the VALID non-auto value is exempt: see _location_mode. + if _location_mode(values) != "specific" and isinstance(ipgeo_iso, str) and ipgeo_iso: try: ipgeo_dt = datetime.fromisoformat(ipgeo_iso) if ipgeo_dt.tzinfo is None: @@ -403,16 +444,17 @@ def _compute_uncollected(values: dict[str, Any]) -> list[str]: # time-location — gate per D3. Fix C: legacy / pre-litclock-dev#337 env files don't # set WEATHER_LOCATION_MODE; the rest of the app treats a missing mode as - # `auto`. Accept None alongside "auto" so those Pis get the grey tier - # instead of the orange false positive this change was meant to remove. + # `auto`. _location_mode normalises None/empty to "auto" (the resolver's + # own normalisation, litclock-dev#836) so those Pis get the grey tier instead of + # the orange false positive this change was meant to remove. An INVALID + # value is not "auto" and stays orange, as it did before. # When the persistent marker is absent (collected is None), preserve the # v0.214.4 env-only behavior (no marker gate); otherwise require the # time-location key to be missing too. if _weather_is_enabled(values): - mode = values.get("weather_location_mode") tl_never_collected = True if collected is None else "time-location" not in collected if ( - mode in ("auto", None, "") + _location_mode(values) == "auto" and tl_never_collected and not values.get("weather_location_name") and not values.get("last_ip_geo_at") diff --git a/src/control_server/routes/settings.py b/src/control_server/routes/settings.py index 7530bce..a023205 100644 --- a/src/control_server/routes/settings.py +++ b/src/control_server/routes/settings.py @@ -804,7 +804,7 @@ def _save_and_apply( # since by definition nothing was written (litclock-dev#414 item #2). return ({"ok": True, "saved": [], "settings": dict(existing)}, 200) - previous_mode = (existing.get("WEATHER_LOCATION_MODE") or "auto").strip() or "auto" + previous_mode = _config.weather_location_mode(existing.get("WEATHER_LOCATION_MODE")) sync_quick_attempted, sync_quick_succeeded = _run_sync_quick_if_needed( env_updates, resolved_country, diff --git a/src/control_server/routes/status.py b/src/control_server/routes/status.py index ccf93bb..21b55d7 100644 --- a/src/control_server/routes/status.py +++ b/src/control_server/routes/status.py @@ -20,6 +20,7 @@ from __future__ import annotations import json +import math import os import stat import time as _time @@ -72,6 +73,18 @@ # cleaned (e.g. update.sh disabled / cron stopped firing). PHASE3_SKIP_FRESH_WINDOW_S = 86400 +# litclock-dev#847 item 1 — the negative-result memo update.sh Phase 4.5 +# writes when the litclock-dev#531 runtime-render validation is deferred, times out or +# fails, and removes when it passes (or the marker is already present). JSON: +# {result, rc, reason, sha, at_unix}. No freshness clamp, unlike the Phase 3 +# marker: it describes the device's CURRENT tier, not a one-off skip, and the +# writer clears it on the state change that makes it stale. +DEFAULT_RUNTIME_VALIDATION_MEMO_FILE = os.environ.get( + "LITCLOCK_RUNTIME_VALIDATION_MEMO_FILE", "/var/lib/litclock/runtime-render-validation.json" +) +RUNTIME_VALIDATION_RESULTS = frozenset({"deferred", "timeout", "failed"}) +MAX_RUNTIME_VALIDATION_MEMO_BYTES = 8 * 1024 + # litclock-dev#274 follow-up — adversarial-review P1: budget for treating a # `state=running` update.status entry as fresh. Past this, assume update.sh # died (SIGKILL / OOM / power loss) without writing the terminal @@ -209,6 +222,54 @@ def _resolve_phase3_skipped_at(phase3_skipped_file: Path | None = None) -> float return float(st.st_mtime) +def _resolve_runtime_validation_memo(memo_file: Path | None = None) -> dict | None: + """The runtime-render validation memo as a fixed-shape dict, or ``None``. + + ``None`` when the file is absent, unreadable, not a regular file, not a + JSON object, or carries a ``result`` outside the writer's three tokens or + an ``at_unix`` that is not a finite, convertible number — the reader never + forwards a shape the PWA has not been told about. Same bounded loader as + the update.status readers (symlinks / FIFOs / oversize files rejected on + the open fd). + + EVERY rejection is a `return None`, never an exception (litclock-dev#854 review, Codex + probes). ``collect_status`` calls this unconditionally and serves both + /api/status and the server-rendered `/`, so one malformed file must not + take the control page down until someone deletes it. Two shapes did: + an unhashable ``result`` (``{"result": []}``) raised TypeError from the + set membership test, and a huge int ``at_unix`` raised OverflowError from + ``float()`` — while ``1e999`` sailed through as ``inf`` and would have + reached the PWA as a JSON literal no parser accepts. + """ + path = memo_file or Path(DEFAULT_RUNTIME_VALIDATION_MEMO_FILE) + data = safe_read_json(path, MAX_RUNTIME_VALIDATION_MEMO_BYTES) + if not isinstance(data, dict): + return None + result = data.get("result") + at_unix = data.get("at_unix") + # isinstance BEFORE the membership test: `in` hashes its left operand. + if not isinstance(result, str) or result not in RUNTIME_VALIDATION_RESULTS: + return None + if not isinstance(at_unix, (int, float)) or isinstance(at_unix, bool): + return None + try: + at_unix_f = float(at_unix) + except (OverflowError, ValueError): + return None + if not math.isfinite(at_unix_f): + return None + rc = data.get("rc") + reason = data.get("reason") + sha = data.get("sha") + return { + "result": result, + "at_unix": at_unix_f, + "rc": rc if isinstance(rc, int) and not isinstance(rc, bool) else None, + "reason": reason if isinstance(reason, str) and reason else None, + "sha": sha if isinstance(sha, str) and sha else None, + } + + def _resolve_update_progress( update_status_file: Path | None = None, ) -> tuple[str | None, int | None]: @@ -266,19 +327,32 @@ def _payload_to_last_update(data: dict | None) -> tuple[str | None, str | None] (state=complete) and /var/lib/litclock/last-update.json carry the same finished_at_unix + to_version fields. Returns ``(iso, version)`` on a successful extraction, or ``None`` if the payload is missing / - non-complete / has no usable timestamp.""" + non-complete / has no usable timestamp / did not install anything. + + litclock-dev#847 item 4 (litclock-dev#854 review): ``complete`` means "this RUN + finished cleanly", not "something was installed" — update.sh's two + deliberate no-op early-outs (nothing resolvable, and a blocked SHA) stamp + it before Phase 2 ever calls ``update_status_set_to_version``. A null + ``to_version`` is therefore the marker of a tick that installed nothing, + and this row must not move for one: an always-offline device would + otherwise report a fresh "Last update" every week, for an update that + never happened, with an em-dash where the version belongs. The persistent + mirror agrees by construction — Phase 7 is the only writer of + last-update.json, and its own gate requires ``to_version == $NEW_SHA``. + """ if not isinstance(data, dict) or data.get("state") != "complete": return None finished = data.get("finished_at_unix") to_version = data.get("to_version") + if not isinstance(to_version, str) or not to_version: + return None if not isinstance(finished, (int, float)): return None try: iso = datetime.fromtimestamp(float(finished), tz=UTC).isoformat() except (OSError, OverflowError, ValueError): return None - version = to_version if isinstance(to_version, str) and to_version else None - return iso, version + return iso, to_version def _resolve_last_update( @@ -415,6 +489,7 @@ def collect_status( last_update_file: Path | None = None, lkg_sha_file: Path | None = None, phase3_skipped_file: Path | None = None, + runtime_validation_memo_file: Path | None = None, ) -> dict: """Build the status payload — used by both `/api/status` (jsonified) and the `/` Status-tab template render (server-side first paint per @@ -430,7 +505,8 @@ def collect_status( `phase3_skipped_file` plumbed through for the same reason — the Status hero Phase-3-skip banner (litclock-dev#274 follow-up #5) needs a tmp - path in tests.""" + path in tests. `runtime_validation_memo_file` likewise (litclock-dev#847 + item 1).""" status_path = status_file or Path(DEFAULT_STATUS_FILE) quote_payload = _read_status_file(status_path) @@ -460,6 +536,8 @@ def collect_status( # read, two consumers. None when no update.status file is present # (the common steady-state case). update_state, update_phase = _resolve_update_progress(update_status_file=update_status_file) + # litclock-dev#847 item 1: why this device is (still) on the PNG tier. + runtime_render_validation = _resolve_runtime_validation_memo(memo_file=runtime_validation_memo_file) return { "ok": True, "stale": stale, @@ -493,6 +571,12 @@ def collect_status( # Both null in the common no-update-running state. "update_state": update_state, "update_phase_index": update_phase, + # litclock-dev#847 item 1: null when the last runtime-render validation + # passed (or never ran and was never deferred); otherwise + # {result: deferred|timeout|failed, at_unix, rc, reason, sha}. The PWA + # does not render it yet — a Diagnostics/Status surface is the + # follow-up; the payload is the contract. + "runtime_render_validation": runtime_render_validation, } @@ -507,6 +591,8 @@ def status() -> tuple[object, int]: lkg_sha_cfg = current_app.config.get("LKG_SHA_FILE") lkg_sha_path = Path(lkg_sha_cfg) if lkg_sha_cfg else None phase3_skipped_cfg = current_app.config.get("PHASE3_SKIPPED_FILE") + memo_cfg = current_app.config.get("RUNTIME_VALIDATION_MEMO_FILE") + memo_path = Path(memo_cfg) if memo_cfg else None phase3_skipped_path = Path(phase3_skipped_cfg) if phase3_skipped_cfg else None body = collect_status( status_file=status_path, @@ -516,5 +602,6 @@ def status() -> tuple[object, int]: last_update_file=last_update_path, lkg_sha_file=lkg_sha_path, phase3_skipped_file=phase3_skipped_path, + runtime_validation_memo_file=memo_path, ) return jsonify(body), 200 diff --git a/src/control_server/routes/system.py b/src/control_server/routes/system.py index aa1ccf5..9632d67 100644 --- a/src/control_server/routes/system.py +++ b/src/control_server/routes/system.py @@ -57,10 +57,11 @@ # litclock-dev#396 — gift-flow system-timezone reset. Absolute path so the scoped # sudoers entry (sudoers/020_litclock-control: `timedatectl set-timezone UTC`) -# matches verbatim. Today the call also works via the broad 010_pi-nopasswd -# grant; the scoped entry becomes load-bearing once that grant is dropped -# (litclock-dev#387, not yet shipped). UTC is the neutral default; the recipient's -# first-boot IP-geo overwrites it. +# matches verbatim. The call also works via the broad 010_pi-nopasswd grant, +# which the image keeps (the litclock-dev#387/litclock-dev#82 drop was reversed 2026-07-12); the scoped +# entry is kept anyway because a fixed-argv grant is the right default for a +# call site that needs exactly one command. UTC is the neutral default; the +# recipient's first-boot IP-geo overwrites it. TIMEDATECTL: Final[str] = "/usr/bin/timedatectl" # Own timeout (not SYSTEMCTL_TIMEOUT_S): `timedatectl set-timezone` talks to # systemd-timedated over D-Bus, a different call shape than systemctl. 5s is diff --git a/src/eink_display.py b/src/eink_display.py index ac5beff..8632e37 100755 --- a/src/eink_display.py +++ b/src/eink_display.py @@ -10,11 +10,11 @@ import logging import os import sys -import traceback from PIL import Image, ImageDraw, ImageFont from captive_portal import SETUP_HOSTNAME +from hard_exit import run_and_exit from log import setup_logging # Try to import qrcode, provide helpful message if not installed @@ -1481,6 +1481,28 @@ def save_image(image: Image.Image, path: str): logging.info(f"Image saved to {path}") +def _stderr_note(text): + """A best-effort stderr line for the catalog subcommands' fallback arms. + + PR litclock-dev#851 review (Codex 4): the fallback used `print(..., file=sys.stderr)` + bare. On a broken stderr that raised BEFORE the fallback value was + assigned, so the "always exits 0 with a value" contract failed exactly + when it was being exercised — empty stdout, exit 1. And with fd 2 closed + at startup Python sets sys.stderr to None, where print(file=None) means + STDOUT: the note would land on the very stream scripts/update.sh's smoke + gate compares. Neither may happen; the note is diagnostics, the value is + the contract. + """ + stream = sys.stderr + if stream is None: + return + try: + stream.write(text + "\n") + stream.flush() + except BaseException: + pass + + def _parse_slots(pairs): """--slot NAME=VALUE args → dict. Shared by the status and catalog-get subcommands (slice-1 /review: two diverging copies would silently split @@ -1603,16 +1625,15 @@ def main(): ) # litclock-dev#773 item 2: the OTA smoke gate probed three catalog VALUES, - # so a bundle truncated to just those three passed green with 434 strings + # so a bundle truncated to just those three passed green with 435 strings # gone. Values cannot detect that; only a count can, and the count has to # come from the loader (not from re-reading the JSON) so it measures what # the app will actually see after the filters in ``_catalog``. - catalog_count_parser = subparsers.add_parser( - "catalog-count", help="Print how many strings the active language catalog loaded" - ) - catalog_count_parser.add_argument( - "--language", default=None, help="resolve this language code instead of the active one" - ) + # No --language flag (litclock-dev#840): it never had a caller. The gate + # pins English through the environment (`LITCLOCK_LANGUAGE=en` as a + # command prefix, litclock-dev#763), which is also the only channel the owner's + # language arrives by, so one resolution path serves both. + subparsers.add_parser("catalog-count", help="Print how many strings the active language catalog loaded") handoff_parser = subparsers.add_parser("handoff-splash", help="Display the post-WiFi handoff splash") handoff_parser.add_argument("qr_url", help="PWA QR URL encoded on the splash") @@ -1670,7 +1691,7 @@ def main(): resolved = strings_catalog.get(args.key, **slots) except Exception as exc: # noqa: BLE001 — the contract is "always exits 0 with a value" - print(f"warning: catalog resolution failed ({exc!r}); printing the key", file=sys.stderr) + _stderr_note(f"warning: catalog resolution failed ({exc!r}); printing the key") resolved = args.key print(resolved) return @@ -1686,10 +1707,12 @@ def main(): try: import strings_catalog # noqa: PLC0415 - code = args.language or strings_catalog.active_language() - count = len(strings_catalog._catalog(code)) + # catalog_size is PUBLIC for this caller (litclock-dev#840): the + # gate used to reach into the private `_catalog`, and a rename + # there would have printed 0 and failed every update closed. + count = strings_catalog.catalog_size(strings_catalog.active_language()) except Exception as exc: # noqa: BLE001 — the contract is "always exits 0 with a value" - print(f"warning: catalog count failed ({exc!r}); reporting 0", file=sys.stderr) + _stderr_note(f"warning: catalog count failed ({exc!r}); reporting 0") count = 0 print(count) return @@ -1731,33 +1754,12 @@ def main(): # the boot / hotspot / QR / handoff / shutdown / reset-failed splash, so it # ran the race on every invocation. # - # FLUSHING IS LOAD-BEARING, not tidiness. `os._exit` discards buffered - # writes, and `catalog-get` / `catalog-count` print to stdout which - # scripts/update.sh's OTA smoke gate COMPARES. Python buffers stdout when it - # is a pipe — which is exactly how the gate invokes this — so exiting - # without an explicit flush would return an empty string to the gate and - # fail it on every device, forever. That is the false-RED direction - # litclock-dev#773 exists to prevent, and it would have been introduced by - # the fix for a different bug. - # - # Every step is individually guarded: a flush can itself raise - # (BrokenPipeError on a closed reader), and litclock-dev#813 is the lesson - # that anything unguarded ahead of the exit can cost you the exit. - _exit_code = 0 - try: - main() - except SystemExit as _e: # preserve argparse's codes and the explicit sys.exit(1) - _exit_code = 0 if _e.code is None else (_e.code if isinstance(_e.code, int) else 1) - except BaseException: - traceback.print_exc() - _exit_code = 1 - for _stream in (sys.stdout, sys.stderr): - try: - _stream.flush() - except BaseException: - pass - try: - logging.shutdown() - except BaseException: - pass - os._exit(_exit_code) + # The shim lives in src/hard_exit.py, shared with src/clear.py + # (litclock-dev#837 / litclock-dev#840). What matters HERE: its flush is load-bearing, + # not tidiness — `catalog-get` / `catalog-count` print to a stdout that + # scripts/update.sh's OTA smoke gate compares through a pipe, and an + # unflushed `os._exit` would hand the gate an empty string and revert every + # update on every device, forever. And it preserves argparse's exit codes + # and the explicit sys.exit(1) on a handoff-splash paint failure, which the + # splash scripts branch on. + run_and_exit(main) diff --git a/src/geocoding.py b/src/geocoding.py index 6181c75..34a7b43 100644 --- a/src/geocoding.py +++ b/src/geocoding.py @@ -185,8 +185,12 @@ def set_system_timezone(timezone): # Set the timezone via the root-owned wrapper (litclock-dev#387). We CANNOT call # `sudo timedatectl set-timezone ` directly: sudoers/020 only - # authorizes the wrapper's fixed path (a `set-timezone *` glob would be - # a privilege hole once 010_pi-nopasswd is dropped). The wrapper + # authorizes the wrapper's fixed path (a `set-timezone *` glob would + # let pi set the clock to any string reaching the resolver). The image + # keeps 010_pi-nopasswd — the drop planned under litclock-dev#387/litclock-dev#82 was reversed + # 2026-07-12 — so the wrapper is defense-in-depth rather than the only + # thing standing between pi and root; it is kept because a scoped, + # re-validating grant is the right default for this call site. The wrapper # re-validates the tz in root-owned code — this in-process check is a # UX fast-path, not the security boundary. result = subprocess.run( # noqa: S603,S607 diff --git a/src/hard_exit.py b/src/hard_exit.py new file mode 100644 index 0000000..753fd93 --- /dev/null +++ b/src/hard_exit.py @@ -0,0 +1,142 @@ +"""Terminate a display-touching entry point WITHOUT interpreter finalization. + +One helper for the exit shim that `src/clear.py` and `src/eink_display.py` +used to carry as two AST-identical 18-line copies, held together by an +AST-equivalence test (litclock-dev#815, litclock-dev#840). Both files import +`display_driver`, which pulls in lgpio via gpiozero and spawns the daemon +thread `Thread-1`; it parks in a blocking read and nothing removes it, so at +Py_FinalizeEx it can unwind against half-cleared globals and raise. `os._exit` +skips finalization entirely. (litclock-dev#531 is the shorthand for that +teardown crash — see the NOTE in `src/literary_clock.py` and the CHANGELOG.) + +Stdlib only, on purpose: the helper must be importable on a dev box, in the +fake checkouts the OTA-gate tests build, and before either caller has decided +whether it will touch the panel. + +`src/literary_clock.py`'s three exit arms are deliberately NOT callers, so do +not "fix" that: the painter always exits 0 from its paint block whatever +failed, exits 1 from inside its pre-paint guard, keeps `sys.exit` on +`--dry-run` (no `display_driver` import, and update.sh reads that code) and +has no stdout to flush — a different contract from "run main, translate +SystemExit, flush, exit", pinned statement-by-statement by +`tests/test_exit_without_finalization.py`. + +Why every statement is guarded and the exit sits in a ``finally`` +(litclock-dev#837, found by the Codex adversarial pass in the v0.227.0 port +review): the previous shims ran ``traceback.print_exc()`` bare in their +exception arm. On a broken stderr — reader gone, pipe closed — that write +raises ``BrokenPipeError``, which escaped the arm, skipped the flushes AND the +``os._exit``, and ran the very interpreter finalization the shim exists to +prevent, with `Thread-1` alive. That is litclock-dev#813's lesson repeated a +third time: anything unguarded ahead of the exit can cost you the exit. +Diagnostics here are best-effort; terminating without finalization is not. + +FLUSHING IS LOAD-BEARING, not tidiness. ``os._exit`` discards buffered +writes, and ``catalog-get`` / ``catalog-count`` print to stdout which +``scripts/update.sh``'s OTA smoke gate COMPARES. Python block-buffers stdout +when it is a pipe — exactly how the gate invokes ``eink_display.py`` — so +exiting without an explicit flush would hand the gate an empty string and +revert every update on every device, forever (the false-RED direction +litclock-dev#773 exists to prevent). ``logging.shutdown()`` flushes the log +handlers ``os._exit`` would otherwise skip; it re-raises non-OSError +exceptions while ``logging.raiseExceptions`` is true, so it is guarded too. +""" + +from __future__ import annotations + +import logging +import os +import sys +import traceback +from collections.abc import Callable + + +def exit_code_for(exc: SystemExit) -> int: + """Translate a ``SystemExit`` into the int ``os._exit`` needs. + + ``sys.exit()`` / ``sys.exit(None)`` mean 0; an int in 0..255 is itself + (argparse's 2, the explicit ``sys.exit(1)`` on a handoff-splash paint + failure; bool is an int subclass and passes through as 0/1); any other + payload — ``sys.exit("message")`` — is 1, matching the interpreter. + + CLAMPED, because ``os._exit`` is not ``sys.exit`` (PR litclock-dev#851 review, Codex + adversarial, reproduced with an atexit probe): ``os._exit(2**31)`` and + ``os._exit(-2**31 - 1)`` raise ``OverflowError`` from inside the + ``finally`` that is supposed to be unconditional, the exception escapes + ``run_and_exit``, and interpreter finalization runs after all. Anything + outside 0..255 — including negatives, which the interpreter would wrap + to ``code & 0xFF`` — becomes 1: it is a failure code either way, and 1 is + the one every caller already handles. + """ + code = exc.code + if code is None: + return 0 + if isinstance(code, int): + return code if 0 <= code <= 255 else 1 + return 1 + + +def _report(text: str) -> None: + """Best-effort stderr write: never raises, never lands on stdout.""" + stream = sys.stderr + if stream is None: + return + try: + stream.write(text) + except BaseException: + pass + + +def run_and_exit(main: Callable[[], object]) -> None: + """Run ``main()`` and terminate the process via ``os._exit`` — always. + + Exit code: 0 when ``main`` returns; its ``SystemExit`` code when it raises + one (see :func:`exit_code_for`); 1 for any other exception, whose traceback + is reported on stderr on a best-effort basis. The ``finally`` is the + contract: no failure in reporting, flushing or logging shutdown can reach + the caller, and none can skip the exit. + """ + # 1 until proven otherwise: if something unforeseen unwound past every + # guard below, the finally still exits, and it exits as a FAILURE. + code = 1 + try: + try: + main() + code = 0 + except SystemExit as exc: + # The attribute read in exit_code_for is the ONE expression here + # outside a guard, deliberately: a SystemExit whose `.code` raises + # is how the tests prove the finally exits 1 on an escape past + # every guard. Do not wrap it. + code = exit_code_for(exc) + if exc.code is not None and not isinstance(exc.code, int): + # sys.exit("message"): the interpreter prints the payload to + # stderr before exiting 1. Same, best-effort. + _report(str(exc.code) + "\n") + except BaseException: + # Inside its own guard (litclock-dev#837): print_exc writes to + # stderr and raises BrokenPipeError when nobody is reading it. + try: + if sys.stderr is not None: + # Explicit file= and the None check (PR litclock-dev#851 review): with + # fd 2 closed at startup Python sets sys.stderr to None, and + # traceback.print_exc() then prints to STDOUT — on the + # catalog-count path that pollutes the integer the OTA + # gate compares. + traceback.print_exc(file=sys.stderr) + except BaseException: + pass + code = 1 + for stream in (sys.stdout, sys.stderr): + if stream is None: + continue + try: + stream.flush() + except BaseException: + pass + try: + logging.shutdown() + except BaseException: + pass + finally: + os._exit(code) diff --git a/src/literary_clock.py b/src/literary_clock.py index b99e3f9..289bbe5 100644 --- a/src/literary_clock.py +++ b/src/literary_clock.py @@ -148,13 +148,14 @@ def _render_lead_seconds() -> float: RENDER_LEAD_DEFAULT_S = 4.0 -# 4.0, matching the SHIPPED timer, not a permissive band (/review). + +# The FLOOR is 4.0, matching the SHIPPED timer, not a permissive band (/review). # -# It was 3.0, which this function's own docstring calls broken: "any value below -# ~4 lands in the minute that is ENDING and the clock renders the previous -# minute's quote permanently". With `OnCalendar=*-*-* *:*:56`, a lead of 3.0-3.9 -# targets :59 of the minute that is ending — verified — so the validator -# ACCEPTED the exact failure it exists to reject, and +# It was 3.0, which `_render_lead_seconds`' docstring calls broken: "any value +# below ~4 lands in the minute that is ENDING and the clock renders the +# previous minute's quote permanently". With `OnCalendar=*-*-* *:*:56`, a lead +# of 3.0-3.9 targets :59 of the minute that is ending — verified — so the +# validator ACCEPTED the exact failure it exists to reject, and # tests/test_timer_lead.py asserted 3.0 was honoured, locking it in. # # The real invariant is `lead >= 60 - timer_second`, and this module cannot read @@ -165,6 +166,63 @@ def _render_lead_seconds() -> float: RENDER_LEAD_MAX_S = 30.0 RENDER_LEAD_S = _render_lead_seconds() +# litclock-dev#838 — the hour of the nightly full `epd.Clear()`, parsed the +# same way as the lead above and for the same reason. It used to be a bare +# `int(os.getenv("DISPLAY_CLEAR_HOUR", 2))` INSIDE the paint try, after +# `epd.init()` and before `epd.display()`: an empty or malformed value raised +# there every minute, the paint's `except Exception` swallowed it, and the +# panel froze on the last quote with the process exiting 0 — the litclock-dev#531 +# lookalike signature, and one the Phase 4.5 dry-run cannot see because it +# never sources env.sh. Empty is one uncomment away: since litclock-dev#783 +# all four env.sh writers seed the key, commented, in the sample's +# `export KEY=` unset idiom. +DISPLAY_CLEAR_HOUR_DEFAULT = 2 + + +def _warn_never_raises(fmt: str, *args) -> None: + """`logging.warning`, guaranteed not to raise. + + PR litclock-dev#851 review (Codex 2, downgraded by the Claude pass): `logging.warning` + does NOT raise on a broken pipe or a closed fd — `Handler.handleError` + swallows OSError — only on a Python-level closed TextIOWrapper, which + nothing in this tree does. The guard exists so `_display_clear_hour`'s + "never raises" is literally true rather than true-by-current-handlers. + Design note: `_render_lead_seconds` above has the same bare-warning shape + at IMPORT and is deliberately left as is — a raising warning there is a + raising import, which the litclock-dev#762 tests already cover for the value path, + and refactoring it was out of scope for the review that added this. + """ + try: + logging.warning(fmt, *args) + except BaseException: + pass + + +def _display_clear_hour() -> int: + """`DISPLAY_CLEAR_HOUR`, parsed defensively and bounded to a clock hour. + + Unset, empty or whitespace is the default, SILENTLY — that is the sample's + own unset idiom, merged onto every device by update.sh, and a warning there + would fire every minute on every clock. Anything else that is not an + integer in 0..23 falls back to the default LOUDLY, naming the variable and + the raw value, because the journal is the only thing that separates a + rejected knob from a wedged panel. Never raises. + """ + raw = os.getenv("DISPLAY_CLEAR_HOUR") + if raw is None or not raw.strip(): + return DISPLAY_CLEAR_HOUR_DEFAULT + try: + value = int(raw) + except (TypeError, ValueError): + _warn_never_raises("DISPLAY_CLEAR_HOUR=%r is not an integer; using %s", raw, DISPLAY_CLEAR_HOUR_DEFAULT) + return DISPLAY_CLEAR_HOUR_DEFAULT + if not 0 <= value <= 23: + _warn_never_raises( + "DISPLAY_CLEAR_HOUR=%r is not an hour in 0..23; using %s", raw, DISPLAY_CLEAR_HOUR_DEFAULT + ) + return DISPLAY_CLEAR_HOUR_DEFAULT + return value + # Persistent QR code on the e-ink top strip (litclock-dev#245 A6). 75x75 px at x=713,y=0, # encodes the PWA URL so non-tech users can scan-to-open instead of typing. # Geometry locked in M0 (validated via tools/control-pwa/validate_qr_layout.py); @@ -1032,22 +1090,30 @@ def _stamp_update_failed_glyph(image, draw): epd = None - # These two run BEFORE the paint's try/except, and both sit downstream of - # `import lgpio` — the display_driver import is what constructs the gpiozero - # objects and spawns Thread-1. So an exception here used to propagate out of - # __main__ entirely, skipping the os._exit below and running the very - # finalization this fix exists to avoid. Worse, the most likely failure here - # is a GPIO-busy import error from the previous minute's process, i.e. litclock-dev#531 - # firing on precisely the path the fix targets. + # These run BEFORE the paint's try/except, and the first two sit downstream + # of `import lgpio` — the display_driver import is what constructs the + # gpiozero objects and spawns Thread-1. So an exception here used to + # propagate out of __main__ entirely, skipping the os._exit below and + # running the very finalization this fix exists to avoid. Worse, the most + # likely failure here is a GPIO-busy import error from the previous + # minute's process, i.e. litclock-dev#531 firing on precisely the path the fix + # targets. # # Exits 1, matching the previous behaviour of letting the exception escape. # BaseException, not Exception: a SystemExit or KeyboardInterrupt raised in # here must not slip past to normal finalization either. + # + # The DISPLAY_CLEAR_HOUR parse is here too (litclock-dev#838). It never + # raises by design, and it is read BEFORE the panel is touched so that if + # it ever did, the run would exit 1 with a traceback in the journal rather + # than freeze the panel from inside the paint try with the process + # reporting success — which is what the bare int() it replaces did. try: # Hardware import is lazy — deferred until we actually need to talk to the display. from display_driver import epd7in5 # noqa: E402 image, quote_meta, now = main() + display_clear_hour = _display_clear_hour() except BaseException: # traceback.print_exc() as well as logging: logging.exception is # level-gated, and this guard exists to instrument the one failure most @@ -1085,7 +1151,6 @@ def _stamp_update_failed_glyph(image, draw): epd.init() logging.info("EPD initialized.") - display_clear_hour = int(os.getenv("DISPLAY_CLEAR_HOUR", 2)) # litclock-dev#762 Trap 2: this MUST read the offset target, not # datetime.now(). With the timer at :56 a live now().minute is never 0, # so the hourly full clear would simply stop firing and ghosting would diff --git a/src/location_resolver.py b/src/location_resolver.py index 56eca24..825d900 100644 --- a/src/location_resolver.py +++ b/src/location_resolver.py @@ -390,7 +390,7 @@ def main() -> int: log.warning("could not load env.sh: %r — exiting cleanly", exc) return 0 - mode = (env.get("WEATHER_LOCATION_MODE") or "auto").strip() or "auto" + mode = _config.weather_location_mode(env.get("WEATHER_LOCATION_MODE")) if mode != "auto": # The whole point of this gate: a user in Specific mode picked their # location intentionally; the on-boot reresolve must NEVER overwrite it. diff --git a/src/setup_server.py b/src/setup_server.py index 3f7f260..2c5dc1b 100755 --- a/src/setup_server.py +++ b/src/setup_server.py @@ -579,7 +579,18 @@ def _net_signal(net): return 0 -def _build_wifi_options(networks, placeholder=None): +def _selected_attr(flag): + """``" selected"`` when ``flag``, else nothing — the '] + # Through the helper, not two inline conditionals. The natural shape for + # the first of them — an empty-string branch written before the `if` — + # is a banned substring in tests/test_cross_file_string_parity.py's + # in-code-grammar guard (litclock-dev#532 item 11), which matches raw + # text and cannot tell an HTML attribute from a spliced plural suffix. + # The helper also states the invariant once: the flag picks exactly one + # of the two options to carry `selected`. + placeholder_selected = _selected_attr(not select_manual) + manual_selected = _selected_attr(select_manual) + options = [f''] for net in sorted(networks, key=_net_signal, reverse=True): ssid = html.escape(net["ssid"]) signal = _net_signal(net) @@ -614,7 +653,10 @@ def _build_wifi_options(networks, placeholder=None): security = net.get("security", "") lock = " [Open]" if not security or security == "--" else "" options.append(f'') - options.append(f'') + options.append( + f'" + ) return "\n ".join(options) @@ -636,7 +678,7 @@ def _filter_own_hotspot(networks): return [n for n in networks if n.get("ssid") != HOTSPOT_SSID] -def _wifi_network_options(): +def _wifi_network_options(select_manual=False): """Generate