From ca4eeb22b791e3329e9cb9ee1ddebcbb309b4c8d Mon Sep 17 00:00:00 2001 From: Jared Lunde Date: Fri, 2 Oct 2026 18:34:54 -0700 Subject: [PATCH] ci: run agent test shards in a systemd unit bounded by systemd, not the runner The agent-rest / agent-code-mode wedge leaves the test step in_progress past nextest's global-timeout, the step timeout and the job timeout, all of which need the runner to act. Memory exhaustion is ruled out by measurement under the runner's limits (4 CPUs, 16 GB): run peak ~0.3 GB / 80 tasks, compile anon peak ~3.2 GB. The test step now runs nextest via `sudo systemd-run --wait` with RuntimeMaxSec=11min, MemoryMax=8G, MemorySwapMax=0 and TasksMax=4096, so an overrun fails loudly with a systemd result and the always() steps still run; the unit's cgroup kill replaces the pkill sweep. A ci-resources sampler unit records memory, swap, PSI, pids, disk, top RSS and D-state processes every 5s, uploaded with nextest.log as an artifact. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01JimHGjsfk2Ktm5GxyZJKKk --- .config/nextest.toml | 3 +++ .github/workflows/ci.yml | 57 ++++++++++++++++++++++++++++++++++++---- 2 files changed, 55 insertions(+), 5 deletions(-) diff --git a/.config/nextest.toml b/.config/nextest.toml index e29b0899..a157c0ab 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -26,6 +26,9 @@ fail-fast = false # not a test failure — the runner itself dies, so neither the per-test terminate-after nor the # `global-timeout` below ever gets to fire, and the job sits until GitHub cancels it at 30 minutes with # no log. Four keeps the shards well under a minute each while leaving headroom for the children. +# Measured 2026-10-02 at four threads, 4 CPUs, cgroup-wide: agent-rest and agent-code-mode peak at +# ~0.3 GB and ~80 tasks on a 16 GB runner, so memory does not explain the wedge; CI's `test` step now +# runs nextest in a systemd unit with MemoryMax/TasksMax/RuntimeMaxSec (see ci.yml). test-threads = 4 [profile.deep] diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index dd48d0a7..71c28a2d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -146,20 +146,56 @@ jobs: test ! -e /work echo "AGENT_LIVE_EXEC_CMD=docker exec agent-exec-target {}" >> "$GITHUB_ENV" echo "AGENT_LIVE_WORKDIR=/work" >> "$GITHUB_ENV" + # A resource trend that outlives a wedge: memory, swap, PSI, load, pid count, disk and the + # biggest processes every 5s, appended to a file by a transient *system* unit. It is outside the + # step's process tree, so it can never hold the step's stdout pipe, and it stops itself. + - name: start resource sampler + run: | + : > "$RUNNER_TEMP/resources.log" + sudo systemd-run --unit=ci-resources --collect --quiet -p RuntimeMaxSec=40min \ + -p StandardOutput=append:"$RUNNER_TEMP/resources.log" -p StandardError=append:"$RUNNER_TEMP/resources.log" \ + sh -c 'while :; do + echo "=== $(date -u +%T) load $(cut -d" " -f1-4 /proc/loadavg) pids $(ls -d /proc/[0-9]* | wc -l)" + free -m | tail -2 + for r in memory cpu io; do echo "psi $r $(head -1 /proc/pressure/$r)"; done + df -h / | tail -1 + ps -eo pid,rss,nlwp,stat,etime,comm --sort=-rss | head -6 + ps -eo pid,stat,wchan:24,comm | awk "\$2 ~ /^D/ {print \"D-state:\", \$0}" + sleep 5 + done' - name: compile tests run: cargo nextest run --workspace --profile ci ${{ matrix.extra }} -E '${{ matrix.filter }}' --no-run timeout-minutes: 20 + # The run is a transient system unit (`systemd-run --wait`), not a child of the step. The + # wedge leaves the test step `in_progress` past the nextest global-timeout, the step timeout and + # the job timeout, so every one of those limits needs the runner itself to act, and it doesn't. + # These limits are enforced by systemd instead: + # RuntimeMaxSec ends the run at 11m (nextest's own global-timeout is 10m) and fails the unit + # with result `timeout`. systemd-run returns then even if a process survives + # its SIGKILL, so the step finishes and the steps after it run. + # MemoryMax measured peak of the agent-rest and agent-code-mode runs is ~0.3 GB (cgroup, + # 4 CPUs, 16 GB). 8 GB is ~25x that and half the runner, so a runaway is + # OOM-killed with result `oom-kill` in the log instead of swapping the VM. + # TasksMax peak is ~80 pids/threads. 4096 stops a fork loop long before the pid table. + # When the unit stops, systemd kills everything left in its cgroup, so a leaked `serve` daemon or + # nats-server cannot outlive the step (nextest still prints `LEAK` naming the test). The unit's + # output is a file, never the step's pipe. The environment is passed explicitly because sudo + # resets it, and cargo is passed by absolute path because systemd-run resolves the binary with + # sudo's PATH, not ours. No GNU `timeout` (see above). - name: test run: | set +e LOG="$RUNNER_TEMP/nextest.log" : > "$LOG" echo "nextest output -> $LOG" - cargo nextest run --workspace --profile ci ${{ matrix.extra }} -E '${{ matrix.filter }}' \ - > "$LOG" 2>&1 /dev/null - pkill -9 -f 'nats-server' 2>/dev/null echo "--- last 200 lines of the log file ---" tail -200 "$LOG" exit $status @@ -169,9 +205,20 @@ jobs: if: always() run: | echo "--- disk ---"; df -h / | tail -1 - echo "--- memory ---"; free -m | head -2 + echo "--- memory ---"; free -m | head -3 echo "--- surviving beyond-ai processes (leaked daemons) ---" ps -eo pid,etime,rss,args | grep -E "beyond-ai|nats-server" | grep -v grep || echo " none" + sudo systemctl stop ci-resources 2>/dev/null + echo "--- last resource samples ---"; tail -60 "$RUNNER_TEMP/resources.log" + - name: upload logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: test-${{ matrix.name }}-logs + path: | + ${{ runner.temp }}/nextest.log + ${{ runner.temp }}/resources.log + if-no-files-found: ignore # The gateway's concurrency-sensitive hermetic tests (reliability, billing, streaming, # cancellation, capture), run repeatedly with every core saturated by a CPU hog. Two real races # (D248's GOAWAY resend, D208's capture drain) passed on a fast dev box and failed only on CI's