From 30fba2f9952a12e065994cfd0fc4626f7301b036 Mon Sep 17 00:00:00 2001 From: w4ffl35 <25737761+w4ffl35@users.noreply.github.com> Date: Sat, 19 Sep 2026 05:41:40 -0600 Subject: [PATCH 1/2] Ship the dashboard without the CI queue MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A client-only change had to wait for the seventeen-job CI suite (Python matrices, extras builds that pull Torch) and then queued behind it for the deploy on the same three runners: about two hours for a mirror refresh that takes fifteen minutes to build. scripts/deploy_dashboard_local.sh runs the workflow's own three steps — rsync, rebuild the container, check health — so a maintainer can ship from a workstation with SSH access to the host. It also stops shipping build/, which is generated documentation output the image never reads; pushing it once filled the deployment host's disk and failed the sync halfway. The workflow's rsync excludes it too, and the script first reclaims images and build cache that no container uses, because the image build writes several gigabytes onto a host that runs other services. --- .github/workflows/deploy-hetzner.yml | 4 ++ scripts/deploy_dashboard_local.sh | 55 ++++++++++++++++++++++++++++ 2 files changed, 59 insertions(+) create mode 100755 scripts/deploy_dashboard_local.sh diff --git a/.github/workflows/deploy-hetzner.yml b/.github/workflows/deploy-hetzner.yml index 9d06482..8a9f5db 100644 --- a/.github/workflows/deploy-hetzner.yml +++ b/.github/workflows/deploy-hetzner.yml @@ -42,10 +42,14 @@ jobs: ssh -i ~/.ssh/deploy_key -o StrictHostKeyChecking=yes \ "$HETZNER_USER@$HETZNER_HOST" \ 'mkdir -p /opt/spikeforge' + # `build/` is generated documentation output — megabytes of it, + # gigabytes with its datasets — and the image never reads it. + # Shipping it filled this host's disk and failed the sync halfway. rsync -rl --delete \ --exclude='.git' \ --exclude='.env' \ --exclude='venv' \ + --exclude='build' \ --exclude='client/node_modules' \ --exclude='client/dist' \ --exclude='**/__pycache__' \ diff --git a/scripts/deploy_dashboard_local.sh b/scripts/deploy_dashboard_local.sh new file mode 100755 index 0000000..81559a2 --- /dev/null +++ b/scripts/deploy_dashboard_local.sh @@ -0,0 +1,55 @@ +#!/usr/bin/env bash +# +# Deploy the dashboard to dash.spikeforge.net over SSH, without waiting on CI. +# +# `.github/workflows/deploy-hetzner.yml` does the same three things, but a push +# to main also starts the seventeen-job CI suite (Python matrices, extras +# builds that pull Torch), and the deploy queues behind it on the same three +# runners. For a client-only change that is minutes of work behind hours of +# queue — the mirror refresh took about two hours that way and about fifteen +# minutes this way. +# +# The steps are the workflow's own, so it needs the same access: an SSH host +# that can reach the deployment machine as a user in the docker group, or root. +# Point it with HETZNER_HOST (default: the `hetzner-airunner` ssh alias). +# +# scripts/deploy_dashboard_local.sh +# +set -euo pipefail + +HOST="${HETZNER_HOST:-hetzner-airunner}" +REMOTE_DIR="${SPIKEFORGE_REMOTE_DIR:-/opt/spikeforge}" +COMPOSE="docker compose -p spikeforge --env-file .env -f docker-compose.hetzner.yml" + +cd "$(dirname "$0")/.." + +# `build/` is generated documentation output: megabytes of it, gigabytes with +# its datasets, and the deployment never reads it. Shipping it once filled the +# deployment host's disk and failed the sync halfway. +echo "==> syncing to ${HOST}:${REMOTE_DIR}" +rsync -rl --delete \ + --exclude='.git' \ + --exclude='.env' \ + --exclude='venv' \ + --exclude='build' \ + --exclude='client/node_modules' \ + --exclude='client/dist' \ + --exclude='**/__pycache__' \ + --exclude='*.egg-info' \ + -e "ssh -o BatchMode=yes" \ + ./ "${HOST}:${REMOTE_DIR}/" + +# The image build writes several gigabytes of layers, and the host runs other +# services, so the disk is the thing that fails first. Reclaim only what no +# container uses. +echo "==> reclaiming unused images and build cache" +ssh -o BatchMode=yes "$HOST" \ + "docker image prune -a -f >/dev/null; docker builder prune -a -f >/dev/null; df -h / | tail -1" + +echo "==> building and restarting" +ssh -o BatchMode=yes "$HOST" \ + "cd ${REMOTE_DIR} && ${COMPOSE} up -d --build --remove-orphans" + +echo "==> health" +ssh -o BatchMode=yes "$HOST" \ + "docker inspect --format '{{.State.Health.Status}}' spikeforge-dashboard" From 5f9348c8f1620946fe21904caa53b379a698ac27 Mon Sep 17 00:00:00 2001 From: w4ffl35 <25737761+w4ffl35@users.noreply.github.com> Date: Sat, 19 Sep 2026 06:15:49 -0600 Subject: [PATCH 2/2] Say in the workflow why its dependency caches are off --- .github/workflows/ci.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5063db6..de26ad3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -14,6 +14,12 @@ env: # CPU wheels keep CI light; mirrors the Dockerfile TORCH_INDEX_URL flow. TORCH_INDEX_URL: https://download.pytorch.org/whl/cpu +# Dependency caches are off on these runners. actions/setup-python and +# actions/setup-node resolve their cache globs against the checkout, and the +# self-hosted runner's `_work` is a symlink onto another drive, so the action +# cannot follow the dependency files and fails the job before any step runs +# ("No file ... matched to [**/pyproject.toml]", "Some specified paths were +# not resolved"). The desktop workflow carries the same note for its pip cache. jobs: lint: # Static checks are interpreter-version agnostic; run them once.