diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5063db6..de26ad3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -14,6 +14,12 @@ env: # CPU wheels keep CI light; mirrors the Dockerfile TORCH_INDEX_URL flow. TORCH_INDEX_URL: https://download.pytorch.org/whl/cpu +# Dependency caches are off on these runners. actions/setup-python and +# actions/setup-node resolve their cache globs against the checkout, and the +# self-hosted runner's `_work` is a symlink onto another drive, so the action +# cannot follow the dependency files and fails the job before any step runs +# ("No file ... matched to [**/pyproject.toml]", "Some specified paths were +# not resolved"). The desktop workflow carries the same note for its pip cache. jobs: lint: # Static checks are interpreter-version agnostic; run them once. diff --git a/.github/workflows/deploy-hetzner.yml b/.github/workflows/deploy-hetzner.yml index 9d06482..8a9f5db 100644 --- a/.github/workflows/deploy-hetzner.yml +++ b/.github/workflows/deploy-hetzner.yml @@ -42,10 +42,14 @@ jobs: ssh -i ~/.ssh/deploy_key -o StrictHostKeyChecking=yes \ "$HETZNER_USER@$HETZNER_HOST" \ 'mkdir -p /opt/spikeforge' + # `build/` is generated documentation output — megabytes of it, + # gigabytes with its datasets — and the image never reads it. + # Shipping it filled this host's disk and failed the sync halfway. rsync -rl --delete \ --exclude='.git' \ --exclude='.env' \ --exclude='venv' \ + --exclude='build' \ --exclude='client/node_modules' \ --exclude='client/dist' \ --exclude='**/__pycache__' \ diff --git a/scripts/deploy_dashboard_local.sh b/scripts/deploy_dashboard_local.sh new file mode 100755 index 0000000..81559a2 --- /dev/null +++ b/scripts/deploy_dashboard_local.sh @@ -0,0 +1,55 @@ +#!/usr/bin/env bash +# +# Deploy the dashboard to dash.spikeforge.net over SSH, without waiting on CI. +# +# `.github/workflows/deploy-hetzner.yml` does the same three things, but a push +# to main also starts the seventeen-job CI suite (Python matrices, extras +# builds that pull Torch), and the deploy queues behind it on the same three +# runners. For a client-only change that is minutes of work behind hours of +# queue — the mirror refresh took about two hours that way and about fifteen +# minutes this way. +# +# The steps are the workflow's own, so it needs the same access: an SSH host +# that can reach the deployment machine as a user in the docker group, or root. +# Point it with HETZNER_HOST (default: the `hetzner-airunner` ssh alias). +# +# scripts/deploy_dashboard_local.sh +# +set -euo pipefail + +HOST="${HETZNER_HOST:-hetzner-airunner}" +REMOTE_DIR="${SPIKEFORGE_REMOTE_DIR:-/opt/spikeforge}" +COMPOSE="docker compose -p spikeforge --env-file .env -f docker-compose.hetzner.yml" + +cd "$(dirname "$0")/.." + +# `build/` is generated documentation output: megabytes of it, gigabytes with +# its datasets, and the deployment never reads it. Shipping it once filled the +# deployment host's disk and failed the sync halfway. +echo "==> syncing to ${HOST}:${REMOTE_DIR}" +rsync -rl --delete \ + --exclude='.git' \ + --exclude='.env' \ + --exclude='venv' \ + --exclude='build' \ + --exclude='client/node_modules' \ + --exclude='client/dist' \ + --exclude='**/__pycache__' \ + --exclude='*.egg-info' \ + -e "ssh -o BatchMode=yes" \ + ./ "${HOST}:${REMOTE_DIR}/" + +# The image build writes several gigabytes of layers, and the host runs other +# services, so the disk is the thing that fails first. Reclaim only what no +# container uses. +echo "==> reclaiming unused images and build cache" +ssh -o BatchMode=yes "$HOST" \ + "docker image prune -a -f >/dev/null; docker builder prune -a -f >/dev/null; df -h / | tail -1" + +echo "==> building and restarting" +ssh -o BatchMode=yes "$HOST" \ + "cd ${REMOTE_DIR} && ${COMPOSE} up -d --build --remove-orphans" + +echo "==> health" +ssh -o BatchMode=yes "$HOST" \ + "docker inspect --format '{{.State.Health.Status}}' spikeforge-dashboard"