Repository navigation
Uptime probe #89
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Uptime probe | |
| # Four complete outages on 2026-09-20 went unnoticed until a third-party | |
| # directory emailed to say our connector was unhealthy. This is the cheapest | |
| # thing that would have told us first. | |
| # | |
| # Every five minutes: probe the cloud's health endpoint, the anonymous MCP demo | |
| # endpoint and the marketing site. On failure, open an issue labelled | |
| # `incident` (or comment on the open one, at most every 30 minutes); on | |
| # recovery, comment and close it. The heap percentage that /health now | |
| # reports is written to the run summary on every probe, so the trend is | |
| # visible here without SSH. | |
| # | |
| # A schedule is best-effort on GitHub — delayed under load, never early. In | |
| # its first ten hours this ran three times, not 120. The minute-by-minute | |
| # probe is a systemd timer on the droplet (deploy/cloud/uptime-probe.sh); | |
| # this one remains as an independent second opinion that does not live on | |
| # the machine it watches, and for the issue it opens. | |
| on: | |
| schedule: | |
| - cron: "*/5 * * * *" | |
| workflow_dispatch: | |
| inputs: | |
| simulate_failure: | |
| description: "Pretend the probes failed, to exercise the incident-issue path" | |
| type: boolean | |
| default: false | |
| permissions: | |
| issues: write | |
| contents: read | |
| concurrency: | |
| group: uptime-probe | |
| cancel-in-progress: false | |
| jobs: | |
| probe: | |
| name: Probe production | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 5 | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| REPO: ${{ github.repository }} | |
| TITLE: "Cloud is down or degraded (uptime probe)" | |
| steps: | |
| - name: Probe | |
| id: probe | |
| shell: bash | |
| run: | | |
| set -uo pipefail | |
| fail=0; report="" | |
| note() { report+="$1"$'\n'; echo "$1"; } | |
| # 1. Backend health — 200 and status ok. A 503 here is the heap | |
| # guard reporting the process degraded before it crashes. | |
| code=$(curl -sS -o /tmp/health.json -w '%{http_code}' --max-time 15 https://cloud.anythingmcp.com/health || echo 000) | |
| status=$(jq -r '.status // "?"' /tmp/health.json 2>/dev/null || echo "?") | |
| heap=$(jq -r '(.info.heap // .details.heap // .error.heap) | select(.!=null) | "\(.percent)% of \(.limitMb) MB (\(.usedMb) MB)"' /tmp/health.json 2>/dev/null || true) | |
| if [ "$code" = "200" ] && [ "$status" = "ok" ]; then | |
| note "✅ cloud /health → 200 ok · heap ${heap:-n/a}" | |
| else | |
| fail=1; note "❌ cloud /health → HTTP $code, status=$status · heap ${heap:-n/a} · $(head -c 300 /tmp/health.json 2>/dev/null | tr -d '\n')" | |
| fi | |
| # 2. Anonymous MCP demo endpoint — what directories like Glama probe. | |
| code=$(curl -sS -o /tmp/mcp.txt -w '%{http_code}' --max-time 20 -X POST \ | |
| -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ | |
| -d '{"jsonrpc":"2.0","id":1,"method":"tools/list","params":{}}' \ | |
| https://cloud.anythingmcp.com/mcp/demo || echo 000) | |
| if [ "$code" = "200" ] && grep -q '"tools"' /tmp/mcp.txt; then | |
| note "✅ cloud /mcp/demo tools/list → 200 with tools" | |
| else | |
| fail=1; note "❌ cloud /mcp/demo → HTTP $code · $(head -c 200 /tmp/mcp.txt 2>/dev/null | tr -d '\n')" | |
| fi | |
| # 3. Marketing site. | |
| code=$(curl -sS -o /dev/null -w '%{http_code}' --max-time 15 -L https://anythingmcp.com/ || echo 000) | |
| if [ "$code" = "200" ]; then note "✅ anythingmcp.com → 200"; else fail=1; note "❌ anythingmcp.com → HTTP $code"; fi | |
| # The one path that matters is the one that never runs on a good day. | |
| if [ "${{ inputs.simulate_failure }}" = "true" ]; then | |
| fail=1; note "🧪 simulated failure (workflow_dispatch input) — the issue below is a drill" | |
| fi | |
| { | |
| echo "### Uptime probe — $(date -u +'%Y-%m-%d %H:%M UTC')" | |
| echo | |
| echo "$report" | |
| } >> "$GITHUB_STEP_SUMMARY" | |
| { | |
| echo "fail=$fail" | |
| echo "report<<EOF" | |
| echo "$report" | |
| echo "EOF" | |
| } >> "$GITHUB_OUTPUT" | |
| - name: Open or update the incident issue | |
| if: steps.probe.outputs.fail == '1' | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| gh label create incident --repo "$REPO" --color B60205 --description "Production outage or degradation detected by the uptime probe" 2>/dev/null || true | |
| existing=$(gh issue list --repo "$REPO" --label incident --state open --search "\"$TITLE\" in:title" --json number,updatedAt --jq '.[0]') | |
| when=$(date -u +'%Y-%m-%d %H:%M UTC') | |
| body=$(printf '**%s**\n\n%s\n\nRun: %s/%s/actions/runs/%s' "$when" "${{ steps.probe.outputs.report }}" "$GITHUB_SERVER_URL" "$REPO" "$GITHUB_RUN_ID") | |
| if [ -z "$existing" ] || [ "$existing" = "null" ]; then | |
| gh issue create --repo "$REPO" --label incident --title "$TITLE" --body "$(printf 'The uptime probe failed. It runs every five minutes; this issue is updated at most every thirty and closed automatically on recovery.\n\n%s' "$body")" | |
| else | |
| num=$(echo "$existing" | jq -r .number) | |
| last=$(gh issue view "$num" --repo "$REPO" --json comments --jq '.comments[-1].createdAt // ""') | |
| last_s=$( [ -n "$last" ] && date -d "$last" +%s || echo 0 ) | |
| if [ $(( $(date +%s) - last_s )) -ge 1800 ]; then | |
| gh issue comment "$num" --repo "$REPO" --body "$(printf 'Still failing.\n\n%s' "$body")" | |
| else | |
| echo "Incident #$num already updated within the last 30 minutes; not commenting again." | |
| fi | |
| fi | |
| - name: Close the incident on recovery | |
| if: steps.probe.outputs.fail == '0' | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| existing=$(gh issue list --repo "$REPO" --label incident --state open --search "\"$TITLE\" in:title" --json number --jq '.[0].number // ""') | |
| if [ -n "$existing" ]; then | |
| gh issue comment "$existing" --repo "$REPO" --body "$(printf 'Recovered at %s.\n\n%s' "$(date -u +'%Y-%m-%d %H:%M UTC')" "${{ steps.probe.outputs.report }}")" | |
| gh issue close "$existing" --repo "$REPO" --reason completed | |
| fi |