diff --git a/.env.example b/.env.example index c4bc8c30..83cf93cd 100644 --- a/.env.example +++ b/.env.example @@ -1,18 +1,26 @@ # maintenant Configuration # ======================== -# Operating mode: embedded (default, server + local runtime), server (accepts -# remote agents), or agent (reports to a server, no HTTP interface). +# Operating mode: embedded (default: web server and local runtime, which also +# accepts remote agents when the edition allows it), server (the same, but it +# refuses to start below the Personal edition), or agent (reports to a server, +# no HTTP interface). # MAINTENANT_MODE=embedded # Listen address (use 0.0.0.0 inside containers, 127.0.0.1 on host) MAINTENANT_ADDR=127.0.0.1:8080 + +# Agent gRPC listener (server and embedded modes, Personal edition or above). +# Loopback by default: remote agents cannot connect until you change it. In a +# container use 0.0.0.0:8443 and publish the port. MAINTENANT_GRPC_LISTEN=127.0.0.1:8443 # Log verbosity: debug, info (default), warn, error # MAINTENANT_LOG_LEVEL=info -# SQLite database path +# SQLite database path. The container image sets /data/maintenant.db. Its +# directory also holds the license cache, the telemetry identity and the +# embedded agent's data, even when the database itself is PostgreSQL. MAINTENANT_DB=./maintenant.db # Optional: back the server data set with a PostgreSQL you already operate. # Empty means SQLite, which is the default and the only agent storage. @@ -22,12 +30,18 @@ MAINTENANT_DB=./maintenant.db # Organisation name (displayed on the public status page) # MAINTENANT_ORGANISATION_NAME=Acme Corp -# Public base URL (used for heartbeat ping URLs and subscriber links) +# Public base URL (heartbeat ping URLs, status page email links and the MCP +# OAuth issuer). Default: http:// # MAINTENANT_BASE_URL=https://maintenant.example.com +# gRPC URL shown to agents in the enrollment commands (default: derived from +# the Host of the web request) # MAINTENANT_GRPC_URL=grpcs://maintenant-agents.example.com -# MAINTENANT_STATUS_URL=https://status.example.com (public) +# Canonical URL of the status page: reported as status_url by GET /api/v1/edition +# and opened by the admin link (default: the relative /status) +# MAINTENANT_STATUS_URL=https://status.example.com -# CORS allowed origins (comma-separated, empty = same-origin only) +# Origins allowed to call the API cross-origin and to send state-changing +# requests from a browser (comma-separated, empty = same-origin only) # MAINTENANT_CORS_ORIGINS=http://localhost:5173 # Reverse proxies whose X-Forwarded-For / X-Real-IP / X-Forwarded-Host headers are @@ -41,7 +55,8 @@ MAINTENANT_DB=./maintenant.db # Create HTTP endpoints from Traefik and Caddy docker-proxy container labels (Docker only) # MAINTENANT_PROXY_LABELS=true -# Max request body size in bytes (default: 1MB) +# Max request body size in bytes (default: 1048576). A value that is not a +# positive whole number stops the startup. # MAINTENANT_MAX_BODY_SIZE=1048576 # Update intelligence scan interval (Go duration, default: 24h) @@ -52,12 +67,13 @@ MAINTENANT_DB=./maintenant.db # Raise an alert when a container has been stopped for this long (Go duration, # e.g. 5m). Unset or 0 disables the check; a container that exited with code 0 -# counts as finished, not down. The alert resolves on its own once it runs again. +# or 143 (or 137 when the out-of-memory killer was not the cause) counts as +# completed, not down. The alert resolves on its own once it runs again. # MAINTENANT_CONTAINER_DOWN_AFTER=5m # How long raw resource samples are kept (Go duration, default: 48h). -# The 24h chart is the longest range reading them — 7d is served from the hourly -# rollup — so 24h is the floor. Raw samples dominate database size. +# The 24h chart is the longest range reading them (7d is served from the hourly +# rollup), so 24h is the floor. Raw samples dominate database size. # MAINTENANT_RETENTION_SNAPSHOTS=48h # Time between two retention passes (Go duration, default: 1h, minimum 1m) @@ -66,37 +82,50 @@ MAINTENANT_DB=./maintenant.db # Rows deleted per transaction during retention (default: 1000, range 100-100000) # MAINTENANT_RETENTION_BATCH_SIZE=1000 -# Score below which the security posture raises an alert (unset = no alert) +# Score below which the security posture raises an alert, from 1 to 100, +# checked every 5 minutes (Personal edition; unset or 0 = no alert, a value outside 0-100 or +# not a number stops the startup) # MAINTENANT_SECURITY_SCORE_THRESHOLD=70 # Opt out of anonymous usage telemetry # MAINTENANT_DISABLE_TELEMETRY=true -# Allow notification webhooks pointing at private addresses (SSRF guard off). -# Only on a network where you trust every webhook target. +# Development only: allow http:// and private, loopback or link-local targets +# for notification channels and webhook subscriptions (SSRF guard off). +# Only on a network where you trust every target. # MAINTENANT_ALLOW_PRIVATE_WEBHOOKS=true -# Kubernetes namespaces to monitor (comma-separated, empty = all) +# Kubernetes namespaces to monitor (comma-separated, empty = all except +# kube-system, kube-public and kube-node-lease) # MAINTENANT_K8S_NAMESPACES=default,production -# Kubernetes namespaces to exclude (comma-separated) -# MAINTENANT_K8S_EXCLUDE_NAMESPACES=kube-system +# Kubernetes namespaces to exclude (comma-separated, added to the three above; +# ignored when an allowlist is set) +# MAINTENANT_K8S_EXCLUDE_NAMESPACES=monitoring -# Pro license key (enables Pro features) +# Personal or Pro license key # MAINTENANT_LICENSE_KEY=your-license-key +# GitHub token used when fetching release notes for the changelog (Personal +# edition). Optional: it only raises the GitHub API rate limit. +# GITHUB_TOKEN=ghp_xxx + # MCP Server (Model Context Protocol for AI assistants) # Both credentials are required: /mcp bypasses the reverse-proxy auth, so # without them maintenant refuses to start rather than serving your monitoring # data to anyone. Set MAINTENANT_MCP_ALLOW_UNAUTHENTICATED=true to accept that -# on a trusted network. +# on a trusted network. Use a secret of at least 32 characters +# (openssl rand -hex 32). # MAINTENANT_MCP=true # MAINTENANT_MCP_CLIENT_ID=maintenant-mcp # MAINTENANT_MCP_CLIENT_SECRET=your-secret-here +# Loopback redirect URIs are always accepted; list the others exactly. # MAINTENANT_MCP_ALLOWED_REDIRECT_URIS=https://claude.ai/api/mcp/auth_callback # MAINTENANT_MCP_ALLOW_UNAUTHENTICATED=true -# SMTP configuration (required for email notification channels) +# SMTP configuration (email alert channel with Personal, status page +# subscribers with Pro). Port 465 uses implicit TLS; on other ports STARTTLS is +# used when the server announces it. # MAINTENANT_SMTP_HOST=smtp.example.com # MAINTENANT_SMTP_PORT=587 # MAINTENANT_SMTP_USERNAME=alerts@example.com @@ -104,13 +133,18 @@ MAINTENANT_DB=./maintenant.db # MAINTENANT_SMTP_FROM=maintenant@example.com # Extra root CA to trust, for hosts signed by an internal PKI (step-ca, AD CS...). -# Added to the system roots, never replacing them. Mount it readable by uid 65534. +# Added to the system roots, never replacing them. Applies to endpoint and +# certificate checks, webhooks and channels, outbound heartbeats, SMTP, the +# license server, registries and the agent's connection to its server. Mount it +# readable by uid 65534; an unreadable file stops the startup. # Prefer this over SSL_CERT_FILE, which replaces the whole bundle and fails silently. # MAINTENANT_CA_CERT=/etc/maintenant/ca.pem # ── Multi-host ──────────────────────────────────────────────────────────────── -# An agent needs the server URL and, on first boot only, an enrollment token. -# The token is consumed once; the agent then authenticates with its own key. +# An agent needs the server URL and, on first boot, an enrollment token. The +# token is consumed once; the agent then authenticates with its own key. If the +# server refuses the stored identity (agent revoked or deleted), the agent +# enrolls again with the token when it is still set, and stops otherwise. # MAINTENANT_SERVER=grpcs://maintenant-agents.example.com:8443 # MAINTENANT_ENROLLMENT_TOKEN=paste-the-token-from-the-Agents-page # MAINTENANT_LABEL=web-01 @@ -119,14 +153,17 @@ MAINTENANT_DB=./maintenant.db # operating system. Left empty, the agent finds it from its own pod. # MAINTENANT_NODE_NAME=node-01 -# Where the agent keeps its identity and liveness files (agent mode) +# Where the agent keeps its identity, its spool database and its liveness file +# (agent mode) # MAINTENANT_DATA_DIR=/var/lib/maintenant # Skip TLS verification when reaching the server. Debug only: it defeats the # point of TLS. Use MAINTENANT_CA_CERT for an internal PKI instead. # MAINTENANT_GRPC_INSECURE_SKIP_TLS_VERIFY=true -# Server side: TLS for the gRPC listener agents connect to. +# Server side: TLS for the gRPC listener agents connect to. Set both or neither: +# one without the other stops the startup. With neither, a self-signed +# certificate is generated at each start and a warning is logged. # MAINTENANT_GRPC_TLS_CERT=/etc/maintenant/grpc.crt # MAINTENANT_GRPC_TLS_KEY=/etc/maintenant/grpc.key # Serve gRPC as h2c instead, without TLS. Only behind a reverse proxy that @@ -139,10 +176,11 @@ MAINTENANT_DB=./maintenant.db # MAINTENANT_AGENT_STALE_THRESHOLD_SECONDS=60 # Agent side: local queue holding events while the server is unreachable. -# Setting both budgets to 0 disables it. +# Setting both the memory and the disk budget to 0 disables it. Events older +# than the age limit are dropped from it; 0 turns the limit off. # MAINTENANT_AGENT_SPOOL_MAX_MEMORY_BYTES=16777216 # MAINTENANT_AGENT_SPOOL_MAX_DISK_BYTES=134217728 # MAINTENANT_AGENT_SPOOL_MAX_AGE_SECONDS=86400 -# Also run a local agent alongside the server (server mode, Pro) +# Also run a local agent alongside the server (server mode, Personal edition or above) # MAINTENANT_EMBEDDED_AGENT=true diff --git a/Dockerfile b/Dockerfile index ba75e9de..481e1104 100644 --- a/Dockerfile +++ b/Dockerfile @@ -42,7 +42,8 @@ RUN --mount=type=cache,target=/go/pkg/mod \ FROM alpine:3.21 -RUN apk add --no-cache ca-certificates tzdata setpriv \ +RUN apk upgrade --no-cache \ + && apk add --no-cache ca-certificates tzdata setpriv \ && mkdir -p /data \ && chown 65534:65534 /data @@ -51,6 +52,9 @@ RUN apk add --no-cache ca-certificates tzdata setpriv \ # /tmp as a tiny tmpfs, which SQLITE_FULL-fails the conversion; /data has real space. ENV SQLITE_TMPDIR=/data +# Its directory also holds the licence cache and the update window, PostgreSQL or not. +ENV MAINTENANT_DB=/data/maintenant.db + # Tells the OS identity reader it must not fall back to the image's own # /etc/os-release, which describes the container rather than the host. ENV MAINTENANT_CONTAINER=1 diff --git a/README.md b/README.md index d5831636..66cf7172 100644 --- a/README.md +++ b/README.md @@ -61,11 +61,12 @@ docker compose up -d Open **http://localhost:8080**. Your containers are already there, with their health, restart loops, resources and logs. Nothing to configure. -Docker socket access is automatic: the entrypoint reads the mounted socket's group and grants it to the unprivileged user, on Compose and on Swarm (where `docker stack deploy` silently ignores `group_add`). If containers do not show up, see [Troubleshooting](https://docs.maintenant.dev/troubleshooting/). +Docker socket access is automatic: the entrypoint reads the mounted socket's group and grants it to the unprivileged user, on Compose and on Swarm (where `docker stack deploy` silently ignores `group_add`). A socket owned by group `root` needs `DOCKER_GID=0`. If containers do not show up, see [Troubleshooting](https://docs.maintenant.dev/troubleshooting/). **Kubernetes** ```bash +kubectl create namespace maintenant kubectl apply -f deploy/kubernetes/ ``` @@ -234,11 +235,11 @@ docker run -d --name maintenant-agent --restart unless-stopped \ --enrollment-token=mnt_enr_XXXXXXXXXXXXXXXX --label="prod-worker-01" ``` -Agents detect their local runtime (Docker, Swarm or Kubernetes), stream container state, endpoints, certificates, host CPU/memory/disk, and reconnect on their own. Every entity is attributed to its host, so nothing gets mixed across machines. *Personal: up to 20 remote machines. Pro: unlimited.* +Agents detect their local runtime (Docker, Swarm or Kubernetes), stream container state, endpoints, certificates, host CPU/memory/disk, and reconnect on their own, replaying what they saw while disconnected as history. Every entity is attributed to its host, so nothing gets mixed across machines. *Personal: up to 20 remote machines. Pro: unlimited.* ### [Update intelligence](https://docs.maintenant.dev/features/updates/) -Scans OCI registries and compares digests, so you know which images have an update before you `docker pull` blindly. Compose-aware update and rollback commands, with the right `--project-directory`. No Diun, no Watchtower, no extra container: it is part of the monitor. +Scans OCI registries: newer versions for fixed tags, and for floating tags like `latest` a comparison between the digest your container runs and the one the tag points at now. You know which images have an update before you `docker pull` blindly. Update and rollback commands for Compose, plain Docker, Swarm and Kubernetes, with the right `cd` into the Compose project. No Diun, no Watchtower, no extra container: it is part of the monitor. ### [Host OS end-of-support](https://docs.maintenant.dev/features/host-os/) @@ -246,7 +247,7 @@ Every monitored host reports its distribution and version; maintenant checks it ### [Endpoint monitoring](https://docs.maintenant.dev/features/endpoints/) -HTTP and TCP checks declared as Docker labels, picked up when the container starts. Response times, uptime history, 90-day sparklines, failure and recovery thresholds. +HTTP and TCP checks declared as Docker labels, picked up when the container starts, or added by hand. Endpoints can also be derived from Traefik and Caddy labels (opt-in). Response times, uptime history, 90-day sparklines, failure and recovery thresholds. ```yaml labels: @@ -257,7 +258,7 @@ labels: ### [Heartbeat and cron monitoring](https://docs.maintenant.dev/features/heartbeats/) -Create a monitor, get a URL, add one `curl` to the job. maintenant tracks start and finish, duration, exit code, and alerts when the deadline is missed. +Create a monitor, get a URL, add one `curl` to the job. maintenant tracks start and finish, duration, exit code, and alerts when the deadline is missed. Outbound heartbeats let two maintenant instances watch each other, so a dead monitor does not go unnoticed. ```bash curl -fsS -o /dev/null https://now.example.com/ping/{uuid}/$? @@ -273,21 +274,21 @@ Real-time CPU, memory, network and disk I/O per container and per host, top-cons ### [Network security insights](https://docs.maintenant.dev/features/security/) -Flags what should not be there: ports bound to `0.0.0.0`, exposed database ports, host-network mode, privileged containers, Kubernetes NodePort and LoadBalancer services without a NetworkPolicy. Each image is mapped to its software ecosystem through OCI manifest inspection. **Personal** adds CVE enrichment, a risk score per container and a unified security posture dashboard. +Flags what should not be there: ports bound to `0.0.0.0`, exposed database ports, host-network mode, privileged containers, Kubernetes Services of type NodePort or LoadBalancer, and database ports exposed through them. **Personal** adds CVE enrichment (each image is mapped to its software ecosystem through OCI manifest inspection), a risk score per container and a unified security posture dashboard. ### [Alert engine](https://docs.maintenant.dev/features/alerts/) -One alert pipeline for every source: container restart loops and unhealthy checks, endpoint failures, missed heartbeats, expiring or invalid certificates, CPU and memory thresholds, available updates. Channels are silent by default and routed through **triggers** (severity, source, scope, tags). Silence rules for planned maintenance, exponential backoff on delivery. +One alert pipeline for every source: container restart loops, unhealthy checks and stopped containers, endpoint failures, missed heartbeats, expiring or invalid certificates, CPU and memory thresholds, available updates, Swarm and Kubernetes health, agents going offline, hosts whose OS loses support. Channels are silent by default and routed through **triggers** (severity, source, scope). Alerts can be acknowledged. Silence rules for planned maintenance, three delivery attempts per notification. -Channels: Discord and webhooks (Community), email and Telegram (Personal), Slack and Microsoft Teams (Pro). **Pro** adds [escalation policies](https://docs.maintenant.dev/features/alert-escalation/) that page the on-call, then the backup, then the lead, plus per-entity routing and maintenance windows. +Channels: Discord and webhooks (Community), email and Telegram (Personal), Slack and Microsoft Teams (Pro). **Pro** adds [escalation policies](https://docs.maintenant.dev/features/alert-escalation/) of up to five levels that page the on-call, then the backup, then the lead, plus maintenance windows. ### [Public status page](https://docs.maintenant.dev/features/status-page/) -Real-time status page with severity aggregation across every monitor, live over SSE. **Personal** adds incident timelines, **Pro** adds subscriber notifications (email and webhook) and branding. +Real-time status page with severity aggregation across every monitor, live over SSE. **Personal** adds incident timelines, **Pro** adds email subscribers (double opt-in, through your own SMTP server), maintenance windows and branding. ### [MCP server](https://docs.maintenant.dev/features/mcp/) -Built-in [Model Context Protocol](https://modelcontextprotocol.io/) server. Ask your AI assistant what is burning, read a container's logs, check the alert queue, acknowledge an alert, open an incident. stdio and Streamable HTTP transports, full OAuth2 for remote clients (Claude web, mobile and Desktop). +Built-in [Model Context Protocol](https://modelcontextprotocol.io/) server with 51 tools. Ask your AI assistant what is burning, read a container's logs, check the alert queue, acknowledge an alert, open an incident. stdio and Streamable HTTP transports, OAuth2 with a client id and secret for remote clients (Claude web, mobile and Desktop). --- @@ -296,7 +297,7 @@ Built-in [Model Context Protocol](https://modelcontextprotocol.io/) server. Ask Everything is driven by **Docker labels** and a handful of **environment variables**. No YAML to maintain. - [Environment variables](https://docs.maintenant.dev/getting-started/configuration/): bind address, database, base URL, PostgreSQL DSN, MCP, Kubernetes namespaces, license key, telemetry. -- [Docker labels reference](https://docs.maintenant.dev/guides/docker-labels/): endpoints, TLS, alert severity, restart thresholds, channel routing, grouping, ignore. +- [Docker labels reference](https://docs.maintenant.dev/guides/docker-labels/): endpoints, TLS, alert severity, restart thresholds, update tracking, grouping, ignore. - [REST API](https://docs.maintenant.dev/api/reference/) under `/api/v1/`, plus an SSE event stream.
@@ -331,7 +332,6 @@ services: maintenant.endpoint.http: "http://api:3000/health" maintenant.endpoint.interval: "15s" maintenant.alert.severity: "critical" - maintenant.alert.channels: "ops-webhook" postgres: image: postgres:16 @@ -356,8 +356,8 @@ volumes: - **No built-in authentication, by design.** Like Dozzle and Prometheus, maintenant sits behind your reverse proxy and auth middleware (Traefik or Caddy, Authelia or Authentik). `/ping/{uuid}` and `/status/` are meant to stay public. [Reverse proxy setup](https://docs.maintenant.dev/security/#reverse-proxy-setup). - **Read-only everywhere.** Docker socket mounted `:ro`, read-only RBAC on Kubernetes, read-only agents. maintenant never starts, stops or modifies a container. A [socket proxy](https://docs.maintenant.dev/security/#recommended-docker-socket-proxy) is supported if you would rather not mount the socket at all. -- **Hardened container.** Runs as `nobody`, `read_only` root filesystem, `no-new-privileges`. -- **Anonymous, opt-out telemetry.** One counts-only snapshot per hour, no hostnames, IPs, names, URLs or keys, ever. `MAINTENANT_DISABLE_TELEMETRY=1` turns it off with no background goroutine and no outbound packet. [Exact payload and details](https://docs.maintenant.dev/getting-started/configuration/#telemetry). +- **Hardened container.** Drops to `nobody` (uid 65534) once it has fixed the ownership of its volume, `read_only` root filesystem, `no-new-privileges`. +- **Anonymous, opt-out telemetry.** One snapshot per hour: counts of monitored objects, edition, storage engine, version and runtime figures (OS, architecture, CPU cores, memory). No hostnames, IPs, names, URLs or keys, ever. `MAINTENANT_DISABLE_TELEMETRY=1` turns it off with no background goroutine and no outbound packet. [Exact payload and details](https://docs.maintenant.dev/getting-started/configuration/#telemetry). --- @@ -385,14 +385,14 @@ Community is free forever and runs production infrastructure every day: it is th | Heartbeats | 5 | unlimited | unlimited | | Certificates | 5 | unlimited | unlimited | | Resource history | 7 days | 30 days | 90 days | -| Alert channels | Discord, webhooks | + email, Telegram, advanced filters | + Slack, Teams, escalation, per-entity routing, maintenance windows | +| Alert channels | Discord, webhooks | + email, Telegram, advanced filters | + Slack, Teams, escalation, maintenance windows | | Security | network insights | + CVE enrichment, risk scoring, security posture, OCSP | same | | Status page | 3 components | unlimited, incident timelines | + subscriber notifications, branding | -| Use | anything | your own infrastructure | + running it for others, priority support | +| Use | anything | your own infrastructure | + running it for others, email support | -Personal covers one person on infrastructure they own or run for themselves, freelancers included, and ships with one year of updates (then €59 per extra year; every version released inside a paid year stays licensed for life). Pro adds the right to monitor other people's infrastructure. Enterprise (SSO, audit logs, SLAs, on-prem support): [hello@kolapsis.com](mailto:hello@kolapsis.com). +Personal covers one person on infrastructure they own or run for themselves, freelancers included, and ships with one year of updates (then €59 per extra year; every version released inside a paid year stays licensed for life). Pro adds the right to monitor other people's infrastructure. Volume pricing and custom agreements: [license@maintenant.dev](mailto:license@maintenant.dev). -Paid editions are the same binary, self-hosted the same way. The key is verified against the license server and the signed answer is cached, so being offline for weeks changes nothing. **Your monitoring data never leaves your infrastructure.** +Paid editions are the same binary, self-hosted the same way. The key is verified against the license server and the signed answer is cached: the instance falls back to Community only after 60 days without reaching the server, so a few weeks offline change nothing. **Your monitoring data never leaves your infrastructure.** ```bash MAINTENANT_LICENSE_KEY=your-license-key # Personal or Pro, restart, done diff --git a/SECURITY.md b/SECURITY.md index 928d4455..bcd133c4 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -7,8 +7,8 @@ no backports: upgrade to the current release before reporting an issue. | Version | Supported | | ------- | --------- | -| 1.3.x | Yes | -| < 1.3 | No | +| 1.8.x | Yes | +| < 1.8 | No | The container image `ghcr.io/kolapsis/maintenant` is rebuilt for every release, so the `latest` tag always carries the patched base image. @@ -34,10 +34,10 @@ Please include, as far as you can establish them: We commit to the following, counted from the moment your advisory is received: -- **48 hours** — acknowledgement of receipt. -- **5 days** — first assessment: reproduction, severity, affected versions. -- **30 days** — a fix released, or a written remediation plan with a date. -- **90 days** — coordinated public disclosure, whether or not a fix has shipped. +- **48 hours**: acknowledgement of receipt. +- **5 days**: first assessment: reproduction, severity, affected versions. +- **30 days**: a fix released, or a written remediation plan with a date. +- **90 days**: coordinated public disclosure, whether or not a fix has shipped. If you need a different timeline (an upcoming conference talk, an embargo agreed with another vendor), say so in the advisory and we will coordinate. We credit @@ -46,20 +46,28 @@ not to. ## Supply chain guarantees -Every released image is verifiable by a third party, with no privileged access -to this repository. +Every release is verifiable by a third party, with no privileged access to this +repository. A release publishes two kinds of artifact: the container image and +the static Linux binaries used by the native install script. + +### Container image + +The images tagged for a release (`X.Y.Z`, `X.Y`, `X` and `latest`) are signed +and attested. The `main` and commit-SHA tags, published on every merge to +`main`, and the `demo` image are neither signed nor attested: do not rely on +them where verification matters. Build provenance (SLSA), signed keyless through GitHub OIDC: ```bash -gh attestation verify oci://ghcr.io/kolapsis/maintenant:1.3.7 --owner kOlapsis +gh attestation verify oci://ghcr.io/kolapsis/maintenant:1.8.0 --owner kOlapsis ``` Cosign signature. Both flags are required: without them, `cosign verify` would accept any identity. ```bash -cosign verify ghcr.io/kolapsis/maintenant:1.3.7 \ +cosign verify ghcr.io/kolapsis/maintenant:1.8.0 \ --certificate-identity-regexp "https://github.com/kOlapsis/maintenant/.github/workflows/release.yml@.*" \ --certificate-oidc-issuer "https://token.actions.githubusercontent.com" ``` @@ -69,7 +77,48 @@ Note that `sigstore/cosign-installer@v4` signs with Cosign 3.x. A local Cosign local version before concluding anything. A CycloneDX SBOM is attached to each release as `sbom.cdx.json` and attested in -the registry alongside the image. +the registry alongside the image. `provenance.intoto.jsonl`, also attached to +the release, is the provenance attestation of the image. + +### Linux binaries + +Each release carries `maintenant-vX.Y.Z-linux-amd64`, `maintenant-vX.Y.Z-linux-arm64` +and `install.sh`, plus: + +- `SHA256SUMS`: the checksums of the two binaries and of `install.sh`. +- `SHA256SUMS.bundle`: the keyless Cosign signature of `SHA256SUMS`, with its + certificate. `install.sh` is covered by this signature through the checksum + file, and has no provenance attestation of its own. +- A SLSA build provenance attestation for each binary, stored by GitHub and + checked with `gh attestation verify`. + +```bash +VERSION=v1.8.0 +ARCH=amd64 +BASE=https://github.com/kOlapsis/maintenant/releases/download/${VERSION} + +curl -LO ${BASE}/maintenant-${VERSION}-linux-${ARCH} +curl -LO ${BASE}/SHA256SUMS +curl -LO ${BASE}/SHA256SUMS.bundle + +# Checksum +sha256sum -c SHA256SUMS --ignore-missing + +# Signature of SHA256SUMS (Cosign 3.x) +cosign verify-blob \ + --bundle SHA256SUMS.bundle \ + --certificate-identity-regexp \ + "^https://github\.com/kOlapsis/maintenant/\.github/workflows/release\.yml@refs/tags/v[0-9]+\.[0-9]+\.[0-9]+$" \ + --certificate-oidc-issuer https://token.actions.githubusercontent.com \ + SHA256SUMS + +# SLSA provenance +gh attestation verify maintenant-${VERSION}-linux-${ARCH} --owner kOlapsis +``` + +The native install script runs the checksum check on every install and the +Cosign check when Cosign 3 or later is installed; see +[Supply-chain verification](https://docs.maintenant.dev/install/#supply-chain-verification). ## Hardening the deployment @@ -79,6 +128,8 @@ points matter more than the rest: - **Do not mount `/var/run/docker.sock` directly.** Use a read-only docker-socket-proxy; see [Recommended: Docker Socket Proxy](https://docs.maintenant.dev/security/#recommended-docker-socket-proxy) for the configuration. A mounted socket is equivalent to root on the host. -- **Do not bind the listener to `0.0.0.0`** on a machine reachable from an - untrusted network. Bind to a private interface, or put an authenticating - reverse proxy in front; see [Reverse Proxy Setup](https://docs.maintenant.dev/security/#reverse-proxy-setup). +- **Do not expose the listener to an untrusted network.** maintenant has no + login of its own and listens on `127.0.0.1:8080` by default. Inside a + container it has to listen on `0.0.0.0:8080`, but publish that port only to a + private interface, or put an authenticating reverse proxy in front; see + [Reverse Proxy Setup](https://docs.maintenant.dev/security/#reverse-proxy-setup). diff --git a/cmd/maintenant/image_test.go b/cmd/maintenant/image_test.go new file mode 100644 index 00000000..e8ed7a3b --- /dev/null +++ b/cmd/maintenant/image_test.go @@ -0,0 +1,153 @@ +// Copyright 2026 Benjamin Touchard (kOlapsis) +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestImageKeepsTheDatabaseOnTheDataVolume(t *testing.T) { + dockerfile, err := os.ReadFile(filepath.Join("..", "..", "Dockerfile")) + require.NoError(t, err) + + env := map[string]string{} + var volumes []string + for _, line := range strings.Split(string(dockerfile), "\n") { + fields := strings.Fields(line) + if len(fields) < 2 { + continue + } + switch fields[0] { + case "ENV": + for _, kv := range fields[1:] { + if k, v, ok := strings.Cut(kv, "="); ok { + env[k] = v + } + } + case "VOLUME": + volumes = append(volumes, fields[1:]...) + } + } + + db := env["MAINTENANT_DB"] + require.NotEmpty(t, db, "without MAINTENANT_DB the binary opens ./maintenant.db on the read-only root") + assert.Contains(t, volumes, filepath.Dir(db)) +} + +type entrypointRun struct { + stubs string + log string +} + +// newEntrypointRun stubs every command the entrypoint runs, so it can be driven as root without touching the host. +func newEntrypointRun(t *testing.T) *entrypointRun { + t.Helper() + stubs := t.TempDir() + record := "#!/bin/sh\nprintf '%s %s\\n' \"${0##*/}\" \"$*\" >>\"$STUB_LOG\"\n" + for _, name := range []string{"chown", "mkdir", "setpriv", "stat", "maintenant"} { + require.NoError(t, os.WriteFile(filepath.Join(stubs, name), []byte(record), 0o700)) + } + id := "#!/bin/sh\necho \"$STUB_UID\"\n" + require.NoError(t, os.WriteFile(filepath.Join(stubs, "id"), []byte(id), 0o700)) + return &entrypointRun{stubs: stubs, log: filepath.Join(stubs, "calls.log")} +} + +func (e *entrypointRun) binary() string { return filepath.Join(e.stubs, "maintenant") } + +// run executes the entrypoint as uid and returns the stub calls, one "name args" per entry. +func (e *entrypointRun) run(t *testing.T, uid string, env map[string]string, args ...string) []string { + t.Helper() + cmd := exec.Command("/bin/sh", append([]string{filepath.Join("..", "..", "docker-entrypoint.sh")}, args...)...) + cmd.Env = []string{"PATH=" + e.stubs + ":" + os.Getenv("PATH"), "STUB_LOG=" + e.log, "STUB_UID=" + uid} + for k, v := range env { + cmd.Env = append(cmd.Env, k+"="+v) + } + out, err := cmd.CombinedOutput() + require.NoError(t, err, string(out)) + data, err := os.ReadFile(e.log) + require.NoError(t, err) + return strings.Split(strings.TrimSpace(string(data)), "\n") +} + +func calls(all []string, name string) []string { + var out []string + for _, c := range all { + if cmd, args, _ := strings.Cut(c, " "); cmd == name { + out = append(out, args) + } + } + return out +} + +func chowned(all []string) []string { + var dirs []string + for _, args := range calls(all, "chown") { + f := strings.Fields(args) + dirs = append(dirs, f[len(f)-1]) + } + return dirs +} + +func TestEntrypoint_UnprivilegedRunsTheCommandItself(t *testing.T) { + e := newEntrypointRun(t) + got := e.run(t, "65534", nil, e.binary(), "--mode=agent") + + assert.Equal(t, []string{"maintenant --mode=agent"}, got) +} + +func TestEntrypoint_AgentOwnsItsDataDirectory(t *testing.T) { + dataDir := t.TempDir() + cases := map[string]struct { + env map[string]string + args []string + }{ + "mode and data dir from the environment": { + env: map[string]string{"MAINTENANT_MODE": "agent", "MAINTENANT_DATA_DIR": dataDir}, + }, + "mode and data dir from flags": { + args: []string{"--mode", "agent", "--data-dir=" + dataDir}, + }, + "mode flag, data dir from the environment": { + env: map[string]string{"MAINTENANT_DATA_DIR": dataDir}, + args: []string{"-mode=agent"}, + }, + } + for name, tc := range cases { + t.Run(name, func(t *testing.T) { + e := newEntrypointRun(t) + got := e.run(t, "0", tc.env, append([]string{e.binary()}, tc.args...)...) + + assert.Equal(t, []string{dataDir}, chowned(got)) + assert.Len(t, calls(got, "setpriv"), 1) + }) + } +} + +func TestEntrypoint_ServerOwnsTheDatabaseDirectory(t *testing.T) { + dbDir := t.TempDir() + flagDir := t.TempDir() + + e := newEntrypointRun(t) + got := e.run(t, "0", map[string]string{"MAINTENANT_DB": filepath.Join(dbDir, "maintenant.db")}, e.binary()) + assert.ElementsMatch(t, []string{dbDir, "/data/shm"}, chowned(got)) + assert.Len(t, calls(got, "setpriv"), 1) + + e = newEntrypointRun(t) + got = e.run(t, "0", map[string]string{"MAINTENANT_MODE": "agent"}, e.binary(), "--mode=server", "--db", filepath.Join(flagDir, "m.db")) + assert.ElementsMatch(t, []string{flagDir, "/data/shm"}, chowned(got)) +} + +func TestEntrypoint_NeverHandsTheRootDirectoryOver(t *testing.T) { + e := newEntrypointRun(t) + got := e.run(t, "0", map[string]string{"MAINTENANT_DB": "/maintenant.db"}, e.binary()) + + assert.Equal(t, []string{"/data/shm"}, chowned(got)) +} diff --git a/cmd/maintenant/main.go b/cmd/maintenant/main.go index 5408f8f0..c1db32cc 100644 --- a/cmd/maintenant/main.go +++ b/cmd/maintenant/main.go @@ -127,6 +127,11 @@ func main() { os.Exit(1) } + if err := cfg.ValidateBodySize(); err != nil { + logger.Error("invalid HTTP configuration", "error", err) + os.Exit(1) + } + // A threshold that does not parse must stop startup: falling back to "off" // would leave the operator believing the check runs. if err := cfg.ValidateAlerting(); err != nil { diff --git a/cmd/maintenant/storage_errors.go b/cmd/maintenant/storage_errors.go index a08c843e..aa59573d 100644 --- a/cmd/maintenant/storage_errors.go +++ b/cmd/maintenant/storage_errors.go @@ -24,6 +24,12 @@ func logStorageStartupError(logger *slog.Logger, err error, dsn string) bool { case errors.Is(err, store.ErrInvalidDSN): logger.Error("the database connection string cannot be read", "fix", "expected postgres://user:password@host:5432/database[?sslmode=require]") + case errors.Is(err, store.ErrTLSRefused) && store.DefaultsSSLMode(dsn): + logger.Error("the database server does not accept TLS", "target", target, + "fix", "sslmode=require is added by default for a non-local host: enable TLS on the PostgreSQL server, or set sslmode=disable explicitly in MAINTENANT_DATABASE_URL if the network to the database is trusted") + case errors.Is(err, store.ErrTLSRefused): + logger.Error("the database server does not accept TLS", "target", target, + "fix", "the connection string requires TLS through its sslmode: enable TLS on the PostgreSQL server, or lower sslmode if the network to the database is trusted") case errors.Is(err, store.ErrUnreachable): logger.Error("the database does not answer", "target", target, "fix", "check the host and port, the network route and the firewall; the instance does not fall back to the local file") diff --git a/cmd/maintenant/storage_errors_test.go b/cmd/maintenant/storage_errors_test.go index b714bfcb..7e45d363 100644 --- a/cmd/maintenant/storage_errors_test.go +++ b/cmd/maintenant/storage_errors_test.go @@ -31,6 +31,7 @@ func TestLogStorageStartupError_DistinctFamilies(t *testing.T) { "agent mode": fmt.Errorf("wrapped: %w", app.ErrDatabaseURLInAgentMode), "invalid dsn": fmt.Errorf("open database: %w", store.ErrInvalidDSN), "unreachable": fmt.Errorf("open database: %w", store.ErrUnreachable), + "tls refused": fmt.Errorf("open database: %w", store.ErrTLSRefused), "credentials": fmt.Errorf("open database: %w", store.ErrAuthRefused), "version": fmt.Errorf("open database: %w", store.ErrUnsupportedVersion), "schema from future": fmt.Errorf("run migrations: %w", store.ErrSchemaNewer), @@ -64,6 +65,25 @@ func TestLogStorageStartupError_DistinctFamilies(t *testing.T) { assert.Contains(t, messages["version"], "14") } +func TestLogStorageStartupError_TLSRefusedNamesTheDefault(t *testing.T) { + err := fmt.Errorf("open database: %w", store.ErrTLSRefused) + + var defaulted bytes.Buffer + require.True(t, logStorageStartupError(slog.New(slog.NewTextHandler(&defaulted, nil)), err, + "postgres://app:"+cliSentinelPassword+"@db:5432/maintenant")) + assert.Contains(t, defaulted.String(), "does not accept TLS") + assert.Contains(t, defaulted.String(), "sslmode=require is added by default") + assert.Contains(t, defaulted.String(), "sslmode=disable") + assert.NotContains(t, defaulted.String(), "firewall") + assert.NotContains(t, defaulted.String(), cliSentinelPassword) + + var explicit bytes.Buffer + require.True(t, logStorageStartupError(slog.New(slog.NewTextHandler(&explicit, nil)), err, + "postgres://app@db:5432/maintenant?sslmode=require")) + assert.Contains(t, explicit.String(), "does not accept TLS") + assert.NotContains(t, explicit.String(), "added by default", "the operator asked for TLS") +} + // TestLogStorageStartupError_PassesThroughOtherErrors keeps the classifier // honest: an error that is not a storage startup failure is left to the // caller's generic handler rather than mislabelled. diff --git a/compose.test.yml b/compose.test.yml index 93a42737..36f59213 100644 --- a/compose.test.yml +++ b/compose.test.yml @@ -48,7 +48,6 @@ services: MAINTENANT_DATABASE_URL: "${MAINTENANT_DATABASE_URL:-}" MAINTENANT_ORGANISATION_NAME: "e2e" MAINTENANT_DISABLE_TELEMETRY: "true" - MAINTENANT_TELEMETRY_DATADIR: "/data" MAINTENANT_SMTP_HOST: mailpit MAINTENANT_SMTP_PORT: "1025" MAINTENANT_SMTP_FROM: "e2e@maintenant.test" @@ -57,7 +56,6 @@ services: # (server mode, escalation policies, security posture). Without it the # stack runs in Community, which covers everything storage touches. MAINTENANT_LICENSE_KEY: "${MAINTENANT_LICENSE_KEY:-}" - MAINTENANT_TZ: "Europe/Paris" healthcheck: # Same probe as the image, on a shorter cycle: a test run should not # wait 30s to learn the instance is up. diff --git a/deploy/cloud-init/maintenant.yaml b/deploy/cloud-init/maintenant.yaml index 07570de7..ff4205c3 100644 --- a/deploy/cloud-init/maintenant.yaml +++ b/deploy/cloud-init/maintenant.yaml @@ -1,5 +1,5 @@ #cloud-config -# maintenant — first-boot provisioning for any Ubuntu cloud server. +# maintenant: first-boot provisioning for any Ubuntu cloud server. # # Pass this file as user data when creating the server. Each provider spells # that differently: @@ -8,14 +8,24 @@ # --user-data-from-file deploy/cloud-init/maintenant.yaml # DigitalOcean doctl compute droplet create --image ubuntu-24-04-x64 \ # --user-data-file deploy/cloud-init/maintenant.yaml +# Scaleway scw instance server create image=ubuntu_noble \ +# cloud-init=@deploy/cloud-init/maintenant.yaml +# OVHcloud openstack server create --user-data deploy/cloud-init/maintenant.yaml +# (Public Cloud only: a VPS cannot run cloud-init) +# Vultr vultr-cli instance create --os= \ +# --userdata-file=deploy/cloud-init/maintenant.yaml # # The dashboard is published on 127.0.0.1:8080 only. Reach it through an SSH # tunnel, or put a reverse proxy with authentication in front of it: # -# ssh -L 8080:127.0.0.1:8080 root@ +# ssh -L 8080:127.0.0.1:8080 root@ (OVHcloud images: ubuntu@) # # Per-provider guides (firewall, volumes, private network, managed Kubernetes): # https://docs.maintenant.dev/guides/hetzner/ +# https://docs.maintenant.dev/guides/digitalocean/ +# https://docs.maintenant.dev/guides/scaleway/ +# https://docs.maintenant.dev/guides/ovhcloud/ +# https://docs.maintenant.dev/guides/vultr/ package_update: true package_upgrade: true diff --git a/deploy/helm/maintenant/Chart.yaml b/deploy/helm/maintenant/Chart.yaml index fa93017a..8c013220 100644 --- a/deploy/helm/maintenant/Chart.yaml +++ b/deploy/helm/maintenant/Chart.yaml @@ -2,8 +2,8 @@ apiVersion: v2 name: maintenant description: Self-discovering infrastructure monitoring for Docker and Kubernetes type: application -version: 1.2.0 -appVersion: "1.2.0" +version: 1.3.0 +appVersion: "1.8.0" home: https://github.com/kolapsis/maintenant sources: - https://github.com/kolapsis/maintenant diff --git a/deploy/helm/maintenant/templates/rbac.yaml b/deploy/helm/maintenant/templates/rbac.yaml index 8fe6aec0..6994a5bd 100644 --- a/deploy/helm/maintenant/templates/rbac.yaml +++ b/deploy/helm/maintenant/templates/rbac.yaml @@ -6,17 +6,22 @@ metadata: labels: {{- include "maintenant.labels" . | nindent 4 }} rules: - # Core resources — read-only + # Read-only, exactly the APIs the runtime calls (internal/kubernetes/rbac.go). - apiGroups: [""] - resources: ["pods", "pods/log", "services", "namespaces", "events"] + resources: ["namespaces", "nodes", "pods", "events", "services"] verbs: ["get", "list", "watch"] - # Apps — read-only + - apiGroups: [""] + resources: ["pods/log"] + verbs: ["get"] - apiGroups: ["apps"] - resources: ["deployments", "statefulsets", "daemonsets", "replicasets"] + resources: ["deployments", "statefulsets", "daemonsets"] + verbs: ["get", "list", "watch"] + - apiGroups: ["batch"] + resources: ["jobs"] verbs: ["get", "list", "watch"] - # Metrics — read-only (requires metrics-server) + # Requires metrics-server. - apiGroups: ["metrics.k8s.io"] - resources: ["pods"] + resources: ["pods", "nodes"] verbs: ["get", "list"] --- apiVersion: rbac.authorization.k8s.io/v1 diff --git a/deploy/helm/maintenant/values.yaml b/deploy/helm/maintenant/values.yaml index 75675918..134b1c85 100644 --- a/deploy/helm/maintenant/values.yaml +++ b/deploy/helm/maintenant/values.yaml @@ -10,7 +10,7 @@ imagePullSecrets: [] ## Runtime: "kubernetes" or "docker" runtime: kubernetes -## License key (Pro edition) +## License key (Personal or Pro edition) license: key: "" ## Use an existing secret instead of the inline key above. @@ -60,8 +60,8 @@ resources: ## Extra environment variables injected into the container extraEnv: [] -# - name: MAINTENANT_WEBHOOK_URL -# value: "https://hooks.example.com/..." +# - name: MAINTENANT_BASE_URL +# value: "https://maintenant.example.com" ## External PostgreSQL for the server data set (optional). ## diff --git a/deploy/install/README.md b/deploy/install/README.md index b70148de..d08ff0e7 100644 --- a/deploy/install/README.md +++ b/deploy/install/README.md @@ -1,16 +1,18 @@ -# deploy/install — Maintenant native installer +# deploy/install: Maintenant native installer -This directory contains the self-contained install script, its systemd unit template, and bats test suite. +This directory contains the self-contained install script, its systemd unit template, and bats test suite. User documentation: [Native Linux Install](../../docs/install.md). ## Files | File | Description | |---|---| -| `install.sh` | POSIX install script (served at `https://install.maintenant.dev`) | -| `maintenant.service` | Reference systemd unit (also embedded as heredoc in `install.sh`) | +| `install.sh` | POSIX install script (served at `https://install.maintenant.dev`, and attached to every release as `install.sh`) | +| `maintenant.service` | Reference systemd unit (also embedded as heredoc in `install.sh`; a test checks that the script writes exactly this unit with its default paths) | | `test/setup.bash` | Bats helper: mock commands used by install.sh | | `test/install_basic.bats` | Tests for platform detection, checksum verification, basic install flow | | `test/install_args.bats` | Tests for flag parsing, flag→env mapping, env file merge | +| `test/offline.bats` | Tests for `--binary` and `--sha256sums` (install with no network access) | +| `test/service.bats` | Tests for the service restart, the atomic binary replacement, the summary and the unit written for custom paths | | `test/uninstall.bats` | Tests for uninstall / purge | | `test/pinning.bats` | Tests for version pinning and error messages | @@ -27,26 +29,33 @@ bats deploy/install/test/ bats deploy/install/test/install_basic.bats ``` -## Dev mode (test without touching the system) +The CI also runs ShellCheck on the script (`.github/workflows/shellcheck.yml`), and `go test ./internal/app/...` checks the script against the flag registry in `internal/app/flags.go`: `BOOL_FLAGS`, `VALUE_FLAGS`, the `--help` text and the flag→env mapping must list exactly the registry's configuration flags. Adding a flag to the binary therefore means adding it to the script too. + +## Dev mode + +The script always needs root, and it creates the `maintenant` system user (and adds it to the `docker` group when that group exists). Moving the paths keeps the files away from `/usr/local/bin`, `/etc` and `/var/lib`, but not the user: run it in a throwaway VM or container. ```sh -MAINTENANT_INSTALL_DIR=/tmp/maintenant-test \ -MAINTENANT_DATA_DIR=/tmp/maintenant-data-test \ -MAINTENANT_CONFIG_DIR=/tmp/maintenant-config-test \ -bash deploy/install/install.sh --no-service --skip-cosign +mkdir -p /tmp/maintenant-test +sudo MAINTENANT_INSTALL_DIR=/tmp/maintenant-test \ + MAINTENANT_DATA_DIR=/tmp/maintenant-data-test \ + MAINTENANT_CONFIG_DIR=/tmp/maintenant-config-test \ + bash deploy/install/install.sh --no-service --skip-cosign ``` -## Pinning de version +The install directory must exist before the run: the script writes the binary next to its destination and renames it, and does not create that directory. -```sh -# Install a specific version -MAINTENANT_VERSION=v1.6.0 curl -fsSL https://install.maintenant.dev | sudo -E bash +## Version pinning -# The -E flag preserves the MAINTENANT_VERSION env var through sudo +```sh +# Install a specific version: set the variable on the bash side of the pipe +curl -fsSL https://install.maintenant.dev | sudo MAINTENANT_VERSION=v1.8.0 bash ``` +`MAINTENANT_VERSION=v1.8.0 curl … | sudo bash` does not pin anything: the assignment only reaches `curl`. + Note: version pinning only works for releases ≥ v1.6.0 (the first release containing standalone binaries). ## Script versioning -The published script includes a `SCRIPT_VERSION=__GIT_SHA__` placeholder that the CI sync workflow replaces with the git short SHA. This allows bug reports to identify exactly which script version was executed. +The script carries a `SCRIPT_VERSION=__GIT_SHA__` placeholder. The release workflow (`.github/workflows/release.yml`) replaces it with the release tag when it renders the `install.sh` asset, and the script stamps that value into the header of `/etc/maintenant/maintenant.env`. This allows bug reports to identify exactly which script version was executed. diff --git a/deploy/install/install.sh b/deploy/install/install.sh index 844d49fb..c41662c3 100755 --- a/deploy/install/install.sh +++ b/deploy/install/install.sh @@ -18,6 +18,21 @@ SERVICE_FILE="${SERVICE_FILE:-/etc/systemd/system/maintenant.service}" GITHUB_REPO="kOlapsis/maintenant" GITHUB_API="https://api.github.com" SCRIPT_VERSION="__GIT_SHA__" +NL=' +' + +# Kept in step with internal/app/flags.go by internal/app/install_script_test.go. +BOOL_FLAGS="proxyLabels disableOsEolRefresh disableTelemetry allowPrivateWebhooks \ +mcp mcpAllowUnauthenticated grpc-tls-insecure grpc-insecure-skip-tls-verify embedded-agent" +VALUE_FLAGS="addr baseUrl corsOrigins trustedProxies db organisationName runtime logLevel \ +maxBodySize updateInterval securityScoreThreshold licenseKey \ +smtpHost smtpPort smtpUsername smtpPassword smtpFrom \ +mcpClientId mcpClientSecret mcpAllowedRedirectUris k8sNamespaces k8sExcludeNamespaces \ +statusUrl containerDownAfter retentionSnapshots retentionInterval retentionBatchSize \ +mode server enrollment-token label nodeName grpc-listen grpc-url grpc-tls-cert grpc-tls-key \ +agentRateLimitPerSecond agentStaleThresholdSeconds \ +agentSpoolMaxMemoryBytes agentSpoolMaxDiskBytes agentSpoolMaxAgeSeconds \ +data-dir ca-cert database-url" # ── Color / output ──────────────────────────────────────────────────────────── @@ -38,11 +53,17 @@ abort() { exit "${2:-1}" } +_fs() { + "$@" || abort "Filesystem operation failed: $*" 30 +} + # ── Cleanup trap ────────────────────────────────────────────────────────────── TMPDIR_INSTALL="${TMPDIR_INSTALL:-}" +BINARY_TMP="" cleanup() { if [ -n "${TMPDIR_INSTALL:-}" ]; then rm -rf "$TMPDIR_INSTALL"; fi + if [ -n "${BINARY_TMP:-}" ]; then rm -f "$BINARY_TMP"; fi } trap cleanup EXIT @@ -57,18 +78,31 @@ Script flags: --uninstall Remove Maintenant (keeps data and user by default) --purge With --uninstall: also remove data dir, config, user --skip-cosign Skip cosign signature check (SHA256 still required) + --binary Install this local binary instead of downloading one + (offline install: no network access at all) + --sha256sums With --binary: check it against this SHA256SUMS file --help, -h Show this help -Binary configuration flags (written to /etc/maintenant/maintenant.env): +Binary configuration flags (written to /etc/maintenant/maintenant.env). +Value flags take "--flag value" or "--flag=value". Boolean flags take "--flag" +or "--flag=true|false". Run "maintenant --help" for what each flag does. --addr --baseUrl + --corsOrigins + --trustedProxies --db + --containerDownAfter + --retentionSnapshots + --retentionInterval + --retentionBatchSize --organisationName - --corsOrigins - --runtime + --statusUrl + --runtime + --proxyLabels --logLevel --maxBodySize --updateInterval + --disableOsEolRefresh --securityScoreThreshold --disableTelemetry --allowPrivateWebhooks @@ -85,14 +119,11 @@ Binary configuration flags (written to /etc/maintenant/maintenant.env): --mcpAllowUnauthenticated --k8sNamespaces --k8sExcludeNamespaces - --statusUrl - --retentionSnapshots - --retentionInterval - --retentionBatchSize --mode --server --enrollment-token --label + --nodeName --grpc-listen --grpc-url --grpc-tls-cert @@ -101,9 +132,12 @@ Binary configuration flags (written to /etc/maintenant/maintenant.env): --grpc-insecure-skip-tls-verify --agentRateLimitPerSecond --agentStaleThresholdSeconds - --data-dir + --agentSpoolMaxMemoryBytes + --agentSpoolMaxDiskBytes + --agentSpoolMaxAgeSeconds --embedded-agent --ca-cert + --data-dir --database-url Examples: @@ -114,8 +148,11 @@ Examples: install.sh --mode agent --server grpcs://maintenant.example.com:8443 \ --enrollment-token TOKEN --label web-01 + # Offline, from a binary and its SHA256SUMS copied onto this host + install.sh --binary ./maintenant-v1.2.3-linux-amd64 --sha256sums ./SHA256SUMS + Environment variables: - MAINTENANT_VERSION Version to install (default: latest) + MAINTENANT_VERSION Version to download (default: latest) MAINTENANT_INSTALL_DIR Binary install path (default: /usr/local/bin) MAINTENANT_DATA_DIR Data directory (default: /var/lib/maintenant) MAINTENANT_CONFIG_DIR Config directory (default: /etc/maintenant) @@ -142,9 +179,8 @@ detect_platform() { check_prereqs() { [ "$(id -u)" -eq 0 ] || abort "This script must be run as root (EUID 0)" 11 - # curl or wget - if ! command -v curl >/dev/null 2>&1 && ! command -v wget >/dev/null 2>&1; then - abort "curl or wget is required" 12 + if [ -z "${LOCAL_BINARY:-}" ] && ! command -v curl >/dev/null 2>&1 && ! command -v wget >/dev/null 2>&1; then + abort "curl or wget is required (or install a local binary with --binary)" 12 fi # No tar: the release assets are bare binaries, not archives. @@ -250,8 +286,10 @@ download_and_verify() { BASE_URL="https://github.com/$GITHUB_REPO/releases/download/${VERSION}" log_step "Downloading $ASSET_NAME..." - fetch_url_to "$BASE_URL/$ASSET_NAME" "$TMPDIR_INSTALL/$ASSET_NAME" - fetch_url_to "$BASE_URL/SHA256SUMS" "$TMPDIR_INSTALL/SHA256SUMS" + fetch_url_to "$BASE_URL/$ASSET_NAME" "$TMPDIR_INSTALL/$ASSET_NAME" \ + || abort "Failed to download $ASSET_NAME" 20 + fetch_url_to "$BASE_URL/SHA256SUMS" "$TMPDIR_INSTALL/SHA256SUMS" \ + || abort "Failed to download SHA256SUMS" 20 log_step "Verifying SHA256 checksum..." (cd "$TMPDIR_INSTALL" && sha256sum -c SHA256SUMS --ignore-missing) \ @@ -267,7 +305,8 @@ download_and_verify() { # the signature either. log_warn "cosign 3 or later is required to read the release bundle — skipping signature verification" else - fetch_url_to "$BASE_URL/SHA256SUMS.bundle" "$TMPDIR_INSTALL/SHA256SUMS.bundle" + fetch_url_to "$BASE_URL/SHA256SUMS.bundle" "$TMPDIR_INSTALL/SHA256SUMS.bundle" \ + || abort "Failed to download SHA256SUMS.bundle" 20 log_step "Verifying cosign signature..." if ! cosign verify-blob \ --bundle "$TMPDIR_INSTALL/SHA256SUMS.bundle" \ @@ -278,6 +317,42 @@ download_and_verify() { fi log_info "cosign signature verified" fi + + BINARY_SRC="$TMPDIR_INSTALL/$ASSET_NAME" +} + +# ── use_local_binary ────────────────────────────────────────────────────────── + +use_local_binary() { + [ -f "$LOCAL_BINARY" ] || abort "Binary not found: $LOCAL_BINARY" 2 + TMPDIR_INSTALL=$(mktemp -d) + BINARY_SRC="$LOCAL_BINARY" + VERSION="local" + + if [ -z "${LOCAL_SUMS:-}" ]; then + log_warn "No --sha256sums given: the integrity of $LOCAL_BINARY is not checked" + return + fi + [ -f "$LOCAL_SUMS" ] || abort "SHA256SUMS file not found: $LOCAL_SUMS" 2 + + log_step "Verifying SHA256 checksum against $LOCAL_SUMS..." + LOCAL_SUM=$(sha256sum "$LOCAL_BINARY" | cut -d ' ' -f 1) + MATCHED_ASSET=$(awk -v sum="$LOCAL_SUM" -v suffix="-linux-$ARCH" ' + $1 == sum { + name = $2 + sub(/^\*/, "", name) + if (name ~ /^maintenant-/ && substr(name, length(name) - length(suffix) + 1) == suffix) { + print name + exit + } + }' "$LOCAL_SUMS") + [ -n "$MATCHED_ASSET" ] \ + || abort "SHA256 of $LOCAL_BINARY matches no linux-$ARCH binary listed in $LOCAL_SUMS" 21 + + VERSION="${MATCHED_ASSET#maintenant-}" + VERSION="${VERSION%-linux-"$ARCH"}" + log_info "Checksum matches $MATCHED_ASSET" + log_warn "Offline install: the cosign signature of $LOCAL_SUMS is not verified" } # ── ensure_user ─────────────────────────────────────────────────────────────── @@ -288,7 +363,8 @@ ensure_user() { else log_step "Creating system user $SERVICE_USER..." useradd -r -s /usr/sbin/nologin -d "$DATA_DIR" \ - -c "Maintenant service user" "$SERVICE_USER" + -c "Maintenant service user" "$SERVICE_USER" \ + || abort "Failed to create user $SERVICE_USER" log_info "User $SERVICE_USER created" fi @@ -302,31 +378,126 @@ ensure_user() { fi } +# ── resolve_paths ───────────────────────────────────────────────────────────── +# A flag given now wins over the env file, which wins over the defaults. + +_env_file_value() { + [ -f "$2" ] || return 0 + awk -v key="$1" -v q="'" ' + { + line = $0 + sub(/^[ \t]+/, "", line) + if (!match(line, /^[A-Za-z_][A-Za-z0-9_]*[ \t]*=/)) next + k = substr(line, 1, RLENGTH - 1) + sub(/[ \t]+$/, "", k) + if (k != key) next + v = substr(line, RLENGTH + 1) + sub(/^[ \t]+/, "", v) + sub(/[ \t]+$/, "", v) + if (length(v) >= 2 && (v ~ /^".*"$/ || v ~ ("^" q ".*" q "$"))) v = substr(v, 2, length(v) - 2) + found = v + } + END { printf "%s", found } + ' "$2" +} + +_strip_trailing_slashes() { + p="$1" + while [ "$p" != "/" ]; do + case "$p" in + */) p="${p%/}" ;; + *) break ;; + esac + done + printf '%s' "$p" +} + +resolve_paths() { + ENV_FILE="$CONFIG_DIR/maintenant.env" + + if _has_binary_flag data-dir; then + DATA_DIR=$(_binary_flag_value data-dir) + else + from_file=$(_env_file_value MAINTENANT_DATA_DIR "$ENV_FILE") + [ -z "$from_file" ] || DATA_DIR="$from_file" + fi + DATA_DIR=$(_strip_trailing_slashes "$DATA_DIR") + case "$DATA_DIR" in + /?*) ;; + *) abort "The data directory must be an absolute path other than /: $DATA_DIR" 2 ;; + esac + + if _has_binary_flag db; then + DB_PATH=$(_binary_flag_value db) + else + DB_PATH=$(_env_file_value MAINTENANT_DB "$ENV_FILE") + fi + [ -n "$DB_PATH" ] || DB_PATH="./maintenant.db" + while :; do + case "$DB_PATH" in + ./*) DB_PATH="${DB_PATH#./}" ;; + *) break ;; + esac + done + case "$DB_PATH" in + /*) ;; + *) DB_PATH="$DATA_DIR/$DB_PATH" ;; + esac + DB_DIR=$(dirname "$DB_PATH") + + for unit_path in "$INSTALL_DIR" "$CONFIG_DIR" "$DATA_DIR" "$DB_DIR"; do + case "$unit_path" in + *[!A-Za-z0-9._/@+-]*) + abort "Unsupported character in path (letters, digits and ._/@+- only): $unit_path" 2 ;; + esac + done +} + +# ── prepare_dirs ────────────────────────────────────────────────────────────── + +_refuse_foreign_dir() { + [ -d "$1" ] || return 0 + [ -z "$(find "$1" -prune -user "$SERVICE_USER" 2>/dev/null)" ] || return 0 + [ -z "$(ls -A "$1")" ] && return 0 + abort "$1 already exists, is not empty and does not belong to $SERVICE_USER: use a dedicated directory, or chown it to $SERVICE_USER first" 30 +} + +_own_dir() { + _fs mkdir -p "$1" + _fs chown "$SERVICE_USER:$SERVICE_USER" "$1" + _fs chmod 0750 "$1" +} + +prepare_dirs() { + _refuse_foreign_dir "$DATA_DIR" + _refuse_foreign_dir "$DB_DIR" + + _fs mkdir -p "$CONFIG_DIR" + _fs chown "root:$SERVICE_USER" "$CONFIG_DIR" + _fs chmod 0750 "$CONFIG_DIR" + + _own_dir "$DATA_DIR" + [ "$DB_DIR" = "$DATA_DIR" ] || _own_dir "$DB_DIR" +} + # ── install_binary ──────────────────────────────────────────────────────────── install_binary() { log_step "Installing binary..." - mkdir -p "$DATA_DIR" - chown "$SERVICE_USER:$SERVICE_USER" "$DATA_DIR" - chmod 0750 "$DATA_DIR" - - mkdir -p "$CONFIG_DIR" - chown "root:$SERVICE_USER" "$CONFIG_DIR" - chmod 0750 "$CONFIG_DIR" - - install -m 0755 -o root -g root \ - "$TMPDIR_INSTALL/maintenant-${VERSION}-linux-${ARCH}" \ - "$INSTALL_DIR/maintenant" + BINARY_TMP=$(mktemp "$INSTALL_DIR/.maintenant.XXXXXX") \ + || abort "Cannot create a temporary file in $INSTALL_DIR" 30 + _fs install -m 0755 -o root -g root "$BINARY_SRC" "$BINARY_TMP" + _fs mv -f "$BINARY_TMP" "$INSTALL_DIR/maintenant" + BINARY_TMP="" log_info "Binary installed to $INSTALL_DIR/maintenant" } # ── install_service ─────────────────────────────────────────────────────────── -install_service() { - [ -z "${NO_SERVICE:-}" ] || { log_info "Skipping service installation (--no-service)"; return; } - - log_step "Installing systemd service..." - cat > "$SERVICE_FILE" <<'UNIT' +render_unit() { + rw_paths="$DATA_DIR" + [ "$DB_DIR" = "$DATA_DIR" ] || rw_paths="$rw_paths $DB_DIR" + cat </dev/null 2>&1 && systemctl is-active --quiet maintenant 2>/dev/null; then + log_warn "maintenant.service still runs the previous binary: systemctl restart maintenant" + fi + return + fi + + log_step "Installing systemd service..." + render_unit > "$SERVICE_FILE" || abort "Failed to write $SERVICE_FILE" 30 + + systemctl daemon-reload || abort "systemctl daemon-reload failed" 31 + systemctl enable maintenant || abort "systemctl enable maintenant failed" 31 + if systemctl is-active --quiet maintenant; then + log_step "Restarting the service on the new binary..." + systemctl restart maintenant || abort "systemctl restart maintenant failed" 31 + else + log_step "Starting the service..." + systemctl start maintenant || abort "systemctl start maintenant failed" 31 + fi log_step "Waiting for service to become active..." i=0 @@ -381,7 +573,12 @@ UNIT # ── print_summary ───────────────────────────────────────────────────────────── print_summary() { - LISTEN_ADDR="${MAINTENANT_ADDR:-127.0.0.1:8080}" + if _has_binary_flag addr; then + LISTEN_ADDR=$(_binary_flag_value addr) + else + LISTEN_ADDR=$(_env_file_value MAINTENANT_ADDR "$CONFIG_DIR/maintenant.env") + fi + [ -n "$LISTEN_ADDR" ] || LISTEN_ADDR="127.0.0.1:8080" cat < "$TMPDIR_INSTALL/new_flags.env" if [ ! -f "$ENV_FILE" ]; then - # First creation { printf '%s\n\n' "$HEADER" cat "$TMPDIR_INSTALL/new_flags.env" - } > "$ENV_FILE" - chown "root:$SERVICE_USER" "$ENV_FILE" - chmod 0640 "$ENV_FILE" + } > "$TMPDIR_INSTALL/merged.env" || abort "Cannot write $TMPDIR_INSTALL/merged.env" 30 + _install_env_file log_info "Created $ENV_FILE" return fi - # Merge: read existing, override with new - KEYS_COUNT=0 - UPDATES=0 - - # Read existing non-comment lines - grep -E '^MAINTENANT_[A-Z_]+=.*$' "$ENV_FILE" > "$TMPDIR_INSTALL/existing.env" 2>/dev/null || true - - # Build merged file { printf '%s\n\n' "$HEADER" + awk -v counts="$TMPDIR_INSTALL/merge.counts" ' + BEGIN { head = 1 } + FILENAME == ARGV[1] { + i = index($0, "=") + if (i > 1) { + k = substr($0, 1, i - 1) + if (!(k in val)) order[++n] = k + val[k] = substr($0, i + 1) + } + next + } + head && /^# (Generated by install\.sh$|Last updated: |Script version: |Edit this file directly then: )/ { generated = 1; next } + head && generated && /^[ \t]*$/ { head = 0; next } + { head = 0 } + { + line = $0 + sub(/^[ \t]+/, "", line) + if (match(line, /^[A-Za-z_][A-Za-z0-9_]*[ \t]*=/)) { + k = substr(line, 1, RLENGTH - 1) + sub(/[ \t]+$/, "", k) + if (k in val) { + print k "=" val[k] + written[k] = 1 + updated++ + next + } + } + print + } + END { + for (j = 1; j <= n; j++) { + if (!(order[j] in written)) { + print order[j] "=" val[order[j]] + added++ + } + } + print (updated + 0) " " (added + 0) > counts + } + ' "$TMPDIR_INSTALL/new_flags.env" "$ENV_FILE" + } > "$TMPDIR_INSTALL/merged.env" || abort "Cannot merge into $ENV_FILE" 30 + + read -r UPDATED ADDED < "$TMPDIR_INSTALL/merge.counts" + _install_env_file + log_info "$ENV_FILE updated (${UPDATED} keys replaced, ${ADDED} keys added, other lines kept)" +} - # Start with existing entries, override if in new_flags - while IFS='=' read -r key rest; do - [ -n "$key" ] || continue - val="$rest" - # Check if this key appears in new flags - new_val=$(grep "^${key}=" "$TMPDIR_INSTALL/new_flags.env" | cut -d= -f2-) - if [ -n "$new_val" ]; then - printf '%s=%s\n' "$key" "$new_val" - UPDATES=$((UPDATES + 1)) - else - printf '%s=%s\n' "$key" "$val" - fi - KEYS_COUNT=$((KEYS_COUNT + 1)) - done < "$TMPDIR_INSTALL/existing.env" - - # Add new keys not already in existing - while IFS='=' read -r key rest; do - [ -n "$key" ] || continue - if ! grep -q "^${key}=" "$TMPDIR_INSTALL/existing.env" 2>/dev/null; then - printf '%s=%s\n' "$key" "$rest" - KEYS_COUNT=$((KEYS_COUNT + 1)) - fi - done < "$TMPDIR_INSTALL/new_flags.env" - } > "$TMPDIR_INSTALL/merged.env" - - mv "$TMPDIR_INSTALL/merged.env" "$ENV_FILE" - chown "root:$SERVICE_USER" "$ENV_FILE" - chmod 0640 "$ENV_FILE" - log_info "$ENV_FILE updated (${KEYS_COUNT} keys preserved, ${UPDATES} keys updated)" +_install_env_file() { + _fs chown "root:$SERVICE_USER" "$TMPDIR_INSTALL/merged.env" + _fs chmod 0640 "$TMPDIR_INSTALL/merged.env" + _fs mv -f "$TMPDIR_INSTALL/merged.env" "$ENV_FILE" } # ── uninstall ───────────────────────────────────────────────────────────────── uninstall() { log_step "Uninstalling Maintenant..." + resolve_paths # Stop and disable service systemctl stop maintenant 2>/dev/null || true @@ -623,6 +868,10 @@ _purge() { rm -rf "$DATA_DIR" log_info "Removed $DATA_DIR" fi + case "$DB_DIR/" in + "$DATA_DIR"/*) ;; + *) [ ! -d "$DB_DIR" ] || log_warn "Database directory kept, it lies outside $DATA_DIR: $DB_DIR" ;; + esac if [ -d "$CONFIG_DIR" ]; then rm -rf "$CONFIG_DIR" log_info "Removed $CONFIG_DIR" @@ -670,10 +919,15 @@ main() { log_step "Starting Maintenant installation" detect_platform check_prereqs - resolve_version - download_and_verify + resolve_paths + if [ -n "$LOCAL_BINARY" ]; then + use_local_binary + else + resolve_version + download_and_verify + fi ensure_user - install_binary + prepare_dirs # Apply binary flags to env file if any were provided if [ -n "$BINARY_FLAG_KEYS" ]; then @@ -681,6 +935,7 @@ main() { merge_env_file fi + install_binary install_service print_summary } diff --git a/deploy/install/maintenant.service b/deploy/install/maintenant.service index dca0d3bd..dfbe5063 100644 --- a/deploy/install/maintenant.service +++ b/deploy/install/maintenant.service @@ -1,5 +1,5 @@ [Unit] -Description=Maintenant monitoring +Description=Maintenant infrastructure monitoring Documentation=https://docs.maintenant.dev After=network-online.target Wants=network-online.target @@ -8,14 +8,13 @@ Wants=network-online.target Type=simple User=maintenant Group=maintenant +Environment=MAINTENANT_DATA_DIR=/var/lib/maintenant EnvironmentFile=-/etc/maintenant/maintenant.env ExecStart=/usr/local/bin/maintenant WorkingDirectory=/var/lib/maintenant Restart=on-failure RestartSec=5s LimitNOFILE=65536 - -# Hardening NoNewPrivileges=true PrivateTmp=true ProtectSystem=strict diff --git a/deploy/install/test/install_args.bats b/deploy/install/test/install_args.bats index 23acc5b0..fe713106 100644 --- a/deploy/install/test/install_args.bats +++ b/deploy/install/test/install_args.bats @@ -225,3 +225,181 @@ SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/install.sh" [ "$PERMS" = "640" ] rm -rf "$FAKE_TMPDIR" "$FAKE_CONFIG_DIR" } + +@test "merge_env_file: keeps every line it was not asked to change" { + FAKE_TMPDIR=$(mktemp -d) + FAKE_CONFIG_DIR=$(mktemp -d) + cat > "$FAKE_CONFIG_DIR/maintenant.env" <<'ENV' +# Generated by install.sh +# Last updated: 2020-01-01T00:00:00Z +# Script version: abc1234 +# Edit this file directly then: systemctl restart maintenant + +MAINTENANT_ADDR=127.0.0.1:8080 +# socket proxy in front of the Docker API +DOCKER_HOST=tcp://socket-proxy:2375 +MAINTENANT_K8S_NAMESPACES=prod,staging +ENV + + run bash -c " + NO_COLOR=1 + TMPDIR_INSTALL='$FAKE_TMPDIR' + CONFIG_DIR='$FAKE_CONFIG_DIR' + SERVICE_USER=nobody + chown() { return 0; } + export -f chown + export NO_COLOR TMPDIR_INSTALL CONFIG_DIR SERVICE_USER + _INSTALL_SH_TESTING=1 . '$SCRIPT' + parse_maintenant_flags --addr 0.0.0.0:9000 --logLevel debug + merge_env_file + " + [ "$status" -eq 0 ] + ENV_FILE="$FAKE_CONFIG_DIR/maintenant.env" + grep -qx 'MAINTENANT_ADDR=0.0.0.0:9000' "$ENV_FILE" + grep -qx 'MAINTENANT_LOG_LEVEL=debug' "$ENV_FILE" + grep -qx 'DOCKER_HOST=tcp://socket-proxy:2375' "$ENV_FILE" + grep -qx 'MAINTENANT_K8S_NAMESPACES=prod,staging' "$ENV_FILE" + grep -qx '# socket proxy in front of the Docker API' "$ENV_FILE" + [ "$(grep -c '^MAINTENANT_ADDR=' "$ENV_FILE")" -eq 1 ] + [ "$(grep -c '^# Generated by install.sh$' "$ENV_FILE")" -eq 1 ] + [ -z "$(grep '2020-01-01' "$ENV_FILE")" ] + rm -rf "$FAKE_TMPDIR" "$FAKE_CONFIG_DIR" +} + +@test "merge_env_file: a flag given twice keeps its last value" { + FAKE_TMPDIR=$(mktemp -d) + FAKE_CONFIG_DIR=$(mktemp -d) + printf 'MAINTENANT_ADDR=127.0.0.1:8080\n' > "$FAKE_CONFIG_DIR/maintenant.env" + + run bash -c " + NO_COLOR=1 + TMPDIR_INSTALL='$FAKE_TMPDIR' + CONFIG_DIR='$FAKE_CONFIG_DIR' + SERVICE_USER=nobody + chown() { return 0; } + export -f chown + export NO_COLOR TMPDIR_INSTALL CONFIG_DIR SERVICE_USER + _INSTALL_SH_TESTING=1 . '$SCRIPT' + parse_maintenant_flags --addr 0.0.0.0:1 --addr 0.0.0.0:2 + merge_env_file + " + [ "$status" -eq 0 ] + [ "$(grep '^MAINTENANT_ADDR=' "$FAKE_CONFIG_DIR/maintenant.env")" = "MAINTENANT_ADDR=0.0.0.0:2" ] + rm -rf "$FAKE_TMPDIR" "$FAKE_CONFIG_DIR" +} + +# ── boolean and value flags ─────────────────────────────────────────────────── + +parse_and_print() { + bash -c " + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + parse_maintenant_flags $* + printf '%s' \"\$BINARY_FLAG_KEYS\" | while IFS= read -r k; do + printf '%s=%s\n' \"\$k\" \"\$(_binary_flag_value \"\$k\")\" + done + " +} + +@test "parse_maintenant_flags: a boolean flag does not swallow the next flag" { + run parse_and_print --disableOsEolRefresh --addr 0.0.0.0:8080 + [ "$status" -eq 0 ] + [[ "$output" == *"disableOsEolRefresh=true"* ]] + [[ "$output" == *"addr=0.0.0.0:8080"* ]] +} + +@test "parse_maintenant_flags: a boolean flag may come last" { + run parse_and_print --addr 0.0.0.0:8080 --proxyLabels + [ "$status" -eq 0 ] + [[ "$output" == *"proxyLabels=true"* ]] +} + +@test "parse_maintenant_flags: every boolean flag of the binary works bare" { + run bash -c " + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + for f in \$BOOL_FLAGS; do + parse_maintenant_flags --\$f --addr 0.0.0.0:8080 || exit 1 + [ \"\$(_binary_flag_value \$f)\" = true ] || { echo \"--\$f\"; exit 1; } + [ \"\$(_binary_flag_value addr)\" = 0.0.0.0:8080 ] || { echo \"--\$f\"; exit 1; } + done + " + [ "$status" -eq 0 ] +} + +@test "parse_maintenant_flags: a boolean flag takes =false" { + run parse_and_print --disableTelemetry=false --mcp=TRUE + [ "$status" -eq 0 ] + [[ "$output" == *"disableTelemetry=false"* ]] + [[ "$output" == *"mcp=true"* ]] +} + +@test "parse_maintenant_flags: a boolean flag refuses a value that is not a boolean" { + run parse_and_print --mcp=maybe + [ "$status" -eq 2 ] +} + +@test "parse_maintenant_flags: a value flag takes --flag=value" { + run parse_and_print --addr=0.0.0.0:9000 --smtpPassword=--starts-with-dashes + [ "$status" -eq 0 ] + [[ "$output" == *"addr=0.0.0.0:9000"* ]] + [[ "$output" == *"smtpPassword=--starts-with-dashes"* ]] +} + +@test "parse_maintenant_flags: a value flag refuses the next flag as its value" { + run parse_and_print --addr --mcp + [ "$status" -eq 2 ] +} + +@test "parse_maintenant_flags: a flag the binary does not know is refused" { + run parse_and_print --adr 0.0.0.0:8080 + [ "$status" -eq 2 ] + [[ "$output" == *"Unknown argument: --adr"* ]] +} + +@test "parse_maintenant_flags: the binary's action flags are refused" { + run parse_and_print --copy-store-to postgres://db/maintenant + [ "$status" -eq 2 ] + run parse_and_print --yes + [ "$status" -eq 2 ] +} + +@test "parse_maintenant_flags: a value on two lines is refused" { + run bash -c " + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + parse_maintenant_flags --label \"\$(printf 'web\nMAINTENANT_MODE=server')\" + " + [ "$status" -eq 2 ] +} + +@test "parse_maintenant_flags: flags the help used to omit are accepted" { + run parse_and_print --proxyLabels --disableOsEolRefresh --nodeName node-1 \ + --trustedProxies 10.0.0.0/8 --containerDownAfter 5m \ + --agentSpoolMaxMemoryBytes 1048576 --agentSpoolMaxDiskBytes 10485760 \ + --agentSpoolMaxAgeSeconds 3600 + [ "$status" -eq 0 ] + [[ "$output" == *"nodeName=node-1"* ]] + [[ "$output" == *"agentSpoolMaxAgeSeconds=3600"* ]] +} + +@test "--help lists the flags it used to omit" { + run bash "$SCRIPT" --help + [ "$status" -eq 0 ] + for f in proxyLabels disableOsEolRefresh nodeName trustedProxies containerDownAfter \ + agentSpoolMaxMemoryBytes agentSpoolMaxDiskBytes agentSpoolMaxAgeSeconds binary sha256sums; do + [[ "$output" == *"--$f"* ]] || { echo "missing --$f"; return 1; } + done +} + +@test "parse_maintenant_flags: --binary and --sha256sums are script flags" { + run bash -c " + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + parse_maintenant_flags --binary ./maintenant --sha256sums=./SHA256SUMS + echo \"LOCAL_BINARY=\$LOCAL_BINARY LOCAL_SUMS=\$LOCAL_SUMS KEYS=[\$BINARY_FLAG_KEYS]\" + " + [ "$status" -eq 0 ] + [[ "$output" == *"LOCAL_BINARY=./maintenant LOCAL_SUMS=./SHA256SUMS KEYS=[]"* ]] +} + +@test "parse_maintenant_flags: --sha256sums without --binary exits 2" { + run parse_and_print --sha256sums ./SHA256SUMS + [ "$status" -eq 2 ] +} diff --git a/deploy/install/test/install_basic.bats b/deploy/install/test/install_basic.bats index 68dbd958..4c189139 100644 --- a/deploy/install/test/install_basic.bats +++ b/deploy/install/test/install_basic.bats @@ -129,6 +129,20 @@ run_script() { rm -rf "$FAKE_TMPDIR" } +@test "download_and_verify: a failed download exits 20" { + run bash -c " + VERSION='v1.0.0' + ARCH='amd64' + SKIP_COSIGN=1 + NO_COLOR=1 + export VERSION ARCH SKIP_COSIGN NO_COLOR + _INSTALL_SH_TESTING=1 . '$SCRIPT' + fetch_url_to() { return 22; } + download_and_verify + " + [ "$status" -eq 20 ] +} + # ── full flow --no-service ──────────────────────────────────────────────────── @test "full install with --no-service succeeds" { diff --git a/deploy/install/test/offline.bats b/deploy/install/test/offline.bats new file mode 100644 index 00000000..2e50c1bc --- /dev/null +++ b/deploy/install/test/offline.bats @@ -0,0 +1,101 @@ +#!/usr/bin/env bats +# Tests for the offline install from a local binary (--binary, --sha256sums) +bats_require_minimum_version 1.5.0 + +load 'setup' + +SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/install.sh" + +setup() { + WORK=$(mktemp -d) + mkdir -p "$WORK/media" "$WORK/bin" "$WORK/data" "$WORK/etc" + printf 'maintenant binary' > "$WORK/media/maintenant-v1.2.3-linux-amd64" + (cd "$WORK/media" && sha256sum maintenant-v1.2.3-linux-amd64 > SHA256SUMS) +} + +teardown() { + rm -rf "$WORK" +} + +# Every network path fails and leaves a trace in $WORK/network.log. +run_offline() { + local svc + svc=$(id -un) + run bash -c " + NO_COLOR=1 + uname() { case \"\$1\" in -s) echo Linux;; -m) echo x86_64;; esac; } + id() { echo '0'; } + useradd() { return 0; } + getent() { return 1; } + chown() { return 0; } + install() { cp \"\${@: -2:1}\" \"\${@: -1}\"; } + curl() { echo \"curl \$*\" >> '$WORK/network.log'; return 7; } + wget() { echo \"wget \$*\" >> '$WORK/network.log'; return 4; } + export -f uname id useradd getent chown install curl wget + INSTALL_DIR='$WORK/bin' + DATA_DIR='$WORK/data' + CONFIG_DIR='$WORK/etc' + SERVICE_USER='$svc' + export NO_COLOR INSTALL_DIR DATA_DIR CONFIG_DIR + _INSTALL_SH_TESTING=1 . '$SCRIPT' + fetch_url() { echo \"fetch_url \$*\" >> '$WORK/network.log'; return 1; } + fetch_url_to() { echo \"fetch_url_to \$*\" >> '$WORK/network.log'; return 1; } + main --no-service $* + " +} + +@test "offline install: installs the local binary without any network access" { + run_offline --binary "$WORK/media/maintenant-v1.2.3-linux-amd64" --sha256sums "$WORK/media/SHA256SUMS" --addr 0.0.0.0:8080 + [ "$status" -eq 0 ] + [ ! -e "$WORK/network.log" ] + [ "$(cat "$WORK/bin/maintenant")" = "maintenant binary" ] + grep -qx 'MAINTENANT_ADDR=0.0.0.0:8080' "$WORK/etc/maintenant.env" + [[ "$output" == *"Version : v1.2.3"* ]] +} + +@test "offline install: a renamed binary is found in SHA256SUMS by its checksum" { + cp "$WORK/media/maintenant-v1.2.3-linux-amd64" "$WORK/media/maintenant" + run_offline --binary "$WORK/media/maintenant" --sha256sums "$WORK/media/SHA256SUMS" + [ "$status" -eq 0 ] + [[ "$output" == *"Version : v1.2.3"* ]] +} + +@test "offline install: a checksum mismatch exits 21" { + printf 'tampered' > "$WORK/media/maintenant-v1.2.3-linux-amd64" + run_offline --binary "$WORK/media/maintenant-v1.2.3-linux-amd64" --sha256sums "$WORK/media/SHA256SUMS" + [ "$status" -eq 21 ] + [ ! -e "$WORK/bin/maintenant" ] +} + +@test "offline install: a binary listed for another architecture exits 21" { + (cd "$WORK/media" && sha256sum maintenant-v1.2.3-linux-amd64 \ + | sed 's/linux-amd64/linux-arm64/' > SHA256SUMS) + run_offline --binary "$WORK/media/maintenant-v1.2.3-linux-amd64" --sha256sums "$WORK/media/SHA256SUMS" + [ "$status" -eq 21 ] +} + +@test "offline install: a missing binary exits 2" { + run_offline --binary "$WORK/media/nope" + [ "$status" -eq 2 ] +} + +@test "offline install: without --sha256sums it warns and installs" { + run_offline --binary "$WORK/media/maintenant-v1.2.3-linux-amd64" + [ "$status" -eq 0 ] + [[ "$output" == *"integrity of"*"is not checked"* ]] + [ ! -e "$WORK/network.log" ] +} + +@test "check_prereqs: an offline install needs neither curl nor wget" { + run bash -c " + id() { echo '0'; } + install() { :; } + useradd() { :; } + export -f id install useradd + PATH=/usr/bin/no-such-dir + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + parse_maintenant_flags --no-service --binary ./maintenant + check_prereqs + " + [ "$status" -eq 0 ] +} diff --git a/deploy/install/test/service.bats b/deploy/install/test/service.bats new file mode 100644 index 00000000..8d265749 --- /dev/null +++ b/deploy/install/test/service.bats @@ -0,0 +1,243 @@ +#!/usr/bin/env bats +# Tests for the binary swap, the systemd unit and the directories it may write +bats_require_minimum_version 1.5.0 + +load 'setup' + +SCRIPT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/install.sh" +UNIT="$(cd "$(dirname "$BATS_TEST_FILENAME")/.." && pwd)/maintenant.service" + +setup() { + WORK=$(mktemp -d) +} + +teardown() { + rm -rf "$WORK" +} + +run_install_service() { + run bash -c " + NO_COLOR=1 + SERVICE_FILE='$WORK/maintenant.service' + export NO_COLOR SERVICE_FILE + systemctl() { + echo \"\$*\" >> '$WORK/systemctl.log' + case \"\$*\" in + start*|restart*|'enable --now'*) touch '$WORK/active' ;; + is-active*) [ -e '$WORK/active' ] ;; + esac + } + sleep() { :; } + journalctl() { :; } + export -f systemctl sleep journalctl + [ -z '${ALREADY_ACTIVE:-}' ] || touch '$WORK/active' + _INSTALL_SH_TESTING=1 . '$SCRIPT' + DATA_DIR=/var/lib/maintenant + DB_DIR=/var/lib/maintenant + install_service + " +} + +@test "install_service: restarts a service that is already running" { + ALREADY_ACTIVE=1 run_install_service + [ "$status" -eq 0 ] + grep -qx 'restart maintenant' "$WORK/systemctl.log" +} + +@test "install_service: starts a service that is not running" { + run_install_service + [ "$status" -eq 0 ] + grep -qx 'start maintenant' "$WORK/systemctl.log" + [ -z "$(grep '^restart' "$WORK/systemctl.log")" ] +} + +@test "install_service: --no-service warns when the running service keeps the old binary" { + NO_SERVICE=1 ALREADY_ACTIVE=1 run_install_service + [ "$status" -eq 0 ] + [[ "$output" == *"still runs the previous binary"* ]] + [ -z "$(grep -E '^(start|restart|enable)' "$WORK/systemctl.log")" ] +} + +@test "install_service: a failing restart exits 31" { + run bash -c " + NO_COLOR=1 + SERVICE_FILE='$WORK/maintenant.service' + export NO_COLOR SERVICE_FILE + systemctl() { + case \"\$*\" in + restart*) return 1 ;; + esac + return 0 + } + export -f systemctl + _INSTALL_SH_TESTING=1 . '$SCRIPT' + DATA_DIR=/var/lib/maintenant + DB_DIR=/var/lib/maintenant + install_service + " + [ "$status" -eq 31 ] +} + +summary_for() { + run bash -c " + MAINTENANT_ADDR=10.9.9.9:1 + export MAINTENANT_ADDR + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + CONFIG_DIR='$WORK/etc' + VERSION=v1.0.0 + parse_maintenant_flags $* + print_summary + " +} + +@test "print_summary: shows the address given as a flag" { + mkdir -p "$WORK/etc" + printf 'MAINTENANT_ADDR=0.0.0.0:7000\n' > "$WORK/etc/maintenant.env" + summary_for --addr 0.0.0.0:9000 + [ "$status" -eq 0 ] + [[ "$output" == *"Listens : http://0.0.0.0:9000"$'\n'* ]] +} + +@test "print_summary: shows the address already in the env file" { + mkdir -p "$WORK/etc" + printf 'MAINTENANT_ADDR="0.0.0.0:7000"\n' > "$WORK/etc/maintenant.env" + summary_for --logLevel debug + [ "$status" -eq 0 ] + [[ "$output" == *"Listens : http://0.0.0.0:7000"$'\n'* ]] +} + +@test "print_summary: falls back to the binary's default address" { + summary_for + [ "$status" -eq 0 ] + [[ "$output" == *"Listens : http://127.0.0.1:8080"$'\n'* ]] +} + +@test "install_binary: writes a temporary file next to the binary, then renames it" { + mkdir -p "$WORK/bin" "$WORK/tmp" + printf 'old' > "$WORK/bin/maintenant" + printf 'new' > "$WORK/tmp/maintenant-v1.0.0-linux-amd64" + + run bash -c " + NO_COLOR=1 + export NO_COLOR + install() { + echo \"\${@: -1}\" >> '$WORK/install.log' + cp \"\${@: -2:1}\" \"\${@: -1}\" + } + export -f install + _INSTALL_SH_TESTING=1 . '$SCRIPT' + INSTALL_DIR='$WORK/bin' + TMPDIR_INSTALL='$WORK/tmp' + VERSION=v1.0.0 + ARCH=amd64 + BINARY_SRC='$WORK/tmp/maintenant-v1.0.0-linux-amd64' + install_binary + " + [ "$status" -eq 0 ] + [ "$(cat "$WORK/bin/maintenant")" = "new" ] + written=$(cat "$WORK/install.log") + [ "$(dirname "$written")" = "$WORK/bin" ] + [ "$written" != "$WORK/bin/maintenant" ] + [ "$(ls -A "$WORK/bin")" = "maintenant" ] +} + +@test "the reference unit is what the script writes with its defaults" { + run bash -c " + unset DATA_DIR MAINTENANT_DATA_DIR CONFIG_DIR MAINTENANT_CONFIG_DIR INSTALL_DIR MAINTENANT_INSTALL_DIR SERVICE_USER + _INSTALL_SH_TESTING=1 . '$SCRIPT' + _env_file_value() { :; } + resolve_paths + render_unit + " + [ "$status" -eq 0 ] + [ "$output" = "$(cat "$UNIT")" ] +} + +render_for() { + run bash -c " + unset DATA_DIR MAINTENANT_DATA_DIR + _INSTALL_SH_TESTING=1 NO_COLOR=1 . '$SCRIPT' + CONFIG_DIR='$WORK/etc' + parse_maintenant_flags $* + resolve_paths + render_unit + " +} + +@test "unit: --db outside the data dir is writable" { + render_for --db /srv/maintenant-db/maintenant.db + [ "$status" -eq 0 ] + [[ "$output" == *"ReadWritePaths=/var/lib/maintenant /srv/maintenant-db"$'\n'* ]] +} + +@test "unit: --data-dir moves the working directory, the writable path and MAINTENANT_DATA_DIR" { + render_for --data-dir /srv/maintenant/ + [ "$status" -eq 0 ] + [[ "$output" == *"WorkingDirectory=/srv/maintenant"$'\n'* ]] + [[ "$output" == *"ReadWritePaths=/srv/maintenant"$'\n'* ]] + [[ "$output" == *"Environment=MAINTENANT_DATA_DIR=/srv/maintenant"$'\n'* ]] +} + +@test "unit: a relative --db lives in the data dir" { + render_for --data-dir /srv/maintenant --db ./db/maintenant.db + [ "$status" -eq 0 ] + [[ "$output" == *"ReadWritePaths=/srv/maintenant /srv/maintenant/db"$'\n'* ]] +} + +@test "unit: paths already in the env file are kept on a rerun without flags" { + mkdir -p "$WORK/etc" + printf 'MAINTENANT_DATA_DIR="/srv/agent"\nMAINTENANT_DB=/srv/db/maintenant.db\n' > "$WORK/etc/maintenant.env" + render_for --logLevel debug + [ "$status" -eq 0 ] + [[ "$output" == *"WorkingDirectory=/srv/agent"$'\n'* ]] + [[ "$output" == *"ReadWritePaths=/srv/agent /srv/db"$'\n'* ]] +} + +@test "resolve_paths: a relative data dir exits 2" { + render_for --data-dir data + [ "$status" -eq 2 ] +} + +@test "resolve_paths: a path the unit file cannot carry exits 2" { + render_for --db '"/srv/my db/maintenant.db"' + [ "$status" -eq 2 ] +} + +@test "prepare_dirs: creates the data and database directories for the service user" { + run bash -c " + NO_COLOR=1 + export NO_COLOR + chown() { echo \"\$*\" >> '$WORK/chown.log'; } + export -f chown + _INSTALL_SH_TESTING=1 . '$SCRIPT' + SERVICE_USER=\$(id -un) + CONFIG_DIR='$WORK/etc' + parse_maintenant_flags --data-dir '$WORK/data' --db '$WORK/db/maintenant.db' + resolve_paths + prepare_dirs + " + [ "$status" -eq 0 ] + [ -d "$WORK/data" ] + [ -d "$WORK/db" ] + grep -q ":.* $WORK/data\$" "$WORK/chown.log" + grep -q ":.* $WORK/db\$" "$WORK/chown.log" +} + +@test "prepare_dirs: refuses a non-empty directory that belongs to someone else" { + mkdir -p "$WORK/shared" + touch "$WORK/shared/other-app.conf" + run bash -c " + NO_COLOR=1 + export NO_COLOR + chown() { echo \"\$*\" >> '$WORK/chown.log'; } + export -f chown + _INSTALL_SH_TESTING=1 . '$SCRIPT' + SERVICE_USER=nobody + CONFIG_DIR='$WORK/etc' + parse_maintenant_flags --data-dir '$WORK/data' --db '$WORK/shared/maintenant.db' + resolve_paths + prepare_dirs + " + [ "$status" -eq 30 ] + [ ! -e "$WORK/chown.log" ] +} diff --git a/deploy/install/test/uninstall.bats b/deploy/install/test/uninstall.bats index 95e7dbcb..b6c04ef1 100644 --- a/deploy/install/test/uninstall.bats +++ b/deploy/install/test/uninstall.bats @@ -99,6 +99,37 @@ run_uninstall() { rm -rf "$FAKE_INSTALL_DIR" "$FAKE_DATA_DIR" "$FAKE_CONFIG_DIR" "$SERVICE_FILE" 2>/dev/null || true } +@test "uninstall --purge: removes the data dir set in the env file" { + FAKE_INSTALL_DIR=$(mktemp -d) + FAKE_DATA_DIR=$(mktemp -d) + FAKE_CONFIG_DIR=$(mktemp -d) + CUSTOM_DATA_DIR=$(mktemp -d) + touch "$CUSTOM_DATA_DIR/agent-identity.json" + printf 'MAINTENANT_DATA_DIR=%s\n' "$CUSTOM_DATA_DIR" > "$FAKE_CONFIG_DIR/maintenant.env" + + run bash -c " + NO_COLOR=1 + id() { echo 0; } + systemctl() { return 0; } + getent() { return 1; } + userdel() { return 0; } + export -f id systemctl getent userdel + INSTALL_DIR='$FAKE_INSTALL_DIR' + DATA_DIR='$FAKE_DATA_DIR' + CONFIG_DIR='$FAKE_CONFIG_DIR' + SERVICE_FILE='/tmp/no-such-service-$$.service' + export INSTALL_DIR DATA_DIR CONFIG_DIR SERVICE_FILE NO_COLOR + _INSTALL_SH_TESTING=1 . '$SCRIPT' + parse_maintenant_flags --uninstall --purge + check_prereqs + uninstall + " + [ "$status" -eq 0 ] + [ ! -d "$CUSTOM_DATA_DIR" ] + + rm -rf "$FAKE_INSTALL_DIR" "$FAKE_DATA_DIR" "$FAKE_CONFIG_DIR" "$CUSTOM_DATA_DIR" +} + @test "uninstall on system without maintenant returns 0 (idempotent)" { FAKE_INSTALL_DIR=$(mktemp -d) FAKE_DATA_DIR=$(mktemp -d) diff --git a/deploy/kubernetes/deployment.yaml b/deploy/kubernetes/deployment.yaml index c607556b..426f23a9 100644 --- a/deploy/kubernetes/deployment.yaml +++ b/deploy/kubernetes/deployment.yaml @@ -1,9 +1,12 @@ -# Apply with: kubectl apply -n -f deployment.yaml -# No namespace is hardcoded — kubectl -n sets it. +# Apply with: +# kubectl create namespace maintenant +# kubectl apply -f deploy/kubernetes/ +# The namespace is fixed here and in rbac.yaml; change both together. apiVersion: apps/v1 kind: Deployment metadata: name: maintenant + namespace: maintenant labels: app.kubernetes.io/name: maintenant app.kubernetes.io/component: monitoring @@ -105,6 +108,7 @@ apiVersion: v1 kind: PersistentVolumeClaim metadata: name: maintenant-data + namespace: maintenant spec: accessModes: - ReadWriteOnce @@ -119,6 +123,7 @@ apiVersion: v1 kind: Service metadata: name: maintenant + namespace: maintenant labels: app.kubernetes.io/name: maintenant spec: diff --git a/deploy/kubernetes/rbac.yaml b/deploy/kubernetes/rbac.yaml index 7634d2be..7213794f 100644 --- a/deploy/kubernetes/rbac.yaml +++ b/deploy/kubernetes/rbac.yaml @@ -1,28 +1,35 @@ -# Apply with: kubectl apply -n -f rbac.yaml -# The ServiceAccount is created in the target namespace automatically. -# The ClusterRoleBinding references it, so update the namespace below -# if you are NOT using "maintenant" as your namespace. +# Apply with: +# kubectl create namespace maintenant +# kubectl apply -f deploy/kubernetes/ +# The namespace is fixed here, in the ClusterRoleBinding subject and in +# deployment.yaml; change all of them together. apiVersion: v1 kind: ServiceAccount metadata: name: maintenant + namespace: maintenant --- apiVersion: rbac.authorization.k8s.io/v1 kind: ClusterRole metadata: name: maintenant rules: - # Core resources — read-only + # Read-only, exactly the APIs the runtime calls (internal/kubernetes/rbac.go). - apiGroups: [""] - resources: ["pods", "pods/log", "services", "namespaces", "events"] + resources: ["namespaces", "nodes", "pods", "events", "services"] verbs: ["get", "list", "watch"] - # Apps — read-only + - apiGroups: [""] + resources: ["pods/log"] + verbs: ["get"] - apiGroups: ["apps"] - resources: ["deployments", "statefulsets", "daemonsets", "replicasets"] + resources: ["deployments", "statefulsets", "daemonsets"] + verbs: ["get", "list", "watch"] + - apiGroups: ["batch"] + resources: ["jobs"] verbs: ["get", "list", "watch"] - # Metrics — read-only (requires metrics-server) + # Requires metrics-server. - apiGroups: ["metrics.k8s.io"] - resources: ["pods"] + resources: ["pods", "nodes"] verbs: ["get", "list"] --- apiVersion: rbac.authorization.k8s.io/v1 @@ -36,4 +43,4 @@ roleRef: subjects: - kind: ServiceAccount name: maintenant - namespace: maintenant # ← change this to match your namespace + namespace: maintenant diff --git a/docker-entrypoint.sh b/docker-entrypoint.sh index 2f3ebf14..a8feffe1 100755 --- a/docker-entrypoint.sh +++ b/docker-entrypoint.sh @@ -8,18 +8,43 @@ if [ "${1#-}" != "$1" ]; then set -- /app/maintenant "$@" fi -# Ensure the data directory used by the chosen mode is writable by the -# unprivileged runtime user. Covers both named volumes (Docker creates them -# root:root when the target path doesn't exist in the image) and bind mounts -# (host ownership leaks into the container). -case " $* " in - *" --mode=agent "*|*" --mode agent "*) - mkdir -p /var/lib/maintenant - chown 65534:65534 /var/lib/maintenant +# Already unprivileged (runAsUser, --user): no chown possible, and setpriv needs CAP_SETGID. +if [ "$(id -u)" != "0" ]; then + exec "$@" +fi + +mode="${MAINTENANT_MODE:-embedded}" +db="${MAINTENANT_DB:-./maintenant.db}" +data_dir="${MAINTENANT_DATA_DIR:-/var/lib/maintenant}" +prev="" +for arg in "$@"; do + case "$prev" in + --mode | -mode) mode="$arg" ;; + --db | -db) db="$arg" ;; + --data-dir | -data-dir) data_dir="$arg" ;; + esac + case "$arg" in + --mode=* | -mode=*) mode="${arg#*=}" ;; + --db=* | -db=*) db="${arg#*=}" ;; + --data-dir=* | -data-dir=*) data_dir="${arg#*=}" ;; + esac + prev="$arg" +done + +# Volumes and bind mounts arrive root-owned: hand the directory itself, never its content nor /, to the runtime user. +own_dir() { + mkdir -p -- "$1" + [ "$(cd -- "$1" && pwd -P)" != "/" ] || return 0 + chown 65534:65534 -- "$1" +} + +case "$mode" in + agent) + own_dir "$data_dir" ;; *) - mkdir -p /data/shm - chown 65534:65534 /data/shm + own_dir "$(dirname -- "$db")" + own_dir /data/shm ;; esac diff --git a/docs/api/reference.md b/docs/api/reference.md index 7f782605..b154655f 100644 --- a/docs/api/reference.md +++ b/docs/api/reference.md @@ -1,35 +1,117 @@ # API Reference -All endpoints are under `/api/v1/`. Responses are JSON. Errors follow a standard format: +The REST API lives under `/api/v1/`. Requests and responses are JSON unless a route says otherwise. Besides the REST API, the server exposes the public ping endpoints (`/ping/`), the public status page (`/status/`), the MCP endpoint (`/mcp`) with its OAuth routes, and the Server-Sent Events streams. Every route is listed on this page. + +--- + +## Conventions + +### Authentication + +maintenant has no built-in authentication for the dashboard or the REST API. Every `/api/` route answers anyone who can reach the listener, so put the instance behind a reverse proxy or an authentication proxy (see [Security](../security.md)). Three surfaces are open on purpose: + +- `/ping/{uuid}`: heartbeat pings. The UUID is the only secret. +- `/status/`: the public status page, its JSON snapshot, its event stream and its subscription links. +- `/mcp`: protected by its own OAuth 2.0 flow (see [MCP and OAuth routes](#mcp-and-oauth-routes)). + +The health check `/api/v1/health` is an ordinary `/api/` route, so it is exposed like the others. It reports the version, the runtime and the storage state. + +### Errors + +A failed request answers with a JSON body of this shape: ```json { "error": { - "code": "not_found", - "message": "Container not found" + "code": "QUOTA_EXCEEDED", + "message": "The Community edition is limited to 10 endpoints.", + "resource": "endpoints", + "limit": 10, + "required_edition": "personal" } } ``` +`code` and `message` are always present. The other fields (`feature`, `resource`, `limit`, `required_edition`, `window`, `max_window`) only appear on edition and quota refusals, so a client never has to parse the message. + +| Code | Status | Meaning | +|------|:------:|---------| +| `EDITION_REQUIRED` | 403 | The running edition does not open this feature. Carries `feature` (the capability) and `required_edition`. A history window the edition does not open also carries `window` and `max_window` (the largest window the edition opens). | +| `QUOTA_EXCEEDED` | 403 | A creation would pass the edition cap on endpoints, heartbeat monitors, certificate monitors or status page components. Carries `resource`, `limit` and `required_edition`, the lowest edition that lifts the cap. | +| `HOST_LIMIT_REACHED` | 409 | `POST /api/v1/agents/enrollment-tokens` when the edition's cap on remote agents is reached (20 on Personal). Same fields as `QUOTA_EXCEEDED`. | +| `STORAGE_UNAVAILABLE` | 503 | The database is unreachable. Only the routes that read through the store answer this, the others answer `INTERNAL_ERROR`. Clients should retry. | +| `INTERNAL_ERROR` | 500 | Unexpected failure, including a recovered panic. | +| `CROSS_ORIGIN_REFUSED` | 403 | A browser sent a write (`POST`, `PUT`, `PATCH`, `DELETE`) from an origin the server does not trust. See [Cross-origin writes](#cross-origin-writes). | +| `DEMO_MODE` | 403 | The instance runs in demo mode. See [Demo mode](#demo-mode). | +| `rate_limited` | 429 | The client IP went over a rate limit. Carries a `Retry-After` header. | + +The remaining codes are specific to a route and are listed with it. Most are upper case (`INVALID_JSON`, `NOT_FOUND`), but several handlers use lower case codes: webhooks, risk scoring, alert triggers, escalation policies, the status page administration (whose unexpected failures answer `internal` instead of `INTERNAL_ERROR`) and personalization, the public status page routes and `rate_limited`. Codes are given here exactly as the server returns them. + +Three answers are not JSON: + +- A route the server did not register (a feature that is not wired, such as the outbound heartbeats in demo mode) and a wrong method on a known path answer the plain-text `404` and `405` of the Go router. +- A request still running after 10 seconds is cut with `503` and the plain-text body `request timeout`. Streams are exempt. The work the request started can still finish, for example a synchronous certificate scan. +- The OAuth routes answer with the RFC 6749 shape (`{"error": "...", "error_description": "..."}`). + +### Editions and quotas + +`GET /api/v1/edition` returns the edition table the server applies, so a client never needs its own copy. Editions are ordered Community, Personal, Pro. + +| Edition | Capabilities opened | +|---------|---------------------| +| Community | `alert_routing`, `swarm_dashboard`, `k8s_cluster`, `resource_history` | +| Personal | Community, plus `multihost`, `cve_enrichment`, `risk_scoring`, `changelog`, `incidents`, `smtp`, `alert_advanced_filters`, `security_posture`, `ocsp_stapling`, `telegram` | +| Pro | Personal, plus `slack`, `teams`, `alert_escalation`, `maintenance_windows`, `subscribers`, `personalization` | + +| Resource | Community | Personal | Pro | +|----------|:---------:|:--------:|:---:| +| Standalone endpoints | 10 | unlimited | unlimited | +| Heartbeat monitors | 5 | unlimited | unlimited | +| Standalone certificate monitors | 5 | unlimited | unlimited | +| Status page components | 3 | unlimited | unlimited | +| Remote agents | 0 | 20 | unlimited | +| History window (resources) | 7 days | 30 days | 90 days | + +The caps are read from the running edition at each creation, so a licence change applies without a restart. The tables under each section give the edition a route needs in the **Edition** column. A dash means the route is open in every edition. + +### Host scope + +The local runtime is itself an agent, identified by the sentinel id `00000000-0000-0000-0000-000000000000`. Routes that list monitored entities accept an `agent_id` query parameter: `local` selects the server's own runtime, any other value selects that agent. Without it, list routes cover every host, with these exceptions: `GET /api/v1/resources/summary` defaults to the local host, and the single-entity Kubernetes routes (`workloads/{id}`, `pods/{namespace}/{name}`) look in the local runtime unless `agent_id` is given. The parameter is never validated: an unknown agent yields an empty list or a `404`. + +### Limits + +- **Rate limits**: per client IP, in token buckets. `/api/` allows 50 requests per second (burst 200). `/ping/`, `/status/`, `/mcp` and `/oauth/` share a tighter bucket of 10 requests per second (burst 20). Status page subscriptions are limited to 5 per hour. Client addresses behind a reverse proxy are only read from forwarded headers for the proxies listed in `MAINTENANT_TRUSTED_PROXIES`. +- **Body size**: 1 MiB for `/api/` by default (`MAINTENANT_MAX_BODY_SIZE`), whatever the method. A value that is not a positive whole number of bytes stops the server at startup. Past the limit the JSON decoding fails and the route answers its invalid body error. Public `/status/` routes accept 4 KiB. +- **Timeout**: 10 seconds for every route except the event streams. + +### Cross-origin writes + +The API refuses a browser write that comes from another origin, unless that origin is listed in `MAINTENANT_CORS_ORIGINS`: the answer is `403 CROSS_ORIGIN_REFUSED`. Same-origin requests, requests whose `Sec-Fetch-Site` is `none` and requests without `Origin` and `Sec-Fetch-Site` headers (curl, scripts) pass. A browser old enough to send no `Sec-Fetch-Site` header is refused behind a proxy that rewrites the `Host` header. The wildcard `*` in `MAINTENANT_CORS_ORIGINS` allows cross-origin reads but no longer cross-origin writes. `/ping/`, `/status/`, `/mcp` and `/oauth/` are not concerned. + +### Demo mode + +A demo build of maintenant is read-only. Every request other than `GET`, `HEAD` and `OPTIONS` answers `403 DEMO_MODE` (only `POST /api/v1/escalation-policies/overlap-probe`, which changes nothing, is let through). `/ping/`, `/mcp`, `/oauth/` and `/.well-known/oauth-*` answer `403 DEMO_MODE` for every method, and the outbound heartbeat routes are not registered. A request carrying the `X-Maintenant-Demo-Token` header with the value of `MAINTENANT_DEMO_TOKEN` bypasses the guard. `GET /api/v1/edition` reports `"demo": true`. + --- -## Health +## Health and instance information -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/health` | Health check, returns `{"status": "ok", "version": "...", "runtime": {...}, "storage": {...}}` | -| `GET` | `/api/v1/runtime/status` | Runtime info (docker/kubernetes, connection state) | -| `GET` | `/api/v1/edition` | Edition and feature flags | +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/health` | Health check | — | +| `GET` | `/api/v1/runtime/status` | Active runtime and its connection state | — | +| `GET` | `/api/v1/edition` | Edition, features, quotas and tiers | — | +| `GET` | `/api/v1/license/status` | Licence state | — | -The `storage` object reports the engine backing this instance, whether it -answers, and how many other instances beat on the same database: +### Health + +`GET /api/v1/health` always answers `200`: ```json -{ "engine": "postgres", "connected": true, "peers": 0 } +{ "status": "ok", "version": "1.8.0", "runtime": { "name": "docker", "connected": true }, "storage": { "engine": "sqlite", "connected": true, "peers": 0 } } ``` -`engine` is `sqlite` or `postgres`. It never carries the connection string, the -host or any credential. +`version`, `runtime` and `storage` are left out when they are not known. The `storage` object reports the engine backing this instance, whether it answers, and how many other instances beat on the same database. `engine` is `sqlite` or `postgres`. It never carries the connection string, the host or any credential. !!! warning "This endpoint answers 200 during a database outage" @@ -40,147 +122,357 @@ host or any credential. `503 STORAGE_UNAVAILABLE` in the meantime, which clients should ride out rather than treat as data loss. +### Runtime status + +`GET /api/v1/runtime/status` returns `runtime` (`docker` or `kubernetes`), `context` (`docker`, `swarm` or `kubernetes`), `connected`, `label` (`Containers`, `Services` or `Workloads`), `detected_at` and `metadata`. On a Swarm manager `metadata` holds `cluster_id`, `is_manager`, `manager_count`, `worker_count` and `service_count`. The manager and worker counts come from `docker info` and are refreshed every 60 seconds. A Swarm worker reports the `docker` context. + +### Edition + +`GET /api/v1/edition` returns: + +| Field | Description | +|-------|-------------| +| `edition` | `community`, `personal` or `pro` | +| `organisation_name` | Value of `MAINTENANT_ORGANISATION_NAME` (default `Maintenant`) | +| `status_url` | Public status page URL: the value of `MAINTENANT_STATUS_URL`, empty when it is not set | +| `demo` | `true` in demo mode | +| `features` | Every capability mapped to a boolean: does the running edition open it. `smtp` is also `false` until SMTP is configured | +| `feature_editions` | Every capability mapped to the lowest edition that opens it | +| `quotas` | For each capped resource present (`agent_hosts`, `endpoints`, `heartbeats`, `certificates`, `status_components`), an object `{ "used": n, "limit": n }`. A limit of `-1` means unlimited | +| `tiers` | The cap of every resource for every edition: `{ "community": {...}, "personal": {...}, "pro": {...} }` | +| `suspended_channels` | `{ "count": n, "channels": [{ "id", "name", "type", "required_edition" }] }`: enabled channels the running edition no longer opens | +| `resource_history` | `{ "max_window", "max_window_seconds", "windows": [{ "window", "seconds", "min_edition" }] }`. The window catalogue is the same in every edition | + +### Licence + +`GET /api/v1/license/status` always returns the same eight keys: `status`, `edition`, `plan`, `message`, `verified_at`, `expires_at`, `updates_until` and `update_grace_until`. Dates are RFC 3339 strings, or an empty string when the licence has none (a perpetual licence has no `expires_at`). Without a licence key the status is `inactive` and the edition `community`. Other statuses include `active`, `grace`, `expired`, `revoked`, `update_window_grace` and `update_window_ended`. + --- ## Containers +All routes are open in every edition. + | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/containers` | List all containers with groups | -| `GET` | `/api/v1/containers/{id}` | Get container details with uptime stats | +| `GET` | `/api/v1/containers` | List containers, grouped | +| `GET` | `/api/v1/containers/{id}` | Get a container with its uptime | | `GET` | `/api/v1/containers/{id}/transitions` | List state transitions | | `GET` | `/api/v1/containers/{id}/logs` | Fetch recent logs | -| `GET` | `/api/v1/containers/{id}/logs/stream` | Stream logs in real time (SSE) | +| `GET` | `/api/v1/containers/{id}/logs/stream` | Follow logs in real time (SSE) | +| `GET` | `/api/v1/containers/{id}/uptime/daily` | Daily uptime percentages | +| `GET` | `/api/v1/containers/{id}/endpoints` | List the endpoints of a container | | `DELETE` | `/api/v1/containers/{id}` | Remove a container from monitoring | -| `GET` | `/api/v1/containers/{id}/endpoints` | List endpoints for a container | + +`GET /api/v1/containers` query parameters: `archived=true` also returns archived containers, `group` (a custom or orchestration group name), `state` (for example `running`, `exited`, `completed`, `restarting`, `paused`, `created`, `dead`) and `agent_id`. A container that stopped with exit code 0 or 143 is `completed`, and so is one that ended with 137 unless the out-of-memory killer sent it: an OOM kill is `exited`, like any other crash. Containers marked `maintenant.ignore` are never listed. The answer is `{ "groups": [{ "name", "source", "containers": [...] }], "total", "archived_count" }`, plus `"stale": true` when the local runtime is disconnected. `source` is `label`, `namespace`, `compose`, `orchestration` or `default`. Each container carries its identity (`id`, `external_id`, `agent_id`, `name`, `image`), `state`, `health_status` (`null` without a health check), `has_health_check`, group and orchestration fields, `is_ignored`, `alert_severity`, `restart_threshold`, `archived`, timestamps, Kubernetes and Swarm details where they apply, `security_insight_count` and `security_highest_severity`. A container reported by a remote agent also carries `agent_hostname` and `agent_label`, and `stale` plus `agent_offline` when that agent has no live stream. + +`GET /api/v1/containers/{id}` returns the same core fields as one flat object, plus `uptime` and, for a Kubernetes workload, `container_names`. `uptime` holds a percentage for `24h` and for every longer window the edition's history cap opens: `7d` on Community, plus `30d` on Personal and `90d` on Pro. Use the daily route for day-by-day history. A Swarm task also carries `swarm_service_id`, `swarm_service_name`, `swarm_node_id` and `swarm_task_slot`. + +`GET /api/v1/containers/{id}/transitions` query parameters: `since` and `until` (RFC 3339; `since` defaults to 24 hours ago, an invalid value removes the bound), `limit` (default 50) and `offset`. The answer is `{ "container_id", "transitions", "total", "has_more" }`, newest first. + +`GET /api/v1/containers/{id}/logs` query parameters: `lines` (default 100, at most 500) and `timestamps=true`. The answer is `{ "container_id", "container_name", "lines", "total_lines", "truncated" }`. Logs of a remote container are fetched from its agent (15-second deadline). + +`GET /api/v1/containers/{id}/logs/stream` is a Server-Sent Events stream with its own `lines` parameter (default 100, at most 500) and, on Kubernetes, `container` to pick a container of the pod. It sends `container.log_line` events (`container_id`, `line`, `stream`, `timestamp`), a keep-alive comment every 25 seconds while the container is silent, and ends with a `container.log_error` event (`container_id`, `error`: `container stopped`, `agent disconnected` or the agent's error). The route exists when the local runtime can read logs or an agent can serve them. + +`DELETE /api/v1/containers/{id}` removes the container and its history with a hard delete and answers `204`. It emits `container.archived`. + +Errors: + +- `404 CONTAINER_NOT_FOUND`: no such container (also on the logs, stream and `endpoints` routes). +- `409 CONTAINER_RUNNING`: the container is running and cannot be deleted. +- `502 RUNTIME_UNAVAILABLE`: the local runtime or agent support is not wired for logs. `502 LOGS_UNAVAILABLE`: the runtime or the agent could not return the logs. +- Logs of a remote container: `503 AGENT_OFFLINE`, `501 AGENT_TOO_OLD` (the agent predates the log command), `429 LOGS_BUSY` (too many concurrent log requests for that agent) and `504 LOGS_TIMEOUT`. +- `logs` and `logs/stream` answer `503 RUNTIME_UNAVAILABLE` when the local runtime is disconnected. + +`GET /api/v1/containers/{id}/uptime/daily` (and the same route for endpoints and heartbeats) takes `days` (default 90, at most 365) and answers `{ "monitor_id", "monitor_type", "days": [{ "date", "uptime_percent", "incident_count" }] }`: one entry per UTC day, most recent first. `uptime_percent` is `null` for a day without data. Completed days come from the daily aggregates, which are kept for 365 days; the current day is computed from the raw rows. The day of a heartbeat is weighted by the time the monitor was up: a missed deadline counts as down until the next successful ping, and the days before its first ping have no data. --- ## Endpoints -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/endpoints` | List all monitored endpoints | -| `GET` | `/api/v1/endpoints/{id}` | Get endpoint details | -| `GET` | `/api/v1/endpoints/{id}/checks` | List check results | -| `POST` | `/api/v1/endpoints/{id}/check` | Probe now and return the refreshed endpoint | -| `GET` | `/api/v1/endpoints/{id}/uptime/daily` | Daily uptime percentages | +Endpoint monitors are discovered from labels or created through the API (a standalone endpoint). All routes are open in every edition, except that creating a standalone endpoint is capped. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/endpoints` | List endpoints | — | +| `POST` | `/api/v1/endpoints` | Create a standalone endpoint (10 on Community) | — | +| `GET` | `/api/v1/endpoints/{id}` | Get an endpoint with its uptime | — | +| `PUT` | `/api/v1/endpoints/{id}` | Update a standalone endpoint | — | +| `DELETE` | `/api/v1/endpoints/{id}` | Delete an endpoint | — | +| `GET` | `/api/v1/endpoints/{id}/checks` | List check results | — | +| `POST` | `/api/v1/endpoints/{id}/check` | Probe now and return the refreshed endpoint | — | +| `GET` | `/api/v1/endpoints/{id}/uptime/daily` | Daily uptime percentages | — | + +An endpoint has `id`, `endpoint_type` (`http` or `tcp`), `target`, `status` (`up`, `down`, `degraded` or `unknown`), `alert_state` (`normal` or `alerting`), the consecutive counters, the last check fields, `source` (`label` or `standalone`), `name`, `agent_id` and a `config` object (`interval` and `timeout` as duration strings such as `30s`, `failure_threshold`, `recovery_threshold`, `method`, `expected_status`, `tls_verify`, `headers`, `max_redirects`). A new standalone endpoint defaults to a 30-second interval, a 10-second timeout, failure threshold 3, recovery threshold 2, method `GET`, expected status `2xx`, TLS verification on and 5 redirects. + +- `GET /api/v1/endpoints` query parameters: `status`, `container` (container name), `type`, `source`, `include_inactive=true` and `agent_id`. The answer is `{ "endpoints": [...], "total" }`. `stale` and `agent_offline` mark an endpoint probed by an agent with no live stream. +- `POST /api/v1/endpoints` body: `name` (required), `target` (required; an absolute `http` or `https` URL for `http`, a `host:port` address with a port from 1 to 65535 for `tcp`), `endpoint_type` (required, `http` or `tcp`), `interval` (duration, at least 5s, default 30s; an interval below 5s in a `maintenant.endpoint.*.interval` label is raised to 5s), `timeout` (duration, at least 1s, default 10s, not above the interval), `method` (default `GET`) and `headers` (object). Thresholds, expected status, TLS verification and redirects cannot be set through the API. Answers `201` with `{ "endpoint": {...} }`. +- `PUT /api/v1/endpoints/{id}` accepts the same fields, all optional; an omitted or empty field keeps its value and `headers` replaces the whole map. The resulting target is validated for the endpoint type as on creation. Only standalone endpoints can be updated. +- `GET /api/v1/endpoints/{id}` answers `{ "endpoint", "uptime": { "1h", "24h", "7d", "30d" } }`. +- `GET /api/v1/endpoints/{id}/checks` takes `limit` (default 50, at most 500), `offset` and `since` (Unix seconds). The answer is `{ "endpoint_id", "checks", "total", "has_more" }`. +- `POST /api/v1/endpoints/{id}/check` probes synchronously and answers `{ "endpoint", "uptime" }`. + +Errors: + +- `400 INVALID_JSON`, `400 INVALID_INPUT` (the message names the field), `400 NOT_STANDALONE` (updating a label-discovered endpoint). +- `403 QUOTA_EXCEEDED`: creating past the cap (`resource: "endpoints"`, `limit: 10` on Community). +- `404 ENDPOINT_NOT_FOUND`. +- `409 ENDPOINT_LIVE`: deleting an endpoint whose container still runs. It becomes deletable once its container is gone. +- `409 AGENT_PROBED`: `check` on an endpoint an agent probes itself. `409 CHECK_IN_PROGRESS`: a check is already running. --- ## Heartbeats +All routes are open in every edition, except that creating a heartbeat monitor is capped. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/heartbeats` | List heartbeat monitors | — | +| `POST` | `/api/v1/heartbeats` | Create a heartbeat monitor (5 on Community) | — | +| `GET` | `/api/v1/heartbeats/{id}` | Get a heartbeat monitor and its ping snippets | — | +| `PUT` | `/api/v1/heartbeats/{id}` | Update a heartbeat monitor | — | +| `DELETE` | `/api/v1/heartbeats/{id}` | Delete a heartbeat monitor | — | +| `POST` | `/api/v1/heartbeats/{id}/pause` | Pause deadline checking | — | +| `POST` | `/api/v1/heartbeats/{id}/resume` | Resume deadline checking | — | +| `GET` | `/api/v1/heartbeats/{id}/executions` | List executions | — | +| `GET` | `/api/v1/heartbeats/{id}/pings` | List raw pings | — | +| `GET` | `/api/v1/heartbeats/{id}/uptime/daily` | Daily uptime percentages | — | + +A heartbeat has `id` (which is also its ping token), `name`, `status` (`new`, `up`, `down`, `started` or `paused`), `alert_state`, `interval_seconds`, `grace_seconds`, the last ping and deadline fields, the consecutive counters and timestamps. The deadline is the last ping plus the interval plus the grace period. + +- `POST /api/v1/heartbeats` body: `name` (required, 1 to 255 characters), `interval_seconds` (required, 60 to 604800) and `grace_seconds` (0 to the interval, default 0). Answers `201` with the heartbeat object. The cap counts every heartbeat, paused and agent-relayed ones included. +- `PUT /api/v1/heartbeats/{id}` takes the same fields, all optional. +- `GET /api/v1/heartbeats`: query parameters `status` and `agent_id`; answers `{ "heartbeats", "total" }`. +- `GET /api/v1/heartbeats/{id}` answers `{ "heartbeat", "snippets" }`; `snippets` holds ready-to-paste `curl`, `wget`, `python`, `go`, `bash` and `docker_healthcheck` snippets once `MAINTENANT_BASE_URL` is set. +- `executions` (`limit` default 20, at most 500, and `offset`) answers `{ "executions", "total" }`, newest first; each has an `outcome` of `success`, `failure`, `timeout` or `in_progress`. `pings` (`limit` default 50, at most 500, and `offset`) answers `{ "pings", "total" }` with `expected_at` and `grace_deadline` computed for each ping. +- `pause`, `resume` and `PUT` answer `200` with the heartbeat; `DELETE` answers `204`. Deleting is a hard delete: the monitor, its pings, executions, pauses and daily uptime are removed, and its id answers `404` afterwards, on these routes and on `/ping/`. + +Errors: + +- `400 INVALID_JSON`, `400 INVALID_INPUT` (including `heartbeat is already paused` on `pause` and `heartbeat is not paused` on `resume`). A ping received while the heartbeat is paused puts it back to monitoring. +- `403 QUOTA_EXCEEDED`: creating past the cap (`resource: "heartbeats"`, `limit: 5` on Community). +- `404 NOT_FOUND`: unknown heartbeat, including on `DELETE`, `pause` and `resume`. + +### Ping endpoints (public) + +These routes need no credentials: the UUID of the heartbeat is the secret. They share the tight rate limit of `/ping/` and answer `403 DEMO_MODE` in demo mode. + | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/heartbeats` | List all heartbeat monitors | -| `POST` | `/api/v1/heartbeats` | Create a heartbeat monitor | -| `GET` | `/api/v1/heartbeats/{id}` | Get a heartbeat monitor | -| `PUT` | `/api/v1/heartbeats/{id}` | Update a heartbeat monitor | -| `DELETE` | `/api/v1/heartbeats/{id}` | Delete a heartbeat monitor | -| `POST` | `/api/v1/heartbeats/{id}/pause` | Pause deadline checking | -| `POST` | `/api/v1/heartbeats/{id}/resume` | Resume deadline checking | -| `GET` | `/api/v1/heartbeats/{id}/executions` | List executions | -| `GET` | `/api/v1/heartbeats/{id}/pings` | List raw pings | -| `GET` | `/api/v1/heartbeats/{id}/uptime/daily` | Daily uptime percentages | - -### Outbound Heartbeats +| `GET/POST` | `/ping/{uuid}` | Simple ping (success) | +| `GET/POST` | `/ping/{uuid}/start` | Signal job start | +| `GET/POST` | `/ping/{uuid}/{exit_code}` | Ping with exit code (0 = success) | + +- `/ping/{uuid}` and `/ping/{uuid}/start` answer `200 {"ok": true}`. `/ping/{uuid}/{exit_code}` answers `200 {"ok": true, "exit_code": n}`. A non-zero exit code raises an alert. +- Errors: `404 HEARTBEAT_NOT_FOUND` (unknown or deleted heartbeat), `400 INVALID_EXIT_CODE` (not an integer between 0 and 255, checked before the lookup) and `500 INTERNAL_ERROR`. +- A `POST` body of up to 10 KiB is read but never stored. +- The `source_ip` of a ping is the client address resolved like the rate limit does: forwarded headers are believed only when the connection comes from a proxy listed in `MAINTENANT_TRUSTED_PROXIES`. + +### Outbound heartbeats + +Outbound heartbeats make this instance ping an external URL so another system notices when it stops. The routes are not registered in demo mode. All are open in every edition. See [Outbound Heartbeats](../features/heartbeats.md#outbound-heartbeats). | Method | Endpoint | Description | |--------|----------|-------------| | `GET` | `/api/v1/outbound-heartbeats` | List all targets | | `POST` | `/api/v1/outbound-heartbeats` | Create a target | -| `PUT` | `/api/v1/outbound-heartbeats/{id}` | Update a target | +| `PUT` | `/api/v1/outbound-heartbeats/{id}` | Replace a target | | `DELETE` | `/api/v1/outbound-heartbeats/{id}` | Delete a target | | `POST` | `/api/v1/outbound-heartbeats/{id}/send` | Send a ping now | -See [Outbound Heartbeats](../features/heartbeats.md#outbound-heartbeats). +A target has `id`, `name`, `url`, `interval_seconds`, `enabled`, `last_sent_at`, `last_status_code`, `last_error`, `created_at` and `updated_at`. The body of `POST` and `PUT` holds `name` (1 to 255 characters), `url` (an absolute HTTPS URL that does not resolve to a private or internal address), `interval_seconds` (30 to 86400) and `enabled` (default `true`; `PUT` replaces the target, so omitting `enabled` enables it). The list answers `{ "outbound_heartbeats": [...] }`. `send` works on a disabled target and answers `200` with the refreshed object: a failed send is recorded in `last_error` and `last_status_code`, not returned as an HTTP error. Errors: `400 INVALID_JSON`, `400 INVALID_INPUT` and `404 NOT_FOUND`. -### Ping Endpoints (Public) +--- -These routes do not require authentication: +## Certificates -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET/POST` | `/ping/{uuid}` | Simple ping (success) | -| `GET/POST` | `/ping/{uuid}/start` | Signal job start | -| `GET/POST` | `/ping/{uuid}/{exit_code}` | Ping with exit code (0 = success) | +All routes are open in every edition, except that creating a standalone monitor is capped. ---- +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/certificates` | List certificate monitors | — | +| `POST` | `/api/v1/certificates` | Create a standalone certificate monitor (5 on Community) | — | +| `GET` | `/api/v1/certificates/{id}` | Get a monitor with its latest check and chain | — | +| `PUT` | `/api/v1/certificates/{id}` | Update a monitor | — | +| `DELETE` | `/api/v1/certificates/{id}` | Delete a monitor | — | +| `GET` | `/api/v1/certificates/{id}/checks` | List check history | — | +| `POST` | `/api/v1/certificates/{id}/check` | Scan now and return the refreshed monitor | — | -## Certificates +A monitor has `id`, `hostname`, `port`, `server_name`, `source` (`auto`, `standalone` or `label`), `status` (`valid`, `expiring`, `expired`, `error` or `unknown`), `check_interval_seconds`, `warning_thresholds` (days, default `[30, 14, 7, 3, 1]`), the last check and next check times, `agent_id` and timestamps. -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/certificates` | List all certificate monitors | -| `POST` | `/api/v1/certificates` | Create a standalone certificate monitor | -| `GET` | `/api/v1/certificates/{id}` | Get certificate details | -| `PUT` | `/api/v1/certificates/{id}` | Update a certificate monitor | -| `DELETE` | `/api/v1/certificates/{id}` | Delete a certificate monitor | -| `GET` | `/api/v1/certificates/{id}/checks` | List check history | -| `POST` | `/api/v1/certificates/{id}/check` | Scan now and return the refreshed monitor | +- `GET /api/v1/certificates` query parameters: `status`, `source` and `agent_id`. Each monitor carries a `latest_check` object, and the answer is `{ "certificates", "total" }`. +- `POST /api/v1/certificates` body: `hostname` (required), `port` (default 443), `server_name` (a bare host name, optional), `check_interval_seconds` (3600 to 604800, default 43200) and `warning_thresholds`. The first TLS check runs before the answer, so `201` returns `{ "certificate", "latest_check" }`. +- `PUT /api/v1/certificates/{id}` accepts `check_interval_seconds` and `warning_thresholds` only; other fields are ignored. +- `GET /api/v1/certificates/{id}` returns `{ "certificate", "latest_check" }` where `latest_check` holds the subject, issuer, SANs, validity dates, `days_remaining`, `chain_valid`, `chain_error`, `hostname_match`, the `chain` and, for a stapled response, `ocsp_stapled`, `ocsp_status`, `ocsp_produced_at`, `ocsp_next_update` and `ocsp_error`. +- `GET /api/v1/certificates/{id}/checks` takes `limit` (default 50, at most 500) and `offset` and answers `{ "monitor_id", "checks", "total", "has_more" }`. Each check lists `ocsp_stapled` and `ocsp_status` only; the other OCSP fields are on `latest_check`. A scan pushed by an endpoint probe or by an agent is stored once per monitor interval, or sooner when its outcome changes, so the history holds one row per interval or change. + +Errors: + +- `400 INVALID_JSON`, `400 INVALID_INPUT`, `400 CANNOT_DELETE_AUTO` (an auto-detected monitor is removed by removing its label). +- `403 QUOTA_EXCEEDED` (`resource: "certificates"`, `limit: 5` on Community). +- `404 NOT_FOUND`. +- `409 DUPLICATE_MONITOR`, `409 ALREADY_AUTO_DETECTED`, `409 AGENT_SCANNED` (the agent scans this one itself) and `409 CHECK_IN_PROGRESS`. --- ## Resources +All routes are open in every edition. The history window is capped by the edition. + | Method | Endpoint | Description | Edition | |--------|----------|-------------|:-------:| -| `GET` | `/api/v1/containers/{id}/resources/current` | Current CPU, memory, network, I/O | | -| `GET` | `/api/v1/containers/{id}/resources/history` | Historical metrics (`?range=24h`) | Pro | -| `GET` | `/api/v1/containers/{id}/resources/alerts` | Get alert thresholds | | -| `PUT` | `/api/v1/containers/{id}/resources/alerts` | Set alert thresholds | | -| `GET` | `/api/v1/resources/summary` | Aggregate resource summary (`?agent_id=local\|` to scope to a host) | | -| `GET` | `/api/v1/resources/top` | Top consumers (`?sort=cpu&limit=10`, `?agent_id=` to scope to a host) | | -| `GET` | `/api/v1/resources/hosts` | List hosts (local + agents) with current CPU/memory/disk | Pro | +| `GET` | `/api/v1/containers/{id}/resources/current` | Current CPU, memory, network, I/O | — | +| `GET` | `/api/v1/containers/{id}/resources/history` | Historical metrics (`?range=24h`; window `30d` needs Personal, `90d` needs Pro) | — | +| `GET` | `/api/v1/containers/{id}/resources/alerts` | Get alert thresholds | — | +| `PUT` | `/api/v1/containers/{id}/resources/alerts` | Set alert thresholds | — | +| `GET` | `/api/v1/resources/summary` | Aggregate resource summary of one host | — | +| `GET` | `/api/v1/resources/top` | Top consumers (a `period` of `30d` needs Personal, `90d` needs Pro) | — | +| `GET` | `/api/v1/resources/hosts` | List hosts (local and agents) with current CPU, memory and disk | — | + +- `current` answers `container_id`, `cpu_percent` (100 is one core), `mem_used`, `mem_limit`, `mem_percent`, `net_rx_bytes`, `net_tx_bytes`, `block_read_bytes`, `block_write_bytes` and `timestamp`. It returns `404 NOT_FOUND` when no sample exists yet. For a container on a remote agent, only a sample received in the last 35 seconds counts. +- `history` takes `range`: `1h` (the default), `6h`, `24h`, `7d`, `30d` or `90d`. The answer is `{ "container_id", "range", "granularity", "points" }` with `granularity` of `raw`, `1m`, `5m`, `1h` or `1d`. Errors: `400 INVALID_RANGE` for an unknown window, `403 EDITION_REQUIRED` for a window the edition does not open (with `feature: "resource_history"`, `required_edition`, `window` and `max_window`). +- `alerts` (GET) answers `container_id`, `cpu_threshold`, `mem_threshold`, `enabled`, `alert_state` and `last_alerted_at`; without configuration it returns 90, 90 and `enabled: false`. `PUT` takes `cpu_threshold` (1 to 1000, required), `mem_threshold` (1 to 100) and `enabled` (omitted means `false`). An alert fires after two consecutive breaching samples, and CPU and memory fire and resolve independently. Errors: `400 INVALID_THRESHOLD`. +- `summary` takes `agent_id` (`local` or absent for the local host) and answers the host totals: `agent_id` (an empty string for the local host), `available`, `total_cpu_percent`, `cpu_count`, `total_mem_used`, `total_mem_limit`, `total_mem_percent`, `total_net_rx_rate`, `total_net_tx_rate`, `container_count`, `disk_total`, `disk_used`, `disk_percent` and `timestamp`. The container count and the network figures cover the containers of that host, including those of a remote agent. `total_net_rx_rate` and `total_net_tx_rate` are throughputs in bytes per second, computed from the last two samples of each container. A container with a single sample, a counter that went backwards (restart) or a runtime that does not report network counters adds nothing to them. +- `top` requires `metric` (`cpu` or `memory`, otherwise `400 INVALID_METRIC`), and takes `limit` (default 5; a larger value is capped at 20), `agent_id` (absent means every host) and `period` (a history window). The hour or day in progress counts in a `period` ranking, not only the closed hours and days. Without `period` the ranking uses the latest samples and is open in every edition. The answer is `{ "metric", "period", "consumers": [{ "container_id", "container_name", "value", "percent", "rank" }] }`. A bad `period` answers `400 INVALID_PERIOD`, or `403 EDITION_REQUIRED` as for `history`. +- `hosts` answers `{ "hosts": [{ "agent_id", "hostname", "label", "is_local", "available", "cpu_percent", "mem_used", "mem_total", "mem_percent", "disk_total", "disk_used", "disk_percent", "container_count" }] }`, the local host first. --- -## Agents +## Kubernetes + +All routes are open in every edition. They are read from the store, per agent, so they also serve clusters monitored by a remote agent. The routes that need metrics-server (`workloads/{id}/resources`, `nodes/{name}/resources`) answer `200` with `"metrics_available": false` and a message when the data is not available, which is always the case for a remote agent. + +| Method | Endpoint | Description | +|--------|----------|-------------| +| `GET` | `/api/v1/kubernetes/namespaces` | List namespaces | +| `GET` | `/api/v1/kubernetes/workloads` | List workloads, grouped by namespace | +| `GET` | `/api/v1/kubernetes/workloads/{id}` | Workload with its pods and events | +| `GET` | `/api/v1/kubernetes/workloads/{id}/resources` | Per-pod CPU and memory from metrics-server | +| `GET` | `/api/v1/kubernetes/pods` | List pods | +| `GET` | `/api/v1/kubernetes/pods/{namespace}/{name}` | Pod with its events | +| `GET` | `/api/v1/kubernetes/nodes` | List nodes | +| `GET` | `/api/v1/kubernetes/nodes/{name}/resources` | Node CPU and memory from metrics-server | +| `GET` | `/api/v1/kubernetes/cluster` | Cluster overview | + +- Every route takes `agent_id`. +- `workloads` takes `namespaces` (comma-separated), `kind` (`Deployment`, `StatefulSet`, `DaemonSet` or `Job`) and `status` (`healthy`, `degraded`, `progressing` or `failed`). It answers `{ "groups": [{ "namespace", "workloads" }], "total" }`. A workload has `id` (`namespace/Kind/name`), `name`, `namespace`, `kind`, `images`, `ready_replicas`, `desired_replicas`, `status` and `created_at`. +- `{id}` is the workload id with its slashes URL-encoded (`default%2FDeployment%2Fweb`). +- `pods` takes `namespaces`, `workload` (id prefix), `node` and `status` (for example `Running`, `Pending`). `{ "pods", "total" }`. +- `cluster` answers `namespace_count`, `node_count`, `node_ready_count`, `pod_status`, `workload_count`, `workload_healthy`, `cluster_health` (`healthy` or `degraded`) and a per-namespace summary. +- Errors: `404 K8S_WORKLOAD_NOT_FOUND`, `404 K8S_POD_NOT_FOUND`, `400 INVALID_ID` (bad workload id encoding) and `500 K8S_ERROR`. -Multi-host agent management (`--mode=server`). All endpoints require **Personal** or above. Below that they return `403 EDITION_REQUIRED`, naming the capability and the edition that grants it. See [Multi-Host Monitoring](../features/multihost.md). +--- + +## Swarm + +All routes are open in every edition. Services, tasks and nodes are read from the store, per agent. `info`, `dashboard`, `cluster`, `update-status` and `resources` read the server's own live Swarm and answer `409 SWARM_NOT_ACTIVE` when the local runtime is not a Swarm manager. | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/agents` | List enrolled agents, their connection state and their host OS (`os`) | -| `GET` | `/api/v1/agents/{id}` | Get an agent | -| `PATCH` | `/api/v1/agents/{id}` | Update an agent's display label | -| `POST` | `/api/v1/agents/{id}/revoke` | Revoke an agent (closes its stream, stops retries) | -| `DELETE` | `/api/v1/agents/{id}` | Delete an agent and purge all its events | -| `GET` | `/api/v1/agents/metrics` | Aggregate fleet metrics (counts by status) | -| `POST` | `/api/v1/agents/enrollment-tokens` | Create a one-time enrollment token (`{ "ttl_hours": 24 }`) | -| `GET` | `/api/v1/agents/enrollment-tokens` | List enrollment tokens (masked) | -| `GET` | `/api/v1/agents/enrollment-tokens/{token_id}` | Get an enrollment token (masked) | -| `DELETE` | `/api/v1/agents/enrollment-tokens/{token_id}` | Delete an enrollment token | +| `GET` | `/api/v1/swarm/info` | Swarm state of the local runtime (`{ "active": false }` when there is none) | +| `GET` | `/api/v1/swarm/services` | List services (`?stack=`, `?agent_id=`) | +| `GET` | `/api/v1/swarm/services/{serviceID}` | Service with its tasks | +| `GET` | `/api/v1/swarm/services/{serviceID}/update-status` | Rolling update state (live) | +| `GET` | `/api/v1/swarm/services/{serviceID}/resources` | Per-task CPU, memory and network (live) | +| `GET` | `/api/v1/swarm/tasks` | List tasks (`?service=`, `?node=`, `?state=`, `?agent_id=`) | +| `GET` | `/api/v1/swarm/nodes` | List nodes (`?agent_id=`) | +| `GET` | `/api/v1/swarm/nodes/{nodeID}` | Node with its tasks (live) | +| `GET` | `/api/v1/swarm/dashboard` | Cluster, node and service summary (live) | +| `GET` | `/api/v1/swarm/cluster` | Cluster health and counts (live) | + +`dashboard` answers `cluster` (counts and task health), `nodes` and `services`. + +`info` answers `active`, `cluster_id`, `is_manager`, `manager_count`, `worker_count` and `created_at`. The counts (also in `dashboard`, `cluster` and `GET /api/v1/runtime/status`) and the creation date come from `docker info` and are refreshed every 60 seconds. The server detects a Swarm as soon as Docker connects, so activating Swarm later needs no restart. + +Errors: `404 SWARM_SERVICE_NOT_FOUND`, `404 SWARM_NODE_NOT_FOUND`, `409 SWARM_NOT_ACTIVE`, `409 SWARM_NODES_NOT_AVAILABLE` (node monitoring is not available) and `500 INTERNAL_ERROR`. + +--- + +## Agents + +Multi-host agent management. Every route requires **Personal** or above (capability `multihost`) and answers `403 EDITION_REQUIRED` below that. See [Multi-Host Monitoring](../features/multihost.md). + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/agents` | List enrolled agents | Personal | +| `GET` | `/api/v1/agents/{id}` | Get an agent | Personal | +| `PATCH` | `/api/v1/agents/{id}` | Update an agent's display label | Personal | +| `POST` | `/api/v1/agents/{id}/revoke` | Revoke an agent (closes its stream, stops retries) | Personal | +| `DELETE` | `/api/v1/agents/{id}` | Delete an agent and purge all its events | Personal | +| `GET` | `/api/v1/agents/metrics` | Fleet counts | Personal | +| `POST` | `/api/v1/agents/enrollment-tokens` | Create a one-time enrollment token | Personal | +| `GET` | `/api/v1/agents/enrollment-tokens` | List enrollment tokens (masked) | Personal | +| `GET` | `/api/v1/agents/enrollment-tokens/{token_id}` | Get an enrollment token (masked) | Personal | +| `DELETE` | `/api/v1/agents/enrollment-tokens/{token_id}` | Delete an enrollment token | Personal | + +An agent has `agent_id`, `hostname`, `label`, `os_arch`, `agent_version`, `detected_runtime` (`docker`, `swarm` or `kubernetes`), `status` (`active` or `revoked`), `connection_state` (`connected` or `disconnected`), `last_seen_at`, `created_at`, `revoked_at`, `revoked_by`, `spool` and `os`. `spool` (`queued`, `draining`, `dropped_since_connect`, `reported_at`) is `null` while the agent is disconnected. `os` holds the host operating system (`id`, `version_id`, `pretty_name`, `source`, `unavailable_reason`, `reported_at`) and its end-of-support state. An agent is `connected` while its stream is live or it was seen within `MAINTENANT_AGENT_STALE_THRESHOLD_SECONDS` (60 by default). + +- `GET /api/v1/agents` takes `status` (`active`, `revoked` or `all`) and `connection_state` (`connected` or `disconnected`); the answer is `{ "agents": [...] }`. +- `PATCH` takes `{ "label": "..." }` (required, up to 64 characters, empty clears it) and answers the updated agent. +- `revoke` answers the agent and `delete` answers `204`. +- `POST /api/v1/agents/enrollment-tokens` takes an optional `{ "ttl_hours": n }` (default 24, at most 168). The `201` answer carries `token_id`, `token` (shown only here), `token_masked`, `created_at`, `expires_at`, `install_templates` (`standalone`, `docker_run`, `docker_compose`, `kubernetes`) and `warnings` (for example `public_url_appears_local`, `public_url_plaintext_refused`). +- `GET .../enrollment-tokens` lists unconsumed, unexpired tokens; `include_expired=true` and `include_consumed=true` widen the list. The answer is `{ "tokens": [...] }`. +- `GET /api/v1/agents/metrics` answers `total`, `by_status`, `by_runtime`, `by_connection_state` and `total_events_per_second_observed_5m`. + +Errors: `404 NOT_FOUND`, `400 INVALID_JSON`, `400 MISSING_FIELD`, `400 LABEL_TOO_LONG`, `409 TOKEN_CONSUMED` (deleting a consumed token) and `409 HOST_LIMIT_REACHED` (creating a token when the cap is reached: 20 active remote agents on Personal, the error names `pro` as the edition that lifts it). --- ## Alerts +All routes are open in every edition. + | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/alerts` | List all alerts (including resolved) | -| `GET` | `/api/v1/alerts/active` | List active (unresolved) alerts | -| `GET` | `/api/v1/alerts/{id}` | Get alert details | +| `GET` | `/api/v1/alerts` | List alerts (including resolved), newest first | +| `GET` | `/api/v1/alerts/active` | List active alerts, by severity | +| `GET` | `/api/v1/alerts/{id}` | Get an alert | +| `POST` | `/api/v1/alerts/{id}/acknowledge` | Acknowledge an active alert | + +- `GET /api/v1/alerts` takes `source`, `severity` and `status` (`active`, `resolved` or `silenced`), `before` (RFC 3339; returns alerts fired before it), `before_id` and `limit` (1 to 200, default 50). The answer is `{ "alerts": [...], "has_more": bool }`, ordered by `fired_at` then `id`, newest first. To page, pass the `fired_at` and the `id` of the last alert as `before` and `before_id`: alerts fired in the same second as the cursor are neither skipped nor repeated. `before_id` without `before` is refused. Errors: `400 INVALID_PARAM`. +- `GET /api/v1/alerts/active` answers `{ "critical": [...], "warning": [...], "info": [...] }`. Acknowledged and resolved alerts are left out. +- `GET /api/v1/alerts/{id}` answers `404 NOT_FOUND` for an unknown alert. +- A daily job purges the alerts that are not active (resolved or silenced) once they are older than 90 days, counted from their resolution, or from their creation when they never resolved. An alert that is still active is never purged, however old. +- An alert has `id`, `source`, `alert_type`, `severity` (`critical`, `warning` or `info`), `status`, `message`, `entity_type`, `entity_id`, `entity_name`, `details` (a string holding JSON), `fired_at`, `resolved_at`, `resolved_by_id`, `acknowledged_at`, `acknowledged_by`, `escalated_at` (set when an escalation policy notifies a level for the alert) and `created_at`. +- `POST .../acknowledge` takes `{ "acknowledged_by": "..." }` (required). It answers the alert, emits `alert.acknowledged` and stops its running escalation. The dashboard, the MCP server and the security posture acknowledgments all go through this one path, so they behave the same. An acknowledged alert whose severity later rises does not start a new escalation. Errors: `400 INVALID_BODY`, `400 INVALID_REQUEST`, `404 NOT_FOUND` (unknown alert) and `409 CONFLICT` (the alert is not active or is already acknowledged). --- ## Notification Channels -Channels are silent by default. They only fire when referenced by an [Alert Trigger](#alert-triggers) or an [Escalation Policy](#escalation-policies). +Channels are silent by default. They only fire when referenced by an [Alert Trigger](#alert-triggers) or an [Escalation Policy](#escalation-policies). Every route is open, but the channel type needs an edition. | Method | Endpoint | Description | |--------|----------|-------------| | `GET` | `/api/v1/channels` | List notification channels | -| `POST` | `/api/v1/channels` | Create a channel (slack, discord, teams, webhook, email) | +| `POST` | `/api/v1/channels` | Create a channel | | `PUT` | `/api/v1/channels/{id}` | Update a channel | | `DELETE` | `/api/v1/channels/{id}` | Delete a channel | -| `POST` | `/api/v1/channels/{id}/test` | Send a test alert | +| `POST` | `/api/v1/channels/{id}/test` | Send a test notification | + +| Channel `type` | Destination (`url`) | Edition | +|----------------|---------------------|:-------:| +| `webhook` (default) | HTTPS URL | — | +| `discord` | Discord webhook URL | — | +| `email` | Recipient address (needs `MAINTENANT_SMTP_*`) | Personal | +| `telegram` | Chat id (`-1001234567890`) or public `@username` | Personal | +| `slack` | Slack webhook URL | Pro | +| `teams` | Teams webhook URL | Pro | + +A channel has `id`, `name`, `type`, `url`, `headers` (a string holding a JSON object), `config` (a string holding JSON, for example `{"thread_id": "42"}` on Telegram), `has_secret`, `enabled`, `health` (`healthy` or `failing`, on the list only), `suspended`, `required_edition` and timestamps. The secret (a Telegram bot token) is never returned, only `has_secret`; the URL and headers are returned as stored. + +- `POST` body: `name` (required), `url` (required), `type`, `headers`, `secret` (required for Telegram), `config` and `enabled` (default `true`). A URL must be HTTPS and must not resolve to a private or internal address, unless `MAINTENANT_ALLOW_PRIVATE_WEBHOOKS` is set. Answers `201` with the channel. +- `PUT` takes the same fields, all optional. A secret cannot be cleared. A request that only sets `enabled` to `false` is always accepted, even after the edition dropped. +- `test` answers `404 NOT_FOUND` for an unknown channel. Otherwise it answers `200`: `{ "status": "delivered", "response_code": n }` or `{ "status": "failed", "error": "..." }`. +- Creating, updating or testing a channel of a type the edition does not open answers `403 EDITION_REQUIRED` (`feature` is the capability: `slack`, `teams`, `smtp` or `telegram`). After a downgrade such a channel is `suspended`: it stops delivering and `GET /api/v1/edition` lists it under `suspended_channels`. +- Errors: `400 INVALID_BODY`, `400 VALIDATION_ERROR`, `404 NOT_FOUND` and `409 DUPLICATE_NAME` on create. --- ## Alert Triggers -Triggers route alerts to channels based on filters (severity, source, scope, tag). +Triggers route alerts to channels based on filters. A trigger matches when all of its non-empty filters match. | Method | Endpoint | Description | |--------|----------|-------------| @@ -190,45 +482,63 @@ Triggers route alerts to channels based on filters (severity, source, scope, tag | `PUT` | `/api/v1/alert-triggers/{id}` | Update a trigger | | `DELETE` | `/api/v1/alert-triggers/{id}` | Delete a trigger | -Filters `filter_scopes` and `filter_tags` require Personal or above; `filter_severities` and `filter_sources` are available on every edition. +A trigger has `id`, `name`, `filter_severities`, `filter_sources` and `filter_scopes` (comma-separated strings, empty matches everything; a scope is `entity_type:entity_id`), `enabled`, `notify_on_resolve`, `channel_ids` and timestamps. The list answers `{ "triggers": [...] }`. -`notify_on_resolve` (boolean, default `true`) controls whether the trigger also relays `alert.resolved` events; set it to `false` for a channel that should only receive failures. +- `POST` body: `name` (required, up to 120 characters), `channel_ids` (required, at least one existing channel), the three filters, `enabled` (default `true`) and `notify_on_resolve` (default `true`, set it to `false` for a channel that should only receive failures). +- `PUT` replaces the name, the filters and the channels; `enabled` and `notify_on_resolve` keep their value when omitted. +- `filter_scopes` requires **Personal** (capability `alert_advanced_filters`). Below Personal, a non-empty value that differs from the stored one answers `403 EDITION_REQUIRED`; keeping or clearing the stored value is allowed. `filter_severities` and `filter_sources` are open in every edition. +- Errors (lower case codes): `400 validation_failed`, `404 trigger_not_found`, `409 name_conflict`. Invalid JSON answers `400 INVALID_BODY`. --- -## Escalation Policies :material-crown:{ title="Pro" } +## Escalation Policies -Multi-level escalation chains for unacknowledged alerts. +Multi-level escalation chains for unacknowledged alerts. Every route requires **Pro** (capability `alert_escalation`) and answers `403 EDITION_REQUIRED` below it. -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/escalation-policies` | List all policies | -| `POST` | `/api/v1/escalation-policies` | Create a policy | -| `GET` | `/api/v1/escalation-policies/{id}` | Get a policy | -| `PUT` | `/api/v1/escalation-policies/{id}` | Update a policy | -| `PATCH` | `/api/v1/escalation-policies/{id}/active` | Activate / deactivate | -| `DELETE` | `/api/v1/escalation-policies/{id}` | Delete a policy | -| `POST` | `/api/v1/escalation-policies/overlap-probe` | Detect overlapping policies | -| `GET` | `/api/v1/escalation-policies/{id}/runs` | List recent runs for a policy | -| `GET` | `/api/v1/alerts/{id}/escalation-runs` | List runs for an alert | -| `GET` | `/api/v1/escalation-runs/{id}` | Get run detail and deliveries | - -Endpoints return `403 edition_required` on Community. +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/escalation-policies` | List policies (`?active=true` keeps the active ones) | Pro | +| `POST` | `/api/v1/escalation-policies` | Create a policy | Pro | +| `GET` | `/api/v1/escalation-policies/{id}` | Get a policy | Pro | +| `PUT` | `/api/v1/escalation-policies/{id}` | Replace a policy | Pro | +| `PATCH` | `/api/v1/escalation-policies/{id}/active` | Activate or deactivate | Pro | +| `DELETE` | `/api/v1/escalation-policies/{id}` | Delete a policy | Pro | +| `POST` | `/api/v1/escalation-policies/overlap-probe` | Detect overlapping policies | Pro | +| `GET` | `/api/v1/escalation-policies/{id}/runs` | List recent runs of a policy | Pro | +| `GET` | `/api/v1/alerts/{alert_id}/escalation-runs` | List the runs of an alert | Pro | +| `GET` | `/api/v1/escalation-runs/{run_id}` | Get a run with its deliveries | Pro | + +A policy is `{ "name", "active", "filters": { "severities", "scopes": [{ "kind", "ref_id" }] }, "levels": [{ "delay_seconds", "channel_ids" }] }`. Validation: a name of 1 to 120 characters, one to 5 levels, each delay between 60 and 86400 seconds and at least 60 seconds after the previous level, and at least one channel per level. `active` has no default: leaving it out creates or replaces the policy as inactive. The delays of the levels count from the start of the run. + +- `GET /api/v1/escalation-policies` answers `{ "policies": [...], "limits": { "max_active", "max_levels", "current_active" } }`. `max_active` is `-1` (no cap on active policies) and `max_levels` is 5. +- `overlap-probe` takes a policy body and answers `{ "overlapping": [{ "policy_id", "policy_name", "shared_channels", "filter_intersection" }] }`: policies whose filters intersect and that share at least one channel. +- `PATCH .../active` takes `{ "active": bool }` and answers `{ "id", "active", "updated_at" }`. +- `runs` takes `limit` (default 50, at most 200) and `cursor` (a run id; returns older runs) and answers `{ "runs": [...] }`. A run has a `status` (`active`, `paused_by_maintenance`, `stopped_by_ack`, `stopped_by_resolution`, `stopped_by_policy_deletion`, `stopped_by_policy_disabled`, `stopped_by_edition_downgrade` or `exhausted`); its deliveries have a `status` of `pending`, `sent`, `failed` or `abandoned`. Deactivating a policy (`PUT` with `active` false, or `PATCH .../active`) stops its running runs with `stopped_by_policy_disabled`. +- Errors (lower case codes): `400 validation_failed`, `400 invalid_body`, `404 policy_not_found`, `404 run_not_found` and `500 internal_error`. --- ## Silence Rules +All routes are open in every edition. + | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/silence` | List active silence rules | +| `GET` | `/api/v1/silence` | List silence rules (`?active=true` keeps the ones in effect) | | `POST` | `/api/v1/silence` | Create a silence rule | | `DELETE` | `/api/v1/silence/{id}` | Cancel a silence rule | +- `POST` body: `duration_seconds` (required, above zero) and optional `entity_type`, `entity_id`, `source` and `reason`. All three filters empty silence everything; otherwise an alert must match every filter that is set. Answers `201` with the rule. +- The list answers `{ "rules": [...] }`. A rule has `id`, `entity_type`, `entity_id`, `source`, `reason`, `starts_at`, `duration_seconds`, `expires_at`, `is_active`, `cancelled_at` and `created_at`. +- `DELETE` cancels the rule (`is_active` becomes `false`, the rule stays listed) and answers `204`, even for an unknown id. +- Errors: `400 INVALID_BODY`, `400 VALIDATION_ERROR`. + --- ## Webhooks +Webhook subscriptions deliver raw events to a URL. To notify a chat service, use a [channel](#notification-channels) instead: a webhook URL of Discord, Slack or Teams is refused. All routes are open in every edition. + | Method | Endpoint | Description | |--------|----------|-------------| | `GET` | `/api/v1/webhooks` | List webhook subscriptions | @@ -236,6 +546,12 @@ Endpoints return `403 edition_required` on Community. | `DELETE` | `/api/v1/webhooks/{id}` | Delete a webhook subscription | | `POST` | `/api/v1/webhooks/{id}/test` | Send a test payload | +- `POST` body: `name` (required, 1 to 100 characters), `url` (required, HTTPS, not resolving to a private or internal address unless `MAINTENANT_ALLOW_PRIVATE_WEBHOOKS` is set), `secret` (optional, enables the signature) and `event_types` (default `["*"]`). Valid event types are `*`, `container.state_changed`, `endpoint.status_changed`, `heartbeat.status_changed`, `certificate.status_changed`, `alert.fired` and `alert.resolved`. Answers `201` with the subscription (the secret is never returned). +- The list answers `{ "webhooks": [...] }`. A subscription has `id`, `name`, `url`, `event_types`, `is_active`, `last_delivery_status` (`delivered` or `failed`), `last_delivery_at`, `failure_count` (consecutive failures) and `created_at`. +- After 10 consecutive failed deliveries the subscription is deactivated (`is_active` becomes `false`). A successful delivery, a test included, resets `failure_count` and reactivates it. +- `test` sends `{"type": "test", "timestamp": "...", "data": {"message": "maintenant webhook test"}}` synchronously, with the headers and the signature of a real delivery, and records the outcome like one. It answers `200` with `{ "status": "delivered", "http_status": n }` or, on failure, `{ "status": "failed", "error": "..." }` plus `http_status` when the target answered. +- Errors use lower case codes: `400 invalid_json`, `400 invalid_input`, `404 not_found` and `500 internal_error`. + ### Delivery format Each delivery is a `POST` with a JSON body of the shape `{type, timestamp, data}`, where `data` is the raw event payload for that `type`: @@ -255,7 +571,7 @@ Each delivery is a `POST` with a JSON body of the shape `{type, timestamp, data} } ``` -The `data` fields depend on `type` (`container.state_changed`, `endpoint.status_changed`, `heartbeat.status_changed`, `certificate.status_changed`, `alert.fired`, `alert.resolved`). Subscribe to specific types or `*` for all. +The `data` fields are those of the [SSE event](#sse-event-stream) of the same type. A subscription to `container.state_changed` also receives the `container.discovered` payloads, under that type, and a subscription to `endpoint.status_changed` also receives the `endpoint.discovered` and `endpoint.removed` payloads. Subscribe to specific types or `*` for all. A delivery is attempted up to three times, with a 1 second then a 5 second wait between attempts. The events sent to one URL are delivered one at a time in the order they happened, so a retry never lets a later event overtake an earlier one. Headers on every delivery: @@ -263,7 +579,7 @@ Headers on every delivery: |--------|-------| | `X-maintenant-Event` | the event `type` | | `X-maintenant-Delivery` | a unique delivery UUID | -| `X-maintenant-Signature` | `sha256=` — present only when the subscription has a secret | +| `X-maintenant-Signature` | `sha256=`, present only when the subscription has a secret | When a secret is set, verify authenticity by computing `HMAC-SHA256(secret, raw_request_body)` and comparing (constant-time) against the hex digest in `X-maintenant-Signature`. The signature is computed over the exact bytes of the request body. @@ -274,122 +590,248 @@ When a secret is set, verify authenticity by computing `HMAC-SHA256(secret, raw_ ## Status Page (Admin) +The admin routes of the public status page. See [Status Page](../features/status-page.md). The routes below sit under `/api/v1/status/`; the public page is served under `/status/` (see [Public status page](#public-status-page)). + ### Components -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/status/components` | List components | -| `POST` | `/api/v1/status/components` | Create a component | -| `PUT` | `/api/v1/status/components/{id}` | Update a component | -| `DELETE` | `/api/v1/status/components/{id}` | Delete a component | +All routes are open in every edition, except that creating a component is capped. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/status/components` | List components | — | +| `POST` | `/api/v1/status/components` | Create a component (3 on Community) | — | +| `PUT` | `/api/v1/status/components/{id}` | Update a component | — | +| `DELETE` | `/api/v1/status/components/{id}` | Delete a component | — | -### Incidents :material-star-four-points:{ title="Personal" } +The list is a bare JSON array (`null` when empty), ordered by `display_order`. A component has `id`, `composition_mode` (`explicit` or `match-all`), `monitors` (`[{ "type", "id" }]`), `match_all_type`, `display_name`, `display_order`, `visible`, `derived_status`, `status_override`, `effective_status`, `auto_incident`, `needs_attention` and timestamps. Statuses are `operational`, `degraded`, `partial_outage`, `major_outage` and `under_maintenance`. -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/status/incidents` | List all incidents | -| `POST` | `/api/v1/status/incidents` | Create an incident | -| `PUT` | `/api/v1/status/incidents/{id}` | Update an incident | -| `DELETE` | `/api/v1/status/incidents/{id}` | Delete an incident | -| `POST` | `/api/v1/status/incidents/{id}/updates` | Add an incident update | +- `POST` body: `display_name` (required), `composition_mode` (default `explicit`), `monitors` (an `explicit` component needs at least one, each of type `container`, `endpoint`, `heartbeat` or `certificate`), `match_all_type` (required in `match-all` mode, empty otherwise), `display_order`, `visible` (default `true`) and `auto_incident`. `403 QUOTA_EXCEEDED` when the cap is reached (`resource: "status_components"`, `limit: 3` on Community). +- `PUT` takes the same fields, all optional. `composition_mode` and `match_all_type` cannot change, a `match-all` component's monitors cannot be edited, and `status_override` set to an empty string clears the override. A `status_override` must be one of the five statuses above, otherwise the answer is `400 validation`. While a maintenance window runs it forces `under_maintenance` on its components but remembers the override an operator had set: that override comes back when the last window holding the component ends. +- `DELETE` answers `204`. +- Creating, updating or deleting a component emits `status.component_created`, `status.component_updated` or `status.component_deleted` on the dashboard stream, and on the public stream followed by `status.global_changed` when the component is visible (or was visible before the update or delete), so open pages refresh. A hidden component never reaches the public stream. +- Errors (lower case codes): `400 invalid_body`, `400 validation`, `404 not_found` and `500 internal`. -### Maintenance Windows :material-crown:{ title="Pro" } +### Incidents -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/status/maintenance` | List maintenance windows | -| `POST` | `/api/v1/status/maintenance` | Schedule a maintenance window | -| `PUT` | `/api/v1/status/maintenance/{id}` | Update a maintenance window | -| `DELETE` | `/api/v1/status/maintenance/{id}` | Delete a maintenance window | +Reading is open in every edition. Writing requires **Personal** (capability `incidents`). -### Subscribers :material-crown:{ title="Pro" } +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/status/incidents` | List all incidents | — | +| `POST` | `/api/v1/status/incidents` | Create an incident | Personal | +| `PUT` | `/api/v1/status/incidents/{id}` | Update an incident's title, severity or components | Personal | +| `DELETE` | `/api/v1/status/incidents/{id}` | Delete an incident | Personal | +| `POST` | `/api/v1/status/incidents/{id}/updates` | Add an incident update | Personal | -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/status/subscribers` | List email subscribers | +- `GET` takes `status`, `severity`, `limit` (1 to 100, default 20) and `offset`, and answers `{ "incidents": [...], "total" }`. An incident has `id`, `title`, `severity`, `status`, `is_maintenance`, `components`, `updates` and timestamps. +- `POST` body: `title` and `severity` (`minor`, `major` or `critical`) are required; `status` (`investigating`, `identified`, `monitoring` or `resolved`) defaults to `investigating`; `component_ids` and `message` (the first update) are optional. It emits `status.incident_created` and emails the confirmed subscribers when subscriptions are open. +- `PUT` takes `title`, `severity` and `component_ids`, all optional. Leaving `component_ids` out keeps the linked components, and an empty list removes them. +- `POST .../updates` takes `status` (one of the four statuses above) and `message` (both required). An update with status `resolved` resolves the incident and emits `status.incident_resolved`; any other update emits `status.incident_updated`. +- A `component_ids` list that holds an empty, unknown or repeated id is refused with `400 validation` before anything is written, on create and on update. +- Errors (lower case codes): `400 validation` (a missing field, a value outside the lists above or an invalid component id), `404 not_found`. + +### Maintenance windows + +Reading is open in every edition. Writing requires **Pro** (capability `maintenance_windows`). + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/status/maintenance` | List maintenance windows | — | +| `POST` | `/api/v1/status/maintenance` | Schedule a maintenance window | Pro | +| `PUT` | `/api/v1/status/maintenance/{id}` | Update a maintenance window | Pro | +| `DELETE` | `/api/v1/status/maintenance/{id}` | Delete a maintenance window | Pro | + +- `GET` takes `status` (`upcoming`, `active` or `completed`) and `limit` (1 to 100, default 20) and answers a bare array (`null` when empty). Without `status`, running windows come first, then upcoming ones (soonest first), then completed ones (latest first). +- `POST` body: `title`, `starts_at` and `ends_at` (RFC 3339, `ends_at` not before `starts_at`) are required; `description` and `component_ids` are optional. An empty, unknown or repeated component id answers `400 validation`. +- `PUT` takes the same fields, all optional (leaving `component_ids` out keeps the linked components), and answers `400 validation` when the resulting `ends_at` (the new one, or the stored one when it is left out) is before `starts_at`, and `409 conflict` while the window is active. A scheduler checks the windows every 60 seconds: starting one opens a "Scheduled Maintenance" incident, puts its components under maintenance and emits `status.maintenance_started`; ending it emits `status.maintenance_ended`. When several windows hold the same component, it stays under maintenance until the last one ends. +- `DELETE` on a running window ends it the way its scheduled end would: the incident is resolved, the components are released and `status.maintenance_ended` is emitted. + +### Subscribers + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/status/subscribers` | List email subscribers (addresses masked) | Pro | + +Answers `{ "subscribers": [{ "id", "email", "confirmed", "created_at" }], "total", "confirmed" }`. + +### SMTP test + +Status page emails are sent through the SMTP server configured with the `MAINTENANT_SMTP_*` variables. There is no route to read or change that configuration. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `POST` | `/api/v1/status/smtp/test` | Send a test email | Personal | + +Body: `{ "to": "address" }`. Answers `200 {"status": "sent"}`. Errors: `400 not_configured` (no SMTP host), `400 invalid_body`, `400 validation` (invalid address) and `502 smtp_failed` (the message is the SMTP error). + +### Personalization + +Branding of the public page. Every route requires **Pro** (capability `personalization`), reads included, and answers `403 EDITION_REQUIRED` below it. Errors use the codes `validation_error`, `payload_too_large`, `unsupported_mime`, `active_svg`, `not_found` and `internal_error`. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/status-page/settings` | Get the settings | Pro | +| `PUT` | `/api/v1/status-page/settings` | Replace the settings | Pro | +| `GET` | `/api/v1/status-page/assets/{role}` | Download an asset | Pro | +| `PUT` | `/api/v1/status-page/assets/{role}` | Upload an asset (multipart, part `file`, optional `alt_text`) | Pro | +| `DELETE` | `/api/v1/status-page/assets/{role}` | Delete an asset | Pro | +| `GET` | `/api/v1/status-page/footer-links` | List footer links | Pro | +| `POST` | `/api/v1/status-page/footer-links` | Add a footer link | Pro | +| `PUT` | `/api/v1/status-page/footer-links/order` | Reorder footer links (`{ "ids": [...] }`) | Pro | +| `PUT` | `/api/v1/status-page/footer-links/{id}` | Update a footer link | Pro | +| `DELETE` | `/api/v1/status-page/footer-links/{id}` | Delete a footer link | Pro | +| `GET` | `/api/v1/status-page/faq` | List FAQ items | Pro | +| `POST` | `/api/v1/status-page/faq` | Add an FAQ item | Pro | +| `PUT` | `/api/v1/status-page/faq/order` | Reorder FAQ items (`{ "ids": [...] }`) | Pro | +| `PUT` | `/api/v1/status-page/faq/{id}` | Update an FAQ item | Pro | +| `DELETE` | `/api/v1/status-page/faq/{id}` | Delete an FAQ item | Pro | + +- **Settings** (`PUT` replaces every field): `title` (1 to 100 characters), `subtitle` (200), `colors` (`bg`, `surface`, `border`, `text`, `accent`, `status_operational`, `status_degraded`, `status_partial`, `status_major`, each `#RRGGBB` or `#RRGGBBAA`), `announcement` (`enabled`, `message_md` up to 1000 characters, `url` over HTTP or HTTPS), `footer_text_md` (500), `locale` (`en` or `fr`), `timezone` (an IANA name or empty) and `date_format` (`relative` or `absolute`). Markdown is rendered to sanitized HTML on the server. The answer adds `version` and, when the palette misses WCAG AA contrast, a `warnings.contrast` list that never blocks the save. +- **Assets**: `role` is `logo` (200 KiB; PNG, JPEG, WebP or SVG), `favicon` (50 KiB; PNG, ICO or SVG) or `hero` (500 KiB; PNG, JPEG or WebP). The type is sniffed from the content. An SVG is read whole and needs no XML declaration, but one that carries a script, an event handler, a `javascript:` link, a `foreignObject` or embedded HTML is refused. An oversized file answers `400 payload_too_large`, a wrong type `400 unsupported_mime` and an SVG with active content `400 active_svg`. +- **Footer links** take `label` (1 to 60 characters) and `url` (HTTP or HTTPS). **FAQ items** take `question` (1 to 200 characters) and `answer_md` (up to 4000). Lists answer `{ "items": [...] }`. + +--- + +## Public status page -### SMTP Configuration :material-star-four-points:{ title="Personal" } +These routes need no credentials, are not under `/api/`, share the 10 requests per second bucket of the public surfaces and cap bodies at 4 KiB. The page can be framed from any origin. See [Status Page](../features/status-page.md). | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/status/smtp` | Get SMTP configuration | -| `PUT` | `/api/v1/status/smtp` | Update SMTP configuration | -| `POST` | `/api/v1/status/smtp/test` | Send a test email | +| `GET` | `/status/` | The status page (the dashboard application at the status route). `/status` redirects to it | +| `GET` | `/status/api` | JSON snapshot of the current status | +| `GET` | `/status/events` | Event stream of the status page (SSE) | +| `GET` | `/status/feed.atom` | Atom feed of the ongoing incidents and of those resolved in the last 30 days | +| `GET` | `/status/settings.json` | Branding for the page: colors, assets, footer, FAQ | +| `POST` | `/status/subscribe` | Subscribe an email address to updates | +| `GET` | `/status/confirm?token=` | Confirm a subscription | +| `GET` | `/status/unsubscribe?token=` | Unsubscribe | + +- `GET /status/api` answers `global_status`, `global_message`, `updated_at`, `components` (`[{ "id", "name", "status", "monitors" }]`, visible components only; `monitors` is `[{ "type", "id", "name", "status" }]`), `active_incidents`, `upcoming_maintenance` (the next five windows, soonest first), `subscriptions_enabled` and, once the page has been personalized, `personalization_version`. List fields are `null` when empty. It sends `Access-Control-Allow-Origin: *`. +- Incident and maintenance `components` lists on the public surfaces (this route, the stream, the Atom feed and the notification emails) name visible components only: a hidden component is never named, and an incident opened automatically by a hidden component's alert is not created. The Atom feed builds its links from `MAINTENANT_STATUS_URL`, or from `MAINTENANT_BASE_URL` followed by `/status` when it is not set, never from the `Host` header of the request. +- `GET /status/settings.json` returns the branding with images inlined as `data:` URLs (there is no public asset URL). Below Pro it returns the default settings. It sends an `ETag` (`"v"`), answers `304` to a matching `If-None-Match` and sends `Access-Control-Allow-Origin: *`. +- `POST /status/subscribe` takes `{ "email": "..." }` (`Content-Type: application/json`) or a form field `email` (`application/x-www-form-urlencoded`). Any other content type answers `415 unsupported_media_type`. Subscriptions are open when SMTP is configured and the edition is **Pro**; otherwise the route answers `503 subscriptions_unavailable`. The answer is always `200 {"status": "confirmation_sent"}`, whether the address is new, pending or already confirmed, so nothing reveals who is subscribed. A pending address gets a new link, which replaces the old one, and a confirmed address gets no mail. Errors: `400 invalid_email`, `400 invalid_body`, `413 body_too_large`, `429 rate_limited` (5 per hour and per IP, with `Retry-After`) and `500 subscription_failed`. +- `confirm` and `unsubscribe` answer small HTML pages. The link in the confirmation mail is valid for 24 hours. `confirm` answers `503` while subscriptions are closed; `unsubscribe` keeps working. +- Plain-text errors on the page, API and feed routes: `503 Status page not available` when the application is not embedded and `500 Internal Server Error`. --- ## Updates +All routes are open in every edition. Two response fields depend on the edition: `cve_counts` in the summary and the changelog and CVE fields of a container's update (Personal). + | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/updates` | List available updates (`?status=&update_type=`) | +| `GET` | `/api/v1/updates` | List updates (`?status=&update_type=`) | | `GET` | `/api/v1/updates/summary` | Update summary with counts (`os_counts` for host operating systems) | | `GET` | `/api/v1/updates/hosts` | Every monitored host with its OS identity and end-of-support status, plus the support table in use | | `POST` | `/api/v1/updates/scan` | Trigger a manual scan | | `GET` | `/api/v1/updates/scan/{scan_id}` | Get scan status | -| `GET` | `/api/v1/updates/dry-run` | Preview what a scan would check | +| `GET` | `/api/v1/updates/dry-run` | Dry run: list the available updates | | `GET` | `/api/v1/updates/container/{container_id}` | Update details for a container | -| `POST` | `/api/v1/updates/pin/{container_id}` | Pin current version | -| `DELETE` | `/api/v1/updates/pin/{container_id}` | Unpin version | +| `POST` | `/api/v1/updates/pin/{container_id}` | Pin the current version | +| `DELETE` | `/api/v1/updates/pin/{container_id}` | Unpin the version | | `GET` | `/api/v1/updates/exclusions` | List exclusions | | `POST` | `/api/v1/updates/exclusions` | Create an exclusion | | `DELETE` | `/api/v1/updates/exclusions/{id}` | Delete an exclusion | +`{container_id}` is the container's external id (the runtime id; for Kubernetes `namespace/Kind/name`, slashes included). + +- `GET /api/v1/updates` takes `status` (`available` or `pinned`) and `update_type` (`major`, `minor`, `patch`, `digest_only` or `unknown`). The answer is `{ "updates": [...], "last_scan", "next_scan" }` (both dates are the zero time until a scan finishes). An update has `id`, `container_id`, `container_name`, `image`, `current_tag`, `current_digest`, `latest_tag`, `latest_digest`, `update_type`, `risk_score`, `status`, `detected_at`, and `pin_reason` when pinned. `risk_score` (Personal and above) adds up at most 75 points: update type (up to 20), CVE severity (30), network exposure (10, from the container's security insights), restarts over the last 24 hours (10) and breaking changes (5). Container criticality and dependents no longer count. +- `summary` answers `last_scan`, `next_scan`, `scan_status` (`running`, `completed`, `failed` or `idle`), `counts` (`critical`, `recommended`, `available`, `up_to_date`, `pinned`), `cve_counts` (Personal) and `os_counts` (`ended`, `ending_soon`, `unknown`, `untracked`, `supported`). +- `hosts` answers `{ "hosts": [...], "eol_table": { "source", "fetched_at", "refresh_enabled", "last_refresh_error" } }`. A host has `agent_id`, `hostname`, `label`, `is_local`, `runtime`, `connection_state` and an `os` object with its `support` state (`unknown`, `untracked`, `supported`, `security_only`, `ending_soon` or `ended`). +- `scan` answers `202 {"status": "running", "started_at": "..."}` without a scan id: the id arrives in the `update.scan_started` event. `409 SCAN_IN_PROGRESS` when a scan is running. +- `scan/{scan_id}` answers `scan_id`, `status`, `started_at`, `containers_scanned`, `updates_found`, `errors` and `completed_at`; `404 NOT_FOUND` for an unknown scan. +- `dry-run` answers `{ "would_update": [{ "container_id", "container_name", "image", "current_tag", "latest_tag", "update_type" }] }`, the updates with status `available`. +- `container/{container_id}` adds `pinned`, `pin_reason`, `update_command`, `rollback_command`, `tag_include` and `tag_exclude`. The commands are shell commands that depend on the workload: Compose or standalone Docker commands, `docker service update --image ` for a Swarm task, `kubectl set image` for a Kubernetes workload or a bare pod. For a Swarm task or a Kubernetes workload whose tag was republished under the same name, the command sets `@`, because setting an unchanged reference rolls nothing out. On Personal and above it also carries `source_url`, `previous_digest`, `changelog_url`, `changelog_summary`, `has_breaking_changes` and `active_cves`. Errors: `404 NOT_FOUND`. +- `pin` takes an optional `{ "reason": "..." }` and answers `200 { "container_id", "pinned_tag", "pinned_digest", "reason", "pinned_at" }`; `404 NOT_FOUND` when the container has no update data. `DELETE` always answers `204`. +- `exclusions` (POST) takes `pattern` (required) and `pattern_type` (`image` or `tag`), and answers `201`. Posting a pattern that already exists answers `200` with the stored exclusion. Errors: `400 INVALID_JSON`, `400 INVALID_PATTERN`, `400 INVALID_TYPE`. + --- ## Security -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/security/insights` | List all network security insights | -| `GET` | `/api/v1/security/insights/{container_id}` | Get insights for a specific container | -| `GET` | `/api/v1/security/summary` | Aggregated counts by severity and type | +Insights are open in every edition. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/security/insights` | List all insights, by container | — | +| `GET` | `/api/v1/security/insights/{container_id}` | Insights of a container | — | +| `GET` | `/api/v1/security/summary` | Aggregated counts by severity and type | — | -### Security Posture :material-star-four-points:{ title="Personal" } +The list answers `{ "containers": [...], "summary": {...} }` and the summary route answers the same `summary` object (`total_containers_monitored`, `total_containers_affected`, `total_insights`, `by_severity`, `by_type`). An insight has `type` (`port_exposed_all_interfaces`, `database_port_exposed`, `privileged_container`, `host_network_mode`, `service_load_balancer` or `service_node_port`), `severity` (`critical`, `high` or `medium`), `container_id`, `container_name`, `title`, `description`, `details` and `detected_at`. `{container_id}` is the maintenant container id; an unknown one answers `404 CONTAINER_NOT_FOUND`. -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/security/posture` | Global infrastructure posture score | -| `GET` | `/api/v1/security/posture/containers` | Per-container posture scores (`?limit=&offset=`) | -| `GET` | `/api/v1/security/posture/containers/{id}` | Posture score for a single container | -| `POST` | `/api/v1/security/acknowledgments` | Acknowledge a finding | -| `DELETE` | `/api/v1/security/acknowledgments/{id}` | Revoke an acknowledgment | -| `GET` | `/api/v1/security/acknowledgments` | List acknowledgments (`?container_id=`) | +### Security Posture + +Requires **Personal** (capability `security_posture`). Every route answers `403 EDITION_REQUIRED` below it. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/security/posture` | Global infrastructure posture score | Personal | +| `GET` | `/api/v1/security/posture/containers` | Per-container posture scores (`?limit=&offset=`) | Personal | +| `GET` | `/api/v1/security/posture/containers/{container_id}` | Posture score for one container | Personal | +| `POST` | `/api/v1/security/acknowledgments` | Acknowledge a finding | Personal | +| `GET` | `/api/v1/security/acknowledgments` | List acknowledgments (`?container_id=`) | Personal | +| `DELETE` | `/api/v1/security/acknowledgments/{id}` | Revoke an acknowledgment | Personal | + +- The global score has `score`, `color` (`green` from 80, `yellow` from 60, `orange` from 40, else `red`), `container_count`, `scored_count`, `is_partial`, `categories` (`tls`, `cves`, `updates`, `network_exposure`, `image_age`), `top_risks` and `computed_at`. Every scoring of the infrastructure evaluates the posture threshold alert (`MAINTENANT_SECURITY_SCORE_THRESHOLD`): a read of this route, the MCP posture tool, and, when a threshold is set, a background check every 5 minutes. +- `containers` takes `limit` (default 50) and `offset` and answers `{ "containers", "total", "limit", "offset" }`, worst score first. All containers are scored in one batch (certificates and updates are read once, not once per container); a scoring failure answers `500 INTERNAL_ERROR` instead of dropping containers from the list. +- `POST acknowledgments` takes `container_id` and `finding_type` (required), `finding_key`, `acknowledged_by` and `reason`. Answers `201`. Errors: `400 INVALID_BODY`, `400 MISSING_FIELDS`, `404 NOT_FOUND`, `409 ALREADY_ACKNOWLEDGED`. --- -## CVE Intelligence :material-star-four-points:{ title="Personal" } +## CVE Intelligence -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/cve` | List known CVEs across all containers | -| `GET` | `/api/v1/cve/{container_id}` | List CVEs for a specific container | +Requires **Personal** (capability `cve_enrichment`). The check is made by the handler, which answers `403 EDITION_REQUIRED` below it. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/cve` | List known CVEs across all containers (`?severity=&container_id=`) | Personal | +| `GET` | `/api/v1/cve/{container_id}` | List CVEs for a specific container | Personal | + +`{container_id}` is the external id. The list answers `{ "cves": [{ "cve_id", "cvss_score", "severity", "summary", "first_detected_at", "affected_containers" }], "total", "by_severity" }`; `total` and `by_severity` count container-CVE pairs, not distinct CVEs. The container route answers `{ "container_id", "cves": [...] }`. --- -## Risk Scoring :material-star-four-points:{ title="Personal" } +## Risk Scoring -| Method | Endpoint | Description | -|--------|----------|-------------| -| `GET` | `/api/v1/risk` | Risk scores for all containers | -| `GET` | `/api/v1/risk/{container_id}` | Risk score for a specific container | -| `GET` | `/api/v1/risk/{container_id}/history` | Risk score history | +Requires **Personal** (capability `risk_scoring`). Every route answers `403 EDITION_REQUIRED` below it. Error codes here are lower case. + +| Method | Endpoint | Description | Edition | +|--------|----------|-------------|:-------:| +| `GET` | `/api/v1/risk` | Risk scores for all containers | Personal | +| `GET` | `/api/v1/risk/{container_id}` | Risk score of a container; with `?period=24h\|7d\|30d`, its score history | Personal | + +The list answers `{ "containers": [{ "container_id", "container_name", "risk_score", "level" }], "host_risk_score", "host_risk_level" }`. `level` is `critical` from 81, `high` from 61, `moderate` from 31, else `low`. With `period`, the container route answers `{ "container_id", "history": [{ "score", "recorded_at" }] }`. Errors: `400 invalid_period`, `404 not_found` (no risk data), `500 internal`. --- -## License +## Dashboard | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/license/status` | Current license status and edition info | +| `GET` | `/api/v1/dashboard/sparklines` | Response times of the last 20 checks of every active endpoint | + +Answers an object keyed `endpoint:`, each value an array of response times in milliseconds, oldest first. Open in every edition. --- -## Dashboard +## MCP and OAuth routes + +The MCP server is off unless `MAINTENANT_MCP` is enabled, and none of these routes exist in demo mode. When `MAINTENANT_MCP_CLIENT_ID` and `MAINTENANT_MCP_CLIENT_SECRET` are set, `/mcp` requires an OAuth access token and the `/.well-known/` and `/oauth/` routes exist. Without them the instance refuses to start, unless `MAINTENANT_MCP_ALLOW_UNAUTHENTICATED` leaves `/mcp` open, in which case only `/mcp` is served. See [MCP Server](../features/mcp.md). | Method | Endpoint | Description | |--------|----------|-------------| -| `GET` | `/api/v1/dashboard/sparklines` | Sparkline data for all endpoints | +| `GET`, `POST`, `DELETE` | `/mcp`, `/mcp/` | MCP over streamable HTTP. Needs `Authorization: Bearer ` when OAuth is configured | +| `GET` | `/.well-known/oauth-authorization-server` | Authorization server metadata | +| `GET` | `/.well-known/oauth-protected-resource` | Protected resource metadata for `/mcp` | +| `GET` | `/oauth/authorize` | Authorization endpoint (authorization code with PKCE) | +| `POST` | `/oauth/token` | Token endpoint | + +- `/oauth/authorize` takes `response_type=code`, `client_id`, `redirect_uri`, `code_challenge`, `code_challenge_method=S256` and an optional `state`. It asks for no login: a valid request is approved at once and redirected with a `code`. The redirect URI must be a loopback address or one of `MAINTENANT_MCP_ALLOWED_REDIRECT_URIS`. The code lives 10 minutes and works once. +- `/oauth/token` takes form fields. `grant_type=authorization_code` needs `code`, `client_id`, `client_secret`, `redirect_uri` and `code_verifier`; `grant_type=refresh_token` needs `refresh_token`, `client_id` and `client_secret`. The client authenticates with its secret in the form (`client_secret_post`). The answer is `{ "access_token", "token_type": "Bearer", "expires_in": 3600, "refresh_token" }`: access tokens last 1 hour, refresh tokens 30 days, and each refresh token is used once. +- OAuth errors use the RFC 6749 shape: `invalid_request`, `invalid_client` (401), `invalid_grant`, `unsupported_grant_type`, `unsupported_response_type`, `unauthorized_client` and `server_error`. `/mcp` answers `401` with a `WWW-Authenticate` header pointing to the protected resource metadata when the token is missing, expired or revoked. +- The issuer of the metadata is `MAINTENANT_BASE_URL`. --- @@ -401,36 +843,114 @@ Connect to the real-time event stream: GET /api/v1/containers/events ``` -This is a Server-Sent Events (SSE) endpoint. Each event has a `type` field and a JSON `data` payload. +This is a Server-Sent Events (SSE) endpoint. The server writes each event as: + +``` +event: container.state_changed +data: {"id":"a1b2c3d4e5f6","state":"running",...} +``` + +The SSE `event:` field is the event type, and `data:` holds the JSON payload only (there is no `{type, data}` envelope). There is no `id:` field and no event on connection: a client only receives events broadcast after it connects, and refetches state after a reconnect. While the stream is silent, the server sends the comment `: keepalive` every 25 seconds so that a proxy does not cut it; clients ignore comments. The `status.*` events of the status page are sent on this stream as well. A client that falls 64 events behind loses events instead of slowing the others. The response sets `Cache-Control: no-cache` and `X-Accel-Buffering: no`. ### Event Types -| Event | Source | Description | -|-------|--------|-------------| -| `container.state_changed` | Container | State transition (running, stopped, etc.) | -| `container.health_changed` | Container | Health check status change | -| `container.restart_alert` | Container | Restart loop detected | -| `endpoint.check_result` | Endpoint | Check completed (up/down, response time) | -| `endpoint.alert` | Endpoint | Consecutive failure threshold reached | -| `endpoint.recovery` | Endpoint | Endpoint recovered | -| `heartbeat.pinged` | Heartbeat | Ping received | -| `heartbeat.deadline_missed` | Heartbeat | Missed deadline | -| `certificate.alert` | Certificate | Expiry warning | -| `certificate.recovery` | Certificate | Certificate renewed | -| `resource.snapshot` | Resource | New metrics snapshot | -| `resource.alert` | Resource | Threshold exceeded | -| `resource.recovery` | Resource | Usage returned to normal | -| `update.scan_started` | Update | Scan in progress | -| `update.scan_completed` | Update | Scan finished | -| `update.detected` | Update | New update found | -| `update.pinned` | Update | Version pinned | -| `update.unpinned` | Update | Version unpinned | -| `storage.availability_changed` | Storage | Database became unreachable, or answered again (`{engine, connected}`) | - -### Status Page SSE - -The public status page has its own event stream: +| Event | Emitted when | Payload keys | +|-------|--------------|--------------| +| `container.discovered` | A container the store did not know appears (not for ignored containers) | the container object | +| `container.state_changed` | A container changes state | `id`, `state`, `previous_state`, `health_status`, `exit_code`, `timestamp`, `agent_id` | +| `container.health_changed` | A container's health status changes | `id`, `container_name`, `health_status`, `previous_health`, `timestamp`, `agent_id` | +| `container.archived` | A container is destroyed, vanished while offline, missing from an agent inventory or deleted through the API | `id`, `archived_at`, `agent_id` (the API delete sends `id` only) | +| `container.restart_alert` | A restart loop is detected | `container_id`, `container_name`, `restart_count`, `threshold`, `severity`, `timestamp`, `agent_id` | +| `container.restart_recovery` | The restart rate is back under the threshold | `container_id`, `container_name`, `timestamp`, `agent_id` | +| `endpoint.discovered` | An endpoint is discovered from labels or created through the API | `endpoint_id`, `container_name`, `endpoint_type`, `target`, plus `agent_id` for an agent's endpoint and `source` (`standalone`) and `name` for a standalone one | +| `endpoint.status_changed` | An endpoint changes status after a check, or becomes `unknown` because its container stopped | `endpoint_id`, `container_name`, `target`, `previous_status`, `new_status`, `error`, `timestamp`, plus `response_time_ms`, `http_status` and `agent_id` after a check | +| `endpoint.removed` | An endpoint is deactivated or deleted | `endpoint_id`, `reason` (`label_removed`, `container_destroyed`, `container_gone` or `user_deleted`), plus `container_name` and `agent_id` when they apply | +| `endpoint.alert` | An endpoint reaches its failure threshold | `endpoint_id`, `container_name`, `target`, `consecutive_failures`, `threshold`, `last_error`, `timestamp` | +| `endpoint.recovery` | An alerting endpoint reaches its recovery threshold | `endpoint_id`, `container_name`, `target`, `consecutive_successes`, `threshold`, `timestamp` | +| `endpoint.config_error` | An endpoint label cannot be parsed | `endpoint_id` (`null`), `container_name`, `label_key`, `error`, `timestamp`, plus `agent_id` for an agent's container | +| `heartbeat.created` | A heartbeat monitor is created | `heartbeat_id`, `name`, `status` | +| `heartbeat.ping_received` | A ping arrives (`ping_type`: `success`, `start` or `exit_code`) | `heartbeat_id`, `ping_type`, `status`, plus `exit_code` or `agent_id` | +| `heartbeat.status_changed` | A heartbeat changes status (ping, missed deadline, pause, resume) | `heartbeat_id`, `old_status`, `new_status`, `agent_id` on a ping | +| `heartbeat.alert` | A deadline is missed or a ping carries a non-zero exit code | `heartbeat_id`, `name`, `alert_type`, `details` | +| `heartbeat.recovery` | A failing heartbeat recovers | `heartbeat_id`, `name`, `agent_id` on a ping | +| `heartbeat.deleted` | A heartbeat monitor is deleted | `heartbeat_id` | +| `certificate.created` | A certificate monitor is created or auto-detected | `monitor_id`, `hostname`, `port`, `source`, plus `server_name` and `agent_id` when they apply | +| `certificate.check_completed` | A check finishes, including failed ones | `monitor_id`, `hostname`, `status`, `checked_at`, plus `subject_cn`, `issuer_cn`, `not_after`, `days_remaining`, `chain_valid`, `hostname_match` when known | +| `certificate.status_changed` | A monitor changes status | `monitor_id`, `hostname`, `previous_status`, `new_status`, `days_remaining`, `timestamp` | +| `certificate.alert` | Expiry threshold, invalid chain, hostname mismatch, revoked OCSP response or expiry | `monitor_id`, `hostname`, `port`, `alert_type`, `severity`, `timestamp` and details | +| `certificate.recovery` | A certificate alert clears: the certificate was renewed, or the chain, hostname or OCSP problem is gone | `monitor_id`, `hostname`, `port`, `previous_alert_type` (`expiring`, `expired`, `chain_invalid`, `hostname_mismatch` or `ocsp_revoked`), `days_remaining`, `timestamp`, plus `new_not_after` and `server_name` when they apply | +| `certificate.deleted` | A certificate monitor is deleted | `monitor_id`, `hostname` | +| `resource.snapshot` | A sample is stored (live samples only) | `container_id`, `cpu_percent`, `mem_used`, `mem_limit`, `mem_percent`, `net_rx_bytes`, `net_tx_bytes`, `block_read_bytes`, `block_write_bytes`, `timestamp`, `agent_id` | +| `resource.alert` | CPU or memory stays over its threshold (one event per metric) | `container_id`, `container_name`, `alert_type` (`cpu` or `memory`), `current_value`, `threshold`, `timestamp` | +| `resource.recovery` | CPU or memory returns to normal (one event per metric) | `container_id`, `container_name`, `recovered_type` (`cpu` or `memory`), `current_value`, `threshold`, `timestamp` | +| `alert.fired` | An alert is raised, or its severity escalates | the alert: `id`, `source`, `alert_type`, `severity`, `status`, `message`, `entity_type`, `entity_id`, `entity_name`, `details` (an object), `fired_at`, `created_at` | +| `alert.silenced` | An alert is raised while a silence rule or a maintenance window matches | same as `alert.fired` | +| `alert.resolved` | An alert resolves | same as `alert.fired`, with `resolved_at` | +| `alert.acknowledged` | An alert is acknowledged, from the REST route, the MCP server or a security posture acknowledgment | the alert as stored: `details` is a string holding JSON | +| `channel.created`, `channel.updated` | A channel changes (REST or MCP) | the channel object | +| `channel.deleted` | A channel is deleted | `id` | +| `trigger.created`, `trigger.updated` | A trigger changes | the trigger object | +| `trigger.deleted` | A trigger is deleted | `id` | +| `silence.created` | A silence rule is created | the rule object | +| `silence.cancelled` | A silence rule is cancelled | `id` | +| `runtime.context_changed` | Swarm is activated or deactivated (deactivating Swarm also stops the Swarm manager loops) | `previous`, `current`, `message`, `detected_at` | +| `runtime.availability_changed` | The runtime connects or is lost (the Docker event stream closing once the daemon stops answering counts as a loss), and once when the server starts | `name`, `connected` | +| `storage.availability_changed` | The database becomes unreachable, or answers again | `engine`, `connected` | +| `update.scan_started` | A scan starts | `scan_id`, `started_at` | +| `update.scan_completed` | A scan ends | `scan_id`, `updates_found`, `errors` | +| `update.detected` | A scan finds an update (on every scan, for each container) | `container_id`, `container_uid`, `container_name`, `image`, `current_tag`, `latest_tag`, `update_type`, `risk_score`, `alert_on`, plus `update_command`, `rollback_command` | +| `update.resolved` | An update is no longer pending | `container_id`, `container_uid`, `container_name` | +| `security.insights_changed` | A container's insights change | `container_id`, `container_name`, `highest_severity`, `count`, `change` | +| `security.insights_resolved` | All insights of a container are gone | `container_id`, `container_name` | +| `security.posture_changed` | The posture score moved by 5 points or more, or changed colour (needs `MAINTENANT_SECURITY_SCORE_THRESHOLD`; evaluated at every scoring) | `score`, `previous_score`, `color` | +| `swarm.service_discovered` | A Swarm service is created | `service_id`, `name`, `mode`, `desired_replicas`, `stack_name`, `image` | +| `swarm.service_updated` | A Swarm service is updated, or stays under-replicated for 5 minutes | `service_id`, `name`, `desired_replicas`, `running_replicas`, plus `image` for an update and `replica_alert` for under-replication | +| `swarm.service_removed` | A Swarm service is removed | `service_id`, `name` | +| `swarm.status` | The Swarm state this node manages changes: Swarm is activated (`active: true`) or deactivated (`active: false`, no other key), or the manager or worker counts move. It is not sent at startup, before any client listens | `active`, plus `is_manager`, `cluster_id`, `manager_count`, `worker_count` when active | +| `swarm.node_status_changed` | A node's status or availability changes | `node_id`, `hostname`, `role`, `old_status`, `new_status`, `old_availability`, `new_availability` | +| `swarm.node_updated` | A node joins, or its role, host name, engine version or address changes | `node_id`, `hostname`, `role`, `status`, `availability`, `engine_version`, `address`, `task_count` | +| `swarm.task_failed` | Swarm marks a task failed, on any node, with an exit code other than 0 and 143 (137 counts). Tasks stopped by a rolling update or a scale-down do not count | `task_id`, `service_id`, `service_name`, `node_id`, `container_id`, `error`, `exit_code`, `timestamp` | +| `swarm.crash_loop_detected` | A service has 3 failed tasks within 5 minutes | `service_id`, `service_name`, `failure_count`, `window_minutes`, `last_error`, `timestamp` | +| `swarm.crash_loop_recovered` | A crash-looping service is stable for 10 minutes | `service_id`, `service_name`, `timestamp` | +| `swarm.update_progress` | A rolling update progresses | `service_id`, `service_name`, `state`, `tasks_updated`, `tasks_total`, `new_image`, `message`, `timestamp` | +| `swarm.update_completed` | A rolling update completes or rolls back | `service_id`, `service_name`, `state`, `message`, `started_at`, `completed_at` | +| `swarm.topology_changed` | A remote agent reported a new Swarm topology | `agent_id` | +| `kubernetes.topology_changed` | A remote agent reported a new Kubernetes topology | `agent_id` | +| `kubernetes.workload_changed` | An alert of a workload of the server's own cluster is raised or resolved | `id`, `namespace`, `name` | +| `kubernetes.pod_changed` | An alert of a pod of the server's own cluster is raised or resolved | `namespace`, `name` | +| `kubernetes.node_changed` | An alert of a node of the server's own cluster is raised or resolved | `name` | +| `agent.created` | An agent enrols | `agent_id`, `hostname`, `label`, `runtime`, `status` | +| `agent.updated` | An agent's label or host OS changes | `agent_id`, `label` for a label change, the agent object for a host OS change | +| `agent.revoked`, `agent.deleted` | An agent is revoked or deleted | `agent_id` | +| `agent.connected`, `agent.disconnected` | An agent stream opens or closes (or goes stale) | `agent_id` | + +Events carrying `agent_id` use the sentinel `00000000-0000-0000-0000-000000000000` for the local runtime. Data that a remote agent replays from its spool after a reconnection is stored for history and does not raise events. + +!!! note "Events the server defines but does not send" + + `runtime.status`, `update.pinned` and `update.unpinned` are declared but never emitted. + +### Container log stream + +`GET /api/v1/containers/{id}/logs/stream` is a separate stream per container (see [Containers](#containers)). It sends `container.log_line` and, at the end, `container.log_error`. + +### Status page stream + +The public status page has its own event stream, separate from the one above: ``` GET /status/events ``` + +It carries only the status page events, so the public page never receives dashboard data. The dashboard stream above receives the same events without the visibility filter: it names every component. + +| Event | Emitted when | Payload keys | +|-------|--------------|--------------| +| `status.component_changed` | A monitor linked to a component changes state. Sent to the public stream only for a visible component | `component_id`, `name`, `status`, `monitors` | +| `status.component_created`, `status.component_updated`, `status.component_deleted` | An administrator creates, edits or deletes a component | `component_id` only. Sent to the public stream only when the component is visible, or was before the change | +| `status.global_changed` | Right after every `status.component_changed`, `status.component_created`, `status.component_updated` and `status.component_deleted` | `status`, `message` | +| `status.incident_created` | An incident is created, manually or automatically | `id`, `title`, `severity`, `status`, `components` (visible components only on the public stream) | +| `status.incident_updated` | An update is added to an incident | `id`, `status`, `message` | +| `status.incident_resolved` | An incident is resolved | `id`, `title` | +| `status.maintenance_started` | A maintenance window starts | `id`, `title`, `components` (visible components only on the public stream) | +| `status.maintenance_ended` | A maintenance window ends | `id`, `title`, `components` (visible components only on the public stream) | diff --git a/docs/architecture.md b/docs/architecture.md index 5c5fc338..78a46d29 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -3,53 +3,70 @@ ## Overview ``` -┌──────────────────────────────────────────────────────┐ -│ Single Go Binary │ -│ │ -│ ┌────────────────────────────────────────────┐ │ -│ │ Vue 3 + TypeScript + Tailwind (embed.FS) │ │ -│ │ Real-time SSE · uPlot charts · PWA │ │ -│ └────────────────────────────────────────────┘ │ -│ | │ -│ ┌────────────────────────────────────────────┐ │ -│ │ REST API v1 + SSE Broker │ │ -│ │ MCP Server (stdio + HTTP) │ │ -│ └────────────────────────────────────────────┘ │ -│ | | │ -│ ┌─────────────┐ ┌──────────────────────┐ │ -│ │ Docker │ │ Kubernetes │ │ -│ │ Runtime │ │ Runtime │ │ -│ └─────────────┘ └──────────────────────┘ │ -│ | | │ -│ ┌────────────────────────────────────────────┐ │ -│ │ Containers · Endpoints · Heartbeats · │ │ -│ │ Certificates · Resources · Alerts · │ │ -│ │ Updates · Security · Status Page · │ │ -│ │ Webhooks │ │ -│ └────────────────────────────────────────────┘ │ -│ | │ -│ ┌────────────────────────────────────────────┐ │ -│ │ SQLite (WAL · single-writer · zero │ │ -│ │ external dependencies) │ │ -│ └────────────────────────────────────────────┘ │ -└──────────────────────────────────────────────────────┘ +┌──────────────────────────────────────────────────────────────────┐ +│ Single Go Binary │ +│ │ +│ ┌────────────────────────────────────────────────────────────┐ │ +│ │ Vue 3 + TypeScript + Tailwind (embed.FS) │ │ +│ │ SSE live updates · uPlot charts · PWA │ │ +│ └────────────────────────────────────────────────────────────┘ │ +│ | │ +│ ┌────────────────────────────────────────────────────────────┐ │ +│ │ REST API v1 · SSE brokers · public status page │ │ +│ │ MCP server (stdio + HTTP) │ │ +│ └────────────────────────────────────────────────────────────┘ │ +│ | │ +│ ┌─────────────────┐ ┌─────────────────┐ ┌──────────────────┐ │ +│ │ Docker / Swarm │ │ Kubernetes │ │ Agent gRPC │ │ +│ │ runtime │ │ runtime │ │ server │ │ +│ └─────────────────┘ └─────────────────┘ └──────────────────┘ │ +│ | │ +│ ┌────────────────────────────────────────────────────────────┐ │ +│ │ Containers · Endpoints · Heartbeats · Certificates │ │ +│ │ Resources · Alerts · Updates · Security · Webhooks │ │ +│ │ Status page · Outbound heartbeats · Host OS support │ │ +│ └────────────────────────────────────────────────────────────┘ │ +│ | │ +│ ┌────────────────────────────────────────────────────────────┐ │ +│ │ SQLite (WAL, single writer) by default, │ │ +│ │ PostgreSQL 14+ when the operator supplies one │ │ +│ └────────────────────────────────────────────────────────────┘ │ +└──────────────────────────────────────────────────────────────────┘ ``` +Remote hosts run the same binary in agent mode and stream to the gRPC server of a server or embedded instance. See [Multi-host agents](#multi-host-agents). + +--- + +## Operating modes + +One binary, three modes, selected with `MAINTENANT_MODE` (`--mode`): + +| Mode | What runs | +|------|-----------| +| `embedded` (default) | The HTTP server (dashboard, API, MCP, status page), the monitoring of the local runtime and the storage engine. On Personal and above the agent gRPC listener also starts, so agents can enrol against an embedded instance. | +| `server` | Same as `embedded`, but the instance refuses to start below Personal. It is the only mode where `--embedded-agent` has an effect: it runs a local agent inside the process. | +| `agent` | No HTTP server and no store of its own apart from the event spool (an external database URL is refused). The agent detects its local runtime, enrols against a server and streams to it. It starts even when no runtime answers, reports the host alone and retries the runtime in the background. It keeps its identity (`identity.json`), its event spool (`spool.db`) and a liveness file in `MAINTENANT_DATA_DIR`. | + +Two more entry points share the binary: `maintenant --mcp-stdio` serves the MCP server over stdin and stdout, and `maintenant healthcheck` backs the image `HEALTHCHECK` (it reads the agent liveness file when there is one, otherwise it calls `/api/v1/health` on the configured address). `--copy-store-to` copies an installation into an empty PostgreSQL database and exits. + --- ## Design Philosophy -**Single binary** — The Vue 3 frontend is compiled to static assets and embedded in the Go binary via `embed.FS`. One file to deploy, nothing else to configure. +**Single binary**: The Vue 3 frontend is compiled to static assets and embedded in the Go binary via `embed.FS`. One file to deploy. There is no build per edition: the paid code (`internal/commercial` and `frontend/src/commercial`) is always compiled in and a licence key unlocks it at runtime. The core is Apache 2.0, the paid code is under the commercial licence. -**Zero external dependencies** — SQLite by default, with nothing to install: no Redis, no message queue, no database to administer. The binary runs anywhere Go compiles. An operator watching a fleet *may* point the server at a PostgreSQL they already run, which is the only way agent identities survive losing the server's machine; absent that setting, nothing changes. Agents store their state in SQLite, always. +**Zero external dependencies**: SQLite by default, with nothing to install: no Redis, no message queue, no database to administer. An operator watching a fleet *may* point the server at a PostgreSQL they already run, which is the only way agent identities survive losing the server's machine; absent that setting, nothing changes. Agents store their state in SQLite, always. -**Real-time by default** — Every state change is pushed to the browser via Server-Sent Events (SSE). No polling, no stale data. +**Live updates**: State changes are pushed to the browser over Server-Sent Events. The client reconnects with a backoff and its stores refetch after a reconnect. A few widgets also poll (host resources, sparklines, scan progress). -**Read-only** — maintenant never modifies your containers. It observes the Docker socket or Kubernetes API in read-only mode. +**Read-only**: maintenant never changes your containers. It reads from the Docker socket or the Kubernetes API, and the update and rollback commands it shows are for you to run. -**Label-driven** — Monitoring is configured through Docker labels directly on your containers. No separate config files to maintain. +**Label-driven**: Monitoring is declared where the workload is defined: Docker labels, Swarm `deploy.labels` and, for a few settings, Kubernetes annotations. Manual endpoints, heartbeats, certificates, channels and the status page are managed in the interface or through the API, and the instance itself is configured with `MAINTENANT_*` variables or their matching flags. -**Runtime-agnostic** — Docker and Kubernetes are abstracted behind a common `Runtime` interface. maintenant auto-detects the runtime at startup or can be forced via `MAINTENANT_RUNTIME`. +**Runtime-agnostic**: Docker (with Swarm detection) and Kubernetes sit behind a common `Runtime` interface in `internal/runtime`. At startup the runtime is chosen in this order: `MAINTENANT_RUNTIME`, then `KUBERNETES_SERVICE_HOST` (in cluster), then a kubeconfig whose cluster answers, then the Docker socket. A kubeconfig whose cluster does not answer falls back to Docker with a warning. When the runtime is lost the instance keeps serving in degraded mode and reconnects in the background, retrying with a growing delay (1 second at first, 30 seconds at most) and logging the cause of each failure. A Swarm is detected as soon as Docker connects, so enabling Swarm later needs no restart. + +**Editions are data, not builds**: `internal/extension` defines the editions (Community, Personal, Pro), the capabilities, the quotas and the history windows. The tier table that says which edition opens what lives in `internal/commercial/tiers`, and `internal/extpoint` declares the implementations a licensed build plugs into the core (update enrichment, posture scoring, notification channels, status page extras, maintenance suppression, escalation, multi-host). A nil implementation keeps the Community behaviour. `GET /api/v1/edition` publishes the table so the frontend never holds one of its own. --- @@ -59,57 +76,99 @@ | Technology | Purpose | |-----------|---------| -| **Go** (>= 1.25) | Application runtime | +| **Go** 1.26.6 | Application runtime (`go` directive of `go.mod`) | +| **`net/http`** (stdlib) | HTTP server, REST API, SSE, cross-origin protection | | **SQLite** (WAL mode) | Persistence by default, single-writer pattern | +| **`github.com/mattn/go-sqlite3`** | SQLite driver (CGO, the only CGO dependency) | | **PostgreSQL** 14+ (optional) | Persistence for the server data set, when the operator supplies one | -| **`net/http`** (stdlib) | HTTP server, REST API, SSE | -| **`github.com/docker/docker`** | Docker SDK for container discovery and events | -| **`k8s.io/client-go`** | Kubernetes API client | -| **`k8s.io/metrics`** | Kubernetes metrics API | -| **`github.com/mattn/go-sqlite3`** | SQLite driver (CGO — the only one) | | **`github.com/jackc/pgx/v5`** | PostgreSQL driver (pure Go) | -| **`github.com/google/go-containerregistry`** | OCI registry scanning | +| **`github.com/golang-migrate/migrate/v4`** | Embedded schema migrations for both engines | +| **`github.com/moby/moby/client`** and **`api`** | Docker SDK for container discovery, events and Swarm | +| **`k8s.io/client-go`**, **`api`**, **`apimachinery`** | Kubernetes API client | +| **`k8s.io/metrics`** | Kubernetes metrics API | +| **`google.golang.org/grpc`** and **`protobuf`** | Agent protocol (`proto/ingest.proto`) | +| **`github.com/google/go-containerregistry`** | Registry queries for image updates | +| **`github.com/Masterminds/semver/v3`** | Image tag comparison | | **`github.com/modelcontextprotocol/go-sdk`** | MCP server (AI assistant integration) | +| **`golang.org/x/crypto`** | OCSP response parsing for certificate monitoring | +| **`golang.org/x/time`**, **`golang.org/x/sync`** | Rate limiting, concurrent probes and collectors | +| **`github.com/yuin/goldmark`** and **`github.com/microcosm-cc/bluemonday`** | Markdown rendering and sanitizing for the status page | +| **`github.com/google/uuid`** | UUID generation behind `internal/uid` | +| **`github.com/kolapsis/shm`** | Anonymous usage telemetry SDK (opt-out with `MAINTENANT_DISABLE_TELEMETRY`) | | **`embed.FS`** | Frontend embedding | ### Frontend | Technology | Purpose | |-----------|---------| -| **Vue 3** | UI framework (Composition API) | -| **TypeScript 5.9** | Type safety | +| **Vue 3** (3.5) | UI framework (Composition API) | +| **TypeScript** 5.9 | Type safety | | **Pinia** | State management (SSE-connected stores) | -| **Tailwind CSS 4** | Styling | +| **Vue Router** | Client-side routing | +| **Tailwind CSS** 4 | Styling | | **uPlot** | Lightweight time-series charts (~40 KB) | -| **Vite** | Build tooling | -| **vite-plugin-pwa** | Progressive Web App support | +| **lucide-vue-next** | Icons | +| **`@tanstack/vue-virtual`** | Virtualized lists | +| **Vite** 7 | Build tooling | +| **vite-plugin-pwa**, **workbox-window** | Progressive Web App support | +| **Vitest** | Unit tests | --- ## Project Structure ``` -cmd/maintenant/ Entry point, service wiring +cmd/maintenant/ Entry point: flag parsing, modes, copy-store, healthcheck web/ Embedded frontend (embed.FS) +proto/ ingest.proto, the agent protocol internal/ Private packages - alert/ Alert engine, notifier, formatters (webhook, discord) - api/v1/ HTTP handlers, SSE broker, router + app/ Service wiring, configuration and flag registry, lifecycle, + HTTP server assembly, storage supervision + api/v1/ HTTP handlers, SSE broker, router, middleware + agent/ Agent mode: runtime detection, enrollment, collectors, + endpoint and certificate probes, event spool + agentauth/ Byte-exact signing payload of the agent handshake + agentevent/ Observation time and replay flag carried by agent events + agentpb/ Generated gRPC and protobuf stubs + agentproto/ Command helpers and public gRPC URL resolution + alert/ Alert engine, notifier, webhook and Discord senders, down detector + escalation/ Escalation policy, run and service contracts certificate/ TLS certificate monitoring + commercial/ Paid code, commercial licence + channels/ Email, Telegram, Slack and Teams channels + escalation/ Escalation service and runner + license/ Licence verification, cache and update window + maintenance/ Alert suppression during maintenance windows + multihost/ Agent gRPC server, sessions, rate limit + posture/ Security posture scorer + statuspage/ Incidents, maintenance windows, subscriber mail, personalization + tiers/ Tier table: capabilities, quotas, history caps per edition + updates/ CVE, changelog and risk enrichment container/ Container model, service, uptime docker/ Docker runtime implementation endpoint/ Endpoint monitoring (HTTP/TCP) - event/ Event types and dispatching - extension/ Extension point interfaces + no-ops (used by Pro) - heartbeat/ Heartbeat/cron monitoring - kubernetes/ Kubernetes runtime implementation - license/ License validation and management - mcp/ MCP server (Model Context Protocol) - ratelimit/ Per-IP rate limiting middleware - resource/ Resource metrics collection - runtime/ Runtime abstraction interface - security/ Network security analysis, posture scoring + eol/ Host operating system end of support + event/ SSE event type constants + extension/ Editions, capabilities, quotas, history windows, policy contract + extpoint/ Implementations a licensed build plugs into the core + heartbeat/ Heartbeat and cron monitoring + hoststat/ Host CPU, memory and disk from /proc + kubernetes/ Kubernetes runtime implementation and topology ingest + mcp/ MCP server (Model Context Protocol), OAuth for /mcp + outbound/ Outbound heartbeats + proxylabels/ Endpoint labels derived from Traefik and Caddy labels + ratelimit/ Per-IP rate limiting middleware, client IP resolution + resource/ Resource metrics collection, rollups, history + retry/ Exponential backoff helper + runtime/ Runtime interface and detection + security/ Security insights (published ports, network mode, privileges) + ssrf/ Outbound URL guard for webhooks and channels status/ Public status page (handler, subscribers) store/ Store layer, dialect, migrations, writer, copy + swarm/ Swarm discovery, nodes, crash loops, rolling updates + telemetry/ Anonymous usage telemetry + trust/ Root certificates for every outbound TLS connection + uid/ Entity identifiers (UUIDv7 and deterministic UUIDv5) update/ Update intelligence, registry scanning webhook/ Webhook dispatcher @@ -118,10 +177,15 @@ frontend/src/ components/ Reusable UI components ui/ Generic UI primitives dashboard/ Dashboard-specific widgets + heartbeats/ Heartbeat widgets + commercial/ Paid frontend code, commercial licence + (components, composables, services, stores, types) stores/ Pinia stores (SSE-connected) - services/ API client functions + services/ API client functions, SSE bus, session guard composables/ Vue composables layouts/ Page layouts + locales/ Status page translations + types/ Shared TypeScript types utils/ Utility functions router/ Vue Router configuration ``` @@ -133,30 +197,80 @@ frontend/src/ ### Container Event ``` -Docker/K8s Event +Docker / Kubernetes event → Runtime.StreamEvents() → container.Service.ProcessEvent() - → SQLite (persist state transition) - → SSE Broker.Broadcast() - → Browser (real-time update) - → Alert Engine (evaluate rules) - → Webhook Dispatcher (deliver to channels) + → store (persist state transition) + → event callback (internal/app/wiring.go) + → SSE broker → browsers + → webhook dispatcher (observer) + → status page (component status) + → alert engine (restart loop, health change) ``` ### Endpoint Check ``` -Check Engine (ticker) - → HTTP/TCP request +Check engine (one ticker per endpoint) + → HTTP/TCP probe → endpoint.Service.ProcessCheckResult() - → SQLite (persist check result) - → SSE Broker.Broadcast() - → Browser (update sparkline) - → Alert Detector (evaluate thresholds) - → Alert Engine → Notifier → Webhook channels - → Certificate Service (auto-detect TLS from HTTPS) + → store (persist check result and status) + → event callback (endpoint.status_changed, internal/app/wiring.go) + → SSE broker → browsers + → webhook dispatcher (observer) + → status page (component status, on a status change) + → alert callback (internal/app/wiring.go) + → alert engine (consecutive failures, untrusted certificate) + → SSE broker (endpoint.alert, endpoint.recovery) + → certificate service (auto-detect TLS on HTTPS targets) +``` + +### Remote agent event + +``` +Agent collector → spool → gRPC Push stream + → multihost server (authentication, per-agent rate limit) + → dispatcher + → container, endpoint, heartbeat, resource and certificate services + → Swarm and Kubernetes topology ingest ``` +The services store what an agent sends under its `agent_id`. Endpoint results and heartbeat pings go through the same service methods as the ones produced locally, so they follow the same path afterwards (alert thresholds, status page, deadlines). + +### Alerts + +The alert engine does not listen to the SSE broker. Each service hands it alert events directly: the callbacks wired in `internal/app/wiring.go` push an `alert.Event` into `alertEngine.EventChannel()` and to the status page (`HandleAlertEvent`), next to the SSE broadcast. + +``` +alert.Event → alert engine + → deduplicate on source, type, entity + → silence rules and maintenance windows + → persist the alert + → SSE broker (alert.fired, alert.silenced, alert.resolved) + → escalation (when the edition opens it) + → alert triggers → channels → notifier queue (workers, retries) +``` + +Channels are silent by default: an alert reaches a channel only through an alert trigger or an escalation policy. The notifier tries a delivery up to three times, waiting 1 second and then 5 seconds. The notifications of one alert to one channel go through the same worker one after the other, so a recovery never overtakes the alert it resolves. + +Webhook subscriptions are a separate path. The webhook dispatcher is an observer of the SSE broker: it reacts to six event types (`container.state_changed`, `endpoint.status_changed`, `heartbeat.status_changed`, `certificate.status_changed`, `alert.fired`, `alert.resolved`) and sends them through the notifier's worker pool. The events sent to one URL are delivered in order. The dispatcher records each delivery on the subscription (last status, consecutive failures) and deactivates it after 10 failures in a row; a success reactivates it. + +--- + +## Multi-host agents + +Multi-host monitoring needs Personal or above. The gRPC protocol is defined in `proto/ingest.proto` (service `Ingest`). + +- **Enrollment**: an operator creates a one-time enrollment token (24 hours by default, 7 days at most). The agent generates an Ed25519 key pair and calls `RegisterAgent` with the token. When the server revokes the agent or no longer knows it, the agent discards its spool and enrols a fresh identity once with the token it is configured with. Without a usable token it exits with an error that names `MAINTENANT_ENROLLMENT_TOKEN`. +- **Authentication**: every `Push` stream starts with a random 32-byte nonce from the server. The agent answers with a signature over the nonce, its id and a timestamp. A clock difference above 300 seconds, a revoked agent or an unknown agent is refused. +- **Stream**: the agent pushes container events and inventories, endpoint and certificate probe results, resource samples, Swarm and Kubernetes topologies and the host OS identity. The server answers with acknowledgements and errors (`agent_revoked`, `rate_limited`, ...). Commands are the one exception to the push-only stream: the server can ask an agent for container logs. +- **Probing**: the agent probes the endpoints and certificates of its own containers itself, from their labels. The server never dials them. +- **Spool**: events are queued in memory, then in `spool.db`, while the server is unreachable, and replayed after the reconnect. Bounds: `MAINTENANT_AGENT_SPOOL_MAX_MEMORY_BYTES`, `MAINTENANT_AGENT_SPOOL_MAX_DISK_BYTES` and `MAINTENANT_AGENT_SPOOL_MAX_AGE_SECONDS`. State snapshots are never queued, and a replayed event is stored for history without raising live events or alerts. +- **Transport**: TLS. Without `MAINTENANT_GRPC_TLS_CERT` and `MAINTENANT_GRPC_TLS_KEY` the server generates a self-signed certificate and logs a warning. `MAINTENANT_GRPC_TLS_INSECURE` serves plain h2c behind a trusted reverse proxy, and the embedded agent then dials `grpc://` instead of `grpcs://`. A certificate without its key, or the reverse, stops the startup. The listener defaults to `127.0.0.1:8443`. +- **Identity**: the server's own runtime is an agent too, with the sentinel id `00000000-0000-0000-0000-000000000000` (`uid.LocalAgent`), so every entity carries an `agent_id`. + +See [Multi-Host Monitoring](features/multihost.md) and [Agent Setup](guides/agent-setup.md). + --- ## Storage engines @@ -166,7 +280,7 @@ storage; PostgreSQL backs the server data set when an operator supplies a connection string. Every engine difference goes through a `Dialect`, and there are six of them: placeholder syntax, batched deletes, opening PRAGMAs, error classification, write serialization, and SQL-side UUID generation for rollups. -Nothing else in the ~50 query files knows which engine it runs on — the UUID +Nothing else in the query files knows which engine it runs on: the UUID rework had already made the schema portable (TEXT keys, epoch-second BIGINTs, `ON CONFLICT ... DO UPDATE`). @@ -175,8 +289,8 @@ full history; PostgreSQL starts from a single baseline numbered at the SQLite head of the day it was written (28). From 29 onward, a migration is written for both engines under the same number, or not at all. That rule is enforced by a test, not by discipline: it migrates a fresh database on each engine and -compares the two heads — tables, columns, types, defaults, indexes, foreign -keys, constraints — failing on any divergence. +compares the two heads (tables, columns, types, defaults, indexes, foreign +keys, constraints) and fails on any divergence. **Why it exists.** The server holds one class of data the fleet cannot rebuild: agent identities and enrolments. Kept on the machine that runs the process, @@ -184,9 +298,13 @@ that data makes the instance irreplaceable. Detached, a replacement instance started elsewhere picks the fleet back up with no action on any monitored host. **What the product does not do:** it never installs, backs up or supervises the -database, and it does not orchestrate failover — no leader election, no mutual +database, and it does not orchestrate failover: no leader election, no mutual exclusion. Instances register in a table and beat; a second one is *reported*, -never arbitrated. Exclusion belongs to the operator's cluster manager. +never arbitrated. Exclusion belongs to the operator's cluster manager. While the +database is unreachable `/api/v1/health` still answers 200 and reports it in +`storage.connected`, the reads that need the database answer +`503 STORAGE_UNAVAILABLE`, and a `storage.availability_changed` event tells the +interface. --- @@ -194,22 +312,70 @@ never arbitrated. Exclusion belongs to the operator's cluster manager. maintenant uses SQLite in WAL (Write-Ahead Logging) mode with a single-writer pattern: -- **One writer goroutine** — All writes are serialized through a channel-based writer to avoid `SQLITE_BUSY` errors -- **Multiple readers** — Read queries run concurrently without blocking -- **Automatic migrations** — Schema migrations run at startup using embedded SQL files. A schema newer than the binary is refused rather than written into, on either engine. -- **Resource history** — Charts up to 24h group raw samples; the 7-day range reads the hourly rollup, which holds exactly the buckets it displays. Raw samples are therefore only kept for 48 hours, and they are what dominates database size. -- **Retention cleanup** — Background goroutine prunes old data (transitions: 90 days, check results: 30 days, heartbeat pings: 30 days, resource snapshots: 48 hours, resource hourly: 90 days, resource daily: 1 year). A pass runs at startup and then hourly, deleting in batches of 1000 rows until each table is drained. If a pass hits its 2-minute-per-table budget it reschedules itself a minute later instead of waiting for the next hour, so a backlog is cleared instead of accumulating. Tunable with `MAINTENANT_RETENTION_SNAPSHOTS`, `MAINTENANT_RETENTION_INTERVAL` and `MAINTENANT_RETENTION_BATCH_SIZE`. -- **Bounded WAL** — Every pooled connection sets `journal_size_limit` (64 MiB) and `wal_autocheckpoint` (1000 pages) through a driver `ConnectHook`; pages freed by retention are returned to the filesystem with `incremental_vacuum` in slices +- **One writer goroutine**: All writes are serialized through a channel-based writer to avoid `SQLITE_BUSY` errors +- **Multiple readers**: Read queries run concurrently without blocking +- **Automatic migrations**: Schema migrations run at startup using embedded SQL files. A schema newer than the binary is refused rather than written into, on either engine. +- **Bounded WAL**: Every pooled connection sets `journal_size_limit` (64 MiB) and `wal_autocheckpoint` (1000 pages) through a driver `ConnectHook`; pages freed by retention are returned to the filesystem with `incremental_vacuum` in slices + +### Resource history + +Containers are sampled every 10 seconds. A rollup runs every 5 minutes and feeds the hourly and daily tables. A history window reads the table that covers it, so the raw samples only need to cover the shortest ranges. + +| Window | Read from | Resolution | Opened by | +|--------|-----------|------------|-----------| +| `1h` | raw samples | 10 seconds | Community | +| `6h` | raw samples | 1 minute | Community | +| `24h` | raw samples | 5 minutes | Community | +| `7d` | hourly rollup | 1 hour | Community | +| `30d` | hourly rollup | 1 hour | Personal | +| `90d` | daily rollup | 1 day | Pro | + +The edition caps the window, not the live values. The cap is 7 days on Community, 30 days on Personal and 90 days on Pro (`internal/commercial/tiers`, `internal/extension`). + +### Retention + +A background goroutine prunes old data. A pass runs at startup and then hourly (`MAINTENANT_RETENTION_INTERVAL`), deleting in batches of 1000 rows (`MAINTENANT_RETENTION_BATCH_SIZE`) until each table is drained. If a pass hits its 2-minute-per-table budget it reschedules itself a minute later instead of waiting for the next hour, so a backlog is cleared instead of accumulating. + +| Data | Kept | +|------|------| +| Container state transitions | 90 days, except the latest transition of each container, which is kept | +| Archived containers | 30 days after archival, with their update, CVE, pin and risk history | +| Endpoint check results | 30 days | +| Inactive endpoints | 30 days | +| Heartbeat pings and executions | 30 days | +| Certificate check results | 30 days | +| Daily uptime aggregates (endpoints, heartbeats, containers) | 365 days | +| Raw resource samples | 48 hours (`MAINTENANT_RETENTION_SNAPSHOTS`, at least 24 hours) | +| Hourly resource rollups | 90 days | +| Daily resource rollups | 1 year | +| Alerts | 90 days (daily pass) | +| Update scan records and image updates | 30 days (daily pass) | +| Escalation runs and deliveries | 90 days (nightly at 03:00 local time) | +| Unconsumed enrollment tokens | 7 days after expiry (hourly pass) | + +Before the raw transitions, check results and heartbeat pings are deleted, the same pass writes their daily uptime aggregates (`endpoint_uptime_daily`, `heartbeat_uptime_daily`, `container_uptime_daily`). A day whose aggregate could not be written keeps its raw rows until a later pass succeeds. The uptime API serves up to 365 days from the aggregates and computes the days not aggregated yet from the raw rows. + +--- + +## HTTP layer + +The top-level mux routes `/mcp` (when MCP is enabled) and the OAuth endpoints (when a client id and secret are also set), `/status/*` (public status page), `/api/`, `/ping/` and, for everything else, the embedded single-page application with a fallback to `index.html`. Three per-IP token buckets protect it: 10 requests per second (burst 20) for `/ping/`, `/status/`, `/mcp` and `/oauth/`, 50 per second (burst 200) for `/api/`, and 5 per hour for status page subscriptions. Client addresses are read from forwarded headers only for the proxies listed in `MAINTENANT_TRUSTED_PROXIES`. + +Around the mux, from the outside in: security headers and a content security policy, a 10-second timeout for every request except the streaming ones, and the demo guard. The API router adds panic recovery, request logging, a request id, CORS, the cross-origin guard for unsafe methods, and the body size limit. The server itself uses a 5-second read timeout and no write timeout, which SSE requires. The API has no authentication of its own: see [Security](security.md) for what to put in front of it. --- ## SSE Architecture -The SSE broker is the central hub for real-time updates: +The SSE brokers are the central hub for real-time updates. There are two, one per audience: + +1. **Services** emit events when state changes (container state, heartbeat ping, alert fired, agent connected) +2. **The main broker** (`GET /api/v1/containers/events`) fans out events to every connected dashboard. A client that falls 64 events behind loses events instead of blocking the others. +3. **The status broker** (`GET /status/events`) carries only `status.*` events, so the public status page never sees dashboard events. The same `status.*` events also go to the main broker, for the dashboard. +4. **The webhook dispatcher** observes the main broker and delivers six event types to external URLs. + +Both streams send a keep-alive comment every 25 seconds so that a proxy does not cut an idle connection. -1. **Services** emit events when state changes (container state, check result, alert fired) -2. **SSE Broker** fans out events to all connected browser clients -3. **Webhook Dispatcher** observes the broker and delivers events to external channels -4. **Alert Engine** processes events and generates alerts +The alert engine is not a subscriber: services call it directly (see [Alerts](#alerts)). -Each browser tab maintains a single SSE connection to `/api/v1/containers/events`. The Pinia stores dispatch received events to the appropriate component. +Each browser tab keeps a single connection to `/api/v1/containers/events`, shared by all Pinia stores through `sseBus`. The stream is closed after the tab has been hidden for 60 seconds and reopened when it becomes visible, and the stores refetch after a reconnect. Container logs use a separate stream per container, `/api/v1/containers/{id}/logs/stream`. See the [API reference](api/reference.md#sse-event-stream) for the event list. diff --git a/docs/features/alert-escalation.md b/docs/features/alert-escalation.md index ec0550fe..e4c84a99 100644 --- a/docs/features/alert-escalation.md +++ b/docs/features/alert-escalation.md @@ -2,26 +2,32 @@ **Availability**: Maintenant Pro -Escalation policies automatically route unacknowledged alerts through a chain of notification levels, each with a configurable delay and a distinct set of channels. +Escalation policies automatically route unacknowledged alerts through a chain of up to 5 notification levels, each with a configurable delay and a distinct set of channels. -> **Pattern: reserved-escalation channel.** Since channels are silent by default (they only receive alerts when wired through an [Alert Trigger](alerts.md#alert-triggers)), you can create a channel that exists *only* for an escalation level. The directrice technique's email referenced in Level 3 of a policy, with no trigger using it, will *only* be notified after T+1h of unacknowledged escalation — never at the initial dispatch. This is the cleanest way to model "last-resort" destinations without duplicate notifications. +> **Pattern: reserved-escalation channel.** Since channels are silent by default (they only receive alerts when wired through an [Alert Trigger](alerts.md#alert-triggers)), you can create a channel that exists *only* for an escalation level. A manager's email referenced in Level 3 of a policy, with no trigger using it, will *only* be notified after 1 hour of unacknowledged escalation, never at the initial dispatch. This is the cleanest way to model "last-resort" destinations without duplicate notifications. Such a channel gets the escalation, acknowledgment and exhaustion notices, but no notice when the alert resolves. --- ## Concept -When an alert fires and remains unacknowledged, an active escalation policy that matches the alert will start an **escalation run**. The run tracks which notification level is currently due and dispatches notifications to the configured channels at each level's delay. The chain stops as soon as the alert is acknowledged or resolved. +When an alert fires, every active escalation policy whose filters match it starts an **escalation run**. The run tracks which notification level is due next and sends the notifications of that level when its delay has passed. The chain stops as soon as the alert is acknowledged or resolved. + +A level's `delay_seconds` is counted **from the start of the run** (the moment the alert fires), not from the previous level. A level with a delay of 900 fires 15 minutes after the alert, whatever the level before it had. Runs are evaluated once a minute, so a level can fire up to a minute after its delay. ``` -Alert fires +Alert fires (T+0) │ - ├─ Level 1 (delay 5 min) → Notify #slack-oncall + ├─ Level 1 (delay 300) → T+5 min, if still unacknowledged → Notify #slack-oncall │ - ├─ Level 2 (delay 15 min, if still unacknowledged) → Page +33-6-XX + ├─ Level 2 (delay 900) → T+15 min, if still unacknowledged → Call the pager webhook │ - └─ Level 3 (delay 1 hour, if still unacknowledged) → Email management + └─ Level 3 (delay 3600) → T+1 hour, if still unacknowledged → Email management ``` +A run keeps the policy as it was when the run started. Editing a policy changes the alerts raised afterwards, not the runs already in progress. Deactivating or deleting a policy stops its runs (`stopped_by_policy_disabled`, `stopped_by_policy_deletion`). + +Policies only see alerts that are active. An alert raised while a matching [silence rule](alerts.md#silence-rules) or maintenance window is in force is silenced and starts no run. If an alert becomes more severe, policies that match its new severity start a run too; runs already started continue untouched. An alert that is already acknowledged starts no new run when its severity rises, and a run still going when the alert is acknowledged stops at its next evaluation. Each time a level is sent, the alert's `escalated_at` is set to that time. + --- ## Examples @@ -32,18 +38,18 @@ Notify the on-call Slack channel 5 minutes after an alert fires. ```json { - "name": "Critical pager — Level 1 only", + "name": "Critical pager: Level 1 only", "active": true, "filters": { "severities": ["critical"] }, "levels": [ - { "delay_seconds": 300, "channel_ids": [1] } + { "delay_seconds": 300, "channel_ids": ["0195f3c8-5e2a-7b14-9a30-7d1c4f2e8a66"] } ] } ``` ### Multi-level policy -Escalate progressively over an hour. +Escalate progressively over an hour: the levels fire 5 minutes, 15 minutes and 1 hour after the alert. ```json { @@ -51,85 +57,123 @@ Escalate progressively over an hour. "active": true, "filters": { "severities": ["critical", "warning"] }, "levels": [ - { "delay_seconds": 300, "channel_ids": [1] }, - { "delay_seconds": 900, "channel_ids": [2] }, - { "delay_seconds": 3600, "channel_ids": [3] } + { "delay_seconds": 300, "channel_ids": ["0195f3c8-5e2a-7b14-9a30-7d1c4f2e8a66"] }, + { "delay_seconds": 900, "channel_ids": ["0195f3c9-0b71-7e52-8d04-3a9f6c1b2d40"] }, + { "delay_seconds": 3600, "channel_ids": ["0195f3c9-44d8-7a3c-b6e1-92f0d5a7c318"] } ] } ``` -### Filter by tag +### Filter by entity -Only escalate alerts tagged `prod`. +Only escalate alerts about one endpoint. A scope names an alert's entity: `kind` is the `entity_type` of the alerts to match, from the Entity column of the [alert sources](alerts.md#alert-sources) (`container`, `endpoint`, `heartbeat`, `certificate`, `agent`, `swarm_service`, `workload` and so on), and `ref_id` is the `entity_id` those alerts carry (a UUID for a container, endpoint, heartbeat, certificate or agent). Neither is checked when the policy is saved: a scope that names nothing matches no alert. ```json { - "name": "Prod-only chain", + "name": "Checkout endpoint chain", "active": true, "filters": { "severities": ["critical"], - "tags": ["prod"] + "scopes": [{ "kind": "endpoint", "ref_id": "0195f3c4-11aa-7f00-8c55-2b9d0e3a4c18" }] }, "levels": [ - { "delay_seconds": 300, "channel_ids": [1] } + { "delay_seconds": 300, "channel_ids": ["0195f3c8-5e2a-7b14-9a30-7d1c4f2e8a66"] } ] } ``` +An empty `severities` or `scopes` matches everything. An alert must satisfy both filters to match. The policy editor of the **Escalation** page edits the name, the activation, the severities (`warning`, `critical`) and the levels, up to 5; it shows the scopes of a policy read-only and keeps them when you save. Scopes are set through the API. The API also accepts `info` as a severity. + +--- + +## Validation + +| Field | Rule | +|-------|------| +| `name` | Required, at most 120 characters | +| `levels` | At least one, at most 5 | +| `levels[].delay_seconds` | Between 60 and 86400 | +| Consecutive levels | Each level is at least 60 seconds after the previous one, so delays must increase | +| `levels[].channel_ids` | At least one channel ID per level | + +A request that breaks a rule returns `400 validation_failed` with the field in the message. The channel IDs are not checked against existing channels when the policy is saved: a level that points at a deleted channel skips it when it fires. + --- ## Interactions ### Acknowledgment stops escalation -When an alert is acknowledged, any active escalation run is immediately stopped with status `stopped_by_ack`. A notification is sent on all channels that have already received an escalation notification, informing them the alert has been acknowledged. +When an alert is acknowledged, any active escalation run is immediately stopped with status `stopped_by_ack`. Every channel that received a notification of the run (a delivery that is `sent` or still `pending`) gets one acknowledgment notice, once, even if it appeared in several levels. The notice reuses the recovery format of the channel, and its message is: + +``` +Acknowledged by at