Repository navigation
Expand file tree
/
Copy pathteploy.yml
More file actions
252 lines (239 loc) · 12.7 KB
/
Copy pathteploy.yml
File metadata and controls
252 lines (239 loc) · 12.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
# Observe — Teploy deployment template.
#
# First time:
# teploy setup # installs Docker + Caddy on the server
# use the volume_ownership contract below for a fresh state directory
# teploy deploy # builds image, ships it, switches traffic
#
# Subsequent deploys:
# teploy deploy
#
# Required env (export before running):
# TEPLOY_OBSERVE_JWT_SECRET — 32+ random bytes, `openssl rand -hex 32`
# TEPLOY_OBSERVE_SESSION_SALT — another random 32-byte string
# TEPLOY_OBSERVE_ADMIN_PASSWORD — initial admin password (change via /settings after)
#
# Shell-exported vars above only apply to the deploy that exports them — if
# you forget to re-export before a later `teploy deploy`, these silently
# fall back to empty, and Observe generates a fresh random JWT secret/salt
# on every restart, invalidating all sessions (only a WARN log, easy to
# miss). For a persistent deploy, set these once via
# `teploy secret set OBSERVE_JWT_SECRET=... OBSERVE_SESSION_SALT=...
# OBSERVE_ADMIN_PASSWORD=...` instead — secrets set that way always
# override this file's env: block on every future deploy, no re-export
# needed. To reset an already-seeded admin's password later (the plain
# OBSERVE_ADMIN_PASSWORD only seeds on first boot), use the
# OBSERVE_RESET_ADMIN_PASSWORD escape hatch the same way, then unset it.
app: observe
domain: observe.example.com
server: observe.example.com
# Ingress + binds. The dashboard/read API (`port`) is deliberately NOT on
# 0.0.0.0: anything reaching it gets a login page, so it is pinned to the
# tailnet address. The INGEST listener is the opposite - it must stay on
# 0.0.0.0 because the tunnel fronting the public ingest hostname may dial
# either address. Losing either silently breaks a live install: without
# `publish` nothing listens on 3001 at all and every SDK write is refused.
ingress: host
bind: 100.108.123.49
publish:
- "0.0.0.0:3001:3001"
# `port` is the app listener: dashboard + read API + admin. Anything Teploy
# routes `domain` to can reach the dashboard, so point `domain` here only for
# an instance you're happy to have a login page on (or gate it behind
# Cloudflare Access / a tailnet).
#
# To collect telemetry from browsers or off-network servers, run the
# ingest-only listener instead and give it its own hostname:
#
# env:
# OBSERVE_INGEST_ADDR: ":3001"
# OBSERVE_PUBLIC_URL: "https://observe.example.com"
#
# then route ingest.example.com -> 3001. That port serves telemetry writes
# and 404s everything else, so the dashboard is not merely unrouted there —
# it isn't listening on it.
port: 3000
# Teploy readiness must probe SQL/WAL health rather than the SPA fallback.
health:
mode: http
path: /healthz
platform: linux/amd64
# Build mode: with no `image` set, Teploy builds from the repo Dockerfile.
# Default (build_local unset) builds on the server — best for this vendored
# Go module, since it compiles natively for the server's arch with no local
# Docker or cross-build needed. Set `build_local: true` to build on your
# machine and ship the image over SSH instead, or set
# `image: ghcr.io/useteploy/teploy-observe` to pull a prebuilt image.
# Graceful shutdown — give ingest buffer 30s to flush before SIGKILL.
stop_timeout: 30
# Nucleus runs as a Teploy accessory (separate container, sidecar to observe).
# Teploy names the accessory's network alias `<app>-<name>`, so observe
# reaches it at `observe-nucleus:5432` (see OBSERVE_NUCLEUS_URL below).
accessories:
nucleus:
# Pin the engine. `:latest` is republished on every push to Nucleus's main,
# so with a floating tag an unrelated redeploy swaps the storage engine
# underneath a live database — and the tag moves between the day you test
# and the day you deploy. Bump this deliberately, after reading the
# release's migration notes.
#
# v0.1.8 (2026-08-21), up from v0.1.5. Verified before bumping: Observe's
# suite passes against it, 34 packages, same as 0.1.6 — while v0.1.5 fails
# auth/cohorts/errors because migration 28 cannot run on a fresh database
# (`events_pre028 not found in storage`). The one package that still fails
# on BOTH engines is `internal/cohorts`, which is Observe's own known
# `events_recent` gap (no `distinct_id`), order-dependent and unrelated to
# the engine. v0.1.6's breaking change applies: SKIP LOCKED, NOWAIT and
# partial CREATE INDEX ... WHERE now error rather than being ignored —
# checked, neither Observe nor Ship uses them.
#
# v1.0.0 (2026-08-31), up from v0.1.8. Two wire-protocol fixes are why this
# release exists. The RowDescription a client received did not always
# describe the DataRows that followed it — 20,800 of 125,419 statements
# over a ten-second concurrent read were answered with a description that
# did not match their rows, and a client reading fields positionally throws
# while one reading by name silently decodes a row with fields missing.
# And Describe could EXECUTE a side-effecting statement, because whether to
# describe statically or by running the statement was decided by a text
# scan that a comment between a function name and its arguments defeats.
#
# v1.1.1 (2026-09-20), up from v1.0.0. WAL format v2 lands: versioned
# checksummed framing, stable 64-bit MVCC ids, atomic CommitV2 records,
# lossless schema codec, tail GC at VACUUM, and the ACQUIRE SNAPSHOT LEASE
# primitive observe's backups now run inside. Also fixes two a0732f5c-era
# races that admitted duplicate PRIMARY KEYs under concurrent DML+DDL, and
# a same-transaction CREATE+RENAME visibility failure that broke fresh
# installs at migration 027 (repo-built engines). First open of an existing
# data dir upgrades the WAL in place and preserves the original byte-for-
# byte as mvcc.wal.v1 (retired on the next clean open) — that backup is
# the rollback path; DB_FORMAT_VERSION gates cross-format binary swaps.
# v1.1.0's arm64 layer was built with a drifting cross toolchain (needed
# GLIBC_2.38); v1.1.1 pins cross 0.2.5 and loads on bookworm. Verified:
# observe's full suite (44 packages, -p 1) green against the v1.1.x image.
#
# v1.2.1 (2026-10-04), up from v1.1.1. Atomic ALTER rewrites: ADD/DROP
# COLUMN and ALTER COLUMN TYPE now adopt the new layout, rewrite every
# row, rebuild the indexes and verify every tuple decodes as one
# all-or-nothing step; any failure restores the pre-change state. Fixes
# the L9 live-store corruption (llm_traces undecodable after several
# in-transaction ALTER ADDs; half-applied ALTERs recorded as migrated —
# the token_source-yes/cost_source-no state on infra-home). The live
# store still needs the in-place catalog repair from
# Neutron's nucleus/docs/HANDOVER_TEPLOY_OBSERVE_ADD_COLUMN.md before
# migration 054 can re-run. New refusal to respect in migration files:
# ALTER on a table this transaction has already written to errors —
# order ALTERs before DML. CREATE UNIQUE INDEX now enforces uniqueness
# on populated tables. Verified: observe's full suite (48 packages,
# -p 1) green against the v1.2.1 image, plus a fresh migration boot
# (001-056).
#
# v1.2.2 (2026-10-04), up from v1.2.1. Closes the L9-era gap this repo
# hit in the wild: 1.2.1 enforced CREATE UNIQUE INDEX only on the
# parsed path — simple-protocol autocommit statements (wire-level
# INSERT/UPDATE) took the SQL OLTP fast path, which bypassed the check,
# so duplicates landed under a unique index (neutron#69). The fast path
# now declines unique-indexed tables (fail-closed) and violations come
# back Postgres-shaped (23505, against the index name) — cross-replica
# survey-response dedupe can move from upstream-blocked to engine-
# enforced. Also boxes the statement dispatcher's futures (-27% debug
# stack). Verified: observe's full suite green against the v1.2.2
# image, fresh migration boot included.
#
# v1.3.0 (2026-10-06), up from v1.2.2. The polyglot release: DROP SCHEMA
# (RESTRICT) — cleanupResidue schemas can now be dropped; TEXT columns
# advertise PostgreSQL's OID 25 (strict clients that verify wire type
# identity now pass); quoted-identifier role names stored by value;
# BS.1770-4 loudness + GCC-PHAT sync; qualified ORM SDKs in all three
# languages. Verified: observe's full suite green against the v1.3.0
# image, fresh migration boot included.
image: ghcr.io/neutron-build/nucleus:v1.3.0
port: 5432
# Hard cgroup cap, and the only bound the kernel enforces.
# NUCLEUS_MAX_MEMORY_MB below is the engine's accounting of its OWN
# allocations: a leak in a path it does not count sails past it. This
# engine once reached 30 GB RSS with that budget set to 16 GB and the host
# OOM killer took an unrelated service down with it. Keep this ABOVE the
# engine budget so graceful eviction runs first and the kernel cap is only
# the backstop; size both to the host.
memory: 10g
env:
# Single-node sidecar on a private network: allow no-auth SQL and the
# cluster/replication transports to bind non-loopback without a token.
# Auth: the Tailscale overlay IS the transport boundary for this
# single-node deployment. SCRAM via teploy secret injection is the
# follow-up (tracked); the exposed-password incident is closed (rotated).
NUCLEUS_PASSWORD: "secret:NUCLEUS_PASSWORD"
# Memory budget shared by all engine subsystems. The engine defaults to
# 512 MB and REJECTS WRITES at 90% RSS — far too small once a real
# dataset is resident (a ~5 GB observe store idles above 3 GB RSS).
# Size to roughly 2x resident data, below host headroom.
NUCLEUS_MAX_MEMORY_MB: "8192"
volumes:
nucleus-data: /data
# Self-hosted S3: the backup + restore-test target for this stack (no
# external object storage needed). Loopback-published so the host-side
# `aws` CLI that teploy backup shells out to can reach it; never exposed
# publicly. Console (:9001) stays network-internal.
minio:
image: minio/minio:latest
port: 9000
command: server /data --console-address :9001
publish:
- 127.0.0.1:9100:9000
env:
MINIO_ROOT_USER: teploy-backup
MINIO_ROOT_PASSWORD: secret:MINIO_ROOT_PASSWORD
volumes:
minio-data: /data
env:
OBSERVE_ADDR: ":3000"
OBSERVE_INGEST_ADDR: ":3001"
# Suite header: the `Teploy [Product v]` switcher in the top-left renders
# only when it knows where its siblings live. These are plain URLs, not
# secrets, and they belong here rather than being set by hand -- a deploy
# from a config that omits them silently drops the switcher, which is
# exactly how it disappeared on 2026-08-25.
TEPLOY_NAV_DASH_URL: "http://100.108.123.49:3456"
# Ship moved to infra-home 2026-08-27 (its sandboxes run on compute-1, which
# is what lets its control plane sit beside the forge).
TEPLOY_NAV_SHIP_URL: "http://100.108.123.49:7460"
# The teploy secret OBSERVE_NUCLEUS_URL (with SCRAM password) overrides
# this default at deploy time via the env-file merge.
OBSERVE_NUCLEUS_URL: "postgres://observe-nucleus:5432/observe"
OBSERVE_SITE_ID: "default"
OBSERVE_SESSION_SALT: "${TEPLOY_OBSERVE_SESSION_SALT}"
OBSERVE_JWT_SECRET: "${TEPLOY_OBSERVE_JWT_SECRET}"
OBSERVE_ADMIN_USER: "admin"
OBSERVE_ADMIN_PASSWORD: "${TEPLOY_OBSERVE_ADMIN_PASSWORD}"
OBSERVE_BUFFER_SIZE: "100000"
OBSERVE_FLUSH_INTERVAL_MS: "2000"
OBSERVE_FLUSH_SIZE: "500"
OBSERVE_RAW_RETENTION_DAYS: "30"
OBSERVE_HOURLY_RETENTION_DAYS: "365"
OBSERVE_SEED_DEMO: "false"
# Storage ownership is provisioned before container start by the supported
# CLI contract below, under its deployment fence, using root/passwordless sudo.
# It accepts an empty root or an already matching populated root; it refuses
# mismatched populated roots and validates parent/final symlinks.
# To migrate an existing root-owned store, as root provision ONLY this directory:
# state=/deployments/observe/volumes/observe-state
# install -d -m 0700 -o 10001 -g 10001 "$state"
# find "$state" -xdev -exec chown -h 10001:10001 {} +
# The second command preserves existing audit keys and WAL bytes and does not
# follow symlinks or cross filesystem boundaries. Keep this path in sync with
# app: above and Teploy's deployment directory if either is customized.
# Do not make this directory world-writable or run the application as root.
volumes:
observe-state: /var/lib/observe
volume_ownership:
observe-state:
uid: 10001
gid: 10001
mode: "0700"
# Optional: Slack webhook on deploy success/failure.
# notifications:
# webhook: https://hooks.slack.com/services/YOUR/WEBHOOK/URL
hooks:
post_deploy: |
echo "Observe deployed to https://${TEPLOY_DOMAIN}/"
echo "Admin: admin / \${OBSERVE_ADMIN_PASSWORD} — change via /settings."