Repository navigation
Expand file tree
/
Copy pathdocker-compose.prod.yml
More file actions
303 lines (295 loc) · 13.7 KB
/
Copy pathdocker-compose.prod.yml
File metadata and controls
303 lines (295 loc) · 13.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
name: spire-codex
# Env shared verbatim by the backend and lake-ingest services: both must see
# the same data dirs and stats knobs so they build the same store shapes.
x-shared-env: &shared-env
DATA_DIR: /data
BETA_DATA_DIR: /data-beta
# Optional cap on how many release versions keep their own per-version
# stats bucket. Empty = code default (unbounded: every release version
# with a run stays in the stats-page version dropdowns). Setting N drops
# versions older than the newest N from the dropdowns and the
# /api/runs/list version filter, in exchange for fewer per-version Elo
# fits and blob shards per rebuild/persist.
ENTITY_STATS_RECENT_VERSIONS_N: ${ENTITY_STATS_RECENT_VERSIONS_N:-}
# on = stats_core merges the lake-built deep item tables (deep_tables.json)
# into the summary docs; off keeps the legacy tables in place.
LAKE_DEEP_TABLES: ${LAKE_DEEP_TABLES:-}
# Startup files -> run_blobs backfill on the first lease cycle after
# boot; only the refresh-lease holder runs it. Set to off to skip (e.g.
# while diagnosing Mongo load).
RUN_BLOBS_STARTUP_BACKFILL: "${RUN_BLOBS_STARTUP_BACKFILL:-on}"
# MongoDB — when MONGO_URL is set, runs_db.py uses the Mongo path. Falls
# through to local /data/runs.db SQLite otherwise.
MONGO_URL: ${MONGO_URL:-}
# Application cache (issue #388). Fail-safe: unset or unreachable ->
# cache calls no-op and the existing data paths run unchanged. Uses the
# container name (not the service alias) because prod and beta share
# the external nginx_web-network and a bare "redis" alias would be
# ambiguous between the two stacks.
REDIS_URL: "${REDIS_URL:-redis://spire-codex-redis:6379/0}"
SENTRY_DSN: ${SENTRY_DSN:-}
ENVIRONMENT: ${ENVIRONMENT:-production}
services:
backend:
image: ptrlrd/spire-codex-backend:latest
container_name: spire-codex-backend
restart: unless-stopped
# Memory budget, 15.6G box: mongo 5G + backend 6G + lake-ingest 5G
# (runs to completion, not resident) + redis 2.5G + frontend 1G. Caps, not
# reservations — real steady usage is well under total, and redis
# never approaches its cap. Each of the 4 web workers holds its own copy
# of the entity-stats snapshot (~4.5G total), which left a 5g cap with
# near-zero headroom: profile-insight walks (all blobs for one account
# in a worker) pushed workers into swap and crawled for 15+ minutes
# (2026-08-27). 6g gives walks real room; the lasting fix is retiring
# the per-worker snapshot copies once entity brackets serve from the
# lake.
# With lake-ingest moved to its own box, its 5g budget can go to the
# backend: set BACKEND_MEM_LIMIT=8g in the box .env.
mem_limit: ${BACKEND_MEM_LIMIT:-6g}
# Multiple workers so a burst of POST /api/runs (Overwolf launch
# backfills) doesn't serialise behind a single thread and block
# everything else (GET /api/runs/stats, /api/runs/list etc.).
#
# The shell wrapper sets up prometheus_client's multiprocess
# collector dir before uvicorn forks workers: each worker writes
# its counters into per-pid files under PROMETHEUS_MULTIPROC_DIR,
# and /metrics aggregates across them. Without this, each scrape
# returns one random worker's view, which Prometheus reads as a
# counter reset on every cross-worker bounce — `rate()` then
# multiplies by 4-ish and dashboards show wildly inflated traffic.
# The dir must be empty at boot so we don't include stale files
# from previously-exited worker pids.
command:
- sh
- -c
- >-
mkdir -p "$$PROMETHEUS_MULTIPROC_DIR" &&
rm -rf "$$PROMETHEUS_MULTIPROC_DIR"/* &&
exec uvicorn app.main:app --host 0.0.0.0 --port 8000 --workers ${BACKEND_WEB_WORKERS:-4} --timeout-worker-healthcheck ${BACKEND_WORKER_HEALTHCHECK:-120}
volumes:
- ./data:/data
- ./data-beta:/data-beta:ro
# The static tree (gallery PNG sources, ZIP downloads, audio) is
# bind-mounted instead of baked into the image; it arrives via git
# pull at deploy, exactly like data/. Image bytes themselves are
# served from the CDN (nginx 301s /static/images/*).
- ./backend/static:/app/static:ro
- ./lake:/lake:ro
- /opt/spire-sim:/opt/spire-sim:ro
- ./.secrets:/secrets:ro
environment:
<<: *shared-env
FEEDBACK_WEBHOOK_URL: ${FEEDBACK_WEBHOOK_URL:-}
# Cloudflare purge from the admin cache tab. Same token/zone the
# autodeploy script uses; set them in the box's .env.
CF_TOKEN: ${CF_TOKEN:-}
CF_ZONE: ${CF_ZONE:-}
# Umami analytics on the admin overview, proxied server-side.
UMAMI_URL: ${UMAMI_URL:-}
UMAMI_USERNAME: ${UMAMI_USERNAME:-}
UMAMI_PASSWORD: ${UMAMI_PASSWORD:-}
UMAMI_WEBSITE_ID: ${UMAMI_WEBSITE_ID:-}
GUIDE_WEBHOOK_URL: ${GUIDE_WEBHOOK_URL:-}
RESEND_API_KEY: ${RESEND_API_KEY:-}
UNINSTALL_FORWARD_FROM: ${UNINSTALL_FORWARD_FROM:-}
UNINSTALL_FORWARD_TO: ${UNINSTALL_FORWARD_TO:-}
R2_ACCESS_KEY_ID: ${R2_ACCESS_KEY_ID:-}
R2_SECRET_ACCESS_KEY: ${R2_SECRET_ACCESS_KEY:-}
R2_ENDPOINT: ${R2_ENDPOINT:-}
R2_BUCKET: ${R2_BUCKET:-}
R2_PUBLIC_BASE_URL: ${R2_PUBLIC_BASE_URL:-}
ADMIN_IDS: ${ADMIN_IDS:-}
# Server-side secret (in neither repo) so the stored daily DAU hashes
# aren't brute-forceable back to steam ids. The count works without it,
# but set it. See routers/telemetry.py.
TELEMETRY_SALT: ${TELEMETRY_SALT:-}
# Min times a relic must be offered at an Ancient screen before its
# community take-rate is published to the in-game tip. Empty here ->
# the code default (20); the beta .env sets it to 1 so take-rates show
# on a small corpus. See services/community_stats.py.
COMMUNITY_ANCIENT_MIN_OFFERS: ${COMMUNITY_ANCIENT_MIN_OFFERS:-}
STEAM_WEB_API_KEY: ${STEAM_WEB_API_KEY:-}
# Thank You page: Ko-fi webhook token, GitHub contributor sources,
# logins kept off the public list, optional token for the rate limit.
KOFI_VERIFICATION_TOKEN: ${KOFI_VERIFICATION_TOKEN:-}
THANKS_GITHUB_REPOS: ${THANKS_GITHUB_REPOS:-}
THANKS_GITHUB_EXCLUDE: ${THANKS_GITHUB_EXCLUDE:-}
GITHUB_TOKEN: ${GITHUB_TOKEN:-}
OVERWOLF_STORE_ID: ${OVERWOLF_STORE_ID:-}
OVERWOLF_EXTENSION_ID: ${OVERWOLF_EXTENSION_ID:-}
SUPPORTER_EMAIL_SALT: ${SUPPORTER_EMAIL_SALT:-}
GITHUB_APP_ID: ${GITHUB_APP_ID:-}
GITHUB_APP_INSTALLATION_ID: ${GITHUB_APP_INSTALLATION_ID:-}
GITHUB_APP_REPO: ${GITHUB_APP_REPO:-}
GITHUB_APP_PRIVATE_KEY_PATH: ${GITHUB_APP_PRIVATE_KEY_PATH:-/secrets/knowledge-demon.private-key.pem}
# Card-render QA mount. Set to /data/qa once the rendered PNGs +
# index.html are rsync'd onto the host volume. Empty/missing →
# the /qa endpoint isn't registered.
QA_DIR: ${QA_DIR:-}
# User accounts (Steam/Discord/Twitch OAuth + JWT sessions). Values live
# in the host .env; passed through so the auth endpoints work.
JWT_SECRET: ${JWT_SECRET:-}
DISCORD_CLIENT_ID: ${DISCORD_CLIENT_ID:-}
DISCORD_CLIENT_SECRET: ${DISCORD_CLIENT_SECRET:-}
TWITCH_CLIENT_ID: ${TWITCH_CLIENT_ID:-}
TWITCH_CLIENT_SECRET: ${TWITCH_CLIENT_SECRET:-}
# Patreon link for the paid API-key tier (auth_patreon router). Unset =
# the feature stays dormant (connect bounces with patreon_unconfigured).
PATREON_CLIENT_ID: ${PATREON_CLIENT_ID:-}
PATREON_CLIENT_SECRET: ${PATREON_CLIENT_SECRET:-}
PATREON_WEBHOOK_SECRET: ${PATREON_WEBHOOK_SECRET:-}
FRONTEND_URL: "${FRONTEND_URL:-https://spire-codex.com}"
SPIRE_CODEX_PUBLIC_BASE: "${SPIRE_CODEX_PUBLIC_BASE:-https://spire-codex.com}"
# prometheus_client multiprocess mode. Set in the env (not the
# CMD) so it's exported into every worker as a real env var
# before prometheus_client is imported. The dir is recreated +
# cleaned by the CMD shell wrapper above.
#
# Side effect: prometheus_client disables its built-in per-process
# collectors (process_cpu_seconds_total, process_resident_memory_bytes,
# gc, platform) under multiproc, because they only make sense
# per-pid. Any Grafana panel reading those metrics must switch
# to node_exporter (`node_cpu_*`, `node_memory_*`) or cAdvisor
# (`container_cpu_*`, `container_memory_*`).
PROMETHEUS_MULTIPROC_DIR: /tmp/prom_multiproc
# The heavy stats walk runs only in the rebuilder service below, so a
# web deploy recreating this container no longer kills an in-flight
# rebuild (which is how the snapshot sat stale for days). Set
# BACKEND_STATS_REFRESHER=on in the box .env to revert to the old
# single-container behavior.
STATS_REFRESHER: "${BACKEND_STATS_REFRESHER:-off}"
# Workers AI credentials for /api/search/semantic and the
# build_embeddings corpus script; unset leaves semantic search off.
WORKERS_AI_TOKEN: ${WORKERS_AI_TOKEN:-}
CF_ACCOUNT_ID: ${CF_ACCOUNT_ID:-}
# Ordering only (no health condition): the backend must boot fine with
# Redis down, since the cache layer is fail-safe by design.
depends_on:
- redis
stop_grace_period: 15s
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request;urllib.request.urlopen('http://127.0.0.1:8000/health',timeout=3)"]
interval: 10s
timeout: 5s
retries: 6
start_period: 30s
networks:
- nginx_web-network
logging:
driver: json-file
options:
max-size: "10m"
max-file: "5"
# Analytics-lake ingest (incremental extract + parquet + store rebuild).
# profiles: never started by `up -d`; host cron runs it every 6 hours:
# 17 */6 * * * cd /var/www/spire-codex && docker compose -f docker-compose.prod.yml run --rm lake-ingest >> /var/log/lake-ingest.log 2>&1
# Overlap-safe: the ingest takes /lake/ingest.lock (flock) and exits if a
# prior run still holds it. Keep 6h until three consecutive cycles publish
# complete generations (see /lake/ingest_metrics.jsonl) and the max cycle
# time sits well under the interval; only then tighten the cadence.
lake-ingest:
image: ptrlrd/spire-codex-backend:latest
profiles: ["ops"]
entrypoint: ["python", "/lab/ingest.py"]
env_file: .env
environment:
DATA_DIR: /data
# Build frame.parquet from the lake (one COPY) instead of the 3h Mongo
# doc walk. Flip in .env after lab/validate_frame_lake.py passes.
FRAME_FROM_LAKE: ${FRAME_FROM_LAKE:-}
# Defaults fit the shared serving box; the dedicated ingest box raises
# them in its .env. The container cap must clear LAKE_BUILD_MEMORY plus
# ~1.5g of Python-side headroom, and cpus must not exceed the box's
# cores or docker refuses to start the container.
mem_limit: ${LAKE_INGEST_MEM_LIMIT:-5g}
# Serving always wins the box: 5 of the 8 cores at most (the hostname's
# "4vcpu" is the droplet's birth name — nproc says 8), and an eighth of
# the default CPU weight so under contention the kernel hands cycles to
# the backend/Mongo first. An idle box still runs the ingest full speed.
cpus: ${LAKE_INGEST_CPUS:-5}
cpu_shares: 128
volumes:
- ./lab:/lab:ro
- ./data:/data:ro
- ./lake:/lake
- /opt/spire-sim:/opt/spire-sim:ro
networks:
- nginx_web-network
redis:
image: redis:7-alpine
# Must stay ABOVE redis's own --maxmemory so it evicts by LRU (its
# intended behavior) instead of being OOM-killed at the cgroup boundary.
# The repo says 2gb but the running prod instance reports a 3gb cap
# (config drift, found 2026-08-23), so this clears the higher of the two.
mem_limit: 3500m
container_name: spire-codex-redis
# Cache only: hard memory cap with LRU eviction, no persistence (no AOF,
# no RDB snapshots). Everything in here is rebuildable from Mongo / the
# run files on a miss, so losing it on restart costs one warmup, not data.
command: redis-server --maxmemory 2gb --maxmemory-policy allkeys-lru --appendonly no --save ""
# Redis has no auth: REDIS_BIND stays loopback or the VPC private IP
# for the ingest box, never 0.0.0.0.
ports:
- "${REDIS_BIND:-127.0.0.1}:6379:6379"
restart: unless-stopped
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 30s
timeout: 3s
retries: 3
networks:
- nginx_web-network
logging:
driver: json-file
options:
max-size: "10m"
max-file: "5"
frontend:
image: ptrlrd/spire-codex-frontend:latest
container_name: spire-codex-frontend
restart: unless-stopped
# Next server sits ~400M plus up to 256M of in-memory page cache
# (cacheMaxMemorySize). The heap is pinned below so the cache can't
# crowd it: a 1g cap gave Node a ~512M heap and the old 512M cache
# filled it, crash-looping the frontend (2026-10-07).
mem_limit: ${FRONTEND_MEM_LIMIT:-1536m}
stop_grace_period: 15s
depends_on:
- backend
volumes:
- ./data:/data:ro
- next-static:/app/.next/static
- next-cache:/app/.next/cache/fetch-cache
environment:
- HOSTNAME=0.0.0.0
- NODE_OPTIONS=--max-old-space-size=${FRONTEND_HEAP_MB:-1024}
- API_INTERNAL_URL=http://spire-codex-backend:8000
- NEXT_PUBLIC_SITE_URL=https://spire-codex.com
healthcheck:
test:
[
"CMD",
"node",
"-e",
"fetch('http://127.0.0.1:3000/robots.txt').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))",
]
interval: 10s
timeout: 5s
retries: 6
start_period: 20s
networks:
- nginx_web-network
logging:
driver: json-file
options:
max-size: "10m"
max-file: "5"
networks:
nginx_web-network:
external: true
volumes:
next-static:
name: spire-codex_next-static
external: true
next-cache: