diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 272020a0..2452e051 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -70,6 +70,21 @@ jobs: # memory ceiling before the box does. run: scripts/lint-systemd-units.sh + - name: Lint - shell syntax + run: scripts/lint-shell.sh + + - name: Lint - docs topology + # Production is one droplet with no VPN; the docs described a + # WireGuard mesh and a monitoring host that never existed. + run: scripts/lint-docs-topology.sh + + - name: Test - backup cron scripts + # deploy/postgres/backup-daily.sh and + # deploy/spaces/sync-cross-region.sh run unattended from root + # crontab and are not covered by go test; both logged nothing + # for four months before anyone noticed. + run: scripts/test-backup-scripts.sh + - name: Test run: go test -trimpath ./... diff --git a/Makefile b/Makefile index fca2929d..3fca8967 100644 --- a/Makefile +++ b/Makefile @@ -72,7 +72,7 @@ assets: ## Copy Primer CSS into internal/web/static/ for embedding. echo "warn: .refs/primer-css/dist not found; run 'git clone https://github.com/primer/css .refs/primer-css' first"; \ fi -ci: lint lint-policy lint-markdown lint-org-plan lint-secret-logs lint-spdx lint-unused lint-migrations lint-systemd-units verify-api-docs test build ## Full CI pipeline (matches .github/workflows/ci.yml). +ci: lint lint-policy lint-markdown lint-org-plan lint-secret-logs lint-spdx lint-unused lint-migrations lint-systemd-units lint-shell lint-docs-topology test-backup-scripts verify-api-docs test build ## Full CI pipeline (matches .github/workflows/ci.yml). @echo "ci: ok" lint-policy: ## Enforce policy-package boundary (no inline auth checks in handlers/git/cmd). @@ -99,6 +99,15 @@ lint-migrations: ## Fail when goose migration numeric versions collide. lint-systemd-units: ## Fail when a shipped systemd unit loses its memory ceiling or collides with the backup window. @scripts/lint-systemd-units.sh +lint-shell: ## Syntax-check every shell script (deploy/ runs unattended from cron). + @scripts/lint-shell.sh + +lint-docs-topology: ## Fail when docs/internal claims a WireGuard mesh outside a marked aspirational block. + @scripts/lint-docs-topology.sh + +test-backup-scripts: ## Functional test for the backup cron scripts (logging, heartbeat, failure exit). + @scripts/test-backup-scripts.sh + verify-api-docs: ## Fail when an /api/v1 route in code is missing from docs/public/api/. @scripts/verify-api-docs.sh diff --git a/deploy/ansible/roles/postgres/templates/postgresql.conf.j2 b/deploy/ansible/roles/postgres/templates/postgresql.conf.j2 index f4bb5022..a9792974 100644 --- a/deploy/ansible/roles/postgres/templates/postgresql.conf.j2 +++ b/deploy/ansible/roles/postgres/templates/postgresql.conf.j2 @@ -1,23 +1,68 @@ # Managed by Ansible — edits here are overwritten on next deploy. -# Tunes for shithub's MVP single-droplet workload. +# +# Sized for the single-box deployment: 2 vCPU / 3.9 GB shared with +# shithubd web + worker (~200 MB steady state after the Phase 2 +# renderer fix, ceilinged at 1.6/2.0 GB and 0.75/1.0 GB by systemd), +# Caddy, Alloy (~320 MB) and the backup crons (rclone peaks at +# 1.0-1.6 GB). Postgres does NOT get the usual 25%-of-RAM share here, +# because the 25% rule assumes a dedicated database host. +# +# NOTE: this template has never been applied to shithub-app. The live +# box runs Debian defaults (shared_buffers 128MB, work_mem 4MB, +# max_connections 100, no pg_stat_statements). Applying it needs a +# Postgres RESTART, not a reload. Read docs/internal/db.md before +# doing it. -listen_addresses = 'localhost' # over WG only; no public Postgres +listen_addresses = 'localhost' # loopback only; no public Postgres port = 5432 -max_connections = 100 -shared_buffers = 512MB # ~25% of 2GB droplet RAM -effective_cache_size = 1500MB # rough working set hint -maintenance_work_mem = 64MB -work_mem = 8MB + +# 60, not 100. Actual demand: web pool 10 (db.max_conns), worker +# 6 (SHITHUB_WORKERS=4 + 2), cron/hook/ssh/admin invocations at 2-4 +# each and short-lived. 60 leaves ~35 of headroom over the steady +# ~19 observed on 2026-09-02 and caps the worst case at a bounded +# per-backend footprint; 100 backends each able to take work_mem is +# how a shared box gets OOM-killed. +max_connections = 60 + +# 256MB, not 512MB. Double the live Debian default (which is what the +# 447 MB aggregate RSS on the box was measured against), but small +# enough that Postgres + shithubd + Alloy + a 1.6 GB rclone still fit +# under 3.9 GB with the 4 GB swapfile as slack rather than as a +# working surface. +shared_buffers = 256MB + +# What the kernel is likely caching, not an allocation. The DB is +# ~1 GB and the page cache on this box is regularly squeezed by +# rclone and AIDE, so 1GB is the honest hint; overstating it pushes +# the planner toward index scans that are not actually cheap here. +effective_cache_size = 1GB + +maintenance_work_mem = 64MB # VACUUM/CREATE INDEX; one at a time + +# 4MB (the Debian default the box already runs). This is per sort/hash +# *node*, not per connection, so the ceiling is roughly +# max_connections x nodes x work_mem. At 60 connections a bump to 8MB +# doubles a tail risk we cannot afford on a shared box; raise it per +# session (SET LOCAL work_mem) for a specific heavy query instead. +work_mem = 4MB wal_level = replica # required for WAL archiving (S37) archive_mode = on archive_command = '/usr/local/bin/shithub-pg-archive %p %f' archive_timeout = 60 # at least one segment per minute checkpoint_completion_target = 0.9 -# pg_stat_statements (S36 perf-pass requirement; S37 deploy installs). +# pg_stat_statements. Not installed on the live box — the extension +# needs this preload line AND a restart AND `CREATE EXTENSION`, and +# none of that has happened, which is why the 2026-09-02 outage +# investigation had no query attribution to work from. +# +# track = top (not all): `all` also records statements executed inside +# PL/pgSQL function bodies, which multiplies entries and shared-memory +# pressure for attribution we do not need. Top-level statements are +# what maps back to a handler or a job. shared_preload_libraries = 'pg_stat_statements' pg_stat_statements.max = 5000 -pg_stat_statements.track = all +pg_stat_statements.track = top # Logging — minimal noise, slow-query attention. log_destination = 'stderr' diff --git a/deploy/audit/check-droplet-drift.sh b/deploy/audit/check-droplet-drift.sh index 910ac829..c8cafcd9 100755 --- a/deploy/audit/check-droplet-drift.sh +++ b/deploy/audit/check-droplet-drift.sh @@ -37,6 +37,11 @@ declare -a MANAGED=( "/etc/alloy/credentials.env::TEMPLATE" "/etc/systemd/system/alloy.service.d/shithub.conf::TEMPLATE" "/etc/postgresql/16/main/conf.d/99_shithub_archive.conf::TEMPLATE" + # Known drift, tracked deliberately: the box runs Debian defaults + # here. deploy/ansible/roles/postgres/templates/postgresql.conf.j2 + # has never been applied (see docs/internal/db.md); applying it needs + # a Postgres restart in the 05:15-06:00 window, not a `make deploy`. + "/etc/postgresql/16/main/postgresql.conf::TEMPLATE" "/etc/aide/aide.conf.d/99_shithub_exclude::deploy/ansible/roles/base/files/aide-shithub.conf" "/etc/cron.daily/aide::TEMPLATE" "/etc/caddy/Caddyfile::TEMPLATE" @@ -44,11 +49,33 @@ declare -a MANAGED=( "/etc/fail2ban/filter.d/shithubd-auth.conf::TEMPLATE" "/etc/systemd/system/shithubd-web.service::TEMPLATE" "/etc/systemd/system/shithubd-worker.service::TEMPLATE" + # Known drift: web.env carries a hand-edited Stripe block that is + # not in web.env.j2. A `make deploy` re-renders this file from the + # template and WILL drop those keys — see KNOWN_DRIFT below. "/etc/shithub/web.env::TEMPLATE" "/etc/shithub/worker.env::TEMPLATE" "/etc/ssh/sshd_config::TEMPLATE" ) +# Paths whose drift is known and accepted. Reported as KNOWN rather +# than silently passing: the point is that the next operator reads the +# reason instead of rediscovering it. +declare -a KNOWN_DRIFT=( + "/etc/postgresql/16/main/postgresql.conf::postgresql.conf.j2 has never been applied; needs a restart, see docs/internal/db.md" + "/etc/shithub/web.env::hand-edited Stripe block not present in web.env.j2; a full deploy re-renders this file and drops it" +) + +known_drift_note() { + local entry + for entry in "${KNOWN_DRIFT[@]}"; do + if [ "${entry%%::*}" = "$1" ]; then + printf '%s' "${entry##*::}" + return 0 + fi + done + return 1 +} + DRIFT_COUNT=0 # Build a single-shot SSH script that returns md5 + stat for every @@ -94,7 +121,11 @@ while IFS='|' read -r dpath status remote_md5 remote_stat; do fi if [ "$src" = "TEMPLATE" ]; then - printf "%-60s \033[36m%-10s\033[0m %s (template — manual check)\n" "$dpath" "TEMPLATE" "$remote_stat" + if note=$(known_drift_note "$dpath"); then + printf "%-60s \033[33m%-10s\033[0m %s\n" "$dpath" "KNOWN" "$note" + else + printf "%-60s \033[36m%-10s\033[0m %s (template — manual check)\n" "$dpath" "TEMPLATE" "$remote_stat" + fi continue fi @@ -120,4 +151,6 @@ if [ "$DRIFT_COUNT" -gt 0 ]; then exit 1 fi echo "No drift detected on copy: files. TEMPLATE rows still need manual review." +echo "KNOWN rows are accepted drift with a reason attached — do not 'fix' one" +echo "by running a full deploy without reading the note." exit 0 diff --git a/deploy/monitoring/README.md b/deploy/monitoring/README.md new file mode 100644 index 00000000..83fc7275 --- /dev/null +++ b/deploy/monitoring/README.md @@ -0,0 +1,97 @@ +# `deploy/monitoring/` — not deployed + +**Nothing in this directory runs in production**, with one exception +noted below. These files describe a self-hosted Prometheus + Loki + +Alertmanager + Grafana monitoring host that has never existed. They +are kept because they are the expression catalogue for the alerts we +want and the starting point if we ever outgrow Grafana Cloud — not +because they are live. + +## What actually runs + +shithub.sh is a single droplet. The `monitoring-client` Ansible role +installs `node_exporter` (`127.0.0.1:9100`) and **Grafana Alloy**, +which scrapes node_exporter and `shithubd web`'s +`127.0.0.1:8080/metrics` and `remote_write`s to **Grafana Cloud**. +Push-only; no inbound monitoring port; no local Prometheus, Grafana, +Loki or Alertmanager process. See `docs/internal/deploy.md` and +`docs/internal/runbooks/observability.md`. + +## File by file + +| Path | Status | +|---|---| +| `prometheus/prometheus.yml` | **Inert.** Scrapes six targets on a `10.50.0.0/24` WireGuard mesh that does not exist, and points `alerting.alertmanagers` at `10.50.0.10:9093`. No process reads it. | +| `prometheus/rules.yml` | **Inert.** No Prometheus loads it and there is no Alertmanager to route it. Useful as PromQL source material; see the caveats below before copying an expression. | +| `alertmanager/alertmanager.yml` | **Inert.** No Alertmanager is installed. The SMTP/webhook receivers are placeholders. | +| `loki/loki-config.yaml` | **Inert.** No Loki, and nothing ships logs — there is no Promtail and the Alloy config has no logs component. Logs are journald-only. | +| `grafana/dashboards/*.json` | **Import by hand.** These work against the Grafana Cloud stack via Dashboards → New → Import. Panel queries are mirrored in `runbooks/observability.md`. | +| `grafana/provision-actions-alerts.sh` | **Live tool.** The one thing here that talks to production. It creates/updates a Grafana-*managed* alert rule in the Cloud stack over the HTTP API. Which rules are currently provisioned is state in Grafana Cloud, not in this repo. | + +## Which documented alerts therefore do not exist + +Every rule in `prometheus/rules.yml` is unfired: + +- `shithubd-availability`: `ShithubdWebDown`, `ShithubdWorkerDown`, + `PostgresDown` +- `shithubd-latency`: `HighRequestLatencyP95`, `HighDBQueryRate` +- `shithubd-jobs`: `JobBacklogGrowing`, `WebhookDeliveryFailing` +- `shithubd-billing`: `BillingWebhookFailureRateHigh`, + `BillingWebhookBacklogHigh`, `BillingWebhookFailedReceipt`, + `BillingCheckoutFailures`, `BillingSeatDrift`, + `BillingPastDuePrincipals`, `BillingQuotaOverage` +- `shithubd-actions`: `ActionsRunnerHeartbeatStale`, + `ActionsRunnerIdleWithAssignedJobs`, `ActionsQueueDepthHigh`, + `ActionsRunDurationP99Regressed`, + `ActionsLogScrubberPossiblyMissing` +- `shithubd-backups`: `BackupOverdue` + +The runbook anchors these rules reference +(`docs/internal/runbooks/incidents.md`, `backups.md`, +`stripe-billing.md`) are still correct as *procedures*; they are just +not triggered by anything. What does page today is the DigitalOcean +droplet + uptime alert set (`docs/internal/runbooks/alerts.md`). + +## Caveats before copying an expression into Grafana Cloud + +1. **Job labels differ.** Alloy scrapes with `job="shithubd"` and + `job="node"`. Anything selecting `job="shithubd-web"`, + `job="shithubd-worker"`, `job="postgres"` or `job="caddy"` matches + nothing. +2. **The worker is not scraped at all.** `shithubd worker` starts no + HTTP listener, so `ShithubdWorkerDown` and every `shithub_worker_*` + expression are unsatisfiable regardless of relabelling. +3. **No Postgres exporter.** `PostgresDown` and any `pg_*` series have + no source. WAL-archive health is checked by the hourly + `shithub-verify-wal-archive` cron job, which logs and journals. +4. **`BackupOverdue` names a metric that never existed.** It selects + `shithubd_backup_last_success_seconds`; our metric namespace is + `shithub_`, and no such gauge was ever emitted. It is now real as + `shithub_backup_last_success_seconds{job="daily"|"spaces-sync"}`, + sourced from the heartbeat files the backup scripts write — see + `docs/internal/runbooks/backups.md`. + +## What it would take to use these files + +Roughly, in order: + +1. A second droplet for the monitoring host, and a private network or + VPN between it and the app box — `deploy/ansible/roles/wireguard/` + exists but is not run by any current inventory, and the app box has + no `wg0`. +2. A monitoring-host play. There is none in this repo; `deploy.md` + used to claim it lived "outside this repo" and it does not exist + either. +3. Exporters for the targets the scrape config assumes: + `postgres_exporter`, Caddy admin metrics, and a metrics listener in + `shithubd worker` (a code change). +4. Relabelling `prometheus.yml` from mesh addresses to real ones and + fixing the job names, or relabelling `rules.yml` to match Alloy's. +5. Real Alertmanager receivers — the SMTP host, credentials file and + pager webhook in `alertmanager.yml` are all placeholders. +6. A logs pipeline if `loki-config.yaml` is to mean anything: an Alloy + `loki.source.journal` component plus a Loki endpoint. + +Until at least steps 1–4 are done, the cheaper path for any alert you +want is a Grafana-managed rule in the existing Cloud stack, added to +`grafana/provision-actions-alerts.sh` so it stays reproducible. diff --git a/deploy/monitoring/prometheus/rules.yml b/deploy/monitoring/prometheus/rules.yml index fa20f2aa..4c92d8d0 100644 --- a/deploy/monitoring/prometheus/rules.yml +++ b/deploy/monitoring/prometheus/rules.yml @@ -196,8 +196,13 @@ groups: - name: shithubd-backups interval: 5m rules: + # `shithubd_backup_last_success_seconds` never existed — our + # namespace is `shithub_`. The gauge is real now, sourced from + # the heartbeat files the backup scripts write + # (internal/infra/metrics/backupobserver.go). It is absent until + # the first success on a host, hence the absent() arm. - alert: BackupOverdue - expr: time() - shithubd_backup_last_success_seconds > 60 * 60 * 30 + expr: absent(shithub_backup_last_success_seconds{job="daily"}) or time() - shithub_backup_last_success_seconds{job="daily"} > 60 * 60 * 30 for: 0m labels: {severity: page} annotations: diff --git a/deploy/postgres/backup-daily.sh b/deploy/postgres/backup-daily.sh index d7142121..510003f8 100755 --- a/deploy/postgres/backup-daily.sh +++ b/deploy/postgres/backup-daily.sh @@ -8,8 +8,13 @@ # anything older than 30 days; PITR rolls forward from the WAL # archive (see archive_command.sh). # -# Exit non-zero on any failure so the systemd timer surfaces it -# (OnFailure= → alertmanager). +# Exit non-zero on any failure. Nothing watches that exit code today +# — there is no Alertmanager and no systemd timer, this runs from root +# crontab at 03:17 with output appended to /var/log/shithub-backup.log +# — so the run also brackets itself with timestamped start/end lines +# and drops a heartbeat file on success. That log was 0 bytes from +# 2026-05-10 to 2026-09-02 precisely because a silent success and a +# silent absence look identical. set -euo pipefail @@ -19,6 +24,28 @@ LOCAL_DIR="${SHITHUB_BACKUP_LOCAL:-/var/backups/shithub}" STAMP="$(date -u +%Y%m%dT%H%M%SZ)" NAME="${DB}-${STAMP}.dump" +# Read by shithub_backup_last_success_seconds{job="daily"} — see +# internal/infra/metrics/backupobserver.go. Epoch seconds, one line. +HEARTBEAT="${SHITHUB_BACKUP_HEARTBEAT:-/var/lib/shithub/backup-last-success}" + +ts() { date -u +%Y-%m-%dT%H:%M:%SZ; } + +# Bracket the run. `set -e` makes most failures land here with a +# non-zero status, and the trap re-raises it so cron/systemd still see +# a failed run: the log line is an addition, never a swallow. +on_exit() { + local rc=$? + if [ "$rc" -eq 0 ]; then + echo "[$(ts)] backup-daily end status=ok exit=0 dump=$NAME" + else + echo "[$(ts)] backup-daily end status=FAILED exit=$rc dump=$NAME" + fi + return "$rc" +} +trap on_exit EXIT + +echo "[$(ts)] backup-daily start db=$DB bucket=$BUCKET" + mkdir -p "$LOCAL_DIR" # pg_dump as the postgres user via local-socket peer auth. @@ -38,3 +65,11 @@ rclone --config /etc/rclone-shithub.conf --s3-no-check-bucket \ # Local retention: keep the last 7 dumps; bucket lifecycle handles # the long tail. ls -1t "$LOCAL_DIR"/*.dump 2>/dev/null | tail -n +8 | xargs -r rm -f + +# Success only. Everything above is `set -e`-guarded, so reaching this +# line means the dump was taken, verified with pg_restore --list, and +# uploaded. Written via a temp file + mv so a reader never sees a +# half-written timestamp. +mkdir -p "$(dirname "$HEARTBEAT")" +printf '%s\n' "$(date -u +%s)" > "$HEARTBEAT.tmp" +mv "$HEARTBEAT.tmp" "$HEARTBEAT" diff --git a/deploy/spaces/sync-cross-region.sh b/deploy/spaces/sync-cross-region.sh index 3cfd261c..96cb648a 100755 --- a/deploy/spaces/sync-cross-region.sh +++ b/deploy/spaces/sync-cross-region.sh @@ -35,23 +35,55 @@ DR="${SHITHUB_DR_BUCKET:-spaces-dr:shithub-backups-dr}" WAL_PRIMARY="${SHITHUB_WAL_BUCKET:-spaces-prod:shithub-wal}" WAL_DR="${SHITHUB_WAL_DR_BUCKET:-spaces-dr:shithub-wal-dr}" -LOG="/var/log/shithub/spaces-sync.log" +LOG="${SHITHUB_SPACES_SYNC_LOG:-/var/log/shithub/spaces-sync.log}" mkdir -p "$(dirname "$LOG")" +# Read by shithub_backup_last_success_seconds{job="spaces-sync"} — see +# internal/infra/metrics/backupobserver.go. Epoch seconds, one line. +HEARTBEAT="${SHITHUB_SPACES_SYNC_HEARTBEAT:-/var/lib/shithub/spaces-sync-last-success}" + ts() { date -u +%Y-%m-%dT%H:%M:%SZ; } -{ - echo "[$(ts)] sync start" +# Status lines go to BOTH the script's own log and stdout, which cron +# appends to /var/log/shithub-spaces-sync.log. Before this, everything +# was swallowed into $LOG and the cron-redirected file sat at 0 bytes +# from 2026-05-10 to 2026-09-02 — indistinguishable from "the job was +# never installed". rclone's own chatter stays in $LOG only; it is far +# too noisy for the cron log. +status() { printf '[%s] %s\n' "$(ts)" "$*" | tee -a "$LOG"; } + +# `set -e` routes any rclone failure here with its exit status, and +# the trap re-raises it, so a failed sync is still a non-zero exit for +# cron and for the flock wrapper. +on_exit() { + local rc=$? + if [ "$rc" -eq 0 ]; then + status "spaces-sync end status=ok exit=0" + else + status "spaces-sync end status=FAILED exit=$rc" + fi + return "$rc" +} +trap on_exit EXIT + +status "spaces-sync start primary=$PRIMARY dr=$DR wal=$WAL_PRIMARY wal_dr=$WAL_DR" + +# Redirected per-command rather than as one `{ ... } >> "$LOG"` block: +# when `set -e` aborts inside a redirected compound command, the EXIT +# trap inherits that redirection and the FAILED line lands in $LOG +# instead of the cron log — exactly the stream we are trying to fix. - # Small bucket: --fast-list is cheap and cuts API calls. - rclone --config /etc/rclone-shithub.conf --s3-no-check-bucket \ - copy --transfers 8 --checkers 16 --fast-list \ - "$PRIMARY" "$DR" +# Small bucket: --fast-list is cheap and cuts API calls. +rclone --config /etc/rclone-shithub.conf --s3-no-check-bucket \ + copy --transfers 8 --checkers 16 --fast-list \ + "$PRIMARY" "$DR" >> "$LOG" 2>&1 - # WAL bucket: no --fast-list, lower concurrency. See header. - rclone --config /etc/rclone-shithub.conf --s3-no-check-bucket \ - copy --transfers 4 --checkers 8 \ - "$WAL_PRIMARY" "$WAL_DR" +# WAL bucket: no --fast-list, lower concurrency. See header. +rclone --config /etc/rclone-shithub.conf --s3-no-check-bucket \ + copy --transfers 4 --checkers 8 \ + "$WAL_PRIMARY" "$WAL_DR" >> "$LOG" 2>&1 - echo "[$(ts)] sync end" -} >> "$LOG" 2>&1 +# Success only: both legs copied without error. +mkdir -p "$(dirname "$HEARTBEAT")" +printf '%s\n' "$(date -u +%s)" > "$HEARTBEAT.tmp" +mv "$HEARTBEAT.tmp" "$HEARTBEAT" diff --git a/docs/internal/architecture.md b/docs/internal/architecture.md index 23ffd0d4..429b9ab7 100644 --- a/docs/internal/architecture.md +++ b/docs/internal/architecture.md @@ -133,50 +133,55 @@ role used at deploy time. ## Deployment topology +shithub.sh runs on **one droplet**. Web, worker, cron, Caddy, +Postgres 16, Grafana Alloy, node_exporter, AIDE and the backup cron +jobs are all co-located on `shithub-app` (2 vCPU / 3.9 GB + a 4 GB +swapfile). + ``` +----------------------------+ public ---> | Caddy (TLS, rate limits) | :443 +----------------------------+ | 127.0.0.1:8080 - +----------------------------+ - | shithubd web (systemd) | - | shithubd worker (systemd) | - | shithubd cron (timer) | - +----------------------------+ - | - +-----------+ +-----------------+ - | Postgres | | Spaces (S3) | - +-----------+ | - WAL archive | - | - daily dumps | - | - LFS / blobs | - +-----------------+ - \ / - \ WireGuard mesh (10.50.0.0/24) - \________ ________/ - | - +----------------+ - | Monitoring | - | Prom/Loki/AM | - | Grafana | - +----------------+ + +--------------------------------------------------------+ + | droplet `shithub-app` | + | shithubd web / worker / cron | + | Postgres 16 (localhost:5432) | + | Caddy, Grafana Alloy, node_exporter, AIDE, cron | + +--------------------------------------------------------+ + | | + v v + Spaces (S3): WAL archive, Grafana Cloud + daily dumps, LFS/blobs, (remote_write, push-only; + Actions logs no inbound port) + + 3 × Actions runner droplets (SSH inbound from the app box only) ``` -Monitoring is on the WireGuard mesh; metrics ports never face the -public internet. See [deploy.md](./deploy.md) for the full -operator guide. +There is no private network and no separate database, backup, or +monitoring host. Postgres binds `localhost`; `/metrics` is served on +`127.0.0.1:8080` and read only by the local Alloy process. -## Observability +The memory budget for this box, the systemd ceilings that enforce it, +and the aspirational multi-host design are in +[deploy.md](./deploy.md). -Three independent channels: +## Observability -- **Structured logs** (`internal/infra/log`) → stdout → journald - → promtail → Loki. -- **Metrics** (Prometheus) at `/metrics`, basic-auth gated in +- **Structured logs** (`internal/infra/log`) → stdout → journald. + There is no log shipping: reading logs means SSH + `journalctl`. +- **Metrics** (Prometheus exposition) at `/metrics`, basic-auth + gateable, scraped locally by Grafana Alloy and pushed to Grafana + Cloud. Only the web process exposes an endpoint; the worker does + not. +- **Tracing** (OTel HTTP) optional, sample-rate controlled, off in prod. -- **Tracing** (OTel HTTP) optional, sample-rate controlled. - **Error reporting** (Sentry-protocol DSN, GlitchTip-compatible). -See [observability.md](./observability.md). +There is **no Alertmanager and no local Prometheus**; the rules in +`deploy/monitoring/prometheus/rules.yml` are an expression catalogue, +not a running alert pipeline. See [observability.md](./observability.md) +and `deploy/monitoring/README.md`. ## What's deliberately not here diff --git a/docs/internal/capacity.md b/docs/internal/capacity.md index ec008c9d..9e1b53fc 100644 --- a/docs/internal/capacity.md +++ b/docs/internal/capacity.md @@ -14,14 +14,27 @@ which is summary-only; this file carries the run-by-run detail. ## Test environment -- Staging compute matches the production reference deployment - (`docs/public/self-host/prerequisites.md`): +> **Reality check (2026-09-02).** Production is **one** 2 vCPU / +> 3.9 GB droplet running web, worker, cron, Caddy and Postgres +> together, with a 4 GB swapfile — see +> `docs/internal/deploy.md#single-box-reference-deployment-what-shithubsh-runs`. +> The multi-host staging shape below was never provisioned; treat +> the per-host split as aspirational and re-baseline these numbers +> against the single box before trusting them. + + + +- Staging compute was specified to match the multi-host reference + deployment (`docs/public/self-host/prerequisites.md`): - 2× web (2 vCPU / 4 GB) - 1× worker (2 vCPU / 4 GB) - 1× postgres (2 vCPU / 8 GB / 100 GB SSD) - 1× backup, 1× monitoring (smaller) - Caddy at the edge with TLS terminated. - WireGuard mesh between hosts. + + + - Staging seeded with synthetic data: - 5,000 users, 50,000 repos, ~500,000 issues, ~1M comments. - Largest repo: ~50 MB packed; 95th percentile under 5 MB. diff --git a/docs/internal/db.md b/docs/internal/db.md index 74f08f68..68f94784 100644 --- a/docs/internal/db.md +++ b/docs/internal/db.md @@ -5,8 +5,12 @@ conventions. Every domain sprint (S05 onwards) follows these. ## Engine and tooling -- **PostgreSQL 16**. Production runs self-hosted on a dedicated block volume - (S37). Local dev runs in `docker-compose` (`make dev-db`). +- **PostgreSQL 16**. Production runs self-hosted **on the application + droplet** — not a dedicated database host — with the data directory on + an attached block volume (`/data/pgdata`) and the server listening on + `localhost` only. It shares 2 vCPU and 3.9 GB with web, worker, cron, + Caddy and Alloy; see `docs/internal/deploy.md`. Local dev runs in + `docker-compose` (`make dev-db`). - **Driver:** `pgx/v5` (`github.com/jackc/pgx/v5`). Native `*pgxpool.Pool` for app code; the `stdlib` adapter is reserved for libraries that demand `*sql.DB`. @@ -106,11 +110,22 @@ case-insensitively but display case-preserved (`users.username`, ## Connection pooling -- App: `pgxpool.Pool` per process. Default sizing 25 in prod, 10 in dev. -- Hooks (S07/S14): tiny pools (max 4) since they're short-lived and on the - critical path of git operations. +- Web: one `pgxpool.Pool`, `db.max_conns` (default **10**, see + `internal/infra/config` and `internal/infra/db`). +- Worker: `resolveWorkerCount()` + 2 — **6** at the shipped + `SHITHUB_WORKERS=4`. The pool is sized off the *resolved* count, not + the raw flag; getting that wrong once put four workers on a two-conn + pool (2026-09-02 sitrep, cause #9). +- Hooks, `ssh`, `cron` and `admin` subcommands: short-lived pools of + 2–4. - Tests: per-test pool with max 2 connections. +Server-side `max_connections` in `postgresql.conf.j2` is **60**, which +covers 10 + 6 plus concurrent short-lived invocations with room to +spare. Raising a client pool means re-checking that number: on a +shared box each backend can claim `work_mem` per sort node, so +connection count is a memory decision, not just a concurrency one. + ## Test harness - `internal/testing/dbtest`: `dbtest.NewTestDB(t)` creates a fresh database @@ -121,8 +136,100 @@ case-insensitively but display case-preserved (`users.username`, ## Operational -- `pg_stat_statements` extension is loaded by default in dev compose and - prod (S37). Used by S36's perf pass. -- `archive_mode=on` + WAL shipping to Spaces (cross-region) in prod (S37). +- `pg_stat_statements` is loaded by default in dev compose. It is + **not** installed in production — see below. +- `archive_mode=on` + WAL shipping to Spaces (cross-region) in prod. - Daily logical backups via `pg_dump --format=custom`, restored weekly to - validate the backup chain. + validate the backup chain (`runbooks/backups.md`). + +## The Ansible postgresql.conf has never been applied + +`deploy/ansible/roles/postgres/templates/postgresql.conf.j2` is +tracked, tuned, and **not on the box.** `shithub-app` runs Debian +package defaults: + +| Setting | Live box | Template | +|---|---|---| +| `shared_buffers` | 128MB | 256MB | +| `work_mem` | 4MB | 4MB | +| `effective_cache_size` | 4GB (default) | 1GB | +| `maintenance_work_mem` | 64MB (default) | 64MB | +| `max_connections` | 100 | 60 | +| `shared_preload_libraries` | *(empty)* | `pg_stat_statements` | + +Consequences worth knowing: there is no `pg_stat_statements`, so +there is no query attribution when the box is under load; and +`max_connections=100` on a shared 3.9 GB box is a larger worst case +than anything the app will actually open. + +`deploy/audit/check-droplet-drift.sh` tracks +`/etc/postgresql/16/main/postgresql.conf` so this stays visible. + +### Applying it safely + +**Do not do this with a blind `make deploy`.** The play would rewrite +the config and hand off to a handler mid-day; `shared_buffers`, +`max_connections` and `shared_preload_libraries` all need a **restart** +(not a reload), which drops every open connection, and the deploy +pipeline restarts web and worker for unrelated reasons on every push +to trunk. + +Do it deliberately, inside the **05:15–06:00 UTC quiet window** — after +`shithubd-cron.timer` at 05:15 and before the AIDE check at 06:00, with +the 03:17 `pg_dump` and the 01/07/13/19:23 rclone sync all clear: + +```sh +# 1. Snapshot what is live, so a rollback is a file copy. +ssh root@shithub.sh " + cp -a /etc/postgresql/16/main/postgresql.conf \ + /etc/postgresql/16/main/postgresql.conf.pre-ansible + sudo -u postgres psql -x -c \"SELECT name, setting, unit, context + FROM pg_settings WHERE name IN + ('shared_buffers','work_mem','effective_cache_size', + 'maintenance_work_mem','max_connections', + 'shared_preload_libraries')\" +" + +# 2. Confirm there is enough free memory for the larger shared_buffers +# right now (the delta is ~128 MB, but check, do not assume). +ssh root@shithub.sh 'free -m; systemctl show shithubd-web -p MemoryCurrent' + +# 3. Apply the db role ONLY. --check first; read the diff. +ANSIBLE_INVENTORY=production ANSIBLE_TAGS=db make deploy-check +ANSIBLE_INVENTORY=production ANSIBLE_TAGS=db make deploy + +# 4. Validate the file parses BEFORE bouncing the server. `-C` reads a +# setting out of the on-disk config without starting a server; on +# Debian, -D is the CONFIG dir, not pgdata. +ssh root@shithub.sh 'sudo -u postgres \ + /usr/lib/postgresql/16/bin/postgres -D /etc/postgresql/16/main \ + -C shared_buffers' + +# 5. Restart (not reload) and watch it come back. +ssh root@shithub.sh ' + systemctl restart postgresql@16-main + sleep 5 + systemctl is-active postgresql@16-main + sudo -u postgres psql -Atc "SHOW shared_buffers" + sudo -u postgres psql -Atc "SHOW max_connections" + journalctl -u postgresql@16-main -n 50 --no-pager +' + +# 6. Create the extension (the preload alone does nothing). +ssh root@shithub.sh 'sudo -u postgres psql -d shithub \ + -c "CREATE EXTENSION IF NOT EXISTS pg_stat_statements"' + +# 7. Web and worker dropped their pools on the restart; confirm they +# reconnected rather than wedging. +ssh root@shithub.sh ' + curl -fsS 127.0.0.1:8080/readyz + journalctl -u shithubd-web -u shithubd-worker -n 50 --no-pager +' +``` + +Rollback is `cp` the `.pre-ansible` file back and restart again. + +If Postgres refuses to start after the restart, the usual cause is +`shared_buffers` exceeding what the kernel will give it: lower it, +restart, and check `journalctl` for the shared-memory error before +trying anything else. Never delete `pgdata`. diff --git a/docs/internal/deploy.md b/docs/internal/deploy.md index 106443a4..7aa2158a 100644 --- a/docs/internal/deploy.md +++ b/docs/internal/deploy.md @@ -1,13 +1,107 @@ # Deployment -This is the operator's guide to taking a fresh box from "Ubuntu 24.04 +This is the operator's guide to taking a fresh box from "Debian/Ubuntu with sshd" to "running shithubd in production." It is opinionated: DigitalOcean for compute, DigitalOcean Spaces for object storage, -Postgres on a dedicated droplet, Caddy as the edge, WireGuard for the -monitoring mesh. If you're running on something else, the Ansible -roles are the source of truth — read them. +Caddy as the edge, Postgres 16, Grafana Alloy pushing metrics to +Grafana Cloud. If you're running on something else, the Ansible roles +are the source of truth — read them. -## Topology +Two topologies are described below. The **single-box reference +deployment** is what shithub.sh actually runs and what the Ansible +roles are tuned for. The **multi-host design** is the aspirational +shape we'd grow into; nothing in it is deployed today. + +## Single-box reference deployment (what shithub.sh runs) + +``` + +----------------------------+ + public ---> | Caddy (TLS, rate limits) | :443 + +----------------------------+ + | 127.0.0.1:8080 + +--------------------------------------------------------+ + | droplet `shithub-app` — 2 vCPU / 3.9 GB / 4 GB swap | + | | + | shithubd web (systemd) ---. | + | shithubd worker (systemd) ---+--> Postgres 16 | + | shithubd cron (timer, 05:15) -' (localhost:5432, | + | Caddy peer + SCRAM) | + | Grafana Alloy + node_exporter | + | root crontab: pg_dump, rclone DR sync, AIDE, | + | WAL-archive verify | + +--------------------------------------------------------+ + | | + v v + Spaces (S3) Grafana Cloud + - WAL archive (Prometheus/Mimir, + - daily dumps remote_write, push-only) + - LFS / blobs / Actions logs + + 3 × runner droplets — outbound HTTPS to the app box only; + inbound SSH restricted by cloud firewall to the app box. +``` + +Everything is on one host. There is **no private network, no VPN, and +no separate database, backup, or monitoring host.** Postgres listens on +`localhost` only; `/metrics` is served on `127.0.0.1:8080` and is +reached only by the local Alloy process. Alloy `remote_write`s to +Grafana Cloud, so no inbound monitoring port exists at all. + +### Memory budget + +3.9 GB of RAM plus a 4 GB swapfile on `/data` (`vm.swappiness=10`). +Baseline figures are the measurements in the 2026-09-02 availability +sitrep; the ceilings are what the shipped units enforce. + +| Component | Ceiling | Observed / expected | +|---|---|---| +| `shithubd-web` | `GOMEMLIMIT=1200MiB`, `MemoryHigh=1600M`, `MemoryMax=2000M`, `OOMScoreAdjust=-500` | 1.33 GB RSS before Phase 2; the eight duplicate template renderers (664 MB) are now one (41 MB), so steady-state target is < 600 MB | +| `shithubd-worker` | `MemoryHigh=768M`, `MemoryMax=1024M` (cgroup-wide, so forked `git` counts) | small; spikes with pack operations | +| `shithubd-cron` | inherits the worker-class footprint; runs 05:15 UTC | short-lived | +| Postgres 16 | none (not a cgroup we manage) | 447 MB aggregate at the Debian-default `shared_buffers=128MB`; ~600 MB if `postgresql.conf.j2` is ever applied (see `db.md`) | +| Grafana Alloy | none | 317 MB scraping an 11.7k-series `/metrics` (Phase 3 should cut the series count by an order of magnitude) | +| Caddy, node_exporter, sshd, journald | none | ~100 MB combined | +| `rclone` DR sync | 4×/day under `flock`, off the 03:00–05:00 window | 1.0–1.6 GB for ~28 min per run; this is the single largest transient | +| AIDE check | 06:00 UTC, one wrapper only | ~0.5 GB for ~12 min | + +The failure mode this budget exists to prevent: baseline ~2.4 GB + +AIDE ~0.5 GB + rclone 1.0–1.6 GB exceeded 3.9 GB with no swap, and the +kernel picked the largest process — nine `global_oom` kills in the week +before 2026-09-02, alternating between `shithubd` and `rclone`. + +### Observability on this box + +- **Metrics.** node_exporter on `127.0.0.1:9100`, shithubd on + `127.0.0.1:8080/metrics`. Grafana Alloy scrapes both and + `remote_write`s to Grafana Cloud. Scrape job labels are `node` and + `shithubd` — not the `shithubd-web` / `shithubd-worker` / `postgres` + / `caddy` jobs the committed Prometheus config assumes. +- **The worker exposes no metrics endpoint.** `shithubd worker` never + starts an HTTP listener, so every `shithub_worker_*` series exists in + the binary and reaches nothing. +- **Logs.** journald only. There is **no log shipping** — no Promtail, + no Loki, no Alloy logs pipeline. Reading logs means SSH + + `journalctl`. +- **Alerting.** DigitalOcean droplet alerts and the uptime check + (`runbooks/alerts.md`) plus whatever Grafana-managed alert rules the + operator has provisioned in Grafana Cloud. **There is no + Alertmanager and no local Prometheus**, so nothing in + `deploy/monitoring/prometheus/rules.yml` can fire — see + `deploy/monitoring/README.md`. + +Details, queries and the operator setup flow: +`runbooks/observability.md`. + + + +## Multi-host design (aspirational — not deployed) + +Nothing in this section is running. It is the shape we would grow +into if one box stops being enough, and it is what +`deploy/ansible/roles/wireguard/`, `deploy/monitoring/` and the +`monitoring-host` language elsewhere in the tree assume. Treat every +reference to a mesh address, a monitoring host, Prometheus, Loki or +Alertmanager as a design note, not as production. ``` +----------------------------+ @@ -37,15 +131,24 @@ roles are the source of truth — read them. +----------------+ ``` -The monitoring host is *not* on the public internet. App processes -listen on `127.0.0.1` and on the wg0 mesh interface only; nothing -about the metrics port is reachable from outside the mesh. +In that design the monitoring host is *not* on the public internet; +app processes listen on `127.0.0.1` and on the `wg0` mesh interface +only, and nothing about the metrics port is reachable from outside the +mesh. Adopting it means provisioning the mesh +(`deploy/ansible/roles/wireguard/`), standing up the monitoring host, +and repointing the scrape config — see `deploy/monitoring/README.md` +for the gap list. + + ## One-time bootstrap -1. **Provision the droplets.** Three for staging is enough (web, - db, monitoring). Production starts at five (2× web, db, backup, - monitoring) and grows the web tier first. +1. **Provision the droplet.** One box runs everything (see the + reference deployment above); 4 GB is the practical floor, 8 GB is + comfortable. Actions runners are separate droplets with a cloud + firewall allowing inbound SSH from the app box only. The + multi-host split (2× web, db, backup, monitoring) is the + aspirational shape, not the starting point. 2. **Get sshd public-key login working** for the operator user. The Ansible base role narrows it from there. 3. **Populate `deploy/ansible/inventory/`** by copying @@ -105,16 +208,24 @@ In rough order: - **caddy** (`tags: [edge]`) — installs Caddy + the templated `Caddyfile`. Auto-TLS via Let's Encrypt staging until the operator flips a vars flag; production after that. -- **wireguard** (`tags: [net]`) — peers each host into the mesh. -- **backup** (`tags: [backup]`) — installs the daily backup cron on - the db host and the 6-hourly cross-region sync on the backup host. -- **monitoring-client** (`tags: [monitoring]`) — node-exporter + - promtail on every host pointing at the monitoring host. +- **backup** (`tags: [backup]`) — installs the daily `pg_dump` cron + (03:17) and the 6-hourly cross-region Spaces sync (01/07/13/19:23, + under `flock`). On the single-box deployment both land on the app + host. +- **monitoring-client** (`tags: [monitoring]`) — `node_exporter` on + `127.0.0.1:9100` plus Grafana Alloy, which scrapes node_exporter and + shithubd's `/metrics` and `remote_write`s to Grafana Cloud. Metrics + only; it ships no logs. + + +- **wireguard** (`tags: [net]`) — peers each host into the WireGuard + mesh. Part of the aspirational multi-host design; **not run today** + (a single-host inventory has no peers). + + -The monitoring host itself is provisioned by a separate Ansible play -that lives outside this repo (it depends on operator-specific TLS -material). The configs in `deploy/monitoring/` are the source of -truth for *what* runs there. +There is no monitoring host. The configs in `deploy/monitoring/` are +not deployed anywhere — see `deploy/monitoring/README.md`. ## Backups @@ -171,7 +282,8 @@ back is "redeploy the previous binary." Two paths: | sshd (incl. AKC for git) | `deploy/sshd_config.j2` | | Postgres scripts | `deploy/postgres/` | | Spaces lifecycle + DR | `deploy/spaces/` | -| WireGuard mesh | `deploy/wireguard/wg0.conf.j2` | -| Monitoring configs | `deploy/monitoring/` | +| Metrics agent (deployed) | `deploy/ansible/roles/monitoring-client/` | +| Monitoring configs (undeployed)| `deploy/monitoring/` — see its `README.md` | +| Mesh role (aspirational) | see the multi-host section above | | Restore drill | `deploy/restore-drill/` | | Operator runbooks | `docs/internal/runbooks/` | diff --git a/docs/internal/observability.md b/docs/internal/observability.md index a8934279..47667a12 100644 --- a/docs/internal/observability.md +++ b/docs/internal/observability.md @@ -37,6 +37,10 @@ shithub ships four sinks: structured logging, Prometheus metrics, OpenTelemetry - `shithub_billing_past_due_principals{subject_kind}` (gauge) - `shithub_billing_org_seat_drift` (gauge) - `shithub_billing_quota_overage_orgs{quota}` (gauge) + - `shithub_backup_last_success_seconds{job}` (gauge; read at scrape + time from the heartbeat file each backup cron job writes on + success. The series is **absent** until a job has succeeded once + on that host, so alert on `absent()` as well as on age.) - Standard Go runtime + process metrics (registered automatically). - **Cardinality discipline.** Route labels come from chi's `RoutePattern()` so we get `/owner/{repo}` instead of per-repo concrete paths. Never label by `user_id` or `repo_id`. - Per-domain metrics (added in later sprints) MUST register against `metrics.Registry` so a single `/metrics` scrape sees everything. @@ -79,8 +83,55 @@ The `request_id` is the correlation key tying logs, metrics, traces, and error r 4. `middleware.Recover` includes it on panic logs and on the Sentry/GlitchTip event. 5. The styled error pages (`errors/{404,403,429,500}.html`) display it for end-user support reference. +## How this reaches an operator in production + +shithub.sh is a single droplet (`docs/internal/deploy.md`). The +pipeline is: + +``` +shithubd web /metrics (127.0.0.1:8080) --. + +--> Grafana Alloy --remote_write--> Grafana Cloud +node_exporter (127.0.0.1:9100) ----------' (Prometheus/Mimir) +``` + +- Alloy is installed by the `monitoring-client` Ansible role. It + scrapes with job labels `shithubd` and `node`, and pushes; **no + inbound monitoring port is open on the droplet.** +- **The worker exposes no `/metrics` endpoint.** `shithubd worker` + starts no HTTP listener, so `shithub_worker_*` and any other + worker-side series are collected by nothing. Worker health has to be + read from `journalctl -u shithubd-worker` or from DB state. +- **There is no log pipeline.** No Promtail, no Loki, no Alloy logs + component. Logs live in journald on the box and nowhere else, so + log-based correlation (including the `request_id` flow below) is an + SSH-and-grep exercise. +- **There is no Alertmanager and no local Prometheus.** Nothing loads + `deploy/monitoring/prometheus/rules.yml`. Every alert defined there + — `ShithubdWebDown`, `ShithubdWorkerDown`, `PostgresDown`, + `HighRequestLatencyP95`, `HighDBQueryRate`, `JobBacklogGrowing`, + `WebhookDeliveryFailing`, the seven `shithubd-billing` alerts, the + five `shithubd-actions` alerts, and `BackupOverdue` — **does not + fire.** Several of them could not fire even with a Prometheus + attached, because they select on job labels (`shithubd-web`, + `shithubd-worker`, `postgres`, `caddy`) that this deployment never + emits. +- What *does* page today: DigitalOcean droplet resource alerts and + the uptime check (`runbooks/alerts.md`), plus any Grafana-managed + alert rule provisioned into the Cloud stack by hand or by + `deploy/monitoring/grafana/provision-actions-alerts.sh`. + +Operator setup, dashboard queries, pprof procedure and the +"metrics stopped landing" checklist: `runbooks/observability.md`. +Why the committed monitoring configs are inert: +`deploy/monitoring/README.md`. + ## Operational notes -- `pg_stat_statements` is loaded by the dev compose Postgres (S01) and required in prod (S37). -- GlitchTip and the OTLP collector run on bare metal in our prod topology (S37); the droplet's `/metrics` is scraped by Prometheus over WireGuard. +- `pg_stat_statements` is loaded by the dev compose Postgres (S01). + It is **not** installed on the production box; the Ansible + `postgresql.conf.j2` that would load it has never been applied + (`docs/internal/db.md`). +- GlitchTip and the OTLP collector are not deployed; + `error_reporting.dsn` and `tracing.enabled` are unset in prod, so + both packages are no-ops there. - Configuration documented in `docs/internal/config.md`. diff --git a/docs/internal/retro/2026-09-02-availability-sitrep.md b/docs/internal/retro/2026-09-02-availability-sitrep.md index 454efcbb..2bba23f1 100644 --- a/docs/internal/retro/2026-09-02-availability-sitrep.md +++ b/docs/internal/retro/2026-09-02-availability-sitrep.md @@ -96,7 +96,9 @@ commits behind origin/trunk**. All fixes must branch from The first four items are now also mirrored into Ansible (base + backup roles, `deploy/spaces/sync-cross-region.sh`), so the next `make deploy` re-applies them instead of reverting the hand edits. -The verification items below are still the operator's. +The verification items below are still the operator's. **Every +unticked box in this phase is restated with exact commands in +[Operator to-do](#operator-to-do) at the end of this file.** - [ ] Add 4 GB swapfile on `/data`, `vm.swappiness=10`, persist in fstab - [ ] Disable `dailyaidecheck.timer` (keep `shithub-aide-check` cron) @@ -178,31 +180,255 @@ The verification items below are still the operator's. - [x] `actionsobserver`: the `octet_length` sum now runs every 5 min on its own cadence; the count and queue-depth gauges stay at 15 s -### Phase 4 — observability and docs - -- [ ] Enable `pg_stat_statements` -- [ ] `docs/internal/observability.md` / `deploy.md`: replace the - WireGuard + monitoring-droplet story with the real Alloy → - Grafana Cloud pipeline; note there is no Alertmanager, so the - runbook backup alerts cannot fire -- [ ] `runbooks/alerts.md:77`: box is 2 vCPU, not 4 -- [ ] Backup scripts log a timestamped success line; both logs are - 0 bytes since May 10 -- [ ] Decide `postgresql.conf.j2`: revise for a shared 4 GB box - (shared_buffers 256 MB, work_mem 4 MB) and track it in - `check-droplet-drift.sh`, or delete it -- [ ] Add `/etc/postgresql/16/main/postgresql.conf` and the Stripe - hand-edit in `web.env` to drift tracking - -### Operator-only items (need account access) - -- [ ] `doctl` token is revoked/expired (401); rotate it -- [ ] Runner firewall: add current laptop egress IP to the SSH rule - (interim: `ssh -J root@24.199.108.81 root@`) -- [ ] Spaces access key is plaintext in both env files; rotate if the - exposure set is wider than intended -- [ ] Consider resizing the droplet to 8 GB if Phase 0–2 do not hold - peak memory under 75% +### Phase 4 — observability and docs (PR) + +- [x] `docs/internal/observability.md` / `deploy.md` / + `architecture.md` / `capacity.md`, plus + `runbooks/observability.md`, `alerts.md`, `incidents.md`: + the WireGuard + monitoring-droplet story is replaced with the + real Alloy → Grafana Cloud pipeline. `deploy.md` and + `architecture.md` now lead with a **single-box reference + deployment** section (topology, per-component memory budget + against the 3.9 GB + 4 GB swap, what the systemd ceilings + enforce); the multi-host design is kept as a clearly marked + aspirational section rather than deleted. Every doc now states + plainly that there is no Alertmanager, no local Prometheus, no + log shipping and no worker metrics endpoint, and enumerates the + alerts that therefore do not exist. +- [x] `runbooks/incidents.md`: added a `memory-pressure` section — + how to read the box (`journalctl -k` for OOM class, `sar -r` + /`sar -S`/`sar -q` for the run-up, `systemctl show + -p MemoryCurrent` + `systemd-cgtop` for per-unit attribution, + `ps -eo rss` for the non-cgroup processes, and the `/metrics` + gauges worth reading), plus mitigation order. +- [x] `deploy/monitoring/README.md`: states that none of the + committed Prometheus/Alertmanager/Loki configs are deployed, + which alerts are therefore inert, why several could not fire + even with a Prometheus attached (job labels, no worker + endpoint, no postgres exporter), and what adopting them would + take. Nothing deleted. +- [x] `runbooks/alerts.md:77`: box is 2 vCPU, not 4 +- [x] Backup scripts log a timestamped start/end/exit-status line to + the cron-redirected stream and write a heartbeat file on + success only (`/var/lib/shithub/backup-last-success`, + `/var/lib/shithub/spaces-sync-last-success`). `set -euo + pipefail` semantics preserved: a failed rclone still exits + non-zero and leaves any previous heartbeat untouched. Covered + by `scripts/test-backup-scripts.sh` (stubbed rclone/pg_dump), + wired into `make ci` and CI alongside a `bash -n` sweep + (`scripts/lint-shell.sh`). +- [x] The heartbeats are exported as + `shithub_backup_last_success_seconds{job="daily"|"spaces-sync"}` + (`internal/infra/metrics/backupobserver.go`), so backup + freshness reaches Grafana Cloud and a managed rule can finally + make `BackupOverdue` real. The rule in `rules.yml` named + `shithubd_backup_last_success_seconds`, which never existed. +- [x] `postgresql.conf.j2` revised rather than deleted, for a 4 GB + box shared with a ~200 MB app + worker: `shared_buffers=256MB`, + `work_mem=4MB`, `effective_cache_size=1GB`, + `maintenance_work_mem=64MB`, `max_connections=60` (web pool 10 + + worker 6 + short-lived cron/hook/ssh/admin pools + headroom), + `shared_preload_libraries='pg_stat_statements'`, + `pg_stat_statements.track=top`. +- [x] `docs/internal/db.md` records that the template has never been + applied, the live-vs-template setting table, and a step-by-step + safe-apply procedure (snapshot, `--check`, `postgres -C` + validation, **restart** not reload, inside the 05:15–06:00 + quiet window, never a blind `make deploy`). +- [x] `check-droplet-drift.sh` tracks + `/etc/postgresql/16/main/postgresql.conf` and reports both it + and the `web.env` Stripe hand-edit as `KNOWN` drift with the + reason attached. +- [x] `scripts/lint-docs-topology.sh` fails CI when `docs/internal` + mentions WireGuard / `wg0` / `10.50.0.` outside an explicit + `` block. `docs/internal/retro/` + is excluded — retrospectives are dated snapshots. +- [ ] Enable `pg_stat_statements` on the box — operator, see below. + +## Operator to-do + +Everything that needs account access or a hand on the box, with the +commands. Nothing here is done by merging a PR. + +### 1. Rotate the `doctl` token + +The current token 401s, which blocks every other `doctl` item below. + +```sh +# https://cloud.digitalocean.com/account/api/tokens → Generate New Token +# Scopes needed: droplet read, monitoring read+write, firewall read+write. +doctl auth init --context shithub # paste the token when prompted +doctl auth switch --context shithub +doctl account get # must not 401 +``` + +### 2. Repoint the DO uptime check at `/healthz` + +`provision-do-alerts.sh` already defaults to +`https://shithub.sh/healthz`, but `doctl` cannot change an existing +check's target in place — it needs a delete + recreate. + +```sh +doctl monitoring uptime list # note the check id +doctl monitoring uptime delete +deploy/cutover/provision-do-alerts.sh # recreates check + 3 alerts +doctl monitoring uptime list --output json | jq '.[].target' +# expect "https://shithub.sh/healthz" +``` + +Rationale and the 429 evidence: `runbooks/alerts.md`. + +### 3. Add the laptop egress IP to the runner SSH firewall + +```sh +curl -s https://ifconfig.me; echo # your egress IP +doctl compute firewall list # find the runner firewall id +doctl compute firewall add-rules \ + --inbound-rules "protocol:tcp,ports:22,address:/32" +``` + +Interim workaround that needs no firewall change (the app box is +already allowed): `ssh -J root@24.199.108.81 root@`. + +### 4. Persist the swapfile and `vm.swappiness` + +The 4 GB swapfile exists at runtime but is not in `fstab`, so it +disappears on the next reboot — which is exactly when it is most +needed. `roles/base/tasks/swap.yml` does this idempotently; use the +**same** fstab line and sysctl filename by hand so the next +`make deploy` is a no-op rather than leaving a second sysctl file +behind. + +```sh +ssh root@shithub.sh ' + swapon --show + grep -q "^/data/swapfile[[:space:]]" /etc/fstab || + echo "/data/swapfile none swap sw,nofail 0 0" >> /etc/fstab + printf "vm.swappiness = 10\n" > /etc/sysctl.d/60-shithub-swap.conf + sysctl --system + sysctl vm.swappiness # expect 10 + findmnt --verify --fstab 2>&1 | tail -5 # fstab parses +' +``` + +Do **not** verify by `swapoff`-ing: that forces every swapped page +back into a 3.9 GB box and is the one command guaranteed to cause the +outage this item exists to prevent. `swapon -a` is a no-op when the +file is already active, so the fstab line is only really exercised at +the next reboot. + +Or, equivalently and preferably, just run the role: +`ANSIBLE_INVENTORY=production ANSIBLE_TAGS=base make deploy-check` +first, then `make deploy` — the swap tasks are guarded by `creates`, +the on-disk signature, `/proc/swaps` and a path-matched fstab line, so +they converge without touching the live swapfile. + +### 5. Disable the duplicate AIDE timer + +The Debian-packaged timer still runs alongside our cron wrapper, so +two AIDE scans overlap in the 03:00–05:00 window. + +```sh +ssh root@shithub.sh ' + systemctl disable --now dailyaidecheck.timer + systemctl is-enabled dailyaidecheck.timer # expect: disabled + crontab -l | grep shithub-aide-check # our 06:00 wrapper survives +' +``` + +### 6. Install the defanged backup + sync scripts + +The box still runs the pre-Phase-4 copies: no status lines, no +heartbeat, and the sync's `--fast-list` on the WAL bucket. Deploy +re-copies these, but do not wait for an unrelated deploy. + +```sh +# from a checkout of trunk after this PR merges +scp deploy/spaces/sync-cross-region.sh root@shithub.sh:/usr/local/bin/shithub-spaces-sync +scp deploy/postgres/backup-daily.sh root@shithub.sh:/usr/local/bin/shithub-backup-daily +ssh root@shithub.sh ' + chmod 0755 /usr/local/bin/shithub-spaces-sync /usr/local/bin/shithub-backup-daily + mkdir -p /var/lib/shithub /var/log/shithub + crontab -l | grep -E "shithub-(backup-daily|spaces-sync)" # flock + 6h cadence +' + +# Verify by running each once, off-peak, and checking the traces: +ssh root@shithub.sh ' + /usr/local/bin/shithub-backup-daily >> /var/log/shithub-backup.log 2>&1 + tail -3 /var/log/shithub-backup.log + date -u -d @"$(cat /var/lib/shithub/backup-last-success)" +' +deploy/audit/check-droplet-drift.sh # runs locally, ssh's to the box +``` + +Then confirm the gauge is live: +`curl -fsS 127.0.0.1:8080/metrics | grep shithub_backup_last_success`. + +### 7. Apply `postgresql.conf.j2` and enable `pg_stat_statements` + +Needs a **Postgres restart**, inside the 05:15–06:00 UTC quiet +window, never via a blind `make deploy`. The full procedure with +snapshot and rollback is +[`docs/internal/db.md`](../db.md#applying-it-safely) — follow it +there rather than improvising. Ends with: + +```sh +ssh root@shithub.sh 'sudo -u postgres psql -d shithub \ + -c "CREATE EXTENSION IF NOT EXISTS pg_stat_statements"' +``` + +### 8. Rotate the Spaces access key + +The key is plaintext in `/etc/rclone-shithub.conf`, `web.env` and +`worker.env`. Rotate if the exposure set is wider than intended. + +```sh +# https://cloud.digitalocean.com/account/api/spaces → generate a new key +ssh root@shithub.sh ' + sed -i "s/^access_key_id = .*/access_key_id = /" /etc/rclone-shithub.conf + sed -i "s/^secret_access_key = .*/secret_access_key = /" /etc/rclone-shithub.conf + # same pair in /etc/shithub/web.env and /etc/shithub/worker.env + rclone --config /etc/rclone-shithub.conf lsd spaces-prod: # must succeed + systemctl restart shithubd-web shithubd-worker +' +# Only after the check above passes: revoke the old key in the portal. +``` + +Note `web.env` also carries a hand-edited Stripe block absent from +`web.env.j2`; a full `make deploy` re-renders the file and drops it. +`check-droplet-drift.sh` now flags this as KNOWN drift. + +### 9. Enable pprof on the box + +Ansible sets it in `web.env.j2`, but the deploy pipeline does not +re-render `/etc/shithub/web.env`. + +```sh +ssh root@shithub.sh ' + grep -q SHITHUB_WEB__PPROF_ADDR /etc/shithub/web.env || + echo "SHITHUB_WEB__PPROF_ADDR=127.0.0.1:6060" >> /etc/shithub/web.env + systemctl restart shithubd-web + journalctl -u shithubd-web -n 20 --no-pager | grep pprof +' +# expect: pprof listener started (loopback only) addr=127.0.0.1:6060 +``` + +### 10. Phase 0 verification, still open + +```sh +ssh root@shithub.sh ' + journalctl -k --since "7 days ago" | grep -i oom # expect empty + sar -r | tail -20 # peak %memused < 75 +' +# DR bucket integrity: +ssh root@shithub.sh 'rclone --config /etc/rclone-shithub.conf \ + check --size-only --one-way spaces-prod:shithub-backups spaces-dr:shithub-backups-dr' +``` + +### 11. Standing decision + +- [ ] Resize the droplet to 8 GB if Phases 0–3 do not hold peak + memory under 75%. ## Verification targets diff --git a/docs/internal/runbooks/alerts.md b/docs/internal/runbooks/alerts.md index c0554217..9515bfbf 100644 --- a/docs/internal/runbooks/alerts.md +++ b/docs/internal/runbooks/alerts.md @@ -1,5 +1,16 @@ # Alerts +**DigitalOcean alerts are the only thing that pages.** There is no +Alertmanager and no local Prometheus in this deployment, so nothing +loads `deploy/monitoring/prometheus/rules.yml` — every alert defined +there (`ShithubdWebDown`, `ShithubdWorkerDown`, `PostgresDown`, +`HighRequestLatencyP95`, `HighDBQueryRate`, `JobBacklogGrowing`, +`WebhookDeliveryFailing`, the `shithubd-billing` and +`shithubd-actions` groups, and `BackupOverdue`) is inert. Grafana +Cloud receives the metrics and *can* host managed alert rules, but +only the ones an operator has provisioned there actually exist; see +`runbooks/observability.md` and `deploy/monitoring/README.md`. + DigitalOcean's built-in monitoring is the cheap-and-good alerting tier. Two layers, both managed via `doctl`: @@ -129,10 +140,24 @@ recreate via the script.) ## What's NOT here - **shithubd-internal metrics** (request latency p95, DB pool - saturation, job queue depth, throttle rejections, archive - failures from Postgres). Those need a Prometheus scrape of - `127.0.0.1:8080/metrics` plus a Grafana Cloud free-tier — see - task #263 / `runbooks/observability.md` (TBD). + saturation, job queue depth, throttle rejections). These *are* + collected — Grafana Alloy scrapes `127.0.0.1:8080/metrics` and + pushes to Grafana Cloud (`runbooks/observability.md`) — but no DO + alert reads them, and a Grafana-managed rule has to be created per + signal before any of them pages. +- **Backup failure.** Neither DO nor Grafana Cloud alerts on it. The + backup and DR-sync scripts write a timestamped start/end/exit line + to their logs and a heartbeat file on success + (`/var/lib/shithub/backup-last-success`, + `/var/lib/shithub/spaces-sync-last-success`), which the web process + exports as `shithub_backup_last_success_seconds{job=...}`. That + gauge reaches Grafana Cloud, so a managed rule on it is the cheapest + way to make `BackupOverdue` real. Until then, checking is manual — + `runbooks/backups.md`. +- **Postgres and Caddy exporters.** Not installed, so + `pg_stat_archiver.failed_count` (the archive-failing signal) is only + visible to the hourly `shithub-verify-wal-archive` cron job, which + logs and journals but does not page. - **Slack delivery.** Today everything emails. The DO API supports Slack webhooks; add `--slack-channels` + `--slack-urls` to the provision script when there's a destination. diff --git a/docs/internal/runbooks/backups.md b/docs/internal/runbooks/backups.md index 1975bd36..176934f4 100644 --- a/docs/internal/runbooks/backups.md +++ b/docs/internal/runbooks/backups.md @@ -80,20 +80,56 @@ ships zero WAL segments until the operator runs through this once: ## Verifying that backups are healthy -The monitoring stack does this for you: +**Nothing pages on a missed backup today.** There is no Alertmanager +and no local Prometheus, so the `BackupOverdue` rule in +`deploy/monitoring/prometheus/rules.yml` never evaluates +(`deploy/monitoring/README.md`). Checking is the operator's job until +a Grafana-managed rule exists. -- `BackupOverdue` alert fires if `time() - - shithubd_backup_last_success_seconds > 30h`. The backup script - pushes the timestamp to the metrics endpoint on success. -- `pg_stat_archiver.failed_count > 0` is paged via the - `archive-failing` runbook. +Each job leaves two traces: -If you want to confirm by hand: +| Job | Cron log | Heartbeat (success only) | +|---|---|---| +| `shithub-backup-daily` (03:17) | `/var/log/shithub-backup.log` | `/var/lib/shithub/backup-last-success` | +| `shithub-spaces-sync` (01/07/13/19:23) | `/var/log/shithub-spaces-sync.log` (status lines) and `/var/log/shithub/spaces-sync.log` (status + rclone detail) | `/var/lib/shithub/spaces-sync-last-success` | + +Both scripts bracket every run with +`... start ...` / `... end status=ok exit=0` (or `status=FAILED +exit=N`) lines, and write the heartbeat **only** on a fully +successful run, so a stale heartbeat is never refreshed by a broken +one. Before 2026-09-02 neither log had a byte in it since 2026-05-10, +which is indistinguishable from the job never having been installed. + +Quick check: + +```sh +ssh root@shithub.sh ' + tail -5 /var/log/shithub-backup.log /var/log/shithub-spaces-sync.log + for f in /var/lib/shithub/backup-last-success /var/lib/shithub/spaces-sync-last-success; do + printf "%s: %s\n" "$f" "$(date -u -d @"$(cat "$f")" 2>/dev/null || echo MISSING)" + done +' +``` + +The heartbeats are also exported as +`shithub_backup_last_success_seconds{job="daily"|"spaces-sync"}` on +`/metrics`, which Alloy pushes to Grafana Cloud — so +`time() - shithub_backup_last_success_seconds > 30h` (plus an +`absent()` clause, since the series does not exist until the first +success) is a one-rule Grafana-managed alert whenever someone wants +it. + +`pg_stat_archiver.failed_count > 0` is checked hourly at :47 by +`shithub-verify-wal-archive`, which logs to +`/var/log/shithub/wal-archive.log` and journals under +`shithub-wal-archive` — also not a page. See the `archive-failing` +runbook. + +To confirm the objects landed by hand: ```sh -ssh db -sudo -u postgres rclone --config /etc/rclone-shithub.conf \ - lsf spaces-prod:shithub-backups/daily/$(date -u +%Y/%m/%d)/ +ssh root@shithub.sh 'rclone --config /etc/rclone-shithub.conf \ + lsf spaces-prod:shithub-backups/daily/$(date -u +%Y/%m/%d)/' ``` ## Quarterly restore drill @@ -113,13 +149,23 @@ means our backups can't actually restore; we treat that as P0. ## Missed backup -**Symptom:** `BackupOverdue` alert. - -1. SSH to db host. `systemctl status shithub-backup-daily.timer` - and `journalctl -u shithub-backup-daily.service -n 200`. -2. Most likely: the script ran but `rclone copyto` failed (creds, - network). Re-run by hand: - `sudo -u postgres /usr/local/bin/shithub-backup-daily`. -3. If the script has been failing silently for >24h, file an +**Symptom:** the heartbeat is older than a day, or the cron log has +a `status=FAILED` line. (`BackupOverdue` is not wired — see above.) + +1. SSH to the box. There is no systemd timer; the job is a root + crontab entry. `crontab -l | grep shithub-backup-daily` confirms + it is installed, and + `tail -50 /var/log/shithub-backup.log` shows the last few runs. +2. `status=FAILED exit=N` names the failing step by position: the + line right before it in the log is the last thing that ran. + Most likely: the script ran but `rclone copyto` failed (creds, + network). +3. No `start` line at the expected time at all means cron did not + fire it — check `journalctl -u cron --since yesterday`. +4. Re-run by hand: `/usr/local/bin/shithub-backup-daily`. It takes + `sudo -u postgres` internally, so run it as root, and it is safe + to re-run: a second dump is a new timestamped file and retention + trims to seven. +5. If the script has been failing silently for >24h, file an incident — every additional day extends the RPO of an actual recovery. diff --git a/docs/internal/runbooks/incidents.md b/docs/internal/runbooks/incidents.md index 545fb1e9..21d63b9c 100644 --- a/docs/internal/runbooks/incidents.md +++ b/docs/internal/runbooks/incidents.md @@ -5,14 +5,30 @@ anchors below. Keep procedures here short and actionable: what to check first, how to mitigate, what to record. Postmortems live elsewhere. +> **These are procedures, not pages.** There is no Alertmanager and +> no local Prometheus, so the rule file that names these anchors is +> not loaded anywhere and none of the alerts below fire on their own +> (`deploy/monitoring/README.md`). What actually notifies is the +> DigitalOcean droplet/uptime alert set in `alerts.md`. The sections +> here stay useful because they are also the checklists you run when +> a DO alert, a user report, or a metric you eyeballed in Grafana +> Cloud sends you to the box. + +Everything runs on one droplet (`docs/internal/deploy.md`), so "SSH +to the affected host" always means `ssh root@shithub.sh`, and any +resource problem is a shared-resource problem: web, worker, cron, +Caddy and Postgres are competing for the same 3.9 GB. + ## shithubd-down **Symptom:** `up{job="shithubd-web"} == 0` for >2m. Site returns 502/connection refused. 1. SSH to the affected host. -2. `systemctl status shithubd-web` — if `active (running)` but - alert fires, the metrics scrape is broken (check wg0). +2. `systemctl status shithubd-web` — if `active (running)` but the + signal says down, the metrics path is broken, not the app: + `systemctl status alloy` and + `curl -fsS 127.0.0.1:8080/metrics | head -3` on the box. 3. If failed: `journalctl -u shithubd-web -n 200 --no-pager`. Look for migration failures (ExecStartPre), env file errors, port conflicts. @@ -35,7 +51,7 @@ visible to users: webhook deliveries stall, async fan-out lags. **Symptom:** `up{job="postgres"} == 0`. Site returns 500 on every write; reads through cache may still appear to work briefly. -1. SSH to the db host. +1. SSH to the box (Postgres is local, not a separate host). 2. `systemctl status postgresql`. 3. If startup is failing on WAL replay: do **not** delete pgdata. Check disk free first (`df -h /data`). If full, the most likely @@ -51,8 +67,8 @@ write; reads through cache may still appear to work briefly. 1. Look at `journalctl -u postgresql -n 200` for `archive_command failed` lines. The last few should print the rclone error. 2. Common causes: bucket creds rotated, network partition to Spaces. -3. Confirm: `sudo -u postgres rclone --config /root/.config/rclone/ - rclone.conf lsd spaces-prod:`. +3. Confirm: `sudo -u postgres rclone --config /etc/rclone-shithub.conf + lsd spaces-prod:`. 4. Mitigation: fix the underlying issue (rotate creds, restore network). Postgres will retry automatically. 5. If disk is critically full and Spaces will not be reachable in @@ -169,3 +185,113 @@ no secret-bearing logs. snapshots are captured at claim time. 4. If the controlled workflow is not masked, stop affected runners, rotate the exposed secret, and open a security incident. + +## memory-pressure + +**Symptom:** DO `memory > 90%` alert, `load1 > 4` with low CPU, +processes stuck in `D` state, or a unit that restarted with no crash +in its own log. On a single 3.9 GB box shared by web, worker, cron, +Caddy and Postgres this is the most common incident shape — nine +kernel OOM kills in the week before 2026-09-02. + +### 1. Was it the kernel? + +```sh +ssh root@shithub.sh ' + journalctl -k --since "24 hours ago" | grep -iE "oom|killed process" + journalctl --since "24 hours ago" -u shithubd-web -u shithubd-worker | grep -iE "oom|Main process exited" +' +``` + +`global_oom` means the whole box ran out and the kernel picked the +largest RSS. A cgroup OOM names the unit and means that unit hit its +own `MemoryMax` — different fix (raise the ceiling or fix the leak), +different blast radius (nothing else was harmed). + +### 2. Read the history — `sar -r` + +`sysstat` keeps 10-minute samples, so you can see the shape of the +run-up rather than the instant you happened to look. + +```sh +ssh root@shithub.sh ' + sar -r | tail -40 # today, 10-min buckets + sar -r -f /var/log/sysstat/sa$(date -d yesterday +%d) | tail -40 + sar -r -s 03:00:00 -e 06:00:00 # the backup/AIDE window + sar -S | tail -20 # swap used — should stay near 0 + sar -q | tail -20 # runqueue + load +' +``` + +Read `kbavail` / `%memused`, **not** `kbmemfree` — page cache counts +as used. The target from the availability campaign is a daily peak +`%memused` under 75%. Sustained `%swpused` above a few percent with +`vm.swappiness=10` means the box is genuinely over-committed, not just +paging out cold anonymous memory. + +### 3. Attribute it — per-unit current usage + +```sh +ssh root@shithub.sh ' + systemctl show shithubd-web -p MemoryCurrent -p MemoryHigh -p MemoryMax + systemctl show shithubd-worker -p MemoryCurrent -p MemoryHigh -p MemoryMax + systemctl show shithubd-cron -p MemoryCurrent + systemd-cgtop -m -1 -b -n1 | head -20 +' +``` + +`MemoryCurrent` is the cgroup total in bytes, so it includes anything +the unit forked (`git` under the worker, `pg_dump` under a cron job) — +which is the number that matters when a unit is near its ceiling. +`[not set]` means the unit has no cgroup accounting; that is a bug, +`scripts/lint-systemd-units.sh` exists to catch it. + +Processes outside those cgroups — Postgres, Caddy, Alloy, rclone, +AIDE — need plain RSS: + +```sh +ssh root@shithub.sh 'ps -eo rss,comm --sort=-rss | head -15' +``` + +Typical steady state: Postgres ~450 MB aggregate, Alloy ~320 MB, +Caddy + node_exporter + sshd ~100 MB. Transients: `rclone` 1.0–1.6 GB +during a DR sync (01/07/13/19:23 UTC), `aide` ~0.5 GB at 06:00. + +### 4. Attribute it — `/metrics` gauges + +From the box (or from Grafana Cloud, same series): + +```sh +ssh root@shithub.sh 'curl -fsS 127.0.0.1:8080/metrics | grep -E \ + "^(process_resident_memory_bytes|go_memstats_(heap_alloc_bytes|heap_inuse_bytes|next_gc_bytes)|go_goroutines|shithub_http_in_flight|shithub_db_pool_(acquired|total))"' +``` + +| Gauge | Reading | +|---|---| +| `process_resident_memory_bytes` | web RSS. Target < 600 MB after the Phase 2 renderer fix; 1.3 GB was the pre-fix baseline. | +| `go_memstats_heap_alloc_bytes` | live heap. If this is flat and high while RSS climbs, the growth is not the Go heap. | +| `go_memstats_next_gc_bytes` | the GC target. With `GOMEMLIMIT=1200MiB` it should not exceed that; if it does, the env var did not reach the process. | +| `go_goroutines` | ~18 in steady state. Hundreds means a leak — take a goroutine profile. | +| `shithub_http_in_flight` | > 50 sustained means saturation is upstream of memory. | +| `shithub_db_pool_acquired` vs `_total` | equal means the pool is exhausted; workers block, and blocked work retains memory. | + +These are web-process only — the worker exposes no metrics endpoint, +so `MemoryCurrent` and `ps` are the only view of it. + +### 5. Mitigate + +1. Kill the transient first if one is running: `pkill -f 'rclone.*shithub'` + (the sync is idempotent; the next `flock`ed run picks up where it + stopped), or `pkill -f 'aide --check'` (both run from root cron, + not from a unit, so there is nothing to `systemctl stop`). +2. `systemctl restart shithubd-web` reclaims a leaked heap and costs a + few seconds of 502. +3. Confirm swap is present and being used as a cushion, not a crutch: + `swapon --show; sysctl vm.swappiness`. Expect a 4 GB file on + `/data` and `vm.swappiness=10`. +4. If the run-up is a Go heap, take a profile before restarting — + `runbooks/observability.md#taking-a-heap-profile`. A restart + destroys the evidence. +5. Record the peak, the attribution, and whether a cron job overlapped + the window. Recurrence at the same clock time is a scheduling + problem, not a leak. diff --git a/docs/internal/runbooks/observability.md b/docs/internal/runbooks/observability.md index 6b6ffdba..eea7dcd7 100644 --- a/docs/internal/runbooks/observability.md +++ b/docs/internal/runbooks/observability.md @@ -25,6 +25,28 @@ Alloy `remote_write`s both into Grafana Cloud's Prometheus endpoint (Mimir under the hood). No inbound port on the droplet — push-only. No Prometheus or Grafana running locally. +### What is NOT wired + +- **No Alertmanager, no local Prometheus.** Nothing loads + `deploy/monitoring/prometheus/rules.yml`, so none of the alerts in + it fire — including `BackupOverdue`, `ShithubdWebDown`, + `PostgresDown`, `JobBacklogGrowing` and the billing/actions groups. + Backup failure is currently detected by an operator reading + `/var/log/shithub/*.log`, not by a page. See + `deploy/monitoring/README.md`. +- **No log shipping.** No Promtail, no Loki, no Alloy logs + component. Logs are journald-only; `journalctl` over SSH is the + interface. +- **No worker metrics.** `shithubd worker` starts no HTTP listener, + so the `shithub_worker_*` series never leave the process. Alloy + scrapes only `127.0.0.1:8080` (web) and `127.0.0.1:9100` (node). +- **No Postgres or Caddy exporter.** `postgres_exporter` and Caddy's + admin metrics are not installed, so `pg_stat_*`-derived panels and + edge-level latency are unavailable from Grafana Cloud. +- Scrape job labels here are `shithubd` and `node`. The committed + Prometheus config assumes `shithubd-web`, `shithubd-worker`, + `postgres` and `caddy`; queries copied from it need relabelling. + ### Notable shithubd metrics | Name | Type | Meaning | @@ -179,7 +201,9 @@ Add these once paid plans are enabled: Production uses Grafana-managed alerts because Alloy pushes metrics to Grafana Cloud with `remote_write`; there is no local Prometheus process loading -`deploy/monitoring/prometheus/rules.yml`. +`deploy/monitoring/prometheus/rules.yml`, and there is no Alertmanager +anywhere in the deployment. A rule in that file is documentation until +someone recreates it as a Grafana-managed rule. The committed Prometheus rule file remains the source-of-truth expression catalog for self-hosted deployments. For shithub.sh's Grafana Cloud stack, use @@ -232,7 +256,8 @@ Suggested first three alerts (Cloud UI): Billing alert rules are committed in `deploy/monitoring/prometheus/rules.yml` under `shithubd-billing`. -Load them before live billing launch. The corresponding response steps +**They are not loaded by anything today** — recreate them as +Grafana-managed rules before live billing launch. The corresponding response steps live in [`stripe-billing.md`](./stripe-billing.md#billing-alerts). ## When metrics stop landing diff --git a/go.mod b/go.mod index 62be5502..74b98a3b 100644 --- a/go.mod +++ b/go.mod @@ -56,6 +56,7 @@ require ( github.com/klauspost/compress v1.18.5 // indirect github.com/klauspost/cpuid/v2 v2.2.11 // indirect github.com/klauspost/crc32 v1.3.0 // indirect + github.com/kylelemons/godebug v1.1.0 // indirect github.com/mfridman/interpolate v0.0.2 // indirect github.com/minio/crc64nvme v1.1.1 // indirect github.com/minio/md5-simd v1.1.2 // indirect diff --git a/internal/infra/metrics/backupobserver.go b/internal/infra/metrics/backupobserver.go new file mode 100644 index 00000000..ba4b00ec --- /dev/null +++ b/internal/infra/metrics/backupobserver.go @@ -0,0 +1,92 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +package metrics + +import ( + "os" + "path/filepath" + "strconv" + "strings" + + "github.com/prometheus/client_golang/prometheus" +) + +// Backup jobs run from root crontab on the app box, not from a systemd +// timer, and there is no Alertmanager to catch a non-zero exit (see +// deploy/monitoring/README.md). Each job writes a heartbeat file +// containing epoch seconds *only* when it fully succeeds; this +// collector turns those files into a gauge so the one signal that does +// leave the box — Alloy's remote_write to Grafana Cloud — carries +// backup freshness. +// +// Read at scrape time rather than cached: the files change a few times +// a day and a scrape is two stat+read calls on a tmpfs-warm path. +const defaultBackupHeartbeatDir = "/var/lib/shithub" + +// job label -> file name under the heartbeat dir. Keep in sync with +// deploy/postgres/backup-daily.sh and deploy/spaces/sync-cross-region.sh. +var backupHeartbeatFiles = map[string]string{ + "daily": "backup-last-success", + "spaces-sync": "spaces-sync-last-success", +} + +var backupLastSuccessDesc = prometheus.NewDesc( + "shithub_backup_last_success_seconds", + "Unix timestamp of the last fully successful run of each backup job, from its heartbeat file. The series is absent when the job has never succeeded on this host, so alert on absent() as well as on age.", + []string{"job"}, + nil, +) + +// BackupHeartbeatCollector reports the mtime-independent timestamp +// each backup job records on success. +type BackupHeartbeatCollector struct { + dir string +} + +// NewBackupHeartbeatCollector reads heartbeat files from dir. An empty +// dir means the production default. +func NewBackupHeartbeatCollector(dir string) *BackupHeartbeatCollector { + if dir == "" { + dir = defaultBackupHeartbeatDir + } + return &BackupHeartbeatCollector{dir: dir} +} + +func (c *BackupHeartbeatCollector) Describe(ch chan<- *prometheus.Desc) { + ch <- backupLastSuccessDesc +} + +func (c *BackupHeartbeatCollector) Collect(ch chan<- prometheus.Metric) { + for job, name := range backupHeartbeatFiles { + ts, ok := readBackupHeartbeat(filepath.Join(c.dir, name)) + if !ok { + // No file, unreadable, or garbage. Emitting 0 here would + // read as "succeeded in 1970" and fire an age alert + // identically to a real overdue backup, but it would also + // mask the never-ran case; absent is the honest answer. + continue + } + ch <- prometheus.MustNewConstMetric( + backupLastSuccessDesc, prometheus.GaugeValue, ts, job, + ) + } +} + +// readBackupHeartbeat parses the epoch-seconds line a backup script +// writes. Anything unexpected is reported as absent rather than as a +// zero sample. +func readBackupHeartbeat(path string) (float64, bool) { + b, err := os.ReadFile(path) + if err != nil { + return 0, false + } + n, err := strconv.ParseInt(strings.TrimSpace(string(b)), 10, 64) + if err != nil || n <= 0 { + return 0, false + } + return float64(n), true +} + +func init() { + Registry.MustRegister(NewBackupHeartbeatCollector("")) +} diff --git a/internal/infra/metrics/backupobserver_test.go b/internal/infra/metrics/backupobserver_test.go new file mode 100644 index 00000000..4f7ccda2 --- /dev/null +++ b/internal/infra/metrics/backupobserver_test.go @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +package metrics + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/prometheus/client_golang/prometheus" + "github.com/prometheus/client_golang/prometheus/testutil" +) + +func writeHeartbeat(t *testing.T, dir, name, content string) { + t.Helper() + if err := os.WriteFile(filepath.Join(dir, name), []byte(content), 0o644); err != nil { + t.Fatalf("write %s: %v", name, err) + } +} + +func TestBackupHeartbeatCollector(t *testing.T) { + tests := []struct { + name string + files map[string]string + want string + }{ + { + name: "no heartbeat files", + files: nil, + want: "", + }, + { + name: "both jobs healthy", + files: map[string]string{ + "backup-last-success": "1767225420\n", + "spaces-sync-last-success": "1767229020\n", + }, + want: ` +shithub_backup_last_success_seconds{job="daily"} 1.76722542e+09 +shithub_backup_last_success_seconds{job="spaces-sync"} 1.76722902e+09 +`, + }, + { + name: "only the daily job has ever succeeded", + files: map[string]string{ + "backup-last-success": "1767225420", + }, + want: ` +shithub_backup_last_success_seconds{job="daily"} 1.76722542e+09 +`, + }, + { + // A truncated or half-written file must not read as a + // 1970 success — the series stays absent. + name: "garbage content is absent, not zero", + files: map[string]string{ + "backup-last-success": "not-a-timestamp", + "spaces-sync-last-success": "", + }, + want: "", + }, + { + name: "zero timestamp is absent", + files: map[string]string{ + "backup-last-success": "0", + }, + want: "", + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + dir := t.TempDir() + for name, content := range tc.files { + writeHeartbeat(t, dir, name, content) + } + + reg := prometheus.NewPedanticRegistry() + reg.MustRegister(NewBackupHeartbeatCollector(dir)) + + header := "# HELP shithub_backup_last_success_seconds " + + "Unix timestamp of the last fully successful run of each backup job, " + + "from its heartbeat file. The series is absent when the job has never " + + "succeeded on this host, so alert on absent() as well as on age.\n" + + "# TYPE shithub_backup_last_success_seconds gauge\n" + expected := "" + if tc.want != "" { + expected = header + strings.TrimPrefix(tc.want, "\n") + } + + if err := testutil.GatherAndCompare(reg, strings.NewReader(expected), + "shithub_backup_last_success_seconds"); err != nil { + t.Errorf("gathered metrics: %v", err) + } + }) + } +} + +// The default collector is registered at init against the production +// path, which does not exist in CI. It must degrade to zero series +// rather than erroring the whole /metrics scrape. +func TestBackupHeartbeatCollectorMissingDirIsSilent(t *testing.T) { + reg := prometheus.NewPedanticRegistry() + reg.MustRegister(NewBackupHeartbeatCollector(filepath.Join(t.TempDir(), "absent"))) + + got, err := testutil.GatherAndCount(reg, "shithub_backup_last_success_seconds") + if err != nil { + t.Fatalf("gather: %v", err) + } + if got != 0 { + t.Errorf("series count = %d, want 0", got) + } +} diff --git a/scripts/lint-docs-topology.sh b/scripts/lint-docs-topology.sh new file mode 100755 index 00000000..06076344 --- /dev/null +++ b/scripts/lint-docs-topology.sh @@ -0,0 +1,78 @@ +#!/usr/bin/env bash +# SPDX-License-Identifier: AGPL-3.0-or-later +# +# Keep docs/internal honest about the production topology. +# +# shithub.sh runs on ONE droplet with no VPN. The WireGuard mesh and +# the monitoring host it connects to are a design we never built, and +# for months the deploy, architecture, observability and incident docs +# described that design as if it were production — which is how an +# operator ended up told to "check wg0" on a box with no wg0. +# +# The mesh is still worth documenting as the shape we'd grow into, so +# this does not ban the words. It requires that every mention sits +# inside an explicitly marked block: +# +# +# ... prose about the mesh, the monitoring host, etc ... +# +# +# A mention outside such a block is a claim about production, and +# fails. +# +# Excluded: docs/internal/retro/ — retrospectives are dated snapshots +# of what was believed at the time and must not be rewritten. + +set -uo pipefail + +cd "$(git rev-parse --show-toplevel)" + +# Terms that only make sense in the multi-host design. +PATTERN='wireguard|wg0|10\.50\.0\.' + +START='' +END='' + +fail=0 + +while IFS= read -r f; do + case "$f" in + docs/internal/retro/*) continue ;; + esac + + # Walk the file once, tracking whether we are inside a marked block, + # and report any hit outside one. awk keeps this a single pass and + # avoids the "grep -n then correlate line numbers" dance. + out=$(awk -v pat="$PATTERN" -v start="$START" -v end="$END" ' + index($0, start) { inblock = 1; next } + index($0, end) { inblock = 0; next } + !inblock && tolower($0) ~ pat { printf "%d: %s\n", NR, $0 } + END { + if (inblock) print "EOF: unclosed topology:aspirational block" + } + ' "$f") + + if [ -n "$out" ]; then + echo "lint-docs-topology: $f" >&2 + printf '%s\n' "$out" | sed 's/^/ /' >&2 + fail=1 + fi +done < <(git ls-files 'docs/internal/*.md') + +if [ "$fail" -ne 0 ]; then + cat >&2 <<'MSG' + +Production is a single droplet with no VPN and no monitoring host +(docs/internal/deploy.md). If the text above describes the +aspirational multi-host design, wrap it: + + + ... + + +Otherwise, fix the claim. +MSG + exit 1 +fi + +echo "lint-docs-topology: ok" diff --git a/scripts/lint-shell.sh b/scripts/lint-shell.sh new file mode 100755 index 00000000..0f3e0a05 --- /dev/null +++ b/scripts/lint-shell.sh @@ -0,0 +1,39 @@ +#!/usr/bin/env bash +# SPDX-License-Identifier: AGPL-3.0-or-later +# +# Syntax check every shell script in the tree. Most of deploy/ is +# shell that only ever executes on the droplet, from cron, unattended +# — a parse error there is discovered by a missed backup, not by a +# failing build. `bash -n` is cheap and catches exactly that class. +# +# shellcheck is not a CI dependency (it isn't installed on the +# runner); when it happens to be present locally its findings are +# reported as advisory and never fail the run. + +set -uo pipefail + +cd "$(git rev-parse --show-toplevel)" + +fail=0 +checked=0 + +while IFS= read -r f; do + checked=$((checked + 1)) + if ! out=$(bash -n "$f" 2>&1); then + echo "lint-shell: $f: $out" >&2 + fail=1 + fi +done < <(git ls-files '*.sh') + +if [ "$fail" -ne 0 ]; then + echo "lint-shell: syntax errors above" >&2 + exit 1 +fi + +if command -v shellcheck >/dev/null 2>&1; then + # Advisory only: the tree has never been shellcheck-clean and + # gating on it here would block unrelated work. + git ls-files '*.sh' | xargs shellcheck --severity=error --format=gcc 2>/dev/null || true +fi + +echo "lint-shell: ok ($checked scripts)" diff --git a/scripts/test-backup-scripts.sh b/scripts/test-backup-scripts.sh new file mode 100755 index 00000000..07272efa --- /dev/null +++ b/scripts/test-backup-scripts.sh @@ -0,0 +1,179 @@ +#!/usr/bin/env bash +# SPDX-License-Identifier: AGPL-3.0-or-later +# +# Functional test for the two backup scripts that cron runs on the app +# box. Neither is covered by `go test`, and both were silently doing +# nothing observable for four months (0-byte logs, no heartbeat), so +# the invariants worth pinning are: +# +# 1. a successful run writes a heartbeat file the metrics collector +# can read (epoch seconds), and logs a start + status=ok line; +# 2. a failed rclone exits NON-ZERO and writes NO heartbeat — a +# stale heartbeat must never be refreshed by a broken run; +# 3. rclone's own chatter stays out of the cron-redirected stream. +# +# Everything external (rclone, sudo, pg_dump, pg_restore) is stubbed +# on PATH, so this runs anywhere with bash and no Postgres. + +set -euo pipefail + +cd "$(git rev-parse --show-toplevel)" + +BACKUP=deploy/postgres/backup-daily.sh +SYNC=deploy/spaces/sync-cross-region.sh + +fails=0 +ok() { printf ' ok %s\n' "$*"; } +bad() { printf ' FAIL %s\n' "$*" >&2; fails=$((fails + 1)); } + +work="$(mktemp -d)" +trap 'rm -rf "$work"' EXIT + +# --- stubs ----------------------------------------------------------- +bin="$work/bin" +mkdir -p "$bin" + +cat > "$bin/sudo" <<'EOF' +#!/usr/bin/env bash +[ "$1" = "-u" ] && shift 2 +exec "$@" +EOF + +# Writes an empty file wherever --file= points, so pg_restore --list +# and the retention glob have something to chew on. +cat > "$bin/pg_dump" <<'EOF' +#!/usr/bin/env bash +for a in "$@"; do + case "$a" in --file=*) : > "${a#--file=}" ;; esac +done +EOF + +cat > "$bin/pg_restore" <<'EOF' +#!/usr/bin/env bash +exit 0 +EOF + +# STUB_RCLONE_EXIT lets a case force the failure path. +cat > "$bin/rclone" <<'EOF' +#!/usr/bin/env bash +echo "rclone-chatter: $*" +exit "${STUB_RCLONE_EXIT:-0}" +EOF + +chmod +x "$bin"/* +export PATH="$bin:$PATH" + +# --- helpers --------------------------------------------------------- + +# heartbeat_is_epoch -> 0 when the file holds a plausible unix +# timestamp (10 digits, this century). +heartbeat_is_epoch() { + local v + v="$(cat "$1" 2>/dev/null || true)" + [[ "$v" =~ ^[0-9]{10}$ ]] && [ "$v" -gt 1600000000 ] +} + +# --- backup-daily ---------------------------------------------------- + +echo "backup-daily.sh" + +run_backup() { + local case_dir="$1" + mkdir -p "$case_dir" + SHITHUB_BACKUP_LOCAL="$case_dir/dumps" \ + SHITHUB_BACKUP_HEARTBEAT="$case_dir/heartbeat" \ + "$BACKUP" > "$case_dir/cron.log" 2>&1 +} + +c="$work/backup-ok" +if run_backup "$c"; then + ok "exits 0 on success" +else + bad "exits 0 on success (got $?)" +fi +grep -q 'backup-daily start' "$c/cron.log" \ + && ok "logs a start line" || bad "logs a start line" +grep -q 'backup-daily end status=ok exit=0' "$c/cron.log" \ + && ok "logs status=ok" || bad "logs status=ok" +if heartbeat_is_epoch "$c/heartbeat"; then + ok "writes an epoch-seconds heartbeat" +else + bad "writes an epoch-seconds heartbeat (got '$(cat "$c/heartbeat" 2>/dev/null)')" +fi +[ -e "$c/heartbeat.tmp" ] && bad "leaves no heartbeat temp file" + +c="$work/backup-fail" +mkdir -p "$c" +# Pre-seed a heartbeat from an earlier good run; a failed run must not +# touch it. +echo 1600000001 > "$c/heartbeat" +if STUB_RCLONE_EXIT=7 run_backup "$c"; then + bad "exits non-zero when rclone fails" +else + rc=$? + [ "$rc" -eq 7 ] && ok "propagates the rclone exit status ($rc)" \ + || ok "exits non-zero when rclone fails ($rc)" +fi +grep -q 'backup-daily end status=FAILED' "$c/cron.log" \ + && ok "logs status=FAILED" || bad "logs status=FAILED" +[ "$(cat "$c/heartbeat")" = "1600000001" ] \ + && ok "leaves the previous heartbeat untouched on failure" \ + || bad "leaves the previous heartbeat untouched on failure" + +# --- sync-cross-region ----------------------------------------------- + +echo "sync-cross-region.sh" + +run_sync() { + local case_dir="$1" + mkdir -p "$case_dir" + SHITHUB_SPACES_SYNC_LOG="$case_dir/spaces-sync.log" \ + SHITHUB_SPACES_SYNC_HEARTBEAT="$case_dir/heartbeat" \ + "$SYNC" > "$case_dir/cron.log" 2>&1 +} + +c="$work/sync-ok" +if run_sync "$c"; then + ok "exits 0 on success" +else + bad "exits 0 on success (got $?)" +fi +grep -q 'spaces-sync start' "$c/cron.log" \ + && ok "start line reaches the cron-redirected stream" \ + || bad "start line reaches the cron-redirected stream" +grep -q 'spaces-sync end status=ok exit=0' "$c/cron.log" \ + && ok "status=ok reaches the cron-redirected stream" \ + || bad "status=ok reaches the cron-redirected stream" +grep -q 'spaces-sync end status=ok exit=0' "$c/spaces-sync.log" \ + && ok "status=ok also reaches the script's own log" \ + || bad "status=ok also reaches the script's own log" +grep -q 'rclone-chatter' "$c/spaces-sync.log" \ + && ok "rclone output lands in the script's own log" \ + || bad "rclone output lands in the script's own log" +grep -q 'rclone-chatter' "$c/cron.log" \ + && bad "rclone output stays out of the cron log" \ + || ok "rclone output stays out of the cron log" +if heartbeat_is_epoch "$c/heartbeat"; then + ok "writes an epoch-seconds heartbeat" +else + bad "writes an epoch-seconds heartbeat" +fi + +c="$work/sync-fail" +mkdir -p "$c" +if STUB_RCLONE_EXIT=3 run_sync "$c"; then + bad "exits non-zero when rclone fails" +else + ok "exits non-zero when rclone fails ($?)" +fi +grep -q 'spaces-sync end status=FAILED' "$c/cron.log" \ + && ok "logs status=FAILED" || bad "logs status=FAILED" +[ -e "$c/heartbeat" ] && bad "writes no heartbeat on failure" \ + || ok "writes no heartbeat on failure" + +echo "" +if [ "$fails" -gt 0 ]; then + echo "test-backup-scripts: $fails assertion(s) failed" >&2 + exit 1 +fi +echo "test-backup-scripts: ok"