Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,21 @@ jobs:
# memory ceiling before the box does.
run: scripts/lint-systemd-units.sh

- name: Lint - shell syntax
run: scripts/lint-shell.sh

- name: Lint - docs topology
# Production is one droplet with no VPN; the docs described a
# WireGuard mesh and a monitoring host that never existed.
run: scripts/lint-docs-topology.sh

- name: Test - backup cron scripts
# deploy/postgres/backup-daily.sh and
# deploy/spaces/sync-cross-region.sh run unattended from root
# crontab and are not covered by go test; both logged nothing
# for four months before anyone noticed.
run: scripts/test-backup-scripts.sh

- name: Test
run: go test -trimpath ./...

Expand Down
11 changes: 10 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,7 @@ assets: ## Copy Primer CSS into internal/web/static/ for embedding.
echo "warn: .refs/primer-css/dist not found; run 'git clone https://github.com/primer/css .refs/primer-css' first"; \
fi

ci: lint lint-policy lint-markdown lint-org-plan lint-secret-logs lint-spdx lint-unused lint-migrations lint-systemd-units verify-api-docs test build ## Full CI pipeline (matches .github/workflows/ci.yml).
ci: lint lint-policy lint-markdown lint-org-plan lint-secret-logs lint-spdx lint-unused lint-migrations lint-systemd-units lint-shell lint-docs-topology test-backup-scripts verify-api-docs test build ## Full CI pipeline (matches .github/workflows/ci.yml).
@echo "ci: ok"

lint-policy: ## Enforce policy-package boundary (no inline auth checks in handlers/git/cmd).
Expand All @@ -99,6 +99,15 @@ lint-migrations: ## Fail when goose migration numeric versions collide.
lint-systemd-units: ## Fail when a shipped systemd unit loses its memory ceiling or collides with the backup window.
@scripts/lint-systemd-units.sh

lint-shell: ## Syntax-check every shell script (deploy/ runs unattended from cron).
@scripts/lint-shell.sh

lint-docs-topology: ## Fail when docs/internal claims a WireGuard mesh outside a marked aspirational block.
@scripts/lint-docs-topology.sh

test-backup-scripts: ## Functional test for the backup cron scripts (logging, heartbeat, failure exit).
@scripts/test-backup-scripts.sh

verify-api-docs: ## Fail when an /api/v1 route in code is missing from docs/public/api/.
@scripts/verify-api-docs.sh

Expand Down
63 changes: 54 additions & 9 deletions deploy/ansible/roles/postgres/templates/postgresql.conf.j2
Original file line number Diff line number Diff line change
@@ -1,23 +1,68 @@
# Managed by Ansible — edits here are overwritten on next deploy.
# Tunes for shithub's MVP single-droplet workload.
#
# Sized for the single-box deployment: 2 vCPU / 3.9 GB shared with
# shithubd web + worker (~200 MB steady state after the Phase 2
# renderer fix, ceilinged at 1.6/2.0 GB and 0.75/1.0 GB by systemd),
# Caddy, Alloy (~320 MB) and the backup crons (rclone peaks at
# 1.0-1.6 GB). Postgres does NOT get the usual 25%-of-RAM share here,
# because the 25% rule assumes a dedicated database host.
#
# NOTE: this template has never been applied to shithub-app. The live
# box runs Debian defaults (shared_buffers 128MB, work_mem 4MB,
# max_connections 100, no pg_stat_statements). Applying it needs a
# Postgres RESTART, not a reload. Read docs/internal/db.md before
# doing it.

listen_addresses = 'localhost' # over WG only; no public Postgres
listen_addresses = 'localhost' # loopback only; no public Postgres
port = 5432
max_connections = 100
shared_buffers = 512MB # ~25% of 2GB droplet RAM
effective_cache_size = 1500MB # rough working set hint
maintenance_work_mem = 64MB
work_mem = 8MB

# 60, not 100. Actual demand: web pool 10 (db.max_conns), worker
# 6 (SHITHUB_WORKERS=4 + 2), cron/hook/ssh/admin invocations at 2-4
# each and short-lived. 60 leaves ~35 of headroom over the steady
# ~19 observed on 2026-09-02 and caps the worst case at a bounded
# per-backend footprint; 100 backends each able to take work_mem is
# how a shared box gets OOM-killed.
max_connections = 60

# 256MB, not 512MB. Double the live Debian default (which is what the
# 447 MB aggregate RSS on the box was measured against), but small
# enough that Postgres + shithubd + Alloy + a 1.6 GB rclone still fit
# under 3.9 GB with the 4 GB swapfile as slack rather than as a
# working surface.
shared_buffers = 256MB

# What the kernel is likely caching, not an allocation. The DB is
# ~1 GB and the page cache on this box is regularly squeezed by
# rclone and AIDE, so 1GB is the honest hint; overstating it pushes
# the planner toward index scans that are not actually cheap here.
effective_cache_size = 1GB

maintenance_work_mem = 64MB # VACUUM/CREATE INDEX; one at a time

# 4MB (the Debian default the box already runs). This is per sort/hash
# *node*, not per connection, so the ceiling is roughly
# max_connections x nodes x work_mem. At 60 connections a bump to 8MB
# doubles a tail risk we cannot afford on a shared box; raise it per
# session (SET LOCAL work_mem) for a specific heavy query instead.
work_mem = 4MB
wal_level = replica # required for WAL archiving (S37)
archive_mode = on
archive_command = '/usr/local/bin/shithub-pg-archive %p %f'
archive_timeout = 60 # at least one segment per minute
checkpoint_completion_target = 0.9

# pg_stat_statements (S36 perf-pass requirement; S37 deploy installs).
# pg_stat_statements. Not installed on the live box — the extension
# needs this preload line AND a restart AND `CREATE EXTENSION`, and
# none of that has happened, which is why the 2026-09-02 outage
# investigation had no query attribution to work from.
#
# track = top (not all): `all` also records statements executed inside
# PL/pgSQL function bodies, which multiplies entries and shared-memory
# pressure for attribution we do not need. Top-level statements are
# what maps back to a handler or a job.
shared_preload_libraries = 'pg_stat_statements'
pg_stat_statements.max = 5000
pg_stat_statements.track = all
pg_stat_statements.track = top

# Logging — minimal noise, slow-query attention.
log_destination = 'stderr'
Expand Down
35 changes: 34 additions & 1 deletion deploy/audit/check-droplet-drift.sh
Original file line number Diff line number Diff line change
Expand Up @@ -37,18 +37,45 @@ declare -a MANAGED=(
"/etc/alloy/credentials.env::TEMPLATE"
"/etc/systemd/system/alloy.service.d/shithub.conf::TEMPLATE"
"/etc/postgresql/16/main/conf.d/99_shithub_archive.conf::TEMPLATE"
# Known drift, tracked deliberately: the box runs Debian defaults
# here. deploy/ansible/roles/postgres/templates/postgresql.conf.j2
# has never been applied (see docs/internal/db.md); applying it needs
# a Postgres restart in the 05:15-06:00 window, not a `make deploy`.
"/etc/postgresql/16/main/postgresql.conf::TEMPLATE"
"/etc/aide/aide.conf.d/99_shithub_exclude::deploy/ansible/roles/base/files/aide-shithub.conf"
"/etc/cron.daily/aide::TEMPLATE"
"/etc/caddy/Caddyfile::TEMPLATE"
"/etc/fail2ban/jail.d/shithub.local::TEMPLATE"
"/etc/fail2ban/filter.d/shithubd-auth.conf::TEMPLATE"
"/etc/systemd/system/shithubd-web.service::TEMPLATE"
"/etc/systemd/system/shithubd-worker.service::TEMPLATE"
# Known drift: web.env carries a hand-edited Stripe block that is
# not in web.env.j2. A `make deploy` re-renders this file from the
# template and WILL drop those keys — see KNOWN_DRIFT below.
"/etc/shithub/web.env::TEMPLATE"
"/etc/shithub/worker.env::TEMPLATE"
"/etc/ssh/sshd_config::TEMPLATE"
)

# Paths whose drift is known and accepted. Reported as KNOWN rather
# than silently passing: the point is that the next operator reads the
# reason instead of rediscovering it.
declare -a KNOWN_DRIFT=(
"/etc/postgresql/16/main/postgresql.conf::postgresql.conf.j2 has never been applied; needs a restart, see docs/internal/db.md"
"/etc/shithub/web.env::hand-edited Stripe block not present in web.env.j2; a full deploy re-renders this file and drops it"
)

known_drift_note() {
local entry
for entry in "${KNOWN_DRIFT[@]}"; do
if [ "${entry%%::*}" = "$1" ]; then
printf '%s' "${entry##*::}"
return 0
fi
done
return 1
}

DRIFT_COUNT=0

# Build a single-shot SSH script that returns md5 + stat for every
Expand Down Expand Up @@ -94,7 +121,11 @@ while IFS='|' read -r dpath status remote_md5 remote_stat; do
fi

if [ "$src" = "TEMPLATE" ]; then
printf "%-60s \033[36m%-10s\033[0m %s (template — manual check)\n" "$dpath" "TEMPLATE" "$remote_stat"
if note=$(known_drift_note "$dpath"); then
printf "%-60s \033[33m%-10s\033[0m %s\n" "$dpath" "KNOWN" "$note"
else
printf "%-60s \033[36m%-10s\033[0m %s (template — manual check)\n" "$dpath" "TEMPLATE" "$remote_stat"
fi
continue
fi

Expand All @@ -120,4 +151,6 @@ if [ "$DRIFT_COUNT" -gt 0 ]; then
exit 1
fi
echo "No drift detected on copy: files. TEMPLATE rows still need manual review."
echo "KNOWN rows are accepted drift with a reason attached — do not 'fix' one"
echo "by running a full deploy without reading the note."
exit 0
97 changes: 97 additions & 0 deletions deploy/monitoring/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
# `deploy/monitoring/` — not deployed

**Nothing in this directory runs in production**, with one exception
noted below. These files describe a self-hosted Prometheus + Loki +
Alertmanager + Grafana monitoring host that has never existed. They
are kept because they are the expression catalogue for the alerts we
want and the starting point if we ever outgrow Grafana Cloud — not
because they are live.

## What actually runs

shithub.sh is a single droplet. The `monitoring-client` Ansible role
installs `node_exporter` (`127.0.0.1:9100`) and **Grafana Alloy**,
which scrapes node_exporter and `shithubd web`'s
`127.0.0.1:8080/metrics` and `remote_write`s to **Grafana Cloud**.
Push-only; no inbound monitoring port; no local Prometheus, Grafana,
Loki or Alertmanager process. See `docs/internal/deploy.md` and
`docs/internal/runbooks/observability.md`.

## File by file

| Path | Status |
|---|---|
| `prometheus/prometheus.yml` | **Inert.** Scrapes six targets on a `10.50.0.0/24` WireGuard mesh that does not exist, and points `alerting.alertmanagers` at `10.50.0.10:9093`. No process reads it. |
| `prometheus/rules.yml` | **Inert.** No Prometheus loads it and there is no Alertmanager to route it. Useful as PromQL source material; see the caveats below before copying an expression. |
| `alertmanager/alertmanager.yml` | **Inert.** No Alertmanager is installed. The SMTP/webhook receivers are placeholders. |
| `loki/loki-config.yaml` | **Inert.** No Loki, and nothing ships logs — there is no Promtail and the Alloy config has no logs component. Logs are journald-only. |
| `grafana/dashboards/*.json` | **Import by hand.** These work against the Grafana Cloud stack via Dashboards → New → Import. Panel queries are mirrored in `runbooks/observability.md`. |
| `grafana/provision-actions-alerts.sh` | **Live tool.** The one thing here that talks to production. It creates/updates a Grafana-*managed* alert rule in the Cloud stack over the HTTP API. Which rules are currently provisioned is state in Grafana Cloud, not in this repo. |

## Which documented alerts therefore do not exist

Every rule in `prometheus/rules.yml` is unfired:

- `shithubd-availability`: `ShithubdWebDown`, `ShithubdWorkerDown`,
`PostgresDown`
- `shithubd-latency`: `HighRequestLatencyP95`, `HighDBQueryRate`
- `shithubd-jobs`: `JobBacklogGrowing`, `WebhookDeliveryFailing`
- `shithubd-billing`: `BillingWebhookFailureRateHigh`,
`BillingWebhookBacklogHigh`, `BillingWebhookFailedReceipt`,
`BillingCheckoutFailures`, `BillingSeatDrift`,
`BillingPastDuePrincipals`, `BillingQuotaOverage`
- `shithubd-actions`: `ActionsRunnerHeartbeatStale`,
`ActionsRunnerIdleWithAssignedJobs`, `ActionsQueueDepthHigh`,
`ActionsRunDurationP99Regressed`,
`ActionsLogScrubberPossiblyMissing`
- `shithubd-backups`: `BackupOverdue`

The runbook anchors these rules reference
(`docs/internal/runbooks/incidents.md`, `backups.md`,
`stripe-billing.md`) are still correct as *procedures*; they are just
not triggered by anything. What does page today is the DigitalOcean
droplet + uptime alert set (`docs/internal/runbooks/alerts.md`).

## Caveats before copying an expression into Grafana Cloud

1. **Job labels differ.** Alloy scrapes with `job="shithubd"` and
`job="node"`. Anything selecting `job="shithubd-web"`,
`job="shithubd-worker"`, `job="postgres"` or `job="caddy"` matches
nothing.
2. **The worker is not scraped at all.** `shithubd worker` starts no
HTTP listener, so `ShithubdWorkerDown` and every `shithub_worker_*`
expression are unsatisfiable regardless of relabelling.
3. **No Postgres exporter.** `PostgresDown` and any `pg_*` series have
no source. WAL-archive health is checked by the hourly
`shithub-verify-wal-archive` cron job, which logs and journals.
4. **`BackupOverdue` names a metric that never existed.** It selects
`shithubd_backup_last_success_seconds`; our metric namespace is
`shithub_`, and no such gauge was ever emitted. It is now real as
`shithub_backup_last_success_seconds{job="daily"|"spaces-sync"}`,
sourced from the heartbeat files the backup scripts write — see
`docs/internal/runbooks/backups.md`.

## What it would take to use these files

Roughly, in order:

1. A second droplet for the monitoring host, and a private network or
VPN between it and the app box — `deploy/ansible/roles/wireguard/`
exists but is not run by any current inventory, and the app box has
no `wg0`.
2. A monitoring-host play. There is none in this repo; `deploy.md`
used to claim it lived "outside this repo" and it does not exist
either.
3. Exporters for the targets the scrape config assumes:
`postgres_exporter`, Caddy admin metrics, and a metrics listener in
`shithubd worker` (a code change).
4. Relabelling `prometheus.yml` from mesh addresses to real ones and
fixing the job names, or relabelling `rules.yml` to match Alloy's.
5. Real Alertmanager receivers — the SMTP host, credentials file and
pager webhook in `alertmanager.yml` are all placeholders.
6. A logs pipeline if `loki-config.yaml` is to mean anything: an Alloy
`loki.source.journal` component plus a Loki endpoint.

Until at least steps 1–4 are done, the cheaper path for any alert you
want is a Grafana-managed rule in the existing Cloud stack, added to
`grafana/provision-actions-alerts.sh` so it stays reproducible.
7 changes: 6 additions & 1 deletion deploy/monitoring/prometheus/rules.yml
Original file line number Diff line number Diff line change
Expand Up @@ -196,8 +196,13 @@ groups:
- name: shithubd-backups
interval: 5m
rules:
# `shithubd_backup_last_success_seconds` never existed — our
# namespace is `shithub_`. The gauge is real now, sourced from
# the heartbeat files the backup scripts write
# (internal/infra/metrics/backupobserver.go). It is absent until
# the first success on a host, hence the absent() arm.
- alert: BackupOverdue
expr: time() - shithubd_backup_last_success_seconds > 60 * 60 * 30
expr: absent(shithub_backup_last_success_seconds{job="daily"}) or time() - shithub_backup_last_success_seconds{job="daily"} > 60 * 60 * 30
for: 0m
labels: {severity: page}
annotations:
Expand Down
39 changes: 37 additions & 2 deletions deploy/postgres/backup-daily.sh
Original file line number Diff line number Diff line change
Expand Up @@ -8,8 +8,13 @@
# anything older than 30 days; PITR rolls forward from the WAL
# archive (see archive_command.sh).
#
# Exit non-zero on any failure so the systemd timer surfaces it
# (OnFailure= → alertmanager).
# Exit non-zero on any failure. Nothing watches that exit code today
# — there is no Alertmanager and no systemd timer, this runs from root
# crontab at 03:17 with output appended to /var/log/shithub-backup.log
# — so the run also brackets itself with timestamped start/end lines
# and drops a heartbeat file on success. That log was 0 bytes from
# 2026-05-10 to 2026-09-02 precisely because a silent success and a
# silent absence look identical.

set -euo pipefail

Expand All @@ -19,6 +24,28 @@ LOCAL_DIR="${SHITHUB_BACKUP_LOCAL:-/var/backups/shithub}"
STAMP="$(date -u +%Y%m%dT%H%M%SZ)"
NAME="${DB}-${STAMP}.dump"

# Read by shithub_backup_last_success_seconds{job="daily"} — see
# internal/infra/metrics/backupobserver.go. Epoch seconds, one line.
HEARTBEAT="${SHITHUB_BACKUP_HEARTBEAT:-/var/lib/shithub/backup-last-success}"

ts() { date -u +%Y-%m-%dT%H:%M:%SZ; }

# Bracket the run. `set -e` makes most failures land here with a
# non-zero status, and the trap re-raises it so cron/systemd still see
# a failed run: the log line is an addition, never a swallow.
on_exit() {
local rc=$?
if [ "$rc" -eq 0 ]; then
echo "[$(ts)] backup-daily end status=ok exit=0 dump=$NAME"
else
echo "[$(ts)] backup-daily end status=FAILED exit=$rc dump=$NAME"
fi
return "$rc"
}
trap on_exit EXIT

echo "[$(ts)] backup-daily start db=$DB bucket=$BUCKET"

mkdir -p "$LOCAL_DIR"

# pg_dump as the postgres user via local-socket peer auth.
Expand All @@ -38,3 +65,11 @@ rclone --config /etc/rclone-shithub.conf --s3-no-check-bucket \
# Local retention: keep the last 7 dumps; bucket lifecycle handles
# the long tail.
ls -1t "$LOCAL_DIR"/*.dump 2>/dev/null | tail -n +8 | xargs -r rm -f

# Success only. Everything above is `set -e`-guarded, so reaching this
# line means the dump was taken, verified with pg_restore --list, and
# uploaded. Written via a temp file + mv so a reader never sees a
# half-written timestamp.
mkdir -p "$(dirname "$HEARTBEAT")"
printf '%s\n' "$(date -u +%s)" > "$HEARTBEAT.tmp"
mv "$HEARTBEAT.tmp" "$HEARTBEAT"
Loading
Loading