diff --git a/.github/workflows/cd-apps.yml b/.github/workflows/cd-apps.yml index 5ba0f06..6277543 100644 --- a/.github/workflows/cd-apps.yml +++ b/.github/workflows/cd-apps.yml @@ -115,8 +115,14 @@ jobs: docker image prune -af docker compose ps - # Annotate deploy in Grafana - GRAFANA_TOKEN=$(grep GRAFANA_SERVICE_TOKEN .env | cut -d= -f2- 2>/dev/null) + # Annotate deploy in Grafana. The token lives in the + # decrypted SOPS env file we just scp'd to /opt/trails-cool + # — that file is `app.env`, not `.env`. (Pre-fix this read + # the wrong path, so annotations were silently no-op'ing + # every deploy.) `2>/dev/null` keeps a missing token from + # failing the deploy; `|| true` keeps the curl from + # failing the deploy if Grafana itself is unhealthy. + GRAFANA_TOKEN=$(grep GRAFANA_SERVICE_TOKEN app.env 2>/dev/null | cut -d= -f2-) if [ -n "$GRAFANA_TOKEN" ]; then docker compose exec -T grafana curl -sf -X POST \ -H "Authorization: Bearer $GRAFANA_TOKEN" \ diff --git a/infrastructure/Caddyfile b/infrastructure/Caddyfile index f2a61b6..6d80f0a 100644 --- a/infrastructure/Caddyfile +++ b/infrastructure/Caddyfile @@ -28,7 +28,17 @@ output stdout format json } - reverse_proxy journal:3000 + reverse_proxy journal:3000 { + # During an `apps` deploy the journal container is briefly down + # (~10–30s) while compose swaps containers. Without these, + # Caddy returns 502 immediately and the `caddy-502-rate` alert + # trips on every deploy. With them, Caddy holds and retries + # against the upstream for up to 30s — restart becomes + # invisible to clients. A real outage longer than 30s still + # 502s and correctly trips the alert. + lb_try_duration 30s + lb_try_interval 250ms + } } www.{$DOMAIN:trails.cool} { @@ -53,5 +63,9 @@ planner.{$DOMAIN:trails.cool} { output stdout format json } - reverse_proxy planner:3001 + reverse_proxy planner:3001 { + # Same rationale as the journal block — see the comment there. + lb_try_duration 30s + lb_try_interval 250ms + } } diff --git a/infrastructure/grafana/provisioning/alerting/alerts.yml b/infrastructure/grafana/provisioning/alerting/alerts.yml index aaae096..1ce2546 100644 --- a/infrastructure/grafana/provisioning/alerting/alerts.yml +++ b/infrastructure/grafana/provisioning/alerting/alerts.yml @@ -206,6 +206,12 @@ groups: annotations: summary: "BRouter host metrics scrape has been failing for 2+ minutes — the dedicated host, vSwitch, or cAdvisor may be down" + # The threshold here is intentionally `> 0` for 2m — *any* + # sustained 502 stream is real. Deploy-time restarts no longer + # produce 502s thanks to `lb_try_duration` on Caddy's reverse + # proxy (see `infrastructure/Caddyfile`); if 502s appear here + # it means the upstream has been unreachable for longer than + # Caddy's retry window, which is a genuine outage. - uid: caddy-502-rate title: Caddy 502 errors detected condition: B