trails/infrastructure/grafana/provisioning/alerting/alerts.yml
Ullrich Schäfer 96de8831cf
poi-index: Grafana dashboard/alert, docs, privacy manifest
- planner dashboard: replace Overpass panels with POI index freshness +
  serving panels (age, rows/category, import status, API request rate/errors)
- alerts: replace overpass-upstream-unhealthy with poi-index-stale (>6 weeks)
- architecture.md: POI data flow now the self-hosted index
- privacy manifest (DE+EN): Overpass is Journal-surface-backfill-only;
  Planner POIs served same-origin from /api/pois
- self-host-overpass README + roadmap: superseded for POIs by poi-index
- poi-extract README: self-hoster story (optional pipeline, regional extract,
  graceful empty index)
- map-core tsconfig: exclude *.sync.test.ts from tsc to keep it zero-dep

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-12 22:54:51 +02:00

326 lines
12 KiB
YAML

apiVersion: 1
groups:
- orgId: 1
name: trails.cool alerts
folder: trails.cool
interval: 1m
rules:
- uid: disk-usage-high
title: Disk usage > 80%
condition: B
data:
- refId: A
relativeTimeRange: { from: 600, to: 0 }
datasourceUid: prometheus
model:
expr: (1 - node_filesystem_avail_bytes{mountpoint="/"} / node_filesystem_size_bytes{mountpoint="/"}) * 100
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [80], type: gt }
operator: { type: and }
reducer: { type: last }
for: 15m
annotations:
summary: "Disk usage is above 80%"
- uid: app-health-failing
title: App health check failing
condition: B
data:
- refId: A
relativeTimeRange: { from: 300, to: 0 }
datasourceUid: prometheus
model:
expr: up{job=~"journal|planner"}
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [1], type: lt }
operator: { type: and }
reducer: { type: last }
for: 2m
annotations:
summary: "{{ $labels.job }} is down"
- uid: error-rate-high
title: Error rate > 5%
condition: B
noDataState: OK
data:
- refId: A
relativeTimeRange: { from: 600, to: 0 }
datasourceUid: prometheus
model:
expr: (sum(rate(caddy_http_request_duration_seconds_count{code=~"5.."}[5m])) or vector(0)) / clamp_min(sum(rate(caddy_http_request_duration_seconds_count[5m])), 0.001) * 100
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [5], type: gt }
operator: { type: and }
reducer: { type: last }
for: 5m
annotations:
summary: "Error rate is above 5%"
- uid: container-restart-loop
title: Container restart loop
condition: B
data:
- refId: A
relativeTimeRange: { from: 300, to: 0 }
datasourceUid: prometheus
model:
expr: changes(container_start_time_seconds{name=~"trails-cool.*"}[5m])
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [2], type: gt }
operator: { type: and }
reducer: { type: last }
for: 1m
annotations:
summary: "{{ $labels.name }} has restarted more than 2 times in 5 minutes"
- uid: db-connections-high
title: PostgreSQL connections > 80
condition: B
data:
- refId: A
relativeTimeRange: { from: 300, to: 0 }
datasourceUid: prometheus
model:
expr: sum(pg_stat_activity_count{datname="trails"})
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [80], type: gt }
operator: { type: and }
reducer: { type: last }
for: 5m
annotations:
summary: "PostgreSQL active connections above 80 — approaching default max_connections limit"
- uid: app-crash-log
title: Application crash detected
condition: C
data:
- refId: A
relativeTimeRange: { from: 300, to: 0 }
datasourceUid: loki
model:
expr: sum(count_over_time({service=~"journal|planner"} |~ "uncaughtException|unhandledRejection|ERR_|FATAL|segfault|OOMKilled"[5m]))
instant: true
- refId: B
datasourceUid: __expr__
model:
type: reduce
expression: A
reducer: last
- refId: C
datasourceUid: __expr__
model:
type: threshold
expression: B
conditions:
- evaluator: { params: [0], type: gt }
operator: { type: and }
reducer: { type: last }
for: 0s
noDataState: OK
annotations:
summary: "Crash signature detected in application logs"
- uid: background-job-failures
title: Background job failures
condition: C
noDataState: OK
data:
- refId: A
relativeTimeRange: { from: 3600, to: 0 }
datasourceUid: postgres
model:
rawSql: "SELECT count(*) AS failed FROM pgboss.job WHERE state = 'failed' AND completed_on > now() - interval '1 hour'"
format: table
- refId: B
datasourceUid: __expr__
model:
type: reduce
expression: A
reducer: last
- refId: C
datasourceUid: __expr__
model:
type: threshold
expression: B
conditions:
- evaluator: { params: [0], type: gt }
operator: { type: and }
reducer: { type: last }
for: 0s
annotations:
summary: "Background jobs have failed in the last hour — check Grafana Service Health dashboard"
- uid: brouter-scrape-down
title: BRouter host unreachable
condition: B
noDataState: Alerting
data:
- refId: A
relativeTimeRange: { from: 300, to: 0 }
datasourceUid: prometheus
model:
expr: up{job="brouter-cadvisor"}
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [1], type: lt }
operator: { type: and }
reducer: { type: last }
for: 2m
annotations:
summary: "BRouter host metrics scrape has been failing for 2+ minutes — the dedicated host, vSwitch, or cAdvisor may be down"
# The threshold here is intentionally `> 0` for 2m — *any*
# sustained 502 stream is real. Deploy-time restarts no longer
# produce 502s thanks to `lb_try_duration` on Caddy's reverse
# proxy (see `infrastructure/Caddyfile`); if 502s appear here
# it means the upstream has been unreachable for longer than
# Caddy's retry window, which is a genuine outage.
- uid: caddy-502-rate
title: Caddy 502 errors detected
condition: B
noDataState: OK
data:
- refId: A
relativeTimeRange: { from: 300, to: 0 }
datasourceUid: prometheus
model:
expr: sum(rate(caddy_http_request_duration_seconds_count{code="502"}[5m])) or vector(0)
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [0], type: gt }
operator: { type: and }
reducer: { type: last }
for: 2m
annotations:
summary: "Caddy is returning 502 errors — journal or planner upstream unreachable"
# Prometheus self-health. A corrupt WAL segment once made every
# compaction fail for days, silently collapsing metric retention to
# the ~2h head block (size-retention kept deleting fresh blocks).
# This fires on any compaction failure OR WAL corruption in the last
# hour. Depends on the `prometheus` self-scrape job.
- uid: prometheus-compaction-failing
title: Prometheus TSDB compaction failing
condition: B
noDataState: OK
data:
- refId: A
relativeTimeRange: { from: 3600, to: 0 }
datasourceUid: prometheus
model:
expr: increase(prometheus_tsdb_compactions_failed_total[1h]) + increase(prometheus_tsdb_wal_corruptions_total[1h])
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
- evaluator: { params: [0], type: gt }
operator: { type: and }
reducer: { type: last }
for: 10m
annotations:
summary: "Prometheus TSDB compaction or WAL is failing — metric retention is at risk. Check the WAL for a corrupt segment (this silently capped retention to ~2h before)."
# POI index freshness. POIs are served from the self-hosted
# `planner.pois` index, refreshed by a monthly extract+import. If the
# index age exceeds ~6 weeks the monthly refresh has missed a cycle
# (extract job, vSwitch transfer, checksum, guard, or import failed) and
# the data is going stale. `poi_index_age_seconds` is published by the
# Planner from its own DB; a growing value also covers a silently-failing
# import (the guard refuses to swap, so age keeps climbing). Import-run
# detail lives on the Planner dashboard (poi_import_last_status).
- uid: poi-index-stale
title: POI index is stale
condition: B
noDataState: OK
data:
- refId: A
relativeTimeRange: { from: 600, to: 0 }
datasourceUid: prometheus
model:
expr: max(poi_index_age_seconds)
instant: true
- refId: B
datasourceUid: __expr__
model:
type: threshold
expression: A
conditions:
# 6 weeks = 3628800s — one missed monthly refresh.
- evaluator: { params: [3628800], type: gt }
operator: { type: and }
reducer: { type: last }
for: 1h
annotations:
summary: "POI index age exceeds 6 weeks — the monthly refresh has missed a cycle. Check the poi-extract timer on the BRouter host and poi-import on the flagship (Planner dashboard: Last Import Status)."
contactPoints:
# Single "default" contact point with multiple integrations: every alert
# fan-outs to both email and Pushover. Grafana resolves $VAR / ${VAR}
# tokens from the container's env at startup, so the secrets never land
# in the provisioning YAML or the git history. See SOPS
# `secrets.infra.env` for PUSHOVER_*.
- orgId: 1
name: default
receivers:
- uid: email-default
type: email
settings:
addresses: admin@trails.cool
- uid: pushover-ullrich
type: pushover
settings:
apiToken: $PUSHOVER_API_TOKEN
userKey: $PUSHOVER_USER_KEY_ULLRICH
# High-priority push (bypass quiet hours) on alert; default
# priority on resolve. Tweak if this gets noisy.
priority: "1"
okPriority: "0"
policies:
- orgId: 1
receiver: default