Public Overpass instances go bad without warning — we just lived through a day where overpass.private.coffee (our single upstream) returned 504s after 60s while lz4.overpass-api.de served the same query in 0.6s. Survive that by trying a list of upstreams in order until one responds usefully. Server changes: - OVERPASS_URLS env — comma-separated list, tried in order. Falls back to OVERPASS_URL for backward compat, then to a built-in default pair (lz4.overpass-api.de, overpass-api.de). - Per-upstream timeout of 10s via AbortSignal.timeout, so a saturated instance fails over in seconds, not minutes. - Client abort propagation via AbortSignal.any — if the browser cancels (e.g. user panned the map), the in-flight upstream fetch is aborted too. Stops wasting upstream capacity on requests nobody wants. - Rate-limited-body detection (Overpass returns 200 with a `rate_limited` marker when throttling) triggers failover. - Per-upstream labels on overpass_upstream_requests_total and overpass_upstream_duration_seconds so the dashboard can break out health + latency by instance. Histogram buckets extended to 30s. Compose: OVERPASS_URLS passthrough with the default pair hard-wired, overridable via SOPS. Tests: 8 cases covering the new fetchWithFailover helper — happy path, 5xx failover, rate_limited failover, network error failover, timeout tagging, all-upstreams-fail, client abort stops the loop, per-attempt latency observation. Follow-ups left out of scope: - Client-side debounce (UI concern). - Moving `.observe()` after body read for accuracy in measurement. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
217 lines
6.9 KiB
YAML
217 lines
6.9 KiB
YAML
services:
|
|
caddy:
|
|
image: caddy:2
|
|
restart: unless-stopped
|
|
ports:
|
|
- "80:80"
|
|
- "443:443"
|
|
- "443:443/udp"
|
|
environment:
|
|
DOMAIN: ${DOMAIN:-trails.cool}
|
|
volumes:
|
|
- ./Caddyfile:/etc/caddy/Caddyfile:ro
|
|
- caddy_data:/data
|
|
- caddy_config:/config
|
|
depends_on:
|
|
- journal
|
|
- planner
|
|
|
|
journal:
|
|
image: ghcr.io/trails-cool/journal:latest
|
|
restart: unless-stopped
|
|
environment:
|
|
DOMAIN: ${DOMAIN:-trails.cool}
|
|
ORIGIN: https://${DOMAIN:-trails.cool}
|
|
PLANNER_URL: https://planner.${DOMAIN:-trails.cool}
|
|
DATABASE_URL: postgres://trails:${POSTGRES_PASSWORD:-trails}@postgres:5432/trails
|
|
JWT_SECRET: ${JWT_SECRET:-change-me-in-production}
|
|
SESSION_SECRET: ${SESSION_SECRET:-change-me-in-production}
|
|
NODE_ENV: production
|
|
PORT: 3000
|
|
SENTRY_RELEASE: ${SENTRY_RELEASE:-}
|
|
SMTP_URL: ${SMTP_URL:-}
|
|
SMTP_FROM: ${SMTP_FROM:-trails.cool <noreply@trails.cool>}
|
|
WAHOO_CLIENT_ID: ${WAHOO_CLIENT_ID:-}
|
|
WAHOO_CLIENT_SECRET: ${WAHOO_CLIENT_SECRET:-}
|
|
WAHOO_WEBHOOK_TOKEN: ${WAHOO_WEBHOOK_TOKEN:-}
|
|
# Demo-activity-bot. Disabled by default everywhere; flip to "true"
|
|
# only in prod. When unset, DEMO_BOT_PERSONA uses the built-in
|
|
# Bruno/Berlin persona. See docs/demo-persona.md for the schema.
|
|
DEMO_BOT_ENABLED: ${DEMO_BOT_ENABLED:-}
|
|
DEMO_BOT_RETENTION_DAYS: ${DEMO_BOT_RETENTION_DAYS:-}
|
|
DEMO_BOT_REGION: ${DEMO_BOT_REGION:-}
|
|
DEMO_BOT_PERSONA: ${DEMO_BOT_PERSONA:-}
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "curl -sf http://localhost:3000/api/health || exit 1"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
retries: 3
|
|
depends_on:
|
|
postgres:
|
|
condition: service_healthy
|
|
|
|
planner:
|
|
image: ghcr.io/trails-cool/planner:latest
|
|
restart: unless-stopped
|
|
environment:
|
|
BROUTER_URL: http://brouter:17777
|
|
# Ordered failover list for the Overpass proxy. The code defaults
|
|
# to the same pair if unset; declaring it here makes the prod
|
|
# upstream explicit and easy to reshuffle via env when a
|
|
# community instance misbehaves. Override per-env with
|
|
# OVERPASS_URLS in the SOPS file.
|
|
OVERPASS_URLS: ${OVERPASS_URLS:-https://lz4.overpass-api.de/api/interpreter,https://overpass-api.de/api/interpreter}
|
|
DATABASE_URL: postgres://trails:${POSTGRES_PASSWORD:-trails}@postgres:5432/trails
|
|
NODE_ENV: production
|
|
PORT: 3001
|
|
SENTRY_RELEASE: ${SENTRY_RELEASE:-}
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "curl -sf http://localhost:3001/health || exit 1"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
retries: 3
|
|
depends_on:
|
|
postgres:
|
|
condition: service_healthy
|
|
brouter:
|
|
condition: service_started
|
|
|
|
brouter:
|
|
image: ghcr.io/trails-cool/brouter:latest
|
|
restart: unless-stopped
|
|
volumes:
|
|
- ./segments:/data/segments
|
|
# Segments can be pulled from:
|
|
# - https://brouter.de/brouter/segments4/ (official, weekly updates)
|
|
# - A trails.cool CDN mirror (later)
|
|
|
|
postgres:
|
|
image: postgis/postgis:16-3.4
|
|
restart: unless-stopped
|
|
environment:
|
|
POSTGRES_USER: trails
|
|
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:-trails}
|
|
POSTGRES_DB: trails
|
|
volumes:
|
|
- pgdata:/var/lib/postgresql/data
|
|
- ./postgres/init-grafana-user.sql:/docker-entrypoint-initdb.d/init-grafana-user.sql:ro
|
|
command:
|
|
- "postgres"
|
|
- "-c"
|
|
- "shared_preload_libraries=pg_stat_statements"
|
|
- "-c"
|
|
- "pg_stat_statements.track=all"
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U trails"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 5
|
|
|
|
postgres-exporter:
|
|
image: prometheuscommunity/postgres-exporter:latest
|
|
restart: unless-stopped
|
|
environment:
|
|
DATA_SOURCE_NAME: postgresql://trails:${POSTGRES_PASSWORD:-trails}@postgres:5432/trails?sslmode=disable
|
|
PG_EXPORTER_EXTEND_QUERY_PATH: /etc/postgres-exporter/queries.yml
|
|
volumes:
|
|
- ./postgres/queries.yml:/etc/postgres-exporter/queries.yml:ro
|
|
depends_on:
|
|
postgres:
|
|
condition: service_healthy
|
|
|
|
node-exporter:
|
|
image: prom/node-exporter:latest
|
|
restart: unless-stopped
|
|
pid: host
|
|
volumes:
|
|
- /proc:/host/proc:ro
|
|
- /sys:/host/sys:ro
|
|
- /:/rootfs:ro
|
|
command:
|
|
- "--path.procfs=/host/proc"
|
|
- "--path.sysfs=/host/sys"
|
|
- "--path.rootfs=/rootfs"
|
|
- "--collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host|etc)($$|/)"
|
|
|
|
cadvisor:
|
|
image: gcr.io/cadvisor/cadvisor:latest
|
|
restart: unless-stopped
|
|
privileged: true
|
|
volumes:
|
|
- /:/rootfs:ro
|
|
- /var/run:/var/run:ro
|
|
- /sys:/sys:ro
|
|
- /var/lib/docker/:/var/lib/docker:ro
|
|
- /dev/disk/:/dev/disk:ro
|
|
|
|
promtail:
|
|
image: grafana/promtail:latest
|
|
restart: unless-stopped
|
|
volumes:
|
|
- /var/run/docker.sock:/var/run/docker.sock:ro
|
|
- /var/lib/docker/containers:/var/lib/docker/containers:ro
|
|
- ./promtail/promtail-config.yml:/etc/promtail/config.yml:ro
|
|
command: ["-config.file=/etc/promtail/config.yml"]
|
|
depends_on:
|
|
- loki
|
|
|
|
prometheus:
|
|
image: prom/prometheus:latest
|
|
restart: unless-stopped
|
|
volumes:
|
|
- ./prometheus/prometheus.yml:/etc/prometheus/prometheus.yml:ro
|
|
- prometheus_data:/prometheus
|
|
command:
|
|
- "--config.file=/etc/prometheus/prometheus.yml"
|
|
- "--storage.tsdb.retention.time=15d"
|
|
- "--storage.tsdb.retention.size=1GB"
|
|
|
|
loki:
|
|
image: grafana/loki:latest
|
|
restart: unless-stopped
|
|
volumes:
|
|
- ./loki/loki-config.yml:/etc/loki/local-config.yaml:ro
|
|
- loki_data:/loki
|
|
command: ["-config.file=/etc/loki/local-config.yaml"]
|
|
|
|
grafana:
|
|
image: grafana/grafana:latest
|
|
restart: unless-stopped
|
|
environment:
|
|
GF_SERVER_ROOT_URL: https://grafana.internal.${DOMAIN:-trails.cool}
|
|
GF_AUTH_GITHUB_ENABLED: "true"
|
|
GF_AUTH_GITHUB_CLIENT_ID: ${GF_AUTH_GITHUB_CLIENT_ID:-}
|
|
GF_AUTH_GITHUB_CLIENT_SECRET: ${GF_AUTH_GITHUB_CLIENT_SECRET:-}
|
|
GF_AUTH_GITHUB_ALLOWED_ORGANIZATIONS: trails-cool
|
|
GF_AUTH_GITHUB_SCOPES: user:email,read:org
|
|
GF_AUTH_OAUTH_ALLOW_INSECURE_EMAIL_LOOKUP: "true"
|
|
GF_AUTH_DISABLE_LOGIN_FORM: "true"
|
|
GF_SMTP_ENABLED: "true"
|
|
GF_SMTP_HOST: ${GF_SMTP_HOST:-}
|
|
GF_SMTP_USER: ${GF_SMTP_USER:-}
|
|
GF_SMTP_PASSWORD: ${GF_SMTP_PASSWORD:-}
|
|
GF_SMTP_FROM_ADDRESS: noreply@trails.cool
|
|
GF_SMTP_FROM_NAME: trails.cool Grafana
|
|
GRAFANA_DB_PASSWORD: ${GRAFANA_DB_PASSWORD:-}
|
|
volumes:
|
|
- ./grafana/provisioning:/etc/grafana/provisioning:ro
|
|
- ./grafana/dashboards:/var/lib/grafana/dashboards:ro
|
|
- grafana_data:/var/lib/grafana
|
|
depends_on:
|
|
- prometheus
|
|
- loki
|
|
|
|
# garage:
|
|
# image: dxflrs/garage:v1.0
|
|
# restart: unless-stopped
|
|
# volumes:
|
|
# - garage_data:/var/lib/garage
|
|
# - ./garage.toml:/etc/garage.toml:ro
|
|
|
|
volumes:
|
|
caddy_data:
|
|
caddy_config:
|
|
pgdata:
|
|
prometheus_data:
|
|
loki_data:
|
|
grafana_data:
|