* HH-445: deploy production observability and runbooks * fix(ops): share production database DSN * fix(HH-445): enforce database TLS gate * fix(HH-445): preserve production serve command * fix(prod): require external database dependencies * fix(prod): unify database host rejection gates * test(prod): enforce exact database TLS runbook contract --------- Co-authored-by: Rogee <rogee@ipao.vip>
160 lines
5.7 KiB
YAML
160 lines
5.7 KiB
YAML
groups:
|
|
- name: gochat-application
|
|
rules:
|
|
- alert: GoChatAppDown
|
|
expr: up{job="gochat"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: GoChat application is down
|
|
description: GoChat metrics have been unreachable for more than one minute.
|
|
|
|
- alert: GoChatDatabaseReadinessFailed
|
|
expr: probe_success{job="gochat-database-readiness"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: GoChat database readiness failed
|
|
description: The application cannot complete its PostgreSQL dependency check.
|
|
|
|
- alert: GoChatRedisReadinessFailed
|
|
expr: probe_success{job="gochat-redis-readiness"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: GoChat Redis readiness failed
|
|
description: The application cannot complete its Redis dependency check.
|
|
|
|
- alert: GoChatHighErrorRate
|
|
expr: |
|
|
sum by (job, instance) (rate(http_request_errors_total{job="gochat"}[5m]))
|
|
/ sum by (job, instance) (rate(http_requests_total{job="gochat"}[5m])) > 0.05
|
|
and sum by (job, instance) (rate(http_requests_total{job="gochat"}[5m])) > 0
|
|
for: 5m
|
|
labels:
|
|
severity: warning
|
|
annotations:
|
|
summary: GoChat error rate above 5%
|
|
description: Error rate is {{ $value | humanizePercentage }} over the last 5 minutes.
|
|
|
|
- alert: GoChatHighMemory
|
|
expr: gochat_go_memory_alloc_bytes / 1024 / 1024 > 400
|
|
for: 5m
|
|
labels:
|
|
severity: warning
|
|
annotations:
|
|
summary: GoChat memory usage above 400 MB
|
|
description: Allocated heap is {{ $value | humanize }} MB.
|
|
|
|
- alert: GoChatHighGoroutines
|
|
expr: gochat_go_goroutines > 1000
|
|
for: 5m
|
|
labels:
|
|
severity: warning
|
|
annotations:
|
|
summary: GoChat goroutine count above 1000
|
|
description: GoChat has {{ $value | humanize }} goroutines.
|
|
|
|
- name: gochat-workers
|
|
rules:
|
|
- alert: GoChatWorkerDown
|
|
expr: |
|
|
time() - max(container_last_seen{container_label_com_docker_compose_service="worker"}) > 60
|
|
or absent(container_last_seen{container_label_com_docker_compose_service="worker"})
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: GoChat worker is down
|
|
description: cAdvisor has not observed a production worker container for more than one minute.
|
|
|
|
- alert: GoChatCriticalQueueBacklog
|
|
expr: |
|
|
sum by (queue) (gochat_background_jobs_total{queue=~"critical|high",status=~"queued|retrying"}) > 25
|
|
or max by (queue) (gochat_background_jobs_oldest_seconds{queue=~"critical|high",status=~"queued|retrying"}) > 300
|
|
for: 5m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: Critical background queue is delayed
|
|
description: Queue {{ $labels.queue }} exceeds 25 ready jobs or its oldest job is over five minutes old.
|
|
|
|
- alert: GoChatWorkerJobsStuck
|
|
expr: max by (queue) (gochat_background_jobs_oldest_seconds{status="running"}) > 900
|
|
for: 5m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: Background jobs are stuck
|
|
description: Queue {{ $labels.queue }} has a running job older than 15 minutes.
|
|
|
|
- name: gochat-infrastructure
|
|
rules:
|
|
- alert: GoChatPostgresExporterDown
|
|
expr: up{job="postgres-exporter"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: PostgreSQL exporter is down
|
|
description: Prometheus cannot scrape the PostgreSQL exporter.
|
|
|
|
- alert: GoChatRedisExporterDown
|
|
expr: up{job="redis-exporter"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: Redis exporter is down
|
|
description: Prometheus cannot scrape the Redis exporter.
|
|
|
|
- alert: GoChatRedisMemoryHigh
|
|
expr: redis_memory_max_bytes > 0 and redis_memory_used_bytes / redis_memory_max_bytes > 0.8
|
|
for: 5m
|
|
labels:
|
|
severity: warning
|
|
annotations:
|
|
summary: Redis memory usage above 80%
|
|
description: Redis is approaching its configured memory ceiling.
|
|
|
|
- alert: GoChatPostgresConnectionsHigh
|
|
expr: sum(pg_stat_activity_count) / max(pg_settings_max_connections) > 0.8
|
|
for: 5m
|
|
labels:
|
|
severity: warning
|
|
annotations:
|
|
summary: PostgreSQL connection usage above 80%
|
|
description: PostgreSQL is approaching its connection limit.
|
|
|
|
- alert: GoChatBackupStale
|
|
expr: |
|
|
time() - gochat_backup_last_success_timestamp_seconds > gochat_backup_rpo_target_seconds
|
|
or absent(gochat_backup_last_success_timestamp_seconds)
|
|
for: 5m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: GoChat backup is stale
|
|
description: No successful encrypted off-site backup exists inside the declared RPO.
|
|
|
|
- alert: GoChatAlertmanagerDown
|
|
expr: up{job="alertmanager"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: Alertmanager is down
|
|
description: Prometheus cannot deliver notifications to Alertmanager.
|
|
|
|
- alert: ShangwutongReadinessFailed
|
|
expr: probe_success{job="shangwutong-readiness"} == 0
|
|
for: 1m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: Shangwutong readiness failed
|
|
description: The production Connector readiness endpoint is failing.
|