Files
gochat/backend/configs/prometheus_alerts.yml
T
Rogeeandrogee fb83285617 HH-445: deploy production observability and runbooks (#96)
* HH-445: deploy production observability and runbooks

* fix(ops): share production database DSN

* fix(HH-445): enforce database TLS gate

* fix(HH-445): preserve production serve command

* fix(prod): require external database dependencies

* fix(prod): unify database host rejection gates

* test(prod): enforce exact database TLS runbook contract

---------

Co-authored-by: Rogee <rogee@ipao.vip>
2026-08-22 19:39:57 +08:00

160 lines
5.7 KiB
YAML

groups:
- name: gochat-application
rules:
- alert: GoChatAppDown
expr: up{job="gochat"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: GoChat application is down
description: GoChat metrics have been unreachable for more than one minute.
- alert: GoChatDatabaseReadinessFailed
expr: probe_success{job="gochat-database-readiness"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: GoChat database readiness failed
description: The application cannot complete its PostgreSQL dependency check.
- alert: GoChatRedisReadinessFailed
expr: probe_success{job="gochat-redis-readiness"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: GoChat Redis readiness failed
description: The application cannot complete its Redis dependency check.
- alert: GoChatHighErrorRate
expr: |
sum by (job, instance) (rate(http_request_errors_total{job="gochat"}[5m]))
/ sum by (job, instance) (rate(http_requests_total{job="gochat"}[5m])) > 0.05
and sum by (job, instance) (rate(http_requests_total{job="gochat"}[5m])) > 0
for: 5m
labels:
severity: warning
annotations:
summary: GoChat error rate above 5%
description: Error rate is {{ $value | humanizePercentage }} over the last 5 minutes.
- alert: GoChatHighMemory
expr: gochat_go_memory_alloc_bytes / 1024 / 1024 > 400
for: 5m
labels:
severity: warning
annotations:
summary: GoChat memory usage above 400 MB
description: Allocated heap is {{ $value | humanize }} MB.
- alert: GoChatHighGoroutines
expr: gochat_go_goroutines > 1000
for: 5m
labels:
severity: warning
annotations:
summary: GoChat goroutine count above 1000
description: GoChat has {{ $value | humanize }} goroutines.
- name: gochat-workers
rules:
- alert: GoChatWorkerDown
expr: |
time() - max(container_last_seen{container_label_com_docker_compose_service="worker"}) > 60
or absent(container_last_seen{container_label_com_docker_compose_service="worker"})
for: 1m
labels:
severity: critical
annotations:
summary: GoChat worker is down
description: cAdvisor has not observed a production worker container for more than one minute.
- alert: GoChatCriticalQueueBacklog
expr: |
sum by (queue) (gochat_background_jobs_total{queue=~"critical|high",status=~"queued|retrying"}) > 25
or max by (queue) (gochat_background_jobs_oldest_seconds{queue=~"critical|high",status=~"queued|retrying"}) > 300
for: 5m
labels:
severity: critical
annotations:
summary: Critical background queue is delayed
description: Queue {{ $labels.queue }} exceeds 25 ready jobs or its oldest job is over five minutes old.
- alert: GoChatWorkerJobsStuck
expr: max by (queue) (gochat_background_jobs_oldest_seconds{status="running"}) > 900
for: 5m
labels:
severity: critical
annotations:
summary: Background jobs are stuck
description: Queue {{ $labels.queue }} has a running job older than 15 minutes.
- name: gochat-infrastructure
rules:
- alert: GoChatPostgresExporterDown
expr: up{job="postgres-exporter"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: PostgreSQL exporter is down
description: Prometheus cannot scrape the PostgreSQL exporter.
- alert: GoChatRedisExporterDown
expr: up{job="redis-exporter"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: Redis exporter is down
description: Prometheus cannot scrape the Redis exporter.
- alert: GoChatRedisMemoryHigh
expr: redis_memory_max_bytes > 0 and redis_memory_used_bytes / redis_memory_max_bytes > 0.8
for: 5m
labels:
severity: warning
annotations:
summary: Redis memory usage above 80%
description: Redis is approaching its configured memory ceiling.
- alert: GoChatPostgresConnectionsHigh
expr: sum(pg_stat_activity_count) / max(pg_settings_max_connections) > 0.8
for: 5m
labels:
severity: warning
annotations:
summary: PostgreSQL connection usage above 80%
description: PostgreSQL is approaching its connection limit.
- alert: GoChatBackupStale
expr: |
time() - gochat_backup_last_success_timestamp_seconds > gochat_backup_rpo_target_seconds
or absent(gochat_backup_last_success_timestamp_seconds)
for: 5m
labels:
severity: critical
annotations:
summary: GoChat backup is stale
description: No successful encrypted off-site backup exists inside the declared RPO.
- alert: GoChatAlertmanagerDown
expr: up{job="alertmanager"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: Alertmanager is down
description: Prometheus cannot deliver notifications to Alertmanager.
- alert: ShangwutongReadinessFailed
expr: probe_success{job="shangwutong-readiness"} == 0
for: 1m
labels:
severity: critical
annotations:
summary: Shangwutong readiness failed
description: The production Connector readiness endpoint is failing.