HH-445: deploy production observability and runbooks (#96)

* HH-445: deploy production observability and runbooks

* fix(ops): share production database DSN

* fix(HH-445): enforce database TLS gate

* fix(HH-445): preserve production serve command

* fix(prod): require external database dependencies

* fix(prod): unify database host rejection gates

* test(prod): enforce exact database TLS runbook contract

---------

Co-authored-by: Rogee <rogee@ipao.vip>
This commit is contained in:
Rogee
2026-08-22 19:39:57 +08:00
committed by GitHub
co-authored by rogee
parent 61376a57fd
commit fb83285617
28 changed files with 1622 additions and 220 deletions
+17
View File
@@ -0,0 +1,17 @@
route:
receiver: operations-webhook
group_by: [alertname, job]
group_wait: 30s
group_interval: 5m
repeat_interval: 4h
routes:
- receiver: operations-webhook
matchers:
- severity="critical"
repeat_interval: 30m
receivers:
- name: operations-webhook
webhook_configs:
- url_file: /run/secrets/alertmanager-webhook-url
send_resolved: true
+7
View File
@@ -0,0 +1,7 @@
modules:
http_2xx:
prober: http
timeout: 5s
http:
preferred_ip_protocol: ip4
valid_status_codes: [200]
+26
View File
@@ -0,0 +1,26 @@
gochat_background_jobs:
query: |
SELECT
queue,
status,
COUNT(*)::double precision AS total,
COALESCE(MAX(EXTRACT(EPOCH FROM (CURRENT_TIMESTAMP - CASE
WHEN status = 'running' THEN COALESCE(locked_at, updated_at)
ELSE scheduled_at
END))), 0)::double precision AS oldest_seconds
FROM background_jobs
WHERE status IN ('queued', 'retrying', 'running')
GROUP BY queue, status
metrics:
- queue:
usage: LABEL
description: Background job queue.
- status:
usage: LABEL
description: Background job state.
- total:
usage: GAUGE
description: Current background jobs by queue and state.
- oldest_seconds:
usage: GAUGE
description: Age of the oldest background job by queue and state.
+79
View File
@@ -0,0 +1,79 @@
global:
scrape_interval: 15s
evaluation_interval: 15s
rule_files:
- /etc/prometheus/rules/*.yml
alerting:
alertmanagers:
- static_configs:
- targets: [alertmanager:9093]
scrape_configs:
- job_name: prometheus
static_configs:
- targets: [prometheus:9090]
- job_name: alertmanager
static_configs:
- targets: [alertmanager:9093]
- job_name: gochat
metrics_path: /metrics
static_configs:
- targets: [gochat:3000]
- job_name: postgres-exporter
static_configs:
- targets: [postgres-exporter:9187]
- job_name: redis-exporter
static_configs:
- targets: [redis-exporter:9121]
- job_name: cadvisor
static_configs:
- targets: [cadvisor:8080]
- job_name: node-exporter
static_configs:
- targets: [node-exporter:9100]
- job_name: gochat-readiness
metrics_path: /probe
params:
module: [http_2xx]
static_configs:
- targets: ["http://gochat:3000/ready"]
relabel_configs: &blackbox-relabel
- source_labels: [__address__]
target_label: __param_target
- source_labels: [__param_target]
target_label: instance
- target_label: __address__
replacement: blackbox-exporter:9115
- job_name: gochat-database-readiness
metrics_path: /probe
params:
module: [http_2xx]
static_configs:
- targets: ["http://gochat:3000/health?check=database"]
relabel_configs: *blackbox-relabel
- job_name: gochat-redis-readiness
metrics_path: /probe
params:
module: [http_2xx]
static_configs:
- targets: ["http://gochat:3000/health?check=redis"]
relabel_configs: *blackbox-relabel
- job_name: shangwutong-readiness
metrics_path: /probe
params:
module: [http_2xx]
static_configs:
- targets: ["http://shangwutong:9100/readyz"]
relabel_configs: *blackbox-relabel