- POST/PUT /api/gateways 成功后后台立即探活一次 /healthz 并覆盖落库,不阻塞请求 - probeAndRecordGateway 抽出单网关探活+落库复用;探活超时改为 var 供测试注入 - 网关页保存成功后延迟 2.5s 刷新一次,尽快展示最新健康状态 - 测试:立即探活、慢网关不阻塞、不可达落 unhealthy
84 lines
3.1 KiB
Go
84 lines
3.1 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"net/http"
|
|
"sync"
|
|
"time"
|
|
|
|
hub "git.ipao.vip/rogee/creator-hub/internal/environment"
|
|
"github.com/sirupsen/logrus"
|
|
)
|
|
|
|
// gatewayHealthProbeTimeout 是单次 /healthz 探活的超时;测试可临时替换。
|
|
var gatewayHealthProbeTimeout = 5 * time.Second
|
|
|
|
// gatewayHealthStore 是网关健康检查所需的存储能力;生产实现为 *hub.Store,测试使用内存桩。
|
|
type gatewayHealthStore interface {
|
|
ListGateways(ctx context.Context) ([]hub.Gateway, error)
|
|
RecordGatewayCheck(ctx context.Context, name, status, reason string) (hub.Gateway, error)
|
|
}
|
|
|
|
// classifyGatewayHealth 将一次 /healthz 探活结果归一为持久化状态与原因。
|
|
// 连接失败或 5xx 判为 unhealthy;其余(含 401 令牌不匹配)说明进程可达,判为 healthy
|
|
// 并把异常状态写进 reason 供排查。
|
|
func classifyGatewayHealth(status int, callErr error) (string, string) {
|
|
if callErr != nil {
|
|
reason := callErr.Error()
|
|
if len(reason) > 300 {
|
|
reason = reason[:300]
|
|
}
|
|
return "unhealthy", reason
|
|
}
|
|
if status >= 500 {
|
|
return "unhealthy", "gateway returned status " + http.StatusText(status)
|
|
}
|
|
if status >= 400 {
|
|
return "healthy", "gateway responded with status " + http.StatusText(status)
|
|
}
|
|
return "healthy", ""
|
|
}
|
|
|
|
// RecordGatewayHealthChecks 执行一轮网关健康探活并落库;供 workers 定时任务与测试调用。
|
|
func RecordGatewayHealthChecks(ctx context.Context, store *hub.Store) error {
|
|
return recordGatewayHealthChecks(ctx, store)
|
|
}
|
|
|
|
// recordGatewayHealthChecks 对全部注册网关并发探活一次,每个网关只覆盖写入最近一次结果。
|
|
func recordGatewayHealthChecks(ctx context.Context, store gatewayHealthStore) error {
|
|
gateways, err := store.ListGateways(ctx)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
var wg sync.WaitGroup
|
|
for _, gateway := range gateways {
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
probeAndRecordGateway(ctx, store, gateway)
|
|
}()
|
|
}
|
|
wg.Wait()
|
|
return nil
|
|
}
|
|
|
|
// probeAndRecordGateway 对单个网关探活一次并覆盖写入最近一次结果;落库失败不阻断其余网关,仅记日志。
|
|
func probeAndRecordGateway(ctx context.Context, store gatewayHealthStore, gateway hub.Gateway) {
|
|
status, _, callErr := gatewayCall(ctx, gateway, http.MethodGet, "/healthz", nil, gatewayHealthProbeTimeout)
|
|
healthStatus, reason := classifyGatewayHealth(status, callErr)
|
|
if _, recordErr := store.RecordGatewayCheck(ctx, gateway.Name, healthStatus, reason); recordErr != nil {
|
|
logrus.WithFields(logrus.Fields{
|
|
"service": "control-plane",
|
|
"event_type": "gateway_health_check",
|
|
"gateway": gateway.Name,
|
|
"probed": healthStatus,
|
|
}).WithError(recordErr).Warn("gateway health check result was not persisted")
|
|
}
|
|
}
|
|
|
|
// probeGatewayAsync 在后台对单个网关立即探活一次并落库,不阻塞当前请求;
|
|
// 用 WithoutCancel 脱离请求生命周期,请求结束后探活与落库仍会完成。
|
|
func probeGatewayAsync(ctx context.Context, store gatewayHealthStore, gateway hub.Gateway) {
|
|
go probeAndRecordGateway(context.WithoutCancel(ctx), store, gateway)
|
|
}
|