feat(watchdog): 主机侧巡检定时器按异常邮件通知平台所有者,operator 增加 phase 与 build_info 指标、卡死存活探针与告警规则

This commit is contained in:
Lemon-miaow committed 2026-09-24 18:06:19 +08:00
1 parent d50492b86f
commit d17524cd67
26 files changed
+2686 -34

No files matched your search

+98
View File
@@ -0,0 +1,98 @@
package metrics
import (
"os"
"reflect"
"regexp"
"strings"
"testing"
"github.com/prometheus/client_golang/prometheus"
"sigs.k8s.io/yaml"
)
// The shipped alert rules live in deploy/alerts. promtool tests what they do
// (felis-alerts_test.yml); these tests pin what promtool cannot see from there.
type ruleGroups struct {
Groups []struct {
Name string `json:"name"`
Rules []struct {
Alert string `json:"alert"`
Expr string `json:"expr"`
} `json:"rules"`
} `json:"groups"`
}
func readYAML(t *testing.T, path string, into any) {
t.Helper()
raw, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
if err := yaml.Unmarshal(raw, into); err != nil {
t.Fatalf("parse %s: %v", path, err)
}
}
// TestPrometheusRuleMatchesPlainRules: the PrometheusRule twin carries exactly
// the groups of the promtool-tested plain file.
func TestPrometheusRuleMatchesPlainRules(t *testing.T) {
var plain, twin map[string]any
readYAML(t, "../../deploy/alerts/felis-alerts.yaml", &plain)
readYAML(t, "../../deploy/alerts/felis-prometheusrule.yaml", &twin)
spec, _ := twin["spec"].(map[string]any)
if !reflect.DeepEqual(plain["groups"], spec["groups"]) {
t.Fatal("deploy/alerts/felis-prometheusrule.yaml spec.groups differs from felis-alerts.yaml groups; copy the plain file's groups over")
}
}
// TestAlertRulesUseRealMetrics: every felis_* series a rule reads is one this
// package exports (or the backup timer's textfile series), so a renamed metric
// cannot leave an alert silently watching nothing.
func TestAlertRulesUseRealMetrics(t *testing.T) {
reg := prometheus.NewRegistry()
if err := Register(reg); err != nil {
t.Fatal(err)
}
families, err := reg.Gather()
if err != nil {
t.Fatal(err)
}
known := map[string]bool{}
for _, mf := range families {
known[mf.GetName()] = true
}
// Collectors with no children yet gather nothing; name them from their
// descriptors instead.
for _, c := range Collectors() {
ch := make(chan *prometheus.Desc, 4)
c.Describe(ch)
close(ch)
for d := range ch {
if m := regexp.MustCompile(`fqName: "([^"]+)"`).FindStringSubmatch(d.String()); m != nil {
known[m[1]] = true
}
}
}
var rules ruleGroups
readYAML(t, "../../deploy/alerts/felis-alerts.yaml", &rules)
series := regexp.MustCompile(`felis_[a-z_]+`)
for _, g := range rules.Groups {
for _, r := range g.Rules {
for _, name := range series.FindAllString(r.Expr, -1) {
if strings.HasPrefix(name, "felis_db_backup_") {
continue // written by felis-db-backup.timer for node-exporter
}
base := name
for _, suffix := range []string{"_bucket", "_count", "_sum"} {
base = strings.TrimSuffix(base, suffix)
}
if !known[name] && !known[base] {
t.Errorf("alert %s reads %s, which no felis collector exports", r.Alert, name)
}
}
}
}
}
+45
View File
@@ -32,6 +32,18 @@ var (
Help: "Current number of Minecraft servers known to the operator, by desired state.",
}, []string{"state"})
// ServerPhase is 1 for each server's current phase and absent for every other
// phase. role is the server's system role (login, lobby), empty for a user
// server, so an alert can single out the login gate; desired is its
// desiredState, so a server that is down on purpose can be told from one that
// failed to come up. SyncServerPhases republishes it from a full List, so a
// deleted server's series goes away instead of freezing at its last phase.
ServerPhase = prometheus.NewGaugeVec(prometheus.GaugeOpts{
Namespace: namespace,
Name: "server_phase",
Help: "1 for each Minecraft server's current phase, by server, system role, phase and desired state.",
}, []string{"server", "role", "phase", "desired"})
// StartDurationSeconds observes the wall-clock time from desiredState=Running
// to a server reporting ready. Buckets are tuned for Minecraft cold starts
// (seconds to a few minutes), not the default sub-second web-latency buckets.
@@ -107,8 +119,25 @@ var (
Name: "audit_write_failures_total",
Help: "Audit rows the API failed to write.",
})
// BuildInfo is 1 for the process serving it, labelled by component
// ("operator", "api") and version. Both processes register every collector,
// so this is the one series that says which of them a scrape reached: an
// alert on absent(felis_build_info{component="api"}) fires when felis-api is
// down or no longer scraped, where every other felis_* series would still be
// present from the operator.
BuildInfo = prometheus.NewGaugeVec(prometheus.GaugeOpts{
Namespace: namespace,
Name: "build_info",
Help: "1 for the Felis component serving these metrics, by component and version.",
}, []string{"component", "version"})
)
// SetBuildInfo marks this process as component at version on felis_build_info.
func SetBuildInfo(component, version string) {
BuildInfo.WithLabelValues(component, version).Set(1)
}
// OTPPurposes are the email-code doors OTPLockoutsTotal is labelled by.
var OTPPurposes = []string{"onboard_email", "login_email", "op_login", "migrate_confirm"}
@@ -149,12 +178,27 @@ func SyncServerGauge(states []string) {
}
}
// ServerPhaseSample is one server's felis_server_phase series.
type ServerPhaseSample struct {
Server, Role, Phase, Desired string
}
// SyncServerPhases republishes felis_server_phase from a full snapshot of the
// fleet, Resetting first for the same reason SyncServerGauge does.
func SyncServerPhases(samples []ServerPhaseSample) {
ServerPhase.Reset()
for _, s := range samples {
ServerPhase.WithLabelValues(s.Server, s.Role, s.Phase, s.Desired).Set(1)
}
}
// Collectors returns every felis_* collector in a stable order. Production and
// tests register the same slice, so the test asserting the full set is exposed
// also pins the production surface.
func Collectors() []prometheus.Collector {
return []prometheus.Collector{
ServersTotal,
ServerPhase,
StartDurationSeconds,
ImageBuildFailuresTotal,
ReaperWorldsDeletedTotal,
@@ -164,6 +208,7 @@ func Collectors() []prometheus.Collector {
AuthFailuresTotal,
SessionsRevokedTotal,
AuditWriteFailuresTotal,
BuildInfo,
}
}
+24
View File
@@ -13,9 +13,11 @@ import (
// silent dashboard/alert break.
var wantNames = []string{
"felis_servers_total",
"felis_server_phase",
"felis_start_duration_seconds",
"felis_image_build_failures_total",
"felis_reaper_worlds_deleted_total",
"felis_build_info",
}
func TestRegisterExposesNamedFelisMetrics(t *testing.T) {
@@ -28,6 +30,8 @@ func TestRegisterExposesNamedFelisMetrics(t *testing.T) {
// family — this proves the exported vars are the ones actually registered,
// not shadow copies.
ServersTotal.WithLabelValues("Running").Set(3)
ServerPhase.WithLabelValues("survival", "", "Running", "Running").Set(1)
SetBuildInfo("operator", "v1.2.3")
StartDurationSeconds.Observe(12.5)
ImageBuildFailuresTotal.Inc()
ReaperWorldsDeletedTotal.Add(2)
@@ -54,6 +58,26 @@ func TestRegisterExposesNamedFelisMetrics(t *testing.T) {
}
}
// TestSyncServerPhasesDropsDeletedServers: each sync publishes exactly the
// servers in the snapshot, so a server that left the fleet or changed phase keeps
// no stale series behind.
func TestSyncServerPhasesDropsDeletedServers(t *testing.T) {
SyncServerPhases([]ServerPhaseSample{
{Server: "login", Role: "login", Phase: "Starting", Desired: "Running"},
{Server: "survival", Phase: "Running", Desired: "Running"},
})
SyncServerPhases([]ServerPhaseSample{
{Server: "login", Role: "login", Phase: "Running", Desired: "Running"},
})
if n := testutil.CollectAndCount(ServerPhase); n != 1 {
t.Fatalf("series after the second sync = %d, want 1", n)
}
if got := testutil.ToFloat64(ServerPhase.WithLabelValues("login", "login", "Running", "Running")); got != 1 {
t.Errorf("login Running = %v, want 1", got)
}
ServerPhase.Reset()
}
func TestSyncServerGaugeResetsStaleStates(t *testing.T) {
// Self-contained (Reset->sync->assert within this one test) so it never
// clobbers another test's ServersTotal children. The correctness point is the