feat(watchdog): 主机侧巡检定时器按异常邮件通知平台所有者,operator 增加 phase 与 build_info 指标、卡死存活探针与告警规则
This commit is contained in:
26 files changed
+2686
-34
No files matched your search
@@ -0,0 +1,98 @@
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"os"
|
||||
"reflect"
|
||||
"regexp"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
// The shipped alert rules live in deploy/alerts. promtool tests what they do
|
||||
// (felis-alerts_test.yml); these tests pin what promtool cannot see from there.
|
||||
|
||||
type ruleGroups struct {
|
||||
Groups []struct {
|
||||
Name string `json:"name"`
|
||||
Rules []struct {
|
||||
Alert string `json:"alert"`
|
||||
Expr string `json:"expr"`
|
||||
} `json:"rules"`
|
||||
} `json:"groups"`
|
||||
}
|
||||
|
||||
func readYAML(t *testing.T, path string, into any) {
|
||||
t.Helper()
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := yaml.Unmarshal(raw, into); err != nil {
|
||||
t.Fatalf("parse %s: %v", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPrometheusRuleMatchesPlainRules: the PrometheusRule twin carries exactly
|
||||
// the groups of the promtool-tested plain file.
|
||||
func TestPrometheusRuleMatchesPlainRules(t *testing.T) {
|
||||
var plain, twin map[string]any
|
||||
readYAML(t, "../../deploy/alerts/felis-alerts.yaml", &plain)
|
||||
readYAML(t, "../../deploy/alerts/felis-prometheusrule.yaml", &twin)
|
||||
spec, _ := twin["spec"].(map[string]any)
|
||||
if !reflect.DeepEqual(plain["groups"], spec["groups"]) {
|
||||
t.Fatal("deploy/alerts/felis-prometheusrule.yaml spec.groups differs from felis-alerts.yaml groups; copy the plain file's groups over")
|
||||
}
|
||||
}
|
||||
|
||||
// TestAlertRulesUseRealMetrics: every felis_* series a rule reads is one this
|
||||
// package exports (or the backup timer's textfile series), so a renamed metric
|
||||
// cannot leave an alert silently watching nothing.
|
||||
func TestAlertRulesUseRealMetrics(t *testing.T) {
|
||||
reg := prometheus.NewRegistry()
|
||||
if err := Register(reg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
families, err := reg.Gather()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
known := map[string]bool{}
|
||||
for _, mf := range families {
|
||||
known[mf.GetName()] = true
|
||||
}
|
||||
// Collectors with no children yet gather nothing; name them from their
|
||||
// descriptors instead.
|
||||
for _, c := range Collectors() {
|
||||
ch := make(chan *prometheus.Desc, 4)
|
||||
c.Describe(ch)
|
||||
close(ch)
|
||||
for d := range ch {
|
||||
if m := regexp.MustCompile(`fqName: "([^"]+)"`).FindStringSubmatch(d.String()); m != nil {
|
||||
known[m[1]] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var rules ruleGroups
|
||||
readYAML(t, "../../deploy/alerts/felis-alerts.yaml", &rules)
|
||||
series := regexp.MustCompile(`felis_[a-z_]+`)
|
||||
for _, g := range rules.Groups {
|
||||
for _, r := range g.Rules {
|
||||
for _, name := range series.FindAllString(r.Expr, -1) {
|
||||
if strings.HasPrefix(name, "felis_db_backup_") {
|
||||
continue // written by felis-db-backup.timer for node-exporter
|
||||
}
|
||||
base := name
|
||||
for _, suffix := range []string{"_bucket", "_count", "_sum"} {
|
||||
base = strings.TrimSuffix(base, suffix)
|
||||
}
|
||||
if !known[name] && !known[base] {
|
||||
t.Errorf("alert %s reads %s, which no felis collector exports", r.Alert, name)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -32,6 +32,18 @@ var (
|
||||
Help: "Current number of Minecraft servers known to the operator, by desired state.",
|
||||
}, []string{"state"})
|
||||
|
||||
// ServerPhase is 1 for each server's current phase and absent for every other
|
||||
// phase. role is the server's system role (login, lobby), empty for a user
|
||||
// server, so an alert can single out the login gate; desired is its
|
||||
// desiredState, so a server that is down on purpose can be told from one that
|
||||
// failed to come up. SyncServerPhases republishes it from a full List, so a
|
||||
// deleted server's series goes away instead of freezing at its last phase.
|
||||
ServerPhase = prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Namespace: namespace,
|
||||
Name: "server_phase",
|
||||
Help: "1 for each Minecraft server's current phase, by server, system role, phase and desired state.",
|
||||
}, []string{"server", "role", "phase", "desired"})
|
||||
|
||||
// StartDurationSeconds observes the wall-clock time from desiredState=Running
|
||||
// to a server reporting ready. Buckets are tuned for Minecraft cold starts
|
||||
// (seconds to a few minutes), not the default sub-second web-latency buckets.
|
||||
@@ -107,8 +119,25 @@ var (
|
||||
Name: "audit_write_failures_total",
|
||||
Help: "Audit rows the API failed to write.",
|
||||
})
|
||||
|
||||
// BuildInfo is 1 for the process serving it, labelled by component
|
||||
// ("operator", "api") and version. Both processes register every collector,
|
||||
// so this is the one series that says which of them a scrape reached: an
|
||||
// alert on absent(felis_build_info{component="api"}) fires when felis-api is
|
||||
// down or no longer scraped, where every other felis_* series would still be
|
||||
// present from the operator.
|
||||
BuildInfo = prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Namespace: namespace,
|
||||
Name: "build_info",
|
||||
Help: "1 for the Felis component serving these metrics, by component and version.",
|
||||
}, []string{"component", "version"})
|
||||
)
|
||||
|
||||
// SetBuildInfo marks this process as component at version on felis_build_info.
|
||||
func SetBuildInfo(component, version string) {
|
||||
BuildInfo.WithLabelValues(component, version).Set(1)
|
||||
}
|
||||
|
||||
// OTPPurposes are the email-code doors OTPLockoutsTotal is labelled by.
|
||||
var OTPPurposes = []string{"onboard_email", "login_email", "op_login", "migrate_confirm"}
|
||||
|
||||
@@ -149,12 +178,27 @@ func SyncServerGauge(states []string) {
|
||||
}
|
||||
}
|
||||
|
||||
// ServerPhaseSample is one server's felis_server_phase series.
|
||||
type ServerPhaseSample struct {
|
||||
Server, Role, Phase, Desired string
|
||||
}
|
||||
|
||||
// SyncServerPhases republishes felis_server_phase from a full snapshot of the
|
||||
// fleet, Resetting first for the same reason SyncServerGauge does.
|
||||
func SyncServerPhases(samples []ServerPhaseSample) {
|
||||
ServerPhase.Reset()
|
||||
for _, s := range samples {
|
||||
ServerPhase.WithLabelValues(s.Server, s.Role, s.Phase, s.Desired).Set(1)
|
||||
}
|
||||
}
|
||||
|
||||
// Collectors returns every felis_* collector in a stable order. Production and
|
||||
// tests register the same slice, so the test asserting the full set is exposed
|
||||
// also pins the production surface.
|
||||
func Collectors() []prometheus.Collector {
|
||||
return []prometheus.Collector{
|
||||
ServersTotal,
|
||||
ServerPhase,
|
||||
StartDurationSeconds,
|
||||
ImageBuildFailuresTotal,
|
||||
ReaperWorldsDeletedTotal,
|
||||
@@ -164,6 +208,7 @@ func Collectors() []prometheus.Collector {
|
||||
AuthFailuresTotal,
|
||||
SessionsRevokedTotal,
|
||||
AuditWriteFailuresTotal,
|
||||
BuildInfo,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -13,9 +13,11 @@ import (
|
||||
// silent dashboard/alert break.
|
||||
var wantNames = []string{
|
||||
"felis_servers_total",
|
||||
"felis_server_phase",
|
||||
"felis_start_duration_seconds",
|
||||
"felis_image_build_failures_total",
|
||||
"felis_reaper_worlds_deleted_total",
|
||||
"felis_build_info",
|
||||
}
|
||||
|
||||
func TestRegisterExposesNamedFelisMetrics(t *testing.T) {
|
||||
@@ -28,6 +30,8 @@ func TestRegisterExposesNamedFelisMetrics(t *testing.T) {
|
||||
// family — this proves the exported vars are the ones actually registered,
|
||||
// not shadow copies.
|
||||
ServersTotal.WithLabelValues("Running").Set(3)
|
||||
ServerPhase.WithLabelValues("survival", "", "Running", "Running").Set(1)
|
||||
SetBuildInfo("operator", "v1.2.3")
|
||||
StartDurationSeconds.Observe(12.5)
|
||||
ImageBuildFailuresTotal.Inc()
|
||||
ReaperWorldsDeletedTotal.Add(2)
|
||||
@@ -54,6 +58,26 @@ func TestRegisterExposesNamedFelisMetrics(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestSyncServerPhasesDropsDeletedServers: each sync publishes exactly the
|
||||
// servers in the snapshot, so a server that left the fleet or changed phase keeps
|
||||
// no stale series behind.
|
||||
func TestSyncServerPhasesDropsDeletedServers(t *testing.T) {
|
||||
SyncServerPhases([]ServerPhaseSample{
|
||||
{Server: "login", Role: "login", Phase: "Starting", Desired: "Running"},
|
||||
{Server: "survival", Phase: "Running", Desired: "Running"},
|
||||
})
|
||||
SyncServerPhases([]ServerPhaseSample{
|
||||
{Server: "login", Role: "login", Phase: "Running", Desired: "Running"},
|
||||
})
|
||||
if n := testutil.CollectAndCount(ServerPhase); n != 1 {
|
||||
t.Fatalf("series after the second sync = %d, want 1", n)
|
||||
}
|
||||
if got := testutil.ToFloat64(ServerPhase.WithLabelValues("login", "login", "Running", "Running")); got != 1 {
|
||||
t.Errorf("login Running = %v, want 1", got)
|
||||
}
|
||||
ServerPhase.Reset()
|
||||
}
|
||||
|
||||
func TestSyncServerGaugeResetsStaleStates(t *testing.T) {
|
||||
// Self-contained (Reset->sync->assert within this one test) so it never
|
||||
// clobbers another test's ServersTotal children. The correctness point is the
|
||||
|
||||
Reference in new issue
Block a user