feat(watchdog): 主机侧巡检定时器按异常邮件通知平台所有者,operator 增加 phase 与 build_info 指标、卡死存活探针与告警规则
This commit is contained in:
26 files changed
+2686
-34
No files matched your search
@@ -13,8 +13,8 @@ import (
|
||||
// defaultGaugeInterval is the republish cadence used when none is configured.
|
||||
const defaultGaugeInterval = 30 * time.Second
|
||||
|
||||
// GaugeSyncer periodically republishes felis_servers_total (spec §23) from a
|
||||
// full List of MinecraftServers.
|
||||
// GaugeSyncer periodically republishes felis_servers_total (spec §23) and
|
||||
// felis_server_phase from a full List of MinecraftServers.
|
||||
//
|
||||
// felis_servers_total is a fleet-wide gauge partitioned by desiredState, which a
|
||||
// per-object Reconcile fundamentally cannot maintain: one reconcile observes a
|
||||
@@ -62,7 +62,8 @@ func (g *GaugeSyncer) Start(ctx context.Context) error {
|
||||
}
|
||||
}
|
||||
|
||||
// SyncOnce Lists the fleet once and republishes felis_servers_total from it.
|
||||
// SyncOnce Lists the fleet once and republishes felis_servers_total and
|
||||
// felis_server_phase from it.
|
||||
// Separated from Start so the List->translate->gauge path is unit-testable
|
||||
// against a fake client, leaving only the ticker loop untested.
|
||||
func (g *GaugeSyncer) SyncOnce(ctx context.Context) error {
|
||||
@@ -71,9 +72,35 @@ func (g *GaugeSyncer) SyncOnce(ctx context.Context) error {
|
||||
return err
|
||||
}
|
||||
metrics.SyncServerGauge(serverStates(list.Items))
|
||||
metrics.SyncServerPhases(serverPhases(list.Items))
|
||||
return nil
|
||||
}
|
||||
|
||||
// serverPhases maps each server to its felis_server_phase labels. A server the
|
||||
// operator has not reported on yet is Unknown; desiredState defaults as in
|
||||
// serverStates.
|
||||
func serverPhases(items []v1alpha1.MinecraftServer) []metrics.ServerPhaseSample {
|
||||
out := make([]metrics.ServerPhaseSample, 0, len(items))
|
||||
for i := range items {
|
||||
ms := &items[i]
|
||||
phase := string(ms.Status.Phase)
|
||||
if phase == "" {
|
||||
phase = string(v1alpha1.PhaseUnknown)
|
||||
}
|
||||
desired := string(ms.Spec.DesiredState)
|
||||
if desired == "" {
|
||||
desired = string(v1alpha1.DesiredStopped)
|
||||
}
|
||||
out = append(out, metrics.ServerPhaseSample{
|
||||
Server: ms.Name,
|
||||
Role: ms.Labels[v1alpha1.LabelSystemRole],
|
||||
Phase: phase,
|
||||
Desired: desired,
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// serverStates maps each server to its desiredState, defaulting an unset state
|
||||
// to Stopped (the CRD default, spec §4). Pure, so the labelling rule is covered
|
||||
// by SyncOnce's fake-client test rather than only at runtime.
|
||||
|
||||
@@ -48,3 +48,30 @@ func TestGaugeSyncerPublishesFleetStateFromCluster(t *testing.T) {
|
||||
t.Errorf("servers_total{state=Stopped} = %v, want 2", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGaugeSyncerPublishesServerPhases: every server gets one felis_server_phase
|
||||
// series carrying its system role, observed phase (Unknown before the first
|
||||
// report) and desired state.
|
||||
func TestGaugeSyncerPublishesServerPhases(t *testing.T) {
|
||||
login := gaugeServer("login", v1alpha1.DesiredRunning)
|
||||
login.Labels = map[string]string{v1alpha1.LabelSystemRole: "login"}
|
||||
login.Status.Phase = v1alpha1.PhaseFailed
|
||||
fresh := gaugeServer("fresh", "")
|
||||
c := fake.NewClientBuilder().WithScheme(newScheme(t)).WithObjects(login, fresh).Build()
|
||||
|
||||
if err := (&operator.GaugeSyncer{Client: c}).SyncOnce(context.Background()); err != nil {
|
||||
t.Fatalf("SyncOnce: %v", err)
|
||||
}
|
||||
defer metrics.ServerPhase.Reset()
|
||||
if n := testutil.CollectAndCount(metrics.ServerPhase); n != 2 {
|
||||
t.Fatalf("felis_server_phase series = %d, want 2", n)
|
||||
}
|
||||
for _, labels := range [][]string{
|
||||
{"login", "login", "Failed", "Running"},
|
||||
{"fresh", "", "Unknown", "Stopped"},
|
||||
} {
|
||||
if got := testutil.ToFloat64(metrics.ServerPhase.WithLabelValues(labels...)); got != 1 {
|
||||
t.Errorf("felis_server_phase%v = %v, want 1", labels, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -78,6 +78,8 @@ type Reconciler struct {
|
||||
// (internal/maintenance). It is the manager's uncached API reader, so the
|
||||
// operator needs jobs:list and no Job informer. Nil skips the check.
|
||||
Jobs client.Reader
|
||||
// Watch records the passes in flight for the liveness probe. Nil skips it.
|
||||
Watch *ReconcileWatch
|
||||
}
|
||||
|
||||
func (r *Reconciler) now() metav1.Time {
|
||||
@@ -109,8 +111,18 @@ func (r *Reconciler) SetupWithManager(mgr ctrl.Manager) error {
|
||||
// The same server is never reconciled twice at once regardless.
|
||||
const maxConcurrentReconciles = 4
|
||||
|
||||
// Reconcile drives a single MinecraftServer toward spec.desiredState.
|
||||
// Reconcile drives a single MinecraftServer toward spec.desiredState. One pass
|
||||
// is bounded by reconcileTimeout and reported to r.Watch while it runs.
|
||||
func (r *Reconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, reconcileTimeout)
|
||||
defer cancel()
|
||||
if r.Watch != nil {
|
||||
defer r.Watch.begin(req.Name)()
|
||||
}
|
||||
return r.reconcile(ctx, req)
|
||||
}
|
||||
|
||||
func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) {
|
||||
var server v1alpha1.MinecraftServer
|
||||
if err := r.Get(ctx, req.NamespacedName, &server); err != nil {
|
||||
// Deletion is handled by owner references on the children.
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
package operator
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// reconcileTimeout bounds one reconcile pass. The slowest legitimate pass waits
|
||||
// on RCON: 5s for a probe, defaultSaveTimeout for the world save ahead of a stop.
|
||||
// A pass still running past this is waiting on something that will not answer;
|
||||
// cancelling its context makes the calls that honour it return, and the server
|
||||
// is retried with backoff.
|
||||
const reconcileTimeout = 3 * time.Minute
|
||||
|
||||
// stuckReconcileAfter is how long one pass may run before the operator reports
|
||||
// itself unhealthy. It is well past reconcileTimeout, so only a pass blocked in a
|
||||
// call that ignores its context gets here. Each such pass holds one of the
|
||||
// maxConcurrentReconciles workers for good; a few of them stall every server's
|
||||
// start and stop while the Pod still looks healthy. Failing the liveness probe
|
||||
// gets the operator restarted, the one fix for a goroutine that never returns.
|
||||
const stuckReconcileAfter = 10 * time.Minute
|
||||
|
||||
// ReconcileWatch tracks the reconcile passes in flight so the liveness probe can
|
||||
// tell a busy operator from a wedged one. An idle operator has nothing in flight
|
||||
// and is healthy: a quiet fleet reconciles rarely, so the time since the last
|
||||
// pass says nothing about whether the next one would run.
|
||||
type ReconcileWatch struct {
|
||||
// StuckAfter defaults to stuckReconcileAfter.
|
||||
StuckAfter time.Duration
|
||||
// Now defaults to time.Now.
|
||||
Now func() time.Time
|
||||
|
||||
mu sync.Mutex
|
||||
next uint64
|
||||
inflight map[uint64]inflightPass
|
||||
}
|
||||
|
||||
type inflightPass struct {
|
||||
server string
|
||||
started time.Time
|
||||
}
|
||||
|
||||
func (w *ReconcileWatch) now() time.Time {
|
||||
if w.Now != nil {
|
||||
return w.Now()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
// begin records a pass for server and returns the func that ends it.
|
||||
func (w *ReconcileWatch) begin(server string) func() {
|
||||
w.mu.Lock()
|
||||
defer w.mu.Unlock()
|
||||
if w.inflight == nil {
|
||||
w.inflight = map[uint64]inflightPass{}
|
||||
}
|
||||
w.next++
|
||||
id := w.next
|
||||
w.inflight[id] = inflightPass{server: server, started: w.now()}
|
||||
return func() {
|
||||
w.mu.Lock()
|
||||
delete(w.inflight, id)
|
||||
w.mu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// Check is a healthz.Checker: it fails while any pass has been running longer
|
||||
// than StuckAfter, naming the server and how long.
|
||||
func (w *ReconcileWatch) Check(_ *http.Request) error {
|
||||
limit := w.StuckAfter
|
||||
if limit <= 0 {
|
||||
limit = stuckReconcileAfter
|
||||
}
|
||||
now := w.now()
|
||||
w.mu.Lock()
|
||||
defer w.mu.Unlock()
|
||||
var oldest *inflightPass
|
||||
for id := range w.inflight {
|
||||
p := w.inflight[id]
|
||||
if oldest == nil || p.started.Before(oldest.started) {
|
||||
oldest = &p
|
||||
}
|
||||
}
|
||||
if oldest != nil && now.Sub(oldest.started) > limit {
|
||||
return fmt.Errorf("reconcile of %s has been running for %s (limit %s)",
|
||||
oldest.server, now.Sub(oldest.started).Truncate(time.Second), limit)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
package operator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
|
||||
)
|
||||
|
||||
// TestReconcileWatchCheck: an idle or busy operator is healthy; one pass running
|
||||
// past the limit fails the check and names the server; the check recovers once
|
||||
// that pass returns.
|
||||
func TestReconcileWatchCheck(t *testing.T) {
|
||||
now := time.Unix(1_700_000_000, 0)
|
||||
w := &ReconcileWatch{StuckAfter: 10 * time.Minute, Now: func() time.Time { return now }}
|
||||
if err := w.Check(nil); err != nil {
|
||||
t.Fatalf("idle: %v", err)
|
||||
}
|
||||
|
||||
endOld := w.begin("survival")
|
||||
now = now.Add(9 * time.Minute)
|
||||
endNew := w.begin("creative")
|
||||
if err := w.Check(nil); err != nil {
|
||||
t.Fatalf("9m in: %v", err)
|
||||
}
|
||||
|
||||
now = now.Add(2 * time.Minute)
|
||||
err := w.Check(nil)
|
||||
if err == nil || !strings.Contains(err.Error(), "survival") || !strings.Contains(err.Error(), "11m0s") {
|
||||
t.Fatalf("11m in: err = %v, want survival stuck for 11m0s", err)
|
||||
}
|
||||
|
||||
endOld()
|
||||
if err := w.Check(nil); err != nil {
|
||||
t.Fatalf("after the stuck pass returned: %v", err)
|
||||
}
|
||||
endNew()
|
||||
}
|
||||
|
||||
// TestReconcileBoundedAndWatched: Reconcile hands the pass a deadline and holds
|
||||
// an in-flight entry exactly while it runs.
|
||||
func TestReconcileBoundedAndWatched(t *testing.T) {
|
||||
scheme := runtime.NewScheme()
|
||||
if err := v1alpha1.AddToScheme(scheme); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
watch := &ReconcileWatch{}
|
||||
var deadline time.Duration
|
||||
var inflight int
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithInterceptorFuncs(interceptor.Funcs{
|
||||
Get: func(ctx context.Context, c client.WithWatch, key client.ObjectKey, obj client.Object, opts ...client.GetOption) error {
|
||||
if d, ok := ctx.Deadline(); ok && deadline == 0 {
|
||||
deadline = time.Until(d)
|
||||
}
|
||||
watch.mu.Lock()
|
||||
inflight = len(watch.inflight)
|
||||
watch.mu.Unlock()
|
||||
return c.Get(ctx, key, obj, opts...)
|
||||
},
|
||||
}).Build()
|
||||
r := &Reconciler{Client: cl, Scheme: scheme, Watch: watch}
|
||||
|
||||
if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: types.NamespacedName{Namespace: "minecraft", Name: "missing"}}); err != nil {
|
||||
t.Fatalf("Reconcile: %v", err)
|
||||
}
|
||||
if deadline <= 0 || deadline > reconcileTimeout {
|
||||
t.Errorf("pass deadline = %s, want within %s", deadline, reconcileTimeout)
|
||||
}
|
||||
if inflight != 1 {
|
||||
t.Errorf("in-flight passes during the pass = %d, want 1", inflight)
|
||||
}
|
||||
if n := len(watch.inflight); n != 0 {
|
||||
t.Errorf("in-flight passes after the pass = %d, want 0", n)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user