fix(operator): 回到 Starting 时清掉 readySignalAt,每次启动都记入启动时长指标

This commit is contained in:
Lemon-miaow committed 2026-09-27 13:32:25 +08:00
1 parent 67a7e27f8d
commit c272e2ea24
4 files changed
+52 -2

No files matched your search

@@ -328,7 +328,8 @@ type MinecraftServerStatus struct {
Players PlayersStatus `json:"players,omitempty"`
// LiveMotd is the MOTD currently advertised for the active phase.
LiveMotd string `json:"liveMotd,omitempty"`
// ReadySignalAt is when the first RCON probe succeeded.
// ReadySignalAt is when the first RCON probe of the current run succeeded.
// Every Starting pass clears it, so each start is measured once.
ReadySignalAt *metav1.Time `json:"readySignalAt,omitempty"`
// StartRequestedAt is when the current start attempt was first observed
// (the first Starting reconcile after desiredState=Running). It anchors the
+5
View File
@@ -923,6 +923,11 @@ func (r *Reconciler) markStarting(server *v1alpha1.MinecraftServer, reason, msg
t := r.now()
server.Status.StartRequestedAt = &t
}
// A server back in Starting is on its way to a new ready: a pod that dropped
// out under a Running server, a retry after Failed, an auto-restart. Keeping
// the last run's ready time would make markRunningReady treat the start as
// already observed, and every start after the first would go unmeasured.
server.Status.ReadySignalAt = nil
server.Status.Endpoint = v1alpha1.EndpointStatus{Mode: v1alpha1.EndpointFallback, Address: server.Spec.FallbackServer}
server.Status.LiveMotd = server.Spec.Motd.Starting
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
+42
View File
@@ -1025,6 +1025,48 @@ func TestReconcileRunning_StartDurationObservedOnce(t *testing.T) {
}
}
// A pod that drops out of readiness under a Running server sends it back through
// Starting; its way back to ready is a start like any other, and is observed as
// one, measured from the new Starting pass. Before, readySignalAt kept the first
// run's time, so every start after the first went unobserved.
func TestReconcileRunning_ObservesStartDurationAfterAPodBlip(t *testing.T) {
r, c := newReconciler(t, fakeProber{}, runningServer(), rconSecret())
base := time.Date(2026, 6, 25, 12, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
reconcile(t, r, "survival") // Starting
markPodReady(t, c, "survival")
reconcile(t, r, "survival") // Running: the first start's observation
beforeCount, beforeSum := startDurationState(t)
clock = base.Add(10 * time.Minute)
sts := getSTS(t, c, "survival")
sts.Status.ReadyReplicas = 0
if err := c.Status().Update(context.Background(), sts); err != nil {
t.Fatalf("update sts status: %v", err)
}
reconcile(t, r, "survival") // the pod is not ready: back to Starting
if s := getServer(t, c, "survival"); s.Status.Phase != v1alpha1.PhaseStarting || s.Status.ReadySignalAt != nil {
t.Fatalf("after the blip: phase %s, readySignalAt %v; want Starting and none", s.Status.Phase, s.Status.ReadySignalAt)
}
clock = base.Add(10*time.Minute + 45*time.Second)
markPodReady(t, c, "survival")
reconcile(t, r, "survival") // Running again
afterCount, afterSum := startDurationState(t)
if got := afterCount - beforeCount; got != 1 {
t.Fatalf("histogram sample count delta = %d, want exactly 1 observation for the restart", got)
}
if got := afterSum - beforeSum; got != 45 {
t.Errorf("observed restart duration = %vs, want 45s", got)
}
if s := getServer(t, c, "survival"); s.Status.ReadySignalAt == nil || !s.Status.ReadySignalAt.Equal(ptrTime(metav1.NewTime(clock))) {
t.Errorf("readySignalAt = %v, want the restart's ready time %v", s.Status.ReadySignalAt, clock)
}
}
// A server whose RCON Secret does not exist yet is the normal case on first
// reconcile — felis-api writes spec.rcon.secretRef but holds secrets:get, not
// create, so the name it points at is a promise the operator has to keep. Before