fix(operator): 回到 Starting 时清掉 readySignalAt,每次启动都记入启动时长指标
This commit is contained in:
4 files changed
+52
-2
No files matched your search
@@ -420,7 +420,9 @@ spec:
|
|||||||
status ping is never sufficient — spec §5).
|
status ping is never sufficient — spec §5).
|
||||||
type: boolean
|
type: boolean
|
||||||
readySignalAt:
|
readySignalAt:
|
||||||
description: ReadySignalAt is when the first RCON probe succeeded.
|
description: |-
|
||||||
|
ReadySignalAt is when the first RCON probe of the current run succeeded.
|
||||||
|
Every Starting pass clears it, so each start is measured once.
|
||||||
format: date-time
|
format: date-time
|
||||||
type: string
|
type: string
|
||||||
startRequestedAt:
|
startRequestedAt:
|
||||||
|
|||||||
@@ -328,7 +328,8 @@ type MinecraftServerStatus struct {
|
|||||||
Players PlayersStatus `json:"players,omitempty"`
|
Players PlayersStatus `json:"players,omitempty"`
|
||||||
// LiveMotd is the MOTD currently advertised for the active phase.
|
// LiveMotd is the MOTD currently advertised for the active phase.
|
||||||
LiveMotd string `json:"liveMotd,omitempty"`
|
LiveMotd string `json:"liveMotd,omitempty"`
|
||||||
// ReadySignalAt is when the first RCON probe succeeded.
|
// ReadySignalAt is when the first RCON probe of the current run succeeded.
|
||||||
|
// Every Starting pass clears it, so each start is measured once.
|
||||||
ReadySignalAt *metav1.Time `json:"readySignalAt,omitempty"`
|
ReadySignalAt *metav1.Time `json:"readySignalAt,omitempty"`
|
||||||
// StartRequestedAt is when the current start attempt was first observed
|
// StartRequestedAt is when the current start attempt was first observed
|
||||||
// (the first Starting reconcile after desiredState=Running). It anchors the
|
// (the first Starting reconcile after desiredState=Running). It anchors the
|
||||||
|
|||||||
@@ -923,6 +923,11 @@ func (r *Reconciler) markStarting(server *v1alpha1.MinecraftServer, reason, msg
|
|||||||
t := r.now()
|
t := r.now()
|
||||||
server.Status.StartRequestedAt = &t
|
server.Status.StartRequestedAt = &t
|
||||||
}
|
}
|
||||||
|
// A server back in Starting is on its way to a new ready: a pod that dropped
|
||||||
|
// out under a Running server, a retry after Failed, an auto-restart. Keeping
|
||||||
|
// the last run's ready time would make markRunningReady treat the start as
|
||||||
|
// already observed, and every start after the first would go unmeasured.
|
||||||
|
server.Status.ReadySignalAt = nil
|
||||||
server.Status.Endpoint = v1alpha1.EndpointStatus{Mode: v1alpha1.EndpointFallback, Address: server.Spec.FallbackServer}
|
server.Status.Endpoint = v1alpha1.EndpointStatus{Mode: v1alpha1.EndpointFallback, Address: server.Spec.FallbackServer}
|
||||||
server.Status.LiveMotd = server.Spec.Motd.Starting
|
server.Status.LiveMotd = server.Spec.Motd.Starting
|
||||||
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
|
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
|
||||||
|
|||||||
@@ -1025,6 +1025,48 @@ func TestReconcileRunning_StartDurationObservedOnce(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A pod that drops out of readiness under a Running server sends it back through
|
||||||
|
// Starting; its way back to ready is a start like any other, and is observed as
|
||||||
|
// one, measured from the new Starting pass. Before, readySignalAt kept the first
|
||||||
|
// run's time, so every start after the first went unobserved.
|
||||||
|
func TestReconcileRunning_ObservesStartDurationAfterAPodBlip(t *testing.T) {
|
||||||
|
r, c := newReconciler(t, fakeProber{}, runningServer(), rconSecret())
|
||||||
|
base := time.Date(2026, 6, 25, 12, 0, 0, 0, time.UTC)
|
||||||
|
clock := base
|
||||||
|
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
|
||||||
|
|
||||||
|
reconcile(t, r, "survival") // Starting
|
||||||
|
markPodReady(t, c, "survival")
|
||||||
|
reconcile(t, r, "survival") // Running: the first start's observation
|
||||||
|
beforeCount, beforeSum := startDurationState(t)
|
||||||
|
|
||||||
|
clock = base.Add(10 * time.Minute)
|
||||||
|
sts := getSTS(t, c, "survival")
|
||||||
|
sts.Status.ReadyReplicas = 0
|
||||||
|
if err := c.Status().Update(context.Background(), sts); err != nil {
|
||||||
|
t.Fatalf("update sts status: %v", err)
|
||||||
|
}
|
||||||
|
reconcile(t, r, "survival") // the pod is not ready: back to Starting
|
||||||
|
if s := getServer(t, c, "survival"); s.Status.Phase != v1alpha1.PhaseStarting || s.Status.ReadySignalAt != nil {
|
||||||
|
t.Fatalf("after the blip: phase %s, readySignalAt %v; want Starting and none", s.Status.Phase, s.Status.ReadySignalAt)
|
||||||
|
}
|
||||||
|
|
||||||
|
clock = base.Add(10*time.Minute + 45*time.Second)
|
||||||
|
markPodReady(t, c, "survival")
|
||||||
|
reconcile(t, r, "survival") // Running again
|
||||||
|
|
||||||
|
afterCount, afterSum := startDurationState(t)
|
||||||
|
if got := afterCount - beforeCount; got != 1 {
|
||||||
|
t.Fatalf("histogram sample count delta = %d, want exactly 1 observation for the restart", got)
|
||||||
|
}
|
||||||
|
if got := afterSum - beforeSum; got != 45 {
|
||||||
|
t.Errorf("observed restart duration = %vs, want 45s", got)
|
||||||
|
}
|
||||||
|
if s := getServer(t, c, "survival"); s.Status.ReadySignalAt == nil || !s.Status.ReadySignalAt.Equal(ptrTime(metav1.NewTime(clock))) {
|
||||||
|
t.Errorf("readySignalAt = %v, want the restart's ready time %v", s.Status.ReadySignalAt, clock)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// A server whose RCON Secret does not exist yet is the normal case on first
|
// A server whose RCON Secret does not exist yet is the normal case on first
|
||||||
// reconcile — felis-api writes spec.rcon.secretRef but holds secrets:get, not
|
// reconcile — felis-api writes spec.rcon.secretRef but holds secrets:get, not
|
||||||
// create, so the name it points at is a promise the operator has to keep. Before
|
// create, so the name it points at is a promise the operator has to keep. Before
|
||||||
|
|||||||
Reference in new issue
Block a user