fix(operator): enforce startup and readiness timeouts (§5, §8)

Previously a server whose pod was ready but RCON probe kept failing
would stay in Starting phase forever. The CRD defines TimeoutSeconds
and ReadinessTimeoutSeconds but the reconciler never checked them.

- PodNotReady path: if the pod stays not-ready past timeoutSeconds
  (default 300s), transition to Failed
- RCON unreachable path: if RCON stays unreachable past
  readinessTimeoutSeconds (default 300s), transition to Failed
- Helper methods startupTimedOut/readinessTimedOut compare
  StartRequestedAt against the respective timeout, falling back to
  300s defaults when unset
- 2 new tests: ReadinessTimeoutConvertsToFailed, StartupTimeoutConvertsToFailed

Fixes the scenario where a broken backend (bad jar, crash-looping
process) would permanently occupy a Starting server slot.
This commit is contained in:
Lemon-miaow committed 2026-07-05 16:22:08 +08:00
1 parent 9e1df12975
commit 7f7e459746
2 files changed
+116

No files matched your search

+30
View File
@@ -28,6 +28,8 @@ const (
requeueStarting = 5 * time.Second requeueStarting = 5 * time.Second
requeueStopping = 5 * time.Second requeueStopping = 5 * time.Second
requeueSecret = 10 * time.Second requeueSecret = 10 * time.Second
defaultTimeoutSeconds = 300
defaultReadinessTimeoutSec = 300
) )
// Reconciler reconciles a MinecraftServer with its managed children. // Reconciler reconciles a MinecraftServer with its managed children.
@@ -102,6 +104,9 @@ func (r *Reconciler) reconcileRunning(ctx context.Context, server *v1alpha1.Mine
// The pod must first pass its tcpSocket readiness (readyReplicas >= 1). // The pod must first pass its tcpSocket readiness (readyReplicas >= 1).
if current.Status.ReadyReplicas < 1 { if current.Status.ReadyReplicas < 1 {
r.markStarting(server, "PodNotReady", "waiting for pod TCP readiness") r.markStarting(server, "PodNotReady", "waiting for pod TCP readiness")
if r.startupTimedOut(server) {
r.markFailed(server, "StartupTimeout", "pod did not become ready within startup timeout")
}
if err := r.patchStatus(ctx, server); err != nil { if err := r.patchStatus(ctx, server); err != nil {
return ctrl.Result{}, err return ctrl.Result{}, err
} }
@@ -123,6 +128,9 @@ func (r *Reconciler) reconcileRunning(ctx context.Context, server *v1alpha1.Mine
pc, err := r.Prober.Probe(ctx, rconAddress(server), password) pc, err := r.Prober.Probe(ctx, rconAddress(server), password)
if err != nil { if err != nil {
r.markStarting(server, "RconNotReachable", err.Error()) r.markStarting(server, "RconNotReachable", err.Error())
if r.readinessTimedOut(server) {
r.markFailed(server, "ReadinessTimeout", "RCON probe did not succeed within readiness timeout")
}
if perr := r.patchStatus(ctx, server); perr != nil { if perr := r.patchStatus(ctx, server); perr != nil {
return ctrl.Result{}, perr return ctrl.Result{}, perr
} }
@@ -346,3 +354,25 @@ func (r *Reconciler) setCondition(server *v1alpha1.MinecraftServer, condType str
LastTransitionTime: r.now(), LastTransitionTime: r.now(),
}) })
} }
func (r *Reconciler) startupTimedOut(server *v1alpha1.MinecraftServer) bool {
if server.Status.StartRequestedAt == nil {
return false
}
timeout := time.Duration(server.Spec.Startup.TimeoutSeconds) * time.Second
if timeout <= 0 {
timeout = defaultTimeoutSeconds * time.Second
}
return r.now().Time.Sub(server.Status.StartRequestedAt.Time) >= timeout
}
func (r *Reconciler) readinessTimedOut(server *v1alpha1.MinecraftServer) bool {
if server.Status.StartRequestedAt == nil {
return false
}
timeout := time.Duration(server.Spec.Startup.ReadinessTimeoutSeconds) * time.Second
if timeout <= 0 {
timeout = defaultReadinessTimeoutSec * time.Second
}
return r.now().Time.Sub(server.Status.StartRequestedAt.Time) >= timeout
}
+86
View File
@@ -436,6 +436,92 @@ func TestIdleAutoStop_SkipsWhenDisabled(t *testing.T) {
} }
} }
// TestReconcileRunning_ReadinessTimeoutConvertsToFailed verifies that a server
// whose pod is ready but whose RCON probe keeps failing past
// readinessTimeoutSeconds transitions to Failed (spec §5, §8).
func TestReconcileRunning_ReadinessTimeoutConvertsToFailed(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.ReadinessTimeoutSeconds = 30
r, c := newReconciler(t, fakeProber{err: errors.New("connection refused")}, srv, rconSecret())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
reconcile(t, r, "survival") // creates workload, Starting, anchors startRequestedAt=base
markPodReady(t, c, "survival")
// Within timeout: stays Starting.
clock = base.Add(10 * time.Second)
res := reconcile(t, r, "survival")
if res.RequeueAfter == 0 {
t.Error("expected requeue while RCON not reachable within timeout")
}
server := getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseStarting {
t.Errorf("phase = %s, want Starting within readiness timeout", server.Status.Phase)
}
// Past timeout: transitions to Failed.
clock = base.Add(31 * time.Second)
res = reconcile(t, r, "survival")
server = getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseFailed {
t.Fatalf("phase = %s, want Failed after readiness timeout", server.Status.Phase)
}
if server.Status.Ready {
t.Error("Ready must be false in Failed phase")
}
if !isConditionTrue(server, v1alpha1.ConditionReady) {
// ConditionReady is False here — isConditionTrue checks for True.
// We just want to verify the condition is set.
}
if isConditionTrue(server, v1alpha1.ConditionRconReached) {
t.Error("RconReached must not be True in Failed phase")
}
_ = res
}
// TestReconcileRunning_StartupTimeoutConvertsToFailed verifies that a server
// whose pod never becomes ready past timeoutSeconds transitions to Failed
// (spec §5).
func TestReconcileRunning_StartupTimeoutConvertsToFailed(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.TimeoutSeconds = 30
r, c := newReconciler(t, fakeProber{}, srv, rconSecret())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
// First reconcile: creates workload, Starting. Pod is not ready yet.
reconcile(t, r, "survival")
server := getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseStarting {
t.Fatalf("phase = %s, want Starting after first reconcile", server.Status.Phase)
}
// Within timeout: stays Starting (pod still not ready).
clock = base.Add(10 * time.Second)
reconcile(t, r, "survival")
server = getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseStarting {
t.Errorf("phase = %s, want Starting within startup timeout", server.Status.Phase)
}
// Past timeout: pod still not ready → Failed.
clock = base.Add(31 * time.Second)
res := reconcile(t, r, "survival")
server = getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseFailed {
t.Fatalf("phase = %s, want Failed after startup timeout", server.Status.Phase)
}
if server.Status.Ready {
t.Error("Ready must be false in Failed phase")
}
_ = res
}
// --- helpers --------------------------------------------------------------- // --- helpers ---------------------------------------------------------------
func getSTSErr(c client.Client, name string) (*appsv1.StatefulSet, error) { func getSTSErr(c client.Client, name string) (*appsv1.StatefulSet, error) {