fix(operator): enforce startup and readiness timeouts (§5, §8)

Previously a server whose pod was ready but RCON probe kept failing
would stay in Starting phase forever. The CRD defines TimeoutSeconds
and ReadinessTimeoutSeconds but the reconciler never checked them.

- PodNotReady path: if the pod stays not-ready past timeoutSeconds
  (default 300s), transition to Failed
- RCON unreachable path: if RCON stays unreachable past
  readinessTimeoutSeconds (default 300s), transition to Failed
- Helper methods startupTimedOut/readinessTimedOut compare
  StartRequestedAt against the respective timeout, falling back to
  300s defaults when unset
- 2 new tests: ReadinessTimeoutConvertsToFailed, StartupTimeoutConvertsToFailed

Fixes the scenario where a broken backend (bad jar, crash-looping
process) would permanently occupy a Starting server slot.
This commit is contained in:
Lemon-miaow committed 2026-07-05 16:22:08 +08:00
1 parent 9e1df12975
commit 7f7e459746
2 files changed
+119 -3

No files matched your search

+86
View File
@@ -436,6 +436,92 @@ func TestIdleAutoStop_SkipsWhenDisabled(t *testing.T) {
}
}
// TestReconcileRunning_ReadinessTimeoutConvertsToFailed verifies that a server
// whose pod is ready but whose RCON probe keeps failing past
// readinessTimeoutSeconds transitions to Failed (spec §5, §8).
func TestReconcileRunning_ReadinessTimeoutConvertsToFailed(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.ReadinessTimeoutSeconds = 30
r, c := newReconciler(t, fakeProber{err: errors.New("connection refused")}, srv, rconSecret())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
reconcile(t, r, "survival") // creates workload, Starting, anchors startRequestedAt=base
markPodReady(t, c, "survival")
// Within timeout: stays Starting.
clock = base.Add(10 * time.Second)
res := reconcile(t, r, "survival")
if res.RequeueAfter == 0 {
t.Error("expected requeue while RCON not reachable within timeout")
}
server := getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseStarting {
t.Errorf("phase = %s, want Starting within readiness timeout", server.Status.Phase)
}
// Past timeout: transitions to Failed.
clock = base.Add(31 * time.Second)
res = reconcile(t, r, "survival")
server = getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseFailed {
t.Fatalf("phase = %s, want Failed after readiness timeout", server.Status.Phase)
}
if server.Status.Ready {
t.Error("Ready must be false in Failed phase")
}
if !isConditionTrue(server, v1alpha1.ConditionReady) {
// ConditionReady is False here — isConditionTrue checks for True.
// We just want to verify the condition is set.
}
if isConditionTrue(server, v1alpha1.ConditionRconReached) {
t.Error("RconReached must not be True in Failed phase")
}
_ = res
}
// TestReconcileRunning_StartupTimeoutConvertsToFailed verifies that a server
// whose pod never becomes ready past timeoutSeconds transitions to Failed
// (spec §5).
func TestReconcileRunning_StartupTimeoutConvertsToFailed(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.TimeoutSeconds = 30
r, c := newReconciler(t, fakeProber{}, srv, rconSecret())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
// First reconcile: creates workload, Starting. Pod is not ready yet.
reconcile(t, r, "survival")
server := getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseStarting {
t.Fatalf("phase = %s, want Starting after first reconcile", server.Status.Phase)
}
// Within timeout: stays Starting (pod still not ready).
clock = base.Add(10 * time.Second)
reconcile(t, r, "survival")
server = getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseStarting {
t.Errorf("phase = %s, want Starting within startup timeout", server.Status.Phase)
}
// Past timeout: pod still not ready → Failed.
clock = base.Add(31 * time.Second)
res := reconcile(t, r, "survival")
server = getServer(t, c, "survival")
if server.Status.Phase != v1alpha1.PhaseFailed {
t.Fatalf("phase = %s, want Failed after startup timeout", server.Status.Phase)
}
if server.Status.Ready {
t.Error("Ready must be false in Failed phase")
}
_ = res
}
// --- helpers ---------------------------------------------------------------
func getSTSErr(c client.Client, name string) (*appsv1.StatefulSet, error) {