fix(felis): 启动失败的服可在面板重试或停止,玩家入服时直接告知启动失败

This commit is contained in:
Lemon-miaow committed 2026-09-26 22:08:29 +08:00
1 parent c1025274e1
commit 22d1e834ad
28 files changed
+775 -75

No files matched your search

+18 -4
View File
@@ -1748,6 +1748,7 @@ type fakeCluster struct {
wakeErr map[string]error
acquired []string // "name:kind" per admitted AcquireMaintenance
released []string // names per ReleaseMaintenance
retried []string // names per RetryStart
}
func newFakeCluster() *fakeCluster {
@@ -1794,6 +1795,16 @@ func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.De
c.desired[n] = s
return nil
}
// RetryStart records the retry on top of the plain start, so a test tells a
// Failed server's retry apart from an ordinary wake.
func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
if err := c.SetDesiredState(ctx, n, v1alpha1.DesiredRunning); err != nil {
return err
}
c.retried = append(c.retried, n)
return nil
}
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
if err := c.maintErr[n]; err != nil {
return err
@@ -2092,9 +2103,11 @@ func TestMyServersJoinsLiveState(t *testing.T) {
cl := newFakeCluster()
cl.list = []ServerInfo{
{Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running",
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true},
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true,
AutoRestarts: 3, StartGaveUp: true},
{Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running",
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true},
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true,
AutoRestarts: 3, StartGaveUp: true},
}
api := newTestAPI(repo, cl)
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
@@ -2122,7 +2135,8 @@ func TestMyServersJoinsLiveState(t *testing.T) {
mine, open := got["mine"], got["open"]
for k, want := range map[string]any{"displayName": "My World", "phase": "Running",
"desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true,
"playersOnline": float64(2), "playersMax": float64(20)} {
"playersOnline": float64(2), "playersMax": float64(20),
"autoRestarts": float64(3), "startGaveUp": true} {
if mine[k] != want {
t.Errorf("own row %s = %v, want %v", k, mine[k], want)
}
@@ -2130,7 +2144,7 @@ func TestMyServersJoinsLiveState(t *testing.T) {
if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) {
t.Errorf("claimable row public fields = %v", open)
}
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown"} {
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown", "autoRestarts", "startGaveUp"} {
if _, ok := open[k]; ok {
t.Errorf("claimable row carries owner detail %s = %v", k, open[k])
}
+4
View File
@@ -117,6 +117,10 @@ type Cluster interface {
// *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore,
// backup or file write holds the world volume.
SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
// RetryStart is SetDesiredState(Running) for a server whose start Failed: it
// also asks the operator to start it over with a fresh auto-restart budget
// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
RetryStart(ctx context.Context, name string) error
// AcquireMaintenance admits one world-volume operation (internal/maintenance
// kind): ErrNotStopped unless the server is fully stopped, a
// *MaintenanceBusyError while another operation holds the volume. The check
+12
View File
@@ -152,6 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
writeError(w, r, err)
return
}
// A start whose automatic restarts are spent (or that can never succeed as
// configured) stays down until a person looks at it. The 202 this used to
// return queued the player for a server nothing was starting. The join leaves
// the restart budget alone, or every player who tried to join would buy
// another three crash loops; velocity tells them and queues no one. A Failed
// server still inside its backoff is not this: its next attempt is coming, so
// it gets the 202 and the player waits for it.
if info.StartGaveUp && info.DesiredState == string(v1alpha1.DesiredRunning) {
writeError(w, r, newError(http.StatusConflict, "start_failed",
"the server failed to start and its automatic retries are spent; its owner can retry from the panel"))
return
}
if !a.limiter().allowed(name, a.WakeCooldown) {
writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly"))
return
+131
View File
@@ -0,0 +1,131 @@
package api
import (
"net/http"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
)
// A start that Failed holds desiredState Running already, so the wake that
// re-writes Running changes nothing. These pin the two ways out: a person's
// wake in the panel starts it over (RetryStart), and a player's join onto one
// whose retries are spent is told so (409 start_failed) and never resets them.
func failedServer(gaveUp bool) *ServerInfo {
return &ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseFailed),
AutoRestarts: v1alpha1.MaxAutoRestarts, StartGaveUp: gaveUp}
}
func TestWakeOfFailedServerRetriesStart(t *testing.T) {
mk := func(info *ServerInfo) (*API, *fakeCluster, *fakeRepo) {
repo := newFakeRepo()
cl := newFakeCluster()
cl.byName["survival"] = info
api := newTestAPI(repo, cl)
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
return api, cl, repo
}
wake := func(api *API) int {
return do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil).Code
}
t.Run("retries spent: the wake starts it over", func(t *testing.T) {
api, cl, repo := mk(failedServer(true))
if code := wake(api); code != http.StatusAccepted {
t.Fatalf("code = %d, want 202", code)
}
if len(cl.retried) != 1 || cl.retried[0] != "survival" {
t.Fatalf("retried = %v, want [survival]", cl.retried)
}
if len(repo.audits) != 1 || repo.audits[0].Action != "retry_start" {
t.Fatalf("audits = %+v, want one retry_start", repo.audits)
}
})
t.Run("inside the backoff: a person asking skips the wait", func(t *testing.T) {
api, cl, _ := mk(failedServer(false))
if code := wake(api); code != http.StatusAccepted || len(cl.retried) != 1 {
t.Fatalf("code = %d retried = %v, want 202 and one retry", code, cl.retried)
}
})
t.Run("stopped: a plain wake", func(t *testing.T) {
api, cl, repo := mk(&ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)})
if code := wake(api); code != http.StatusAccepted {
t.Fatalf("code = %d, want 202", code)
}
if len(cl.retried) != 0 || cl.desired["survival"] != v1alpha1.DesiredRunning {
t.Fatalf("retried = %v desired = %q, want a plain start", cl.retried, cl.desired["survival"])
}
if len(repo.audits) != 1 || repo.audits[0].Action != "wake" {
t.Fatalf("audits = %+v, want one wake", repo.audits)
}
})
t.Run("maintenance holds the world: refused, nothing retried", func(t *testing.T) {
api, cl, _ := mk(failedServer(true))
cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore}
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil)
if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("retried = %v, want none", cl.retried)
}
})
}
func TestInternalWakeOfFailedServer(t *testing.T) {
body := `{"mc_uuid":"` + wakeUUID + `"}`
t.Run("retries spent: 409 start_failed, budget and cooldown untouched", func(t *testing.T) {
api, cl := newInternalWakeAPI("public")
cl.byName["survival"] = failedServer(true)
api.WakeCooldown = time.Minute
repo := api.Repo.(*fakeRepo)
w := internalWake(api, body)
if w.Code != http.StatusConflict || decodeErr(t, w) != "start_failed" {
t.Fatalf("code = %d body %s, want 409 start_failed", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
}
if len(repo.audits) != 0 {
t.Fatalf("nothing was woken, so nothing is audited: %+v", repo.audits)
}
// The owner fixes it and starts it from the panel; the next join is not
// held back by a cooldown the refused one never spent.
cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseStarting)}
if w := internalWake(api, body); w.Code != http.StatusAccepted {
t.Fatalf("join after the fix: code = %d body %s, want 202", w.Code, w.Body.String())
}
})
t.Run("inside the backoff: 202, the player waits for the next attempt", func(t *testing.T) {
api, cl := newInternalWakeAPI("public")
cl.byName["survival"] = failedServer(false)
if w := internalWake(api, body); w.Code != http.StatusAccepted {
t.Fatalf("code = %d body %s, want 202", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
}
})
t.Run("forbidden player: 403 comes first", func(t *testing.T) {
api, cl := newInternalWakeAPI("ownerOnly")
info := failedServer(true)
info.AutostartPolicy = "ownerOnly"
cl.byName["survival"] = info
if w := internalWake(api, body); w.Code != http.StatusForbidden {
t.Fatalf("code = %d, want 403", w.Code)
}
})
}
+20 -4
View File
@@ -56,9 +56,23 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
return
}
// Refused with 409 maintenance_in_progress while a restore, backup or file
// write holds the world volume: starting on a half-written world corrupts it.
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
// A start that Failed already holds desiredState Running, so writing Running
// again changes nothing and the server stayed dead once its automatic restarts
// were spent. A person pressing start on it means "try again": RetryStart
// makes the operator start it over with a fresh restart budget. Only this face
// does that; a player's join never resets the budget (handleInternalWake).
//
// Either write is refused with 409 maintenance_in_progress while a restore,
// backup or file write holds the world volume: starting on a half-written
// world corrupts it.
action := "wake"
if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) {
action = "retry_start"
err = a.Cluster.RetryStart(r.Context(), name)
} else {
err = a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning)
}
if err != nil {
a.writeLookupError(w, r, err)
return
}
@@ -67,7 +81,7 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
// held at capacity should retry the instant a slot frees, not wait out a
// cooldown their refused wake never earned).
a.limiter().record(name)
a.audit(r, "wake", name)
a.audit(r, action, name)
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
}
@@ -267,6 +281,8 @@ func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) {
v.DesiredState = info.DesiredState
v.AutostartPolicy = info.AutostartPolicy
v.PlayerCountUnknown = info.PlayerCountUnknown
v.AutoRestarts = info.AutoRestarts
v.StartGaveUp = info.StartGaveUp
}
}
}
+17
View File
@@ -246,6 +246,17 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a
// against, the same as AcquireMaintenance's: whichever of a racing wake and
// admission writes second gets a conflict, re-reads, and sees the other.
func (k *K8sCluster) start(ctx context.Context, name string) error {
return k.startWith(ctx, name, false)
}
// RetryStart starts a Failed server over: the same guarded write as start, plus
// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and
// recreates the pod.
func (k *K8sCluster) RetryStart(ctx context.Context, name string) error {
return k.startWith(ctx, name, true)
}
func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer
if err := k.getServer(ctx, name, &ms); err != nil {
@@ -263,6 +274,12 @@ func (k *K8sCluster) start(ctx context.Context, name string) error {
// A lock still on the object here no longer holds anything (Holder said
// so): drop it in the same write.
delete(ms.Annotations, maintenance.Annotation)
if retryFailed {
if ms.Annotations == nil {
ms.Annotations = map[string]string{}
}
ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339)
}
return k.c.Patch(ctx, &ms, patch)
})
}
@@ -214,6 +214,39 @@ func TestStartRespectsMaintenance(t *testing.T) {
}
})
t.Run("retry start -> Running plus the retry request, a plain start leaves none", func(t *testing.T) {
ms := stoppedServer()
ms.Spec.DesiredState = v1alpha1.DesiredRunning
ms.Status.Phase = v1alpha1.PhaseFailed
k, c := lockCluster(t, ms)
if err := k.SetDesiredState(ctx, "survival", v1alpha1.DesiredRunning); err != nil {
t.Fatalf("start: %v", err)
}
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
t.Fatalf("a plain start asked for a retry: %v", ann)
}
if err := k.RetryStart(ctx, "survival"); err != nil {
t.Fatalf("retry: %v", err)
}
ann, desired := annotations(t, c)
if desired != v1alpha1.DesiredRunning || ann[v1alpha1.AnnotationStartRetry] != lockNow.Format(time.RFC3339) {
t.Fatalf("desiredState = %q annotations = %v, want Running and the retry stamped %s",
desired, ann, lockNow.Format(time.RFC3339))
}
})
t.Run("retry start under a fresh lock -> busy, no retry request", func(t *testing.T) {
ms := stoppedServer()
ms.Annotations = map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lockNow)}
k, c := lockCluster(t, ms)
if err := k.RetryStart(ctx, "survival"); !errors.Is(err, ErrMaintenanceInProgress) {
t.Fatalf("err = %v, want maintenance in progress", err)
}
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
t.Fatalf("a refused retry left its request: %v", ann)
}
})
t.Run("stop ignores the lock", func(t *testing.T) {
ms := stoppedServer()
ms.Spec.DesiredState = v1alpha1.DesiredRunning
+5 -3
View File
@@ -21,9 +21,9 @@ type ServerRecord struct {
// may auto-start, or may claim. Everything after Claimable is NOT stored in
// Postgres — handleMyServers joins it best-effort from the CRD status
// (Cluster.ListServers) at read time, so a cluster hiccup renders 0/0 and the
// cached phase, never a 500. DesiredState, AutostartPolicy and
// PlayerCountUnknown are owner detail and stay empty on rows the caller does
// not own, the same split publicServerInfo makes on the status route.
// cached phase, never a 500. DesiredState, AutostartPolicy, PlayerCountUnknown,
// AutoRestarts and StartGaveUp are owner detail and stay empty on rows the caller
// does not own, the same split publicServerInfo makes on the status route.
type MyServerView struct {
Name string `json:"name"`
Subdomain string `json:"subdomain"`
@@ -36,6 +36,8 @@ type MyServerView struct {
DesiredState string `json:"desiredState,omitempty"`
AutostartPolicy string `json:"autostartPolicy,omitempty"`
PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"`
AutoRestarts int32 `json:"autoRestarts,omitempty"`
StartGaveUp bool `json:"startGaveUp,omitempty"`
}
// ServerOwnership is one live server's claim state as the fleet read joins it.
@@ -31,6 +31,13 @@ const (
// server; and felis-api never writes it, so marking a server takes kubectl on
// the cluster, which fits a switch that drops the forwarding secret.
LabelForwarding = GroupName + "/forwarding"
// AnnotationStartRetry is felis-api asking the operator to start a Failed
// server over: its value is the request time (RFC 3339). Re-patching
// desiredState to the Running it already holds changes nothing the operator
// can see, so a person pressing "retry" in the panel had no way through once
// the automatic restarts were spent. The operator takes the request once —
// fresh restart budget, new start anchor, pod recreated — and removes it.
AnnotationStartRetry = GroupName + "/start-retry"
)
// ForwardingLegacy is the LabelForwarding value that selects legacy forwarding.
+33
View File
@@ -195,6 +195,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu
desired = v1alpha1.DesiredStopped
}
prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts
if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok {
if err := r.takeStartRetry(ctx, &server, desired); err != nil {
return ctrl.Result{}, err
}
}
var res ctrl.Result
var err error
if desired == v1alpha1.DesiredStopped {
@@ -233,6 +238,33 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine
r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg))
}
// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed
// and meant to run starts over as if freshly woken — pod recreated, restart
// budget back to zero, a new start anchor — and in any other state the request
// is stale and only removed. The fresh status is written before the request is
// removed, so a pass that fails in between leaves the request to be taken again
// (at the cost of one more pod recreate), never a removed request with the old
// spent budget still in place.
func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error {
if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed {
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}}
if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil {
return err
}
server.Status.AutoRestarts = 0
server.Status.StartRequestedAt = nil
r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod")
if err := r.patchStatus(ctx, server); err != nil {
return err
}
log.FromContext(ctx).Info("start retried")
r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod")
}
patch := client.MergeFrom(server.DeepCopy())
delete(server.Annotations, v1alpha1.AnnotationStartRetry)
return r.Patch(ctx, server, patch)
}
func (r *Reconciler) event(server *v1alpha1.MinecraftServer, eventType, reason, msg string) {
if r.Recorder != nil {
r.Recorder.Event(server, eventType, reason, msg)
@@ -807,6 +839,7 @@ func (r *Reconciler) markFailed(server *v1alpha1.MinecraftServer, reason, msg st
server.Status.Phase = v1alpha1.PhaseFailed
server.Status.Ready = false
server.Status.ObservedGeneration = server.Generation
server.Status.LiveMotd = server.Spec.Motd.Failed
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
r.setCondition(server, v1alpha1.ConditionProvisioned, metav1.ConditionFalse, reason, msg)
}
+119
View File
@@ -0,0 +1,119 @@
package operator_test
import (
"context"
"testing"
"time"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
)
// requestStartRetry is felis-api's RetryStart as the operator sees it.
func requestStartRetry(t *testing.T, c client.Client) {
t.Helper()
s := getServer(t, c, "survival")
patch := client.MergeFrom(s.DeepCopy())
if s.Annotations == nil {
s.Annotations = map[string]string{}
}
s.Annotations[v1alpha1.AnnotationStartRetry] = "2026-07-01T12:00:00Z"
if err := c.Patch(context.Background(), s, patch); err != nil {
t.Fatalf("annotate: %v", err)
}
}
func podPresent(t *testing.T, c client.Client) bool {
t.Helper()
err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival-0"}, &corev1.Pod{})
if err != nil && !apierrors.IsNotFound(err) {
t.Fatalf("get pod: %v", err)
}
return err == nil
}
// Once the automatic restarts are spent, a retry request starts the server over
// with the whole budget back: the pod is recreated, the start re-anchored, and
// the next timeout is retried automatically again.
func TestStartRetryRevivesAServerThatGaveUp(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.TimeoutSeconds = 30
srv.Spec.Motd.Failed = "broken"
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
at := func(offset time.Duration) *v1alpha1.MinecraftServer {
clock = base.Add(offset)
reconcile(t, r, "survival")
return getServer(t, c, "survival")
}
recreatePod := func() {
if err := c.Create(context.Background(), gamePod()); err != nil {
t.Fatal(err)
}
}
// Spend the three automatic restarts (autorestart_test.go pins the schedule).
at(0)
for _, due := range []time.Duration{90 * time.Second, 240 * time.Second, 510 * time.Second} {
at(due)
recreatePod()
}
s := at(time.Hour)
if !v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("setup: not given up: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts)
}
if s.Status.LiveMotd != "broken" {
t.Fatalf("Failed advertises liveMotd %q, want the failed MOTD", s.Status.LiveMotd)
}
requestStartRetry(t, c)
s = at(2 * time.Hour)
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
t.Fatal("the retry request was left on the server")
}
if podPresent(t, c) {
t.Fatal("the retry kept the failed pod")
}
if s.Status.Phase != v1alpha1.PhaseStarting || s.Status.AutoRestarts != 0 || v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("after retry: phase=%s restarts=%d gaveUp=%v", s.Status.Phase, s.Status.AutoRestarts, v1alpha1.StartGaveUp(&s.Status))
}
if s.Status.StartRequestedAt == nil || !s.Status.StartRequestedAt.Time.Equal(base.Add(2*time.Hour)) {
t.Fatalf("start not re-anchored at the retry: %v", s.Status.StartRequestedAt)
}
recreatePod()
// The budget is really back: this start times out and is retried on its own.
if s := at(2*time.Hour + 89*time.Second); s.Status.Phase != v1alpha1.PhaseFailed || v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("retried start timing out: phase=%s gaveUp=%v, want Failed with retries left", s.Status.Phase, v1alpha1.StartGaveUp(&s.Status))
}
if s := at(2*time.Hour + 90*time.Second); s.Status.AutoRestarts != 1 || podPresent(t, c) {
t.Fatalf("retried start: restarts=%d pod=%v, want the first automatic restart", s.Status.AutoRestarts, podPresent(t, c))
}
}
// A request that finds the server anywhere but Failed is stale (the start
// already recovered, or someone stopped it): it is removed and nothing else.
func TestStaleStartRetryIsOnlyRemoved(t *testing.T) {
srv := runningServer()
srv.Annotations = map[string]string{v1alpha1.AnnotationStartRetry: "2026-07-01T12:00:00Z"}
anchor := metav1.NewTime(fixedNow().Add(-10 * time.Second))
srv.Status = v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting, AutoRestarts: 2, StartRequestedAt: &anchor}
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
reconcile(t, r, "survival")
s := getServer(t, c, "survival")
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
t.Fatal("the stale retry request was left on the server")
}
if !podPresent(t, c) || s.Status.AutoRestarts != 2 || !s.Status.StartRequestedAt.Equal(&anchor) {
t.Fatalf("a stale request touched the start: pod=%v restarts=%d anchor=%v",
podPresent(t, c), s.Status.AutoRestarts, s.Status.StartRequestedAt)
}
}