diff --git a/docs/openapi.yaml b/docs/openapi.yaml index 170192d..141062e 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -566,6 +566,13 @@ components: playerCountUnknown: type: boolean description: Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players. + autoRestarts: + type: integer + format: int32 + description: Owned rows only. How often the operator recreated the pod of this start after it timed out. + startGaveUp: + type: boolean + description: Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over. BackupView: type: object @@ -1165,7 +1172,9 @@ paths: description: >- velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the - server's autostartPolicy and the per-server wake cooldown. + server's autostartPolicy and the per-server wake cooldown. A server whose + start failed with its automatic retries spent answers 409 start_failed: a + join never resets the retry budget, so velocity queues no one for it. x-felis-face: [internal] x-felis-tier: service x-felis-callers: [velocity] @@ -1203,7 +1212,11 @@ paths: '404': $ref: '#/components/responses/NotFound' '409': - description: A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started. + description: >- + Nothing was started. maintenance_in_progress: a restore, backup or + file write holds the server's world volume. start_failed: the last + start failed and its automatic retries are spent (ServerInfo.startGaveUp); + the server stays down until a person starts it from the panel. content: application/json: schema: { $ref: '#/components/schemas/Error' } @@ -1767,6 +1780,10 @@ paths: tags: [servers] operationId: wake summary: Wake your own server. + description: >- + On a server whose start Failed (phase Failed, desiredState Running) this + is a retry: the operator recreates the pod and starts it over with a + fresh automatic-restart budget. It is audited as retry_start. x-felis-face: [external] x-felis-tier: app security: [{ sessionCookie: [] }] diff --git a/internal/api/api_test.go b/internal/api/api_test.go index 0af415c..ea40359 100644 --- a/internal/api/api_test.go +++ b/internal/api/api_test.go @@ -1748,6 +1748,7 @@ type fakeCluster struct { wakeErr map[string]error acquired []string // "name:kind" per admitted AcquireMaintenance released []string // names per ReleaseMaintenance + retried []string // names per RetryStart } func newFakeCluster() *fakeCluster { @@ -1794,6 +1795,16 @@ func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.De c.desired[n] = s return nil } + +// RetryStart records the retry on top of the plain start, so a test tells a +// Failed server's retry apart from an ordinary wake. +func (c *fakeCluster) RetryStart(ctx context.Context, n string) error { + if err := c.SetDesiredState(ctx, n, v1alpha1.DesiredRunning); err != nil { + return err + } + c.retried = append(c.retried, n) + return nil +} func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error { if err := c.maintErr[n]; err != nil { return err @@ -2092,9 +2103,11 @@ func TestMyServersJoinsLiveState(t *testing.T) { cl := newFakeCluster() cl.list = []ServerInfo{ {Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running", - AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true}, + AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true, + AutoRestarts: 3, StartGaveUp: true}, {Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running", - AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true}, + AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true, + AutoRestarts: 3, StartGaveUp: true}, } api := newTestAPI(repo, cl) api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}} @@ -2122,7 +2135,8 @@ func TestMyServersJoinsLiveState(t *testing.T) { mine, open := got["mine"], got["open"] for k, want := range map[string]any{"displayName": "My World", "phase": "Running", "desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true, - "playersOnline": float64(2), "playersMax": float64(20)} { + "playersOnline": float64(2), "playersMax": float64(20), + "autoRestarts": float64(3), "startGaveUp": true} { if mine[k] != want { t.Errorf("own row %s = %v, want %v", k, mine[k], want) } @@ -2130,7 +2144,7 @@ func TestMyServersJoinsLiveState(t *testing.T) { if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) { t.Errorf("claimable row public fields = %v", open) } - for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown"} { + for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown", "autoRestarts", "startGaveUp"} { if _, ok := open[k]; ok { t.Errorf("claimable row carries owner detail %s = %v", k, open[k]) } diff --git a/internal/api/cluster.go b/internal/api/cluster.go index eb0f2a4..c7d16bf 100644 --- a/internal/api/cluster.go +++ b/internal/api/cluster.go @@ -117,6 +117,10 @@ type Cluster interface { // *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore, // backup or file write holds the world volume. SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error + // RetryStart is SetDesiredState(Running) for a server whose start Failed: it + // also asks the operator to start it over with a fresh auto-restart budget + // (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way. + RetryStart(ctx context.Context, name string) error // AcquireMaintenance admits one world-volume operation (internal/maintenance // kind): ErrNotStopped unless the server is fully stopped, a // *MaintenanceBusyError while another operation holds the volume. The check diff --git a/internal/api/handlers_internal.go b/internal/api/handlers_internal.go index f683b20..6ed7270 100644 --- a/internal/api/handlers_internal.go +++ b/internal/api/handlers_internal.go @@ -152,6 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) { writeError(w, r, err) return } + // A start whose automatic restarts are spent (or that can never succeed as + // configured) stays down until a person looks at it. The 202 this used to + // return queued the player for a server nothing was starting. The join leaves + // the restart budget alone, or every player who tried to join would buy + // another three crash loops; velocity tells them and queues no one. A Failed + // server still inside its backoff is not this: its next attempt is coming, so + // it gets the 202 and the player waits for it. + if info.StartGaveUp && info.DesiredState == string(v1alpha1.DesiredRunning) { + writeError(w, r, newError(http.StatusConflict, "start_failed", + "the server failed to start and its automatic retries are spent; its owner can retry from the panel")) + return + } if !a.limiter().allowed(name, a.WakeCooldown) { writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly")) return diff --git a/internal/api/handlers_start_failed_test.go b/internal/api/handlers_start_failed_test.go new file mode 100644 index 0000000..ead7caf --- /dev/null +++ b/internal/api/handlers_start_failed_test.go @@ -0,0 +1,131 @@ +package api + +import ( + "net/http" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/maintenance" +) + +// A start that Failed holds desiredState Running already, so the wake that +// re-writes Running changes nothing. These pin the two ways out: a person's +// wake in the panel starts it over (RetryStart), and a player's join onto one +// whose retries are spent is told so (409 start_failed) and never resets them. + +func failedServer(gaveUp bool) *ServerInfo { + return &ServerInfo{Name: "survival", AutostartPolicy: "public", + DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseFailed), + AutoRestarts: v1alpha1.MaxAutoRestarts, StartGaveUp: gaveUp} +} + +func TestWakeOfFailedServerRetriesStart(t *testing.T) { + mk := func(info *ServerInfo) (*API, *fakeCluster, *fakeRepo) { + repo := newFakeRepo() + cl := newFakeCluster() + cl.byName["survival"] = info + api := newTestAPI(repo, cl) + api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}} + return api, cl, repo + } + wake := func(api *API) int { + return do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil).Code + } + + t.Run("retries spent: the wake starts it over", func(t *testing.T) { + api, cl, repo := mk(failedServer(true)) + if code := wake(api); code != http.StatusAccepted { + t.Fatalf("code = %d, want 202", code) + } + if len(cl.retried) != 1 || cl.retried[0] != "survival" { + t.Fatalf("retried = %v, want [survival]", cl.retried) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "retry_start" { + t.Fatalf("audits = %+v, want one retry_start", repo.audits) + } + }) + + t.Run("inside the backoff: a person asking skips the wait", func(t *testing.T) { + api, cl, _ := mk(failedServer(false)) + if code := wake(api); code != http.StatusAccepted || len(cl.retried) != 1 { + t.Fatalf("code = %d retried = %v, want 202 and one retry", code, cl.retried) + } + }) + + t.Run("stopped: a plain wake", func(t *testing.T) { + api, cl, repo := mk(&ServerInfo{Name: "survival", AutostartPolicy: "public", + DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)}) + if code := wake(api); code != http.StatusAccepted { + t.Fatalf("code = %d, want 202", code) + } + if len(cl.retried) != 0 || cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("retried = %v desired = %q, want a plain start", cl.retried, cl.desired["survival"]) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "wake" { + t.Fatalf("audits = %+v, want one wake", repo.audits) + } + }) + + t.Run("maintenance holds the world: refused, nothing retried", func(t *testing.T) { + api, cl, _ := mk(failedServer(true)) + cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" { + t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String()) + } + if len(cl.retried) != 0 { + t.Fatalf("retried = %v, want none", cl.retried) + } + }) +} + +func TestInternalWakeOfFailedServer(t *testing.T) { + body := `{"mc_uuid":"` + wakeUUID + `"}` + + t.Run("retries spent: 409 start_failed, budget and cooldown untouched", func(t *testing.T) { + api, cl := newInternalWakeAPI("public") + cl.byName["survival"] = failedServer(true) + api.WakeCooldown = time.Minute + repo := api.Repo.(*fakeRepo) + + w := internalWake(api, body) + if w.Code != http.StatusConflict || decodeErr(t, w) != "start_failed" { + t.Fatalf("code = %d body %s, want 409 start_failed", w.Code, w.Body.String()) + } + if len(cl.retried) != 0 { + t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried) + } + if len(repo.audits) != 0 { + t.Fatalf("nothing was woken, so nothing is audited: %+v", repo.audits) + } + // The owner fixes it and starts it from the panel; the next join is not + // held back by a cooldown the refused one never spent. + cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public", + DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseStarting)} + if w := internalWake(api, body); w.Code != http.StatusAccepted { + t.Fatalf("join after the fix: code = %d body %s, want 202", w.Code, w.Body.String()) + } + }) + + t.Run("inside the backoff: 202, the player waits for the next attempt", func(t *testing.T) { + api, cl := newInternalWakeAPI("public") + cl.byName["survival"] = failedServer(false) + if w := internalWake(api, body); w.Code != http.StatusAccepted { + t.Fatalf("code = %d body %s, want 202", w.Code, w.Body.String()) + } + if len(cl.retried) != 0 { + t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried) + } + }) + + t.Run("forbidden player: 403 comes first", func(t *testing.T) { + api, cl := newInternalWakeAPI("ownerOnly") + info := failedServer(true) + info.AutostartPolicy = "ownerOnly" + cl.byName["survival"] = info + if w := internalWake(api, body); w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + }) +} diff --git a/internal/api/handlers_user.go b/internal/api/handlers_user.go index 490270f..f3dc37a 100644 --- a/internal/api/handlers_user.go +++ b/internal/api/handlers_user.go @@ -56,9 +56,23 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) { return } - // Refused with 409 maintenance_in_progress while a restore, backup or file - // write holds the world volume: starting on a half-written world corrupts it. - if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil { + // A start that Failed already holds desiredState Running, so writing Running + // again changes nothing and the server stayed dead once its automatic restarts + // were spent. A person pressing start on it means "try again": RetryStart + // makes the operator start it over with a fresh restart budget. Only this face + // does that; a player's join never resets the budget (handleInternalWake). + // + // Either write is refused with 409 maintenance_in_progress while a restore, + // backup or file write holds the world volume: starting on a half-written + // world corrupts it. + action := "wake" + if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) { + action = "retry_start" + err = a.Cluster.RetryStart(r.Context(), name) + } else { + err = a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning) + } + if err != nil { a.writeLookupError(w, r, err) return } @@ -67,7 +81,7 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) { // held at capacity should retry the instant a slot frees, not wait out a // cooldown their refused wake never earned). a.limiter().record(name) - a.audit(r, "wake", name) + a.audit(r, action, name) writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"}) } @@ -267,6 +281,8 @@ func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) { v.DesiredState = info.DesiredState v.AutostartPolicy = info.AutostartPolicy v.PlayerCountUnknown = info.PlayerCountUnknown + v.AutoRestarts = info.AutoRestarts + v.StartGaveUp = info.StartGaveUp } } } diff --git a/internal/api/k8scluster.go b/internal/api/k8scluster.go index 014e7a3..f1919d1 100644 --- a/internal/api/k8scluster.go +++ b/internal/api/k8scluster.go @@ -246,6 +246,17 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a // against, the same as AcquireMaintenance's: whichever of a racing wake and // admission writes second gets a conflict, re-reads, and sees the other. func (k *K8sCluster) start(ctx context.Context, name string) error { + return k.startWith(ctx, name, false) +} + +// RetryStart starts a Failed server over: the same guarded write as start, plus +// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and +// recreates the pod. +func (k *K8sCluster) RetryStart(ctx context.Context, name string) error { + return k.startWith(ctx, name, true) +} + +func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error { return retry.RetryOnConflict(retry.DefaultRetry, func() error { var ms v1alpha1.MinecraftServer if err := k.getServer(ctx, name, &ms); err != nil { @@ -263,6 +274,12 @@ func (k *K8sCluster) start(ctx context.Context, name string) error { // A lock still on the object here no longer holds anything (Holder said // so): drop it in the same write. delete(ms.Annotations, maintenance.Annotation) + if retryFailed { + if ms.Annotations == nil { + ms.Annotations = map[string]string{} + } + ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339) + } return k.c.Patch(ctx, &ms, patch) }) } diff --git a/internal/api/k8scluster_maintenance_test.go b/internal/api/k8scluster_maintenance_test.go index d5a2715..efe734a 100644 --- a/internal/api/k8scluster_maintenance_test.go +++ b/internal/api/k8scluster_maintenance_test.go @@ -214,6 +214,39 @@ func TestStartRespectsMaintenance(t *testing.T) { } }) + t.Run("retry start -> Running plus the retry request, a plain start leaves none", func(t *testing.T) { + ms := stoppedServer() + ms.Spec.DesiredState = v1alpha1.DesiredRunning + ms.Status.Phase = v1alpha1.PhaseFailed + k, c := lockCluster(t, ms) + if err := k.SetDesiredState(ctx, "survival", v1alpha1.DesiredRunning); err != nil { + t.Fatalf("start: %v", err) + } + if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" { + t.Fatalf("a plain start asked for a retry: %v", ann) + } + if err := k.RetryStart(ctx, "survival"); err != nil { + t.Fatalf("retry: %v", err) + } + ann, desired := annotations(t, c) + if desired != v1alpha1.DesiredRunning || ann[v1alpha1.AnnotationStartRetry] != lockNow.Format(time.RFC3339) { + t.Fatalf("desiredState = %q annotations = %v, want Running and the retry stamped %s", + desired, ann, lockNow.Format(time.RFC3339)) + } + }) + + t.Run("retry start under a fresh lock -> busy, no retry request", func(t *testing.T) { + ms := stoppedServer() + ms.Annotations = map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lockNow)} + k, c := lockCluster(t, ms) + if err := k.RetryStart(ctx, "survival"); !errors.Is(err, ErrMaintenanceInProgress) { + t.Fatalf("err = %v, want maintenance in progress", err) + } + if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" { + t.Fatalf("a refused retry left its request: %v", ann) + } + }) + t.Run("stop ignores the lock", func(t *testing.T) { ms := stoppedServer() ms.Spec.DesiredState = v1alpha1.DesiredRunning diff --git a/internal/api/repo.go b/internal/api/repo.go index f9da6ba..82c155b 100644 --- a/internal/api/repo.go +++ b/internal/api/repo.go @@ -21,9 +21,9 @@ type ServerRecord struct { // may auto-start, or may claim. Everything after Claimable is NOT stored in // Postgres — handleMyServers joins it best-effort from the CRD status // (Cluster.ListServers) at read time, so a cluster hiccup renders 0/0 and the -// cached phase, never a 500. DesiredState, AutostartPolicy and -// PlayerCountUnknown are owner detail and stay empty on rows the caller does -// not own, the same split publicServerInfo makes on the status route. +// cached phase, never a 500. DesiredState, AutostartPolicy, PlayerCountUnknown, +// AutoRestarts and StartGaveUp are owner detail and stay empty on rows the caller +// does not own, the same split publicServerInfo makes on the status route. type MyServerView struct { Name string `json:"name"` Subdomain string `json:"subdomain"` @@ -36,6 +36,8 @@ type MyServerView struct { DesiredState string `json:"desiredState,omitempty"` AutostartPolicy string `json:"autostartPolicy,omitempty"` PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"` + AutoRestarts int32 `json:"autoRestarts,omitempty"` + StartGaveUp bool `json:"startGaveUp,omitempty"` } // ServerOwnership is one live server's claim state as the fleet read joins it. diff --git a/internal/apis/felis/v1alpha1/minecraftserver_types.go b/internal/apis/felis/v1alpha1/minecraftserver_types.go index 7a28eae..ba8f9dc 100644 --- a/internal/apis/felis/v1alpha1/minecraftserver_types.go +++ b/internal/apis/felis/v1alpha1/minecraftserver_types.go @@ -31,6 +31,13 @@ const ( // server; and felis-api never writes it, so marking a server takes kubectl on // the cluster, which fits a switch that drops the forwarding secret. LabelForwarding = GroupName + "/forwarding" + // AnnotationStartRetry is felis-api asking the operator to start a Failed + // server over: its value is the request time (RFC 3339). Re-patching + // desiredState to the Running it already holds changes nothing the operator + // can see, so a person pressing "retry" in the panel had no way through once + // the automatic restarts were spent. The operator takes the request once — + // fresh restart budget, new start anchor, pod recreated — and removes it. + AnnotationStartRetry = GroupName + "/start-retry" ) // ForwardingLegacy is the LabelForwarding value that selects legacy forwarding. diff --git a/internal/operator/reconciler.go b/internal/operator/reconciler.go index 5bedfa1..b6b51e5 100644 --- a/internal/operator/reconciler.go +++ b/internal/operator/reconciler.go @@ -195,6 +195,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu desired = v1alpha1.DesiredStopped } prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts + if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok { + if err := r.takeStartRetry(ctx, &server, desired); err != nil { + return ctrl.Result{}, err + } + } var res ctrl.Result var err error if desired == v1alpha1.DesiredStopped { @@ -233,6 +238,33 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg)) } +// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed +// and meant to run starts over as if freshly woken — pod recreated, restart +// budget back to zero, a new start anchor — and in any other state the request +// is stale and only removed. The fresh status is written before the request is +// removed, so a pass that fails in between leaves the request to be taken again +// (at the cost of one more pod recreate), never a removed request with the old +// spent budget still in place. +func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error { + if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed { + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}} + if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil { + return err + } + server.Status.AutoRestarts = 0 + server.Status.StartRequestedAt = nil + r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod") + if err := r.patchStatus(ctx, server); err != nil { + return err + } + log.FromContext(ctx).Info("start retried") + r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod") + } + patch := client.MergeFrom(server.DeepCopy()) + delete(server.Annotations, v1alpha1.AnnotationStartRetry) + return r.Patch(ctx, server, patch) +} + func (r *Reconciler) event(server *v1alpha1.MinecraftServer, eventType, reason, msg string) { if r.Recorder != nil { r.Recorder.Event(server, eventType, reason, msg) @@ -807,6 +839,7 @@ func (r *Reconciler) markFailed(server *v1alpha1.MinecraftServer, reason, msg st server.Status.Phase = v1alpha1.PhaseFailed server.Status.Ready = false server.Status.ObservedGeneration = server.Generation + server.Status.LiveMotd = server.Spec.Motd.Failed r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg) r.setCondition(server, v1alpha1.ConditionProvisioned, metav1.ConditionFalse, reason, msg) } diff --git a/internal/operator/startretry_test.go b/internal/operator/startretry_test.go new file mode 100644 index 0000000..89f8956 --- /dev/null +++ b/internal/operator/startretry_test.go @@ -0,0 +1,119 @@ +package operator_test + +import ( + "context" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + "sigs.k8s.io/controller-runtime/pkg/client" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" +) + +// requestStartRetry is felis-api's RetryStart as the operator sees it. +func requestStartRetry(t *testing.T, c client.Client) { + t.Helper() + s := getServer(t, c, "survival") + patch := client.MergeFrom(s.DeepCopy()) + if s.Annotations == nil { + s.Annotations = map[string]string{} + } + s.Annotations[v1alpha1.AnnotationStartRetry] = "2026-07-01T12:00:00Z" + if err := c.Patch(context.Background(), s, patch); err != nil { + t.Fatalf("annotate: %v", err) + } +} + +func podPresent(t *testing.T, c client.Client) bool { + t.Helper() + err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival-0"}, &corev1.Pod{}) + if err != nil && !apierrors.IsNotFound(err) { + t.Fatalf("get pod: %v", err) + } + return err == nil +} + +// Once the automatic restarts are spent, a retry request starts the server over +// with the whole budget back: the pod is recreated, the start re-anchored, and +// the next timeout is retried automatically again. +func TestStartRetryRevivesAServerThatGaveUp(t *testing.T) { + srv := runningServer() + srv.Spec.Startup.TimeoutSeconds = 30 + srv.Spec.Motd.Failed = "broken" + r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod()) + base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC) + clock := base + r.Now = func() metav1.Time { return metav1.NewTime(clock) } + at := func(offset time.Duration) *v1alpha1.MinecraftServer { + clock = base.Add(offset) + reconcile(t, r, "survival") + return getServer(t, c, "survival") + } + recreatePod := func() { + if err := c.Create(context.Background(), gamePod()); err != nil { + t.Fatal(err) + } + } + + // Spend the three automatic restarts (autorestart_test.go pins the schedule). + at(0) + for _, due := range []time.Duration{90 * time.Second, 240 * time.Second, 510 * time.Second} { + at(due) + recreatePod() + } + s := at(time.Hour) + if !v1alpha1.StartGaveUp(&s.Status) { + t.Fatalf("setup: not given up: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts) + } + if s.Status.LiveMotd != "broken" { + t.Fatalf("Failed advertises liveMotd %q, want the failed MOTD", s.Status.LiveMotd) + } + + requestStartRetry(t, c) + s = at(2 * time.Hour) + if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left { + t.Fatal("the retry request was left on the server") + } + if podPresent(t, c) { + t.Fatal("the retry kept the failed pod") + } + if s.Status.Phase != v1alpha1.PhaseStarting || s.Status.AutoRestarts != 0 || v1alpha1.StartGaveUp(&s.Status) { + t.Fatalf("after retry: phase=%s restarts=%d gaveUp=%v", s.Status.Phase, s.Status.AutoRestarts, v1alpha1.StartGaveUp(&s.Status)) + } + if s.Status.StartRequestedAt == nil || !s.Status.StartRequestedAt.Time.Equal(base.Add(2*time.Hour)) { + t.Fatalf("start not re-anchored at the retry: %v", s.Status.StartRequestedAt) + } + recreatePod() + + // The budget is really back: this start times out and is retried on its own. + if s := at(2*time.Hour + 89*time.Second); s.Status.Phase != v1alpha1.PhaseFailed || v1alpha1.StartGaveUp(&s.Status) { + t.Fatalf("retried start timing out: phase=%s gaveUp=%v, want Failed with retries left", s.Status.Phase, v1alpha1.StartGaveUp(&s.Status)) + } + if s := at(2*time.Hour + 90*time.Second); s.Status.AutoRestarts != 1 || podPresent(t, c) { + t.Fatalf("retried start: restarts=%d pod=%v, want the first automatic restart", s.Status.AutoRestarts, podPresent(t, c)) + } +} + +// A request that finds the server anywhere but Failed is stale (the start +// already recovered, or someone stopped it): it is removed and nothing else. +func TestStaleStartRetryIsOnlyRemoved(t *testing.T) { + srv := runningServer() + srv.Annotations = map[string]string{v1alpha1.AnnotationStartRetry: "2026-07-01T12:00:00Z"} + anchor := metav1.NewTime(fixedNow().Add(-10 * time.Second)) + srv.Status = v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting, AutoRestarts: 2, StartRequestedAt: &anchor} + r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod()) + + reconcile(t, r, "survival") + s := getServer(t, c, "survival") + if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left { + t.Fatal("the stale retry request was left on the server") + } + if !podPresent(t, c) || s.Status.AutoRestarts != 2 || !s.Status.StartRequestedAt.Equal(&anchor) { + t.Fatalf("a stale request touched the start: pod=%v restarts=%d anchor=%v", + podPresent(t, c), s.Status.AutoRestarts, s.Status.StartRequestedAt) + } +} diff --git a/panel/src/components/PhaseBadge.test.ts b/panel/src/components/PhaseBadge.test.ts index 81c2d42..6868243 100644 --- a/panel/src/components/PhaseBadge.test.ts +++ b/panel/src/components/PhaseBadge.test.ts @@ -1,5 +1,5 @@ import { describe, it, expect } from "vitest"; -import { PHASE_COLOR, phaseColor, phaseVariant } from "@/components/PhaseBadge"; +import { PHASE_COLOR, phaseColor, phaseVariant, startFailure } from "@/components/PhaseBadge"; import type { Phase } from "@/lib/types"; // The closed lifecycle set felis-api can emit — the Go source of truth is @@ -47,3 +47,15 @@ describe("unmodelled phase fallback", () => { expect(phaseVariant(future)).toBe(phaseVariant("Unknown")); }); }); + +describe("start failure", () => { + // A Failed phase only reads as a failed start while the server still wants to + // run; a stopped Failed server needs nothing from its owner. + it("reads a failed start only while the server wants to run", () => { + expect(startFailure({ phase: "Running", desiredState: "Running" })).toBeNull(); + expect(startFailure({ phase: "Failed", desiredState: "Stopped" })).toBeNull(); + expect(startFailure({ phase: "Failed" })).toBeNull(); + expect(startFailure({ phase: "Failed", desiredState: "Running" })).toBe("retrying"); + expect(startFailure({ phase: "Failed", desiredState: "Running", startGaveUp: true })).toBe("gaveUp"); + }); +}); diff --git a/panel/src/components/PhaseBadge.tsx b/panel/src/components/PhaseBadge.tsx index fae5b50..5424049 100644 --- a/panel/src/components/PhaseBadge.tsx +++ b/panel/src/components/PhaseBadge.tsx @@ -52,18 +52,54 @@ export function phaseVariant(phase: Phase): BadgeVariant { return VARIANT[phase] ?? VARIANT.Unknown; } -export function PhaseBadge({ phase }: { phase: Phase }) { +/** MAX_AUTO_RESTARTS mirrors the operator's v1alpha1.MaxAutoRestarts: how often it + * recreates the pod of a start that timed out before leaving it Failed. */ +export const MAX_AUTO_RESTARTS = 3; + +export type StartFailure = "retrying" | "gaveUp"; + +/** startFailure reads how a Failed start stands from the owner detail. "retrying" + * means the operator's next automatic attempt is coming; "gaveUp" means nothing + * will start it again until a person does. A Failed server meant to stop, or a + * public view without the owner detail, reads as null: there is no retry to + * offer there, and the badge stays the plain Failed one. */ +export function startFailure(s: { + phase?: Phase; + desiredState?: string; + startGaveUp?: boolean; +}): StartFailure | null { + if (s.phase !== "Failed" || s.desiredState !== "Running") return null; + return s.startGaveUp ? "gaveUp" : "retrying"; +} + +export function PhaseBadge({ + phase, + failure = null, + autoRestarts = 0, +}: { + phase: Phase; + /** From startFailure: a Failed start between automatic retries pulses and says so. */ + failure?: StartFailure | null; + autoRestarts?: number; +}) { const { t } = useTranslation(); + const retrying = phase === "Failed" && failure === "retrying"; + const hint = + phase !== "Failed" || failure === null + ? undefined + : retrying + ? t("servers:server_failed_retrying_body", { used: autoRestarts, max: MAX_AUTO_RESTARTS }) + : t("servers:server_failed_body"); return ( - + - {t(PHASE_KEY[phase] ?? PHASE_KEY.Unknown)} + {retrying ? t("servers:phase_failed_retrying") : t(PHASE_KEY[phase] ?? PHASE_KEY.Unknown)} ); } diff --git a/panel/src/components/PowerButton.test.tsx b/panel/src/components/PowerButton.test.tsx index ceec6f0..f6a7411 100644 --- a/panel/src/components/PowerButton.test.tsx +++ b/panel/src/components/PowerButton.test.tsx @@ -103,4 +103,33 @@ describe("PowerButton", () => { expect(wake).toHaveBeenCalledOnce(); }); + + it("offers a failed server a retry, which is the wake", async () => { + wake.mockResolvedValue(undefined); + const onChanged = vi.fn(); + render(); + + expect(screen.queryByRole("button", { name: t("servers:wake") })).toBeNull(); + await userEvent.click(screen.getByRole("button", { name: t("servers:retry_start") })); + + expect(wake).toHaveBeenCalledWith("lobby"); + expect(stop).not.toHaveBeenCalled(); + expect(onChanged).toHaveBeenCalledOnce(); + }); + + it("stops a failed server without asking, and only the pressed button spins", async () => { + let resolve!: () => void; + stop.mockReturnValue(new Promise((r) => (resolve = r))); + const onChanged = vi.fn(); + render(); + + await userEvent.click(screen.getByRole("button", { name: t("servers:stop") })); + + expect(stop).toHaveBeenCalledWith("lobby"); + expect(screen.queryByText(t("servers:stop_confirm_unknown"))).toBeNull(); + expect(screen.getByRole("button", { name: t("servers:stopping") })).toHaveProperty("disabled", true); + expect(screen.getByRole("button", { name: t("servers:retry_start") })).toHaveProperty("disabled", true); + resolve(); + await vi.waitFor(() => expect(onChanged).toHaveBeenCalledOnce()); + }); }); diff --git a/panel/src/components/PowerButton.tsx b/panel/src/components/PowerButton.tsx index b02fd74..56734b9 100644 --- a/panel/src/components/PowerButton.tsx +++ b/panel/src/components/PowerButton.tsx @@ -1,5 +1,5 @@ import { useState } from "react"; -import { Loader2, Play, Square } from "lucide-react"; +import { Loader2, Play, RotateCcw, Square } from "lucide-react"; import { useTranslation } from "react-i18next"; import { Button } from "@/components/ui/button"; import { api, humanizeError } from "@/lib/api"; @@ -9,6 +9,8 @@ interface Props { name: string; /** A pod is up or on its way (Starting/Running/Stopping): offer Stop. */ live: boolean; + /** Its start Failed while meant to run: offer a retry and a stop. */ + failed?: boolean; playersOnline?: number; /** The operator cannot read the player count, so players may be online. */ playerCountUnknown?: boolean; @@ -21,10 +23,14 @@ interface Props { // PowerButton starts or stops one server. It is busy while the call runs (no // double send), shows why a call was refused (quota, cooldown, a phase that // moved on), and asks before a stop that would disconnect players: the count -// is in the question, and an unreadable count asks too. +// is in the question, and an unreadable count asks too. A server whose start +// failed gets both ways out: retry (the wake, which felis-api turns into a fresh +// start) and stop. Nobody is on a server that never came up, so that stop does +// not ask. export function PowerButton({ name, live, + failed = false, playersOnline, playerCountUnknown, onChanged, @@ -32,13 +38,13 @@ export function PowerButton({ className, }: Props) { const { t } = useTranslation("servers"); - const [busy, setBusy] = useState(false); + const [busy, setBusy] = useState<"wake" | "stop" | null>(null); const [confirming, setConfirming] = useState(false); const [error, setError] = useState(null); async function run(kind: "wake" | "stop") { if (busy) return; - setBusy(true); + setBusy(kind); setError(null); try { await (kind === "wake" ? api.wake(name) : api.stop(name)); @@ -47,20 +53,36 @@ export function PowerButton({ } catch (e) { setError(humanizeError(e)); } finally { - setBusy(false); + setBusy(null); } } const players = playersOnline ?? 0; const askFirst = players > 0 || playerCountUnknown === true; const spinner = ; + const stopButton = (variant: "destructive" | "outline", onClick: () => void) => ( + + ); let control; - if (!live) { + if (failed) { control = ( - + {stopButton("outline", () => void run("stop"))} + + ); + } else if (!live) { + control = ( + ); } else if (confirming) { @@ -71,27 +93,14 @@ export function PowerButton({ ? t("stop_confirm_players", { count: players }) : t("stop_confirm_unknown")} - - + {stopButton("destructive", () => void run("stop"))} ); } else { - control = ( - - ); + control = stopButton("destructive", () => (askFirst ? setConfirming(true) : void run("stop"))); } return ( diff --git a/panel/src/i18n/resources/en-US/servers.json b/panel/src/i18n/resources/en-US/servers.json index dbaa668..526be74 100644 --- a/panel/src/i18n/resources/en-US/servers.json +++ b/panel/src/i18n/resources/en-US/servers.json @@ -11,7 +11,8 @@ "phase_starting": "Starting", "phase_stopping": "Stopping", "phase_stopped": "Stopped", - "phase_failed": "Failed", + "phase_failed": "Failed to start", + "phase_failed_retrying": "Timed out · retrying", "phase_unknown": "Unknown", "online": "online", "console": "Console", @@ -23,12 +24,16 @@ "stopping": "Stopping…", "wake": "Wake", "waking": "Waking…", + "retry_start": "Retry start", + "retrying_start": "Retrying…", "server_asleep_title": "Server is asleep", "server_asleep_body": "Wake it to boot a pod — the live console attaches automatically the moment it starts up.", "server_shutting_down_title": "Server is shutting down", "server_shutting_down_body": "The pod is terminating, so the live console has detached. It will be asleep in a moment.", "server_failed_title": "Server failed to start", - "server_failed_body": "No pod is running, so there is nothing to stream. Wake it to retry; the console attaches once a pod comes back up.", + "server_failed_body": "Its automatic retries are spent. The console holds the log of the last attempt; fix the cause, then press Retry start.", + "server_failed_retrying_title": "Start timed out — retrying", + "server_failed_retrying_body": "The operator recreates the pod and tries again ({{used}} of {{max}} retries used). Press Retry start to go now instead.", "console_offline_title": "Console is offline", "console_offline_body": "The live console attaches automatically as soon as the server is running.", "my_servers_breadcrumb": "My servers", diff --git a/panel/src/i18n/resources/zh-CN/servers.json b/panel/src/i18n/resources/zh-CN/servers.json index 3b97d70..8b38fb1 100644 --- a/panel/src/i18n/resources/zh-CN/servers.json +++ b/panel/src/i18n/resources/zh-CN/servers.json @@ -12,6 +12,7 @@ "phase_stopping": "停止中", "phase_stopped": "已停止", "phase_failed": "启动失败", + "phase_failed_retrying": "启动超时·重试中", "phase_unknown": "未知", "online": "人在线", "console": "控制台", @@ -23,12 +24,16 @@ "stopping": "停止中…", "wake": "启动", "waking": "启动中…", + "retry_start": "重试启动", + "retrying_start": "重试中…", "server_asleep_title": "服务器已休眠", "server_asleep_body": "启动后将拉起 Pod,实时控制台会自动连接。", "server_shutting_down_title": "服务器正在关闭", "server_shutting_down_body": "Pod 正在终止,实时控制台已断开。关闭完成后将进入休眠。", "server_failed_title": "服务器启动失败", - "server_failed_body": "当前无运行中的 Pod,无法获取日志流。重新启动后控制台将自动连接。", + "server_failed_body": "自动重试已用完。控制台里是最后一次启动的日志,找到原因修好后点「重试启动」。", + "server_failed_retrying_title": "启动超时,正在自动重试", + "server_failed_retrying_body": "operator 会重建 Pod 再试(已重试 {{used}}/{{max}} 次)。不想等可以点「重试启动」立即再来。", "console_offline_title": "控制台未连接", "console_offline_body": "服务器运行后,实时控制台将自动连接。", "my_servers_breadcrumb": "我的服务器", diff --git a/panel/src/lib/openapi.gen.ts b/panel/src/lib/openapi.gen.ts index 810cc48..f8b7b36 100644 --- a/panel/src/lib/openapi.gen.ts +++ b/panel/src/lib/openapi.gen.ts @@ -167,7 +167,7 @@ export interface paths { put?: never; /** * Domain-autostart wake driven by velocity for a joining player (spec §9.1). - * @description velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the server's autostartPolicy and the per-server wake cooldown. + * @description velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the server's autostartPolicy and the per-server wake cooldown. A server whose start failed with its automatic retries spent answers 409 start_failed: a join never resets the retry budget, so velocity queues no one for it. */ post: operations["internalWake"]; delete?: never; @@ -419,7 +419,10 @@ export interface paths { }; get?: never; put?: never; - /** Wake your own server. */ + /** + * Wake your own server. + * @description On a server whose start Failed (phase Failed, desiredState Running) this is a retry: the operator recreates the pod and starts it over with a fresh automatic-restart budget. It is audited as retry_start. + */ post: operations["wake"]; delete?: never; options?: never; @@ -2398,6 +2401,13 @@ export interface components { autostartPolicy?: "ownerOnly" | "public" | "allowlist"; /** @description Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players. */ playerCountUnknown?: boolean; + /** + * Format: int32 + * @description Owned rows only. How often the operator recreated the pod of this start after it timed out. + */ + autoRestarts?: number; + /** @description Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over. */ + startGaveUp?: boolean; }; /** @description One world backup (internal/api/repo.go BackupView). backup_ref is withheld (spec §286). */ BackupView: { @@ -3148,7 +3158,7 @@ export interface operations { 401: components["responses"]["Unauthorized"]; 403: components["responses"]["Forbidden"]; 404: components["responses"]["NotFound"]; - /** @description A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started. */ + /** @description Nothing was started. maintenance_in_progress: a restore, backup or file write holds the server's world volume. start_failed: the last start failed and its automatic retries are spent (ServerInfo.startGaveUp); the server stays down until a person starts it from the panel. */ 409: { headers: { [name: string]: unknown; diff --git a/panel/src/lib/types.ts b/panel/src/lib/types.ts index 0aa0587..42ada1a 100644 --- a/panel/src/lib/types.ts +++ b/panel/src/lib/types.ts @@ -36,6 +36,10 @@ export interface MyServerView { autostartPolicy?: AutostartPolicy; /** True while the operator cannot read the player count; a stop may drop players. */ playerCountUnknown?: boolean; + /** How often the operator recreated the pod of this start after it timed out. */ + autoRestarts?: number; + /** True for a Failed server no automatic retry will bring up. */ + startGaveUp?: boolean; } /** ServerStatus is GET /servers/{name}/status (Go ServerInfo). It never carries diff --git a/panel/src/pages/ServerConsole.test.tsx b/panel/src/pages/ServerConsole.test.tsx new file mode 100644 index 0000000..342b914 --- /dev/null +++ b/panel/src/pages/ServerConsole.test.tsx @@ -0,0 +1,78 @@ +// @vitest-environment jsdom +import { describe, it, expect, vi, beforeEach } from "vitest"; +import { render, screen, within } from "@testing-library/react"; +import { MemoryRouter, Route, Routes } from "react-router-dom"; +import { ServerConsole } from "./ServerConsole"; + +const calls = vi.hoisted(() => ({ status: vi.fn(), myServers: vi.fn() })); +vi.mock("@/lib/tier", () => ({ + useTier: () => ({ + loading: false, + identity: { user_id: "admin-1", email: "admin@example.test", role: "admin" }, + isAdmin: true, + isOwner: false, + }), +})); +vi.mock("@/lib/config", async (importActual) => { + const actual = await importActual(); + return { + ...actual, + loadConfig: () => Promise.resolve({ apiBase: "/api/v1", rootDomain: "example.test", gamePort: 25570 }), + }; +}); +vi.mock("@/lib/api", async (importActual) => { + const actual = await importActual(); + return { ...actual, api: { ...actual.api, ...calls } }; +}); +// The live log is a WebSocket; here it only matters whether the page shows it. +vi.mock("@/components/LogConsole", () => ({ LogConsole: () =>
})); + +beforeEach(() => { + calls.status.mockReset(); + calls.myServers.mockReset(); + calls.myServers.mockResolvedValue([]); +}); + +function renderConsole() { + render( + + + } /> + + , + ); +} + +function status(over: Record) { + return { name: "survival", subdomain: "survival", phase: "Failed", ready: false, playersOnline: 0, playersMax: 20, ...over }; +} + +describe("ServerConsole failed start", () => { + it("shows the failed attempt's log under what the owner can do once retries are spent", async () => { + calls.status.mockResolvedValue(status({ desiredState: "Running", startGaveUp: true, autoRestarts: 3 })); + renderConsole(); + + const notice = await screen.findByRole("status", { name: "Server failed to start" }); + expect(within(notice).getByText(/automatic retries are spent/)).toBeTruthy(); + expect(await screen.findByTestId("log-stream")).toBeTruthy(); + expect(screen.getByRole("button", { name: /Retry start/ })).toBeTruthy(); + }); + + it("counts the retries used while the operator is still retrying", async () => { + calls.status.mockResolvedValue(status({ desiredState: "Running", autoRestarts: 1 })); + renderConsole(); + + const notice = await screen.findByRole("status", { name: "Start timed out — retrying" }); + expect(within(notice).getByText(/1 of 3 retries used/)).toBeTruthy(); + expect(screen.getByText("Timed out · retrying")).toBeTruthy(); + }); + + it("asks nothing of the owner once a failed server is being stopped", async () => { + calls.status.mockResolvedValue(status({ desiredState: "Stopped" })); + renderConsole(); + + expect(await screen.findByTestId("log-stream")).toBeTruthy(); + expect(screen.queryByRole("status", { name: /failed to start|timed out/ })).toBeNull(); + expect(screen.queryByRole("button", { name: /Retry start/ })).toBeNull(); + }); +}); diff --git a/panel/src/pages/ServerConsole.tsx b/panel/src/pages/ServerConsole.tsx index 2c835a8..0e05ac7 100644 --- a/panel/src/pages/ServerConsole.tsx +++ b/panel/src/pages/ServerConsole.tsx @@ -1,10 +1,10 @@ -import { useState, useRef, useCallback, useLayoutEffect, type KeyboardEvent } from "react"; +import { useState, useRef, useCallback, useId, useLayoutEffect, type KeyboardEvent } from "react"; import { Link, useParams } from "react-router-dom"; import { Terminal, Moon, Shield, ShieldAlert, HelpCircle, Loader2, Users, Archive, FolderOpen, ChevronRight, type LucideIcon } from "lucide-react"; import { useTranslation } from "react-i18next"; import { Card, CardContent } from "@/components/ui/card"; import { BackLink } from "@/components/BackLink"; -import { PhaseBadge } from "@/components/PhaseBadge"; +import { MAX_AUTO_RESTARTS, PhaseBadge, startFailure, type StartFailure } from "@/components/PhaseBadge"; import { PageHeader } from "@/components/PageHeader"; import { LogConsole } from "@/components/LogConsole"; import { Loading, ErrorState } from "@/components/States"; @@ -39,12 +39,6 @@ function useNotStreamingCopy(phase: Phase): { icon: LucideIcon; title: string; b title: t("server_shutting_down_title"), body: t("server_shutting_down_body"), }; - case "Failed": - return { - icon: ShieldAlert, - title: t("server_failed_title"), - body: t("server_failed_body"), - }; default: return { icon: HelpCircle, @@ -67,6 +61,34 @@ function NotStreaming({ phase }: { phase: Phase }) { ); } +// FailedStartNotice heads the console of a server whose start failed: whether the +// operator will try again on its own, and what the owner can do. The log below it +// is the attempt that failed, which is what they need to find the cause. +function FailedStartNotice({ failure, autoRestarts }: { failure: StartFailure; autoRestarts: number }) { + const { t } = useTranslation("servers"); + const retrying = failure === "retrying"; + const titleId = useId(); + return ( +
+ +
+

+ {retrying ? t("server_failed_retrying_title") : t("server_failed_title")} +

+

+ {retrying + ? t("server_failed_retrying_body", { used: autoRestarts, max: MAX_AUTO_RESTARTS }) + : t("server_failed_body")} +

+
+
+ ); +} + function loadHistory(name: string): string[] { try { const raw = localStorage.getItem(HISTORY_KEY(name)); @@ -206,8 +228,11 @@ export function ServerConsole() { // during BOTH Starting and Running — the operator marks Starting once the pod // is up but RCON is not yet reachable (it only flips to Running after RCON // readiness). Boot logs flow precisely in that Starting window, which is when a - // read most wants them, so the gate streams for both, not Running alone. - const streamable = data?.phase === "Running" || data?.phase === "Starting"; + // read most wants them, so the gate streams for both, not Running alone. A + // Failed start usually leaves its pod behind, and that pod's log is how the + // owner finds out why it failed, so Failed streams too. + const streamable = data?.phase === "Running" || data?.phase === "Starting" || data?.phase === "Failed"; + const failure = data ? startFailure(data) : null; return (
@@ -227,10 +252,11 @@ export function ServerConsole() { subtitle={cfg ? : undefined} actions={
- + ) : cfg ? (
+ {failure && ( + + )} { expect(survival.getByRole("button", { name: "Copy address survival.example.test:25570" })).toBeTruthy(); }); }); + +describe("ServersPage failed starts", () => { + it("offers a failed start a retry and a stop, and tells retrying from given up", async () => { + calls.fleet.mockResolvedValue([ + row("broken", { phase: "Failed", desiredState: "Running", startGaveUp: true, autoRestarts: 3, owned: true }), + row("flaky", { phase: "Failed", desiredState: "Running", autoRestarts: 1, owned: true }), + ]); + render( + + + , + ); + + const broken = await tableRow("broken"); + expect(broken.getByText("Failed to start")).toBeTruthy(); + expect(broken.getByRole("button", { name: /Retry start/ })).toBeTruthy(); + expect(broken.getByRole("button", { name: /Stop/ })).toBeTruthy(); + expect(broken.queryByRole("button", { name: /Wake/ })).toBeNull(); + + const flaky = await tableRow("flaky"); + expect(flaky.getByText("Timed out · retrying")).toBeTruthy(); + expect(flaky.getByRole("button", { name: /Retry start/ })).toBeTruthy(); + }); +}); diff --git a/panel/src/pages/servers/ServersPage.tsx b/panel/src/pages/servers/ServersPage.tsx index f74a78d..0246d30 100644 --- a/panel/src/pages/servers/ServersPage.tsx +++ b/panel/src/pages/servers/ServersPage.tsx @@ -30,7 +30,7 @@ import { SelectContent, SelectItem, } from "@/components/ui/select"; -import { PhaseBadge, PHASE_KEY, PHASE_COLOR } from "@/components/PhaseBadge"; +import { PhaseBadge, PHASE_KEY, PHASE_COLOR, startFailure } from "@/components/PhaseBadge"; import { PowerButton } from "@/components/PowerButton"; import { Loading, ErrorState, EmptyState } from "@/components/States"; import { Pagination } from "@/components/Pagination"; @@ -81,6 +81,8 @@ interface UnifiedServer { playersOnline: number; playersMax: number; playerCountUnknown?: boolean; + autoRestarts?: number; + startGaveUp?: boolean; owner?: string; endpointAddress?: string | null; claimable?: boolean; @@ -133,6 +135,8 @@ export function ServersPage() { playersOnline: s.playersOnline, playersMax: s.playersMax, playerCountUnknown: s.playerCountUnknown, + autoRestarts: s.autoRestarts, + startGaveUp: s.startGaveUp, owner: s.owner, endpointAddress: s.endpointAddress, claimable: s.claimable, @@ -152,6 +156,8 @@ export function ServersPage() { playersOnline: s.playersOnline, playersMax: s.playersMax, playerCountUnknown: s.playerCountUnknown, + autoRestarts: s.autoRestarts, + startGaveUp: s.startGaveUp, owner: s.owned ? t("servers:owned_filter_mine") || "me" : undefined, claimable: s.claimable, owned: s.owned, @@ -509,6 +515,7 @@ function ServerActions({
- +
{address && } @@ -669,7 +676,7 @@ function ServerMobileCard({
- +
{/* Its own line: the join address is what a player came for, so it gets the card's full width rather than sharing it with the badge. */} diff --git a/plugins/velocity/src/main/java/best/lolicon/felis/velocity/MotdResponder.java b/plugins/velocity/src/main/java/best/lolicon/felis/velocity/MotdResponder.java index dc30ca5..e2732c2 100644 --- a/plugins/velocity/src/main/java/best/lolicon/felis/velocity/MotdResponder.java +++ b/plugins/velocity/src/main/java/best/lolicon/felis/velocity/MotdResponder.java @@ -20,12 +20,14 @@ import java.util.Optional; * backend or the wake lever. * *

This is the read-only, phase-aware subset of the responsibility: the MOTD is - * synthesized from the server's lifecycle (online / starting / sleeping) and its - * cached player counts. Mirroring each backend's own MOTD string (by - * pinging ready servers in the background and caching the result) is a richer + * synthesized from the server's lifecycle (online / starting / start failed / + * sleeping) and its cached player counts. Mirroring each backend's own MOTD + * string (by pinging ready servers in the background and caching the result) is a richer * variant deferred to a later slice; nothing here ever pings a sleeping backend. */ public final class MotdResponder { + private static final String PHASE_FAILED = "Failed"; + private final ServerRegistry registry; MotdResponder(ServerRegistry registry) { @@ -55,21 +57,33 @@ public final class MotdResponder { } // The server-list ping carries no client locale, so the MOTD status uses the - // both-languages-in-one-line pattern the modded /link clients share. - private static String statusLine(ServerView v) { + // both-languages-in-one-line pattern the modded /link clients share. A start + // that failed still holds desiredState Running, so it is read first: joining a + // server whose retries are spent wakes nothing, and one between retries is + // waiting out a backoff, which a "starting…" line hid behind a queue that ran out. + static String statusLine(ServerView v) { if (v.ready()) { return "在线 / online"; } + if (v.startGaveUp()) { + return "启动失败,等服主处理 / failed to start — the owner has to restart it"; + } + if (PHASE_FAILED.equals(v.phase())) { + return "启动超时,稍后自动重试 / start timed out — retrying shortly"; + } if ("Running".equals(v.desiredState())) { return "启动中… / starting…"; } return "休眠中,加入即唤醒 / sleeping — join to wake"; } - private static NamedTextColor statusColor(ServerView v) { + static NamedTextColor statusColor(ServerView v) { if (v.ready()) { return NamedTextColor.GREEN; } + if (v.startGaveUp()) { + return NamedTextColor.RED; + } if ("Running".equals(v.desiredState())) { return NamedTextColor.YELLOW; } diff --git a/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java b/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java index 0a6f90f..21ae037 100644 --- a/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java +++ b/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java @@ -505,11 +505,7 @@ public final class WaitingRouter { // stopped the server — it says so and returns false. private boolean stillComing(Player player, boolean zh, Waiter w, ServerView status, long now) { if (status.startGaveUp()) { - player.sendMessage(Component.text( - zh ? "「" + w.serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。" - : "« " + w.serverName + " » failed to start and its automatic retries are spent. " - + "The owner can check its log in the panel and start it again.", - NamedTextColor.RED)); + tellStartFailed(player, zh, w.serverName); return false; } if (DESIRED_STOPPED.equalsIgnoreCase(status.desiredState())) { @@ -655,6 +651,12 @@ public final class WaitingRouter { NamedTextColor.YELLOW)); return; } + if ("start_failed".equals(e.errorCode())) { + // The last start failed with its retries spent. The join does + // not buy another round of them, so there is nothing to wait for. + tellStartFailed(player, zh, serverName); + return; + } logWakeFailure(player, serverName, zh, e); return; case 429: @@ -688,6 +690,16 @@ public final class WaitingRouter { waiting.put(id, new Waiter(serverName, clock.getAsLong(), fromMenu)); } + // The server's start failed and its automatic retries are spent: nothing more is + // coming until a person starts it again from the panel. + private static void tellStartFailed(Player player, boolean zh, String serverName) { + player.sendMessage(Component.text( + zh ? "「" + serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。" + : "« " + serverName + " » failed to start and its automatic retries are spent. " + + "The owner can check its log in the panel and start it again.", + NamedTextColor.RED)); + } + private void logWakeFailure(Player player, String serverName, boolean zh, LinkException e) { log.warn("Felis: wake {} failed (status={}): {}", serverName, e.statusCode(), e.getMessage()); player.sendMessage(Component.text( diff --git a/plugins/velocity/test/best/lolicon/felis/velocity/ServerRegistryTest.java b/plugins/velocity/test/best/lolicon/felis/velocity/ServerRegistryTest.java index befe673..6f873a1 100644 --- a/plugins/velocity/test/best/lolicon/felis/velocity/ServerRegistryTest.java +++ b/plugins/velocity/test/best/lolicon/felis/velocity/ServerRegistryTest.java @@ -2,6 +2,8 @@ package best.lolicon.felis.velocity; import best.lolicon.felis.link.ServerView; +import net.kyori.adventure.text.format.NamedTextColor; + import java.net.InetSocketAddress; import java.util.List; import java.util.Optional; @@ -139,6 +141,24 @@ public final class ServerRegistryTest { assertAddr("a non-numeric port is not a port", "backend:x", 25565, ServerRegistry.parseAddress("backend:x")); + // The server-list MOTD reads a failed start before desiredState Running, + // which a failed start still holds. + assertEq("motd: up", "在线 / online", MotdResponder.statusLine(up("a", "a", "10.43.0.1:25565"))); + assertEq("motd: starting", "启动中… / starting…", MotdResponder.statusLine( + new ServerView("a", "a", "Starting", false, "ownerOnly", "Running", "fallback", "login", 0, 0))); + ServerView backoff = new ServerView("a", "a", "Failed", false, "ownerOnly", "Running", + "fallback", "login", 0, 0, false, 1, false); + assertEq("motd: between retries", "启动超时,稍后自动重试 / start timed out — retrying shortly", + MotdResponder.statusLine(backoff)); + assertEq("motd: between retries is yellow", NamedTextColor.YELLOW, MotdResponder.statusColor(backoff)); + ServerView gaveUp = new ServerView("a", "a", "Failed", false, "ownerOnly", "Running", + "fallback", "login", 0, 0, false, 3, true); + assertEq("motd: retries spent", "启动失败,等服主处理 / failed to start — the owner has to restart it", + MotdResponder.statusLine(gaveUp)); + assertEq("motd: retries spent is red", NamedTextColor.RED, MotdResponder.statusColor(gaveUp)); + assertEq("motd: sleeping", "休眠中,加入即唤醒 / sleeping — join to wake", + MotdResponder.statusLine(down("a", "a"))); + System.out.println("ServerRegistryTest OK (" + checks + " checks)"); } diff --git a/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java b/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java index 7f17a77..88a3b68 100644 --- a/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java +++ b/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java @@ -219,6 +219,7 @@ public final class WaitingRouterTest { String[][] cases = { {"403 forbidden", "You're not allowed to start « gamma »", null}, {"409 maintenance_in_progress", "« gamma » is under maintenance", null}, + {"409 start_failed", "« gamma » failed to start and its automatic retries are spent", null}, {"409 conflict", "Couldn't start « gamma » right now", "wake gamma failed (status=409)"}, {"503 at_capacity", "The cluster is at capacity right now", null}, {"503 unavailable", "Couldn't start « gamma » right now", "wake gamma failed (status=503)"},