fix(felis): 启动失败的服可在面板重试或停止,玩家入服时直接告知启动失败

This commit is contained in:
Lemon-miaow committed 2026-09-26 22:08:29 +08:00
1 parent c1025274e1
commit 22d1e834ad
28 files changed
+775 -75

No files matched your search

+19 -2
View File
@@ -566,6 +566,13 @@ components:
playerCountUnknown:
type: boolean
description: Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players.
autoRestarts:
type: integer
format: int32
description: Owned rows only. How often the operator recreated the pod of this start after it timed out.
startGaveUp:
type: boolean
description: Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over.
BackupView:
type: object
@@ -1165,7 +1172,9 @@ paths:
description: >-
velocity holds no web Principal, so it drives the wake lever with its
service token, identifying the player by online-mode UUID. Gated by the
server's autostartPolicy and the per-server wake cooldown.
server's autostartPolicy and the per-server wake cooldown. A server whose
start failed with its automatic retries spent answers 409 start_failed: a
join never resets the retry budget, so velocity queues no one for it.
x-felis-face: [internal]
x-felis-tier: service
x-felis-callers: [velocity]
@@ -1203,7 +1212,11 @@ paths:
'404':
$ref: '#/components/responses/NotFound'
'409':
description: A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started.
description: >-
Nothing was started. maintenance_in_progress: a restore, backup or
file write holds the server's world volume. start_failed: the last
start failed and its automatic retries are spent (ServerInfo.startGaveUp);
the server stays down until a person starts it from the panel.
content:
application/json:
schema: { $ref: '#/components/schemas/Error' }
@@ -1767,6 +1780,10 @@ paths:
tags: [servers]
operationId: wake
summary: Wake your own server.
description: >-
On a server whose start Failed (phase Failed, desiredState Running) this
is a retry: the operator recreates the pod and starts it over with a
fresh automatic-restart budget. It is audited as retry_start.
x-felis-face: [external]
x-felis-tier: app
security: [{ sessionCookie: [] }]
+18 -4
View File
@@ -1748,6 +1748,7 @@ type fakeCluster struct {
wakeErr map[string]error
acquired []string // "name:kind" per admitted AcquireMaintenance
released []string // names per ReleaseMaintenance
retried []string // names per RetryStart
}
func newFakeCluster() *fakeCluster {
@@ -1794,6 +1795,16 @@ func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.De
c.desired[n] = s
return nil
}
// RetryStart records the retry on top of the plain start, so a test tells a
// Failed server's retry apart from an ordinary wake.
func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
if err := c.SetDesiredState(ctx, n, v1alpha1.DesiredRunning); err != nil {
return err
}
c.retried = append(c.retried, n)
return nil
}
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
if err := c.maintErr[n]; err != nil {
return err
@@ -2092,9 +2103,11 @@ func TestMyServersJoinsLiveState(t *testing.T) {
cl := newFakeCluster()
cl.list = []ServerInfo{
{Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running",
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true},
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true,
AutoRestarts: 3, StartGaveUp: true},
{Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running",
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true},
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true,
AutoRestarts: 3, StartGaveUp: true},
}
api := newTestAPI(repo, cl)
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
@@ -2122,7 +2135,8 @@ func TestMyServersJoinsLiveState(t *testing.T) {
mine, open := got["mine"], got["open"]
for k, want := range map[string]any{"displayName": "My World", "phase": "Running",
"desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true,
"playersOnline": float64(2), "playersMax": float64(20)} {
"playersOnline": float64(2), "playersMax": float64(20),
"autoRestarts": float64(3), "startGaveUp": true} {
if mine[k] != want {
t.Errorf("own row %s = %v, want %v", k, mine[k], want)
}
@@ -2130,7 +2144,7 @@ func TestMyServersJoinsLiveState(t *testing.T) {
if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) {
t.Errorf("claimable row public fields = %v", open)
}
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown"} {
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown", "autoRestarts", "startGaveUp"} {
if _, ok := open[k]; ok {
t.Errorf("claimable row carries owner detail %s = %v", k, open[k])
}
+4
View File
@@ -117,6 +117,10 @@ type Cluster interface {
// *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore,
// backup or file write holds the world volume.
SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
// RetryStart is SetDesiredState(Running) for a server whose start Failed: it
// also asks the operator to start it over with a fresh auto-restart budget
// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
RetryStart(ctx context.Context, name string) error
// AcquireMaintenance admits one world-volume operation (internal/maintenance
// kind): ErrNotStopped unless the server is fully stopped, a
// *MaintenanceBusyError while another operation holds the volume. The check
+12
View File
@@ -152,6 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
writeError(w, r, err)
return
}
// A start whose automatic restarts are spent (or that can never succeed as
// configured) stays down until a person looks at it. The 202 this used to
// return queued the player for a server nothing was starting. The join leaves
// the restart budget alone, or every player who tried to join would buy
// another three crash loops; velocity tells them and queues no one. A Failed
// server still inside its backoff is not this: its next attempt is coming, so
// it gets the 202 and the player waits for it.
if info.StartGaveUp && info.DesiredState == string(v1alpha1.DesiredRunning) {
writeError(w, r, newError(http.StatusConflict, "start_failed",
"the server failed to start and its automatic retries are spent; its owner can retry from the panel"))
return
}
if !a.limiter().allowed(name, a.WakeCooldown) {
writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly"))
return
+131
View File
@@ -0,0 +1,131 @@
package api
import (
"net/http"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
)
// A start that Failed holds desiredState Running already, so the wake that
// re-writes Running changes nothing. These pin the two ways out: a person's
// wake in the panel starts it over (RetryStart), and a player's join onto one
// whose retries are spent is told so (409 start_failed) and never resets them.
func failedServer(gaveUp bool) *ServerInfo {
return &ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseFailed),
AutoRestarts: v1alpha1.MaxAutoRestarts, StartGaveUp: gaveUp}
}
func TestWakeOfFailedServerRetriesStart(t *testing.T) {
mk := func(info *ServerInfo) (*API, *fakeCluster, *fakeRepo) {
repo := newFakeRepo()
cl := newFakeCluster()
cl.byName["survival"] = info
api := newTestAPI(repo, cl)
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
return api, cl, repo
}
wake := func(api *API) int {
return do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil).Code
}
t.Run("retries spent: the wake starts it over", func(t *testing.T) {
api, cl, repo := mk(failedServer(true))
if code := wake(api); code != http.StatusAccepted {
t.Fatalf("code = %d, want 202", code)
}
if len(cl.retried) != 1 || cl.retried[0] != "survival" {
t.Fatalf("retried = %v, want [survival]", cl.retried)
}
if len(repo.audits) != 1 || repo.audits[0].Action != "retry_start" {
t.Fatalf("audits = %+v, want one retry_start", repo.audits)
}
})
t.Run("inside the backoff: a person asking skips the wait", func(t *testing.T) {
api, cl, _ := mk(failedServer(false))
if code := wake(api); code != http.StatusAccepted || len(cl.retried) != 1 {
t.Fatalf("code = %d retried = %v, want 202 and one retry", code, cl.retried)
}
})
t.Run("stopped: a plain wake", func(t *testing.T) {
api, cl, repo := mk(&ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)})
if code := wake(api); code != http.StatusAccepted {
t.Fatalf("code = %d, want 202", code)
}
if len(cl.retried) != 0 || cl.desired["survival"] != v1alpha1.DesiredRunning {
t.Fatalf("retried = %v desired = %q, want a plain start", cl.retried, cl.desired["survival"])
}
if len(repo.audits) != 1 || repo.audits[0].Action != "wake" {
t.Fatalf("audits = %+v, want one wake", repo.audits)
}
})
t.Run("maintenance holds the world: refused, nothing retried", func(t *testing.T) {
api, cl, _ := mk(failedServer(true))
cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore}
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil)
if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("retried = %v, want none", cl.retried)
}
})
}
func TestInternalWakeOfFailedServer(t *testing.T) {
body := `{"mc_uuid":"` + wakeUUID + `"}`
t.Run("retries spent: 409 start_failed, budget and cooldown untouched", func(t *testing.T) {
api, cl := newInternalWakeAPI("public")
cl.byName["survival"] = failedServer(true)
api.WakeCooldown = time.Minute
repo := api.Repo.(*fakeRepo)
w := internalWake(api, body)
if w.Code != http.StatusConflict || decodeErr(t, w) != "start_failed" {
t.Fatalf("code = %d body %s, want 409 start_failed", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
}
if len(repo.audits) != 0 {
t.Fatalf("nothing was woken, so nothing is audited: %+v", repo.audits)
}
// The owner fixes it and starts it from the panel; the next join is not
// held back by a cooldown the refused one never spent.
cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseStarting)}
if w := internalWake(api, body); w.Code != http.StatusAccepted {
t.Fatalf("join after the fix: code = %d body %s, want 202", w.Code, w.Body.String())
}
})
t.Run("inside the backoff: 202, the player waits for the next attempt", func(t *testing.T) {
api, cl := newInternalWakeAPI("public")
cl.byName["survival"] = failedServer(false)
if w := internalWake(api, body); w.Code != http.StatusAccepted {
t.Fatalf("code = %d body %s, want 202", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
}
})
t.Run("forbidden player: 403 comes first", func(t *testing.T) {
api, cl := newInternalWakeAPI("ownerOnly")
info := failedServer(true)
info.AutostartPolicy = "ownerOnly"
cl.byName["survival"] = info
if w := internalWake(api, body); w.Code != http.StatusForbidden {
t.Fatalf("code = %d, want 403", w.Code)
}
})
}
+20 -4
View File
@@ -56,9 +56,23 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
return
}
// Refused with 409 maintenance_in_progress while a restore, backup or file
// write holds the world volume: starting on a half-written world corrupts it.
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
// A start that Failed already holds desiredState Running, so writing Running
// again changes nothing and the server stayed dead once its automatic restarts
// were spent. A person pressing start on it means "try again": RetryStart
// makes the operator start it over with a fresh restart budget. Only this face
// does that; a player's join never resets the budget (handleInternalWake).
//
// Either write is refused with 409 maintenance_in_progress while a restore,
// backup or file write holds the world volume: starting on a half-written
// world corrupts it.
action := "wake"
if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) {
action = "retry_start"
err = a.Cluster.RetryStart(r.Context(), name)
} else {
err = a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning)
}
if err != nil {
a.writeLookupError(w, r, err)
return
}
@@ -67,7 +81,7 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
// held at capacity should retry the instant a slot frees, not wait out a
// cooldown their refused wake never earned).
a.limiter().record(name)
a.audit(r, "wake", name)
a.audit(r, action, name)
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
}
@@ -267,6 +281,8 @@ func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) {
v.DesiredState = info.DesiredState
v.AutostartPolicy = info.AutostartPolicy
v.PlayerCountUnknown = info.PlayerCountUnknown
v.AutoRestarts = info.AutoRestarts
v.StartGaveUp = info.StartGaveUp
}
}
}
+17
View File
@@ -246,6 +246,17 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a
// against, the same as AcquireMaintenance's: whichever of a racing wake and
// admission writes second gets a conflict, re-reads, and sees the other.
func (k *K8sCluster) start(ctx context.Context, name string) error {
return k.startWith(ctx, name, false)
}
// RetryStart starts a Failed server over: the same guarded write as start, plus
// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and
// recreates the pod.
func (k *K8sCluster) RetryStart(ctx context.Context, name string) error {
return k.startWith(ctx, name, true)
}
func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer
if err := k.getServer(ctx, name, &ms); err != nil {
@@ -263,6 +274,12 @@ func (k *K8sCluster) start(ctx context.Context, name string) error {
// A lock still on the object here no longer holds anything (Holder said
// so): drop it in the same write.
delete(ms.Annotations, maintenance.Annotation)
if retryFailed {
if ms.Annotations == nil {
ms.Annotations = map[string]string{}
}
ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339)
}
return k.c.Patch(ctx, &ms, patch)
})
}
@@ -214,6 +214,39 @@ func TestStartRespectsMaintenance(t *testing.T) {
}
})
t.Run("retry start -> Running plus the retry request, a plain start leaves none", func(t *testing.T) {
ms := stoppedServer()
ms.Spec.DesiredState = v1alpha1.DesiredRunning
ms.Status.Phase = v1alpha1.PhaseFailed
k, c := lockCluster(t, ms)
if err := k.SetDesiredState(ctx, "survival", v1alpha1.DesiredRunning); err != nil {
t.Fatalf("start: %v", err)
}
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
t.Fatalf("a plain start asked for a retry: %v", ann)
}
if err := k.RetryStart(ctx, "survival"); err != nil {
t.Fatalf("retry: %v", err)
}
ann, desired := annotations(t, c)
if desired != v1alpha1.DesiredRunning || ann[v1alpha1.AnnotationStartRetry] != lockNow.Format(time.RFC3339) {
t.Fatalf("desiredState = %q annotations = %v, want Running and the retry stamped %s",
desired, ann, lockNow.Format(time.RFC3339))
}
})
t.Run("retry start under a fresh lock -> busy, no retry request", func(t *testing.T) {
ms := stoppedServer()
ms.Annotations = map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lockNow)}
k, c := lockCluster(t, ms)
if err := k.RetryStart(ctx, "survival"); !errors.Is(err, ErrMaintenanceInProgress) {
t.Fatalf("err = %v, want maintenance in progress", err)
}
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
t.Fatalf("a refused retry left its request: %v", ann)
}
})
t.Run("stop ignores the lock", func(t *testing.T) {
ms := stoppedServer()
ms.Spec.DesiredState = v1alpha1.DesiredRunning
+5 -3
View File
@@ -21,9 +21,9 @@ type ServerRecord struct {
// may auto-start, or may claim. Everything after Claimable is NOT stored in
// Postgres — handleMyServers joins it best-effort from the CRD status
// (Cluster.ListServers) at read time, so a cluster hiccup renders 0/0 and the
// cached phase, never a 500. DesiredState, AutostartPolicy and
// PlayerCountUnknown are owner detail and stay empty on rows the caller does
// not own, the same split publicServerInfo makes on the status route.
// cached phase, never a 500. DesiredState, AutostartPolicy, PlayerCountUnknown,
// AutoRestarts and StartGaveUp are owner detail and stay empty on rows the caller
// does not own, the same split publicServerInfo makes on the status route.
type MyServerView struct {
Name string `json:"name"`
Subdomain string `json:"subdomain"`
@@ -36,6 +36,8 @@ type MyServerView struct {
DesiredState string `json:"desiredState,omitempty"`
AutostartPolicy string `json:"autostartPolicy,omitempty"`
PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"`
AutoRestarts int32 `json:"autoRestarts,omitempty"`
StartGaveUp bool `json:"startGaveUp,omitempty"`
}
// ServerOwnership is one live server's claim state as the fleet read joins it.
@@ -31,6 +31,13 @@ const (
// server; and felis-api never writes it, so marking a server takes kubectl on
// the cluster, which fits a switch that drops the forwarding secret.
LabelForwarding = GroupName + "/forwarding"
// AnnotationStartRetry is felis-api asking the operator to start a Failed
// server over: its value is the request time (RFC 3339). Re-patching
// desiredState to the Running it already holds changes nothing the operator
// can see, so a person pressing "retry" in the panel had no way through once
// the automatic restarts were spent. The operator takes the request once —
// fresh restart budget, new start anchor, pod recreated — and removes it.
AnnotationStartRetry = GroupName + "/start-retry"
)
// ForwardingLegacy is the LabelForwarding value that selects legacy forwarding.
+33
View File
@@ -195,6 +195,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu
desired = v1alpha1.DesiredStopped
}
prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts
if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok {
if err := r.takeStartRetry(ctx, &server, desired); err != nil {
return ctrl.Result{}, err
}
}
var res ctrl.Result
var err error
if desired == v1alpha1.DesiredStopped {
@@ -233,6 +238,33 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine
r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg))
}
// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed
// and meant to run starts over as if freshly woken — pod recreated, restart
// budget back to zero, a new start anchor — and in any other state the request
// is stale and only removed. The fresh status is written before the request is
// removed, so a pass that fails in between leaves the request to be taken again
// (at the cost of one more pod recreate), never a removed request with the old
// spent budget still in place.
func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error {
if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed {
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}}
if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil {
return err
}
server.Status.AutoRestarts = 0
server.Status.StartRequestedAt = nil
r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod")
if err := r.patchStatus(ctx, server); err != nil {
return err
}
log.FromContext(ctx).Info("start retried")
r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod")
}
patch := client.MergeFrom(server.DeepCopy())
delete(server.Annotations, v1alpha1.AnnotationStartRetry)
return r.Patch(ctx, server, patch)
}
func (r *Reconciler) event(server *v1alpha1.MinecraftServer, eventType, reason, msg string) {
if r.Recorder != nil {
r.Recorder.Event(server, eventType, reason, msg)
@@ -807,6 +839,7 @@ func (r *Reconciler) markFailed(server *v1alpha1.MinecraftServer, reason, msg st
server.Status.Phase = v1alpha1.PhaseFailed
server.Status.Ready = false
server.Status.ObservedGeneration = server.Generation
server.Status.LiveMotd = server.Spec.Motd.Failed
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
r.setCondition(server, v1alpha1.ConditionProvisioned, metav1.ConditionFalse, reason, msg)
}
+119
View File
@@ -0,0 +1,119 @@
package operator_test
import (
"context"
"testing"
"time"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
)
// requestStartRetry is felis-api's RetryStart as the operator sees it.
func requestStartRetry(t *testing.T, c client.Client) {
t.Helper()
s := getServer(t, c, "survival")
patch := client.MergeFrom(s.DeepCopy())
if s.Annotations == nil {
s.Annotations = map[string]string{}
}
s.Annotations[v1alpha1.AnnotationStartRetry] = "2026-07-01T12:00:00Z"
if err := c.Patch(context.Background(), s, patch); err != nil {
t.Fatalf("annotate: %v", err)
}
}
func podPresent(t *testing.T, c client.Client) bool {
t.Helper()
err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival-0"}, &corev1.Pod{})
if err != nil && !apierrors.IsNotFound(err) {
t.Fatalf("get pod: %v", err)
}
return err == nil
}
// Once the automatic restarts are spent, a retry request starts the server over
// with the whole budget back: the pod is recreated, the start re-anchored, and
// the next timeout is retried automatically again.
func TestStartRetryRevivesAServerThatGaveUp(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.TimeoutSeconds = 30
srv.Spec.Motd.Failed = "broken"
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
at := func(offset time.Duration) *v1alpha1.MinecraftServer {
clock = base.Add(offset)
reconcile(t, r, "survival")
return getServer(t, c, "survival")
}
recreatePod := func() {
if err := c.Create(context.Background(), gamePod()); err != nil {
t.Fatal(err)
}
}
// Spend the three automatic restarts (autorestart_test.go pins the schedule).
at(0)
for _, due := range []time.Duration{90 * time.Second, 240 * time.Second, 510 * time.Second} {
at(due)
recreatePod()
}
s := at(time.Hour)
if !v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("setup: not given up: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts)
}
if s.Status.LiveMotd != "broken" {
t.Fatalf("Failed advertises liveMotd %q, want the failed MOTD", s.Status.LiveMotd)
}
requestStartRetry(t, c)
s = at(2 * time.Hour)
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
t.Fatal("the retry request was left on the server")
}
if podPresent(t, c) {
t.Fatal("the retry kept the failed pod")
}
if s.Status.Phase != v1alpha1.PhaseStarting || s.Status.AutoRestarts != 0 || v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("after retry: phase=%s restarts=%d gaveUp=%v", s.Status.Phase, s.Status.AutoRestarts, v1alpha1.StartGaveUp(&s.Status))
}
if s.Status.StartRequestedAt == nil || !s.Status.StartRequestedAt.Time.Equal(base.Add(2*time.Hour)) {
t.Fatalf("start not re-anchored at the retry: %v", s.Status.StartRequestedAt)
}
recreatePod()
// The budget is really back: this start times out and is retried on its own.
if s := at(2*time.Hour + 89*time.Second); s.Status.Phase != v1alpha1.PhaseFailed || v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("retried start timing out: phase=%s gaveUp=%v, want Failed with retries left", s.Status.Phase, v1alpha1.StartGaveUp(&s.Status))
}
if s := at(2*time.Hour + 90*time.Second); s.Status.AutoRestarts != 1 || podPresent(t, c) {
t.Fatalf("retried start: restarts=%d pod=%v, want the first automatic restart", s.Status.AutoRestarts, podPresent(t, c))
}
}
// A request that finds the server anywhere but Failed is stale (the start
// already recovered, or someone stopped it): it is removed and nothing else.
func TestStaleStartRetryIsOnlyRemoved(t *testing.T) {
srv := runningServer()
srv.Annotations = map[string]string{v1alpha1.AnnotationStartRetry: "2026-07-01T12:00:00Z"}
anchor := metav1.NewTime(fixedNow().Add(-10 * time.Second))
srv.Status = v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting, AutoRestarts: 2, StartRequestedAt: &anchor}
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
reconcile(t, r, "survival")
s := getServer(t, c, "survival")
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
t.Fatal("the stale retry request was left on the server")
}
if !podPresent(t, c) || s.Status.AutoRestarts != 2 || !s.Status.StartRequestedAt.Equal(&anchor) {
t.Fatalf("a stale request touched the start: pod=%v restarts=%d anchor=%v",
podPresent(t, c), s.Status.AutoRestarts, s.Status.StartRequestedAt)
}
}
+13 -1
View File
@@ -1,5 +1,5 @@
import { describe, it, expect } from "vitest";
import { PHASE_COLOR, phaseColor, phaseVariant } from "@/components/PhaseBadge";
import { PHASE_COLOR, phaseColor, phaseVariant, startFailure } from "@/components/PhaseBadge";
import type { Phase } from "@/lib/types";
// The closed lifecycle set felis-api can emit — the Go source of truth is
@@ -47,3 +47,15 @@ describe("unmodelled phase fallback", () => {
expect(phaseVariant(future)).toBe(phaseVariant("Unknown"));
});
});
describe("start failure", () => {
// A Failed phase only reads as a failed start while the server still wants to
// run; a stopped Failed server needs nothing from its owner.
it("reads a failed start only while the server wants to run", () => {
expect(startFailure({ phase: "Running", desiredState: "Running" })).toBeNull();
expect(startFailure({ phase: "Failed", desiredState: "Stopped" })).toBeNull();
expect(startFailure({ phase: "Failed" })).toBeNull();
expect(startFailure({ phase: "Failed", desiredState: "Running" })).toBe("retrying");
expect(startFailure({ phase: "Failed", desiredState: "Running", startGaveUp: true })).toBe("gaveUp");
});
});
+40 -4
View File
@@ -52,18 +52,54 @@ export function phaseVariant(phase: Phase): BadgeVariant {
return VARIANT[phase] ?? VARIANT.Unknown;
}
export function PhaseBadge({ phase }: { phase: Phase }) {
/** MAX_AUTO_RESTARTS mirrors the operator's v1alpha1.MaxAutoRestarts: how often it
* recreates the pod of a start that timed out before leaving it Failed. */
export const MAX_AUTO_RESTARTS = 3;
export type StartFailure = "retrying" | "gaveUp";
/** startFailure reads how a Failed start stands from the owner detail. "retrying"
* means the operator's next automatic attempt is coming; "gaveUp" means nothing
* will start it again until a person does. A Failed server meant to stop, or a
* public view without the owner detail, reads as null: there is no retry to
* offer there, and the badge stays the plain Failed one. */
export function startFailure(s: {
phase?: Phase;
desiredState?: string;
startGaveUp?: boolean;
}): StartFailure | null {
if (s.phase !== "Failed" || s.desiredState !== "Running") return null;
return s.startGaveUp ? "gaveUp" : "retrying";
}
export function PhaseBadge({
phase,
failure = null,
autoRestarts = 0,
}: {
phase: Phase;
/** From startFailure: a Failed start between automatic retries pulses and says so. */
failure?: StartFailure | null;
autoRestarts?: number;
}) {
const { t } = useTranslation();
const retrying = phase === "Failed" && failure === "retrying";
const hint =
phase !== "Failed" || failure === null
? undefined
: retrying
? t("servers:server_failed_retrying_body", { used: autoRestarts, max: MAX_AUTO_RESTARTS })
: t("servers:server_failed_body");
return (
<Badge variant={phaseVariant(phase)} className="gap-1.5 whitespace-nowrap">
<Badge variant={phaseVariant(phase)} className="gap-1.5 whitespace-nowrap" title={hint}>
<span
className={cn(
"h-1.5 w-1.5 rounded-full",
TRANSIENT.has(phase) && "animate-pulse",
(TRANSIENT.has(phase) || retrying) && "animate-pulse",
)}
style={{ backgroundColor: phaseColor(phase) }}
/>
{t(PHASE_KEY[phase] ?? PHASE_KEY.Unknown)}
{retrying ? t("servers:phase_failed_retrying") : t(PHASE_KEY[phase] ?? PHASE_KEY.Unknown)}
</Badge>
);
}
+29
View File
@@ -103,4 +103,33 @@ describe("PowerButton", () => {
expect(wake).toHaveBeenCalledOnce();
});
it("offers a failed server a retry, which is the wake", async () => {
wake.mockResolvedValue(undefined);
const onChanged = vi.fn();
render(<PowerButton name="lobby" live={false} failed onChanged={onChanged} />);
expect(screen.queryByRole("button", { name: t("servers:wake") })).toBeNull();
await userEvent.click(screen.getByRole("button", { name: t("servers:retry_start") }));
expect(wake).toHaveBeenCalledWith("lobby");
expect(stop).not.toHaveBeenCalled();
expect(onChanged).toHaveBeenCalledOnce();
});
it("stops a failed server without asking, and only the pressed button spins", async () => {
let resolve!: () => void;
stop.mockReturnValue(new Promise<void>((r) => (resolve = r)));
const onChanged = vi.fn();
render(<PowerButton name="lobby" live={false} failed playerCountUnknown onChanged={onChanged} />);
await userEvent.click(screen.getByRole("button", { name: t("servers:stop") }));
expect(stop).toHaveBeenCalledWith("lobby");
expect(screen.queryByText(t("servers:stop_confirm_unknown"))).toBeNull();
expect(screen.getByRole("button", { name: t("servers:stopping") })).toHaveProperty("disabled", true);
expect(screen.getByRole("button", { name: t("servers:retry_start") })).toHaveProperty("disabled", true);
resolve();
await vi.waitFor(() => expect(onChanged).toHaveBeenCalledOnce());
});
});
+34 -25
View File
@@ -1,5 +1,5 @@
import { useState } from "react";
import { Loader2, Play, Square } from "lucide-react";
import { Loader2, Play, RotateCcw, Square } from "lucide-react";
import { useTranslation } from "react-i18next";
import { Button } from "@/components/ui/button";
import { api, humanizeError } from "@/lib/api";
@@ -9,6 +9,8 @@ interface Props {
name: string;
/** A pod is up or on its way (Starting/Running/Stopping): offer Stop. */
live: boolean;
/** Its start Failed while meant to run: offer a retry and a stop. */
failed?: boolean;
playersOnline?: number;
/** The operator cannot read the player count, so players may be online. */
playerCountUnknown?: boolean;
@@ -21,10 +23,14 @@ interface Props {
// PowerButton starts or stops one server. It is busy while the call runs (no
// double send), shows why a call was refused (quota, cooldown, a phase that
// moved on), and asks before a stop that would disconnect players: the count
// is in the question, and an unreadable count asks too.
// is in the question, and an unreadable count asks too. A server whose start
// failed gets both ways out: retry (the wake, which felis-api turns into a fresh
// start) and stop. Nobody is on a server that never came up, so that stop does
// not ask.
export function PowerButton({
name,
live,
failed = false,
playersOnline,
playerCountUnknown,
onChanged,
@@ -32,13 +38,13 @@ export function PowerButton({
className,
}: Props) {
const { t } = useTranslation("servers");
const [busy, setBusy] = useState(false);
const [busy, setBusy] = useState<"wake" | "stop" | null>(null);
const [confirming, setConfirming] = useState(false);
const [error, setError] = useState<string | null>(null);
async function run(kind: "wake" | "stop") {
if (busy) return;
setBusy(true);
setBusy(kind);
setError(null);
try {
await (kind === "wake" ? api.wake(name) : api.stop(name));
@@ -47,20 +53,36 @@ export function PowerButton({
} catch (e) {
setError(humanizeError(e));
} finally {
setBusy(false);
setBusy(null);
}
}
const players = playersOnline ?? 0;
const askFirst = players > 0 || playerCountUnknown === true;
const spinner = <Loader2 className="animate-spin" />;
const stopButton = (variant: "destructive" | "outline", onClick: () => void) => (
<Button size={size} variant={variant} onClick={onClick} disabled={busy !== null}>
{busy === "stop" ? spinner : <Square />}
{busy === "stop" ? t("stopping") : t("stop")}
</Button>
);
let control;
if (!live) {
if (failed) {
control = (
<Button size={size} onClick={() => void run("wake")} disabled={busy}>
{busy ? spinner : <Play />}
{busy ? t("waking") : t("wake")}
<div className="flex flex-wrap items-center justify-end gap-2">
<Button size={size} onClick={() => void run("wake")} disabled={busy !== null}>
{busy === "wake" ? spinner : <RotateCcw />}
{busy === "wake" ? t("retrying_start") : t("retry_start")}
</Button>
{stopButton("outline", () => void run("stop"))}
</div>
);
} else if (!live) {
control = (
<Button size={size} onClick={() => void run("wake")} disabled={busy !== null}>
{busy === "wake" ? spinner : <Play />}
{busy === "wake" ? t("waking") : t("wake")}
</Button>
);
} else if (confirming) {
@@ -71,27 +93,14 @@ export function PowerButton({
? t("stop_confirm_players", { count: players })
: t("stop_confirm_unknown")}
</span>
<Button size={size} variant="ghost" onClick={() => setConfirming(false)} disabled={busy}>
<Button size={size} variant="ghost" onClick={() => setConfirming(false)} disabled={busy !== null}>
{t("common:cancel")}
</Button>
<Button size={size} variant="destructive" onClick={() => void run("stop")} disabled={busy}>
{busy ? spinner : <Square />}
{busy ? t("stopping") : t("stop")}
</Button>
{stopButton("destructive", () => void run("stop"))}
</div>
);
} else {
control = (
<Button
size={size}
variant="destructive"
onClick={() => (askFirst ? setConfirming(true) : void run("stop"))}
disabled={busy}
>
{busy ? spinner : <Square />}
{busy ? t("stopping") : t("stop")}
</Button>
);
control = stopButton("destructive", () => (askFirst ? setConfirming(true) : void run("stop")));
}
return (
+7 -2
View File
@@ -11,7 +11,8 @@
"phase_starting": "Starting",
"phase_stopping": "Stopping",
"phase_stopped": "Stopped",
"phase_failed": "Failed",
"phase_failed": "Failed to start",
"phase_failed_retrying": "Timed out · retrying",
"phase_unknown": "Unknown",
"online": "online",
"console": "Console",
@@ -23,12 +24,16 @@
"stopping": "Stopping…",
"wake": "Wake",
"waking": "Waking…",
"retry_start": "Retry start",
"retrying_start": "Retrying…",
"server_asleep_title": "Server is asleep",
"server_asleep_body": "Wake it to boot a pod — the live console attaches automatically the moment it starts up.",
"server_shutting_down_title": "Server is shutting down",
"server_shutting_down_body": "The pod is terminating, so the live console has detached. It will be asleep in a moment.",
"server_failed_title": "Server failed to start",
"server_failed_body": "No pod is running, so there is nothing to stream. Wake it to retry; the console attaches once a pod comes back up.",
"server_failed_body": "Its automatic retries are spent. The console holds the log of the last attempt; fix the cause, then press Retry start.",
"server_failed_retrying_title": "Start timed out — retrying",
"server_failed_retrying_body": "The operator recreates the pod and tries again ({{used}} of {{max}} retries used). Press Retry start to go now instead.",
"console_offline_title": "Console is offline",
"console_offline_body": "The live console attaches automatically as soon as the server is running.",
"my_servers_breadcrumb": "My servers",
+6 -1
View File
@@ -12,6 +12,7 @@
"phase_stopping": "停止中",
"phase_stopped": "已停止",
"phase_failed": "启动失败",
"phase_failed_retrying": "启动超时·重试中",
"phase_unknown": "未知",
"online": "人在线",
"console": "控制台",
@@ -23,12 +24,16 @@
"stopping": "停止中…",
"wake": "启动",
"waking": "启动中…",
"retry_start": "重试启动",
"retrying_start": "重试中…",
"server_asleep_title": "服务器已休眠",
"server_asleep_body": "启动后将拉起 Pod,实时控制台会自动连接。",
"server_shutting_down_title": "服务器正在关闭",
"server_shutting_down_body": "Pod 正在终止,实时控制台已断开。关闭完成后将进入休眠。",
"server_failed_title": "服务器启动失败",
"server_failed_body": "当前无运行中的 Pod,无法获取日志流。重新启动后控制台将自动连接。",
"server_failed_body": "自动重试已用完。控制台里是最后一次启动的日志,找到原因修好后点「重试启动」。",
"server_failed_retrying_title": "启动超时,正在自动重试",
"server_failed_retrying_body": "operator 会重建 Pod 再试(已重试 {{used}}/{{max}} 次)。不想等可以点「重试启动」立即再来。",
"console_offline_title": "控制台未连接",
"console_offline_body": "服务器运行后,实时控制台将自动连接。",
"my_servers_breadcrumb": "我的服务器",
+13 -3
View File
@@ -167,7 +167,7 @@ export interface paths {
put?: never;
/**
* Domain-autostart wake driven by velocity for a joining player (spec §9.1).
* @description velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the server's autostartPolicy and the per-server wake cooldown.
* @description velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the server's autostartPolicy and the per-server wake cooldown. A server whose start failed with its automatic retries spent answers 409 start_failed: a join never resets the retry budget, so velocity queues no one for it.
*/
post: operations["internalWake"];
delete?: never;
@@ -419,7 +419,10 @@ export interface paths {
};
get?: never;
put?: never;
/** Wake your own server. */
/**
* Wake your own server.
* @description On a server whose start Failed (phase Failed, desiredState Running) this is a retry: the operator recreates the pod and starts it over with a fresh automatic-restart budget. It is audited as retry_start.
*/
post: operations["wake"];
delete?: never;
options?: never;
@@ -2398,6 +2401,13 @@ export interface components {
autostartPolicy?: "ownerOnly" | "public" | "allowlist";
/** @description Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players. */
playerCountUnknown?: boolean;
/**
* Format: int32
* @description Owned rows only. How often the operator recreated the pod of this start after it timed out.
*/
autoRestarts?: number;
/** @description Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over. */
startGaveUp?: boolean;
};
/** @description One world backup (internal/api/repo.go BackupView). backup_ref is withheld (spec §286). */
BackupView: {
@@ -3148,7 +3158,7 @@ export interface operations {
401: components["responses"]["Unauthorized"];
403: components["responses"]["Forbidden"];
404: components["responses"]["NotFound"];
/** @description A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started. */
/** @description Nothing was started. maintenance_in_progress: a restore, backup or file write holds the server's world volume. start_failed: the last start failed and its automatic retries are spent (ServerInfo.startGaveUp); the server stays down until a person starts it from the panel. */
409: {
headers: {
[name: string]: unknown;
+4
View File
@@ -36,6 +36,10 @@ export interface MyServerView {
autostartPolicy?: AutostartPolicy;
/** True while the operator cannot read the player count; a stop may drop players. */
playerCountUnknown?: boolean;
/** How often the operator recreated the pod of this start after it timed out. */
autoRestarts?: number;
/** True for a Failed server no automatic retry will bring up. */
startGaveUp?: boolean;
}
/** ServerStatus is GET /servers/{name}/status (Go ServerInfo). It never carries
+78
View File
@@ -0,0 +1,78 @@
// @vitest-environment jsdom
import { describe, it, expect, vi, beforeEach } from "vitest";
import { render, screen, within } from "@testing-library/react";
import { MemoryRouter, Route, Routes } from "react-router-dom";
import { ServerConsole } from "./ServerConsole";
const calls = vi.hoisted(() => ({ status: vi.fn(), myServers: vi.fn() }));
vi.mock("@/lib/tier", () => ({
useTier: () => ({
loading: false,
identity: { user_id: "admin-1", email: "[email protected]", role: "admin" },
isAdmin: true,
isOwner: false,
}),
}));
vi.mock("@/lib/config", async (importActual) => {
const actual = await importActual<typeof import("@/lib/config")>();
return {
...actual,
loadConfig: () => Promise.resolve({ apiBase: "/api/v1", rootDomain: "example.test", gamePort: 25570 }),
};
});
vi.mock("@/lib/api", async (importActual) => {
const actual = await importActual<typeof import("@/lib/api")>();
return { ...actual, api: { ...actual.api, ...calls } };
});
// The live log is a WebSocket; here it only matters whether the page shows it.
vi.mock("@/components/LogConsole", () => ({ LogConsole: () => <div data-testid="log-stream" /> }));
beforeEach(() => {
calls.status.mockReset();
calls.myServers.mockReset();
calls.myServers.mockResolvedValue([]);
});
function renderConsole() {
render(
<MemoryRouter initialEntries={["/servers/survival"]}>
<Routes>
<Route path="/servers/:name" element={<ServerConsole />} />
</Routes>
</MemoryRouter>,
);
}
function status(over: Record<string, unknown>) {
return { name: "survival", subdomain: "survival", phase: "Failed", ready: false, playersOnline: 0, playersMax: 20, ...over };
}
describe("ServerConsole failed start", () => {
it("shows the failed attempt's log under what the owner can do once retries are spent", async () => {
calls.status.mockResolvedValue(status({ desiredState: "Running", startGaveUp: true, autoRestarts: 3 }));
renderConsole();
const notice = await screen.findByRole("status", { name: "Server failed to start" });
expect(within(notice).getByText(/automatic retries are spent/)).toBeTruthy();
expect(await screen.findByTestId("log-stream")).toBeTruthy();
expect(screen.getByRole("button", { name: /Retry start/ })).toBeTruthy();
});
it("counts the retries used while the operator is still retrying", async () => {
calls.status.mockResolvedValue(status({ desiredState: "Running", autoRestarts: 1 }));
renderConsole();
const notice = await screen.findByRole("status", { name: "Start timed out — retrying" });
expect(within(notice).getByText(/1 of 3 retries used/)).toBeTruthy();
expect(screen.getByText("Timed out · retrying")).toBeTruthy();
});
it("asks nothing of the owner once a failed server is being stopped", async () => {
calls.status.mockResolvedValue(status({ desiredState: "Stopped" }));
renderConsole();
expect(await screen.findByTestId("log-stream")).toBeTruthy();
expect(screen.queryByRole("status", { name: /failed to start|timed out/ })).toBeNull();
expect(screen.queryByRole("button", { name: /Retry start/ })).toBeNull();
});
});
+41 -12
View File
@@ -1,10 +1,10 @@
import { useState, useRef, useCallback, useLayoutEffect, type KeyboardEvent } from "react";
import { useState, useRef, useCallback, useId, useLayoutEffect, type KeyboardEvent } from "react";
import { Link, useParams } from "react-router-dom";
import { Terminal, Moon, Shield, ShieldAlert, HelpCircle, Loader2, Users, Archive, FolderOpen, ChevronRight, type LucideIcon } from "lucide-react";
import { useTranslation } from "react-i18next";
import { Card, CardContent } from "@/components/ui/card";
import { BackLink } from "@/components/BackLink";
import { PhaseBadge } from "@/components/PhaseBadge";
import { MAX_AUTO_RESTARTS, PhaseBadge, startFailure, type StartFailure } from "@/components/PhaseBadge";
import { PageHeader } from "@/components/PageHeader";
import { LogConsole } from "@/components/LogConsole";
import { Loading, ErrorState } from "@/components/States";
@@ -39,12 +39,6 @@ function useNotStreamingCopy(phase: Phase): { icon: LucideIcon; title: string; b
title: t("server_shutting_down_title"),
body: t("server_shutting_down_body"),
};
case "Failed":
return {
icon: ShieldAlert,
title: t("server_failed_title"),
body: t("server_failed_body"),
};
default:
return {
icon: HelpCircle,
@@ -67,6 +61,34 @@ function NotStreaming({ phase }: { phase: Phase }) {
);
}
// FailedStartNotice heads the console of a server whose start failed: whether the
// operator will try again on its own, and what the owner can do. The log below it
// is the attempt that failed, which is what they need to find the cause.
function FailedStartNotice({ failure, autoRestarts }: { failure: StartFailure; autoRestarts: number }) {
const { t } = useTranslation("servers");
const retrying = failure === "retrying";
const titleId = useId();
return (
<div
role="status"
aria-labelledby={titleId}
className="flex items-start gap-3 border-b border-zinc-800 bg-red-950/40 px-4 py-3 text-sm text-zinc-300"
>
<ShieldAlert className="mt-0.5 h-4 w-4 shrink-0 text-red-400" />
<div className="space-y-0.5">
<p id={titleId} className="font-medium text-zinc-100">
{retrying ? t("server_failed_retrying_title") : t("server_failed_title")}
</p>
<p>
{retrying
? t("server_failed_retrying_body", { used: autoRestarts, max: MAX_AUTO_RESTARTS })
: t("server_failed_body")}
</p>
</div>
</div>
);
}
function loadHistory(name: string): string[] {
try {
const raw = localStorage.getItem(HISTORY_KEY(name));
@@ -206,8 +228,11 @@ export function ServerConsole() {
// during BOTH Starting and Running — the operator marks Starting once the pod
// is up but RCON is not yet reachable (it only flips to Running after RCON
// readiness). Boot logs flow precisely in that Starting window, which is when a
// read most wants them, so the gate streams for both, not Running alone.
const streamable = data?.phase === "Running" || data?.phase === "Starting";
// read most wants them, so the gate streams for both, not Running alone. A
// Failed start usually leaves its pod behind, and that pod's log is how the
// owner finds out why it failed, so Failed streams too.
const streamable = data?.phase === "Running" || data?.phase === "Starting" || data?.phase === "Failed";
const failure = data ? startFailure(data) : null;
return (
<div className="flex flex-col lg:h-[calc(100vh-3.5rem)] lg:min-h-[35rem] gap-4 min-h-0">
@@ -227,10 +252,11 @@ export function ServerConsole() {
subtitle={cfg ? <CopyAddress address={joinAddress(data.subdomain, cfg)} /> : undefined}
actions={
<div className="flex items-center gap-2">
<PhaseBadge phase={data.phase} />
<PhaseBadge phase={data.phase} failure={failure} autoRestarts={data.autoRestarts} />
<PowerButton
name={name}
live={streamable || data.phase === "Stopping"}
live={data.phase === "Running" || data.phase === "Starting" || data.phase === "Stopping"}
failed={failure !== null}
playersOnline={data.playersOnline}
playerCountUnknown={data.playerCountUnknown}
onChanged={reload}
@@ -251,6 +277,9 @@ export function ServerConsole() {
</div>
) : cfg ? (
<div className="flex-1 flex flex-col lg:min-h-0 min-h-0 bg-black">
{failure && (
<FailedStartNotice failure={failure} autoRestarts={data.autoRestarts ?? 0} />
)}
<LogConsole
key={name}
url={consoleStreamURL(cfg.apiBase, name)}
@@ -139,3 +139,27 @@ describe("ServersPage addresses", () => {
expect(survival.getByRole("button", { name: "Copy address survival.example.test:25570" })).toBeTruthy();
});
});
describe("ServersPage failed starts", () => {
it("offers a failed start a retry and a stop, and tells retrying from given up", async () => {
calls.fleet.mockResolvedValue([
row("broken", { phase: "Failed", desiredState: "Running", startGaveUp: true, autoRestarts: 3, owned: true }),
row("flaky", { phase: "Failed", desiredState: "Running", autoRestarts: 1, owned: true }),
]);
render(
<MemoryRouter>
<ServersPage />
</MemoryRouter>,
);
const broken = await tableRow("broken");
expect(broken.getByText("Failed to start")).toBeTruthy();
expect(broken.getByRole("button", { name: /Retry start/ })).toBeTruthy();
expect(broken.getByRole("button", { name: /Stop/ })).toBeTruthy();
expect(broken.queryByRole("button", { name: /Wake/ })).toBeNull();
const flaky = await tableRow("flaky");
expect(flaky.getByText("Timed out · retrying")).toBeTruthy();
expect(flaky.getByRole("button", { name: /Retry start/ })).toBeTruthy();
});
});
+10 -3
View File
@@ -30,7 +30,7 @@ import {
SelectContent,
SelectItem,
} from "@/components/ui/select";
import { PhaseBadge, PHASE_KEY, PHASE_COLOR } from "@/components/PhaseBadge";
import { PhaseBadge, PHASE_KEY, PHASE_COLOR, startFailure } from "@/components/PhaseBadge";
import { PowerButton } from "@/components/PowerButton";
import { Loading, ErrorState, EmptyState } from "@/components/States";
import { Pagination } from "@/components/Pagination";
@@ -81,6 +81,8 @@ interface UnifiedServer {
playersOnline: number;
playersMax: number;
playerCountUnknown?: boolean;
autoRestarts?: number;
startGaveUp?: boolean;
owner?: string;
endpointAddress?: string | null;
claimable?: boolean;
@@ -133,6 +135,8 @@ export function ServersPage() {
playersOnline: s.playersOnline,
playersMax: s.playersMax,
playerCountUnknown: s.playerCountUnknown,
autoRestarts: s.autoRestarts,
startGaveUp: s.startGaveUp,
owner: s.owner,
endpointAddress: s.endpointAddress,
claimable: s.claimable,
@@ -152,6 +156,8 @@ export function ServersPage() {
playersOnline: s.playersOnline,
playersMax: s.playersMax,
playerCountUnknown: s.playerCountUnknown,
autoRestarts: s.autoRestarts,
startGaveUp: s.startGaveUp,
owner: s.owned ? t("servers:owned_filter_mine") || "me" : undefined,
claimable: s.claimable,
owned: s.owned,
@@ -509,6 +515,7 @@ function ServerActions({
<PowerButton
name={server.name}
live={live}
failed={startFailure(server) !== null}
playersOnline={server.playersOnline}
playerCountUnknown={server.playerCountUnknown}
onChanged={onChanged}
@@ -610,7 +617,7 @@ function ServerRow({
<td className="px-4 py-3 align-middle text-left">
<div className="flex items-center gap-2">
<ServerName server={server} />
<PhaseBadge phase={server.phase} />
<PhaseBadge phase={server.phase} failure={startFailure(server)} autoRestarts={server.autoRestarts} />
</div>
{address && <CopyAddress address={address} className="md:max-w-[16rem]" />}
</td>
@@ -669,7 +676,7 @@ function ServerMobileCard({
<div className="flex min-w-0 items-baseline gap-2">
<ServerName server={server} />
</div>
<PhaseBadge phase={server.phase} />
<PhaseBadge phase={server.phase} failure={startFailure(server)} autoRestarts={server.autoRestarts} />
</div>
{/* Its own line: the join address is what a player came for, so it
gets the card's full width rather than sharing it with the badge. */}
@@ -20,12 +20,14 @@ import java.util.Optional;
* backend or the wake lever.
*
* <p>This is the read-only, phase-aware subset of the responsibility: the MOTD is
* synthesized from the server's lifecycle (online / starting / sleeping) and its
* cached player counts. Mirroring each backend's <em>own</em> MOTD string (by
* pinging ready servers in the background and caching the result) is a richer
* synthesized from the server's lifecycle (online / starting / start failed /
* sleeping) and its cached player counts. Mirroring each backend's <em>own</em> MOTD
* string (by pinging ready servers in the background and caching the result) is a richer
* variant deferred to a later slice; nothing here ever pings a sleeping backend.
*/
public final class MotdResponder {
private static final String PHASE_FAILED = "Failed";
private final ServerRegistry registry;
MotdResponder(ServerRegistry registry) {
@@ -55,21 +57,33 @@ public final class MotdResponder {
}
// The server-list ping carries no client locale, so the MOTD status uses the
// both-languages-in-one-line pattern the modded /link clients share.
private static String statusLine(ServerView v) {
// both-languages-in-one-line pattern the modded /link clients share. A start
// that failed still holds desiredState Running, so it is read first: joining a
// server whose retries are spent wakes nothing, and one between retries is
// waiting out a backoff, which a "starting…" line hid behind a queue that ran out.
static String statusLine(ServerView v) {
if (v.ready()) {
return "在线 / online";
}
if (v.startGaveUp()) {
return "启动失败,等服主处理 / failed to start — the owner has to restart it";
}
if (PHASE_FAILED.equals(v.phase())) {
return "启动超时,稍后自动重试 / start timed out — retrying shortly";
}
if ("Running".equals(v.desiredState())) {
return "启动中… / starting…";
}
return "休眠中,加入即唤醒 / sleeping — join to wake";
}
private static NamedTextColor statusColor(ServerView v) {
static NamedTextColor statusColor(ServerView v) {
if (v.ready()) {
return NamedTextColor.GREEN;
}
if (v.startGaveUp()) {
return NamedTextColor.RED;
}
if ("Running".equals(v.desiredState())) {
return NamedTextColor.YELLOW;
}
@@ -505,11 +505,7 @@ public final class WaitingRouter {
// stopped the server — it says so and returns false.
private boolean stillComing(Player player, boolean zh, Waiter w, ServerView status, long now) {
if (status.startGaveUp()) {
player.sendMessage(Component.text(
zh ? "「" + w.serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。"
: "« " + w.serverName + " » failed to start and its automatic retries are spent. "
+ "The owner can check its log in the panel and start it again.",
NamedTextColor.RED));
tellStartFailed(player, zh, w.serverName);
return false;
}
if (DESIRED_STOPPED.equalsIgnoreCase(status.desiredState())) {
@@ -655,6 +651,12 @@ public final class WaitingRouter {
NamedTextColor.YELLOW));
return;
}
if ("start_failed".equals(e.errorCode())) {
// The last start failed with its retries spent. The join does
// not buy another round of them, so there is nothing to wait for.
tellStartFailed(player, zh, serverName);
return;
}
logWakeFailure(player, serverName, zh, e);
return;
case 429:
@@ -688,6 +690,16 @@ public final class WaitingRouter {
waiting.put(id, new Waiter(serverName, clock.getAsLong(), fromMenu));
}
// The server's start failed and its automatic retries are spent: nothing more is
// coming until a person starts it again from the panel.
private static void tellStartFailed(Player player, boolean zh, String serverName) {
player.sendMessage(Component.text(
zh ? "「" + serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。"
: "« " + serverName + " » failed to start and its automatic retries are spent. "
+ "The owner can check its log in the panel and start it again.",
NamedTextColor.RED));
}
private void logWakeFailure(Player player, String serverName, boolean zh, LinkException e) {
log.warn("Felis: wake {} failed (status={}): {}", serverName, e.statusCode(), e.getMessage());
player.sendMessage(Component.text(
@@ -2,6 +2,8 @@ package best.lolicon.felis.velocity;
import best.lolicon.felis.link.ServerView;
import net.kyori.adventure.text.format.NamedTextColor;
import java.net.InetSocketAddress;
import java.util.List;
import java.util.Optional;
@@ -139,6 +141,24 @@ public final class ServerRegistryTest {
assertAddr("a non-numeric port is not a port", "backend:x", 25565,
ServerRegistry.parseAddress("backend:x"));
// The server-list MOTD reads a failed start before desiredState Running,
// which a failed start still holds.
assertEq("motd: up", "在线 / online", MotdResponder.statusLine(up("a", "a", "10.43.0.1:25565")));
assertEq("motd: starting", "启动中… / starting…", MotdResponder.statusLine(
new ServerView("a", "a", "Starting", false, "ownerOnly", "Running", "fallback", "login", 0, 0)));
ServerView backoff = new ServerView("a", "a", "Failed", false, "ownerOnly", "Running",
"fallback", "login", 0, 0, false, 1, false);
assertEq("motd: between retries", "启动超时,稍后自动重试 / start timed out — retrying shortly",
MotdResponder.statusLine(backoff));
assertEq("motd: between retries is yellow", NamedTextColor.YELLOW, MotdResponder.statusColor(backoff));
ServerView gaveUp = new ServerView("a", "a", "Failed", false, "ownerOnly", "Running",
"fallback", "login", 0, 0, false, 3, true);
assertEq("motd: retries spent", "启动失败,等服主处理 / failed to start — the owner has to restart it",
MotdResponder.statusLine(gaveUp));
assertEq("motd: retries spent is red", NamedTextColor.RED, MotdResponder.statusColor(gaveUp));
assertEq("motd: sleeping", "休眠中,加入即唤醒 / sleeping — join to wake",
MotdResponder.statusLine(down("a", "a")));
System.out.println("ServerRegistryTest OK (" + checks + " checks)");
}
@@ -219,6 +219,7 @@ public final class WaitingRouterTest {
String[][] cases = {
{"403 forbidden", "You're not allowed to start « gamma »", null},
{"409 maintenance_in_progress", "« gamma » is under maintenance", null},
{"409 start_failed", "« gamma » failed to start and its automatic retries are spent", null},
{"409 conflict", "Couldn't start « gamma » right now", "wake gamma failed (status=409)"},
{"503 at_capacity", "The cluster is at capacity right now", null},
{"503 unavailable", "Couldn't start « gamma » right now", "wake gamma failed (status=503)"},