fix(felis): 启动失败的服可在面板重试或停止,玩家入服时直接告知启动失败

This commit is contained in:
Lemon-miaow committed 2026-09-26 22:08:29 +08:00
1 parent c1025274e1
commit 22d1e834ad
28 files changed
+775 -75

No files matched your search

+19 -2
View File
@@ -566,6 +566,13 @@ components:
playerCountUnknown: playerCountUnknown:
type: boolean type: boolean
description: Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players. description: Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players.
autoRestarts:
type: integer
format: int32
description: Owned rows only. How often the operator recreated the pod of this start after it timed out.
startGaveUp:
type: boolean
description: Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over.
BackupView: BackupView:
type: object type: object
@@ -1165,7 +1172,9 @@ paths:
description: >- description: >-
velocity holds no web Principal, so it drives the wake lever with its velocity holds no web Principal, so it drives the wake lever with its
service token, identifying the player by online-mode UUID. Gated by the service token, identifying the player by online-mode UUID. Gated by the
server's autostartPolicy and the per-server wake cooldown. server's autostartPolicy and the per-server wake cooldown. A server whose
start failed with its automatic retries spent answers 409 start_failed: a
join never resets the retry budget, so velocity queues no one for it.
x-felis-face: [internal] x-felis-face: [internal]
x-felis-tier: service x-felis-tier: service
x-felis-callers: [velocity] x-felis-callers: [velocity]
@@ -1203,7 +1212,11 @@ paths:
'404': '404':
$ref: '#/components/responses/NotFound' $ref: '#/components/responses/NotFound'
'409': '409':
description: A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started. description: >-
Nothing was started. maintenance_in_progress: a restore, backup or
file write holds the server's world volume. start_failed: the last
start failed and its automatic retries are spent (ServerInfo.startGaveUp);
the server stays down until a person starts it from the panel.
content: content:
application/json: application/json:
schema: { $ref: '#/components/schemas/Error' } schema: { $ref: '#/components/schemas/Error' }
@@ -1767,6 +1780,10 @@ paths:
tags: [servers] tags: [servers]
operationId: wake operationId: wake
summary: Wake your own server. summary: Wake your own server.
description: >-
On a server whose start Failed (phase Failed, desiredState Running) this
is a retry: the operator recreates the pod and starts it over with a
fresh automatic-restart budget. It is audited as retry_start.
x-felis-face: [external] x-felis-face: [external]
x-felis-tier: app x-felis-tier: app
security: [{ sessionCookie: [] }] security: [{ sessionCookie: [] }]
+18 -4
View File
@@ -1748,6 +1748,7 @@ type fakeCluster struct {
wakeErr map[string]error wakeErr map[string]error
acquired []string // "name:kind" per admitted AcquireMaintenance acquired []string // "name:kind" per admitted AcquireMaintenance
released []string // names per ReleaseMaintenance released []string // names per ReleaseMaintenance
retried []string // names per RetryStart
} }
func newFakeCluster() *fakeCluster { func newFakeCluster() *fakeCluster {
@@ -1794,6 +1795,16 @@ func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.De
c.desired[n] = s c.desired[n] = s
return nil return nil
} }
// RetryStart records the retry on top of the plain start, so a test tells a
// Failed server's retry apart from an ordinary wake.
func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
if err := c.SetDesiredState(ctx, n, v1alpha1.DesiredRunning); err != nil {
return err
}
c.retried = append(c.retried, n)
return nil
}
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error { func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
if err := c.maintErr[n]; err != nil { if err := c.maintErr[n]; err != nil {
return err return err
@@ -2092,9 +2103,11 @@ func TestMyServersJoinsLiveState(t *testing.T) {
cl := newFakeCluster() cl := newFakeCluster()
cl.list = []ServerInfo{ cl.list = []ServerInfo{
{Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running", {Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running",
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true}, AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true,
AutoRestarts: 3, StartGaveUp: true},
{Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running", {Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running",
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true}, AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true,
AutoRestarts: 3, StartGaveUp: true},
} }
api := newTestAPI(repo, cl) api := newTestAPI(repo, cl)
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}} api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
@@ -2122,7 +2135,8 @@ func TestMyServersJoinsLiveState(t *testing.T) {
mine, open := got["mine"], got["open"] mine, open := got["mine"], got["open"]
for k, want := range map[string]any{"displayName": "My World", "phase": "Running", for k, want := range map[string]any{"displayName": "My World", "phase": "Running",
"desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true, "desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true,
"playersOnline": float64(2), "playersMax": float64(20)} { "playersOnline": float64(2), "playersMax": float64(20),
"autoRestarts": float64(3), "startGaveUp": true} {
if mine[k] != want { if mine[k] != want {
t.Errorf("own row %s = %v, want %v", k, mine[k], want) t.Errorf("own row %s = %v, want %v", k, mine[k], want)
} }
@@ -2130,7 +2144,7 @@ func TestMyServersJoinsLiveState(t *testing.T) {
if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) { if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) {
t.Errorf("claimable row public fields = %v", open) t.Errorf("claimable row public fields = %v", open)
} }
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown"} { for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown", "autoRestarts", "startGaveUp"} {
if _, ok := open[k]; ok { if _, ok := open[k]; ok {
t.Errorf("claimable row carries owner detail %s = %v", k, open[k]) t.Errorf("claimable row carries owner detail %s = %v", k, open[k])
} }
+4
View File
@@ -117,6 +117,10 @@ type Cluster interface {
// *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore, // *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore,
// backup or file write holds the world volume. // backup or file write holds the world volume.
SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
// RetryStart is SetDesiredState(Running) for a server whose start Failed: it
// also asks the operator to start it over with a fresh auto-restart budget
// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
RetryStart(ctx context.Context, name string) error
// AcquireMaintenance admits one world-volume operation (internal/maintenance // AcquireMaintenance admits one world-volume operation (internal/maintenance
// kind): ErrNotStopped unless the server is fully stopped, a // kind): ErrNotStopped unless the server is fully stopped, a
// *MaintenanceBusyError while another operation holds the volume. The check // *MaintenanceBusyError while another operation holds the volume. The check
+12
View File
@@ -152,6 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
writeError(w, r, err) writeError(w, r, err)
return return
} }
// A start whose automatic restarts are spent (or that can never succeed as
// configured) stays down until a person looks at it. The 202 this used to
// return queued the player for a server nothing was starting. The join leaves
// the restart budget alone, or every player who tried to join would buy
// another three crash loops; velocity tells them and queues no one. A Failed
// server still inside its backoff is not this: its next attempt is coming, so
// it gets the 202 and the player waits for it.
if info.StartGaveUp && info.DesiredState == string(v1alpha1.DesiredRunning) {
writeError(w, r, newError(http.StatusConflict, "start_failed",
"the server failed to start and its automatic retries are spent; its owner can retry from the panel"))
return
}
if !a.limiter().allowed(name, a.WakeCooldown) { if !a.limiter().allowed(name, a.WakeCooldown) {
writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly")) writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly"))
return return
+131
View File
@@ -0,0 +1,131 @@
package api
import (
"net/http"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
)
// A start that Failed holds desiredState Running already, so the wake that
// re-writes Running changes nothing. These pin the two ways out: a person's
// wake in the panel starts it over (RetryStart), and a player's join onto one
// whose retries are spent is told so (409 start_failed) and never resets them.
func failedServer(gaveUp bool) *ServerInfo {
return &ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseFailed),
AutoRestarts: v1alpha1.MaxAutoRestarts, StartGaveUp: gaveUp}
}
func TestWakeOfFailedServerRetriesStart(t *testing.T) {
mk := func(info *ServerInfo) (*API, *fakeCluster, *fakeRepo) {
repo := newFakeRepo()
cl := newFakeCluster()
cl.byName["survival"] = info
api := newTestAPI(repo, cl)
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
return api, cl, repo
}
wake := func(api *API) int {
return do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil).Code
}
t.Run("retries spent: the wake starts it over", func(t *testing.T) {
api, cl, repo := mk(failedServer(true))
if code := wake(api); code != http.StatusAccepted {
t.Fatalf("code = %d, want 202", code)
}
if len(cl.retried) != 1 || cl.retried[0] != "survival" {
t.Fatalf("retried = %v, want [survival]", cl.retried)
}
if len(repo.audits) != 1 || repo.audits[0].Action != "retry_start" {
t.Fatalf("audits = %+v, want one retry_start", repo.audits)
}
})
t.Run("inside the backoff: a person asking skips the wait", func(t *testing.T) {
api, cl, _ := mk(failedServer(false))
if code := wake(api); code != http.StatusAccepted || len(cl.retried) != 1 {
t.Fatalf("code = %d retried = %v, want 202 and one retry", code, cl.retried)
}
})
t.Run("stopped: a plain wake", func(t *testing.T) {
api, cl, repo := mk(&ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)})
if code := wake(api); code != http.StatusAccepted {
t.Fatalf("code = %d, want 202", code)
}
if len(cl.retried) != 0 || cl.desired["survival"] != v1alpha1.DesiredRunning {
t.Fatalf("retried = %v desired = %q, want a plain start", cl.retried, cl.desired["survival"])
}
if len(repo.audits) != 1 || repo.audits[0].Action != "wake" {
t.Fatalf("audits = %+v, want one wake", repo.audits)
}
})
t.Run("maintenance holds the world: refused, nothing retried", func(t *testing.T) {
api, cl, _ := mk(failedServer(true))
cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore}
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil)
if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("retried = %v, want none", cl.retried)
}
})
}
func TestInternalWakeOfFailedServer(t *testing.T) {
body := `{"mc_uuid":"` + wakeUUID + `"}`
t.Run("retries spent: 409 start_failed, budget and cooldown untouched", func(t *testing.T) {
api, cl := newInternalWakeAPI("public")
cl.byName["survival"] = failedServer(true)
api.WakeCooldown = time.Minute
repo := api.Repo.(*fakeRepo)
w := internalWake(api, body)
if w.Code != http.StatusConflict || decodeErr(t, w) != "start_failed" {
t.Fatalf("code = %d body %s, want 409 start_failed", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
}
if len(repo.audits) != 0 {
t.Fatalf("nothing was woken, so nothing is audited: %+v", repo.audits)
}
// The owner fixes it and starts it from the panel; the next join is not
// held back by a cooldown the refused one never spent.
cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public",
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseStarting)}
if w := internalWake(api, body); w.Code != http.StatusAccepted {
t.Fatalf("join after the fix: code = %d body %s, want 202", w.Code, w.Body.String())
}
})
t.Run("inside the backoff: 202, the player waits for the next attempt", func(t *testing.T) {
api, cl := newInternalWakeAPI("public")
cl.byName["survival"] = failedServer(false)
if w := internalWake(api, body); w.Code != http.StatusAccepted {
t.Fatalf("code = %d body %s, want 202", w.Code, w.Body.String())
}
if len(cl.retried) != 0 {
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
}
})
t.Run("forbidden player: 403 comes first", func(t *testing.T) {
api, cl := newInternalWakeAPI("ownerOnly")
info := failedServer(true)
info.AutostartPolicy = "ownerOnly"
cl.byName["survival"] = info
if w := internalWake(api, body); w.Code != http.StatusForbidden {
t.Fatalf("code = %d, want 403", w.Code)
}
})
}
+20 -4
View File
@@ -56,9 +56,23 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
return return
} }
// Refused with 409 maintenance_in_progress while a restore, backup or file // A start that Failed already holds desiredState Running, so writing Running
// write holds the world volume: starting on a half-written world corrupts it. // again changes nothing and the server stayed dead once its automatic restarts
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil { // were spent. A person pressing start on it means "try again": RetryStart
// makes the operator start it over with a fresh restart budget. Only this face
// does that; a player's join never resets the budget (handleInternalWake).
//
// Either write is refused with 409 maintenance_in_progress while a restore,
// backup or file write holds the world volume: starting on a half-written
// world corrupts it.
action := "wake"
if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) {
action = "retry_start"
err = a.Cluster.RetryStart(r.Context(), name)
} else {
err = a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning)
}
if err != nil {
a.writeLookupError(w, r, err) a.writeLookupError(w, r, err)
return return
} }
@@ -67,7 +81,7 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
// held at capacity should retry the instant a slot frees, not wait out a // held at capacity should retry the instant a slot frees, not wait out a
// cooldown their refused wake never earned). // cooldown their refused wake never earned).
a.limiter().record(name) a.limiter().record(name)
a.audit(r, "wake", name) a.audit(r, action, name)
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"}) writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
} }
@@ -267,6 +281,8 @@ func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) {
v.DesiredState = info.DesiredState v.DesiredState = info.DesiredState
v.AutostartPolicy = info.AutostartPolicy v.AutostartPolicy = info.AutostartPolicy
v.PlayerCountUnknown = info.PlayerCountUnknown v.PlayerCountUnknown = info.PlayerCountUnknown
v.AutoRestarts = info.AutoRestarts
v.StartGaveUp = info.StartGaveUp
} }
} }
} }
+17
View File
@@ -246,6 +246,17 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a
// against, the same as AcquireMaintenance's: whichever of a racing wake and // against, the same as AcquireMaintenance's: whichever of a racing wake and
// admission writes second gets a conflict, re-reads, and sees the other. // admission writes second gets a conflict, re-reads, and sees the other.
func (k *K8sCluster) start(ctx context.Context, name string) error { func (k *K8sCluster) start(ctx context.Context, name string) error {
return k.startWith(ctx, name, false)
}
// RetryStart starts a Failed server over: the same guarded write as start, plus
// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and
// recreates the pod.
func (k *K8sCluster) RetryStart(ctx context.Context, name string) error {
return k.startWith(ctx, name, true)
}
func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error { return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer var ms v1alpha1.MinecraftServer
if err := k.getServer(ctx, name, &ms); err != nil { if err := k.getServer(ctx, name, &ms); err != nil {
@@ -263,6 +274,12 @@ func (k *K8sCluster) start(ctx context.Context, name string) error {
// A lock still on the object here no longer holds anything (Holder said // A lock still on the object here no longer holds anything (Holder said
// so): drop it in the same write. // so): drop it in the same write.
delete(ms.Annotations, maintenance.Annotation) delete(ms.Annotations, maintenance.Annotation)
if retryFailed {
if ms.Annotations == nil {
ms.Annotations = map[string]string{}
}
ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339)
}
return k.c.Patch(ctx, &ms, patch) return k.c.Patch(ctx, &ms, patch)
}) })
} }
@@ -214,6 +214,39 @@ func TestStartRespectsMaintenance(t *testing.T) {
} }
}) })
t.Run("retry start -> Running plus the retry request, a plain start leaves none", func(t *testing.T) {
ms := stoppedServer()
ms.Spec.DesiredState = v1alpha1.DesiredRunning
ms.Status.Phase = v1alpha1.PhaseFailed
k, c := lockCluster(t, ms)
if err := k.SetDesiredState(ctx, "survival", v1alpha1.DesiredRunning); err != nil {
t.Fatalf("start: %v", err)
}
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
t.Fatalf("a plain start asked for a retry: %v", ann)
}
if err := k.RetryStart(ctx, "survival"); err != nil {
t.Fatalf("retry: %v", err)
}
ann, desired := annotations(t, c)
if desired != v1alpha1.DesiredRunning || ann[v1alpha1.AnnotationStartRetry] != lockNow.Format(time.RFC3339) {
t.Fatalf("desiredState = %q annotations = %v, want Running and the retry stamped %s",
desired, ann, lockNow.Format(time.RFC3339))
}
})
t.Run("retry start under a fresh lock -> busy, no retry request", func(t *testing.T) {
ms := stoppedServer()
ms.Annotations = map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lockNow)}
k, c := lockCluster(t, ms)
if err := k.RetryStart(ctx, "survival"); !errors.Is(err, ErrMaintenanceInProgress) {
t.Fatalf("err = %v, want maintenance in progress", err)
}
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
t.Fatalf("a refused retry left its request: %v", ann)
}
})
t.Run("stop ignores the lock", func(t *testing.T) { t.Run("stop ignores the lock", func(t *testing.T) {
ms := stoppedServer() ms := stoppedServer()
ms.Spec.DesiredState = v1alpha1.DesiredRunning ms.Spec.DesiredState = v1alpha1.DesiredRunning
+5 -3
View File
@@ -21,9 +21,9 @@ type ServerRecord struct {
// may auto-start, or may claim. Everything after Claimable is NOT stored in // may auto-start, or may claim. Everything after Claimable is NOT stored in
// Postgres — handleMyServers joins it best-effort from the CRD status // Postgres — handleMyServers joins it best-effort from the CRD status
// (Cluster.ListServers) at read time, so a cluster hiccup renders 0/0 and the // (Cluster.ListServers) at read time, so a cluster hiccup renders 0/0 and the
// cached phase, never a 500. DesiredState, AutostartPolicy and // cached phase, never a 500. DesiredState, AutostartPolicy, PlayerCountUnknown,
// PlayerCountUnknown are owner detail and stay empty on rows the caller does // AutoRestarts and StartGaveUp are owner detail and stay empty on rows the caller
// not own, the same split publicServerInfo makes on the status route. // does not own, the same split publicServerInfo makes on the status route.
type MyServerView struct { type MyServerView struct {
Name string `json:"name"` Name string `json:"name"`
Subdomain string `json:"subdomain"` Subdomain string `json:"subdomain"`
@@ -36,6 +36,8 @@ type MyServerView struct {
DesiredState string `json:"desiredState,omitempty"` DesiredState string `json:"desiredState,omitempty"`
AutostartPolicy string `json:"autostartPolicy,omitempty"` AutostartPolicy string `json:"autostartPolicy,omitempty"`
PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"` PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"`
AutoRestarts int32 `json:"autoRestarts,omitempty"`
StartGaveUp bool `json:"startGaveUp,omitempty"`
} }
// ServerOwnership is one live server's claim state as the fleet read joins it. // ServerOwnership is one live server's claim state as the fleet read joins it.
@@ -31,6 +31,13 @@ const (
// server; and felis-api never writes it, so marking a server takes kubectl on // server; and felis-api never writes it, so marking a server takes kubectl on
// the cluster, which fits a switch that drops the forwarding secret. // the cluster, which fits a switch that drops the forwarding secret.
LabelForwarding = GroupName + "/forwarding" LabelForwarding = GroupName + "/forwarding"
// AnnotationStartRetry is felis-api asking the operator to start a Failed
// server over: its value is the request time (RFC 3339). Re-patching
// desiredState to the Running it already holds changes nothing the operator
// can see, so a person pressing "retry" in the panel had no way through once
// the automatic restarts were spent. The operator takes the request once —
// fresh restart budget, new start anchor, pod recreated — and removes it.
AnnotationStartRetry = GroupName + "/start-retry"
) )
// ForwardingLegacy is the LabelForwarding value that selects legacy forwarding. // ForwardingLegacy is the LabelForwarding value that selects legacy forwarding.
+33
View File
@@ -195,6 +195,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu
desired = v1alpha1.DesiredStopped desired = v1alpha1.DesiredStopped
} }
prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts
if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok {
if err := r.takeStartRetry(ctx, &server, desired); err != nil {
return ctrl.Result{}, err
}
}
var res ctrl.Result var res ctrl.Result
var err error var err error
if desired == v1alpha1.DesiredStopped { if desired == v1alpha1.DesiredStopped {
@@ -233,6 +238,33 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine
r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg)) r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg))
} }
// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed
// and meant to run starts over as if freshly woken — pod recreated, restart
// budget back to zero, a new start anchor — and in any other state the request
// is stale and only removed. The fresh status is written before the request is
// removed, so a pass that fails in between leaves the request to be taken again
// (at the cost of one more pod recreate), never a removed request with the old
// spent budget still in place.
func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error {
if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed {
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}}
if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil {
return err
}
server.Status.AutoRestarts = 0
server.Status.StartRequestedAt = nil
r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod")
if err := r.patchStatus(ctx, server); err != nil {
return err
}
log.FromContext(ctx).Info("start retried")
r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod")
}
patch := client.MergeFrom(server.DeepCopy())
delete(server.Annotations, v1alpha1.AnnotationStartRetry)
return r.Patch(ctx, server, patch)
}
func (r *Reconciler) event(server *v1alpha1.MinecraftServer, eventType, reason, msg string) { func (r *Reconciler) event(server *v1alpha1.MinecraftServer, eventType, reason, msg string) {
if r.Recorder != nil { if r.Recorder != nil {
r.Recorder.Event(server, eventType, reason, msg) r.Recorder.Event(server, eventType, reason, msg)
@@ -807,6 +839,7 @@ func (r *Reconciler) markFailed(server *v1alpha1.MinecraftServer, reason, msg st
server.Status.Phase = v1alpha1.PhaseFailed server.Status.Phase = v1alpha1.PhaseFailed
server.Status.Ready = false server.Status.Ready = false
server.Status.ObservedGeneration = server.Generation server.Status.ObservedGeneration = server.Generation
server.Status.LiveMotd = server.Spec.Motd.Failed
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg) r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
r.setCondition(server, v1alpha1.ConditionProvisioned, metav1.ConditionFalse, reason, msg) r.setCondition(server, v1alpha1.ConditionProvisioned, metav1.ConditionFalse, reason, msg)
} }
+119
View File
@@ -0,0 +1,119 @@
package operator_test
import (
"context"
"testing"
"time"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
)
// requestStartRetry is felis-api's RetryStart as the operator sees it.
func requestStartRetry(t *testing.T, c client.Client) {
t.Helper()
s := getServer(t, c, "survival")
patch := client.MergeFrom(s.DeepCopy())
if s.Annotations == nil {
s.Annotations = map[string]string{}
}
s.Annotations[v1alpha1.AnnotationStartRetry] = "2026-07-01T12:00:00Z"
if err := c.Patch(context.Background(), s, patch); err != nil {
t.Fatalf("annotate: %v", err)
}
}
func podPresent(t *testing.T, c client.Client) bool {
t.Helper()
err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival-0"}, &corev1.Pod{})
if err != nil && !apierrors.IsNotFound(err) {
t.Fatalf("get pod: %v", err)
}
return err == nil
}
// Once the automatic restarts are spent, a retry request starts the server over
// with the whole budget back: the pod is recreated, the start re-anchored, and
// the next timeout is retried automatically again.
func TestStartRetryRevivesAServerThatGaveUp(t *testing.T) {
srv := runningServer()
srv.Spec.Startup.TimeoutSeconds = 30
srv.Spec.Motd.Failed = "broken"
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
clock := base
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
at := func(offset time.Duration) *v1alpha1.MinecraftServer {
clock = base.Add(offset)
reconcile(t, r, "survival")
return getServer(t, c, "survival")
}
recreatePod := func() {
if err := c.Create(context.Background(), gamePod()); err != nil {
t.Fatal(err)
}
}
// Spend the three automatic restarts (autorestart_test.go pins the schedule).
at(0)
for _, due := range []time.Duration{90 * time.Second, 240 * time.Second, 510 * time.Second} {
at(due)
recreatePod()
}
s := at(time.Hour)
if !v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("setup: not given up: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts)
}
if s.Status.LiveMotd != "broken" {
t.Fatalf("Failed advertises liveMotd %q, want the failed MOTD", s.Status.LiveMotd)
}
requestStartRetry(t, c)
s = at(2 * time.Hour)
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
t.Fatal("the retry request was left on the server")
}
if podPresent(t, c) {
t.Fatal("the retry kept the failed pod")
}
if s.Status.Phase != v1alpha1.PhaseStarting || s.Status.AutoRestarts != 0 || v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("after retry: phase=%s restarts=%d gaveUp=%v", s.Status.Phase, s.Status.AutoRestarts, v1alpha1.StartGaveUp(&s.Status))
}
if s.Status.StartRequestedAt == nil || !s.Status.StartRequestedAt.Time.Equal(base.Add(2*time.Hour)) {
t.Fatalf("start not re-anchored at the retry: %v", s.Status.StartRequestedAt)
}
recreatePod()
// The budget is really back: this start times out and is retried on its own.
if s := at(2*time.Hour + 89*time.Second); s.Status.Phase != v1alpha1.PhaseFailed || v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("retried start timing out: phase=%s gaveUp=%v, want Failed with retries left", s.Status.Phase, v1alpha1.StartGaveUp(&s.Status))
}
if s := at(2*time.Hour + 90*time.Second); s.Status.AutoRestarts != 1 || podPresent(t, c) {
t.Fatalf("retried start: restarts=%d pod=%v, want the first automatic restart", s.Status.AutoRestarts, podPresent(t, c))
}
}
// A request that finds the server anywhere but Failed is stale (the start
// already recovered, or someone stopped it): it is removed and nothing else.
func TestStaleStartRetryIsOnlyRemoved(t *testing.T) {
srv := runningServer()
srv.Annotations = map[string]string{v1alpha1.AnnotationStartRetry: "2026-07-01T12:00:00Z"}
anchor := metav1.NewTime(fixedNow().Add(-10 * time.Second))
srv.Status = v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting, AutoRestarts: 2, StartRequestedAt: &anchor}
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
reconcile(t, r, "survival")
s := getServer(t, c, "survival")
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
t.Fatal("the stale retry request was left on the server")
}
if !podPresent(t, c) || s.Status.AutoRestarts != 2 || !s.Status.StartRequestedAt.Equal(&anchor) {
t.Fatalf("a stale request touched the start: pod=%v restarts=%d anchor=%v",
podPresent(t, c), s.Status.AutoRestarts, s.Status.StartRequestedAt)
}
}
+13 -1
View File
@@ -1,5 +1,5 @@
import { describe, it, expect } from "vitest"; import { describe, it, expect } from "vitest";
import { PHASE_COLOR, phaseColor, phaseVariant } from "@/components/PhaseBadge"; import { PHASE_COLOR, phaseColor, phaseVariant, startFailure } from "@/components/PhaseBadge";
import type { Phase } from "@/lib/types"; import type { Phase } from "@/lib/types";
// The closed lifecycle set felis-api can emit — the Go source of truth is // The closed lifecycle set felis-api can emit — the Go source of truth is
@@ -47,3 +47,15 @@ describe("unmodelled phase fallback", () => {
expect(phaseVariant(future)).toBe(phaseVariant("Unknown")); expect(phaseVariant(future)).toBe(phaseVariant("Unknown"));
}); });
}); });
describe("start failure", () => {
// A Failed phase only reads as a failed start while the server still wants to
// run; a stopped Failed server needs nothing from its owner.
it("reads a failed start only while the server wants to run", () => {
expect(startFailure({ phase: "Running", desiredState: "Running" })).toBeNull();
expect(startFailure({ phase: "Failed", desiredState: "Stopped" })).toBeNull();
expect(startFailure({ phase: "Failed" })).toBeNull();
expect(startFailure({ phase: "Failed", desiredState: "Running" })).toBe("retrying");
expect(startFailure({ phase: "Failed", desiredState: "Running", startGaveUp: true })).toBe("gaveUp");
});
});
+40 -4
View File
@@ -52,18 +52,54 @@ export function phaseVariant(phase: Phase): BadgeVariant {
return VARIANT[phase] ?? VARIANT.Unknown; return VARIANT[phase] ?? VARIANT.Unknown;
} }
export function PhaseBadge({ phase }: { phase: Phase }) { /** MAX_AUTO_RESTARTS mirrors the operator's v1alpha1.MaxAutoRestarts: how often it
* recreates the pod of a start that timed out before leaving it Failed. */
export const MAX_AUTO_RESTARTS = 3;
export type StartFailure = "retrying" | "gaveUp";
/** startFailure reads how a Failed start stands from the owner detail. "retrying"
* means the operator's next automatic attempt is coming; "gaveUp" means nothing
* will start it again until a person does. A Failed server meant to stop, or a
* public view without the owner detail, reads as null: there is no retry to
* offer there, and the badge stays the plain Failed one. */
export function startFailure(s: {
phase?: Phase;
desiredState?: string;
startGaveUp?: boolean;
}): StartFailure | null {
if (s.phase !== "Failed" || s.desiredState !== "Running") return null;
return s.startGaveUp ? "gaveUp" : "retrying";
}
export function PhaseBadge({
phase,
failure = null,
autoRestarts = 0,
}: {
phase: Phase;
/** From startFailure: a Failed start between automatic retries pulses and says so. */
failure?: StartFailure | null;
autoRestarts?: number;
}) {
const { t } = useTranslation(); const { t } = useTranslation();
const retrying = phase === "Failed" && failure === "retrying";
const hint =
phase !== "Failed" || failure === null
? undefined
: retrying
? t("servers:server_failed_retrying_body", { used: autoRestarts, max: MAX_AUTO_RESTARTS })
: t("servers:server_failed_body");
return ( return (
<Badge variant={phaseVariant(phase)} className="gap-1.5 whitespace-nowrap"> <Badge variant={phaseVariant(phase)} className="gap-1.5 whitespace-nowrap" title={hint}>
<span <span
className={cn( className={cn(
"h-1.5 w-1.5 rounded-full", "h-1.5 w-1.5 rounded-full",
TRANSIENT.has(phase) && "animate-pulse", (TRANSIENT.has(phase) || retrying) && "animate-pulse",
)} )}
style={{ backgroundColor: phaseColor(phase) }} style={{ backgroundColor: phaseColor(phase) }}
/> />
{t(PHASE_KEY[phase] ?? PHASE_KEY.Unknown)} {retrying ? t("servers:phase_failed_retrying") : t(PHASE_KEY[phase] ?? PHASE_KEY.Unknown)}
</Badge> </Badge>
); );
} }
+29
View File
@@ -103,4 +103,33 @@ describe("PowerButton", () => {
expect(wake).toHaveBeenCalledOnce(); expect(wake).toHaveBeenCalledOnce();
}); });
it("offers a failed server a retry, which is the wake", async () => {
wake.mockResolvedValue(undefined);
const onChanged = vi.fn();
render(<PowerButton name="lobby" live={false} failed onChanged={onChanged} />);
expect(screen.queryByRole("button", { name: t("servers:wake") })).toBeNull();
await userEvent.click(screen.getByRole("button", { name: t("servers:retry_start") }));
expect(wake).toHaveBeenCalledWith("lobby");
expect(stop).not.toHaveBeenCalled();
expect(onChanged).toHaveBeenCalledOnce();
});
it("stops a failed server without asking, and only the pressed button spins", async () => {
let resolve!: () => void;
stop.mockReturnValue(new Promise<void>((r) => (resolve = r)));
const onChanged = vi.fn();
render(<PowerButton name="lobby" live={false} failed playerCountUnknown onChanged={onChanged} />);
await userEvent.click(screen.getByRole("button", { name: t("servers:stop") }));
expect(stop).toHaveBeenCalledWith("lobby");
expect(screen.queryByText(t("servers:stop_confirm_unknown"))).toBeNull();
expect(screen.getByRole("button", { name: t("servers:stopping") })).toHaveProperty("disabled", true);
expect(screen.getByRole("button", { name: t("servers:retry_start") })).toHaveProperty("disabled", true);
resolve();
await vi.waitFor(() => expect(onChanged).toHaveBeenCalledOnce());
});
}); });
+34 -25
View File
@@ -1,5 +1,5 @@
import { useState } from "react"; import { useState } from "react";
import { Loader2, Play, Square } from "lucide-react"; import { Loader2, Play, RotateCcw, Square } from "lucide-react";
import { useTranslation } from "react-i18next"; import { useTranslation } from "react-i18next";
import { Button } from "@/components/ui/button"; import { Button } from "@/components/ui/button";
import { api, humanizeError } from "@/lib/api"; import { api, humanizeError } from "@/lib/api";
@@ -9,6 +9,8 @@ interface Props {
name: string; name: string;
/** A pod is up or on its way (Starting/Running/Stopping): offer Stop. */ /** A pod is up or on its way (Starting/Running/Stopping): offer Stop. */
live: boolean; live: boolean;
/** Its start Failed while meant to run: offer a retry and a stop. */
failed?: boolean;
playersOnline?: number; playersOnline?: number;
/** The operator cannot read the player count, so players may be online. */ /** The operator cannot read the player count, so players may be online. */
playerCountUnknown?: boolean; playerCountUnknown?: boolean;
@@ -21,10 +23,14 @@ interface Props {
// PowerButton starts or stops one server. It is busy while the call runs (no // PowerButton starts or stops one server. It is busy while the call runs (no
// double send), shows why a call was refused (quota, cooldown, a phase that // double send), shows why a call was refused (quota, cooldown, a phase that
// moved on), and asks before a stop that would disconnect players: the count // moved on), and asks before a stop that would disconnect players: the count
// is in the question, and an unreadable count asks too. // is in the question, and an unreadable count asks too. A server whose start
// failed gets both ways out: retry (the wake, which felis-api turns into a fresh
// start) and stop. Nobody is on a server that never came up, so that stop does
// not ask.
export function PowerButton({ export function PowerButton({
name, name,
live, live,
failed = false,
playersOnline, playersOnline,
playerCountUnknown, playerCountUnknown,
onChanged, onChanged,
@@ -32,13 +38,13 @@ export function PowerButton({
className, className,
}: Props) { }: Props) {
const { t } = useTranslation("servers"); const { t } = useTranslation("servers");
const [busy, setBusy] = useState(false); const [busy, setBusy] = useState<"wake" | "stop" | null>(null);
const [confirming, setConfirming] = useState(false); const [confirming, setConfirming] = useState(false);
const [error, setError] = useState<string | null>(null); const [error, setError] = useState<string | null>(null);
async function run(kind: "wake" | "stop") { async function run(kind: "wake" | "stop") {
if (busy) return; if (busy) return;
setBusy(true); setBusy(kind);
setError(null); setError(null);
try { try {
await (kind === "wake" ? api.wake(name) : api.stop(name)); await (kind === "wake" ? api.wake(name) : api.stop(name));
@@ -47,20 +53,36 @@ export function PowerButton({
} catch (e) { } catch (e) {
setError(humanizeError(e)); setError(humanizeError(e));
} finally { } finally {
setBusy(false); setBusy(null);
} }
} }
const players = playersOnline ?? 0; const players = playersOnline ?? 0;
const askFirst = players > 0 || playerCountUnknown === true; const askFirst = players > 0 || playerCountUnknown === true;
const spinner = <Loader2 className="animate-spin" />; const spinner = <Loader2 className="animate-spin" />;
const stopButton = (variant: "destructive" | "outline", onClick: () => void) => (
<Button size={size} variant={variant} onClick={onClick} disabled={busy !== null}>
{busy === "stop" ? spinner : <Square />}
{busy === "stop" ? t("stopping") : t("stop")}
</Button>
);
let control; let control;
if (!live) { if (failed) {
control = ( control = (
<Button size={size} onClick={() => void run("wake")} disabled={busy}> <div className="flex flex-wrap items-center justify-end gap-2">
{busy ? spinner : <Play />} <Button size={size} onClick={() => void run("wake")} disabled={busy !== null}>
{busy ? t("waking") : t("wake")} {busy === "wake" ? spinner : <RotateCcw />}
{busy === "wake" ? t("retrying_start") : t("retry_start")}
</Button>
{stopButton("outline", () => void run("stop"))}
</div>
);
} else if (!live) {
control = (
<Button size={size} onClick={() => void run("wake")} disabled={busy !== null}>
{busy === "wake" ? spinner : <Play />}
{busy === "wake" ? t("waking") : t("wake")}
</Button> </Button>
); );
} else if (confirming) { } else if (confirming) {
@@ -71,27 +93,14 @@ export function PowerButton({
? t("stop_confirm_players", { count: players }) ? t("stop_confirm_players", { count: players })
: t("stop_confirm_unknown")} : t("stop_confirm_unknown")}
</span> </span>
<Button size={size} variant="ghost" onClick={() => setConfirming(false)} disabled={busy}> <Button size={size} variant="ghost" onClick={() => setConfirming(false)} disabled={busy !== null}>
{t("common:cancel")} {t("common:cancel")}
</Button> </Button>
<Button size={size} variant="destructive" onClick={() => void run("stop")} disabled={busy}> {stopButton("destructive", () => void run("stop"))}
{busy ? spinner : <Square />}
{busy ? t("stopping") : t("stop")}
</Button>
</div> </div>
); );
} else { } else {
control = ( control = stopButton("destructive", () => (askFirst ? setConfirming(true) : void run("stop")));
<Button
size={size}
variant="destructive"
onClick={() => (askFirst ? setConfirming(true) : void run("stop"))}
disabled={busy}
>
{busy ? spinner : <Square />}
{busy ? t("stopping") : t("stop")}
</Button>
);
} }
return ( return (
+7 -2
View File
@@ -11,7 +11,8 @@
"phase_starting": "Starting", "phase_starting": "Starting",
"phase_stopping": "Stopping", "phase_stopping": "Stopping",
"phase_stopped": "Stopped", "phase_stopped": "Stopped",
"phase_failed": "Failed", "phase_failed": "Failed to start",
"phase_failed_retrying": "Timed out · retrying",
"phase_unknown": "Unknown", "phase_unknown": "Unknown",
"online": "online", "online": "online",
"console": "Console", "console": "Console",
@@ -23,12 +24,16 @@
"stopping": "Stopping…", "stopping": "Stopping…",
"wake": "Wake", "wake": "Wake",
"waking": "Waking…", "waking": "Waking…",
"retry_start": "Retry start",
"retrying_start": "Retrying…",
"server_asleep_title": "Server is asleep", "server_asleep_title": "Server is asleep",
"server_asleep_body": "Wake it to boot a pod — the live console attaches automatically the moment it starts up.", "server_asleep_body": "Wake it to boot a pod — the live console attaches automatically the moment it starts up.",
"server_shutting_down_title": "Server is shutting down", "server_shutting_down_title": "Server is shutting down",
"server_shutting_down_body": "The pod is terminating, so the live console has detached. It will be asleep in a moment.", "server_shutting_down_body": "The pod is terminating, so the live console has detached. It will be asleep in a moment.",
"server_failed_title": "Server failed to start", "server_failed_title": "Server failed to start",
"server_failed_body": "No pod is running, so there is nothing to stream. Wake it to retry; the console attaches once a pod comes back up.", "server_failed_body": "Its automatic retries are spent. The console holds the log of the last attempt; fix the cause, then press Retry start.",
"server_failed_retrying_title": "Start timed out — retrying",
"server_failed_retrying_body": "The operator recreates the pod and tries again ({{used}} of {{max}} retries used). Press Retry start to go now instead.",
"console_offline_title": "Console is offline", "console_offline_title": "Console is offline",
"console_offline_body": "The live console attaches automatically as soon as the server is running.", "console_offline_body": "The live console attaches automatically as soon as the server is running.",
"my_servers_breadcrumb": "My servers", "my_servers_breadcrumb": "My servers",
+6 -1
View File
@@ -12,6 +12,7 @@
"phase_stopping": "停止中", "phase_stopping": "停止中",
"phase_stopped": "已停止", "phase_stopped": "已停止",
"phase_failed": "启动失败", "phase_failed": "启动失败",
"phase_failed_retrying": "启动超时·重试中",
"phase_unknown": "未知", "phase_unknown": "未知",
"online": "人在线", "online": "人在线",
"console": "控制台", "console": "控制台",
@@ -23,12 +24,16 @@
"stopping": "停止中…", "stopping": "停止中…",
"wake": "启动", "wake": "启动",
"waking": "启动中…", "waking": "启动中…",
"retry_start": "重试启动",
"retrying_start": "重试中…",
"server_asleep_title": "服务器已休眠", "server_asleep_title": "服务器已休眠",
"server_asleep_body": "启动后将拉起 Pod,实时控制台会自动连接。", "server_asleep_body": "启动后将拉起 Pod,实时控制台会自动连接。",
"server_shutting_down_title": "服务器正在关闭", "server_shutting_down_title": "服务器正在关闭",
"server_shutting_down_body": "Pod 正在终止,实时控制台已断开。关闭完成后将进入休眠。", "server_shutting_down_body": "Pod 正在终止,实时控制台已断开。关闭完成后将进入休眠。",
"server_failed_title": "服务器启动失败", "server_failed_title": "服务器启动失败",
"server_failed_body": "当前无运行中的 Pod,无法获取日志流。重新启动后控制台将自动连接。", "server_failed_body": "自动重试已用完。控制台里是最后一次启动的日志,找到原因修好后点「重试启动」。",
"server_failed_retrying_title": "启动超时,正在自动重试",
"server_failed_retrying_body": "operator 会重建 Pod 再试(已重试 {{used}}/{{max}} 次)。不想等可以点「重试启动」立即再来。",
"console_offline_title": "控制台未连接", "console_offline_title": "控制台未连接",
"console_offline_body": "服务器运行后,实时控制台将自动连接。", "console_offline_body": "服务器运行后,实时控制台将自动连接。",
"my_servers_breadcrumb": "我的服务器", "my_servers_breadcrumb": "我的服务器",
+13 -3
View File
@@ -167,7 +167,7 @@ export interface paths {
put?: never; put?: never;
/** /**
* Domain-autostart wake driven by velocity for a joining player (spec §9.1). * Domain-autostart wake driven by velocity for a joining player (spec §9.1).
* @description velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the server's autostartPolicy and the per-server wake cooldown. * @description velocity holds no web Principal, so it drives the wake lever with its service token, identifying the player by online-mode UUID. Gated by the server's autostartPolicy and the per-server wake cooldown. A server whose start failed with its automatic retries spent answers 409 start_failed: a join never resets the retry budget, so velocity queues no one for it.
*/ */
post: operations["internalWake"]; post: operations["internalWake"];
delete?: never; delete?: never;
@@ -419,7 +419,10 @@ export interface paths {
}; };
get?: never; get?: never;
put?: never; put?: never;
/** Wake your own server. */ /**
* Wake your own server.
* @description On a server whose start Failed (phase Failed, desiredState Running) this is a retry: the operator recreates the pod and starts it over with a fresh automatic-restart budget. It is audited as retry_start.
*/
post: operations["wake"]; post: operations["wake"];
delete?: never; delete?: never;
options?: never; options?: never;
@@ -2398,6 +2401,13 @@ export interface components {
autostartPolicy?: "ownerOnly" | "public" | "allowlist"; autostartPolicy?: "ownerOnly" | "public" | "allowlist";
/** @description Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players. */ /** @description Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players. */
playerCountUnknown?: boolean; playerCountUnknown?: boolean;
/**
* Format: int32
* @description Owned rows only. How often the operator recreated the pod of this start after it timed out.
*/
autoRestarts?: number;
/** @description Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over. */
startGaveUp?: boolean;
}; };
/** @description One world backup (internal/api/repo.go BackupView). backup_ref is withheld (spec §286). */ /** @description One world backup (internal/api/repo.go BackupView). backup_ref is withheld (spec §286). */
BackupView: { BackupView: {
@@ -3148,7 +3158,7 @@ export interface operations {
401: components["responses"]["Unauthorized"]; 401: components["responses"]["Unauthorized"];
403: components["responses"]["Forbidden"]; 403: components["responses"]["Forbidden"];
404: components["responses"]["NotFound"]; 404: components["responses"]["NotFound"];
/** @description A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started. */ /** @description Nothing was started. maintenance_in_progress: a restore, backup or file write holds the server's world volume. start_failed: the last start failed and its automatic retries are spent (ServerInfo.startGaveUp); the server stays down until a person starts it from the panel. */
409: { 409: {
headers: { headers: {
[name: string]: unknown; [name: string]: unknown;
+4
View File
@@ -36,6 +36,10 @@ export interface MyServerView {
autostartPolicy?: AutostartPolicy; autostartPolicy?: AutostartPolicy;
/** True while the operator cannot read the player count; a stop may drop players. */ /** True while the operator cannot read the player count; a stop may drop players. */
playerCountUnknown?: boolean; playerCountUnknown?: boolean;
/** How often the operator recreated the pod of this start after it timed out. */
autoRestarts?: number;
/** True for a Failed server no automatic retry will bring up. */
startGaveUp?: boolean;
} }
/** ServerStatus is GET /servers/{name}/status (Go ServerInfo). It never carries /** ServerStatus is GET /servers/{name}/status (Go ServerInfo). It never carries
+78
View File
@@ -0,0 +1,78 @@
// @vitest-environment jsdom
import { describe, it, expect, vi, beforeEach } from "vitest";
import { render, screen, within } from "@testing-library/react";
import { MemoryRouter, Route, Routes } from "react-router-dom";
import { ServerConsole } from "./ServerConsole";
const calls = vi.hoisted(() => ({ status: vi.fn(), myServers: vi.fn() }));
vi.mock("@/lib/tier", () => ({
useTier: () => ({
loading: false,
identity: { user_id: "admin-1", email: "[email protected]", role: "admin" },
isAdmin: true,
isOwner: false,
}),
}));
vi.mock("@/lib/config", async (importActual) => {
const actual = await importActual<typeof import("@/lib/config")>();
return {
...actual,
loadConfig: () => Promise.resolve({ apiBase: "/api/v1", rootDomain: "example.test", gamePort: 25570 }),
};
});
vi.mock("@/lib/api", async (importActual) => {
const actual = await importActual<typeof import("@/lib/api")>();
return { ...actual, api: { ...actual.api, ...calls } };
});
// The live log is a WebSocket; here it only matters whether the page shows it.
vi.mock("@/components/LogConsole", () => ({ LogConsole: () => <div data-testid="log-stream" /> }));
beforeEach(() => {
calls.status.mockReset();
calls.myServers.mockReset();
calls.myServers.mockResolvedValue([]);
});
function renderConsole() {
render(
<MemoryRouter initialEntries={["/servers/survival"]}>
<Routes>
<Route path="/servers/:name" element={<ServerConsole />} />
</Routes>
</MemoryRouter>,
);
}
function status(over: Record<string, unknown>) {
return { name: "survival", subdomain: "survival", phase: "Failed", ready: false, playersOnline: 0, playersMax: 20, ...over };
}
describe("ServerConsole failed start", () => {
it("shows the failed attempt's log under what the owner can do once retries are spent", async () => {
calls.status.mockResolvedValue(status({ desiredState: "Running", startGaveUp: true, autoRestarts: 3 }));
renderConsole();
const notice = await screen.findByRole("status", { name: "Server failed to start" });
expect(within(notice).getByText(/automatic retries are spent/)).toBeTruthy();
expect(await screen.findByTestId("log-stream")).toBeTruthy();
expect(screen.getByRole("button", { name: /Retry start/ })).toBeTruthy();
});
it("counts the retries used while the operator is still retrying", async () => {
calls.status.mockResolvedValue(status({ desiredState: "Running", autoRestarts: 1 }));
renderConsole();
const notice = await screen.findByRole("status", { name: "Start timed out — retrying" });
expect(within(notice).getByText(/1 of 3 retries used/)).toBeTruthy();
expect(screen.getByText("Timed out · retrying")).toBeTruthy();
});
it("asks nothing of the owner once a failed server is being stopped", async () => {
calls.status.mockResolvedValue(status({ desiredState: "Stopped" }));
renderConsole();
expect(await screen.findByTestId("log-stream")).toBeTruthy();
expect(screen.queryByRole("status", { name: /failed to start|timed out/ })).toBeNull();
expect(screen.queryByRole("button", { name: /Retry start/ })).toBeNull();
});
});
+41 -12
View File
@@ -1,10 +1,10 @@
import { useState, useRef, useCallback, useLayoutEffect, type KeyboardEvent } from "react"; import { useState, useRef, useCallback, useId, useLayoutEffect, type KeyboardEvent } from "react";
import { Link, useParams } from "react-router-dom"; import { Link, useParams } from "react-router-dom";
import { Terminal, Moon, Shield, ShieldAlert, HelpCircle, Loader2, Users, Archive, FolderOpen, ChevronRight, type LucideIcon } from "lucide-react"; import { Terminal, Moon, Shield, ShieldAlert, HelpCircle, Loader2, Users, Archive, FolderOpen, ChevronRight, type LucideIcon } from "lucide-react";
import { useTranslation } from "react-i18next"; import { useTranslation } from "react-i18next";
import { Card, CardContent } from "@/components/ui/card"; import { Card, CardContent } from "@/components/ui/card";
import { BackLink } from "@/components/BackLink"; import { BackLink } from "@/components/BackLink";
import { PhaseBadge } from "@/components/PhaseBadge"; import { MAX_AUTO_RESTARTS, PhaseBadge, startFailure, type StartFailure } from "@/components/PhaseBadge";
import { PageHeader } from "@/components/PageHeader"; import { PageHeader } from "@/components/PageHeader";
import { LogConsole } from "@/components/LogConsole"; import { LogConsole } from "@/components/LogConsole";
import { Loading, ErrorState } from "@/components/States"; import { Loading, ErrorState } from "@/components/States";
@@ -39,12 +39,6 @@ function useNotStreamingCopy(phase: Phase): { icon: LucideIcon; title: string; b
title: t("server_shutting_down_title"), title: t("server_shutting_down_title"),
body: t("server_shutting_down_body"), body: t("server_shutting_down_body"),
}; };
case "Failed":
return {
icon: ShieldAlert,
title: t("server_failed_title"),
body: t("server_failed_body"),
};
default: default:
return { return {
icon: HelpCircle, icon: HelpCircle,
@@ -67,6 +61,34 @@ function NotStreaming({ phase }: { phase: Phase }) {
); );
} }
// FailedStartNotice heads the console of a server whose start failed: whether the
// operator will try again on its own, and what the owner can do. The log below it
// is the attempt that failed, which is what they need to find the cause.
function FailedStartNotice({ failure, autoRestarts }: { failure: StartFailure; autoRestarts: number }) {
const { t } = useTranslation("servers");
const retrying = failure === "retrying";
const titleId = useId();
return (
<div
role="status"
aria-labelledby={titleId}
className="flex items-start gap-3 border-b border-zinc-800 bg-red-950/40 px-4 py-3 text-sm text-zinc-300"
>
<ShieldAlert className="mt-0.5 h-4 w-4 shrink-0 text-red-400" />
<div className="space-y-0.5">
<p id={titleId} className="font-medium text-zinc-100">
{retrying ? t("server_failed_retrying_title") : t("server_failed_title")}
</p>
<p>
{retrying
? t("server_failed_retrying_body", { used: autoRestarts, max: MAX_AUTO_RESTARTS })
: t("server_failed_body")}
</p>
</div>
</div>
);
}
function loadHistory(name: string): string[] { function loadHistory(name: string): string[] {
try { try {
const raw = localStorage.getItem(HISTORY_KEY(name)); const raw = localStorage.getItem(HISTORY_KEY(name));
@@ -206,8 +228,11 @@ export function ServerConsole() {
// during BOTH Starting and Running — the operator marks Starting once the pod // during BOTH Starting and Running — the operator marks Starting once the pod
// is up but RCON is not yet reachable (it only flips to Running after RCON // is up but RCON is not yet reachable (it only flips to Running after RCON
// readiness). Boot logs flow precisely in that Starting window, which is when a // readiness). Boot logs flow precisely in that Starting window, which is when a
// read most wants them, so the gate streams for both, not Running alone. // read most wants them, so the gate streams for both, not Running alone. A
const streamable = data?.phase === "Running" || data?.phase === "Starting"; // Failed start usually leaves its pod behind, and that pod's log is how the
// owner finds out why it failed, so Failed streams too.
const streamable = data?.phase === "Running" || data?.phase === "Starting" || data?.phase === "Failed";
const failure = data ? startFailure(data) : null;
return ( return (
<div className="flex flex-col lg:h-[calc(100vh-3.5rem)] lg:min-h-[35rem] gap-4 min-h-0"> <div className="flex flex-col lg:h-[calc(100vh-3.5rem)] lg:min-h-[35rem] gap-4 min-h-0">
@@ -227,10 +252,11 @@ export function ServerConsole() {
subtitle={cfg ? <CopyAddress address={joinAddress(data.subdomain, cfg)} /> : undefined} subtitle={cfg ? <CopyAddress address={joinAddress(data.subdomain, cfg)} /> : undefined}
actions={ actions={
<div className="flex items-center gap-2"> <div className="flex items-center gap-2">
<PhaseBadge phase={data.phase} /> <PhaseBadge phase={data.phase} failure={failure} autoRestarts={data.autoRestarts} />
<PowerButton <PowerButton
name={name} name={name}
live={streamable || data.phase === "Stopping"} live={data.phase === "Running" || data.phase === "Starting" || data.phase === "Stopping"}
failed={failure !== null}
playersOnline={data.playersOnline} playersOnline={data.playersOnline}
playerCountUnknown={data.playerCountUnknown} playerCountUnknown={data.playerCountUnknown}
onChanged={reload} onChanged={reload}
@@ -251,6 +277,9 @@ export function ServerConsole() {
</div> </div>
) : cfg ? ( ) : cfg ? (
<div className="flex-1 flex flex-col lg:min-h-0 min-h-0 bg-black"> <div className="flex-1 flex flex-col lg:min-h-0 min-h-0 bg-black">
{failure && (
<FailedStartNotice failure={failure} autoRestarts={data.autoRestarts ?? 0} />
)}
<LogConsole <LogConsole
key={name} key={name}
url={consoleStreamURL(cfg.apiBase, name)} url={consoleStreamURL(cfg.apiBase, name)}
@@ -139,3 +139,27 @@ describe("ServersPage addresses", () => {
expect(survival.getByRole("button", { name: "Copy address survival.example.test:25570" })).toBeTruthy(); expect(survival.getByRole("button", { name: "Copy address survival.example.test:25570" })).toBeTruthy();
}); });
}); });
describe("ServersPage failed starts", () => {
it("offers a failed start a retry and a stop, and tells retrying from given up", async () => {
calls.fleet.mockResolvedValue([
row("broken", { phase: "Failed", desiredState: "Running", startGaveUp: true, autoRestarts: 3, owned: true }),
row("flaky", { phase: "Failed", desiredState: "Running", autoRestarts: 1, owned: true }),
]);
render(
<MemoryRouter>
<ServersPage />
</MemoryRouter>,
);
const broken = await tableRow("broken");
expect(broken.getByText("Failed to start")).toBeTruthy();
expect(broken.getByRole("button", { name: /Retry start/ })).toBeTruthy();
expect(broken.getByRole("button", { name: /Stop/ })).toBeTruthy();
expect(broken.queryByRole("button", { name: /Wake/ })).toBeNull();
const flaky = await tableRow("flaky");
expect(flaky.getByText("Timed out · retrying")).toBeTruthy();
expect(flaky.getByRole("button", { name: /Retry start/ })).toBeTruthy();
});
});
+10 -3
View File
@@ -30,7 +30,7 @@ import {
SelectContent, SelectContent,
SelectItem, SelectItem,
} from "@/components/ui/select"; } from "@/components/ui/select";
import { PhaseBadge, PHASE_KEY, PHASE_COLOR } from "@/components/PhaseBadge"; import { PhaseBadge, PHASE_KEY, PHASE_COLOR, startFailure } from "@/components/PhaseBadge";
import { PowerButton } from "@/components/PowerButton"; import { PowerButton } from "@/components/PowerButton";
import { Loading, ErrorState, EmptyState } from "@/components/States"; import { Loading, ErrorState, EmptyState } from "@/components/States";
import { Pagination } from "@/components/Pagination"; import { Pagination } from "@/components/Pagination";
@@ -81,6 +81,8 @@ interface UnifiedServer {
playersOnline: number; playersOnline: number;
playersMax: number; playersMax: number;
playerCountUnknown?: boolean; playerCountUnknown?: boolean;
autoRestarts?: number;
startGaveUp?: boolean;
owner?: string; owner?: string;
endpointAddress?: string | null; endpointAddress?: string | null;
claimable?: boolean; claimable?: boolean;
@@ -133,6 +135,8 @@ export function ServersPage() {
playersOnline: s.playersOnline, playersOnline: s.playersOnline,
playersMax: s.playersMax, playersMax: s.playersMax,
playerCountUnknown: s.playerCountUnknown, playerCountUnknown: s.playerCountUnknown,
autoRestarts: s.autoRestarts,
startGaveUp: s.startGaveUp,
owner: s.owner, owner: s.owner,
endpointAddress: s.endpointAddress, endpointAddress: s.endpointAddress,
claimable: s.claimable, claimable: s.claimable,
@@ -152,6 +156,8 @@ export function ServersPage() {
playersOnline: s.playersOnline, playersOnline: s.playersOnline,
playersMax: s.playersMax, playersMax: s.playersMax,
playerCountUnknown: s.playerCountUnknown, playerCountUnknown: s.playerCountUnknown,
autoRestarts: s.autoRestarts,
startGaveUp: s.startGaveUp,
owner: s.owned ? t("servers:owned_filter_mine") || "me" : undefined, owner: s.owned ? t("servers:owned_filter_mine") || "me" : undefined,
claimable: s.claimable, claimable: s.claimable,
owned: s.owned, owned: s.owned,
@@ -509,6 +515,7 @@ function ServerActions({
<PowerButton <PowerButton
name={server.name} name={server.name}
live={live} live={live}
failed={startFailure(server) !== null}
playersOnline={server.playersOnline} playersOnline={server.playersOnline}
playerCountUnknown={server.playerCountUnknown} playerCountUnknown={server.playerCountUnknown}
onChanged={onChanged} onChanged={onChanged}
@@ -610,7 +617,7 @@ function ServerRow({
<td className="px-4 py-3 align-middle text-left"> <td className="px-4 py-3 align-middle text-left">
<div className="flex items-center gap-2"> <div className="flex items-center gap-2">
<ServerName server={server} /> <ServerName server={server} />
<PhaseBadge phase={server.phase} /> <PhaseBadge phase={server.phase} failure={startFailure(server)} autoRestarts={server.autoRestarts} />
</div> </div>
{address && <CopyAddress address={address} className="md:max-w-[16rem]" />} {address && <CopyAddress address={address} className="md:max-w-[16rem]" />}
</td> </td>
@@ -669,7 +676,7 @@ function ServerMobileCard({
<div className="flex min-w-0 items-baseline gap-2"> <div className="flex min-w-0 items-baseline gap-2">
<ServerName server={server} /> <ServerName server={server} />
</div> </div>
<PhaseBadge phase={server.phase} /> <PhaseBadge phase={server.phase} failure={startFailure(server)} autoRestarts={server.autoRestarts} />
</div> </div>
{/* Its own line: the join address is what a player came for, so it {/* Its own line: the join address is what a player came for, so it
gets the card's full width rather than sharing it with the badge. */} gets the card's full width rather than sharing it with the badge. */}
@@ -20,12 +20,14 @@ import java.util.Optional;
* backend or the wake lever. * backend or the wake lever.
* *
* <p>This is the read-only, phase-aware subset of the responsibility: the MOTD is * <p>This is the read-only, phase-aware subset of the responsibility: the MOTD is
* synthesized from the server's lifecycle (online / starting / sleeping) and its * synthesized from the server's lifecycle (online / starting / start failed /
* cached player counts. Mirroring each backend's <em>own</em> MOTD string (by * sleeping) and its cached player counts. Mirroring each backend's <em>own</em> MOTD
* pinging ready servers in the background and caching the result) is a richer * string (by pinging ready servers in the background and caching the result) is a richer
* variant deferred to a later slice; nothing here ever pings a sleeping backend. * variant deferred to a later slice; nothing here ever pings a sleeping backend.
*/ */
public final class MotdResponder { public final class MotdResponder {
private static final String PHASE_FAILED = "Failed";
private final ServerRegistry registry; private final ServerRegistry registry;
MotdResponder(ServerRegistry registry) { MotdResponder(ServerRegistry registry) {
@@ -55,21 +57,33 @@ public final class MotdResponder {
} }
// The server-list ping carries no client locale, so the MOTD status uses the // The server-list ping carries no client locale, so the MOTD status uses the
// both-languages-in-one-line pattern the modded /link clients share. // both-languages-in-one-line pattern the modded /link clients share. A start
private static String statusLine(ServerView v) { // that failed still holds desiredState Running, so it is read first: joining a
// server whose retries are spent wakes nothing, and one between retries is
// waiting out a backoff, which a "starting…" line hid behind a queue that ran out.
static String statusLine(ServerView v) {
if (v.ready()) { if (v.ready()) {
return "在线 / online"; return "在线 / online";
} }
if (v.startGaveUp()) {
return "启动失败,等服主处理 / failed to start — the owner has to restart it";
}
if (PHASE_FAILED.equals(v.phase())) {
return "启动超时,稍后自动重试 / start timed out — retrying shortly";
}
if ("Running".equals(v.desiredState())) { if ("Running".equals(v.desiredState())) {
return "启动中… / starting…"; return "启动中… / starting…";
} }
return "休眠中,加入即唤醒 / sleeping — join to wake"; return "休眠中,加入即唤醒 / sleeping — join to wake";
} }
private static NamedTextColor statusColor(ServerView v) { static NamedTextColor statusColor(ServerView v) {
if (v.ready()) { if (v.ready()) {
return NamedTextColor.GREEN; return NamedTextColor.GREEN;
} }
if (v.startGaveUp()) {
return NamedTextColor.RED;
}
if ("Running".equals(v.desiredState())) { if ("Running".equals(v.desiredState())) {
return NamedTextColor.YELLOW; return NamedTextColor.YELLOW;
} }
@@ -505,11 +505,7 @@ public final class WaitingRouter {
// stopped the server — it says so and returns false. // stopped the server — it says so and returns false.
private boolean stillComing(Player player, boolean zh, Waiter w, ServerView status, long now) { private boolean stillComing(Player player, boolean zh, Waiter w, ServerView status, long now) {
if (status.startGaveUp()) { if (status.startGaveUp()) {
player.sendMessage(Component.text( tellStartFailed(player, zh, w.serverName);
zh ? "「" + w.serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。"
: "« " + w.serverName + " » failed to start and its automatic retries are spent. "
+ "The owner can check its log in the panel and start it again.",
NamedTextColor.RED));
return false; return false;
} }
if (DESIRED_STOPPED.equalsIgnoreCase(status.desiredState())) { if (DESIRED_STOPPED.equalsIgnoreCase(status.desiredState())) {
@@ -655,6 +651,12 @@ public final class WaitingRouter {
NamedTextColor.YELLOW)); NamedTextColor.YELLOW));
return; return;
} }
if ("start_failed".equals(e.errorCode())) {
// The last start failed with its retries spent. The join does
// not buy another round of them, so there is nothing to wait for.
tellStartFailed(player, zh, serverName);
return;
}
logWakeFailure(player, serverName, zh, e); logWakeFailure(player, serverName, zh, e);
return; return;
case 429: case 429:
@@ -688,6 +690,16 @@ public final class WaitingRouter {
waiting.put(id, new Waiter(serverName, clock.getAsLong(), fromMenu)); waiting.put(id, new Waiter(serverName, clock.getAsLong(), fromMenu));
} }
// The server's start failed and its automatic retries are spent: nothing more is
// coming until a person starts it again from the panel.
private static void tellStartFailed(Player player, boolean zh, String serverName) {
player.sendMessage(Component.text(
zh ? "「" + serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。"
: "« " + serverName + " » failed to start and its automatic retries are spent. "
+ "The owner can check its log in the panel and start it again.",
NamedTextColor.RED));
}
private void logWakeFailure(Player player, String serverName, boolean zh, LinkException e) { private void logWakeFailure(Player player, String serverName, boolean zh, LinkException e) {
log.warn("Felis: wake {} failed (status={}): {}", serverName, e.statusCode(), e.getMessage()); log.warn("Felis: wake {} failed (status={}): {}", serverName, e.statusCode(), e.getMessage());
player.sendMessage(Component.text( player.sendMessage(Component.text(
@@ -2,6 +2,8 @@ package best.lolicon.felis.velocity;
import best.lolicon.felis.link.ServerView; import best.lolicon.felis.link.ServerView;
import net.kyori.adventure.text.format.NamedTextColor;
import java.net.InetSocketAddress; import java.net.InetSocketAddress;
import java.util.List; import java.util.List;
import java.util.Optional; import java.util.Optional;
@@ -139,6 +141,24 @@ public final class ServerRegistryTest {
assertAddr("a non-numeric port is not a port", "backend:x", 25565, assertAddr("a non-numeric port is not a port", "backend:x", 25565,
ServerRegistry.parseAddress("backend:x")); ServerRegistry.parseAddress("backend:x"));
// The server-list MOTD reads a failed start before desiredState Running,
// which a failed start still holds.
assertEq("motd: up", "在线 / online", MotdResponder.statusLine(up("a", "a", "10.43.0.1:25565")));
assertEq("motd: starting", "启动中… / starting…", MotdResponder.statusLine(
new ServerView("a", "a", "Starting", false, "ownerOnly", "Running", "fallback", "login", 0, 0)));
ServerView backoff = new ServerView("a", "a", "Failed", false, "ownerOnly", "Running",
"fallback", "login", 0, 0, false, 1, false);
assertEq("motd: between retries", "启动超时,稍后自动重试 / start timed out — retrying shortly",
MotdResponder.statusLine(backoff));
assertEq("motd: between retries is yellow", NamedTextColor.YELLOW, MotdResponder.statusColor(backoff));
ServerView gaveUp = new ServerView("a", "a", "Failed", false, "ownerOnly", "Running",
"fallback", "login", 0, 0, false, 3, true);
assertEq("motd: retries spent", "启动失败,等服主处理 / failed to start — the owner has to restart it",
MotdResponder.statusLine(gaveUp));
assertEq("motd: retries spent is red", NamedTextColor.RED, MotdResponder.statusColor(gaveUp));
assertEq("motd: sleeping", "休眠中,加入即唤醒 / sleeping — join to wake",
MotdResponder.statusLine(down("a", "a")));
System.out.println("ServerRegistryTest OK (" + checks + " checks)"); System.out.println("ServerRegistryTest OK (" + checks + " checks)");
} }
@@ -219,6 +219,7 @@ public final class WaitingRouterTest {
String[][] cases = { String[][] cases = {
{"403 forbidden", "You're not allowed to start « gamma »", null}, {"403 forbidden", "You're not allowed to start « gamma »", null},
{"409 maintenance_in_progress", "« gamma » is under maintenance", null}, {"409 maintenance_in_progress", "« gamma » is under maintenance", null},
{"409 start_failed", "« gamma » failed to start and its automatic retries are spent", null},
{"409 conflict", "Couldn't start « gamma » right now", "wake gamma failed (status=409)"}, {"409 conflict", "Couldn't start « gamma » right now", "wake gamma failed (status=409)"},
{"503 at_capacity", "The cluster is at capacity right now", null}, {"503 at_capacity", "The cluster is at capacity right now", null},
{"503 unavailable", "Couldn't start « gamma » right now", "wake gamma failed (status=503)"}, {"503 unavailable", "Couldn't start « gamma » right now", "wake gamma failed (status=503)"},