fix(felis): 启动失败的服可在面板重试或停止,玩家入服时直接告知启动失败
This commit is contained in:
28 files changed
+775
-75
No files matched your search
@@ -1748,6 +1748,7 @@ type fakeCluster struct {
|
||||
wakeErr map[string]error
|
||||
acquired []string // "name:kind" per admitted AcquireMaintenance
|
||||
released []string // names per ReleaseMaintenance
|
||||
retried []string // names per RetryStart
|
||||
}
|
||||
|
||||
func newFakeCluster() *fakeCluster {
|
||||
@@ -1794,6 +1795,16 @@ func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.De
|
||||
c.desired[n] = s
|
||||
return nil
|
||||
}
|
||||
|
||||
// RetryStart records the retry on top of the plain start, so a test tells a
|
||||
// Failed server's retry apart from an ordinary wake.
|
||||
func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
|
||||
if err := c.SetDesiredState(ctx, n, v1alpha1.DesiredRunning); err != nil {
|
||||
return err
|
||||
}
|
||||
c.retried = append(c.retried, n)
|
||||
return nil
|
||||
}
|
||||
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
|
||||
if err := c.maintErr[n]; err != nil {
|
||||
return err
|
||||
@@ -2092,9 +2103,11 @@ func TestMyServersJoinsLiveState(t *testing.T) {
|
||||
cl := newFakeCluster()
|
||||
cl.list = []ServerInfo{
|
||||
{Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running",
|
||||
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true},
|
||||
AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true,
|
||||
AutoRestarts: 3, StartGaveUp: true},
|
||||
{Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running",
|
||||
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true},
|
||||
AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true,
|
||||
AutoRestarts: 3, StartGaveUp: true},
|
||||
}
|
||||
api := newTestAPI(repo, cl)
|
||||
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
|
||||
@@ -2122,7 +2135,8 @@ func TestMyServersJoinsLiveState(t *testing.T) {
|
||||
mine, open := got["mine"], got["open"]
|
||||
for k, want := range map[string]any{"displayName": "My World", "phase": "Running",
|
||||
"desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true,
|
||||
"playersOnline": float64(2), "playersMax": float64(20)} {
|
||||
"playersOnline": float64(2), "playersMax": float64(20),
|
||||
"autoRestarts": float64(3), "startGaveUp": true} {
|
||||
if mine[k] != want {
|
||||
t.Errorf("own row %s = %v, want %v", k, mine[k], want)
|
||||
}
|
||||
@@ -2130,7 +2144,7 @@ func TestMyServersJoinsLiveState(t *testing.T) {
|
||||
if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) {
|
||||
t.Errorf("claimable row public fields = %v", open)
|
||||
}
|
||||
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown"} {
|
||||
for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown", "autoRestarts", "startGaveUp"} {
|
||||
if _, ok := open[k]; ok {
|
||||
t.Errorf("claimable row carries owner detail %s = %v", k, open[k])
|
||||
}
|
||||
|
||||
@@ -117,6 +117,10 @@ type Cluster interface {
|
||||
// *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore,
|
||||
// backup or file write holds the world volume.
|
||||
SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
|
||||
// RetryStart is SetDesiredState(Running) for a server whose start Failed: it
|
||||
// also asks the operator to start it over with a fresh auto-restart budget
|
||||
// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
|
||||
RetryStart(ctx context.Context, name string) error
|
||||
// AcquireMaintenance admits one world-volume operation (internal/maintenance
|
||||
// kind): ErrNotStopped unless the server is fully stopped, a
|
||||
// *MaintenanceBusyError while another operation holds the volume. The check
|
||||
|
||||
@@ -152,6 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
// A start whose automatic restarts are spent (or that can never succeed as
|
||||
// configured) stays down until a person looks at it. The 202 this used to
|
||||
// return queued the player for a server nothing was starting. The join leaves
|
||||
// the restart budget alone, or every player who tried to join would buy
|
||||
// another three crash loops; velocity tells them and queues no one. A Failed
|
||||
// server still inside its backoff is not this: its next attempt is coming, so
|
||||
// it gets the 202 and the player waits for it.
|
||||
if info.StartGaveUp && info.DesiredState == string(v1alpha1.DesiredRunning) {
|
||||
writeError(w, r, newError(http.StatusConflict, "start_failed",
|
||||
"the server failed to start and its automatic retries are spent; its owner can retry from the panel"))
|
||||
return
|
||||
}
|
||||
if !a.limiter().allowed(name, a.WakeCooldown) {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly"))
|
||||
return
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
)
|
||||
|
||||
// A start that Failed holds desiredState Running already, so the wake that
|
||||
// re-writes Running changes nothing. These pin the two ways out: a person's
|
||||
// wake in the panel starts it over (RetryStart), and a player's join onto one
|
||||
// whose retries are spent is told so (409 start_failed) and never resets them.
|
||||
|
||||
func failedServer(gaveUp bool) *ServerInfo {
|
||||
return &ServerInfo{Name: "survival", AutostartPolicy: "public",
|
||||
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseFailed),
|
||||
AutoRestarts: v1alpha1.MaxAutoRestarts, StartGaveUp: gaveUp}
|
||||
}
|
||||
|
||||
func TestWakeOfFailedServerRetriesStart(t *testing.T) {
|
||||
mk := func(info *ServerInfo) (*API, *fakeCluster, *fakeRepo) {
|
||||
repo := newFakeRepo()
|
||||
cl := newFakeCluster()
|
||||
cl.byName["survival"] = info
|
||||
api := newTestAPI(repo, cl)
|
||||
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
|
||||
return api, cl, repo
|
||||
}
|
||||
wake := func(api *API) int {
|
||||
return do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil).Code
|
||||
}
|
||||
|
||||
t.Run("retries spent: the wake starts it over", func(t *testing.T) {
|
||||
api, cl, repo := mk(failedServer(true))
|
||||
if code := wake(api); code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d, want 202", code)
|
||||
}
|
||||
if len(cl.retried) != 1 || cl.retried[0] != "survival" {
|
||||
t.Fatalf("retried = %v, want [survival]", cl.retried)
|
||||
}
|
||||
if len(repo.audits) != 1 || repo.audits[0].Action != "retry_start" {
|
||||
t.Fatalf("audits = %+v, want one retry_start", repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("inside the backoff: a person asking skips the wait", func(t *testing.T) {
|
||||
api, cl, _ := mk(failedServer(false))
|
||||
if code := wake(api); code != http.StatusAccepted || len(cl.retried) != 1 {
|
||||
t.Fatalf("code = %d retried = %v, want 202 and one retry", code, cl.retried)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("stopped: a plain wake", func(t *testing.T) {
|
||||
api, cl, repo := mk(&ServerInfo{Name: "survival", AutostartPolicy: "public",
|
||||
DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)})
|
||||
if code := wake(api); code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d, want 202", code)
|
||||
}
|
||||
if len(cl.retried) != 0 || cl.desired["survival"] != v1alpha1.DesiredRunning {
|
||||
t.Fatalf("retried = %v desired = %q, want a plain start", cl.retried, cl.desired["survival"])
|
||||
}
|
||||
if len(repo.audits) != 1 || repo.audits[0].Action != "wake" {
|
||||
t.Fatalf("audits = %+v, want one wake", repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("maintenance holds the world: refused, nothing retried", func(t *testing.T) {
|
||||
api, cl, _ := mk(failedServer(true))
|
||||
cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore}
|
||||
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
|
||||
t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
|
||||
}
|
||||
if len(cl.retried) != 0 {
|
||||
t.Fatalf("retried = %v, want none", cl.retried)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestInternalWakeOfFailedServer(t *testing.T) {
|
||||
body := `{"mc_uuid":"` + wakeUUID + `"}`
|
||||
|
||||
t.Run("retries spent: 409 start_failed, budget and cooldown untouched", func(t *testing.T) {
|
||||
api, cl := newInternalWakeAPI("public")
|
||||
cl.byName["survival"] = failedServer(true)
|
||||
api.WakeCooldown = time.Minute
|
||||
repo := api.Repo.(*fakeRepo)
|
||||
|
||||
w := internalWake(api, body)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "start_failed" {
|
||||
t.Fatalf("code = %d body %s, want 409 start_failed", w.Code, w.Body.String())
|
||||
}
|
||||
if len(cl.retried) != 0 {
|
||||
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
|
||||
}
|
||||
if len(repo.audits) != 0 {
|
||||
t.Fatalf("nothing was woken, so nothing is audited: %+v", repo.audits)
|
||||
}
|
||||
// The owner fixes it and starts it from the panel; the next join is not
|
||||
// held back by a cooldown the refused one never spent.
|
||||
cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public",
|
||||
DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseStarting)}
|
||||
if w := internalWake(api, body); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("join after the fix: code = %d body %s, want 202", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("inside the backoff: 202, the player waits for the next attempt", func(t *testing.T) {
|
||||
api, cl := newInternalWakeAPI("public")
|
||||
cl.byName["survival"] = failedServer(false)
|
||||
if w := internalWake(api, body); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d body %s, want 202", w.Code, w.Body.String())
|
||||
}
|
||||
if len(cl.retried) != 0 {
|
||||
t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("forbidden player: 403 comes first", func(t *testing.T) {
|
||||
api, cl := newInternalWakeAPI("ownerOnly")
|
||||
info := failedServer(true)
|
||||
info.AutostartPolicy = "ownerOnly"
|
||||
cl.byName["survival"] = info
|
||||
if w := internalWake(api, body); w.Code != http.StatusForbidden {
|
||||
t.Fatalf("code = %d, want 403", w.Code)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -56,9 +56,23 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// Refused with 409 maintenance_in_progress while a restore, backup or file
|
||||
// write holds the world volume: starting on a half-written world corrupts it.
|
||||
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
|
||||
// A start that Failed already holds desiredState Running, so writing Running
|
||||
// again changes nothing and the server stayed dead once its automatic restarts
|
||||
// were spent. A person pressing start on it means "try again": RetryStart
|
||||
// makes the operator start it over with a fresh restart budget. Only this face
|
||||
// does that; a player's join never resets the budget (handleInternalWake).
|
||||
//
|
||||
// Either write is refused with 409 maintenance_in_progress while a restore,
|
||||
// backup or file write holds the world volume: starting on a half-written
|
||||
// world corrupts it.
|
||||
action := "wake"
|
||||
if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) {
|
||||
action = "retry_start"
|
||||
err = a.Cluster.RetryStart(r.Context(), name)
|
||||
} else {
|
||||
err = a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning)
|
||||
}
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
@@ -67,7 +81,7 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
|
||||
// held at capacity should retry the instant a slot frees, not wait out a
|
||||
// cooldown their refused wake never earned).
|
||||
a.limiter().record(name)
|
||||
a.audit(r, "wake", name)
|
||||
a.audit(r, action, name)
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
|
||||
}
|
||||
|
||||
@@ -267,6 +281,8 @@ func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) {
|
||||
v.DesiredState = info.DesiredState
|
||||
v.AutostartPolicy = info.AutostartPolicy
|
||||
v.PlayerCountUnknown = info.PlayerCountUnknown
|
||||
v.AutoRestarts = info.AutoRestarts
|
||||
v.StartGaveUp = info.StartGaveUp
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -246,6 +246,17 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a
|
||||
// against, the same as AcquireMaintenance's: whichever of a racing wake and
|
||||
// admission writes second gets a conflict, re-reads, and sees the other.
|
||||
func (k *K8sCluster) start(ctx context.Context, name string) error {
|
||||
return k.startWith(ctx, name, false)
|
||||
}
|
||||
|
||||
// RetryStart starts a Failed server over: the same guarded write as start, plus
|
||||
// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and
|
||||
// recreates the pod.
|
||||
func (k *K8sCluster) RetryStart(ctx context.Context, name string) error {
|
||||
return k.startWith(ctx, name, true)
|
||||
}
|
||||
|
||||
func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error {
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.getServer(ctx, name, &ms); err != nil {
|
||||
@@ -263,6 +274,12 @@ func (k *K8sCluster) start(ctx context.Context, name string) error {
|
||||
// A lock still on the object here no longer holds anything (Holder said
|
||||
// so): drop it in the same write.
|
||||
delete(ms.Annotations, maintenance.Annotation)
|
||||
if retryFailed {
|
||||
if ms.Annotations == nil {
|
||||
ms.Annotations = map[string]string{}
|
||||
}
|
||||
ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339)
|
||||
}
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
})
|
||||
}
|
||||
|
||||
@@ -214,6 +214,39 @@ func TestStartRespectsMaintenance(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("retry start -> Running plus the retry request, a plain start leaves none", func(t *testing.T) {
|
||||
ms := stoppedServer()
|
||||
ms.Spec.DesiredState = v1alpha1.DesiredRunning
|
||||
ms.Status.Phase = v1alpha1.PhaseFailed
|
||||
k, c := lockCluster(t, ms)
|
||||
if err := k.SetDesiredState(ctx, "survival", v1alpha1.DesiredRunning); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
|
||||
t.Fatalf("a plain start asked for a retry: %v", ann)
|
||||
}
|
||||
if err := k.RetryStart(ctx, "survival"); err != nil {
|
||||
t.Fatalf("retry: %v", err)
|
||||
}
|
||||
ann, desired := annotations(t, c)
|
||||
if desired != v1alpha1.DesiredRunning || ann[v1alpha1.AnnotationStartRetry] != lockNow.Format(time.RFC3339) {
|
||||
t.Fatalf("desiredState = %q annotations = %v, want Running and the retry stamped %s",
|
||||
desired, ann, lockNow.Format(time.RFC3339))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("retry start under a fresh lock -> busy, no retry request", func(t *testing.T) {
|
||||
ms := stoppedServer()
|
||||
ms.Annotations = map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lockNow)}
|
||||
k, c := lockCluster(t, ms)
|
||||
if err := k.RetryStart(ctx, "survival"); !errors.Is(err, ErrMaintenanceInProgress) {
|
||||
t.Fatalf("err = %v, want maintenance in progress", err)
|
||||
}
|
||||
if ann, _ := annotations(t, c); ann[v1alpha1.AnnotationStartRetry] != "" {
|
||||
t.Fatalf("a refused retry left its request: %v", ann)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("stop ignores the lock", func(t *testing.T) {
|
||||
ms := stoppedServer()
|
||||
ms.Spec.DesiredState = v1alpha1.DesiredRunning
|
||||
|
||||
@@ -21,9 +21,9 @@ type ServerRecord struct {
|
||||
// may auto-start, or may claim. Everything after Claimable is NOT stored in
|
||||
// Postgres — handleMyServers joins it best-effort from the CRD status
|
||||
// (Cluster.ListServers) at read time, so a cluster hiccup renders 0/0 and the
|
||||
// cached phase, never a 500. DesiredState, AutostartPolicy and
|
||||
// PlayerCountUnknown are owner detail and stay empty on rows the caller does
|
||||
// not own, the same split publicServerInfo makes on the status route.
|
||||
// cached phase, never a 500. DesiredState, AutostartPolicy, PlayerCountUnknown,
|
||||
// AutoRestarts and StartGaveUp are owner detail and stay empty on rows the caller
|
||||
// does not own, the same split publicServerInfo makes on the status route.
|
||||
type MyServerView struct {
|
||||
Name string `json:"name"`
|
||||
Subdomain string `json:"subdomain"`
|
||||
@@ -36,6 +36,8 @@ type MyServerView struct {
|
||||
DesiredState string `json:"desiredState,omitempty"`
|
||||
AutostartPolicy string `json:"autostartPolicy,omitempty"`
|
||||
PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"`
|
||||
AutoRestarts int32 `json:"autoRestarts,omitempty"`
|
||||
StartGaveUp bool `json:"startGaveUp,omitempty"`
|
||||
}
|
||||
|
||||
// ServerOwnership is one live server's claim state as the fleet read joins it.
|
||||
|
||||
@@ -31,6 +31,13 @@ const (
|
||||
// server; and felis-api never writes it, so marking a server takes kubectl on
|
||||
// the cluster, which fits a switch that drops the forwarding secret.
|
||||
LabelForwarding = GroupName + "/forwarding"
|
||||
// AnnotationStartRetry is felis-api asking the operator to start a Failed
|
||||
// server over: its value is the request time (RFC 3339). Re-patching
|
||||
// desiredState to the Running it already holds changes nothing the operator
|
||||
// can see, so a person pressing "retry" in the panel had no way through once
|
||||
// the automatic restarts were spent. The operator takes the request once —
|
||||
// fresh restart budget, new start anchor, pod recreated — and removes it.
|
||||
AnnotationStartRetry = GroupName + "/start-retry"
|
||||
)
|
||||
|
||||
// ForwardingLegacy is the LabelForwarding value that selects legacy forwarding.
|
||||
|
||||
@@ -195,6 +195,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu
|
||||
desired = v1alpha1.DesiredStopped
|
||||
}
|
||||
prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts
|
||||
if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok {
|
||||
if err := r.takeStartRetry(ctx, &server, desired); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
}
|
||||
var res ctrl.Result
|
||||
var err error
|
||||
if desired == v1alpha1.DesiredStopped {
|
||||
@@ -233,6 +238,33 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine
|
||||
r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg))
|
||||
}
|
||||
|
||||
// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed
|
||||
// and meant to run starts over as if freshly woken — pod recreated, restart
|
||||
// budget back to zero, a new start anchor — and in any other state the request
|
||||
// is stale and only removed. The fresh status is written before the request is
|
||||
// removed, so a pass that fails in between leaves the request to be taken again
|
||||
// (at the cost of one more pod recreate), never a removed request with the old
|
||||
// spent budget still in place.
|
||||
func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error {
|
||||
if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed {
|
||||
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}}
|
||||
if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil {
|
||||
return err
|
||||
}
|
||||
server.Status.AutoRestarts = 0
|
||||
server.Status.StartRequestedAt = nil
|
||||
r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod")
|
||||
if err := r.patchStatus(ctx, server); err != nil {
|
||||
return err
|
||||
}
|
||||
log.FromContext(ctx).Info("start retried")
|
||||
r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod")
|
||||
}
|
||||
patch := client.MergeFrom(server.DeepCopy())
|
||||
delete(server.Annotations, v1alpha1.AnnotationStartRetry)
|
||||
return r.Patch(ctx, server, patch)
|
||||
}
|
||||
|
||||
func (r *Reconciler) event(server *v1alpha1.MinecraftServer, eventType, reason, msg string) {
|
||||
if r.Recorder != nil {
|
||||
r.Recorder.Event(server, eventType, reason, msg)
|
||||
@@ -807,6 +839,7 @@ func (r *Reconciler) markFailed(server *v1alpha1.MinecraftServer, reason, msg st
|
||||
server.Status.Phase = v1alpha1.PhaseFailed
|
||||
server.Status.Ready = false
|
||||
server.Status.ObservedGeneration = server.Generation
|
||||
server.Status.LiveMotd = server.Spec.Motd.Failed
|
||||
r.setCondition(server, v1alpha1.ConditionReady, metav1.ConditionFalse, reason, msg)
|
||||
r.setCondition(server, v1alpha1.ConditionProvisioned, metav1.ConditionFalse, reason, msg)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
package operator_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
)
|
||||
|
||||
// requestStartRetry is felis-api's RetryStart as the operator sees it.
|
||||
func requestStartRetry(t *testing.T, c client.Client) {
|
||||
t.Helper()
|
||||
s := getServer(t, c, "survival")
|
||||
patch := client.MergeFrom(s.DeepCopy())
|
||||
if s.Annotations == nil {
|
||||
s.Annotations = map[string]string{}
|
||||
}
|
||||
s.Annotations[v1alpha1.AnnotationStartRetry] = "2026-07-01T12:00:00Z"
|
||||
if err := c.Patch(context.Background(), s, patch); err != nil {
|
||||
t.Fatalf("annotate: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func podPresent(t *testing.T, c client.Client) bool {
|
||||
t.Helper()
|
||||
err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival-0"}, &corev1.Pod{})
|
||||
if err != nil && !apierrors.IsNotFound(err) {
|
||||
t.Fatalf("get pod: %v", err)
|
||||
}
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// Once the automatic restarts are spent, a retry request starts the server over
|
||||
// with the whole budget back: the pod is recreated, the start re-anchored, and
|
||||
// the next timeout is retried automatically again.
|
||||
func TestStartRetryRevivesAServerThatGaveUp(t *testing.T) {
|
||||
srv := runningServer()
|
||||
srv.Spec.Startup.TimeoutSeconds = 30
|
||||
srv.Spec.Motd.Failed = "broken"
|
||||
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
|
||||
base := time.Date(2026, 7, 1, 10, 0, 0, 0, time.UTC)
|
||||
clock := base
|
||||
r.Now = func() metav1.Time { return metav1.NewTime(clock) }
|
||||
at := func(offset time.Duration) *v1alpha1.MinecraftServer {
|
||||
clock = base.Add(offset)
|
||||
reconcile(t, r, "survival")
|
||||
return getServer(t, c, "survival")
|
||||
}
|
||||
recreatePod := func() {
|
||||
if err := c.Create(context.Background(), gamePod()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// Spend the three automatic restarts (autorestart_test.go pins the schedule).
|
||||
at(0)
|
||||
for _, due := range []time.Duration{90 * time.Second, 240 * time.Second, 510 * time.Second} {
|
||||
at(due)
|
||||
recreatePod()
|
||||
}
|
||||
s := at(time.Hour)
|
||||
if !v1alpha1.StartGaveUp(&s.Status) {
|
||||
t.Fatalf("setup: not given up: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts)
|
||||
}
|
||||
if s.Status.LiveMotd != "broken" {
|
||||
t.Fatalf("Failed advertises liveMotd %q, want the failed MOTD", s.Status.LiveMotd)
|
||||
}
|
||||
|
||||
requestStartRetry(t, c)
|
||||
s = at(2 * time.Hour)
|
||||
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
|
||||
t.Fatal("the retry request was left on the server")
|
||||
}
|
||||
if podPresent(t, c) {
|
||||
t.Fatal("the retry kept the failed pod")
|
||||
}
|
||||
if s.Status.Phase != v1alpha1.PhaseStarting || s.Status.AutoRestarts != 0 || v1alpha1.StartGaveUp(&s.Status) {
|
||||
t.Fatalf("after retry: phase=%s restarts=%d gaveUp=%v", s.Status.Phase, s.Status.AutoRestarts, v1alpha1.StartGaveUp(&s.Status))
|
||||
}
|
||||
if s.Status.StartRequestedAt == nil || !s.Status.StartRequestedAt.Time.Equal(base.Add(2*time.Hour)) {
|
||||
t.Fatalf("start not re-anchored at the retry: %v", s.Status.StartRequestedAt)
|
||||
}
|
||||
recreatePod()
|
||||
|
||||
// The budget is really back: this start times out and is retried on its own.
|
||||
if s := at(2*time.Hour + 89*time.Second); s.Status.Phase != v1alpha1.PhaseFailed || v1alpha1.StartGaveUp(&s.Status) {
|
||||
t.Fatalf("retried start timing out: phase=%s gaveUp=%v, want Failed with retries left", s.Status.Phase, v1alpha1.StartGaveUp(&s.Status))
|
||||
}
|
||||
if s := at(2*time.Hour + 90*time.Second); s.Status.AutoRestarts != 1 || podPresent(t, c) {
|
||||
t.Fatalf("retried start: restarts=%d pod=%v, want the first automatic restart", s.Status.AutoRestarts, podPresent(t, c))
|
||||
}
|
||||
}
|
||||
|
||||
// A request that finds the server anywhere but Failed is stale (the start
|
||||
// already recovered, or someone stopped it): it is removed and nothing else.
|
||||
func TestStaleStartRetryIsOnlyRemoved(t *testing.T) {
|
||||
srv := runningServer()
|
||||
srv.Annotations = map[string]string{v1alpha1.AnnotationStartRetry: "2026-07-01T12:00:00Z"}
|
||||
anchor := metav1.NewTime(fixedNow().Add(-10 * time.Second))
|
||||
srv.Status = v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting, AutoRestarts: 2, StartRequestedAt: &anchor}
|
||||
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
|
||||
|
||||
reconcile(t, r, "survival")
|
||||
s := getServer(t, c, "survival")
|
||||
if _, left := s.Annotations[v1alpha1.AnnotationStartRetry]; left {
|
||||
t.Fatal("the stale retry request was left on the server")
|
||||
}
|
||||
if !podPresent(t, c) || s.Status.AutoRestarts != 2 || !s.Status.StartRequestedAt.Equal(&anchor) {
|
||||
t.Fatalf("a stale request touched the start: pod=%v restarts=%d anchor=%v",
|
||||
podPresent(t, c), s.Status.AutoRestarts, s.Status.StartRequestedAt)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user