Unverified Commit 22d1e834 authored by Lemon-miaow's avatar Lemon-miaow
Browse files

fix(felis): 启动失败的服可在面板重试或停止,玩家入服时直接告知启动失败

parent c1025274
Loading
Loading
Loading
Loading
+19 −2
Changes for docs/openapi.yaml: 19 added lines, 2 removed lines.
Original line number Diff line number Diff line
@@ -566,6 +566,13 @@ components:
        playerCountUnknown:
          type: boolean
          description: Owned rows only. Present and true while the operator cannot read the player count, so a stop may disconnect players.
        autoRestarts:
          type: integer
          format: int32
          description: Owned rows only. How often the operator recreated the pod of this start after it timed out.
        startGaveUp:
          type: boolean
          description: Owned rows only. Present and true for a Failed server no automatic retry will bring up; waking it from the panel starts it over.

    BackupView:
      type: object
@@ -1165,7 +1172,9 @@ paths:
      description: >-
        velocity holds no web Principal, so it drives the wake lever with its
        service token, identifying the player by online-mode UUID. Gated by the
        server's autostartPolicy and the per-server wake cooldown.
        server's autostartPolicy and the per-server wake cooldown. A server whose
        start failed with its automatic retries spent answers 409 start_failed: a
        join never resets the retry budget, so velocity queues no one for it.
      x-felis-face: [internal]
      x-felis-tier: service
      x-felis-callers: [velocity]
@@ -1203,7 +1212,11 @@ paths:
        '404':
          $ref: '#/components/responses/NotFound'
        '409':
          description: A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started.
          description: >-
            Nothing was started. maintenance_in_progress: a restore, backup or
            file write holds the server's world volume. start_failed: the last
            start failed and its automatic retries are spent (ServerInfo.startGaveUp);
            the server stays down until a person starts it from the panel.
          content:
            application/json:
              schema: { $ref: '#/components/schemas/Error' }
@@ -1767,6 +1780,10 @@ paths:
      tags: [servers]
      operationId: wake
      summary: Wake your own server.
      description: >-
        On a server whose start Failed (phase Failed, desiredState Running) this
        is a retry: the operator recreates the pod and starts it over with a
        fresh automatic-restart budget. It is audited as retry_start.
      x-felis-face: [external]
      x-felis-tier: app
      security: [{ sessionCookie: [] }]
+18 −4
Changes for internal/api/api_test.go: 18 added lines, 4 removed lines.
Original line number Diff line number Diff line
@@ -1748,6 +1748,7 @@ type fakeCluster struct {
	wakeErr  map[string]error
	acquired []string // "name:kind" per admitted AcquireMaintenance
	released []string // names per ReleaseMaintenance
	retried  []string // names per RetryStart
}

func newFakeCluster() *fakeCluster {
@@ -1794,6 +1795,16 @@ func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.De
	c.desired[n] = s
	return nil
}

// RetryStart records the retry on top of the plain start, so a test tells a
// Failed server's retry apart from an ordinary wake.
func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
	if err := c.SetDesiredState(ctx, n, v1alpha1.DesiredRunning); err != nil {
		return err
	}
	c.retried = append(c.retried, n)
	return nil
}
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
	if err := c.maintErr[n]; err != nil {
		return err
@@ -2092,9 +2103,11 @@ func TestMyServersJoinsLiveState(t *testing.T) {
	cl := newFakeCluster()
	cl.list = []ServerInfo{
		{Name: "mine", DisplayName: "My World", Phase: "Running", DesiredState: "Running",
			AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true},
			AutostartPolicy: "ownerOnly", PlayersOnline: 2, PlayersMax: 20, PlayerCountUnknown: true,
			AutoRestarts: 3, StartGaveUp: true},
		{Name: "open", DisplayName: "Open World", Phase: "Running", DesiredState: "Running",
			AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true},
			AutostartPolicy: "public", PlayersMax: 10, PlayerCountUnknown: true,
			AutoRestarts: 3, StartGaveUp: true},
	}
	api := newTestAPI(repo, cl)
	api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
@@ -2122,7 +2135,8 @@ func TestMyServersJoinsLiveState(t *testing.T) {
	mine, open := got["mine"], got["open"]
	for k, want := range map[string]any{"displayName": "My World", "phase": "Running",
		"desiredState": "Running", "autostartPolicy": "ownerOnly", "playerCountUnknown": true,
		"playersOnline": float64(2), "playersMax": float64(20)} {
		"playersOnline": float64(2), "playersMax": float64(20),
		"autoRestarts": float64(3), "startGaveUp": true} {
		if mine[k] != want {
			t.Errorf("own row %s = %v, want %v", k, mine[k], want)
		}
@@ -2130,7 +2144,7 @@ func TestMyServersJoinsLiveState(t *testing.T) {
	if open["displayName"] != "Open World" || open["phase"] != "Running" || open["playersMax"] != float64(10) {
		t.Errorf("claimable row public fields = %v", open)
	}
	for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown"} {
	for _, k := range []string{"desiredState", "autostartPolicy", "playerCountUnknown", "autoRestarts", "startGaveUp"} {
		if _, ok := open[k]; ok {
			t.Errorf("claimable row carries owner detail %s = %v", k, open[k])
		}
+4 −0
Changes for internal/api/cluster.go: 4 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -117,6 +117,10 @@ type Cluster interface {
	// *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore,
	// backup or file write holds the world volume.
	SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
	// RetryStart is SetDesiredState(Running) for a server whose start Failed: it
	// also asks the operator to start it over with a fresh auto-restart budget
	// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
	RetryStart(ctx context.Context, name string) error
	// AcquireMaintenance admits one world-volume operation (internal/maintenance
	// kind): ErrNotStopped unless the server is fully stopped, a
	// *MaintenanceBusyError while another operation holds the volume. The check
+12 −0
Changes for internal/api/handlers_internal.go: 12 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -152,6 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
		writeError(w, r, err)
		return
	}
	// A start whose automatic restarts are spent (or that can never succeed as
	// configured) stays down until a person looks at it. The 202 this used to
	// return queued the player for a server nothing was starting. The join leaves
	// the restart budget alone, or every player who tried to join would buy
	// another three crash loops; velocity tells them and queues no one. A Failed
	// server still inside its backoff is not this: its next attempt is coming, so
	// it gets the 202 and the player waits for it.
	if info.StartGaveUp && info.DesiredState == string(v1alpha1.DesiredRunning) {
		writeError(w, r, newError(http.StatusConflict, "start_failed",
			"the server failed to start and its automatic retries are spent; its owner can retry from the panel"))
		return
	}
	if !a.limiter().allowed(name, a.WakeCooldown) {
		writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly"))
		return
+131 −0
Changes for internal/api/handlers_start_failed_test.go: 131 added lines, 0 removed lines.
Original line number Diff line number Diff line
package api

import (
	"net/http"
	"testing"
	"time"

	"felis.lolicon.best/internal/apis/felis/v1alpha1"
	"felis.lolicon.best/internal/maintenance"
)

// A start that Failed holds desiredState Running already, so the wake that
// re-writes Running changes nothing. These pin the two ways out: a person's
// wake in the panel starts it over (RetryStart), and a player's join onto one
// whose retries are spent is told so (409 start_failed) and never resets them.

func failedServer(gaveUp bool) *ServerInfo {
	return &ServerInfo{Name: "survival", AutostartPolicy: "public",
		DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseFailed),
		AutoRestarts: v1alpha1.MaxAutoRestarts, StartGaveUp: gaveUp}
}

func TestWakeOfFailedServerRetriesStart(t *testing.T) {
	mk := func(info *ServerInfo) (*API, *fakeCluster, *fakeRepo) {
		repo := newFakeRepo()
		cl := newFakeCluster()
		cl.byName["survival"] = info
		api := newTestAPI(repo, cl)
		api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
		return api, cl, repo
	}
	wake := func(api *API) int {
		return do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil).Code
	}

	t.Run("retries spent: the wake starts it over", func(t *testing.T) {
		api, cl, repo := mk(failedServer(true))
		if code := wake(api); code != http.StatusAccepted {
			t.Fatalf("code = %d, want 202", code)
		}
		if len(cl.retried) != 1 || cl.retried[0] != "survival" {
			t.Fatalf("retried = %v, want [survival]", cl.retried)
		}
		if len(repo.audits) != 1 || repo.audits[0].Action != "retry_start" {
			t.Fatalf("audits = %+v, want one retry_start", repo.audits)
		}
	})

	t.Run("inside the backoff: a person asking skips the wait", func(t *testing.T) {
		api, cl, _ := mk(failedServer(false))
		if code := wake(api); code != http.StatusAccepted || len(cl.retried) != 1 {
			t.Fatalf("code = %d retried = %v, want 202 and one retry", code, cl.retried)
		}
	})

	t.Run("stopped: a plain wake", func(t *testing.T) {
		api, cl, repo := mk(&ServerInfo{Name: "survival", AutostartPolicy: "public",
			DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)})
		if code := wake(api); code != http.StatusAccepted {
			t.Fatalf("code = %d, want 202", code)
		}
		if len(cl.retried) != 0 || cl.desired["survival"] != v1alpha1.DesiredRunning {
			t.Fatalf("retried = %v desired = %q, want a plain start", cl.retried, cl.desired["survival"])
		}
		if len(repo.audits) != 1 || repo.audits[0].Action != "wake" {
			t.Fatalf("audits = %+v, want one wake", repo.audits)
		}
	})

	t.Run("maintenance holds the world: refused, nothing retried", func(t *testing.T) {
		api, cl, _ := mk(failedServer(true))
		cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore}
		w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil)
		if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
			t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
		}
		if len(cl.retried) != 0 {
			t.Fatalf("retried = %v, want none", cl.retried)
		}
	})
}

func TestInternalWakeOfFailedServer(t *testing.T) {
	body := `{"mc_uuid":"` + wakeUUID + `"}`

	t.Run("retries spent: 409 start_failed, budget and cooldown untouched", func(t *testing.T) {
		api, cl := newInternalWakeAPI("public")
		cl.byName["survival"] = failedServer(true)
		api.WakeCooldown = time.Minute
		repo := api.Repo.(*fakeRepo)

		w := internalWake(api, body)
		if w.Code != http.StatusConflict || decodeErr(t, w) != "start_failed" {
			t.Fatalf("code = %d body %s, want 409 start_failed", w.Code, w.Body.String())
		}
		if len(cl.retried) != 0 {
			t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
		}
		if len(repo.audits) != 0 {
			t.Fatalf("nothing was woken, so nothing is audited: %+v", repo.audits)
		}
		// The owner fixes it and starts it from the panel; the next join is not
		// held back by a cooldown the refused one never spent.
		cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public",
			DesiredState: string(v1alpha1.DesiredRunning), Phase: string(v1alpha1.PhaseStarting)}
		if w := internalWake(api, body); w.Code != http.StatusAccepted {
			t.Fatalf("join after the fix: code = %d body %s, want 202", w.Code, w.Body.String())
		}
	})

	t.Run("inside the backoff: 202, the player waits for the next attempt", func(t *testing.T) {
		api, cl := newInternalWakeAPI("public")
		cl.byName["survival"] = failedServer(false)
		if w := internalWake(api, body); w.Code != http.StatusAccepted {
			t.Fatalf("code = %d body %s, want 202", w.Code, w.Body.String())
		}
		if len(cl.retried) != 0 {
			t.Fatalf("a join must never reset the restart budget: retried = %v", cl.retried)
		}
	})

	t.Run("forbidden player: 403 comes first", func(t *testing.T) {
		api, cl := newInternalWakeAPI("ownerOnly")
		info := failedServer(true)
		info.AutostartPolicy = "ownerOnly"
		cl.byName["survival"] = info
		if w := internalWake(api, body); w.Code != http.StatusForbidden {
			t.Fatalf("code = %d, want 403", w.Code)
		}
	})
}
Loading