fix(panel): streamline system settings and server creation
Use shared action buttons and default to the login space. Allow staff to save startup-only experience settings while running and request a durable restart through the existing operator flow. Grant the API PVC list permission needed to detect retained worlds before server creation, and show internal failures with a request ID. Cover permission, maintenance, restart and UI behavior with regression checks.
This commit is contained in:
34 files changed
+555
-103
No files matched your search
@@ -566,6 +566,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// (handlers_retire.go); owner/admin-gated inside the handlers.
|
||||
{Method: "PUT", Pattern: "/api/v1/servers/{name}/retirement", h: a.handleRetire},
|
||||
{Method: "DELETE", Pattern: "/api/v1/servers/{name}/retirement", h: a.handleCancelRetire},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/restart", h: a.handleRestart},
|
||||
{Method: "GET", Pattern: "/api/v1/servers/{name}/status", h: a.handleStatus},
|
||||
// Identity self-read (spec §14 tiering): the panel reads this once at boot to
|
||||
// learn its own tier and decide which navigation surfaces to render. App-tier —
|
||||
|
||||
@@ -1935,6 +1935,36 @@ func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
|
||||
c.retried = append(c.retried, n)
|
||||
return nil
|
||||
}
|
||||
func (c *fakeCluster) RestartServer(ctx context.Context, n string) error {
|
||||
return c.RetryStart(ctx, n)
|
||||
}
|
||||
|
||||
func TestRestartAuthorization(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name, server, role, user string
|
||||
staff bool
|
||||
status int
|
||||
}{
|
||||
{"owner of player server", "survival", "user", "owner1", false, 202},
|
||||
{"other player", "survival", "user", "stranger", false, 403},
|
||||
{"Owner console system service", "lobby", "owner", "owner1", true, 202},
|
||||
{"player system service", "lobby", "user", "owner1", false, 400},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
a, _, cl, _ := mkFiles(t)
|
||||
cl.byName[tc.server] = &ServerInfo{Name: tc.server, Phase: "Running", DesiredState: "Running", Ready: true}
|
||||
a.External = staticExternal{p: &Principal{UserID: tc.user, Role: tc.role, ViaAdminAccess: tc.staff}}
|
||||
w := do(a.ExternalHandler(), "POST", "/api/v1/servers/"+tc.server+"/restart", "", nil)
|
||||
if w.Code != tc.status {
|
||||
t.Fatalf("status=%d want=%d body=%s", w.Code, tc.status, w.Body.String())
|
||||
}
|
||||
if (len(cl.retried) == 1) != (tc.status == 202) {
|
||||
t.Fatalf("unauthorized restart or missing request: %v", cl.retried)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
|
||||
if err := c.maintErr[n]; err != nil {
|
||||
return err
|
||||
|
||||
@@ -137,8 +137,10 @@ type Cluster interface {
|
||||
// also asks the operator to start it over with a fresh auto-restart budget
|
||||
// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
|
||||
RetryStart(ctx context.Context, name string) error
|
||||
// RestartServer requests a pod restart without changing desired state.
|
||||
RestartServer(ctx context.Context, name string) error
|
||||
// AcquireMaintenance admits one world-volume operation (internal/maintenance
|
||||
// kind): ErrNotStopped unless the server is fully stopped, a
|
||||
// kind): ErrNotStopped unless fully stopped (except system config writes), a
|
||||
// *MaintenanceBusyError while another operation holds the volume. The check
|
||||
// and the lock are one atomic write against a concurrent wake.
|
||||
AcquireMaintenance(ctx context.Context, name, kind string) error
|
||||
|
||||
@@ -15,6 +15,7 @@ import (
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
|
||||
// FileEditor is the server-file-editor surface the API depends on: list a
|
||||
@@ -233,7 +234,11 @@ func (a *API) handleWriteFile(w http.ResponseWriter, r *http.Request) {
|
||||
// beside it, and a read that overlaps a restore or another change can at worst
|
||||
// show a file mid-change: the sha256 it returned then no longer matches, so a
|
||||
// save built on it is refused with file_changed.
|
||||
release, ok := a.acquireWorld(w, r, name, maintenance.KindFileWrite, "stop the server before editing its files")
|
||||
kind := maintenance.KindFileWrite
|
||||
if liveExperienceFile(r, name) {
|
||||
kind = maintenance.KindConfigWrite
|
||||
}
|
||||
release, ok := a.acquireWorld(w, r, name, kind, "stop the server before editing its files")
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -582,16 +587,12 @@ var sha256Hex = regexp.MustCompile(`^[0-9a-f]{64}$`)
|
||||
// ② ServerByName — an unknown server is 404
|
||||
// ③ owner-or-admin, else 403. An unowned (released) server fails for everyone
|
||||
// but admin, which is the same "must re-claim first" rule restore enforces
|
||||
// ④ stopped gate: the world PVC is RWO and held by a running server, so a file
|
||||
// Job cannot mount it — refuse unless the server is fully stopped. Ready means
|
||||
// it is up; any desiredState other than Stopped means it is up or coming up
|
||||
// and still owns the volume. This yields a specific 409 instead of a Job that
|
||||
// silently fails to mount
|
||||
// ④ stopped gate: world files must not change under a live game process.
|
||||
// Only the system plugin's startup-only config may be read and saved live;
|
||||
// its Job mounts the RWO volume on the same node as the game pod.
|
||||
// ⑤ the FileEditor must be wired, else 503
|
||||
//
|
||||
// Single-sourcing it is what keeps the handlers from drifting: a read path that
|
||||
// forgot the stopped gate would not merely fail, it would hang waiting for a Pod
|
||||
// that can never be scheduled.
|
||||
// All file routes share this gate so the live-config exception stays narrow.
|
||||
//
|
||||
// It returns the validated server name and false if it has already written a
|
||||
// response.
|
||||
@@ -606,7 +607,7 @@ func (a *API) authorizeFileOp(w http.ResponseWriter, r *http.Request) (string, b
|
||||
a.writeLookupError(w, r, err)
|
||||
return "", false
|
||||
}
|
||||
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
|
||||
if !liveExperienceFile(r, name) && (info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped)) {
|
||||
writeError(w, r, newError(http.StatusConflict, "not_stopped",
|
||||
"stop the server before working with its files"))
|
||||
return "", false
|
||||
@@ -636,6 +637,14 @@ func (a *API) authorizeFileOp(w http.ResponseWriter, r *http.Request) (string, b
|
||||
return name, true
|
||||
}
|
||||
|
||||
// Only the startup-only system plugin settings may be edited while running.
|
||||
// World files, uploads, deletes and plugin binaries keep the stopped gate.
|
||||
func liveExperienceFile(r *http.Request, name string) bool {
|
||||
return naming.IsSystemServer(name) && strings.HasSuffix(r.URL.Path, "/file") &&
|
||||
(r.Method == http.MethodGet || r.Method == http.MethodPut) &&
|
||||
r.URL.Query().Get("path") == naming.ExperienceConfigFile
|
||||
}
|
||||
|
||||
// writeFileEditError maps executor errors onto HTTP status codes. The
|
||||
// sentinels are caller-fault and get precise answers; a timeout is reported as 504
|
||||
// so the caller knows to retry rather than believing the edit was rejected; and
|
||||
|
||||
@@ -163,6 +163,49 @@ func mkFiles(t *testing.T) (*API, *fakeRepo, *fakeCluster, *fakeFileEditor) {
|
||||
return api, repo, cl, files
|
||||
}
|
||||
|
||||
func TestRunningSystemExperienceFile(t *testing.T) {
|
||||
for _, name := range []string{"login", "lobby"} {
|
||||
for _, tc := range []struct {
|
||||
method, suffix, body string
|
||||
allowed bool
|
||||
}{
|
||||
{"GET", "/file?path=felis-experience.json", "", true},
|
||||
{"PUT", "/file?path=felis-experience.json", `{"content":"aGk=","content_sha256":"` + hiSum + `"}`, true},
|
||||
{"PUT", "/file?path=felis-experience.json", `{"content":"aGk=","content_sha256":"` + hiSum + `","create_only":true}`, true},
|
||||
{"PUT", "/file?path=world/level.dat", `{"content":"aGk=","content_sha256":"` + hiSum + `"}`, false},
|
||||
{"GET", "/file?path=./felis-experience.json", "", false},
|
||||
{"DELETE", "/file?path=felis-experience.json", "", false},
|
||||
{"GET", "/files", "", false},
|
||||
} {
|
||||
t.Run(name+" "+tc.method+tc.suffix, func(t *testing.T) {
|
||||
a, _, cl, files := mkFiles(t)
|
||||
cl.byName[name] = &ServerInfo{Name: name, Phase: "Running", Ready: true, DesiredState: "Running"}
|
||||
a.External = staticExternal{p: &Principal{UserID: "owner1", Role: "owner", ViaAdminAccess: true}}
|
||||
w := do(a.ExternalHandler(), tc.method, "/api/v1/servers/"+name+tc.suffix, tc.body, jsonHeader)
|
||||
if tc.allowed {
|
||||
if w.Code != http.StatusOK || files.calls != 1 {
|
||||
t.Fatalf("live config: status=%d calls=%d body=%s", w.Code, files.calls, w.Body.String())
|
||||
}
|
||||
if tc.method == "PUT" && (len(cl.acquired) != 1 || cl.acquired[0] != name+":"+maintenance.KindConfigWrite) {
|
||||
t.Fatalf("wrong lock: %v", cl.acquired)
|
||||
}
|
||||
} else if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" || files.calls != 0 {
|
||||
t.Fatalf("world gate: status=%d calls=%d body=%s", w.Code, files.calls, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
t.Run("player cannot edit system config", func(t *testing.T) {
|
||||
a, _, cl, files := mkFiles(t)
|
||||
cl.byName["lobby"] = &ServerInfo{Name: "lobby", Phase: "Running", Ready: true, DesiredState: "Running"}
|
||||
a.External = staticExternal{p: &Principal{UserID: "owner1", Role: "user"}}
|
||||
w := do(a.ExternalHandler(), "GET", "/api/v1/servers/lobby/file?path=felis-experience.json", "", nil)
|
||||
if w.Code < 400 || files.calls != 0 {
|
||||
t.Fatalf("player reached config: status=%d calls=%d", w.Code, files.calls)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestFileEditorStoppedGate is the gate this whole subsystem hinges on. The world
|
||||
// PVC is ReadWriteOnce, but RWO is per node: on a single node a file Job mounts it
|
||||
// right beside a running server, and a write lands under a live world that the
|
||||
|
||||
@@ -92,6 +92,29 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
|
||||
}
|
||||
|
||||
// handleRestart records a restart request for the operator to consume once.
|
||||
func (a *API) handleRestart(w http.ResponseWriter, r *http.Request) {
|
||||
name, ok := a.authorizeServerFiles(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
rec, err := a.managedServerRecord(r.Context(), name)
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
if rec.Retire != nil {
|
||||
writeError(w, r, errServerRetiring)
|
||||
return
|
||||
}
|
||||
if err := a.Cluster.RestartServer(r.Context(), rec.Name); err != nil {
|
||||
a.writeLookupError(w, r, maintenanceError(err, "wait for maintenance to finish before restarting"))
|
||||
return
|
||||
}
|
||||
a.audit(r, "server.restart", rec.Name)
|
||||
writeJSON(w, http.StatusAccepted, map[string]string{"name": rec.Name, "desiredState": "Running"})
|
||||
}
|
||||
|
||||
// handleStop flips desiredState to Stopped. Only the owner or an admin may stop a
|
||||
// server (spec §14: operating someone else's server is admin-tier).
|
||||
func (a *API) handleStop(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
+27
-13
@@ -3,6 +3,7 @@ package api
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -264,22 +265,29 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a
|
||||
// against, the same as AcquireMaintenance's: whichever of a racing wake and
|
||||
// admission writes second gets a conflict, re-reads, and sees the other.
|
||||
func (k *K8sCluster) start(ctx context.Context, name string) error {
|
||||
return k.startWith(ctx, name, false)
|
||||
return k.startWith(ctx, name, "")
|
||||
}
|
||||
|
||||
// RetryStart starts a Failed server over: the same guarded write as start, plus
|
||||
// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and
|
||||
// recreates the pod.
|
||||
func (k *K8sCluster) RetryStart(ctx context.Context, name string) error {
|
||||
return k.startWith(ctx, name, true)
|
||||
return k.startWith(ctx, name, v1alpha1.AnnotationStartRetry)
|
||||
}
|
||||
|
||||
func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error {
|
||||
func (k *K8sCluster) RestartServer(ctx context.Context, name string) error {
|
||||
return k.startWith(ctx, name, v1alpha1.AnnotationRestart)
|
||||
}
|
||||
|
||||
func (k *K8sCluster) startWith(ctx context.Context, name, annotation string) error {
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.getServer(ctx, name, &ms); err != nil {
|
||||
return err
|
||||
}
|
||||
if annotation == v1alpha1.AnnotationRestart && (ms.Spec.DesiredState != v1alpha1.DesiredRunning || ms.Status.Phase != v1alpha1.PhaseRunning) {
|
||||
return newError(http.StatusConflict, "not_running", "start the server before restarting it")
|
||||
}
|
||||
node := ms.Spec.NodeName
|
||||
if node == "" {
|
||||
node = ms.Status.NodeName
|
||||
@@ -308,11 +316,11 @@ func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed boo
|
||||
// A lock still on the object here no longer holds anything (Holder said
|
||||
// so): drop it in the same write.
|
||||
delete(ms.Annotations, maintenance.Annotation)
|
||||
if retryFailed {
|
||||
if annotation != "" {
|
||||
if ms.Annotations == nil {
|
||||
ms.Annotations = map[string]string{}
|
||||
}
|
||||
ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339)
|
||||
ms.Annotations[annotation] = k.clock().UTC().Format(time.RFC3339Nano)
|
||||
}
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
})
|
||||
@@ -320,8 +328,9 @@ func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed boo
|
||||
|
||||
// AcquireMaintenance admits one world-volume operation of the given kind: the
|
||||
// server must be fully stopped (desiredState Stopped, phase Stopped, and no game
|
||||
// pod left, so a pod still saving on its way down is waited out) and nothing else
|
||||
// may hold the volume. Admission writes the maintenance lock under the resourceVersion it
|
||||
// pod left, so a pod still saving on its way down is waited out). System config
|
||||
// writes may run beside the game pod, but not during a pending restart. Nothing
|
||||
// else may hold the volume. Admission writes the lock under the resourceVersion it
|
||||
// checked; the caller creates its Job and then calls ReleaseMaintenance, after
|
||||
// which the Job itself is the lock.
|
||||
func (k *K8sCluster) AcquireMaintenance(ctx context.Context, name, kind string) error {
|
||||
@@ -334,13 +343,18 @@ func (k *K8sCluster) AcquireMaintenance(ctx context.Context, name, kind string)
|
||||
if desired == "" {
|
||||
desired = v1alpha1.DesiredStopped
|
||||
}
|
||||
if desired != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
|
||||
return ErrNotStopped
|
||||
if kind != maintenance.KindConfigWrite || !naming.IsSystemServer(name) {
|
||||
if desired != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
|
||||
return ErrNotStopped
|
||||
}
|
||||
if up, err := k.gamePodExists(ctx, name); err != nil {
|
||||
return err
|
||||
} else if up {
|
||||
return ErrNotStopped
|
||||
}
|
||||
}
|
||||
if up, err := k.gamePodExists(ctx, name); err != nil {
|
||||
return err
|
||||
} else if up {
|
||||
return ErrNotStopped
|
||||
if ms.Annotations[v1alpha1.AnnotationRestart] != "" {
|
||||
return ErrMaintenanceInProgress
|
||||
}
|
||||
holder, held, err := k.maintenanceHolder(ctx, &ms)
|
||||
if err != nil {
|
||||
|
||||
@@ -35,6 +35,54 @@ func stoppedServer() *v1alpha1.MinecraftServer {
|
||||
}
|
||||
}
|
||||
|
||||
func TestSystemConfigSaveAndRestartAdmission(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
ms := stoppedServer()
|
||||
ms.Name = "lobby"
|
||||
ms.Spec.DesiredState = v1alpha1.DesiredRunning
|
||||
ms.Status.Phase, ms.Status.Ready = v1alpha1.PhaseRunning, true
|
||||
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "lobby-0", Namespace: "minecraft",
|
||||
Labels: map[string]string{v1alpha1.LabelServer: "lobby", v1alpha1.LabelComponent: gamePodComponent}}}
|
||||
k, c := lockCluster(t, ms, pod)
|
||||
if err := k.AcquireMaintenance(ctx, "lobby", maintenance.KindConfigWrite); err != nil {
|
||||
t.Fatalf("save beside running pod: %v", err)
|
||||
}
|
||||
if err := k.RestartServer(ctx, "lobby"); !errors.Is(err, ErrMaintenanceInProgress) {
|
||||
t.Fatalf("restart during save: %v", err)
|
||||
}
|
||||
if err := k.ReleaseMaintenance(ctx, "lobby"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := k.RestartServer(ctx, "lobby"); err != nil {
|
||||
t.Fatalf("restart after save: %v", err)
|
||||
}
|
||||
var got v1alpha1.MinecraftServer
|
||||
if err := c.Get(ctx, client.ObjectKeyFromObject(ms), &got); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got.Annotations[v1alpha1.AnnotationRestart] == "" || got.Spec.DesiredState != v1alpha1.DesiredRunning {
|
||||
t.Fatalf("restart not recorded durably: %+v", got)
|
||||
}
|
||||
if err := c.Get(ctx, client.ObjectKeyFromObject(pod), &corev1.Pod{}); err != nil {
|
||||
t.Fatalf("API must leave pod deletion to operator: %v", err)
|
||||
}
|
||||
if err := k.AcquireMaintenance(ctx, "lobby", maintenance.KindConfigWrite); !errors.Is(err, ErrMaintenanceInProgress) {
|
||||
t.Fatalf("save racing a pending restart: %v", err)
|
||||
}
|
||||
|
||||
userServer := stoppedServer()
|
||||
userServer.Spec.DesiredState = v1alpha1.DesiredRunning
|
||||
userServer.Status.Phase = v1alpha1.PhaseRunning
|
||||
k, _ = lockCluster(t, userServer)
|
||||
if err := k.AcquireMaintenance(ctx, "survival", maintenance.KindConfigWrite); !errors.Is(err, ErrNotStopped) {
|
||||
t.Fatalf("live config exception reached a player world: %v", err)
|
||||
}
|
||||
k, _ = lockCluster(t, stoppedServer())
|
||||
if err := k.RestartServer(ctx, "survival"); err == nil {
|
||||
t.Fatal("restart admitted a stopped server")
|
||||
}
|
||||
}
|
||||
|
||||
func lockCluster(t *testing.T, objs ...client.Object) (*K8sCluster, client.Client) {
|
||||
t.Helper()
|
||||
scheme := runtime.NewScheme()
|
||||
|
||||
@@ -37,6 +37,8 @@ func maintenanceLabel(kind string) string {
|
||||
return "a backup"
|
||||
case maintenance.KindFileWrite:
|
||||
return "a file write"
|
||||
case maintenance.KindConfigWrite:
|
||||
return "a settings save"
|
||||
case maintenance.KindExport:
|
||||
return "a world export or file download"
|
||||
case maintenance.KindReap:
|
||||
|
||||
@@ -39,6 +39,8 @@ const (
|
||||
// the automatic restarts were spent. The operator takes the request once —
|
||||
// fresh restart budget, new start anchor, pod recreated — and removes it.
|
||||
AnnotationStartRetry = GroupName + "/start-retry"
|
||||
// AnnotationRestart requests one pod recreation while keeping desiredState Running.
|
||||
AnnotationRestart = GroupName + "/restart"
|
||||
)
|
||||
|
||||
// ForwardingLegacy is the LabelForwarding value that selects legacy forwarding.
|
||||
|
||||
@@ -529,6 +529,9 @@ func write(r *os.Root, name string, content []byte, sum, expect string, createOn
|
||||
if res.Code != "" {
|
||||
return res
|
||||
}
|
||||
if name == naming.ExperienceConfigFile && target != name {
|
||||
return Result{Code: CodeBadPath, Error: "the experience config must be a regular file, not a symlink"}
|
||||
}
|
||||
if expect != "" {
|
||||
if res := checkUnchanged(r, name, target, expect); res.Code != "" {
|
||||
return res
|
||||
|
||||
@@ -45,6 +45,21 @@ func run(root, op, path string, content []byte, expect string) (Result, error) {
|
||||
return Execute(root, Request{Op: op, Path: path, Content: content, Expect: expect})
|
||||
}
|
||||
|
||||
func TestExperienceConfigCannotWriteThroughSymlink(t *testing.T) {
|
||||
root, _ := worldRoot(t)
|
||||
if err := os.Symlink("server.properties", filepath.Join(root, "felis-experience.json")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
res, err := run(root, OpWrite, "felis-experience.json", []byte("{}"), "")
|
||||
if err != nil || res.Code != CodeBadPath {
|
||||
t.Fatalf("symlink write: result=%+v err=%v", res, err)
|
||||
}
|
||||
content, err := os.ReadFile(filepath.Join(root, "server.properties"))
|
||||
if err != nil || string(content) != "motd=hello\n" {
|
||||
t.Fatalf("live world file was changed: content=%q err=%v", content, err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestExecuteContainment is the security test of this package. The world directory
|
||||
// holds attacker-influenced content (players and plugins create files in it), so
|
||||
// each vector below is a path a caller could genuinely supply to try to leave the
|
||||
|
||||
@@ -90,11 +90,12 @@ const (
|
||||
|
||||
// Kinds of holder.
|
||||
const (
|
||||
KindMigration = "migration"
|
||||
KindRestore = "restore"
|
||||
KindBackup = "backup"
|
||||
KindFileWrite = "file-write"
|
||||
KindExport = "export"
|
||||
KindMigration = "migration"
|
||||
KindRestore = "restore"
|
||||
KindBackup = "backup"
|
||||
KindFileWrite = "file-write"
|
||||
KindConfigWrite = "config-write"
|
||||
KindExport = "export"
|
||||
// KindReap is the reaper archiving an idle world and reclaiming its volume.
|
||||
// It runs no Job: the reaper holds the Annotation itself and rewrites it
|
||||
// well inside Grace for as long as it works on the world.
|
||||
|
||||
@@ -46,6 +46,8 @@ const (
|
||||
// reachable only after the login gate passes a player through, so it must
|
||||
// never be used as a fallback target (that would bypass the gate).
|
||||
SystemLobbyServer = "lobby"
|
||||
// ExperienceConfigFile is read by the system plugins only at startup.
|
||||
ExperienceConfigFile = "felis-experience.json"
|
||||
)
|
||||
|
||||
// IsSystemServer identifies platform services whose reserved names may be managed
|
||||
|
||||
@@ -218,9 +218,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu
|
||||
desired = v1alpha1.DesiredStopped
|
||||
}
|
||||
prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts
|
||||
if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok {
|
||||
if err := r.takeStartRetry(ctx, &server, desired); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
for _, annotation := range []string{v1alpha1.AnnotationStartRetry, v1alpha1.AnnotationRestart} {
|
||||
if _, ok := server.Annotations[annotation]; ok {
|
||||
if err := r.takeRestartRequest(ctx, &server, desired, annotation); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
}
|
||||
}
|
||||
var res ctrl.Result
|
||||
@@ -263,30 +265,34 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine
|
||||
r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg))
|
||||
}
|
||||
|
||||
// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed
|
||||
// and meant to run starts over as if freshly woken — pod recreated, restart
|
||||
// budget back to zero, a new start anchor — and in any other state the request
|
||||
// is stale and only removed. The fresh status is written before the request is
|
||||
// takeRestartRequest recreates a Failed pod for a start retry, or a Running pod
|
||||
// for an explicit restart. A request overtaken by a stop is only removed.
|
||||
// The fresh status is written before the request is
|
||||
// removed, so a pass that fails in between leaves the request to be taken again
|
||||
// (at the cost of one more pod recreate), never a removed request with the old
|
||||
// spent budget still in place.
|
||||
func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error {
|
||||
if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed {
|
||||
func (r *Reconciler) takeRestartRequest(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState, annotation string) error {
|
||||
explicit := annotation == v1alpha1.AnnotationRestart
|
||||
if desired == v1alpha1.DesiredRunning && (server.Status.Phase == v1alpha1.PhaseFailed || (explicit && server.Status.Phase == v1alpha1.PhaseRunning)) {
|
||||
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}}
|
||||
if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil {
|
||||
return err
|
||||
}
|
||||
server.Status.AutoRestarts = 0
|
||||
server.Status.StartRequestedAt = nil
|
||||
r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod")
|
||||
reason, message := "StartRetried", "start requested again after it failed; recreated the pod"
|
||||
if explicit {
|
||||
reason, message = "RestartRequested", "restart requested; recreated the pod"
|
||||
}
|
||||
r.markStarting(server, reason, message)
|
||||
if err := r.patchStatus(ctx, server); err != nil {
|
||||
return err
|
||||
}
|
||||
log.FromContext(ctx).Info("start retried")
|
||||
r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod")
|
||||
log.FromContext(ctx).Info(message)
|
||||
r.event(server, corev1.EventTypeNormal, reason, message)
|
||||
}
|
||||
patch := client.MergeFrom(server.DeepCopy())
|
||||
delete(server.Annotations, v1alpha1.AnnotationStartRetry)
|
||||
delete(server.Annotations, annotation)
|
||||
return r.Patch(ctx, server, patch)
|
||||
}
|
||||
|
||||
|
||||
@@ -37,6 +37,38 @@ func podPresent(t *testing.T, c client.Client) bool {
|
||||
return err == nil
|
||||
}
|
||||
|
||||
func TestExplicitRestartIsConsumedOnce(t *testing.T) {
|
||||
srv := runningServer()
|
||||
srv.Status.Phase = v1alpha1.PhaseRunning
|
||||
srv.Annotations = map[string]string{v1alpha1.AnnotationRestart: "2026-07-01T12:00:00Z"}
|
||||
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
|
||||
reconcile(t, r, "survival")
|
||||
s := getServer(t, c, "survival")
|
||||
if s.Annotations[v1alpha1.AnnotationRestart] != "" || s.Spec.DesiredState != v1alpha1.DesiredRunning || s.Status.Phase != v1alpha1.PhaseStarting || podPresent(t, c) {
|
||||
t.Fatalf("restart: desired=%s phase=%s request=%s pod=%v", s.Spec.DesiredState, s.Status.Phase, s.Annotations[v1alpha1.AnnotationRestart], podPresent(t, c))
|
||||
}
|
||||
if err := c.Create(context.Background(), gamePod()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
reconcile(t, r, "survival")
|
||||
if !podPresent(t, c) {
|
||||
t.Fatal("consumed request deleted the replacement pod")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStopSupersedesExplicitRestart(t *testing.T) {
|
||||
srv := runningServer()
|
||||
srv.Spec.DesiredState = v1alpha1.DesiredStopped
|
||||
srv.Status.Phase = v1alpha1.PhaseRunning
|
||||
srv.Annotations = map[string]string{v1alpha1.AnnotationRestart: "2026-07-01T12:00:00Z"}
|
||||
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
|
||||
reconcile(t, r, "survival")
|
||||
s := getServer(t, c, "survival")
|
||||
if s.Annotations[v1alpha1.AnnotationRestart] != "" || s.Spec.DesiredState != v1alpha1.DesiredStopped || s.Status.Phase == v1alpha1.PhaseStarting {
|
||||
t.Fatalf("restart overrode stop: desired=%s phase=%s annotations=%v", s.Spec.DesiredState, s.Status.Phase, s.Annotations)
|
||||
}
|
||||
}
|
||||
|
||||
// Once the automatic restarts are spent, a retry request starts the server over
|
||||
// with the whole budget back: the pod is recreated, the start re-anchored, and
|
||||
// the next timeout is retried automatically again.
|
||||
|
||||
@@ -90,8 +90,8 @@ func ControlPlaneRBAC(p Params) RBAC {
|
||||
// claim. felis-api reads the fleet (velocity's pull, the fleet page, the wake
|
||||
// cap) from an informer cache of minecraftservers, hence watch on that one
|
||||
// resource; everything else goes through a DIRECT client, so it needs no
|
||||
// list/watch beyond the explicit List calls — and the PVC grant is get-only,
|
||||
// mirroring that.
|
||||
// list/watch beyond the explicit List calls. PVC list finds retained world
|
||||
// volumes before creating a server of the same name.
|
||||
//
|
||||
// The read-side grant is deliberately minimal: pods:list + pods/log:get, NOT
|
||||
// pods:get — the streamer lists pods by the server label then reads the chosen
|
||||
@@ -103,9 +103,7 @@ func APIMinecraftRole(p Params) *rbacv1.Role {
|
||||
rules := []rbacv1.PolicyRule{
|
||||
rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "list", "watch", "create", "patch"}),
|
||||
rule([]string{groupCore}, []string{"secrets"}, []string{"get"}),
|
||||
// get-only: WorldVolumeExists does a single direct Get of the world PVC;
|
||||
// nothing in felis-api lists or deletes PVCs.
|
||||
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get"}),
|
||||
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get", "list"}),
|
||||
// list backs GET /servers/{name}/jobs — the async status outlet reads the
|
||||
// backup/restore Jobs back by the server label — and finds the pending
|
||||
// restore chains; patch settles a chain by relabelling its safety-snapshot
|
||||
|
||||
@@ -127,13 +127,15 @@ func TestAPIRole_MinecraftPowersExact(t *testing.T) {
|
||||
if hasRule(mc, groupCore, "secrets", "list") || hasRule(mc, groupCore, "secrets", "watch") {
|
||||
t.Error("felis-api must NOT list or watch secrets (RCON reads are a direct Get by name)")
|
||||
}
|
||||
// WorldVolumeExists (backup/restore pre-gate) does a single direct PVC Get;
|
||||
// nothing in felis-api lists or deletes claims.
|
||||
// WorldVolumeExists reads a claim and lists retained claims on server creation.
|
||||
if !hasRule(mc, groupCore, "persistentvolumeclaims", "get") {
|
||||
t.Error("felis-api must get the world PVC (persistentvolumeclaims:get) for the backup/restore world-volume gate")
|
||||
}
|
||||
if hasRule(mc, groupCore, "persistentvolumeclaims", "list") || hasRule(mc, groupCore, "persistentvolumeclaims", "delete") {
|
||||
t.Error("felis-api must NOT list or delete PVCs (the gate is a single direct Get)")
|
||||
if !hasRule(mc, groupCore, "persistentvolumeclaims", "list") {
|
||||
t.Error("felis-api must list retained PVCs before creating a server")
|
||||
}
|
||||
if hasRule(mc, groupCore, "persistentvolumeclaims", "watch") || hasRule(mc, groupCore, "persistentvolumeclaims", "delete") {
|
||||
t.Error("felis-api must NOT watch or delete PVCs")
|
||||
}
|
||||
// Read-side console (spec §8 读=pods/log follow): list pods to find the
|
||||
// running pod, then read its log subresource — and nothing wider.
|
||||
|
||||
Reference in new issue
Block a user