fix(panel): streamline system settings and server creation

Use shared action buttons and default to the login space. Allow staff to save startup-only experience settings while running and request a durable restart through the existing operator flow.

Grant the API PVC list permission needed to detect retained worlds before server creation, and show internal failures with a request ID. Cover permission, maintenance, restart and UI behavior with regression checks.
This commit is contained in:
Lemon-miaow committed 2026-10-04 16:26:03 +08:00
1 parent e67896979e
commit cb3ed94026
34 files changed
+555 -103

No files matched your search

+1
View File
@@ -566,6 +566,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
// (handlers_retire.go); owner/admin-gated inside the handlers.
{Method: "PUT", Pattern: "/api/v1/servers/{name}/retirement", h: a.handleRetire},
{Method: "DELETE", Pattern: "/api/v1/servers/{name}/retirement", h: a.handleCancelRetire},
{Method: "POST", Pattern: "/api/v1/servers/{name}/restart", h: a.handleRestart},
{Method: "GET", Pattern: "/api/v1/servers/{name}/status", h: a.handleStatus},
// Identity self-read (spec §14 tiering): the panel reads this once at boot to
// learn its own tier and decide which navigation surfaces to render. App-tier —
+30
View File
@@ -1935,6 +1935,36 @@ func (c *fakeCluster) RetryStart(ctx context.Context, n string) error {
c.retried = append(c.retried, n)
return nil
}
func (c *fakeCluster) RestartServer(ctx context.Context, n string) error {
return c.RetryStart(ctx, n)
}
func TestRestartAuthorization(t *testing.T) {
for _, tc := range []struct {
name, server, role, user string
staff bool
status int
}{
{"owner of player server", "survival", "user", "owner1", false, 202},
{"other player", "survival", "user", "stranger", false, 403},
{"Owner console system service", "lobby", "owner", "owner1", true, 202},
{"player system service", "lobby", "user", "owner1", false, 400},
} {
t.Run(tc.name, func(t *testing.T) {
a, _, cl, _ := mkFiles(t)
cl.byName[tc.server] = &ServerInfo{Name: tc.server, Phase: "Running", DesiredState: "Running", Ready: true}
a.External = staticExternal{p: &Principal{UserID: tc.user, Role: tc.role, ViaAdminAccess: tc.staff}}
w := do(a.ExternalHandler(), "POST", "/api/v1/servers/"+tc.server+"/restart", "", nil)
if w.Code != tc.status {
t.Fatalf("status=%d want=%d body=%s", w.Code, tc.status, w.Body.String())
}
if (len(cl.retried) == 1) != (tc.status == 202) {
t.Fatalf("unauthorized restart or missing request: %v", cl.retried)
}
})
}
}
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
if err := c.maintErr[n]; err != nil {
return err
+3 -1
View File
@@ -137,8 +137,10 @@ type Cluster interface {
// also asks the operator to start it over with a fresh auto-restart budget
// (v1alpha1.AnnotationStartRetry). Maintenance refuses it the same way.
RetryStart(ctx context.Context, name string) error
// RestartServer requests a pod restart without changing desired state.
RestartServer(ctx context.Context, name string) error
// AcquireMaintenance admits one world-volume operation (internal/maintenance
// kind): ErrNotStopped unless the server is fully stopped, a
// kind): ErrNotStopped unless fully stopped (except system config writes), a
// *MaintenanceBusyError while another operation holds the volume. The check
// and the lock are one atomic write against a concurrent wake.
AcquireMaintenance(ctx context.Context, name, kind string) error
+19 -10
View File
@@ -15,6 +15,7 @@ import (
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/fileedit"
"felis.lolicon.best/internal/maintenance"
"felis.lolicon.best/internal/naming"
)
// FileEditor is the server-file-editor surface the API depends on: list a
@@ -233,7 +234,11 @@ func (a *API) handleWriteFile(w http.ResponseWriter, r *http.Request) {
// beside it, and a read that overlaps a restore or another change can at worst
// show a file mid-change: the sha256 it returned then no longer matches, so a
// save built on it is refused with file_changed.
release, ok := a.acquireWorld(w, r, name, maintenance.KindFileWrite, "stop the server before editing its files")
kind := maintenance.KindFileWrite
if liveExperienceFile(r, name) {
kind = maintenance.KindConfigWrite
}
release, ok := a.acquireWorld(w, r, name, kind, "stop the server before editing its files")
if !ok {
return
}
@@ -582,16 +587,12 @@ var sha256Hex = regexp.MustCompile(`^[0-9a-f]{64}$`)
// ② ServerByName — an unknown server is 404
// ③ owner-or-admin, else 403. An unowned (released) server fails for everyone
// but admin, which is the same "must re-claim first" rule restore enforces
// ④ stopped gate: the world PVC is RWO and held by a running server, so a file
// Job cannot mount it — refuse unless the server is fully stopped. Ready means
// it is up; any desiredState other than Stopped means it is up or coming up
// and still owns the volume. This yields a specific 409 instead of a Job that
// silently fails to mount
// ④ stopped gate: world files must not change under a live game process.
// Only the system plugin's startup-only config may be read and saved live;
// its Job mounts the RWO volume on the same node as the game pod.
// ⑤ the FileEditor must be wired, else 503
//
// Single-sourcing it is what keeps the handlers from drifting: a read path that
// forgot the stopped gate would not merely fail, it would hang waiting for a Pod
// that can never be scheduled.
// All file routes share this gate so the live-config exception stays narrow.
//
// It returns the validated server name and false if it has already written a
// response.
@@ -606,7 +607,7 @@ func (a *API) authorizeFileOp(w http.ResponseWriter, r *http.Request) (string, b
a.writeLookupError(w, r, err)
return "", false
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
if !liveExperienceFile(r, name) && (info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped)) {
writeError(w, r, newError(http.StatusConflict, "not_stopped",
"stop the server before working with its files"))
return "", false
@@ -636,6 +637,14 @@ func (a *API) authorizeFileOp(w http.ResponseWriter, r *http.Request) (string, b
return name, true
}
// Only the startup-only system plugin settings may be edited while running.
// World files, uploads, deletes and plugin binaries keep the stopped gate.
func liveExperienceFile(r *http.Request, name string) bool {
return naming.IsSystemServer(name) && strings.HasSuffix(r.URL.Path, "/file") &&
(r.Method == http.MethodGet || r.Method == http.MethodPut) &&
r.URL.Query().Get("path") == naming.ExperienceConfigFile
}
// writeFileEditError maps executor errors onto HTTP status codes. The
// sentinels are caller-fault and get precise answers; a timeout is reported as 504
// so the caller knows to retry rather than believing the edit was rejected; and
+43
View File
@@ -163,6 +163,49 @@ func mkFiles(t *testing.T) (*API, *fakeRepo, *fakeCluster, *fakeFileEditor) {
return api, repo, cl, files
}
func TestRunningSystemExperienceFile(t *testing.T) {
for _, name := range []string{"login", "lobby"} {
for _, tc := range []struct {
method, suffix, body string
allowed bool
}{
{"GET", "/file?path=felis-experience.json", "", true},
{"PUT", "/file?path=felis-experience.json", `{"content":"aGk=","content_sha256":"` + hiSum + `"}`, true},
{"PUT", "/file?path=felis-experience.json", `{"content":"aGk=","content_sha256":"` + hiSum + `","create_only":true}`, true},
{"PUT", "/file?path=world/level.dat", `{"content":"aGk=","content_sha256":"` + hiSum + `"}`, false},
{"GET", "/file?path=./felis-experience.json", "", false},
{"DELETE", "/file?path=felis-experience.json", "", false},
{"GET", "/files", "", false},
} {
t.Run(name+" "+tc.method+tc.suffix, func(t *testing.T) {
a, _, cl, files := mkFiles(t)
cl.byName[name] = &ServerInfo{Name: name, Phase: "Running", Ready: true, DesiredState: "Running"}
a.External = staticExternal{p: &Principal{UserID: "owner1", Role: "owner", ViaAdminAccess: true}}
w := do(a.ExternalHandler(), tc.method, "/api/v1/servers/"+name+tc.suffix, tc.body, jsonHeader)
if tc.allowed {
if w.Code != http.StatusOK || files.calls != 1 {
t.Fatalf("live config: status=%d calls=%d body=%s", w.Code, files.calls, w.Body.String())
}
if tc.method == "PUT" && (len(cl.acquired) != 1 || cl.acquired[0] != name+":"+maintenance.KindConfigWrite) {
t.Fatalf("wrong lock: %v", cl.acquired)
}
} else if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" || files.calls != 0 {
t.Fatalf("world gate: status=%d calls=%d body=%s", w.Code, files.calls, w.Body.String())
}
})
}
}
t.Run("player cannot edit system config", func(t *testing.T) {
a, _, cl, files := mkFiles(t)
cl.byName["lobby"] = &ServerInfo{Name: "lobby", Phase: "Running", Ready: true, DesiredState: "Running"}
a.External = staticExternal{p: &Principal{UserID: "owner1", Role: "user"}}
w := do(a.ExternalHandler(), "GET", "/api/v1/servers/lobby/file?path=felis-experience.json", "", nil)
if w.Code < 400 || files.calls != 0 {
t.Fatalf("player reached config: status=%d calls=%d", w.Code, files.calls)
}
})
}
// TestFileEditorStoppedGate is the gate this whole subsystem hinges on. The world
// PVC is ReadWriteOnce, but RWO is per node: on a single node a file Job mounts it
// right beside a running server, and a write lands under a live world that the
+23
View File
@@ -92,6 +92,29 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
}
// handleRestart records a restart request for the operator to consume once.
func (a *API) handleRestart(w http.ResponseWriter, r *http.Request) {
name, ok := a.authorizeServerFiles(w, r)
if !ok {
return
}
rec, err := a.managedServerRecord(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if rec.Retire != nil {
writeError(w, r, errServerRetiring)
return
}
if err := a.Cluster.RestartServer(r.Context(), rec.Name); err != nil {
a.writeLookupError(w, r, maintenanceError(err, "wait for maintenance to finish before restarting"))
return
}
a.audit(r, "server.restart", rec.Name)
writeJSON(w, http.StatusAccepted, map[string]string{"name": rec.Name, "desiredState": "Running"})
}
// handleStop flips desiredState to Stopped. Only the owner or an admin may stop a
// server (spec §14: operating someone else's server is admin-tier).
func (a *API) handleStop(w http.ResponseWriter, r *http.Request) {
+27 -13
View File
@@ -3,6 +3,7 @@ package api
import (
"context"
"errors"
"net/http"
"strings"
"time"
@@ -264,22 +265,29 @@ func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1a
// against, the same as AcquireMaintenance's: whichever of a racing wake and
// admission writes second gets a conflict, re-reads, and sees the other.
func (k *K8sCluster) start(ctx context.Context, name string) error {
return k.startWith(ctx, name, false)
return k.startWith(ctx, name, "")
}
// RetryStart starts a Failed server over: the same guarded write as start, plus
// v1alpha1.AnnotationStartRetry so the operator resets the restart budget and
// recreates the pod.
func (k *K8sCluster) RetryStart(ctx context.Context, name string) error {
return k.startWith(ctx, name, true)
return k.startWith(ctx, name, v1alpha1.AnnotationStartRetry)
}
func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed bool) error {
func (k *K8sCluster) RestartServer(ctx context.Context, name string) error {
return k.startWith(ctx, name, v1alpha1.AnnotationRestart)
}
func (k *K8sCluster) startWith(ctx context.Context, name, annotation string) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer
if err := k.getServer(ctx, name, &ms); err != nil {
return err
}
if annotation == v1alpha1.AnnotationRestart && (ms.Spec.DesiredState != v1alpha1.DesiredRunning || ms.Status.Phase != v1alpha1.PhaseRunning) {
return newError(http.StatusConflict, "not_running", "start the server before restarting it")
}
node := ms.Spec.NodeName
if node == "" {
node = ms.Status.NodeName
@@ -308,11 +316,11 @@ func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed boo
// A lock still on the object here no longer holds anything (Holder said
// so): drop it in the same write.
delete(ms.Annotations, maintenance.Annotation)
if retryFailed {
if annotation != "" {
if ms.Annotations == nil {
ms.Annotations = map[string]string{}
}
ms.Annotations[v1alpha1.AnnotationStartRetry] = k.clock().UTC().Format(time.RFC3339)
ms.Annotations[annotation] = k.clock().UTC().Format(time.RFC3339Nano)
}
return k.c.Patch(ctx, &ms, patch)
})
@@ -320,8 +328,9 @@ func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed boo
// AcquireMaintenance admits one world-volume operation of the given kind: the
// server must be fully stopped (desiredState Stopped, phase Stopped, and no game
// pod left, so a pod still saving on its way down is waited out) and nothing else
// may hold the volume. Admission writes the maintenance lock under the resourceVersion it
// pod left, so a pod still saving on its way down is waited out). System config
// writes may run beside the game pod, but not during a pending restart. Nothing
// else may hold the volume. Admission writes the lock under the resourceVersion it
// checked; the caller creates its Job and then calls ReleaseMaintenance, after
// which the Job itself is the lock.
func (k *K8sCluster) AcquireMaintenance(ctx context.Context, name, kind string) error {
@@ -334,13 +343,18 @@ func (k *K8sCluster) AcquireMaintenance(ctx context.Context, name, kind string)
if desired == "" {
desired = v1alpha1.DesiredStopped
}
if desired != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
return ErrNotStopped
if kind != maintenance.KindConfigWrite || !naming.IsSystemServer(name) {
if desired != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
return ErrNotStopped
}
if up, err := k.gamePodExists(ctx, name); err != nil {
return err
} else if up {
return ErrNotStopped
}
}
if up, err := k.gamePodExists(ctx, name); err != nil {
return err
} else if up {
return ErrNotStopped
if ms.Annotations[v1alpha1.AnnotationRestart] != "" {
return ErrMaintenanceInProgress
}
holder, held, err := k.maintenanceHolder(ctx, &ms)
if err != nil {
@@ -35,6 +35,54 @@ func stoppedServer() *v1alpha1.MinecraftServer {
}
}
func TestSystemConfigSaveAndRestartAdmission(t *testing.T) {
ctx := context.Background()
ms := stoppedServer()
ms.Name = "lobby"
ms.Spec.DesiredState = v1alpha1.DesiredRunning
ms.Status.Phase, ms.Status.Ready = v1alpha1.PhaseRunning, true
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "lobby-0", Namespace: "minecraft",
Labels: map[string]string{v1alpha1.LabelServer: "lobby", v1alpha1.LabelComponent: gamePodComponent}}}
k, c := lockCluster(t, ms, pod)
if err := k.AcquireMaintenance(ctx, "lobby", maintenance.KindConfigWrite); err != nil {
t.Fatalf("save beside running pod: %v", err)
}
if err := k.RestartServer(ctx, "lobby"); !errors.Is(err, ErrMaintenanceInProgress) {
t.Fatalf("restart during save: %v", err)
}
if err := k.ReleaseMaintenance(ctx, "lobby"); err != nil {
t.Fatal(err)
}
if err := k.RestartServer(ctx, "lobby"); err != nil {
t.Fatalf("restart after save: %v", err)
}
var got v1alpha1.MinecraftServer
if err := c.Get(ctx, client.ObjectKeyFromObject(ms), &got); err != nil {
t.Fatal(err)
}
if got.Annotations[v1alpha1.AnnotationRestart] == "" || got.Spec.DesiredState != v1alpha1.DesiredRunning {
t.Fatalf("restart not recorded durably: %+v", got)
}
if err := c.Get(ctx, client.ObjectKeyFromObject(pod), &corev1.Pod{}); err != nil {
t.Fatalf("API must leave pod deletion to operator: %v", err)
}
if err := k.AcquireMaintenance(ctx, "lobby", maintenance.KindConfigWrite); !errors.Is(err, ErrMaintenanceInProgress) {
t.Fatalf("save racing a pending restart: %v", err)
}
userServer := stoppedServer()
userServer.Spec.DesiredState = v1alpha1.DesiredRunning
userServer.Status.Phase = v1alpha1.PhaseRunning
k, _ = lockCluster(t, userServer)
if err := k.AcquireMaintenance(ctx, "survival", maintenance.KindConfigWrite); !errors.Is(err, ErrNotStopped) {
t.Fatalf("live config exception reached a player world: %v", err)
}
k, _ = lockCluster(t, stoppedServer())
if err := k.RestartServer(ctx, "survival"); err == nil {
t.Fatal("restart admitted a stopped server")
}
}
func lockCluster(t *testing.T, objs ...client.Object) (*K8sCluster, client.Client) {
t.Helper()
scheme := runtime.NewScheme()
+2
View File
@@ -37,6 +37,8 @@ func maintenanceLabel(kind string) string {
return "a backup"
case maintenance.KindFileWrite:
return "a file write"
case maintenance.KindConfigWrite:
return "a settings save"
case maintenance.KindExport:
return "a world export or file download"
case maintenance.KindReap:
@@ -39,6 +39,8 @@ const (
// the automatic restarts were spent. The operator takes the request once —
// fresh restart budget, new start anchor, pod recreated — and removes it.
AnnotationStartRetry = GroupName + "/start-retry"
// AnnotationRestart requests one pod recreation while keeping desiredState Running.
AnnotationRestart = GroupName + "/restart"
)
// ForwardingLegacy is the LabelForwarding value that selects legacy forwarding.
+3
View File
@@ -529,6 +529,9 @@ func write(r *os.Root, name string, content []byte, sum, expect string, createOn
if res.Code != "" {
return res
}
if name == naming.ExperienceConfigFile && target != name {
return Result{Code: CodeBadPath, Error: "the experience config must be a regular file, not a symlink"}
}
if expect != "" {
if res := checkUnchanged(r, name, target, expect); res.Code != "" {
return res
+15
View File
@@ -45,6 +45,21 @@ func run(root, op, path string, content []byte, expect string) (Result, error) {
return Execute(root, Request{Op: op, Path: path, Content: content, Expect: expect})
}
func TestExperienceConfigCannotWriteThroughSymlink(t *testing.T) {
root, _ := worldRoot(t)
if err := os.Symlink("server.properties", filepath.Join(root, "felis-experience.json")); err != nil {
t.Fatal(err)
}
res, err := run(root, OpWrite, "felis-experience.json", []byte("{}"), "")
if err != nil || res.Code != CodeBadPath {
t.Fatalf("symlink write: result=%+v err=%v", res, err)
}
content, err := os.ReadFile(filepath.Join(root, "server.properties"))
if err != nil || string(content) != "motd=hello\n" {
t.Fatalf("live world file was changed: content=%q err=%v", content, err)
}
}
// TestExecuteContainment is the security test of this package. The world directory
// holds attacker-influenced content (players and plugins create files in it), so
// each vector below is a path a caller could genuinely supply to try to leave the
+6 -5
View File
@@ -90,11 +90,12 @@ const (
// Kinds of holder.
const (
KindMigration = "migration"
KindRestore = "restore"
KindBackup = "backup"
KindFileWrite = "file-write"
KindExport = "export"
KindMigration = "migration"
KindRestore = "restore"
KindBackup = "backup"
KindFileWrite = "file-write"
KindConfigWrite = "config-write"
KindExport = "export"
// KindReap is the reaper archiving an idle world and reclaiming its volume.
// It runs no Job: the reaper holds the Annotation itself and rewrites it
// well inside Grace for as long as it works on the world.
+2
View File
@@ -46,6 +46,8 @@ const (
// reachable only after the login gate passes a player through, so it must
// never be used as a fallback target (that would bypass the gate).
SystemLobbyServer = "lobby"
// ExperienceConfigFile is read by the system plugins only at startup.
ExperienceConfigFile = "felis-experience.json"
)
// IsSystemServer identifies platform services whose reserved names may be managed
+19 -13
View File
@@ -218,9 +218,11 @@ func (r *Reconciler) reconcile(ctx context.Context, req ctrl.Request) (ctrl.Resu
desired = v1alpha1.DesiredStopped
}
prevPhase, prevRestarts := server.Status.Phase, server.Status.AutoRestarts
if _, ok := server.Annotations[v1alpha1.AnnotationStartRetry]; ok {
if err := r.takeStartRetry(ctx, &server, desired); err != nil {
return ctrl.Result{}, err
for _, annotation := range []string{v1alpha1.AnnotationStartRetry, v1alpha1.AnnotationRestart} {
if _, ok := server.Annotations[annotation]; ok {
if err := r.takeRestartRequest(ctx, &server, desired, annotation); err != nil {
return ctrl.Result{}, err
}
}
}
var res ctrl.Result
@@ -263,30 +265,34 @@ func (r *Reconciler) recordTransition(ctx context.Context, server *v1alpha1.Mine
r.event(server, eventType, reason, fmt.Sprintf("%s → %s: %s", from, phase, msg))
}
// takeStartRetry answers felis-api's AnnotationStartRetry: a server still Failed
// and meant to run starts over as if freshly woken — pod recreated, restart
// budget back to zero, a new start anchor — and in any other state the request
// is stale and only removed. The fresh status is written before the request is
// takeRestartRequest recreates a Failed pod for a start retry, or a Running pod
// for an explicit restart. A request overtaken by a stop is only removed.
// The fresh status is written before the request is
// removed, so a pass that fails in between leaves the request to be taken again
// (at the cost of one more pod recreate), never a removed request with the old
// spent budget still in place.
func (r *Reconciler) takeStartRetry(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState) error {
if desired == v1alpha1.DesiredRunning && server.Status.Phase == v1alpha1.PhaseFailed {
func (r *Reconciler) takeRestartRequest(ctx context.Context, server *v1alpha1.MinecraftServer, desired v1alpha1.DesiredState, annotation string) error {
explicit := annotation == v1alpha1.AnnotationRestart
if desired == v1alpha1.DesiredRunning && (server.Status.Phase == v1alpha1.PhaseFailed || (explicit && server.Status.Phase == v1alpha1.PhaseRunning)) {
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: server.Name + "-0", Namespace: server.Namespace}}
if err := r.Delete(ctx, pod); client.IgnoreNotFound(err) != nil {
return err
}
server.Status.AutoRestarts = 0
server.Status.StartRequestedAt = nil
r.markStarting(server, "StartRetried", "start requested again after it failed; recreated the pod")
reason, message := "StartRetried", "start requested again after it failed; recreated the pod"
if explicit {
reason, message = "RestartRequested", "restart requested; recreated the pod"
}
r.markStarting(server, reason, message)
if err := r.patchStatus(ctx, server); err != nil {
return err
}
log.FromContext(ctx).Info("start retried")
r.event(server, corev1.EventTypeNormal, "StartRetried", "start requested again after it failed; recreated the pod")
log.FromContext(ctx).Info(message)
r.event(server, corev1.EventTypeNormal, reason, message)
}
patch := client.MergeFrom(server.DeepCopy())
delete(server.Annotations, v1alpha1.AnnotationStartRetry)
delete(server.Annotations, annotation)
return r.Patch(ctx, server, patch)
}
+32
View File
@@ -37,6 +37,38 @@ func podPresent(t *testing.T, c client.Client) bool {
return err == nil
}
func TestExplicitRestartIsConsumedOnce(t *testing.T) {
srv := runningServer()
srv.Status.Phase = v1alpha1.PhaseRunning
srv.Annotations = map[string]string{v1alpha1.AnnotationRestart: "2026-07-01T12:00:00Z"}
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
reconcile(t, r, "survival")
s := getServer(t, c, "survival")
if s.Annotations[v1alpha1.AnnotationRestart] != "" || s.Spec.DesiredState != v1alpha1.DesiredRunning || s.Status.Phase != v1alpha1.PhaseStarting || podPresent(t, c) {
t.Fatalf("restart: desired=%s phase=%s request=%s pod=%v", s.Spec.DesiredState, s.Status.Phase, s.Annotations[v1alpha1.AnnotationRestart], podPresent(t, c))
}
if err := c.Create(context.Background(), gamePod()); err != nil {
t.Fatal(err)
}
reconcile(t, r, "survival")
if !podPresent(t, c) {
t.Fatal("consumed request deleted the replacement pod")
}
}
func TestStopSupersedesExplicitRestart(t *testing.T) {
srv := runningServer()
srv.Spec.DesiredState = v1alpha1.DesiredStopped
srv.Status.Phase = v1alpha1.PhaseRunning
srv.Annotations = map[string]string{v1alpha1.AnnotationRestart: "2026-07-01T12:00:00Z"}
r, c := newReconciler(t, fakeProber{}, srv, rconSecret(), gamePod())
reconcile(t, r, "survival")
s := getServer(t, c, "survival")
if s.Annotations[v1alpha1.AnnotationRestart] != "" || s.Spec.DesiredState != v1alpha1.DesiredStopped || s.Status.Phase == v1alpha1.PhaseStarting {
t.Fatalf("restart overrode stop: desired=%s phase=%s annotations=%v", s.Spec.DesiredState, s.Status.Phase, s.Annotations)
}
}
// Once the automatic restarts are spent, a retry request starts the server over
// with the whole budget back: the pod is recreated, the start re-anchored, and
// the next timeout is retried automatically again.
+3 -5
View File
@@ -90,8 +90,8 @@ func ControlPlaneRBAC(p Params) RBAC {
// claim. felis-api reads the fleet (velocity's pull, the fleet page, the wake
// cap) from an informer cache of minecraftservers, hence watch on that one
// resource; everything else goes through a DIRECT client, so it needs no
// list/watch beyond the explicit List calls — and the PVC grant is get-only,
// mirroring that.
// list/watch beyond the explicit List calls. PVC list finds retained world
// volumes before creating a server of the same name.
//
// The read-side grant is deliberately minimal: pods:list + pods/log:get, NOT
// pods:get — the streamer lists pods by the server label then reads the chosen
@@ -103,9 +103,7 @@ func APIMinecraftRole(p Params) *rbacv1.Role {
rules := []rbacv1.PolicyRule{
rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "list", "watch", "create", "patch"}),
rule([]string{groupCore}, []string{"secrets"}, []string{"get"}),
// get-only: WorldVolumeExists does a single direct Get of the world PVC;
// nothing in felis-api lists or deletes PVCs.
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get"}),
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get", "list"}),
// list backs GET /servers/{name}/jobs — the async status outlet reads the
// backup/restore Jobs back by the server label — and finds the pending
// restore chains; patch settles a chain by relabelling its safety-snapshot
+6 -4
View File
@@ -127,13 +127,15 @@ func TestAPIRole_MinecraftPowersExact(t *testing.T) {
if hasRule(mc, groupCore, "secrets", "list") || hasRule(mc, groupCore, "secrets", "watch") {
t.Error("felis-api must NOT list or watch secrets (RCON reads are a direct Get by name)")
}
// WorldVolumeExists (backup/restore pre-gate) does a single direct PVC Get;
// nothing in felis-api lists or deletes claims.
// WorldVolumeExists reads a claim and lists retained claims on server creation.
if !hasRule(mc, groupCore, "persistentvolumeclaims", "get") {
t.Error("felis-api must get the world PVC (persistentvolumeclaims:get) for the backup/restore world-volume gate")
}
if hasRule(mc, groupCore, "persistentvolumeclaims", "list") || hasRule(mc, groupCore, "persistentvolumeclaims", "delete") {
t.Error("felis-api must NOT list or delete PVCs (the gate is a single direct Get)")
if !hasRule(mc, groupCore, "persistentvolumeclaims", "list") {
t.Error("felis-api must list retained PVCs before creating a server")
}
if hasRule(mc, groupCore, "persistentvolumeclaims", "watch") || hasRule(mc, groupCore, "persistentvolumeclaims", "delete") {
t.Error("felis-api must NOT watch or delete PVCs")
}
// Read-side console (spec §8 读=pods/log follow): list pods to find the
// running pod, then read its log subresource — and nothing wider.