fix(restore): replace a finished Job so retries enqueue; replicate felis-config

An E2E audit on a live install found that a FAILED restore held its
deterministic Job name for the rest of the 10-minute TTL, so the next
restore answered 202 'restoring' while nothing ran (ErrAlreadyExists was
treated as success unconditionally). K8sJobs now inspects the colliding
Job: in-flight still coalesces, finished (succeeded or failed) is
deleted and replaced. The minecraft-namespace Role gains jobs:get/delete
for exactly that replacement.

The same audit found the backup Job mounts the felis-config Secret but
the installer only provisions it in the control namespace, so every
backup Job stranded on FailedMount. felis setup now replicates it into
the minecraft namespace beside the service-token and forwarding
secrets.
This commit is contained in:
Lemon-miaow committed 2026-09-22 17:50:04 +08:00
1 parent fd0794d04d
commit 90ccbfede4
6 files changed
+134 -27

No files matched your search

+8 -6
View File
@@ -74,11 +74,13 @@ func ControlPlaneRBAC(p Params) RBAC {
// APIMinecraftRole grants felis-api exactly what it does in the minecraft
// namespace: drive MinecraftServer specs (internal/api.k8scluster — get/list/
// create/patch, never status), read RCON passwords for console writes
// (internal/api.console — secrets:get), create the restore Job
// (internal/restore — jobs:create), and stream the live console for the read
// side (internal/api.logstream — pods:list to find the server's running pod,
// then pods/log:get to follow it; spec §8 读=pods/log follow). felis-api uses a
// DIRECT client, so it needs no list/watch beyond the explicit List calls.
// (internal/api.console — secrets:get), manage the restore Job under its
// deterministic name (internal/restore — jobs:create, plus get/delete so a
// FINISHED Job whose name still blocks a retry can be replaced), and stream the
// live console for the read side (internal/api.logstream — pods:list to find
// the server's running pod, then pods/log:get to follow it; spec §8 读=pods/log
// follow). felis-api uses a DIRECT client, so it needs no list/watch beyond the
// explicit List calls.
//
// The read-side grant is deliberately minimal: pods:list + pods/log:get, NOT
// pods:get — the streamer lists pods by the server label then reads the chosen
@@ -90,7 +92,7 @@ func APIMinecraftRole(p Params) *rbacv1.Role {
return role(p.MinecraftNamespace, "felis-api", ComponentAPI, []rbacv1.PolicyRule{
rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "list", "create", "patch"}),
rule([]string{groupCore}, []string{"secrets"}, []string{"get"}),
rule([]string{groupBatch}, []string{"jobs"}, []string{"create"}),
rule([]string{groupBatch}, []string{"jobs"}, []string{"create", "get", "delete"}),
// Read-side console (spec §8 读=pods/log follow): list pods to find the
// server's running pod, then read its log subresource. Two separate rules so
// the verbs stay tight — list on pods, get on pods/log, and nothing else.
+4 -2
View File
@@ -73,8 +73,10 @@ func TestAPIRole_CreatesJobsInBothNamespaces(t *testing.T) {
if mc.Namespace != "minecraft" {
t.Errorf("felis-api minecraft Role namespace = %q, want minecraft", mc.Namespace)
}
if !hasRule(mc, "batch", "jobs", "create") {
t.Error("felis-api (minecraft) must have batch/jobs:create for the restore Job")
for _, v := range []string{"create", "get", "delete"} {
if !hasRule(mc, "batch", "jobs", v) {
t.Errorf("felis-api (minecraft) must have batch/jobs:%s for the restore-Job lifecycle", v)
}
}
build := roleByName(t, rbac.Roles, "felis-api-builds")