fix(restore): replace a finished Job so retries enqueue; replicate felis-config
An E2E audit on a live install found that a FAILED restore held its deterministic Job name for the rest of the 10-minute TTL, so the next restore answered 202 'restoring' while nothing ran (ErrAlreadyExists was treated as success unconditionally). K8sJobs now inspects the colliding Job: in-flight still coalesces, finished (succeeded or failed) is deleted and replaced. The minecraft-namespace Role gains jobs:get/delete for exactly that replacement. The same audit found the backup Job mounts the felis-config Secret but the installer only provisions it in the control namespace, so every backup Job stranded on FailedMount. felis setup now replicates it into the minecraft namespace beside the service-token and forwarding secrets.
This commit is contained in:
6 files changed
+134
-27
No files matched your search
@@ -74,11 +74,13 @@ func ControlPlaneRBAC(p Params) RBAC {
|
||||
// APIMinecraftRole grants felis-api exactly what it does in the minecraft
|
||||
// namespace: drive MinecraftServer specs (internal/api.k8scluster — get/list/
|
||||
// create/patch, never status), read RCON passwords for console writes
|
||||
// (internal/api.console — secrets:get), create the restore Job
|
||||
// (internal/restore — jobs:create), and stream the live console for the read
|
||||
// side (internal/api.logstream — pods:list to find the server's running pod,
|
||||
// then pods/log:get to follow it; spec §8 读=pods/log follow). felis-api uses a
|
||||
// DIRECT client, so it needs no list/watch beyond the explicit List calls.
|
||||
// (internal/api.console — secrets:get), manage the restore Job under its
|
||||
// deterministic name (internal/restore — jobs:create, plus get/delete so a
|
||||
// FINISHED Job whose name still blocks a retry can be replaced), and stream the
|
||||
// live console for the read side (internal/api.logstream — pods:list to find
|
||||
// the server's running pod, then pods/log:get to follow it; spec §8 读=pods/log
|
||||
// follow). felis-api uses a DIRECT client, so it needs no list/watch beyond the
|
||||
// explicit List calls.
|
||||
//
|
||||
// The read-side grant is deliberately minimal: pods:list + pods/log:get, NOT
|
||||
// pods:get — the streamer lists pods by the server label then reads the chosen
|
||||
@@ -90,7 +92,7 @@ func APIMinecraftRole(p Params) *rbacv1.Role {
|
||||
return role(p.MinecraftNamespace, "felis-api", ComponentAPI, []rbacv1.PolicyRule{
|
||||
rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "list", "create", "patch"}),
|
||||
rule([]string{groupCore}, []string{"secrets"}, []string{"get"}),
|
||||
rule([]string{groupBatch}, []string{"jobs"}, []string{"create"}),
|
||||
rule([]string{groupBatch}, []string{"jobs"}, []string{"create", "get", "delete"}),
|
||||
// Read-side console (spec §8 读=pods/log follow): list pods to find the
|
||||
// server's running pod, then read its log subresource. Two separate rules so
|
||||
// the verbs stay tight — list on pods, get on pods/log, and nothing else.
|
||||
|
||||
@@ -73,8 +73,10 @@ func TestAPIRole_CreatesJobsInBothNamespaces(t *testing.T) {
|
||||
if mc.Namespace != "minecraft" {
|
||||
t.Errorf("felis-api minecraft Role namespace = %q, want minecraft", mc.Namespace)
|
||||
}
|
||||
if !hasRule(mc, "batch", "jobs", "create") {
|
||||
t.Error("felis-api (minecraft) must have batch/jobs:create for the restore Job")
|
||||
for _, v := range []string{"create", "get", "delete"} {
|
||||
if !hasRule(mc, "batch", "jobs", v) {
|
||||
t.Errorf("felis-api (minecraft) must have batch/jobs:%s for the restore-Job lifecycle", v)
|
||||
}
|
||||
}
|
||||
|
||||
build := roleByName(t, rbac.Roles, "felis-api-builds")
|
||||
|
||||
Reference in new issue
Block a user