feat(backup): add backup, restore, and reaper subsystems
Archive-based world backup and restore, plus the reaper that enforces retention and reclaims idle servers.
This commit is contained in:
12 files changed
+2749
No files matched your search
@@ -0,0 +1,225 @@
|
||||
package restore
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
)
|
||||
|
||||
// Label keys applied to restore objects, mirroring internal/build so the two
|
||||
// executors are observable the same way.
|
||||
const (
|
||||
LabelManagedBy = "app.kubernetes.io/managed-by"
|
||||
LabelComponent = "app.kubernetes.io/component"
|
||||
LabelServer = "felis.lolicon.best/server"
|
||||
|
||||
managedByValue = "felis-restore"
|
||||
componentValue = "world-restore"
|
||||
|
||||
worldVolume = "world"
|
||||
backupVolume = "backup"
|
||||
)
|
||||
|
||||
// JobParams are the rendered inputs to a restore Job, derived from a server +
|
||||
// archive ref + Config by the Restorer. jobspec is a pure function of them so
|
||||
// the security-critical Job shape is unit-tested without a cluster.
|
||||
type JobParams struct {
|
||||
Server string
|
||||
WorldPVC string
|
||||
BackupPVC string
|
||||
BackupRef string
|
||||
ArchiveStore string
|
||||
Namespace string
|
||||
ServiceAccount string
|
||||
Image string
|
||||
BackupRoot string
|
||||
WorldsRoot string
|
||||
Deadline time.Duration
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
RunAsUser int64
|
||||
RunAsGroup int64
|
||||
FSGroup int64
|
||||
|
||||
TTLAfterFinished time.Duration
|
||||
}
|
||||
|
||||
// RestoreJobName is the deterministic Job name for a server's restore. It is a
|
||||
// pure function of the server name, which is how CreateRestoreJob detects a
|
||||
// restore already in flight (AlreadyExists) and how the orchestrator's
|
||||
// idempotency holds.
|
||||
func RestoreJobName(server string) string { return "restore-" + server }
|
||||
|
||||
func restoreLabels(p JobParams) map[string]string {
|
||||
return map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
LabelServer: p.Server,
|
||||
}
|
||||
}
|
||||
|
||||
// RestoreJob renders the world-restore Job (spec §7, §16, §22). Every isolation
|
||||
// guarantee lives here and is asserted by jobspec_test.go, because no cluster
|
||||
// runs in this environment:
|
||||
//
|
||||
// - runs under the weak felis-restore SA (never the felis-api SA) with its
|
||||
// token auto-mount disabled, so it cannot reach the K8s API (spec §16, §21);
|
||||
// - mounts EXACTLY two volumes — the world PVC read-write and the backup PVC
|
||||
// read-only — and NO Secret/ConfigMap, so a poisoned archive cannot reach
|
||||
// the felis database or any credential (the four-power red line, spec §22);
|
||||
// - runs as a non-root, fixed uid/gid with an fsGroup so the files it writes
|
||||
// are owned by the same identity the minecraft server later runs as;
|
||||
// - no privilege, no privilege escalation, read-only root filesystem, drop ALL
|
||||
// capabilities — all writes go to the mounted world PVC, nothing else;
|
||||
// - activeDeadlineSeconds + backoffLimit=0 so a wedged or malicious archive
|
||||
// cannot loop or run forever; ttlSecondsAfterFinished GCs the finished Job.
|
||||
//
|
||||
// The container runs `felis restore` (cmd/felis), which extracts the archive at
|
||||
// BackupRef from the backup mount into the world mount. BackupRef is an absolute
|
||||
// path, so the backup PVC MUST be mounted at BackupRoot — the same path the
|
||||
// reaper wrote it under — for the ref to resolve.
|
||||
func RestoreJob(p JobParams) (*batchv1.Job, error) {
|
||||
if p.Image == "" {
|
||||
return nil, fmt.Errorf("restore: image is empty")
|
||||
}
|
||||
if p.WorldPVC == "" || p.BackupPVC == "" {
|
||||
return nil, fmt.Errorf("restore: world and backup PVC names are required")
|
||||
}
|
||||
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
deadline := int64(p.Deadline / time.Second)
|
||||
if deadline <= 0 {
|
||||
deadline = int64(defaultDeadline / time.Second)
|
||||
}
|
||||
ttl := int32(p.TTLAfterFinished / time.Second)
|
||||
if ttl <= 0 {
|
||||
ttl = int32(defaultTTL / time.Second)
|
||||
}
|
||||
|
||||
container := corev1.Container{
|
||||
Name: "restore",
|
||||
Image: p.Image,
|
||||
Command: []string{"felis", "restore"},
|
||||
Args: []string{
|
||||
"--server", p.Server,
|
||||
"--ref", p.BackupRef,
|
||||
"--archive-store", p.ArchiveStore,
|
||||
"--backup-root", p.BackupRoot,
|
||||
"--worlds-root", p.WorldsRoot,
|
||||
},
|
||||
VolumeMounts: []corev1.VolumeMount{
|
||||
{Name: worldVolume, MountPath: p.WorldsRoot},
|
||||
// The archive is only ever read; mounting it read-only means a
|
||||
// compromised restore process cannot mutate other servers' backups.
|
||||
{Name: backupVolume, MountPath: p.BackupRoot, ReadOnly: true},
|
||||
},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||
SecurityContext: &corev1.SecurityContext{
|
||||
Privileged: boolPtr(false),
|
||||
AllowPrivilegeEscalation: boolPtr(false),
|
||||
ReadOnlyRootFilesystem: boolPtr(true),
|
||||
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
||||
},
|
||||
}
|
||||
|
||||
job := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: RestoreJobName(p.Server),
|
||||
Namespace: p.Namespace,
|
||||
Labels: restoreLabels(p),
|
||||
},
|
||||
Spec: batchv1.JobSpec{
|
||||
// One shot: a bad archive must not loop. The TTL GCs the finished Job
|
||||
// so a later restore of the same server is not blocked forever by a
|
||||
// stale completed Job.
|
||||
BackoffLimit: int32Ptr(0),
|
||||
ActiveDeadlineSeconds: int64Ptr(deadline),
|
||||
TTLSecondsAfterFinished: int32Ptr(ttl),
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: restoreLabels(p)},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: p.ServiceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
SecurityContext: &corev1.PodSecurityContext{
|
||||
RunAsNonRoot: boolPtr(true),
|
||||
RunAsUser: int64Ptr(p.RunAsUser),
|
||||
RunAsGroup: int64Ptr(p.RunAsGroup),
|
||||
FSGroup: int64Ptr(p.FSGroup),
|
||||
},
|
||||
Containers: []corev1.Container{container},
|
||||
Volumes: []corev1.Volume{
|
||||
{
|
||||
Name: worldVolume,
|
||||
VolumeSource: corev1.VolumeSource{
|
||||
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
||||
ClaimName: p.WorldPVC,
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
Name: backupVolume,
|
||||
VolumeSource: corev1.VolumeSource{
|
||||
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
||||
ClaimName: p.BackupPVC,
|
||||
ReadOnly: true,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
return job, nil
|
||||
}
|
||||
|
||||
// RestoreServiceAccount renders the weak restore SA (spec §16, §21). Like the
|
||||
// build SA it is created bare: no secrets, token auto-mounting disabled, and —
|
||||
// by having no Role or RoleBinding anywhere — zero K8s API permissions. Its only
|
||||
// capability is filesystem access to the two PVCs the Job mounts.
|
||||
func RestoreServiceAccount(namespace, name string) *corev1.ServiceAccount {
|
||||
return &corev1.ServiceAccount{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: namespace,
|
||||
Labels: map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
},
|
||||
},
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
}
|
||||
}
|
||||
|
||||
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
||||
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||
if cpu == "" {
|
||||
cpu = defaultCPULimit
|
||||
}
|
||||
if mem == "" {
|
||||
mem = defaultMemLimit
|
||||
}
|
||||
cpuQty, err := resource.ParseQuantity(cpu)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("restore: invalid cpu limit %q: %w", cpu, err)
|
||||
}
|
||||
memQty, err := resource.ParseQuantity(mem)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("restore: invalid memory limit %q: %w", mem, err)
|
||||
}
|
||||
return corev1.ResourceList{
|
||||
corev1.ResourceCPU: cpuQty,
|
||||
corev1.ResourceMemory: memQty,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
func int32Ptr(i int32) *int32 { return &i }
|
||||
func int64Ptr(i int64) *int64 { return &i }
|
||||
@@ -0,0 +1,282 @@
|
||||
package restore
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
func sampleJobParams() JobParams {
|
||||
return JobParams{
|
||||
Server: "survival",
|
||||
WorldPVC: "world-survival-0",
|
||||
BackupPVC: "felis-backups",
|
||||
BackupRef: "/backups/survival/2026-06-25.tar.gz",
|
||||
ArchiveStore: "tarLocal",
|
||||
Namespace: defaultNamespace,
|
||||
ServiceAccount: defaultServiceAccount,
|
||||
Image: "registry.felis.svc:5000/felis:1.0",
|
||||
BackupRoot: "/backups",
|
||||
WorldsRoot: "/world",
|
||||
Deadline: 30 * time.Minute,
|
||||
CPULimit: "1",
|
||||
MemLimit: "1Gi",
|
||||
RunAsUser: 1000,
|
||||
RunAsGroup: 1000,
|
||||
FSGroup: 1000,
|
||||
TTLAfterFinished: 10 * time.Minute,
|
||||
}
|
||||
}
|
||||
|
||||
// The restore Pod must run under the weak felis-restore SA — never the
|
||||
// felis-api identity — with its token un-mounted, so it cannot reach the K8s
|
||||
// API. This is the §16/§22 red line asserted on the rendered spec because no
|
||||
// cluster runs here.
|
||||
func TestRestoreJobRunsUnderWeakSA(t *testing.T) {
|
||||
job, err := RestoreJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("RestoreJob: %v", err)
|
||||
}
|
||||
sa := job.Spec.Template.Spec.ServiceAccountName
|
||||
if sa != defaultServiceAccount {
|
||||
t.Errorf("service account = %q, want %q", sa, defaultServiceAccount)
|
||||
}
|
||||
if sa == "felis-api" {
|
||||
t.Fatal("restore Pod must NOT run as the felis-api SA")
|
||||
}
|
||||
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
|
||||
t.Error("AutomountServiceAccountToken must be explicitly false")
|
||||
}
|
||||
}
|
||||
|
||||
// The four-power red line: a restore Pod handles a (potentially poisoned)
|
||||
// archive, so it must mount EXACTLY the two PVCs — world read-write, backup
|
||||
// read-only — and NO Secret or ConfigMap, so it can never reach the felis
|
||||
// database or any credential.
|
||||
func TestRestoreJobMountsOnlyTheTwoPVCsAndNoSecrets(t *testing.T) {
|
||||
job, err := RestoreJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("RestoreJob: %v", err)
|
||||
}
|
||||
vols := job.Spec.Template.Spec.Volumes
|
||||
if len(vols) != 2 {
|
||||
t.Fatalf("expected exactly 2 volumes (world + backup), got %d: %+v", len(vols), vols)
|
||||
}
|
||||
var world, backup *corev1.Volume
|
||||
for i := range vols {
|
||||
v := &vols[i]
|
||||
// The forbidden volume kinds: anything that could carry DB creds or
|
||||
// reach the API.
|
||||
if v.Secret != nil {
|
||||
t.Errorf("volume %q is a Secret — a restore Pod must never mount a Secret", v.Name)
|
||||
}
|
||||
if v.ConfigMap != nil {
|
||||
t.Errorf("volume %q is a ConfigMap — no config/credential injection allowed", v.Name)
|
||||
}
|
||||
if v.Projected != nil || v.DownwardAPI != nil {
|
||||
t.Errorf("volume %q is a projected/downward volume — could surface the SA token", v.Name)
|
||||
}
|
||||
if v.HostPath != nil {
|
||||
t.Errorf("volume %q is a hostPath — no node filesystem access allowed", v.Name)
|
||||
}
|
||||
if v.PersistentVolumeClaim == nil {
|
||||
t.Errorf("volume %q is not a PVC; only the world and backup PVCs are permitted", v.Name)
|
||||
continue
|
||||
}
|
||||
switch v.PersistentVolumeClaim.ClaimName {
|
||||
case "world-survival-0":
|
||||
world = v
|
||||
case "felis-backups":
|
||||
backup = v
|
||||
default:
|
||||
t.Errorf("unexpected PVC %q mounted", v.PersistentVolumeClaim.ClaimName)
|
||||
}
|
||||
}
|
||||
if world == nil {
|
||||
t.Fatal("world PVC not mounted")
|
||||
}
|
||||
if backup == nil {
|
||||
t.Fatal("backup PVC not mounted")
|
||||
}
|
||||
// The backup PVC must be read-only at the volume source: a restore must not
|
||||
// be able to mutate the archive store.
|
||||
if !backup.PersistentVolumeClaim.ReadOnly {
|
||||
t.Error("backup PVC volume source must be ReadOnly")
|
||||
}
|
||||
|
||||
// And the container's mounts must agree: backup read-only, world writable.
|
||||
c := singleContainer(t, job)
|
||||
var backupMount, worldMount *corev1.VolumeMount
|
||||
for i := range c.VolumeMounts {
|
||||
m := &c.VolumeMounts[i]
|
||||
switch m.Name {
|
||||
case backupVolume:
|
||||
backupMount = m
|
||||
case worldVolume:
|
||||
worldMount = m
|
||||
}
|
||||
}
|
||||
if backupMount == nil || !backupMount.ReadOnly {
|
||||
t.Error("backup mount must be ReadOnly")
|
||||
}
|
||||
if worldMount == nil || worldMount.ReadOnly {
|
||||
t.Error("world mount must be writable (the archive extracts into it)")
|
||||
}
|
||||
if backupMount != nil && backupMount.MountPath != "/backups" {
|
||||
t.Errorf("backup mount path = %q, want /backups (absolute refs resolve here)", backupMount.MountPath)
|
||||
}
|
||||
|
||||
// Volumes are only half the red line: a single Env var (e.g. a DATABASE_URL)
|
||||
// or an EnvFrom pulling a whole Secret/ConfigMap into the environment would
|
||||
// hand the restore Pod a credential without ever mounting one. The container
|
||||
// gets ALL of its input from the command flags, so both must be empty.
|
||||
if len(c.Env) != 0 {
|
||||
t.Errorf("restore container must carry no env vars, got %+v", c.Env)
|
||||
}
|
||||
if len(c.EnvFrom) != 0 {
|
||||
t.Errorf("restore container must carry no envFrom sources (no Secret/ConfigMap injection), got %+v", c.EnvFrom)
|
||||
}
|
||||
}
|
||||
|
||||
// A poisoned archive must terminate and not loop or run unbounded; the finished
|
||||
// Job must self-GC.
|
||||
func TestRestoreJobIsBoundedOneShotAndSelfCleaning(t *testing.T) {
|
||||
job, err := RestoreJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("RestoreJob: %v", err)
|
||||
}
|
||||
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
|
||||
t.Error("BackoffLimit must be 0 — a bad archive must not retry")
|
||||
}
|
||||
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds != 1800 {
|
||||
t.Errorf("ActiveDeadlineSeconds must be 1800, got %v", job.Spec.ActiveDeadlineSeconds)
|
||||
}
|
||||
if job.Spec.TTLSecondsAfterFinished == nil || *job.Spec.TTLSecondsAfterFinished != 600 {
|
||||
t.Errorf("TTLSecondsAfterFinished must be 600, got %v", job.Spec.TTLSecondsAfterFinished)
|
||||
}
|
||||
if job.Spec.Template.Spec.RestartPolicy != corev1.RestartPolicyNever {
|
||||
t.Error("RestartPolicy must be Never")
|
||||
}
|
||||
}
|
||||
|
||||
// The container must be non-root, non-privileged, escalation-proof, read-only
|
||||
// root, drop ALL caps, and carry resource limits.
|
||||
func TestRestoreJobContainerIsHardened(t *testing.T) {
|
||||
job, err := RestoreJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("RestoreJob: %v", err)
|
||||
}
|
||||
pod := job.Spec.Template.Spec
|
||||
if pod.SecurityContext == nil || pod.SecurityContext.RunAsNonRoot == nil || !*pod.SecurityContext.RunAsNonRoot {
|
||||
t.Error("pod must set runAsNonRoot=true")
|
||||
}
|
||||
if pod.SecurityContext == nil || pod.SecurityContext.FSGroup == nil || *pod.SecurityContext.FSGroup != 1000 {
|
||||
t.Error("pod must set an fsGroup so restored files are group-owned by the server identity")
|
||||
}
|
||||
c := singleContainer(t, job)
|
||||
sc := c.SecurityContext
|
||||
if sc == nil {
|
||||
t.Fatal("container has no security context")
|
||||
}
|
||||
if sc.Privileged == nil || *sc.Privileged {
|
||||
t.Error("container must not be privileged")
|
||||
}
|
||||
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
|
||||
t.Error("container must set allowPrivilegeEscalation=false")
|
||||
}
|
||||
if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
|
||||
t.Error("container must set readOnlyRootFilesystem=true (writes go only to the world PVC)")
|
||||
}
|
||||
if sc.Capabilities == nil || len(sc.Capabilities.Drop) == 0 || string(sc.Capabilities.Drop[0]) != "ALL" {
|
||||
t.Errorf("container must drop ALL capabilities, got %v", sc.Capabilities)
|
||||
}
|
||||
if c.Resources.Limits.Cpu().IsZero() || c.Resources.Limits.Memory().IsZero() {
|
||||
t.Error("container must carry CPU+memory limits")
|
||||
}
|
||||
}
|
||||
|
||||
// The container must invoke `felis restore` with the archive parameters as
|
||||
// plain flags — and crucially the world PVC name the operator/reaper agree on.
|
||||
func TestRestoreJobInvokesFelisRestoreWithParams(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
job, err := RestoreJob(p)
|
||||
if err != nil {
|
||||
t.Fatalf("RestoreJob: %v", err)
|
||||
}
|
||||
c := singleContainer(t, job)
|
||||
if len(c.Command) < 2 || c.Command[0] != "felis" || c.Command[1] != "restore" {
|
||||
t.Errorf("command = %v, want [felis restore ...]", c.Command)
|
||||
}
|
||||
if !argPairPresent(c.Args, "--server", p.Server) {
|
||||
t.Errorf("args must carry --server %q, got %v", p.Server, c.Args)
|
||||
}
|
||||
if !argPairPresent(c.Args, "--ref", p.BackupRef) {
|
||||
t.Errorf("args must carry --ref %q, got %v", p.BackupRef, c.Args)
|
||||
}
|
||||
if !argPairPresent(c.Args, "--archive-store", p.ArchiveStore) {
|
||||
t.Errorf("args must carry --archive-store %q, got %v", p.ArchiveStore, c.Args)
|
||||
}
|
||||
if !argPairPresent(c.Args, "--backup-root", p.BackupRoot) {
|
||||
t.Errorf("args must carry --backup-root %q, got %v", p.BackupRoot, c.Args)
|
||||
}
|
||||
if !argPairPresent(c.Args, "--worlds-root", p.WorldsRoot) {
|
||||
t.Errorf("args must carry --worlds-root %q, got %v", p.WorldsRoot, c.Args)
|
||||
}
|
||||
if job.Name != "restore-survival" {
|
||||
t.Errorf("job name = %q, want restore-survival (deterministic for idempotency)", job.Name)
|
||||
}
|
||||
}
|
||||
|
||||
// An empty image must be rejected rather than render an unrunnable Job; this is
|
||||
// what lets cmd/felis fall back to a 503 instead of enqueuing junk.
|
||||
func TestRestoreJobRequiresImage(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
p.Image = ""
|
||||
if _, err := RestoreJob(p); err == nil {
|
||||
t.Error("expected error for an empty image")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRestoreJobRejectsBadResourceLimit(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
p.MemLimit = "not-a-quantity"
|
||||
if _, err := RestoreJob(p); err == nil {
|
||||
t.Error("expected error for an unparseable memory limit")
|
||||
}
|
||||
}
|
||||
|
||||
// The restore SA must be bare: no secrets, token automount disabled.
|
||||
func TestRestoreServiceAccountIsBare(t *testing.T) {
|
||||
sa := RestoreServiceAccount(defaultNamespace, defaultServiceAccount)
|
||||
if sa.AutomountServiceAccountToken == nil || *sa.AutomountServiceAccountToken {
|
||||
t.Error("SA must disable token automounting")
|
||||
}
|
||||
if len(sa.Secrets) != 0 {
|
||||
t.Errorf("SA must carry no secrets, got %d", len(sa.Secrets))
|
||||
}
|
||||
if len(sa.ImagePullSecrets) != 0 {
|
||||
t.Errorf("SA must carry no image-pull secrets, got %d", len(sa.ImagePullSecrets))
|
||||
}
|
||||
}
|
||||
|
||||
// ---- helpers ----
|
||||
|
||||
func singleContainer(t *testing.T, job *batchv1.Job) corev1.Container {
|
||||
t.Helper()
|
||||
cs := job.Spec.Template.Spec.Containers
|
||||
if len(cs) != 1 {
|
||||
t.Fatalf("expected exactly one restore container, got %d", len(cs))
|
||||
}
|
||||
return cs[0]
|
||||
}
|
||||
|
||||
func argPairPresent(args []string, flag, val string) bool {
|
||||
for i := 0; i < len(args)-1; i++ {
|
||||
if args[i] == flag && args[i+1] == val {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
package restore
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// K8sJobs is the production Jobs backed by a controller-runtime client (spec
|
||||
// §7, §16). It creates the world-restore Job — nothing more: the restore Job is
|
||||
// one-shot and self-cleaning (ttlSecondsAfterFinished), so there is no phase or
|
||||
// cancel seam, and thus no config to hold (unlike build.K8sJobs, which needs the
|
||||
// namespace to read and delete its Job). Every restore parameter arrives in the
|
||||
// JobParams the Restorer builds from its own (defaulted) Config. The
|
||||
// cluster-bootstrap objects (the weak felis-restore SA) are installed once by
|
||||
// the deployment manifests (spec §21), not per restore, so this binding never
|
||||
// creates them. It is integration-tested against a live cluster, not the
|
||||
// hermetic restore_test.go suite.
|
||||
type K8sJobs struct {
|
||||
c client.Client
|
||||
}
|
||||
|
||||
// NewK8sJobs builds a Jobs over c. The restore Job's parameters all travel in
|
||||
// JobParams, so there is no Config to retain here.
|
||||
func NewK8sJobs(c client.Client) *K8sJobs {
|
||||
return &K8sJobs{c: c}
|
||||
}
|
||||
|
||||
// CreateRestoreJob renders and applies the restore Job. Its name is a
|
||||
// deterministic function of the server (RestoreJobName), so a concurrent restore
|
||||
// of the same server collides on Create; that collision is mapped to
|
||||
// ErrAlreadyExists, which the Restorer treats as success (idempotent enqueue).
|
||||
func (k *K8sJobs) CreateRestoreJob(ctx context.Context, p JobParams) error {
|
||||
job, err := RestoreJob(p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := k.c.Create(ctx, job); err != nil {
|
||||
if apierrors.IsAlreadyExists(err) {
|
||||
return ErrAlreadyExists
|
||||
}
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,222 @@
|
||||
// Package restore implements the world-restore executor (spec §7
|
||||
// POST /servers/{name}/restore-backup, spec §466: a former owner who re-claims a
|
||||
// released server within the retention window restores their archived world).
|
||||
//
|
||||
// felis-api cannot restore a world in-process: the world PVC is RWO and owned by
|
||||
// the operator's StatefulSet, so the API has nothing to mount at request time
|
||||
// (see internal/api.Restorer). This package is the production executor it hands
|
||||
// off to — a one-shot Kubernetes Job in the minecraft namespace that mounts the
|
||||
// target world PVC and the backup store, then runs `felis restore` (cmd/felis)
|
||||
// to extract the archive into the world volume.
|
||||
//
|
||||
// Trust model, mirroring internal/build's weak-SA isolation (spec §16, §21, §22):
|
||||
// the restore Pod runs under a deliberately weak service account with its token
|
||||
// auto-mount disabled, so it cannot reach the K8s API; it is handed ONLY the two
|
||||
// PVCs and the archive parameters as plain flags, never a database URL or any
|
||||
// Secret — the four-power red line that a build/restore Pod must not touch the
|
||||
// felis database or the K8s API. felis-api owns the database and the
|
||||
// authorization decision (handlers_backups.go); this Pod only moves bytes from
|
||||
// the backup PVC onto the world PVC. Every isolation guarantee lives in the pure
|
||||
// jobspec (jobspec.go) and is asserted by jobspec_test.go, because no cluster
|
||||
// runs in this environment.
|
||||
//
|
||||
// The Restorer depends on the Jobs interface, so the orchestration (idempotent
|
||||
// enqueue, error mapping) is unit-tested against an in-memory fake; the
|
||||
// controller-runtime implementation (k8sjobs.go) compiles here but is exercised
|
||||
// only by integration tests against a live cluster.
|
||||
package restore
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
|
||||
// ErrAlreadyExists is returned by a Jobs implementation when a restore Job for a
|
||||
// server already exists (a restore is already in flight). The Restorer treats it
|
||||
// as success — see Restore.
|
||||
var ErrAlreadyExists = errors.New("restore: job already exists")
|
||||
|
||||
// Jobs is the cluster-side restore lifecycle the Restorer depends on. It is an
|
||||
// interface so the orchestration is tested against a fake; the controller-runtime
|
||||
// implementation (K8sJobs) is integration-tested only — it requires a live
|
||||
// cluster. The restore Job is one-shot and self-cleaning (TTL), so unlike the
|
||||
// build subsystem there is no phase-polling or cancel seam: kicking it off is the
|
||||
// whole contract, exactly matching the asynchronous 202 the handler answers.
|
||||
type Jobs interface {
|
||||
// CreateRestoreJob renders and applies the restore Job for p. It returns
|
||||
// ErrAlreadyExists if a Job of the same (deterministic) name already exists.
|
||||
CreateRestoreJob(ctx context.Context, p JobParams) error
|
||||
}
|
||||
|
||||
// Config parameterises the restore executor. Deployment-specific values that
|
||||
// have no safe default — the felis Image to run and the BackupPVC to mount — are
|
||||
// supplied by the caller (cmd/felis sources them from the environment); when
|
||||
// either is empty the caller leaves the API's Restorer nil so the endpoint
|
||||
// reports 503 rather than enqueuing a Job that cannot run.
|
||||
type Config struct {
|
||||
// Namespace is where the world PVCs live and the restore Job runs (the
|
||||
// minecraft namespace). The Job is intentionally co-located with the world it
|
||||
// restores; it never runs in the felis control-plane namespace.
|
||||
Namespace string
|
||||
// ServiceAccount is the weak SA the restore Pod runs as. Like felis-build it
|
||||
// MUST NOT be the felis-api SA and has no Role/RoleBinding anywhere.
|
||||
ServiceAccount string
|
||||
// Image is the felis binary image; the Job runs `felis restore` from it.
|
||||
Image string
|
||||
// ArchiveStore selects the backup backend. Only "tarLocal" is implemented in
|
||||
// this build, mirroring the reaper (cmd/felis buildArchiver).
|
||||
ArchiveStore string
|
||||
// BackupPVC is the name of the backup PVC the archives live on. It must be
|
||||
// RWX so the reaper and concurrent restores can mount it (a helm-slice
|
||||
// contract); restore mounts it read-only.
|
||||
BackupPVC string
|
||||
// BackupRoot is the in-Pod mount path of BackupPVC. It MUST equal the path
|
||||
// the reaper wrote archives under (cfg.Archive.LocalPath), because tarLocal
|
||||
// archive refs are absolute paths — mounting the PVC anywhere else would make
|
||||
// the stored ref unresolvable inside the Pod.
|
||||
BackupRoot string
|
||||
// WorldsRoot is the in-Pod mount path of the world PVC the archive extracts
|
||||
// into.
|
||||
WorldsRoot string
|
||||
// Deadline caps the restore Pod's wall-clock (activeDeadlineSeconds).
|
||||
Deadline time.Duration
|
||||
// CPULimit / MemLimit cap the restore container.
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
// RunAsUser / RunAsGroup / FSGroup are the Pod's runtime identity. FSGroup in
|
||||
// particular MUST match the operator StatefulSet's runtime group so the files
|
||||
// the restore Pod writes are readable by the minecraft server that later
|
||||
// mounts the same world PVC. The default matches the conventional minecraft
|
||||
// container uid; a deployment that runs minecraft as another id overrides it.
|
||||
RunAsUser int64
|
||||
RunAsGroup int64
|
||||
FSGroup int64
|
||||
// TTLAfterFinished is how long a finished restore Job lingers before the Job
|
||||
// controller garbage-collects it. There is no cancel path, so the TTL is the
|
||||
// only cleanup; it also bounds the window in which a re-restore sees a stale
|
||||
// completed Job as ErrAlreadyExists.
|
||||
TTLAfterFinished time.Duration
|
||||
}
|
||||
|
||||
// defaults applied when a Config field is left zero. Image and BackupPVC have no
|
||||
// default on purpose — see Config.
|
||||
const (
|
||||
defaultNamespace = "minecraft"
|
||||
defaultServiceAccount = "felis-restore"
|
||||
defaultArchiveStore = "tarLocal"
|
||||
defaultBackupRoot = "/backups"
|
||||
defaultWorldsRoot = "/world"
|
||||
defaultDeadline = 30 * time.Minute
|
||||
defaultCPULimit = "1"
|
||||
defaultMemLimit = "1Gi"
|
||||
defaultRunAsID = int64(1000)
|
||||
defaultTTL = 10 * time.Minute
|
||||
)
|
||||
|
||||
// withDefaults returns a copy of c with zero fields filled, so a partially
|
||||
// configured Config (or the zero value, in tests) is always usable.
|
||||
func (c Config) withDefaults() Config {
|
||||
if c.Namespace == "" {
|
||||
c.Namespace = defaultNamespace
|
||||
}
|
||||
if c.ServiceAccount == "" {
|
||||
c.ServiceAccount = defaultServiceAccount
|
||||
}
|
||||
if c.ArchiveStore == "" {
|
||||
c.ArchiveStore = defaultArchiveStore
|
||||
}
|
||||
if c.BackupRoot == "" {
|
||||
c.BackupRoot = defaultBackupRoot
|
||||
}
|
||||
if c.WorldsRoot == "" {
|
||||
c.WorldsRoot = defaultWorldsRoot
|
||||
}
|
||||
if c.Deadline <= 0 {
|
||||
c.Deadline = defaultDeadline
|
||||
}
|
||||
if c.CPULimit == "" {
|
||||
c.CPULimit = defaultCPULimit
|
||||
}
|
||||
if c.MemLimit == "" {
|
||||
c.MemLimit = defaultMemLimit
|
||||
}
|
||||
if c.RunAsUser == 0 {
|
||||
c.RunAsUser = defaultRunAsID
|
||||
}
|
||||
if c.RunAsGroup == 0 {
|
||||
c.RunAsGroup = defaultRunAsID
|
||||
}
|
||||
if c.FSGroup == 0 {
|
||||
c.FSGroup = defaultRunAsID
|
||||
}
|
||||
if c.TTLAfterFinished <= 0 {
|
||||
c.TTLAfterFinished = defaultTTL
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// Restorer is the production internal/api.Restorer (the compile-time proof of
|
||||
// that is in internal/api's test, which imports this package; this package never
|
||||
// imports api). It holds no mutable state.
|
||||
type Restorer struct {
|
||||
Jobs Jobs
|
||||
Config Config
|
||||
}
|
||||
|
||||
// Restore enqueues a restore Job that extracts the archive at backupRef into
|
||||
// serverName's world PVC. It returns once the Job is created — the extraction
|
||||
// runs in the Pod — so the handler's 202 ("restoring") is honest.
|
||||
//
|
||||
// It is idempotent: if a restore Job for this server already exists (a restore
|
||||
// is already in flight, or a just-finished one has not yet hit its TTL), the
|
||||
// duplicate enqueue is treated as success rather than surfaced as an error.
|
||||
//
|
||||
// The coalescing key is the Job name (RestoreJobName), which depends only on the
|
||||
// server, NOT on backupRef — so a second request that arrives while one is in
|
||||
// flight is absorbed regardless of the ref it carries, and if the two refs
|
||||
// differ the second is silently dropped (the in-flight restore wins). That is
|
||||
// acceptable here: restore runs only for a Stopped server (handler gate ⑥) and
|
||||
// the handler always passes the latest backup, which for a stopped server does
|
||||
// not change, so concurrent requests carry the same ref in practice. A caller
|
||||
// that genuinely needs a different archive can re-request after the Job clears
|
||||
// its TTL. This keeps the handler's 202 honest without it having to map "already
|
||||
// in progress" onto a 500.
|
||||
func (r *Restorer) Restore(ctx context.Context, serverName, backupRef string) error {
|
||||
if err := r.Jobs.CreateRestoreJob(ctx, r.jobParams(serverName, backupRef)); err != nil {
|
||||
if errors.Is(err, ErrAlreadyExists) {
|
||||
return nil // already enqueued — idempotent
|
||||
}
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// jobParams projects the server, archive ref, and config onto the inputs
|
||||
// jobspec.go renders. The world PVC name is derived from the single shared
|
||||
// naming convention (naming.WorldPVCName), the same one the operator created it
|
||||
// under and the reaper deletes it by.
|
||||
func (r *Restorer) jobParams(serverName, backupRef string) JobParams {
|
||||
cfg := r.Config.withDefaults()
|
||||
return JobParams{
|
||||
Server: serverName,
|
||||
WorldPVC: naming.WorldPVCName(serverName),
|
||||
BackupPVC: cfg.BackupPVC,
|
||||
BackupRef: backupRef,
|
||||
ArchiveStore: cfg.ArchiveStore,
|
||||
Namespace: cfg.Namespace,
|
||||
ServiceAccount: cfg.ServiceAccount,
|
||||
Image: cfg.Image,
|
||||
BackupRoot: cfg.BackupRoot,
|
||||
WorldsRoot: cfg.WorldsRoot,
|
||||
Deadline: cfg.Deadline,
|
||||
CPULimit: cfg.CPULimit,
|
||||
MemLimit: cfg.MemLimit,
|
||||
RunAsUser: cfg.RunAsUser,
|
||||
RunAsGroup: cfg.RunAsGroup,
|
||||
FSGroup: cfg.FSGroup,
|
||||
TTLAfterFinished: cfg.TTLAfterFinished,
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
package restore_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/restore"
|
||||
)
|
||||
|
||||
// fakeJobs is an in-memory Jobs that records the params it was handed and
|
||||
// returns a programmable error, so the orchestration is tested without a
|
||||
// cluster.
|
||||
type fakeJobs struct {
|
||||
calls []restore.JobParams
|
||||
err error
|
||||
}
|
||||
|
||||
func (f *fakeJobs) CreateRestoreJob(_ context.Context, p restore.JobParams) error {
|
||||
f.calls = append(f.calls, p)
|
||||
return f.err
|
||||
}
|
||||
|
||||
func TestRestoreEnqueuesJobWithDerivedParams(t *testing.T) {
|
||||
jobs := &fakeJobs{}
|
||||
r := &restore.Restorer{
|
||||
Jobs: jobs,
|
||||
Config: restore.Config{
|
||||
Image: "registry.internal/felis:test",
|
||||
BackupPVC: "felis-backups",
|
||||
},
|
||||
}
|
||||
|
||||
if err := r.Restore(context.Background(), "survival", "/backups/survival/2026.tar.gz"); err != nil {
|
||||
t.Fatalf("Restore: %v", err)
|
||||
}
|
||||
if len(jobs.calls) != 1 {
|
||||
t.Fatalf("CreateRestoreJob called %d times, want 1", len(jobs.calls))
|
||||
}
|
||||
got := jobs.calls[0]
|
||||
if got.Server != "survival" {
|
||||
t.Errorf("Server = %q, want survival", got.Server)
|
||||
}
|
||||
// The world PVC must come from the shared naming convention, not be invented
|
||||
// here: it is the same name the operator created and the reaper deletes.
|
||||
if got.WorldPVC != "world-survival-0" {
|
||||
t.Errorf("WorldPVC = %q, want world-survival-0", got.WorldPVC)
|
||||
}
|
||||
if got.BackupRef != "/backups/survival/2026.tar.gz" {
|
||||
t.Errorf("BackupRef = %q, want the passed ref", got.BackupRef)
|
||||
}
|
||||
if got.BackupPVC != "felis-backups" {
|
||||
t.Errorf("BackupPVC = %q, want felis-backups", got.BackupPVC)
|
||||
}
|
||||
if got.Image != "registry.internal/felis:test" {
|
||||
t.Errorf("Image = %q, want the configured image", got.Image)
|
||||
}
|
||||
// withDefaults must have filled the unset fields.
|
||||
if got.Namespace == "" || got.ServiceAccount == "" || got.WorldsRoot == "" || got.BackupRoot == "" {
|
||||
t.Errorf("defaults not applied: %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRestoreIsIdempotentOnAlreadyExists(t *testing.T) {
|
||||
jobs := &fakeJobs{err: restore.ErrAlreadyExists}
|
||||
r := &restore.Restorer{Jobs: jobs, Config: restore.Config{Image: "img", BackupPVC: "pvc"}}
|
||||
|
||||
// A restore already in flight is success, not an error: the handler must be
|
||||
// able to answer 202 for a coalesced duplicate request.
|
||||
if err := r.Restore(context.Background(), "survival", "ref"); err != nil {
|
||||
t.Fatalf("Restore on AlreadyExists = %v, want nil (idempotent)", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRestorePropagatesGenericError(t *testing.T) {
|
||||
sentinel := errors.New("apiserver exploded")
|
||||
jobs := &fakeJobs{err: sentinel}
|
||||
r := &restore.Restorer{Jobs: jobs, Config: restore.Config{Image: "img", BackupPVC: "pvc"}}
|
||||
|
||||
err := r.Restore(context.Background(), "survival", "ref")
|
||||
if !errors.Is(err, sentinel) {
|
||||
t.Fatalf("Restore error = %v, want the underlying error propagated", err)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user