Files
Felis/internal/restore/jobspec_test.go
2026-06-29 20:17:17 +08:00

284 lines
9.9 KiB
Go

package restore
import (
"testing"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
)
func sampleJobParams() JobParams {
return JobParams{
Server: "survival",
WorldPVC: "world-survival-0",
BackupPVC: "felis-backups",
BackupRef: "/backups/survival/2026-06-25.tar.gz",
ArchiveStore: "tarLocal",
Namespace: defaultNamespace,
ServiceAccount: defaultServiceAccount,
Image: "registry.felis.svc:5000/felis:1.0",
BackupRoot: "/backups",
WorldsRoot: "/world",
Deadline: 30 * time.Minute,
CPULimit: "1",
MemLimit: "1Gi",
RunAsUser: 1000,
RunAsGroup: 1000,
FSGroup: 1000,
TTLAfterFinished: 10 * time.Minute,
}
}
// The restore Pod must run under the weak felis-restore SA — never the
// felis-api identity — with its token un-mounted, so it cannot reach the K8s
// API. This is the §16/§22 red line asserted on the rendered spec because no
// cluster runs here.
func TestRestoreJobRunsUnderWeakSA(t *testing.T) {
job, err := RestoreJob(sampleJobParams())
if err != nil {
t.Fatalf("RestoreJob: %v", err)
}
sa := job.Spec.Template.Spec.ServiceAccountName
if sa != defaultServiceAccount {
t.Errorf("service account = %q, want %q", sa, defaultServiceAccount)
}
if sa == "felis-api" {
t.Fatal("restore Pod must NOT run as the felis-api SA")
}
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
t.Error("AutomountServiceAccountToken must be explicitly false")
}
}
// The four-power red line: a restore Pod handles a (potentially poisoned)
// archive, so it must mount EXACTLY the two PVCs — world read-write, backup
// read-only — and NO Secret or ConfigMap, so it can never reach the felis
// database or any credential.
func TestRestoreJobMountsOnlyTheTwoPVCsAndNoSecrets(t *testing.T) {
job, err := RestoreJob(sampleJobParams())
if err != nil {
t.Fatalf("RestoreJob: %v", err)
}
vols := job.Spec.Template.Spec.Volumes
if len(vols) != 2 {
t.Fatalf("expected exactly 2 volumes (world + backup), got %d: %+v", len(vols), vols)
}
var world, backup *corev1.Volume
for i := range vols {
v := &vols[i]
// The forbidden volume kinds: anything that could carry DB creds or
// reach the API.
if v.Secret != nil {
t.Errorf("volume %q is a Secret — a restore Pod must never mount a Secret", v.Name)
}
if v.ConfigMap != nil {
t.Errorf("volume %q is a ConfigMap — no config/credential injection allowed", v.Name)
}
if v.Projected != nil || v.DownwardAPI != nil {
t.Errorf("volume %q is a projected/downward volume — could surface the SA token", v.Name)
}
if v.HostPath != nil {
t.Errorf("volume %q is a hostPath — no node filesystem access allowed", v.Name)
}
if v.PersistentVolumeClaim == nil {
t.Errorf("volume %q is not a PVC; only the world and backup PVCs are permitted", v.Name)
continue
}
switch v.PersistentVolumeClaim.ClaimName {
case "world-survival-0":
world = v
case "felis-backups":
backup = v
default:
t.Errorf("unexpected PVC %q mounted", v.PersistentVolumeClaim.ClaimName)
}
}
if world == nil {
t.Fatal("world PVC not mounted")
}
if backup == nil {
t.Fatal("backup PVC not mounted")
}
// The backup PVC must be read-only at the volume source: a restore must not
// be able to mutate the archive store.
if !backup.PersistentVolumeClaim.ReadOnly {
t.Error("backup PVC volume source must be ReadOnly")
}
// And the container's mounts must agree: backup read-only, world writable.
c := singleContainer(t, job)
var backupMount, worldMount *corev1.VolumeMount
for i := range c.VolumeMounts {
m := &c.VolumeMounts[i]
switch m.Name {
case backupVolume:
backupMount = m
case worldVolume:
worldMount = m
}
}
if backupMount == nil || !backupMount.ReadOnly {
t.Error("backup mount must be ReadOnly")
}
if worldMount == nil || worldMount.ReadOnly {
t.Error("world mount must be writable (the archive extracts into it)")
}
if backupMount != nil && backupMount.MountPath != "/backups" {
t.Errorf("backup mount path = %q, want /backups (absolute refs resolve here)", backupMount.MountPath)
}
// Volumes are only half the red line: a single Env var (e.g. a DATABASE_URL)
// or an EnvFrom pulling a whole Secret/ConfigMap into the environment would
// hand the restore Pod a credential without ever mounting one. The container
// gets ALL of its input from the command flags, so both must be empty.
if len(c.Env) != 0 {
t.Errorf("restore container must carry no env vars, got %+v", c.Env)
}
if len(c.EnvFrom) != 0 {
t.Errorf("restore container must carry no envFrom sources (no Secret/ConfigMap injection), got %+v", c.EnvFrom)
}
}
// A poisoned archive must terminate and not loop or run unbounded; the finished
// Job must self-GC.
func TestRestoreJobIsBoundedOneShotAndSelfCleaning(t *testing.T) {
job, err := RestoreJob(sampleJobParams())
if err != nil {
t.Fatalf("RestoreJob: %v", err)
}
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
t.Error("BackoffLimit must be 0 — a bad archive must not retry")
}
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds != 1800 {
t.Errorf("ActiveDeadlineSeconds must be 1800, got %v", job.Spec.ActiveDeadlineSeconds)
}
if job.Spec.TTLSecondsAfterFinished == nil || *job.Spec.TTLSecondsAfterFinished != 600 {
t.Errorf("TTLSecondsAfterFinished must be 600, got %v", job.Spec.TTLSecondsAfterFinished)
}
if job.Spec.Template.Spec.RestartPolicy != corev1.RestartPolicyNever {
t.Error("RestartPolicy must be Never")
}
}
// The container must be non-root, non-privileged, escalation-proof, read-only
// root, drop ALL caps, and carry resource limits.
func TestRestoreJobContainerIsHardened(t *testing.T) {
job, err := RestoreJob(sampleJobParams())
if err != nil {
t.Fatalf("RestoreJob: %v", err)
}
pod := job.Spec.Template.Spec
if pod.SecurityContext == nil || pod.SecurityContext.RunAsNonRoot == nil || !*pod.SecurityContext.RunAsNonRoot {
t.Error("pod must set runAsNonRoot=true")
}
if pod.SecurityContext == nil || pod.SecurityContext.FSGroup == nil || *pod.SecurityContext.FSGroup != 1000 {
t.Error("pod must set an fsGroup so restored files are group-owned by the server identity")
}
c := singleContainer(t, job)
sc := c.SecurityContext
if sc == nil {
t.Fatal("container has no security context")
}
if sc.Privileged == nil || *sc.Privileged {
t.Error("container must not be privileged")
}
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
t.Error("container must set allowPrivilegeEscalation=false")
}
if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
t.Error("container must set readOnlyRootFilesystem=true (writes go only to the world PVC)")
}
if sc.Capabilities == nil || len(sc.Capabilities.Drop) == 0 || string(sc.Capabilities.Drop[0]) != "ALL" {
t.Errorf("container must drop ALL capabilities, got %v", sc.Capabilities)
}
if c.Resources.Limits.Cpu().IsZero() || c.Resources.Limits.Memory().IsZero() {
t.Error("container must carry CPU+memory limits")
}
}
// The container must invoke the felis restore binary by absolute path with the
// archive parameters as plain flags — and crucially the world PVC name the
// operator/reaper agree on.
func TestRestoreJobInvokesFelisRestoreWithParams(t *testing.T) {
p := sampleJobParams()
job, err := RestoreJob(p)
if err != nil {
t.Fatalf("RestoreJob: %v", err)
}
c := singleContainer(t, job)
if len(c.Command) < 2 || c.Command[0] != felisBinaryPath || c.Command[1] != "restore" {
t.Errorf("command = %v, want [%s restore ...]", c.Command, felisBinaryPath)
}
if !argPairPresent(c.Args, "--server", p.Server) {
t.Errorf("args must carry --server %q, got %v", p.Server, c.Args)
}
if !argPairPresent(c.Args, "--ref", p.BackupRef) {
t.Errorf("args must carry --ref %q, got %v", p.BackupRef, c.Args)
}
if !argPairPresent(c.Args, "--archive-store", p.ArchiveStore) {
t.Errorf("args must carry --archive-store %q, got %v", p.ArchiveStore, c.Args)
}
if !argPairPresent(c.Args, "--backup-root", p.BackupRoot) {
t.Errorf("args must carry --backup-root %q, got %v", p.BackupRoot, c.Args)
}
if !argPairPresent(c.Args, "--worlds-root", p.WorldsRoot) {
t.Errorf("args must carry --worlds-root %q, got %v", p.WorldsRoot, c.Args)
}
if job.Name != "restore-survival" {
t.Errorf("job name = %q, want restore-survival (deterministic for idempotency)", job.Name)
}
}
// An empty image must be rejected rather than render an unrunnable Job; this is
// what lets cmd/felis fall back to a 503 instead of enqueuing junk.
func TestRestoreJobRequiresImage(t *testing.T) {
p := sampleJobParams()
p.Image = ""
if _, err := RestoreJob(p); err == nil {
t.Error("expected error for an empty image")
}
}
func TestRestoreJobRejectsBadResourceLimit(t *testing.T) {
p := sampleJobParams()
p.MemLimit = "not-a-quantity"
if _, err := RestoreJob(p); err == nil {
t.Error("expected error for an unparseable memory limit")
}
}
// The restore SA must be bare: no secrets, token automount disabled.
func TestRestoreServiceAccountIsBare(t *testing.T) {
sa := RestoreServiceAccount(defaultNamespace, defaultServiceAccount)
if sa.AutomountServiceAccountToken == nil || *sa.AutomountServiceAccountToken {
t.Error("SA must disable token automounting")
}
if len(sa.Secrets) != 0 {
t.Errorf("SA must carry no secrets, got %d", len(sa.Secrets))
}
if len(sa.ImagePullSecrets) != 0 {
t.Errorf("SA must carry no image-pull secrets, got %d", len(sa.ImagePullSecrets))
}
}
// ---- helpers ----
func singleContainer(t *testing.T, job *batchv1.Job) corev1.Container {
t.Helper()
cs := job.Spec.Template.Spec.Containers
if len(cs) != 1 {
t.Fatalf("expected exactly one restore container, got %d", len(cs))
}
return cs[0]
}
func argPairPresent(args []string, flag, val string) bool {
for i := 0; i < len(args)-1; i++ {
if args[i] == flag && args[i+1] == val {
return true
}
}
return false
}