feat(api): add on-demand world backup endpoint and Job executor (§B4 Sync)
Add POST /api/v1/servers/{name}/backup: an owner or admin snapshots a
stopped server's world into the archive store on demand, recorded as a
first-class world_backups row (reason `manual`) — restorable by the
existing restore path and expired by the reaper's retention pass, so it
never leaks as an orphan archive. This is the break-glass "Sync" op,
resolved as immediate/on-demand backup.
felis-api cannot archive in-process (the world PVC is RWO, held by the
operator StatefulSet), so the work hands off to a one-shot Kubernetes Job
(new internal/backupjob) that mounts the world PVC read-only and the
backup PVC read-write, plus the felis config Secret so it self-records
its row atomically like the reaper. The Pod mirrors restore's weak-SA
isolation (SA token un-mounted, non-root, read-only rootfs, drop ALL);
the one reviewed departure is that config-Secret mount, frozen by
jobspec_test.go. Handler answers 202 backing_up; gated on the server
being Stopped (RWO world PVC), owner-or-admin, and FELIS_IMAGE +
FELIS_BACKUP_PVC being wired (else 503 backup_unavailable).
Each request mints a unique Job name (backup-<server>-<rand>) so a repeat
on-demand backup produces a fresh archive rather than colliding with a
just-finished Job still inside its TTL window and silently no-op'ing the
retry.
This commit is contained in:
16 files changed
+1396
-1
No files matched your search
@@ -58,6 +58,11 @@ type API struct {
|
||||
// authorization paths do not need it (they read the Repo), only the kick-off.
|
||||
Restorer Restorer
|
||||
|
||||
// Backuper starts an on-demand world backup (spec §18/§19 WorldArchiver). Like
|
||||
// Restorer it is optional: when nil the backup route reports 503, so the backup
|
||||
// authorization boundary is exercised before the backup-Job executor is wired.
|
||||
Backuper Backuper
|
||||
|
||||
// Submissions is the user-modpack approval lane (a user-directed extension over
|
||||
// the §16 build subsystem; see internal/submit). It is optional: when
|
||||
// nil the /me/submissions and /submissions routes report 503 rather than 404, so
|
||||
@@ -346,6 +351,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// neither sits behind adminOnly.
|
||||
{Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/backup", h: a.handleBackupNow},
|
||||
// Account linking (spec §10), web side: /start reports link status (it is the
|
||||
// pointer handleClaim's 412 emits), /verify consumes the in-game code and binds
|
||||
// the account. App-tier, not admin — linking your own account is an ordinary
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
package api
|
||||
|
||||
import "context"
|
||||
|
||||
// Backuper starts an on-demand world backup (spec §18/§19 WorldArchiver, run on
|
||||
// demand rather than on the reaper's daily schedule) — the "back up before I touch
|
||||
// it" lever behind POST /api/v1/servers/{name}/backup. Like Restorer it only STARTS
|
||||
// the work: tarring the world PVC into the archive store is pod-filesystem work
|
||||
// felis-api cannot do in-process — the world PVC is RWO and owned by the operator's
|
||||
// StatefulSet, so the API has nothing to mount at request time. The production
|
||||
// implementation therefore hands off to a backup Job (internal/backupjob), which
|
||||
// self-records the world_backups row like the reaper. The call returns once the
|
||||
// backup is enqueued, so the handler answers 202 (backing_up).
|
||||
//
|
||||
// formerOwner is the current owner recorded on the backup row so it can later be
|
||||
// restored (empty when an admin backs up an unowned server). It returns ErrNotFound
|
||||
// if the server is unknown to the execution backend; any other error maps to 500.
|
||||
//
|
||||
// It is an interface so the handler is tested against a fake; the production executor
|
||||
// (internal/backupjob.Backuper) is integration-only, and until it is wired the
|
||||
// API.Backuper is nil so POST /servers/{name}/backup reports 503 — the backup
|
||||
// authorization boundary is exercised without shipping a stub that cannot run.
|
||||
type Backuper interface {
|
||||
Backup(ctx context.Context, serverName, formerOwner string) error
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"felis.lolicon.best/internal/backupjob"
|
||||
)
|
||||
|
||||
// Compile-time proof that the production backup executor satisfies the API's
|
||||
// Backuper interface, mirroring restore_wire_test. The dependency points one way
|
||||
// only: api defines the narrow Backuper port and never imports backupjob in
|
||||
// production (handlers_backups.go depends on the interface); this test is the single
|
||||
// place the concrete type and the port are pinned together.
|
||||
var _ Backuper = (*backupjob.Backuper)(nil)
|
||||
@@ -0,0 +1,178 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
)
|
||||
|
||||
// fakeBackuper records the on-demand backup kick-off the handler makes. It mirrors
|
||||
// fakeRestorer: the handler only STARTS the work, so the fake just captures the
|
||||
// arguments and returns a canned error.
|
||||
type fakeBackuper struct {
|
||||
err error
|
||||
calls int
|
||||
gotName string
|
||||
gotFormerOwn string
|
||||
}
|
||||
|
||||
func (f *fakeBackuper) Backup(_ context.Context, name, formerOwner string) error {
|
||||
f.calls++
|
||||
f.gotName, f.gotFormerOwn = name, formerOwner
|
||||
return f.err
|
||||
}
|
||||
|
||||
// TestBackupNow exercises POST /api/v1/servers/{name}/backup (spec §18/§19 WorldArchiver,
|
||||
// on demand). The default server is stopped and owned by owner1 with a wired
|
||||
// fakeBackuper. The focus is the owner-or-admin gate, the stopped gate (the RWO world
|
||||
// PVC must be free), the 404/503 surface, that the recorded former owner is the
|
||||
// server's CURRENT owner, and that backup is an async 202 kick-off.
|
||||
func TestBackupNow(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
|
||||
mk := func() (*API, *fakeRepo, *fakeCluster, *fakeBackuper) {
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
cl := newFakeCluster()
|
||||
cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped",
|
||||
Ready: false, DesiredState: string(v1alpha1.DesiredStopped)}
|
||||
backuper := &fakeBackuper{}
|
||||
api := newTestAPI(repo, cl)
|
||||
api.Backuper = backuper
|
||||
return api, repo, cl, backuper
|
||||
}
|
||||
|
||||
const path = "/api/v1/servers/survival/backup"
|
||||
|
||||
t.Run("owner backs up -> 202 + backuper(currentOwner) + audit", func(t *testing.T) {
|
||||
api, repo, _, backuper := mk()
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
var resp struct {
|
||||
Name string `json:"name"`
|
||||
Status string `json:"status"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
|
||||
t.Fatalf("body not JSON: %v (%s)", err, w.Body.String())
|
||||
}
|
||||
if resp.Name != "survival" || resp.Status != "backing_up" {
|
||||
t.Fatalf("unexpected response %+v", resp)
|
||||
}
|
||||
if backuper.calls != 1 || backuper.gotName != "survival" || backuper.gotFormerOwn != "owner1" {
|
||||
t.Fatalf("backuper saw (calls=%d,%q,%q), want (1,survival,owner1)",
|
||||
backuper.calls, backuper.gotName, backuper.gotFormerOwn)
|
||||
}
|
||||
if len(repo.audits) != 1 || repo.audits[0].Action != "backup.create" || repo.audits[0].Actor != "[email protected]" {
|
||||
t.Fatalf("audit not written as expected: %+v", repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("admin backs up an unowned server -> 202, empty former owner recorded", func(t *testing.T) {
|
||||
api, repo, _, backuper := mk()
|
||||
repo.byName["survival"].OwnerID = "" // released world
|
||||
api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "[email protected]",
|
||||
Role: "admin", ViaAdminAccess: true}}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if backuper.calls != 1 || backuper.gotFormerOwn != "" {
|
||||
t.Fatalf("backuper saw (calls=%d, former=%q), want (1, empty)", backuper.calls, backuper.gotFormerOwn)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("non-owner -> 403, no backup", func(t *testing.T) {
|
||||
api, _, _, backuper := mk()
|
||||
api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("code = %d, want 403", w.Code)
|
||||
}
|
||||
if backuper.calls != 0 {
|
||||
t.Fatal("a forbidden caller must not start a backup")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unknown server -> 404", func(t *testing.T) {
|
||||
api, _, _, _ := mk()
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/missing/backup", "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("running server -> 409 not_stopped, no backup", func(t *testing.T) {
|
||||
api, _, cl, backuper := mk()
|
||||
cl.byName["survival"].Ready = true
|
||||
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if backuper.calls != 0 {
|
||||
t.Fatal("a running server holds the RWO world PVC — backup must be refused")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("starting server -> 409 not_stopped", func(t *testing.T) {
|
||||
api, _, cl, _ := mk()
|
||||
cl.byName["survival"].Ready = false
|
||||
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil Backuper -> 503 backup_unavailable", func(t *testing.T) {
|
||||
api, _, _, _ := mk()
|
||||
api.Backuper = nil
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "backup_unavailable" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("backuper failure -> 500, not audited", func(t *testing.T) {
|
||||
api, repo, _, backuper := mk()
|
||||
backuper.err = errors.New("kick-off failed")
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusInternalServerError {
|
||||
t.Fatalf("code = %d, want 500 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if len(repo.audits) != 0 {
|
||||
t.Fatalf("a failed backup must not be audited: %+v", repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("backuper reports unknown server -> 404", func(t *testing.T) {
|
||||
api, _, _, backuper := mk()
|
||||
backuper.err = ErrNotFound
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("invalid server name -> 400", func(t *testing.T) {
|
||||
api, _, _, _ := mk()
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/X/backup", "", nil)
|
||||
if w.Code != http.StatusBadRequest {
|
||||
t.Fatalf("code = %d, want 400", w.Code)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -176,3 +176,77 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
"backup_id": backup.ID,
|
||||
})
|
||||
}
|
||||
|
||||
// handleBackupNow starts an on-demand backup of a server's world (POST
|
||||
// /api/v1/servers/{name}/backup; spec §18/§19 WorldArchiver, run on demand). It is
|
||||
// the "back up before I touch it" lever the break-glass console and the owner both
|
||||
// reach for. Authorization mirrors handleRestoreBackup's front half — the shared
|
||||
// "who may act on this server's world" gate — but stops short of restore's backup
|
||||
// resolution and former-owner-match, because a backup is initiated by the CURRENT
|
||||
// owner and records their ownership; there is no prior owner's data to leak:
|
||||
//
|
||||
// ① name validation
|
||||
// ② ServerByName — an unknown server is 404
|
||||
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
|
||||
// released world can still be snapshotted by an operator before disposal)
|
||||
// ④ stopped gate: the world PVC is RWO and held by a running server, so a backup
|
||||
// Job cannot double-mount it — refuse unless the server is fully stopped. This
|
||||
// also guarantees a quiescent, non-torn archive.
|
||||
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
|
||||
// means "enqueued" and the handler answers 202.
|
||||
//
|
||||
// The former owner recorded on the backup is the server's current OwnerID (empty for
|
||||
// an unowned server backed up by an admin), so the resulting world_backups row is
|
||||
// restorable by that owner exactly like an inactivity backup.
|
||||
func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
name := r.PathValue("name")
|
||||
if err := naming.ValidateServerName(name); err != nil {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||
return
|
||||
}
|
||||
|
||||
rec, err := a.Repo.ServerByName(r.Context(), name)
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
if !a.isOwnerOrAdmin(p, rec) {
|
||||
writeError(w, r, errForbidden)
|
||||
return
|
||||
}
|
||||
|
||||
// Stopped gate: the world PVC is RWO and held by a running server, so a backup
|
||||
// Job cannot double-mount it (mirrors the restore gate). Ready means it is up;
|
||||
// any desiredState other than Stopped means it owns the RWO volume.
|
||||
info, err := a.Cluster.GetServer(r.Context(), name)
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
|
||||
writeError(w, r, newError(http.StatusConflict, "not_stopped",
|
||||
"stop the server before backing up its world"))
|
||||
return
|
||||
}
|
||||
|
||||
// Backuper is optional: when unwired the endpoint reports 503 rather than
|
||||
// panicking, so the authorization boundary above is exercised even before the
|
||||
// backup-Job executor is wired (see Backuper).
|
||||
if a.Backuper == nil {
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "backup_unavailable",
|
||||
"backup subsystem is not configured"))
|
||||
return
|
||||
}
|
||||
|
||||
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
|
||||
a.audit(r, p.Email, "backup.create", name)
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||
"name": name,
|
||||
"status": "backing_up",
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,235 @@
|
||||
// Package backupjob implements the on-demand world-backup executor (spec §18/§19
|
||||
// WorldArchiver, run on demand rather than on the reaper's daily schedule). It is
|
||||
// the "back up before I touch it" lever behind POST /api/v1/servers/{name}/backup:
|
||||
// an owner or admin stops a server, then snapshots its world into the archive store
|
||||
// as a first-class world_backups row — restorable by internal/restore and expired
|
||||
// by the reaper's retention pass, so it never leaks as an orphan archive.
|
||||
//
|
||||
// felis-api cannot archive a world in-process: the world PVC is RWO and owned by
|
||||
// the operator's StatefulSet, so the API has nothing to mount at request time (the
|
||||
// same constraint that makes internal/restore a Job). This package is the executor
|
||||
// it hands off to — a one-shot Kubernetes Job in the minecraft namespace that
|
||||
// mounts the target world PVC (read-only) and the backup PVC (read-write), then
|
||||
// runs `felis backup` (cmd/felis) to tar the world into the archive store AND
|
||||
// record the world_backups row.
|
||||
//
|
||||
// Trust model. The backup Pod mirrors internal/restore's weak-SA isolation (a weak
|
||||
// SA with its token un-mounted, so it cannot reach the K8s API) with ONE deliberate
|
||||
// departure, reviewed in jobspec.go: it DOES mount the felis config Secret so it can
|
||||
// self-record its backup row atomically with the archive, exactly like the reaper —
|
||||
// the only other component holding both a world mount and the database. A restore
|
||||
// Pod must stay DB-blind because it processes a poisoned archive; a backup Pod only
|
||||
// reads a world the operator already owns and tars it, so that threat does not apply.
|
||||
//
|
||||
// The Backuper depends on the Jobs interface, so the orchestration (idempotent
|
||||
// enqueue, error mapping) is unit-tested against an in-memory fake; the
|
||||
// controller-runtime implementation (k8sjobs.go) compiles here but is exercised only
|
||||
// by integration tests against a live cluster.
|
||||
package backupjob
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
|
||||
// ErrAlreadyExists is returned by a Jobs implementation when a backup Job for a
|
||||
// server already exists (a backup is already in flight). The Backuper treats it as
|
||||
// success — see Backup.
|
||||
var ErrAlreadyExists = errors.New("backup: job already exists")
|
||||
|
||||
// Jobs is the cluster-side backup lifecycle the Backuper depends on. It is an
|
||||
// interface so the orchestration is tested against a fake; the controller-runtime
|
||||
// implementation (K8sJobs) is integration-tested only — it requires a live cluster.
|
||||
type Jobs interface {
|
||||
// CreateBackupJob renders and applies the backup Job for p. It returns
|
||||
// ErrAlreadyExists if a Job of the same (deterministic) name already exists.
|
||||
CreateBackupJob(ctx context.Context, p JobParams) error
|
||||
}
|
||||
|
||||
// Config parameterises the backup executor. Deployment-specific values that have no
|
||||
// safe default — the felis Image to run and the BackupPVC to mount — are supplied
|
||||
// by the caller (cmd/felis sources them from the environment); when either is empty
|
||||
// the caller leaves the API's Backuper nil so the endpoint reports 503 rather than
|
||||
// enqueuing a Job that cannot run.
|
||||
type Config struct {
|
||||
// Namespace is where the world PVCs live and the backup Job runs (the minecraft
|
||||
// namespace), co-located with the world it snapshots.
|
||||
Namespace string
|
||||
// ServiceAccount is the weak SA the backup Pod runs as. It reuses felis-restore
|
||||
// (bare, no Role/RoleBinding): a backup Pod needs no K8s API access, only
|
||||
// filesystem access to the two PVCs and — via the mounted config Secret, not the
|
||||
// SA — the database.
|
||||
ServiceAccount string
|
||||
// Image is the felis binary image; the Job runs `felis backup` from it.
|
||||
Image string
|
||||
// BackupPVC is the name of the backup PVC the archive is written into (the same
|
||||
// PVC the reaper writes to and restore reads from).
|
||||
BackupPVC string
|
||||
// ConfigSecret is the felis config Secret (felis.toml, carrying the DB URL) the
|
||||
// backup Pod mounts to self-record its world_backups row. Defaults to the name
|
||||
// the control-plane manifests use.
|
||||
ConfigSecret string
|
||||
// ConfigMount is the in-Pod mount path of ConfigSecret (holds felis.toml).
|
||||
ConfigMount string
|
||||
// BackupRoot is the in-Pod mount path of BackupPVC. It MUST equal cfg.Archive.
|
||||
// LocalPath — the path the reaper wrote archives under and restore mounts to
|
||||
// resolve them — because tarLocal archive refs are absolute.
|
||||
BackupRoot string
|
||||
// WorldsRoot is the in-Pod mount path of the world PVC being archived.
|
||||
WorldsRoot string
|
||||
// Deadline caps the backup Pod's wall-clock (activeDeadlineSeconds).
|
||||
Deadline time.Duration
|
||||
// CPULimit / MemLimit cap the backup container.
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
// RunAsUser / RunAsGroup / FSGroup are the Pod's runtime identity. FSGroup MUST
|
||||
// match the operator StatefulSet's runtime group so the read-only world mount is
|
||||
// readable by this Pod's uid.
|
||||
RunAsUser int64
|
||||
RunAsGroup int64
|
||||
FSGroup int64
|
||||
// TTLAfterFinished is how long a finished backup Job lingers before the Job
|
||||
// controller garbage-collects it; it also bounds the window in which a re-backup
|
||||
// sees a stale completed Job as ErrAlreadyExists.
|
||||
TTLAfterFinished time.Duration
|
||||
}
|
||||
|
||||
// defaults applied when a Config field is left zero. Image and BackupPVC have no
|
||||
// default on purpose — see Config.
|
||||
const (
|
||||
defaultNamespace = "minecraft"
|
||||
defaultServiceAccount = "felis-restore"
|
||||
defaultConfigSecret = "felis-config"
|
||||
defaultConfigMount = "/etc/felis"
|
||||
defaultBackupRoot = "/backups"
|
||||
defaultWorldsRoot = "/world"
|
||||
defaultDeadline = 30 * time.Minute
|
||||
defaultCPULimit = "1"
|
||||
defaultMemLimit = "1Gi"
|
||||
defaultRunAsID = int64(1000)
|
||||
defaultTTL = 10 * time.Minute
|
||||
)
|
||||
|
||||
// withDefaults returns a copy of c with zero fields filled, so a partially
|
||||
// configured Config (or the zero value, in tests) is always usable.
|
||||
func (c Config) withDefaults() Config {
|
||||
if c.Namespace == "" {
|
||||
c.Namespace = defaultNamespace
|
||||
}
|
||||
if c.ServiceAccount == "" {
|
||||
c.ServiceAccount = defaultServiceAccount
|
||||
}
|
||||
if c.ConfigSecret == "" {
|
||||
c.ConfigSecret = defaultConfigSecret
|
||||
}
|
||||
if c.ConfigMount == "" {
|
||||
c.ConfigMount = defaultConfigMount
|
||||
}
|
||||
if c.BackupRoot == "" {
|
||||
c.BackupRoot = defaultBackupRoot
|
||||
}
|
||||
if c.WorldsRoot == "" {
|
||||
c.WorldsRoot = defaultWorldsRoot
|
||||
}
|
||||
if c.Deadline <= 0 {
|
||||
c.Deadline = defaultDeadline
|
||||
}
|
||||
if c.CPULimit == "" {
|
||||
c.CPULimit = defaultCPULimit
|
||||
}
|
||||
if c.MemLimit == "" {
|
||||
c.MemLimit = defaultMemLimit
|
||||
}
|
||||
if c.RunAsUser == 0 {
|
||||
c.RunAsUser = defaultRunAsID
|
||||
}
|
||||
if c.RunAsGroup == 0 {
|
||||
c.RunAsGroup = defaultRunAsID
|
||||
}
|
||||
if c.FSGroup == 0 {
|
||||
c.FSGroup = defaultRunAsID
|
||||
}
|
||||
if c.TTLAfterFinished <= 0 {
|
||||
c.TTLAfterFinished = defaultTTL
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// Backuper is the production internal/api.Backuper (the compile-time proof of that
|
||||
// is in internal/api's wire test, which imports this package; this package never
|
||||
// imports api). It holds no mutable state.
|
||||
type Backuper struct {
|
||||
Jobs Jobs
|
||||
Config Config
|
||||
}
|
||||
|
||||
// Backup enqueues a backup Job that tars serverName's world into the archive store
|
||||
// and records the world_backups row. formerOwner is stamped on that row so the owner
|
||||
// can later restore it (empty for an admin backing up an unowned server). It returns
|
||||
// once the Job is created — the archive runs in the Pod — so the handler's 202
|
||||
// ("backing_up") is honest.
|
||||
//
|
||||
// Each call gets a fresh, unique Job name (BackupJobName + random suffix), so an
|
||||
// on-demand backup requested again after a previous one — the console "立即备份"
|
||||
// repeat-tap case — always produces a new archive rather than colliding with a
|
||||
// just-finished Job still inside its TTL window. ErrAlreadyExists is kept only as a
|
||||
// defensive no-op against the astronomically unlikely suffix collision.
|
||||
//
|
||||
// ponytail: unique names mean two truly simultaneous taps can schedule two backup
|
||||
// Pods; both mount the world PVC read-only so neither corrupts anything, and if they
|
||||
// land on different nodes the RWO attach fails one cleanly. Add single-flight-on-
|
||||
// running only if a real double-tap storm ever shows up.
|
||||
func (b *Backuper) Backup(ctx context.Context, serverName, formerOwner string) error {
|
||||
if err := b.Jobs.CreateBackupJob(ctx, b.jobParams(serverName, formerOwner)); err != nil {
|
||||
if errors.Is(err, ErrAlreadyExists) {
|
||||
return nil // suffix collision — treat as enqueued
|
||||
}
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// jobNameSuffix is a short random hex tag that makes each backup Job name unique.
|
||||
// 32 bits is ample: collisions only matter within a single Job's TTL window across
|
||||
// a handful of manual backups.
|
||||
func jobNameSuffix() string {
|
||||
var b [4]byte
|
||||
if _, err := rand.Read(b[:]); err != nil {
|
||||
// ponytail: crypto/rand only fails if the OS RNG is gone — unrecoverable.
|
||||
panic("backupjob: crypto/rand: " + err.Error())
|
||||
}
|
||||
return hex.EncodeToString(b[:])
|
||||
}
|
||||
|
||||
// jobParams projects the server, former owner, and config onto the inputs jobspec.go
|
||||
// renders. The world PVC name is derived from the single shared naming convention
|
||||
// (naming.WorldPVCName), the same one the operator created it under.
|
||||
func (b *Backuper) jobParams(serverName, formerOwner string) JobParams {
|
||||
cfg := b.Config.withDefaults()
|
||||
return JobParams{
|
||||
Server: serverName,
|
||||
JobName: BackupJobName(serverName) + "-" + jobNameSuffix(),
|
||||
FormerOwner: formerOwner,
|
||||
WorldPVC: naming.WorldPVCName(serverName),
|
||||
BackupPVC: cfg.BackupPVC,
|
||||
Namespace: cfg.Namespace,
|
||||
ServiceAccount: cfg.ServiceAccount,
|
||||
Image: cfg.Image,
|
||||
ConfigSecret: cfg.ConfigSecret,
|
||||
ConfigMount: cfg.ConfigMount,
|
||||
BackupRoot: cfg.BackupRoot,
|
||||
WorldsRoot: cfg.WorldsRoot,
|
||||
Deadline: cfg.Deadline,
|
||||
CPULimit: cfg.CPULimit,
|
||||
MemLimit: cfg.MemLimit,
|
||||
RunAsUser: cfg.RunAsUser,
|
||||
RunAsGroup: cfg.RunAsGroup,
|
||||
FSGroup: cfg.FSGroup,
|
||||
TTLAfterFinished: cfg.TTLAfterFinished,
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
package backupjob
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// captureJobs records every JobParams it is handed so the test can inspect the
|
||||
// names the Backuper minted.
|
||||
type captureJobs struct{ got []JobParams }
|
||||
|
||||
func (c *captureJobs) CreateBackupJob(_ context.Context, p JobParams) error {
|
||||
c.got = append(c.got, p)
|
||||
return nil
|
||||
}
|
||||
|
||||
// The core fix: a repeat on-demand backup of the same server must not collide with
|
||||
// the previous Job's name (which would silently no-op the retry within the TTL
|
||||
// window). Two Backup calls must mint two distinct Job names, both under the
|
||||
// server's base prefix.
|
||||
func TestBackupMintsUniqueJobNamePerCall(t *testing.T) {
|
||||
jobs := &captureJobs{}
|
||||
b := &Backuper{Jobs: jobs, Config: Config{Image: "img", BackupPVC: "pvc"}}
|
||||
|
||||
for i := 0; i < 2; i++ {
|
||||
if err := b.Backup(context.Background(), "survival", "usr-1"); err != nil {
|
||||
t.Fatalf("Backup: %v", err)
|
||||
}
|
||||
}
|
||||
if len(jobs.got) != 2 {
|
||||
t.Fatalf("created %d jobs, want 2", len(jobs.got))
|
||||
}
|
||||
base := BackupJobName("survival")
|
||||
for _, p := range jobs.got {
|
||||
if !strings.HasPrefix(p.JobName, base+"-") {
|
||||
t.Errorf("JobName %q must extend the base %q", p.JobName, base)
|
||||
}
|
||||
}
|
||||
if jobs.got[0].JobName == jobs.got[1].JobName {
|
||||
t.Errorf("two backups reused the Job name %q — retry would silently no-op", jobs.got[0].JobName)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,242 @@
|
||||
package backupjob
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
)
|
||||
|
||||
// Label keys applied to backup objects, mirroring internal/restore so the two
|
||||
// executors are observable the same way.
|
||||
const (
|
||||
LabelManagedBy = "app.kubernetes.io/managed-by"
|
||||
LabelComponent = "app.kubernetes.io/component"
|
||||
LabelServer = "felis.lolicon.best/server"
|
||||
|
||||
managedByValue = "felis-backup"
|
||||
componentValue = "world-backup"
|
||||
|
||||
worldVolume = "world"
|
||||
backupVolume = "backup"
|
||||
configVolume = "config"
|
||||
tmpVolume = "tmp"
|
||||
felisBinaryPath = "/usr/local/bin/felis"
|
||||
)
|
||||
|
||||
// JobParams are the rendered inputs to a backup Job, derived from a server +
|
||||
// former owner + Config by the Backuper. jobspec is a pure function of them so
|
||||
// the security-critical Job shape is unit-tested without a cluster.
|
||||
type JobParams struct {
|
||||
Server string
|
||||
// JobName is the unique Job object name for this invocation. The Backuper sets a
|
||||
// fresh per-call name (BackupJobName(server) + random suffix) so every on-demand
|
||||
// request produces its own archive — a deterministic name would collide with a
|
||||
// just-finished Job still inside its TTL window and silently no-op the retry.
|
||||
// Empty falls back to the deterministic BackupJobName (tests, and any direct call).
|
||||
JobName string
|
||||
FormerOwner string // recorded on the world_backups row so the owner can later restore it
|
||||
WorldPVC string
|
||||
BackupPVC string
|
||||
Namespace string
|
||||
ServiceAccount string
|
||||
Image string
|
||||
ConfigSecret string // the felis config Secret mounted for the DB URL (self-recording)
|
||||
ConfigMount string // where the config Secret is mounted (holds felis.toml)
|
||||
BackupRoot string
|
||||
WorldsRoot string
|
||||
Deadline time.Duration
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
RunAsUser int64
|
||||
RunAsGroup int64
|
||||
FSGroup int64
|
||||
|
||||
TTLAfterFinished time.Duration
|
||||
}
|
||||
|
||||
// BackupJobName is the base Job name for a server's on-demand backup — the prefix
|
||||
// the Backuper extends with a per-invocation random suffix (JobParams.JobName) so
|
||||
// repeat backups do not collide. It is distinct from the restore Job name so a
|
||||
// backup and a restore of the same server never collide.
|
||||
func BackupJobName(server string) string { return "backup-" + server }
|
||||
|
||||
func backupLabels(p JobParams) map[string]string {
|
||||
return map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
LabelServer: p.Server,
|
||||
}
|
||||
}
|
||||
|
||||
// BackupJob renders the on-demand world-backup Job (spec §18/§19 WorldArchiver,
|
||||
// run on demand rather than on the reaper's daily schedule). Its isolation mirrors
|
||||
// the restore Job (weak SA, non-root, read-only root fs, drop ALL, one-shot with a
|
||||
// deadline) with ONE deliberate, security-reviewed departure asserted by
|
||||
// jobspec_test.go:
|
||||
//
|
||||
// - It DOES mount the felis config Secret (read-only), because a backup must
|
||||
// self-record its world_backups row — archive-then-insert atomically, exactly
|
||||
// like the reaper, which is the only other component holding both a world mount
|
||||
// and the database. A restore Pod must never touch the DB because it processes a
|
||||
// potentially poisoned archive; a backup Pod only READS a world the operator
|
||||
// already owns and tars it (bytes, never executed), so the poisoned-input threat
|
||||
// that forbids restore's DB access does not apply. Its blast radius is the DB
|
||||
// plus the two PVCs, matching the reaper's trust for a strict subset of the
|
||||
// reaper's operations (it never deletes a PVC and never calls the K8s API — its
|
||||
// SA token stays un-mounted).
|
||||
// - The world PVC is mounted READ-ONLY (backup only reads it; the RWO volume must
|
||||
// be free, which the handler's stopped-gate guarantees), and the backup PVC
|
||||
// READ-WRITE (the archive is written into it) — the mirror image of restore.
|
||||
//
|
||||
// The container runs `/usr/local/bin/felis backup` (cmd/felis), which tars the
|
||||
// world at WorldsRoot into BackupRoot and inserts the world_backups row. BackupRoot
|
||||
// MUST equal the reaper's [archive] local_path so the recorded absolute ref
|
||||
// resolves the same way a later restore Job mounts it.
|
||||
func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||
if p.Image == "" {
|
||||
return nil, fmt.Errorf("backup: image is empty")
|
||||
}
|
||||
if p.WorldPVC == "" || p.BackupPVC == "" {
|
||||
return nil, fmt.Errorf("backup: world and backup PVC names are required")
|
||||
}
|
||||
if p.ConfigSecret == "" {
|
||||
return nil, fmt.Errorf("backup: config secret name is required")
|
||||
}
|
||||
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
deadline := int64(p.Deadline / time.Second)
|
||||
if deadline <= 0 {
|
||||
deadline = int64(defaultDeadline / time.Second)
|
||||
}
|
||||
ttl := int32(p.TTLAfterFinished / time.Second)
|
||||
if ttl <= 0 {
|
||||
ttl = int32(defaultTTL / time.Second)
|
||||
}
|
||||
|
||||
args := []string{
|
||||
"--config", p.ConfigMount + "/felis.toml",
|
||||
"--server", p.Server,
|
||||
"--worlds-root", p.WorldsRoot,
|
||||
}
|
||||
// FormerOwner is optional: an admin backing up an unowned (released) server
|
||||
// records an empty former_owner, exactly as the reaper does for an unowned reap.
|
||||
if p.FormerOwner != "" {
|
||||
args = append(args, "--former-owner", p.FormerOwner)
|
||||
}
|
||||
|
||||
container := corev1.Container{
|
||||
Name: "backup",
|
||||
Image: p.Image,
|
||||
Command: []string{felisBinaryPath, "backup"},
|
||||
Args: args,
|
||||
VolumeMounts: []corev1.VolumeMount{
|
||||
// The world is only read to tar it; mounting it read-only means the
|
||||
// backup process can never mutate the world it snapshots.
|
||||
{Name: worldVolume, MountPath: p.WorldsRoot, ReadOnly: true},
|
||||
{Name: backupVolume, MountPath: p.BackupRoot},
|
||||
{Name: configVolume, MountPath: p.ConfigMount, ReadOnly: true},
|
||||
{Name: tmpVolume, MountPath: "/tmp"},
|
||||
},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||
SecurityContext: &corev1.SecurityContext{
|
||||
Privileged: boolPtr(false),
|
||||
AllowPrivilegeEscalation: boolPtr(false),
|
||||
ReadOnlyRootFilesystem: boolPtr(true),
|
||||
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
||||
},
|
||||
}
|
||||
|
||||
name := p.JobName
|
||||
if name == "" {
|
||||
name = BackupJobName(p.Server)
|
||||
}
|
||||
job := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: p.Namespace,
|
||||
Labels: backupLabels(p),
|
||||
},
|
||||
Spec: batchv1.JobSpec{
|
||||
// One shot: a wedged archive must not loop. The TTL GCs the finished Job
|
||||
// so a later backup of the same server is not blocked forever by a stale
|
||||
// completed Job.
|
||||
BackoffLimit: int32Ptr(0),
|
||||
ActiveDeadlineSeconds: int64Ptr(deadline),
|
||||
TTLSecondsAfterFinished: int32Ptr(ttl),
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: backupLabels(p)},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: p.ServiceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
SecurityContext: &corev1.PodSecurityContext{
|
||||
RunAsNonRoot: boolPtr(true),
|
||||
RunAsUser: int64Ptr(p.RunAsUser),
|
||||
RunAsGroup: int64Ptr(p.RunAsGroup),
|
||||
FSGroup: int64Ptr(p.FSGroup),
|
||||
},
|
||||
Containers: []corev1.Container{container},
|
||||
Volumes: []corev1.Volume{
|
||||
{
|
||||
Name: worldVolume,
|
||||
VolumeSource: corev1.VolumeSource{
|
||||
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
||||
ClaimName: p.WorldPVC,
|
||||
ReadOnly: true,
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
Name: backupVolume,
|
||||
VolumeSource: corev1.VolumeSource{
|
||||
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
||||
ClaimName: p.BackupPVC,
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
Name: configVolume,
|
||||
VolumeSource: corev1.VolumeSource{
|
||||
Secret: &corev1.SecretVolumeSource{SecretName: p.ConfigSecret},
|
||||
},
|
||||
},
|
||||
{Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
return job, nil
|
||||
}
|
||||
|
||||
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
||||
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||
if cpu == "" {
|
||||
cpu = defaultCPULimit
|
||||
}
|
||||
if mem == "" {
|
||||
mem = defaultMemLimit
|
||||
}
|
||||
cpuQty, err := resource.ParseQuantity(cpu)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("backup: invalid cpu limit %q: %w", cpu, err)
|
||||
}
|
||||
memQty, err := resource.ParseQuantity(mem)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("backup: invalid memory limit %q: %w", mem, err)
|
||||
}
|
||||
return corev1.ResourceList{
|
||||
corev1.ResourceCPU: cpuQty,
|
||||
corev1.ResourceMemory: memQty,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
func int32Ptr(i int32) *int32 { return &i }
|
||||
func int64Ptr(i int64) *int64 { return &i }
|
||||
@@ -0,0 +1,220 @@
|
||||
package backupjob
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
func sampleJobParams() JobParams {
|
||||
return JobParams{
|
||||
Server: "survival",
|
||||
FormerOwner: "usr-abc",
|
||||
WorldPVC: "world-survival-0",
|
||||
BackupPVC: "felis-backups",
|
||||
Namespace: defaultNamespace,
|
||||
ServiceAccount: defaultServiceAccount,
|
||||
Image: "registry.felis.svc:5000/felis:1.0",
|
||||
ConfigSecret: defaultConfigSecret,
|
||||
ConfigMount: defaultConfigMount,
|
||||
BackupRoot: "/backups",
|
||||
WorldsRoot: "/world",
|
||||
Deadline: 30 * time.Minute,
|
||||
CPULimit: "1",
|
||||
MemLimit: "1Gi",
|
||||
RunAsUser: 1000,
|
||||
RunAsGroup: 1000,
|
||||
FSGroup: 1000,
|
||||
TTLAfterFinished: 10 * time.Minute,
|
||||
}
|
||||
}
|
||||
|
||||
// The backup Pod runs under the weak felis-restore SA — never the felis-api
|
||||
// identity — with its token un-mounted, so it cannot reach the K8s API. Its DB
|
||||
// access comes from the mounted config Secret, not from any SA permission.
|
||||
func TestBackupJobRunsUnderWeakSAWithNoAPIToken(t *testing.T) {
|
||||
job, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
sa := job.Spec.Template.Spec.ServiceAccountName
|
||||
if sa != defaultServiceAccount {
|
||||
t.Errorf("service account = %q, want %q", sa, defaultServiceAccount)
|
||||
}
|
||||
if sa == "felis-api" {
|
||||
t.Fatal("backup Pod must NOT run as the felis-api SA")
|
||||
}
|
||||
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
|
||||
t.Error("AutomountServiceAccountToken must be explicitly false")
|
||||
}
|
||||
}
|
||||
|
||||
// The deliberate departure from restore's zero-secret isolation: a backup Pod
|
||||
// mounts EXACTLY the two PVCs (world read-only, backup read-write) PLUS the config
|
||||
// Secret read-only (so it can self-record its world_backups row like the reaper) —
|
||||
// and nothing else. This test freezes that exact volume set so a future edit that
|
||||
// widens it (e.g. a second Secret, or a writable world mount) fails loudly.
|
||||
func TestBackupJobMountsTwoPVCsPlusConfigSecretOnly(t *testing.T) {
|
||||
job, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
spec := job.Spec.Template.Spec
|
||||
|
||||
var secretVols, pvcVols int
|
||||
for _, v := range spec.Volumes {
|
||||
switch {
|
||||
case v.Secret != nil:
|
||||
secretVols++
|
||||
if v.Secret.SecretName != defaultConfigSecret {
|
||||
t.Errorf("secret volume = %q, want the config secret %q", v.Secret.SecretName, defaultConfigSecret)
|
||||
}
|
||||
case v.PersistentVolumeClaim != nil:
|
||||
pvcVols++
|
||||
case v.EmptyDir != nil:
|
||||
// the /tmp scratch dir under the read-only root fs — allowed
|
||||
default:
|
||||
t.Errorf("unexpected volume %q: a backup Pod mounts only the two PVCs, the config Secret, and a /tmp emptyDir", v.Name)
|
||||
}
|
||||
}
|
||||
if secretVols != 1 {
|
||||
t.Errorf("secret volumes = %d, want exactly 1 (the config Secret)", secretVols)
|
||||
}
|
||||
if pvcVols != 2 {
|
||||
t.Errorf("PVC volumes = %d, want exactly 2 (world + backup)", pvcVols)
|
||||
}
|
||||
|
||||
// World read-only (backup never mutates the world), backup read-write (the
|
||||
// archive is written into it) — the mirror image of the restore Job.
|
||||
world := mountByName(t, spec.Containers[0].VolumeMounts, worldVolume)
|
||||
if !world.ReadOnly {
|
||||
t.Error("world mount must be read-only — a backup only reads the world")
|
||||
}
|
||||
back := mountByName(t, spec.Containers[0].VolumeMounts, backupVolume)
|
||||
if back.ReadOnly {
|
||||
t.Error("backup mount must be read-write — the archive is written into it")
|
||||
}
|
||||
cfg := mountByName(t, spec.Containers[0].VolumeMounts, configVolume)
|
||||
if !cfg.ReadOnly {
|
||||
t.Error("config Secret mount must be read-only")
|
||||
}
|
||||
|
||||
// The world PVC volume itself is also declared read-only so the RWO claim is
|
||||
// requested read-only (defense in depth beyond the mount flag).
|
||||
for _, v := range spec.Volumes {
|
||||
if v.PersistentVolumeClaim != nil && v.PersistentVolumeClaim.ClaimName == "world-survival-0" && !v.PersistentVolumeClaim.ReadOnly {
|
||||
t.Error("world PVC volume source must be read-only")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The backup container is hardened exactly like the restore/build Job containers:
|
||||
// no privilege, no escalation, read-only root fs, drop ALL capabilities.
|
||||
func TestBackupJobContainerIsHardened(t *testing.T) {
|
||||
job, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
sc := job.Spec.Template.Spec.Containers[0].SecurityContext
|
||||
if sc == nil {
|
||||
t.Fatal("container SecurityContext is nil")
|
||||
}
|
||||
if sc.Privileged == nil || *sc.Privileged {
|
||||
t.Error("Privileged must be false")
|
||||
}
|
||||
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
|
||||
t.Error("AllowPrivilegeEscalation must be false")
|
||||
}
|
||||
if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
|
||||
t.Error("ReadOnlyRootFilesystem must be true")
|
||||
}
|
||||
if sc.Capabilities == nil || len(sc.Capabilities.Drop) != 1 || sc.Capabilities.Drop[0] != "ALL" {
|
||||
t.Error("capabilities must drop ALL")
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot: a wedged archive must not loop, and a deadline caps it.
|
||||
func TestBackupJobIsOneShotWithDeadline(t *testing.T) {
|
||||
job, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
|
||||
t.Error("BackoffLimit must be 0 (no retry loop)")
|
||||
}
|
||||
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds <= 0 {
|
||||
t.Error("ActiveDeadlineSeconds must be set")
|
||||
}
|
||||
if job.Spec.TTLSecondsAfterFinished == nil {
|
||||
t.Error("TTLSecondsAfterFinished must be set so the finished Job is GC'd")
|
||||
}
|
||||
}
|
||||
|
||||
// The command carries the former owner so the recorded backup can be restored by
|
||||
// its owner; an empty former owner (admin backing up an unowned server) omits it.
|
||||
func TestBackupJobArgsCarryServerAndOwner(t *testing.T) {
|
||||
job, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
args := job.Spec.Template.Spec.Containers[0].Args
|
||||
if !argsContain(args, "--server", "survival") {
|
||||
t.Errorf("args missing --server survival: %v", args)
|
||||
}
|
||||
if !argsContain(args, "--former-owner", "usr-abc") {
|
||||
t.Errorf("args missing --former-owner usr-abc: %v", args)
|
||||
}
|
||||
|
||||
p := sampleJobParams()
|
||||
p.FormerOwner = ""
|
||||
unowned, err := BackupJob(p)
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob(unowned): %v", err)
|
||||
}
|
||||
for _, a := range unowned.Spec.Template.Spec.Containers[0].Args {
|
||||
if a == "--former-owner" {
|
||||
t.Error("--former-owner must be omitted when there is no former owner")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
mut func(*JobParams)
|
||||
}{
|
||||
{"no image", func(p *JobParams) { p.Image = "" }},
|
||||
{"no world pvc", func(p *JobParams) { p.WorldPVC = "" }},
|
||||
{"no backup pvc", func(p *JobParams) { p.BackupPVC = "" }},
|
||||
{"no config secret", func(p *JobParams) { p.ConfigSecret = "" }},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
tc.mut(&p)
|
||||
if _, err := BackupJob(p); err == nil {
|
||||
t.Errorf("BackupJob(%s) = nil error, want a validation error", tc.name)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func mountByName(t *testing.T, mounts []corev1.VolumeMount, name string) corev1.VolumeMount {
|
||||
t.Helper()
|
||||
for _, m := range mounts {
|
||||
if m.Name == name {
|
||||
return m
|
||||
}
|
||||
}
|
||||
t.Fatalf("volume mount %q not found", name)
|
||||
return corev1.VolumeMount{}
|
||||
}
|
||||
|
||||
func argsContain(args []string, flag, val string) bool {
|
||||
for i := 0; i+1 < len(args); i++ {
|
||||
if args[i] == flag && args[i+1] == val {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
package backupjob
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// K8sJobs is the production Jobs backed by a controller-runtime client. It creates
|
||||
// the on-demand world-backup Job — nothing more: the backup Job is one-shot and
|
||||
// self-cleaning (ttlSecondsAfterFinished), so there is no phase or cancel seam, and
|
||||
// thus no config to hold. Every backup parameter arrives in the JobParams the
|
||||
// Backuper builds from its own (defaulted) Config. The cluster-bootstrap objects
|
||||
// (the weak felis-restore SA it reuses) are installed once by the deployment
|
||||
// manifests, not per backup, so this binding never creates them. It is
|
||||
// integration-tested against a live cluster, not the hermetic unit suite.
|
||||
type K8sJobs struct {
|
||||
c client.Client
|
||||
}
|
||||
|
||||
// NewK8sJobs builds a Jobs over c.
|
||||
func NewK8sJobs(c client.Client) *K8sJobs {
|
||||
return &K8sJobs{c: c}
|
||||
}
|
||||
|
||||
// CreateBackupJob renders and applies the backup Job. Its name is a deterministic
|
||||
// function of the server (BackupJobName), so a concurrent backup of the same server
|
||||
// collides on Create; that collision is mapped to ErrAlreadyExists, which the
|
||||
// Backuper treats as success (idempotent enqueue).
|
||||
func (k *K8sJobs) CreateBackupJob(ctx context.Context, p JobParams) error {
|
||||
job, err := BackupJob(p)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := k.c.Create(ctx, job); err != nil {
|
||||
if apierrors.IsAlreadyExists(err) {
|
||||
return ErrAlreadyExists
|
||||
}
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
Reference in new issue
Block a user