feat(api): add on-demand world backup endpoint and Job executor (§B4 Sync)

Add POST /api/v1/servers/{name}/backup: an owner or admin snapshots a
stopped server's world into the archive store on demand, recorded as a
first-class world_backups row (reason `manual`) — restorable by the
existing restore path and expired by the reaper's retention pass, so it
never leaks as an orphan archive. This is the break-glass "Sync" op,
resolved as immediate/on-demand backup.

felis-api cannot archive in-process (the world PVC is RWO, held by the
operator StatefulSet), so the work hands off to a one-shot Kubernetes Job
(new internal/backupjob) that mounts the world PVC read-only and the
backup PVC read-write, plus the felis config Secret so it self-records
its row atomically like the reaper. The Pod mirrors restore's weak-SA
isolation (SA token un-mounted, non-root, read-only rootfs, drop ALL);
the one reviewed departure is that config-Secret mount, frozen by
jobspec_test.go. Handler answers 202 backing_up; gated on the server
being Stopped (RWO world PVC), owner-or-admin, and FELIS_IMAGE +
FELIS_BACKUP_PVC being wired (else 503 backup_unavailable).

Each request mints a unique Job name (backup-<server>-<rand>) so a repeat
on-demand backup produces a fresh archive rather than colliding with a
just-finished Job still inside its TTL window and silently no-op'ing the
retry.
This commit is contained in:
flyemoji committed 2026-07-07 10:04:30 +09:00
1 parent fad48ff21d
commit 7a7c0d53ab
16 files changed
+1396 -1

No files matched your search

+30
View File
@@ -13,6 +13,7 @@ import (
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/backupjob"
"felis.lolicon.best/internal/build"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/panel"
@@ -162,6 +163,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
fmt.Fprintln(stderr, "felis api: restore executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — restore endpoint returns 503")
}
// On-demand backup subsystem (spec §18/§19 WorldArchiver, run on demand). Its
// backup Job mirrors the restore Job's weak-SA isolation but additionally mounts
// the config Secret so it self-records the world_backups row (see internal/
// backupjob). It needs the same deployment-specific values as restore, so it is
// wired under the same gate; otherwise the Backuper is left nil and the backup
// endpoint honestly returns 503.
var backuper api.Backuper
if felisImage != "" && backupPVC != "" {
backuper = &backupjob.Backuper{Jobs: backupjob.NewK8sJobs(cl), Config: backupConfig(cfg, felisImage, backupPVC)}
} else {
fmt.Fprintln(stderr, "felis api: backup executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — backup endpoint returns 503")
}
// One PGRepo instance backs both the handlers and the session verifier: the
// SessionAuth that fronts the external face reads sessions/users/settings from
// the same store the auth handlers write to, so a login and the next request
@@ -179,6 +193,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
Internal: api.BearerTokenAuth{Token: token},
Builder: builder,
Restorer: restorer,
Backuper: backuper,
Submissions: submissions,
// The external face is fronted by SessionAuth: it prefers a local-password
// session cookie and otherwise delegates to the Cloudflare-Access JWT verifier,
@@ -359,6 +374,21 @@ func restoreConfig(cfg *config.Config, image, backupPVC string) restore.Config {
}
}
// backupConfig builds the on-demand backup executor's config from felis.toml plus
// the deployment-supplied image and backup PVC. BackupRoot mirrors restoreConfig —
// it MUST equal [archive] local_path so the recorded ref resolves the same way a
// later restore Job mounts it. ConfigSecret/ConfigMount are left to backupjob's
// defaults (the control-plane manifest names), which is the Secret this backup Job
// mounts to self-record its world_backups row.
func backupConfig(cfg *config.Config, image, backupPVC string) backupjob.Config {
return backupjob.Config{
Namespace: cfg.K8s.Namespace,
Image: image,
BackupPVC: backupPVC,
BackupRoot: cfg.Archive.LocalPath,
}
}
// reconcileBuilds polls unfinished builds on an interval and advances any whose
// Job has reached a terminal phase. It exits when ctx is cancelled.
func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
+123
View File
@@ -0,0 +1,123 @@
package main
import (
"crypto/rand"
"encoding/hex"
"flag"
"fmt"
"io"
"time"
"felis.lolicon.best/internal/backup"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/reaper"
"felis.lolicon.best/internal/store"
ctrl "sigs.k8s.io/controller-runtime"
)
// cmdBackup is the in-Pod entrypoint the on-demand backup Job runs. internal/backupjob
// renders a Pod whose command is `/usr/local/bin/felis backup`. It tars the mounted
// world into the archive store AND records the world_backups row, then exits — it is
// NOT a user-facing command and is never invoked by hand.
//
// Unlike `felis restore`, this command DOES hold database credentials (via the mounted
// config Secret) and calls config.Load: a backup must record its row atomically with
// the archive, exactly like the reaper — otherwise a completed archive would leak as an
// orphan file the retention pass never expires. The security review for that departure
// lives in internal/backupjob/jobspec.go. The world is mounted directly at --worlds-root
// (single-PVC mount, like restore), so the archiver's resolver returns that root for any
// PVC; the archive is written into the backup PVC at cfg.Archive.LocalPath.
func cmdBackup(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("backup", flag.ContinueOnError)
fs.SetOutput(stderr)
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
server := fs.String("server", "", "server name whose world is being backed up")
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
if err := fs.Parse(args); err != nil {
return 2
}
if *server == "" {
fmt.Fprintln(stderr, "felis backup: --server is required")
return 2
}
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
if cfg.Archive.Store != "tarLocal" {
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
return 1
}
// Reuse the reaper's retention derivation so an on-demand backup expires on the
// same clock as an inactivity backup — one retention policy, not two.
rcfg, err := reaperConfig(cfg)
if err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
// writes archives with.
archiver := &backup.TarLocal{
BackupRoot: cfg.Archive.LocalPath,
Resolve: func(string) (string, error) {
return *worldsRoot, nil
},
}
ctx := ctrl.SetupSignalHandler()
ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
if err != nil {
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
return 1
}
drv, err := store.Open(ctx, cfg.Database.URL)
if err != nil {
fmt.Fprintf(stderr, "felis backup: open database: %v\n", err)
return 1
}
defer drv.Close()
rec := reaper.BackupRecord{
ID: newBackupID(),
ServerName: *server,
FormerOwner: *formerOwner,
BackupRef: string(ref),
SizeBytes: size,
Reason: "manual",
ExpiresAt: time.Now().Add(rcfg.Retention),
}
if err := reaper.NewPGStore(drv.DB()).InsertBackup(ctx, rec); err != nil {
// The archive is written but unrecorded — an orphan the retention pass would
// never expire. Delete it so a failed backup leaves no leaked bytes, mirroring
// the reaper's archive-then-record atomicity.
if delErr := archiver.Delete(ctx, ref); delErr != nil {
fmt.Fprintf(stderr, "felis backup: record failed (%v) AND orphan archive %s could not be removed: %v\n", err, ref, delErr)
return 1
}
fmt.Fprintf(stderr, "felis backup: record failed, orphan archive removed: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
return 0
}
// newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex
// scheme so a manual and an inactivity backup are indistinguishable downstream.
func newBackupID() string {
var b [16]byte
if _, err := rand.Read(b[:]); err != nil {
// crypto/rand failure is fatal and unrecoverable; a time-based fallback would
// be a weaker ID for no benefit. ponytail: panic is the honest failure here.
panic("felis backup: crypto/rand: " + err.Error())
}
return "bk-" + hex.EncodeToString(b[:])
}
+3
View File
@@ -16,6 +16,7 @@ Commands:
api Run the felis-api HTTP server
reaper Run the world reaper / backup batch
restore Extract a world archive into a world volume (internal Job entrypoint)
backup Archive a world into the backup store and record it (internal Job entrypoint)
manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
@@ -43,6 +44,8 @@ func run(args []string, stdout, stderr io.Writer) int {
return cmdReaper(rest, stdout, stderr)
case "restore":
return cmdRestore(rest, stdout, stderr)
case "backup":
return cmdBackup(rest, stdout, stderr)
case "manifests":
return cmdManifests(rest, stdout, stderr)
case "apply":
@@ -0,0 +1,115 @@
# On-demand world backup (§B4 break-glass "Sync"; felis-api endpoint + Job executor)
- **Type:** feature (addition)
- **Date:** 2026-07-07
- **Area:** `internal/backupjob` (new pkg), `internal/api`, `cmd/felis`, `docs/openapi.yaml` — Go, oracle-verified
- **Commit:** `pending`
- **Task:** #31 Phase B4 break-glass ops — the "Sync" operation, resolved with the user as **immediate/on-demand world backup**. Per the user's "两者都要" decision this is built in two halves: **(this change) the felis-api endpoint that does the real backup-Job orchestration**, and (a follow-up) a break-glass menu peer that calls it while the API is alive.
## What it does
Adds `POST /api/v1/servers/{name}/backup`: an owner or admin snapshots a **stopped**
server's world into the archive store on demand, recorded as a first-class
`world_backups` row (reason `manual`) — restorable later by the existing restore path
and expired by the reaper's retention pass, so it never leaks as an orphan archive.
The backup runs asynchronously as a one-shot Kubernetes Job (the new
`internal/backupjob` package), mirroring how restore and image builds hand off to
Jobs. The handler answers **202 `backing_up`**.
## Why
felis-api cannot archive a world in-process: the world PVC is **RWO** and owned by the
operator's StatefulSet, so the API has nothing to mount at request time — the same
constraint that already makes `internal/restore` a Job. The break-glass console (which
runs direct-to-Postgres) likewise lacks the deployment coordinates (`FELIS_IMAGE`,
`FELIS_BACKUP_PVC`) needed to render the Job. Both point to the same home: the
orchestration belongs in felis-api, which holds those coordinates; other callers
invoke the endpoint.
## Design decisions
- **Backup Job self-records its `world_backups` row.** Unlike the restore Job — which
is deliberately DB-blind because it processes a potentially poisoned archive — the
backup Job **does** mount the felis config Secret and inserts its own backup row,
exactly like the reaper (the only other component holding both a world mount and the
database). This avoids the archive-then-async-record split that would otherwise leak
orphan archives on a crash. The security review for that one departure lives in
`internal/backupjob/jobspec.go` and is frozen by `jobspec_test.go`. Rationale: a
backup only **reads** a world the operator already owns and tars it (bytes, never
executed), so restore's poisoned-input threat does not apply; its blast radius (DB +
two PVCs) is a strict subset of the reaper's, and it never deletes a PVC nor calls
the K8s API (SA token stays un-mounted).
- **World mounted read-only, backup PVC read-write** — the mirror image of restore.
- **Stopped-gate (409 `not_stopped`).** The world PVC is RWO and held by a running
server, so a backup Job cannot double-mount it; the handler refuses unless the server
is fully stopped (`info.Ready || DesiredState != Stopped`). This also guarantees a
quiescent, non-torn archive. Mirrors `handleRestoreBackup`'s gate.
- **Authorization is restore's front half, minus the former-owner match.** Backup is
initiated by the **current** owner and records **their** ownership, so there is no
prior owner's data to leak — the leak guard that restore needs does not apply here.
An admin may back up an unowned (released) world; the recorded former owner is then
empty, exactly as the reaper records for an unowned reap.
- **Unique Job name per request.** Each backup Job is named `backup-<server>-<rand>`,
not a deterministic `backup-<server>`. A deterministic name would collide with a
just-finished Job still inside its `TTLSecondsAfterFinished` window (10m), and the
`AlreadyExists → 202` path would then silently produce **no** archive — the exact
window a user (or the console "立即备份" button) retries in. Unique names make every
request produce its own archive; `ErrAlreadyExists` remains only as a defensive
no-op on the ~impossible suffix collision. Ceiling (documented in `backup.go`): two
truly simultaneous taps may schedule two backup Pods — both mount the world PVC
read-only, so neither corrupts anything; single-flight-on-running is the upgrade
path if a double-tap storm ever appears.
- **One retention clock.** The entrypoint reuses the reaper's `reaperConfig` derivation
so a manual backup expires on the same schedule as an inactivity backup — one policy,
not two. The `"bk-"+hex` id scheme also matches, so manual and inactivity backups are
indistinguishable downstream.
- **Fail-safe on record failure.** If the row insert fails, the entrypoint deletes the
just-written archive so a failed backup leaves no unrecorded bytes.
- **Optional executor, honest 503.** Wired only when `FELIS_IMAGE` + `FELIS_BACKUP_PVC`
are supplied (same gate as restore); otherwise `API.Backuper` is nil and the endpoint
returns 503 `backup_unavailable`, so the authorization boundary is exercised before
the Job executor is deployable.
## Files
| File | Change |
|---|---|
| `internal/backupjob/jobspec.go` | **new** — `BackupJob` renderer + `BackupJobName`; weak SA, token off, hardened container, world RO / backup RW, config-Secret mount |
| `internal/backupjob/backup.go` | **new** — `Backuper` (idempotent enqueue) + `Config`/`withDefaults` |
| `internal/backupjob/k8sjobs.go` | **new** — controller-runtime `CreateBackupJob` (AlreadyExists → idempotent) |
| `internal/backupjob/jobspec_test.go` | **new** — freezes the Job's security shape incl. the deliberate config-Secret mount |
| `internal/backupjob/backup_test.go` | **new** — asserts each `Backup` call mints a unique Job name (repeat-tap must not silently no-op) |
| `cmd/felis/backup.go` | **new** — `felis backup` in-Pod entrypoint: archive + self-record + orphan-cleanup |
| `cmd/felis/run.go` | dispatch `case "backup"` + usage line |
| `internal/api/backuper.go` | **new** — the narrow `Backuper` port |
| `internal/api/handlers_backups.go` | **+`handleBackupNow`** |
| `internal/api/api.go` | `Backuper` field + `POST /servers/{name}/backup` route |
| `internal/api/handlers_backup_now_test.go` | **new** — `fakeBackuper` + handler subtests |
| `internal/api/backuper_wire_test.go` | **new** — compile-time `Backuper = (*backupjob.Backuper)(nil)` |
| `cmd/felis/api.go` | wire `backuper` under the `FELIS_IMAGE`+`FELIS_BACKUP_PVC` gate; `backupConfig` helper |
| `docs/openapi.yaml` | document the `backupNow` operation |
## Verification
WSL oracle (go1.26.4, authoritative for Go):
```
go build ./... && go vet ./... && go test ./... → ALL GREEN
```
The `internal/api` OpenAPI served-route contract test (`TestOpenAPIMatchesServedRoutes`)
initially failed — the new route was served but undocumented — and passes after adding
the `backupNow` operation to `docs/openapi.yaml`. `internal/backupjob` and the new
handler subtests pass. The controller-runtime `K8sJobs` binding is integration-only
(needs a live cluster) and is exercised only by the interface conformance test.
## Self-review outcome
- **ponytail (over-engineering):** the backup Job is a near-mirror of the restore Job,
not a shared parameterization — deliberate, because its security shape differs (it
holds DB creds) and must be asserted independently, not hidden behind a shared knob.
No speculative config; `Config.withDefaults` fills only real deployment values.
- **correctness:** the RWO stopped-gate and the self-recording atomicity were traced to
the reaper and restore before writing; the former-owner asymmetry vs restore is
justified above.
+3 -1
View File
@@ -23,7 +23,9 @@ not yet committed.
## Pending (built + verified, not yet committed)
_None — the break-glass halt (`c2ee21a`) and `/felis migrate` (`c1aa38b`) landed in the ledger below._
| Change | Detail doc | Status |
|---|---|---|
| On-demand world backup — felis-api `POST /servers/{name}/backup` + backup-Job executor (§B4 "Sync", half 1 of 2) | [2026-07-07-on-demand-world-backup.md](2026-07-07-on-demand-world-backup.md) | oracle-green; commit `pending` |
## Committed change ledger
+44
View File
@@ -2584,6 +2584,50 @@ paths:
'503':
$ref: '#/components/responses/ServiceUnavailable'
/api/v1/servers/{name}/backup:
post:
tags: [backups]
operationId: backupNow
summary: Back up a server's world on demand (owner-or-admin; server must be stopped).
description: >-
Snapshots the server's world into the archive store as a first-class
world_backups row (reason "manual"), restorable later like an inactivity
backup. The world PVC is RWO and held by a running server, so the server must
be fully stopped first (409 not_stopped otherwise). The backup runs
asynchronously as a Job, so success is 202 (backing_up).
x-felis-face: [external]
x-felis-tier: app
security: [{ accessJWT: [] }]
parameters:
- { name: name, in: path, required: true, schema: { type: string } }
responses:
'202':
description: Backup started.
content:
application/json:
schema:
type: object
required: [name, status]
properties:
name: { type: string }
status: { type: string, const: backing_up }
'401':
$ref: '#/components/responses/Unauthorized'
'403':
$ref: '#/components/responses/Forbidden'
'404':
description: Unknown server.
content:
application/json:
schema: { $ref: '#/components/schemas/Error' }
'409':
description: Server is not stopped (its world PVC is still mounted).
content:
application/json:
schema: { $ref: '#/components/schemas/Error' }
'503':
$ref: '#/components/responses/ServiceUnavailable'
# ------------------------------------------------------ users (admin tier) ----
/api/v1/users:
get:
+6
View File
@@ -58,6 +58,11 @@ type API struct {
// authorization paths do not need it (they read the Repo), only the kick-off.
Restorer Restorer
// Backuper starts an on-demand world backup (spec §18/§19 WorldArchiver). Like
// Restorer it is optional: when nil the backup route reports 503, so the backup
// authorization boundary is exercised before the backup-Job executor is wired.
Backuper Backuper
// Submissions is the user-modpack approval lane (a user-directed extension over
// the §16 build subsystem; see internal/submit). It is optional: when
// nil the /me/submissions and /submissions routes report 503 rather than 404, so
@@ -346,6 +351,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
// neither sits behind adminOnly.
{Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups},
{Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup},
{Method: "POST", Pattern: "/api/v1/servers/{name}/backup", h: a.handleBackupNow},
// Account linking (spec §10), web side: /start reports link status (it is the
// pointer handleClaim's 412 emits), /verify consumes the in-game code and binds
// the account. App-tier, not admin — linking your own account is an ordinary
+25
View File
@@ -0,0 +1,25 @@
package api
import "context"
// Backuper starts an on-demand world backup (spec §18/§19 WorldArchiver, run on
// demand rather than on the reaper's daily schedule) — the "back up before I touch
// it" lever behind POST /api/v1/servers/{name}/backup. Like Restorer it only STARTS
// the work: tarring the world PVC into the archive store is pod-filesystem work
// felis-api cannot do in-process — the world PVC is RWO and owned by the operator's
// StatefulSet, so the API has nothing to mount at request time. The production
// implementation therefore hands off to a backup Job (internal/backupjob), which
// self-records the world_backups row like the reaper. The call returns once the
// backup is enqueued, so the handler answers 202 (backing_up).
//
// formerOwner is the current owner recorded on the backup row so it can later be
// restored (empty when an admin backs up an unowned server). It returns ErrNotFound
// if the server is unknown to the execution backend; any other error maps to 500.
//
// It is an interface so the handler is tested against a fake; the production executor
// (internal/backupjob.Backuper) is integration-only, and until it is wired the
// API.Backuper is nil so POST /servers/{name}/backup reports 503 — the backup
// authorization boundary is exercised without shipping a stub that cannot run.
type Backuper interface {
Backup(ctx context.Context, serverName, formerOwner string) error
}
+12
View File
@@ -0,0 +1,12 @@
package api
import (
"felis.lolicon.best/internal/backupjob"
)
// Compile-time proof that the production backup executor satisfies the API's
// Backuper interface, mirroring restore_wire_test. The dependency points one way
// only: api defines the narrow Backuper port and never imports backupjob in
// production (handlers_backups.go depends on the interface); this test is the single
// place the concrete type and the port are pinned together.
var _ Backuper = (*backupjob.Backuper)(nil)
+178
View File
@@ -0,0 +1,178 @@
package api
import (
"context"
"encoding/json"
"errors"
"net/http"
"testing"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
)
// fakeBackuper records the on-demand backup kick-off the handler makes. It mirrors
// fakeRestorer: the handler only STARTS the work, so the fake just captures the
// arguments and returns a canned error.
type fakeBackuper struct {
err error
calls int
gotName string
gotFormerOwn string
}
func (f *fakeBackuper) Backup(_ context.Context, name, formerOwner string) error {
f.calls++
f.gotName, f.gotFormerOwn = name, formerOwner
return f.err
}
// TestBackupNow exercises POST /api/v1/servers/{name}/backup (spec §18/§19 WorldArchiver,
// on demand). The default server is stopped and owned by owner1 with a wired
// fakeBackuper. The focus is the owner-or-admin gate, the stopped gate (the RWO world
// PVC must be free), the 404/503 surface, that the recorded former owner is the
// server's CURRENT owner, and that backup is an async 202 kick-off.
func TestBackupNow(t *testing.T) {
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
mk := func() (*API, *fakeRepo, *fakeCluster, *fakeBackuper) {
repo := newFakeRepo()
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
cl := newFakeCluster()
cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped",
Ready: false, DesiredState: string(v1alpha1.DesiredStopped)}
backuper := &fakeBackuper{}
api := newTestAPI(repo, cl)
api.Backuper = backuper
return api, repo, cl, backuper
}
const path = "/api/v1/servers/survival/backup"
t.Run("owner backs up -> 202 + backuper(currentOwner) + audit", func(t *testing.T) {
api, repo, _, backuper := mk()
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusAccepted {
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
}
var resp struct {
Name string `json:"name"`
Status string `json:"status"`
}
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("body not JSON: %v (%s)", err, w.Body.String())
}
if resp.Name != "survival" || resp.Status != "backing_up" {
t.Fatalf("unexpected response %+v", resp)
}
if backuper.calls != 1 || backuper.gotName != "survival" || backuper.gotFormerOwn != "owner1" {
t.Fatalf("backuper saw (calls=%d,%q,%q), want (1,survival,owner1)",
backuper.calls, backuper.gotName, backuper.gotFormerOwn)
}
if len(repo.audits) != 1 || repo.audits[0].Action != "backup.create" || repo.audits[0].Actor != "[email protected]" {
t.Fatalf("audit not written as expected: %+v", repo.audits)
}
})
t.Run("admin backs up an unowned server -> 202, empty former owner recorded", func(t *testing.T) {
api, repo, _, backuper := mk()
repo.byName["survival"].OwnerID = "" // released world
api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "[email protected]",
Role: "admin", ViaAdminAccess: true}}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusAccepted {
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
}
if backuper.calls != 1 || backuper.gotFormerOwn != "" {
t.Fatalf("backuper saw (calls=%d, former=%q), want (1, empty)", backuper.calls, backuper.gotFormerOwn)
}
})
t.Run("non-owner -> 403, no backup", func(t *testing.T) {
api, _, _, backuper := mk()
api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusForbidden {
t.Fatalf("code = %d, want 403", w.Code)
}
if backuper.calls != 0 {
t.Fatal("a forbidden caller must not start a backup")
}
})
t.Run("unknown server -> 404", func(t *testing.T) {
api, _, _, _ := mk()
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/missing/backup", "", nil)
if w.Code != http.StatusNotFound {
t.Fatalf("code = %d, want 404", w.Code)
}
})
t.Run("running server -> 409 not_stopped, no backup", func(t *testing.T) {
api, _, cl, backuper := mk()
cl.byName["survival"].Ready = true
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" {
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
}
if backuper.calls != 0 {
t.Fatal("a running server holds the RWO world PVC — backup must be refused")
}
})
t.Run("starting server -> 409 not_stopped", func(t *testing.T) {
api, _, cl, _ := mk()
cl.byName["survival"].Ready = false
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" {
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
}
})
t.Run("nil Backuper -> 503 backup_unavailable", func(t *testing.T) {
api, _, _, _ := mk()
api.Backuper = nil
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "backup_unavailable" {
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
}
})
t.Run("backuper failure -> 500, not audited", func(t *testing.T) {
api, repo, _, backuper := mk()
backuper.err = errors.New("kick-off failed")
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusInternalServerError {
t.Fatalf("code = %d, want 500 (%s)", w.Code, w.Body.String())
}
if len(repo.audits) != 0 {
t.Fatalf("a failed backup must not be audited: %+v", repo.audits)
}
})
t.Run("backuper reports unknown server -> 404", func(t *testing.T) {
api, _, _, backuper := mk()
backuper.err = ErrNotFound
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusNotFound {
t.Fatalf("code = %d, want 404", w.Code)
}
})
t.Run("invalid server name -> 400", func(t *testing.T) {
api, _, _, _ := mk()
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/X/backup", "", nil)
if w.Code != http.StatusBadRequest {
t.Fatalf("code = %d, want 400", w.Code)
}
})
}
+74
View File
@@ -176,3 +176,77 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
"backup_id": backup.ID,
})
}
// handleBackupNow starts an on-demand backup of a server's world (POST
// /api/v1/servers/{name}/backup; spec §18/§19 WorldArchiver, run on demand). It is
// the "back up before I touch it" lever the break-glass console and the owner both
// reach for. Authorization mirrors handleRestoreBackup's front half — the shared
// "who may act on this server's world" gate — but stops short of restore's backup
// resolution and former-owner-match, because a backup is initiated by the CURRENT
// owner and records their ownership; there is no prior owner's data to leak:
//
// ① name validation
// ② ServerByName — an unknown server is 404
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
// released world can still be snapshotted by an operator before disposal)
// ④ stopped gate: the world PVC is RWO and held by a running server, so a backup
// Job cannot double-mount it — refuse unless the server is fully stopped. This
// also guarantees a quiescent, non-torn archive.
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
// means "enqueued" and the handler answers 202.
//
// The former owner recorded on the backup is the server's current OwnerID (empty for
// an unowned server backed up by an admin), so the resulting world_backups row is
// restorable by that owner exactly like an inactivity backup.
func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
// Stopped gate: the world PVC is RWO and held by a running server, so a backup
// Job cannot double-mount it (mirrors the restore gate). Ready means it is up;
// any desiredState other than Stopped means it owns the RWO volume.
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
writeError(w, r, newError(http.StatusConflict, "not_stopped",
"stop the server before backing up its world"))
return
}
// Backuper is optional: when unwired the endpoint reports 503 rather than
// panicking, so the authorization boundary above is exercised even before the
// backup-Job executor is wired (see Backuper).
if a.Backuper == nil {
writeError(w, r, newError(http.StatusServiceUnavailable, "backup_unavailable",
"backup subsystem is not configured"))
return
}
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
a.writeLookupError(w, r, err)
return
}
a.audit(r, p.Email, "backup.create", name)
writeJSON(w, http.StatusAccepted, map[string]any{
"name": name,
"status": "backing_up",
})
}
+235
View File
@@ -0,0 +1,235 @@
// Package backupjob implements the on-demand world-backup executor (spec §18/§19
// WorldArchiver, run on demand rather than on the reaper's daily schedule). It is
// the "back up before I touch it" lever behind POST /api/v1/servers/{name}/backup:
// an owner or admin stops a server, then snapshots its world into the archive store
// as a first-class world_backups row — restorable by internal/restore and expired
// by the reaper's retention pass, so it never leaks as an orphan archive.
//
// felis-api cannot archive a world in-process: the world PVC is RWO and owned by
// the operator's StatefulSet, so the API has nothing to mount at request time (the
// same constraint that makes internal/restore a Job). This package is the executor
// it hands off to — a one-shot Kubernetes Job in the minecraft namespace that
// mounts the target world PVC (read-only) and the backup PVC (read-write), then
// runs `felis backup` (cmd/felis) to tar the world into the archive store AND
// record the world_backups row.
//
// Trust model. The backup Pod mirrors internal/restore's weak-SA isolation (a weak
// SA with its token un-mounted, so it cannot reach the K8s API) with ONE deliberate
// departure, reviewed in jobspec.go: it DOES mount the felis config Secret so it can
// self-record its backup row atomically with the archive, exactly like the reaper —
// the only other component holding both a world mount and the database. A restore
// Pod must stay DB-blind because it processes a poisoned archive; a backup Pod only
// reads a world the operator already owns and tars it, so that threat does not apply.
//
// The Backuper depends on the Jobs interface, so the orchestration (idempotent
// enqueue, error mapping) is unit-tested against an in-memory fake; the
// controller-runtime implementation (k8sjobs.go) compiles here but is exercised only
// by integration tests against a live cluster.
package backupjob
import (
"context"
"crypto/rand"
"encoding/hex"
"errors"
"time"
"felis.lolicon.best/internal/naming"
)
// ErrAlreadyExists is returned by a Jobs implementation when a backup Job for a
// server already exists (a backup is already in flight). The Backuper treats it as
// success — see Backup.
var ErrAlreadyExists = errors.New("backup: job already exists")
// Jobs is the cluster-side backup lifecycle the Backuper depends on. It is an
// interface so the orchestration is tested against a fake; the controller-runtime
// implementation (K8sJobs) is integration-tested only — it requires a live cluster.
type Jobs interface {
// CreateBackupJob renders and applies the backup Job for p. It returns
// ErrAlreadyExists if a Job of the same (deterministic) name already exists.
CreateBackupJob(ctx context.Context, p JobParams) error
}
// Config parameterises the backup executor. Deployment-specific values that have no
// safe default — the felis Image to run and the BackupPVC to mount — are supplied
// by the caller (cmd/felis sources them from the environment); when either is empty
// the caller leaves the API's Backuper nil so the endpoint reports 503 rather than
// enqueuing a Job that cannot run.
type Config struct {
// Namespace is where the world PVCs live and the backup Job runs (the minecraft
// namespace), co-located with the world it snapshots.
Namespace string
// ServiceAccount is the weak SA the backup Pod runs as. It reuses felis-restore
// (bare, no Role/RoleBinding): a backup Pod needs no K8s API access, only
// filesystem access to the two PVCs and — via the mounted config Secret, not the
// SA — the database.
ServiceAccount string
// Image is the felis binary image; the Job runs `felis backup` from it.
Image string
// BackupPVC is the name of the backup PVC the archive is written into (the same
// PVC the reaper writes to and restore reads from).
BackupPVC string
// ConfigSecret is the felis config Secret (felis.toml, carrying the DB URL) the
// backup Pod mounts to self-record its world_backups row. Defaults to the name
// the control-plane manifests use.
ConfigSecret string
// ConfigMount is the in-Pod mount path of ConfigSecret (holds felis.toml).
ConfigMount string
// BackupRoot is the in-Pod mount path of BackupPVC. It MUST equal cfg.Archive.
// LocalPath — the path the reaper wrote archives under and restore mounts to
// resolve them — because tarLocal archive refs are absolute.
BackupRoot string
// WorldsRoot is the in-Pod mount path of the world PVC being archived.
WorldsRoot string
// Deadline caps the backup Pod's wall-clock (activeDeadlineSeconds).
Deadline time.Duration
// CPULimit / MemLimit cap the backup container.
CPULimit string
MemLimit string
// RunAsUser / RunAsGroup / FSGroup are the Pod's runtime identity. FSGroup MUST
// match the operator StatefulSet's runtime group so the read-only world mount is
// readable by this Pod's uid.
RunAsUser int64
RunAsGroup int64
FSGroup int64
// TTLAfterFinished is how long a finished backup Job lingers before the Job
// controller garbage-collects it; it also bounds the window in which a re-backup
// sees a stale completed Job as ErrAlreadyExists.
TTLAfterFinished time.Duration
}
// defaults applied when a Config field is left zero. Image and BackupPVC have no
// default on purpose — see Config.
const (
defaultNamespace = "minecraft"
defaultServiceAccount = "felis-restore"
defaultConfigSecret = "felis-config"
defaultConfigMount = "/etc/felis"
defaultBackupRoot = "/backups"
defaultWorldsRoot = "/world"
defaultDeadline = 30 * time.Minute
defaultCPULimit = "1"
defaultMemLimit = "1Gi"
defaultRunAsID = int64(1000)
defaultTTL = 10 * time.Minute
)
// withDefaults returns a copy of c with zero fields filled, so a partially
// configured Config (or the zero value, in tests) is always usable.
func (c Config) withDefaults() Config {
if c.Namespace == "" {
c.Namespace = defaultNamespace
}
if c.ServiceAccount == "" {
c.ServiceAccount = defaultServiceAccount
}
if c.ConfigSecret == "" {
c.ConfigSecret = defaultConfigSecret
}
if c.ConfigMount == "" {
c.ConfigMount = defaultConfigMount
}
if c.BackupRoot == "" {
c.BackupRoot = defaultBackupRoot
}
if c.WorldsRoot == "" {
c.WorldsRoot = defaultWorldsRoot
}
if c.Deadline <= 0 {
c.Deadline = defaultDeadline
}
if c.CPULimit == "" {
c.CPULimit = defaultCPULimit
}
if c.MemLimit == "" {
c.MemLimit = defaultMemLimit
}
if c.RunAsUser == 0 {
c.RunAsUser = defaultRunAsID
}
if c.RunAsGroup == 0 {
c.RunAsGroup = defaultRunAsID
}
if c.FSGroup == 0 {
c.FSGroup = defaultRunAsID
}
if c.TTLAfterFinished <= 0 {
c.TTLAfterFinished = defaultTTL
}
return c
}
// Backuper is the production internal/api.Backuper (the compile-time proof of that
// is in internal/api's wire test, which imports this package; this package never
// imports api). It holds no mutable state.
type Backuper struct {
Jobs Jobs
Config Config
}
// Backup enqueues a backup Job that tars serverName's world into the archive store
// and records the world_backups row. formerOwner is stamped on that row so the owner
// can later restore it (empty for an admin backing up an unowned server). It returns
// once the Job is created — the archive runs in the Pod — so the handler's 202
// ("backing_up") is honest.
//
// Each call gets a fresh, unique Job name (BackupJobName + random suffix), so an
// on-demand backup requested again after a previous one — the console "立即备份"
// repeat-tap case — always produces a new archive rather than colliding with a
// just-finished Job still inside its TTL window. ErrAlreadyExists is kept only as a
// defensive no-op against the astronomically unlikely suffix collision.
//
// ponytail: unique names mean two truly simultaneous taps can schedule two backup
// Pods; both mount the world PVC read-only so neither corrupts anything, and if they
// land on different nodes the RWO attach fails one cleanly. Add single-flight-on-
// running only if a real double-tap storm ever shows up.
func (b *Backuper) Backup(ctx context.Context, serverName, formerOwner string) error {
if err := b.Jobs.CreateBackupJob(ctx, b.jobParams(serverName, formerOwner)); err != nil {
if errors.Is(err, ErrAlreadyExists) {
return nil // suffix collision — treat as enqueued
}
return err
}
return nil
}
// jobNameSuffix is a short random hex tag that makes each backup Job name unique.
// 32 bits is ample: collisions only matter within a single Job's TTL window across
// a handful of manual backups.
func jobNameSuffix() string {
var b [4]byte
if _, err := rand.Read(b[:]); err != nil {
// ponytail: crypto/rand only fails if the OS RNG is gone — unrecoverable.
panic("backupjob: crypto/rand: " + err.Error())
}
return hex.EncodeToString(b[:])
}
// jobParams projects the server, former owner, and config onto the inputs jobspec.go
// renders. The world PVC name is derived from the single shared naming convention
// (naming.WorldPVCName), the same one the operator created it under.
func (b *Backuper) jobParams(serverName, formerOwner string) JobParams {
cfg := b.Config.withDefaults()
return JobParams{
Server: serverName,
JobName: BackupJobName(serverName) + "-" + jobNameSuffix(),
FormerOwner: formerOwner,
WorldPVC: naming.WorldPVCName(serverName),
BackupPVC: cfg.BackupPVC,
Namespace: cfg.Namespace,
ServiceAccount: cfg.ServiceAccount,
Image: cfg.Image,
ConfigSecret: cfg.ConfigSecret,
ConfigMount: cfg.ConfigMount,
BackupRoot: cfg.BackupRoot,
WorldsRoot: cfg.WorldsRoot,
Deadline: cfg.Deadline,
CPULimit: cfg.CPULimit,
MemLimit: cfg.MemLimit,
RunAsUser: cfg.RunAsUser,
RunAsGroup: cfg.RunAsGroup,
FSGroup: cfg.FSGroup,
TTLAfterFinished: cfg.TTLAfterFinished,
}
}
+43
View File
@@ -0,0 +1,43 @@
package backupjob
import (
"context"
"strings"
"testing"
)
// captureJobs records every JobParams it is handed so the test can inspect the
// names the Backuper minted.
type captureJobs struct{ got []JobParams }
func (c *captureJobs) CreateBackupJob(_ context.Context, p JobParams) error {
c.got = append(c.got, p)
return nil
}
// The core fix: a repeat on-demand backup of the same server must not collide with
// the previous Job's name (which would silently no-op the retry within the TTL
// window). Two Backup calls must mint two distinct Job names, both under the
// server's base prefix.
func TestBackupMintsUniqueJobNamePerCall(t *testing.T) {
jobs := &captureJobs{}
b := &Backuper{Jobs: jobs, Config: Config{Image: "img", BackupPVC: "pvc"}}
for i := 0; i < 2; i++ {
if err := b.Backup(context.Background(), "survival", "usr-1"); err != nil {
t.Fatalf("Backup: %v", err)
}
}
if len(jobs.got) != 2 {
t.Fatalf("created %d jobs, want 2", len(jobs.got))
}
base := BackupJobName("survival")
for _, p := range jobs.got {
if !strings.HasPrefix(p.JobName, base+"-") {
t.Errorf("JobName %q must extend the base %q", p.JobName, base)
}
}
if jobs.got[0].JobName == jobs.got[1].JobName {
t.Errorf("two backups reused the Job name %q — retry would silently no-op", jobs.got[0].JobName)
}
}
+242
View File
@@ -0,0 +1,242 @@
package backupjob
import (
"fmt"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
// Label keys applied to backup objects, mirroring internal/restore so the two
// executors are observable the same way.
const (
LabelManagedBy = "app.kubernetes.io/managed-by"
LabelComponent = "app.kubernetes.io/component"
LabelServer = "felis.lolicon.best/server"
managedByValue = "felis-backup"
componentValue = "world-backup"
worldVolume = "world"
backupVolume = "backup"
configVolume = "config"
tmpVolume = "tmp"
felisBinaryPath = "/usr/local/bin/felis"
)
// JobParams are the rendered inputs to a backup Job, derived from a server +
// former owner + Config by the Backuper. jobspec is a pure function of them so
// the security-critical Job shape is unit-tested without a cluster.
type JobParams struct {
Server string
// JobName is the unique Job object name for this invocation. The Backuper sets a
// fresh per-call name (BackupJobName(server) + random suffix) so every on-demand
// request produces its own archive — a deterministic name would collide with a
// just-finished Job still inside its TTL window and silently no-op the retry.
// Empty falls back to the deterministic BackupJobName (tests, and any direct call).
JobName string
FormerOwner string // recorded on the world_backups row so the owner can later restore it
WorldPVC string
BackupPVC string
Namespace string
ServiceAccount string
Image string
ConfigSecret string // the felis config Secret mounted for the DB URL (self-recording)
ConfigMount string // where the config Secret is mounted (holds felis.toml)
BackupRoot string
WorldsRoot string
Deadline time.Duration
CPULimit string
MemLimit string
RunAsUser int64
RunAsGroup int64
FSGroup int64
TTLAfterFinished time.Duration
}
// BackupJobName is the base Job name for a server's on-demand backup — the prefix
// the Backuper extends with a per-invocation random suffix (JobParams.JobName) so
// repeat backups do not collide. It is distinct from the restore Job name so a
// backup and a restore of the same server never collide.
func BackupJobName(server string) string { return "backup-" + server }
func backupLabels(p JobParams) map[string]string {
return map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
LabelServer: p.Server,
}
}
// BackupJob renders the on-demand world-backup Job (spec §18/§19 WorldArchiver,
// run on demand rather than on the reaper's daily schedule). Its isolation mirrors
// the restore Job (weak SA, non-root, read-only root fs, drop ALL, one-shot with a
// deadline) with ONE deliberate, security-reviewed departure asserted by
// jobspec_test.go:
//
// - It DOES mount the felis config Secret (read-only), because a backup must
// self-record its world_backups row — archive-then-insert atomically, exactly
// like the reaper, which is the only other component holding both a world mount
// and the database. A restore Pod must never touch the DB because it processes a
// potentially poisoned archive; a backup Pod only READS a world the operator
// already owns and tars it (bytes, never executed), so the poisoned-input threat
// that forbids restore's DB access does not apply. Its blast radius is the DB
// plus the two PVCs, matching the reaper's trust for a strict subset of the
// reaper's operations (it never deletes a PVC and never calls the K8s API — its
// SA token stays un-mounted).
// - The world PVC is mounted READ-ONLY (backup only reads it; the RWO volume must
// be free, which the handler's stopped-gate guarantees), and the backup PVC
// READ-WRITE (the archive is written into it) — the mirror image of restore.
//
// The container runs `/usr/local/bin/felis backup` (cmd/felis), which tars the
// world at WorldsRoot into BackupRoot and inserts the world_backups row. BackupRoot
// MUST equal the reaper's [archive] local_path so the recorded absolute ref
// resolves the same way a later restore Job mounts it.
func BackupJob(p JobParams) (*batchv1.Job, error) {
if p.Image == "" {
return nil, fmt.Errorf("backup: image is empty")
}
if p.WorldPVC == "" || p.BackupPVC == "" {
return nil, fmt.Errorf("backup: world and backup PVC names are required")
}
if p.ConfigSecret == "" {
return nil, fmt.Errorf("backup: config secret name is required")
}
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
if err != nil {
return nil, err
}
deadline := int64(p.Deadline / time.Second)
if deadline <= 0 {
deadline = int64(defaultDeadline / time.Second)
}
ttl := int32(p.TTLAfterFinished / time.Second)
if ttl <= 0 {
ttl = int32(defaultTTL / time.Second)
}
args := []string{
"--config", p.ConfigMount + "/felis.toml",
"--server", p.Server,
"--worlds-root", p.WorldsRoot,
}
// FormerOwner is optional: an admin backing up an unowned (released) server
// records an empty former_owner, exactly as the reaper does for an unowned reap.
if p.FormerOwner != "" {
args = append(args, "--former-owner", p.FormerOwner)
}
container := corev1.Container{
Name: "backup",
Image: p.Image,
Command: []string{felisBinaryPath, "backup"},
Args: args,
VolumeMounts: []corev1.VolumeMount{
// The world is only read to tar it; mounting it read-only means the
// backup process can never mutate the world it snapshots.
{Name: worldVolume, MountPath: p.WorldsRoot, ReadOnly: true},
{Name: backupVolume, MountPath: p.BackupRoot},
{Name: configVolume, MountPath: p.ConfigMount, ReadOnly: true},
{Name: tmpVolume, MountPath: "/tmp"},
},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
ReadOnlyRootFilesystem: boolPtr(true),
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
},
}
name := p.JobName
if name == "" {
name = BackupJobName(p.Server)
}
job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{
Name: name,
Namespace: p.Namespace,
Labels: backupLabels(p),
},
Spec: batchv1.JobSpec{
// One shot: a wedged archive must not loop. The TTL GCs the finished Job
// so a later backup of the same server is not blocked forever by a stale
// completed Job.
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(deadline),
TTLSecondsAfterFinished: int32Ptr(ttl),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: backupLabels(p)},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
SecurityContext: &corev1.PodSecurityContext{
RunAsNonRoot: boolPtr(true),
RunAsUser: int64Ptr(p.RunAsUser),
RunAsGroup: int64Ptr(p.RunAsGroup),
FSGroup: int64Ptr(p.FSGroup),
},
Containers: []corev1.Container{container},
Volumes: []corev1.Volume{
{
Name: worldVolume,
VolumeSource: corev1.VolumeSource{
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
ClaimName: p.WorldPVC,
ReadOnly: true,
},
},
},
{
Name: backupVolume,
VolumeSource: corev1.VolumeSource{
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
ClaimName: p.BackupPVC,
},
},
},
{
Name: configVolume,
VolumeSource: corev1.VolumeSource{
Secret: &corev1.SecretVolumeSource{SecretName: p.ConfigSecret},
},
},
{Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}},
},
},
},
},
}
return job, nil
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
cpu = defaultCPULimit
}
if mem == "" {
mem = defaultMemLimit
}
cpuQty, err := resource.ParseQuantity(cpu)
if err != nil {
return nil, fmt.Errorf("backup: invalid cpu limit %q: %w", cpu, err)
}
memQty, err := resource.ParseQuantity(mem)
if err != nil {
return nil, fmt.Errorf("backup: invalid memory limit %q: %w", mem, err)
}
return corev1.ResourceList{
corev1.ResourceCPU: cpuQty,
corev1.ResourceMemory: memQty,
}, nil
}
func boolPtr(b bool) *bool { return &b }
func int32Ptr(i int32) *int32 { return &i }
func int64Ptr(i int64) *int64 { return &i }
+220
View File
@@ -0,0 +1,220 @@
package backupjob
import (
"testing"
"time"
corev1 "k8s.io/api/core/v1"
)
func sampleJobParams() JobParams {
return JobParams{
Server: "survival",
FormerOwner: "usr-abc",
WorldPVC: "world-survival-0",
BackupPVC: "felis-backups",
Namespace: defaultNamespace,
ServiceAccount: defaultServiceAccount,
Image: "registry.felis.svc:5000/felis:1.0",
ConfigSecret: defaultConfigSecret,
ConfigMount: defaultConfigMount,
BackupRoot: "/backups",
WorldsRoot: "/world",
Deadline: 30 * time.Minute,
CPULimit: "1",
MemLimit: "1Gi",
RunAsUser: 1000,
RunAsGroup: 1000,
FSGroup: 1000,
TTLAfterFinished: 10 * time.Minute,
}
}
// The backup Pod runs under the weak felis-restore SA — never the felis-api
// identity — with its token un-mounted, so it cannot reach the K8s API. Its DB
// access comes from the mounted config Secret, not from any SA permission.
func TestBackupJobRunsUnderWeakSAWithNoAPIToken(t *testing.T) {
job, err := BackupJob(sampleJobParams())
if err != nil {
t.Fatalf("BackupJob: %v", err)
}
sa := job.Spec.Template.Spec.ServiceAccountName
if sa != defaultServiceAccount {
t.Errorf("service account = %q, want %q", sa, defaultServiceAccount)
}
if sa == "felis-api" {
t.Fatal("backup Pod must NOT run as the felis-api SA")
}
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
t.Error("AutomountServiceAccountToken must be explicitly false")
}
}
// The deliberate departure from restore's zero-secret isolation: a backup Pod
// mounts EXACTLY the two PVCs (world read-only, backup read-write) PLUS the config
// Secret read-only (so it can self-record its world_backups row like the reaper) —
// and nothing else. This test freezes that exact volume set so a future edit that
// widens it (e.g. a second Secret, or a writable world mount) fails loudly.
func TestBackupJobMountsTwoPVCsPlusConfigSecretOnly(t *testing.T) {
job, err := BackupJob(sampleJobParams())
if err != nil {
t.Fatalf("BackupJob: %v", err)
}
spec := job.Spec.Template.Spec
var secretVols, pvcVols int
for _, v := range spec.Volumes {
switch {
case v.Secret != nil:
secretVols++
if v.Secret.SecretName != defaultConfigSecret {
t.Errorf("secret volume = %q, want the config secret %q", v.Secret.SecretName, defaultConfigSecret)
}
case v.PersistentVolumeClaim != nil:
pvcVols++
case v.EmptyDir != nil:
// the /tmp scratch dir under the read-only root fs — allowed
default:
t.Errorf("unexpected volume %q: a backup Pod mounts only the two PVCs, the config Secret, and a /tmp emptyDir", v.Name)
}
}
if secretVols != 1 {
t.Errorf("secret volumes = %d, want exactly 1 (the config Secret)", secretVols)
}
if pvcVols != 2 {
t.Errorf("PVC volumes = %d, want exactly 2 (world + backup)", pvcVols)
}
// World read-only (backup never mutates the world), backup read-write (the
// archive is written into it) — the mirror image of the restore Job.
world := mountByName(t, spec.Containers[0].VolumeMounts, worldVolume)
if !world.ReadOnly {
t.Error("world mount must be read-only — a backup only reads the world")
}
back := mountByName(t, spec.Containers[0].VolumeMounts, backupVolume)
if back.ReadOnly {
t.Error("backup mount must be read-write — the archive is written into it")
}
cfg := mountByName(t, spec.Containers[0].VolumeMounts, configVolume)
if !cfg.ReadOnly {
t.Error("config Secret mount must be read-only")
}
// The world PVC volume itself is also declared read-only so the RWO claim is
// requested read-only (defense in depth beyond the mount flag).
for _, v := range spec.Volumes {
if v.PersistentVolumeClaim != nil && v.PersistentVolumeClaim.ClaimName == "world-survival-0" && !v.PersistentVolumeClaim.ReadOnly {
t.Error("world PVC volume source must be read-only")
}
}
}
// The backup container is hardened exactly like the restore/build Job containers:
// no privilege, no escalation, read-only root fs, drop ALL capabilities.
func TestBackupJobContainerIsHardened(t *testing.T) {
job, err := BackupJob(sampleJobParams())
if err != nil {
t.Fatalf("BackupJob: %v", err)
}
sc := job.Spec.Template.Spec.Containers[0].SecurityContext
if sc == nil {
t.Fatal("container SecurityContext is nil")
}
if sc.Privileged == nil || *sc.Privileged {
t.Error("Privileged must be false")
}
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
t.Error("AllowPrivilegeEscalation must be false")
}
if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
t.Error("ReadOnlyRootFilesystem must be true")
}
if sc.Capabilities == nil || len(sc.Capabilities.Drop) != 1 || sc.Capabilities.Drop[0] != "ALL" {
t.Error("capabilities must drop ALL")
}
}
// One-shot: a wedged archive must not loop, and a deadline caps it.
func TestBackupJobIsOneShotWithDeadline(t *testing.T) {
job, err := BackupJob(sampleJobParams())
if err != nil {
t.Fatalf("BackupJob: %v", err)
}
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
t.Error("BackoffLimit must be 0 (no retry loop)")
}
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds <= 0 {
t.Error("ActiveDeadlineSeconds must be set")
}
if job.Spec.TTLSecondsAfterFinished == nil {
t.Error("TTLSecondsAfterFinished must be set so the finished Job is GC'd")
}
}
// The command carries the former owner so the recorded backup can be restored by
// its owner; an empty former owner (admin backing up an unowned server) omits it.
func TestBackupJobArgsCarryServerAndOwner(t *testing.T) {
job, err := BackupJob(sampleJobParams())
if err != nil {
t.Fatalf("BackupJob: %v", err)
}
args := job.Spec.Template.Spec.Containers[0].Args
if !argsContain(args, "--server", "survival") {
t.Errorf("args missing --server survival: %v", args)
}
if !argsContain(args, "--former-owner", "usr-abc") {
t.Errorf("args missing --former-owner usr-abc: %v", args)
}
p := sampleJobParams()
p.FormerOwner = ""
unowned, err := BackupJob(p)
if err != nil {
t.Fatalf("BackupJob(unowned): %v", err)
}
for _, a := range unowned.Spec.Template.Spec.Containers[0].Args {
if a == "--former-owner" {
t.Error("--former-owner must be omitted when there is no former owner")
}
}
}
func TestBackupJobRejectsMissingInputs(t *testing.T) {
for _, tc := range []struct {
name string
mut func(*JobParams)
}{
{"no image", func(p *JobParams) { p.Image = "" }},
{"no world pvc", func(p *JobParams) { p.WorldPVC = "" }},
{"no backup pvc", func(p *JobParams) { p.BackupPVC = "" }},
{"no config secret", func(p *JobParams) { p.ConfigSecret = "" }},
} {
t.Run(tc.name, func(t *testing.T) {
p := sampleJobParams()
tc.mut(&p)
if _, err := BackupJob(p); err == nil {
t.Errorf("BackupJob(%s) = nil error, want a validation error", tc.name)
}
})
}
}
func mountByName(t *testing.T, mounts []corev1.VolumeMount, name string) corev1.VolumeMount {
t.Helper()
for _, m := range mounts {
if m.Name == name {
return m
}
}
t.Fatalf("volume mount %q not found", name)
return corev1.VolumeMount{}
}
func argsContain(args []string, flag, val string) bool {
for i := 0; i+1 < len(args); i++ {
if args[i] == flag && args[i+1] == val {
return true
}
}
return false
}
+43
View File
@@ -0,0 +1,43 @@
package backupjob
import (
"context"
apierrors "k8s.io/apimachinery/pkg/api/errors"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// K8sJobs is the production Jobs backed by a controller-runtime client. It creates
// the on-demand world-backup Job — nothing more: the backup Job is one-shot and
// self-cleaning (ttlSecondsAfterFinished), so there is no phase or cancel seam, and
// thus no config to hold. Every backup parameter arrives in the JobParams the
// Backuper builds from its own (defaulted) Config. The cluster-bootstrap objects
// (the weak felis-restore SA it reuses) are installed once by the deployment
// manifests, not per backup, so this binding never creates them. It is
// integration-tested against a live cluster, not the hermetic unit suite.
type K8sJobs struct {
c client.Client
}
// NewK8sJobs builds a Jobs over c.
func NewK8sJobs(c client.Client) *K8sJobs {
return &K8sJobs{c: c}
}
// CreateBackupJob renders and applies the backup Job. Its name is a deterministic
// function of the server (BackupJobName), so a concurrent backup of the same server
// collides on Create; that collision is mapped to ErrAlreadyExists, which the
// Backuper treats as success (idempotent enqueue).
func (k *K8sJobs) CreateBackupJob(ctx context.Context, p JobParams) error {
job, err := BackupJob(p)
if err != nil {
return err
}
if err := k.c.Create(ctx, job); err != nil {
if apierrors.IsAlreadyExists(err) {
return ErrAlreadyExists
}
return err
}
return nil
}