feat(api): add on-demand world backup endpoint and Job executor (§B4 Sync)
Add POST /api/v1/servers/{name}/backup: an owner or admin snapshots a
stopped server's world into the archive store on demand, recorded as a
first-class world_backups row (reason `manual`) — restorable by the
existing restore path and expired by the reaper's retention pass, so it
never leaks as an orphan archive. This is the break-glass "Sync" op,
resolved as immediate/on-demand backup.
felis-api cannot archive in-process (the world PVC is RWO, held by the
operator StatefulSet), so the work hands off to a one-shot Kubernetes Job
(new internal/backupjob) that mounts the world PVC read-only and the
backup PVC read-write, plus the felis config Secret so it self-records
its row atomically like the reaper. The Pod mirrors restore's weak-SA
isolation (SA token un-mounted, non-root, read-only rootfs, drop ALL);
the one reviewed departure is that config-Secret mount, frozen by
jobspec_test.go. Handler answers 202 backing_up; gated on the server
being Stopped (RWO world PVC), owner-or-admin, and FELIS_IMAGE +
FELIS_BACKUP_PVC being wired (else 503 backup_unavailable).
Each request mints a unique Job name (backup-<server>-<rand>) so a repeat
on-demand backup produces a fresh archive rather than colliding with a
just-finished Job still inside its TTL window and silently no-op'ing the
retry.
This commit is contained in:
16 files changed
+1396
-1
No files matched your search
@@ -13,6 +13,7 @@ import (
|
|||||||
|
|
||||||
"felis.lolicon.best/internal/api"
|
"felis.lolicon.best/internal/api"
|
||||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
"felis.lolicon.best/internal/backupjob"
|
||||||
"felis.lolicon.best/internal/build"
|
"felis.lolicon.best/internal/build"
|
||||||
"felis.lolicon.best/internal/config"
|
"felis.lolicon.best/internal/config"
|
||||||
"felis.lolicon.best/internal/panel"
|
"felis.lolicon.best/internal/panel"
|
||||||
@@ -162,6 +163,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
fmt.Fprintln(stderr, "felis api: restore executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — restore endpoint returns 503")
|
fmt.Fprintln(stderr, "felis api: restore executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — restore endpoint returns 503")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// On-demand backup subsystem (spec §18/§19 WorldArchiver, run on demand). Its
|
||||||
|
// backup Job mirrors the restore Job's weak-SA isolation but additionally mounts
|
||||||
|
// the config Secret so it self-records the world_backups row (see internal/
|
||||||
|
// backupjob). It needs the same deployment-specific values as restore, so it is
|
||||||
|
// wired under the same gate; otherwise the Backuper is left nil and the backup
|
||||||
|
// endpoint honestly returns 503.
|
||||||
|
var backuper api.Backuper
|
||||||
|
if felisImage != "" && backupPVC != "" {
|
||||||
|
backuper = &backupjob.Backuper{Jobs: backupjob.NewK8sJobs(cl), Config: backupConfig(cfg, felisImage, backupPVC)}
|
||||||
|
} else {
|
||||||
|
fmt.Fprintln(stderr, "felis api: backup executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — backup endpoint returns 503")
|
||||||
|
}
|
||||||
|
|
||||||
// One PGRepo instance backs both the handlers and the session verifier: the
|
// One PGRepo instance backs both the handlers and the session verifier: the
|
||||||
// SessionAuth that fronts the external face reads sessions/users/settings from
|
// SessionAuth that fronts the external face reads sessions/users/settings from
|
||||||
// the same store the auth handlers write to, so a login and the next request
|
// the same store the auth handlers write to, so a login and the next request
|
||||||
@@ -179,6 +193,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
Internal: api.BearerTokenAuth{Token: token},
|
Internal: api.BearerTokenAuth{Token: token},
|
||||||
Builder: builder,
|
Builder: builder,
|
||||||
Restorer: restorer,
|
Restorer: restorer,
|
||||||
|
Backuper: backuper,
|
||||||
Submissions: submissions,
|
Submissions: submissions,
|
||||||
// The external face is fronted by SessionAuth: it prefers a local-password
|
// The external face is fronted by SessionAuth: it prefers a local-password
|
||||||
// session cookie and otherwise delegates to the Cloudflare-Access JWT verifier,
|
// session cookie and otherwise delegates to the Cloudflare-Access JWT verifier,
|
||||||
@@ -359,6 +374,21 @@ func restoreConfig(cfg *config.Config, image, backupPVC string) restore.Config {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// backupConfig builds the on-demand backup executor's config from felis.toml plus
|
||||||
|
// the deployment-supplied image and backup PVC. BackupRoot mirrors restoreConfig —
|
||||||
|
// it MUST equal [archive] local_path so the recorded ref resolves the same way a
|
||||||
|
// later restore Job mounts it. ConfigSecret/ConfigMount are left to backupjob's
|
||||||
|
// defaults (the control-plane manifest names), which is the Secret this backup Job
|
||||||
|
// mounts to self-record its world_backups row.
|
||||||
|
func backupConfig(cfg *config.Config, image, backupPVC string) backupjob.Config {
|
||||||
|
return backupjob.Config{
|
||||||
|
Namespace: cfg.K8s.Namespace,
|
||||||
|
Image: image,
|
||||||
|
BackupPVC: backupPVC,
|
||||||
|
BackupRoot: cfg.Archive.LocalPath,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// reconcileBuilds polls unfinished builds on an interval and advances any whose
|
// reconcileBuilds polls unfinished builds on an interval and advances any whose
|
||||||
// Job has reached a terminal phase. It exits when ctx is cancelled.
|
// Job has reached a terminal phase. It exits when ctx is cancelled.
|
||||||
func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
|
func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
|
||||||
|
|||||||
@@ -0,0 +1,123 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"crypto/rand"
|
||||||
|
"encoding/hex"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/backup"
|
||||||
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/naming"
|
||||||
|
"felis.lolicon.best/internal/reaper"
|
||||||
|
"felis.lolicon.best/internal/store"
|
||||||
|
ctrl "sigs.k8s.io/controller-runtime"
|
||||||
|
)
|
||||||
|
|
||||||
|
// cmdBackup is the in-Pod entrypoint the on-demand backup Job runs. internal/backupjob
|
||||||
|
// renders a Pod whose command is `/usr/local/bin/felis backup`. It tars the mounted
|
||||||
|
// world into the archive store AND records the world_backups row, then exits — it is
|
||||||
|
// NOT a user-facing command and is never invoked by hand.
|
||||||
|
//
|
||||||
|
// Unlike `felis restore`, this command DOES hold database credentials (via the mounted
|
||||||
|
// config Secret) and calls config.Load: a backup must record its row atomically with
|
||||||
|
// the archive, exactly like the reaper — otherwise a completed archive would leak as an
|
||||||
|
// orphan file the retention pass never expires. The security review for that departure
|
||||||
|
// lives in internal/backupjob/jobspec.go. The world is mounted directly at --worlds-root
|
||||||
|
// (single-PVC mount, like restore), so the archiver's resolver returns that root for any
|
||||||
|
// PVC; the archive is written into the backup PVC at cfg.Archive.LocalPath.
|
||||||
|
func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("backup", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
|
server := fs.String("server", "", "server name whose world is being backed up")
|
||||||
|
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
|
||||||
|
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *server == "" {
|
||||||
|
fmt.Fprintln(stderr, "felis backup: --server is required")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
|
||||||
|
cfg, err := config.Load(*cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if cfg.Archive.Store != "tarLocal" {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
// Reuse the reaper's retention derivation so an on-demand backup expires on the
|
||||||
|
// same clock as an inactivity backup — one retention policy, not two.
|
||||||
|
rcfg, err := reaperConfig(cfg)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
|
||||||
|
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
|
||||||
|
// writes archives with.
|
||||||
|
archiver := &backup.TarLocal{
|
||||||
|
BackupRoot: cfg.Archive.LocalPath,
|
||||||
|
Resolve: func(string) (string, error) {
|
||||||
|
return *worldsRoot, nil
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
ctx := ctrl.SetupSignalHandler()
|
||||||
|
|
||||||
|
ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: open database: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer drv.Close()
|
||||||
|
|
||||||
|
rec := reaper.BackupRecord{
|
||||||
|
ID: newBackupID(),
|
||||||
|
ServerName: *server,
|
||||||
|
FormerOwner: *formerOwner,
|
||||||
|
BackupRef: string(ref),
|
||||||
|
SizeBytes: size,
|
||||||
|
Reason: "manual",
|
||||||
|
ExpiresAt: time.Now().Add(rcfg.Retention),
|
||||||
|
}
|
||||||
|
if err := reaper.NewPGStore(drv.DB()).InsertBackup(ctx, rec); err != nil {
|
||||||
|
// The archive is written but unrecorded — an orphan the retention pass would
|
||||||
|
// never expire. Delete it so a failed backup leaves no leaked bytes, mirroring
|
||||||
|
// the reaper's archive-then-record atomicity.
|
||||||
|
if delErr := archiver.Delete(ctx, ref); delErr != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: record failed (%v) AND orphan archive %s could not be removed: %v\n", err, ref, delErr)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stderr, "felis backup: record failed, orphan archive removed: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex
|
||||||
|
// scheme so a manual and an inactivity backup are indistinguishable downstream.
|
||||||
|
func newBackupID() string {
|
||||||
|
var b [16]byte
|
||||||
|
if _, err := rand.Read(b[:]); err != nil {
|
||||||
|
// crypto/rand failure is fatal and unrecoverable; a time-based fallback would
|
||||||
|
// be a weaker ID for no benefit. ponytail: panic is the honest failure here.
|
||||||
|
panic("felis backup: crypto/rand: " + err.Error())
|
||||||
|
}
|
||||||
|
return "bk-" + hex.EncodeToString(b[:])
|
||||||
|
}
|
||||||
@@ -16,6 +16,7 @@ Commands:
|
|||||||
api Run the felis-api HTTP server
|
api Run the felis-api HTTP server
|
||||||
reaper Run the world reaper / backup batch
|
reaper Run the world reaper / backup batch
|
||||||
restore Extract a world archive into a world volume (internal Job entrypoint)
|
restore Extract a world archive into a world volume (internal Job entrypoint)
|
||||||
|
backup Archive a world into the backup store and record it (internal Job entrypoint)
|
||||||
manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML
|
manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML
|
||||||
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
||||||
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
||||||
@@ -43,6 +44,8 @@ func run(args []string, stdout, stderr io.Writer) int {
|
|||||||
return cmdReaper(rest, stdout, stderr)
|
return cmdReaper(rest, stdout, stderr)
|
||||||
case "restore":
|
case "restore":
|
||||||
return cmdRestore(rest, stdout, stderr)
|
return cmdRestore(rest, stdout, stderr)
|
||||||
|
case "backup":
|
||||||
|
return cmdBackup(rest, stdout, stderr)
|
||||||
case "manifests":
|
case "manifests":
|
||||||
return cmdManifests(rest, stdout, stderr)
|
return cmdManifests(rest, stdout, stderr)
|
||||||
case "apply":
|
case "apply":
|
||||||
|
|||||||
@@ -0,0 +1,115 @@
|
|||||||
|
# On-demand world backup (§B4 break-glass "Sync"; felis-api endpoint + Job executor)
|
||||||
|
|
||||||
|
- **Type:** feature (addition)
|
||||||
|
- **Date:** 2026-07-07
|
||||||
|
- **Area:** `internal/backupjob` (new pkg), `internal/api`, `cmd/felis`, `docs/openapi.yaml` — Go, oracle-verified
|
||||||
|
- **Commit:** `pending`
|
||||||
|
- **Task:** #31 Phase B4 break-glass ops — the "Sync" operation, resolved with the user as **immediate/on-demand world backup**. Per the user's "两者都要" decision this is built in two halves: **(this change) the felis-api endpoint that does the real backup-Job orchestration**, and (a follow-up) a break-glass menu peer that calls it while the API is alive.
|
||||||
|
|
||||||
|
## What it does
|
||||||
|
|
||||||
|
Adds `POST /api/v1/servers/{name}/backup`: an owner or admin snapshots a **stopped**
|
||||||
|
server's world into the archive store on demand, recorded as a first-class
|
||||||
|
`world_backups` row (reason `manual`) — restorable later by the existing restore path
|
||||||
|
and expired by the reaper's retention pass, so it never leaks as an orphan archive.
|
||||||
|
|
||||||
|
The backup runs asynchronously as a one-shot Kubernetes Job (the new
|
||||||
|
`internal/backupjob` package), mirroring how restore and image builds hand off to
|
||||||
|
Jobs. The handler answers **202 `backing_up`**.
|
||||||
|
|
||||||
|
## Why
|
||||||
|
|
||||||
|
felis-api cannot archive a world in-process: the world PVC is **RWO** and owned by the
|
||||||
|
operator's StatefulSet, so the API has nothing to mount at request time — the same
|
||||||
|
constraint that already makes `internal/restore` a Job. The break-glass console (which
|
||||||
|
runs direct-to-Postgres) likewise lacks the deployment coordinates (`FELIS_IMAGE`,
|
||||||
|
`FELIS_BACKUP_PVC`) needed to render the Job. Both point to the same home: the
|
||||||
|
orchestration belongs in felis-api, which holds those coordinates; other callers
|
||||||
|
invoke the endpoint.
|
||||||
|
|
||||||
|
## Design decisions
|
||||||
|
|
||||||
|
- **Backup Job self-records its `world_backups` row.** Unlike the restore Job — which
|
||||||
|
is deliberately DB-blind because it processes a potentially poisoned archive — the
|
||||||
|
backup Job **does** mount the felis config Secret and inserts its own backup row,
|
||||||
|
exactly like the reaper (the only other component holding both a world mount and the
|
||||||
|
database). This avoids the archive-then-async-record split that would otherwise leak
|
||||||
|
orphan archives on a crash. The security review for that one departure lives in
|
||||||
|
`internal/backupjob/jobspec.go` and is frozen by `jobspec_test.go`. Rationale: a
|
||||||
|
backup only **reads** a world the operator already owns and tars it (bytes, never
|
||||||
|
executed), so restore's poisoned-input threat does not apply; its blast radius (DB +
|
||||||
|
two PVCs) is a strict subset of the reaper's, and it never deletes a PVC nor calls
|
||||||
|
the K8s API (SA token stays un-mounted).
|
||||||
|
- **World mounted read-only, backup PVC read-write** — the mirror image of restore.
|
||||||
|
- **Stopped-gate (409 `not_stopped`).** The world PVC is RWO and held by a running
|
||||||
|
server, so a backup Job cannot double-mount it; the handler refuses unless the server
|
||||||
|
is fully stopped (`info.Ready || DesiredState != Stopped`). This also guarantees a
|
||||||
|
quiescent, non-torn archive. Mirrors `handleRestoreBackup`'s gate.
|
||||||
|
- **Authorization is restore's front half, minus the former-owner match.** Backup is
|
||||||
|
initiated by the **current** owner and records **their** ownership, so there is no
|
||||||
|
prior owner's data to leak — the leak guard that restore needs does not apply here.
|
||||||
|
An admin may back up an unowned (released) world; the recorded former owner is then
|
||||||
|
empty, exactly as the reaper records for an unowned reap.
|
||||||
|
- **Unique Job name per request.** Each backup Job is named `backup-<server>-<rand>`,
|
||||||
|
not a deterministic `backup-<server>`. A deterministic name would collide with a
|
||||||
|
just-finished Job still inside its `TTLSecondsAfterFinished` window (10m), and the
|
||||||
|
`AlreadyExists → 202` path would then silently produce **no** archive — the exact
|
||||||
|
window a user (or the console "立即备份" button) retries in. Unique names make every
|
||||||
|
request produce its own archive; `ErrAlreadyExists` remains only as a defensive
|
||||||
|
no-op on the ~impossible suffix collision. Ceiling (documented in `backup.go`): two
|
||||||
|
truly simultaneous taps may schedule two backup Pods — both mount the world PVC
|
||||||
|
read-only, so neither corrupts anything; single-flight-on-running is the upgrade
|
||||||
|
path if a double-tap storm ever appears.
|
||||||
|
- **One retention clock.** The entrypoint reuses the reaper's `reaperConfig` derivation
|
||||||
|
so a manual backup expires on the same schedule as an inactivity backup — one policy,
|
||||||
|
not two. The `"bk-"+hex` id scheme also matches, so manual and inactivity backups are
|
||||||
|
indistinguishable downstream.
|
||||||
|
- **Fail-safe on record failure.** If the row insert fails, the entrypoint deletes the
|
||||||
|
just-written archive so a failed backup leaves no unrecorded bytes.
|
||||||
|
- **Optional executor, honest 503.** Wired only when `FELIS_IMAGE` + `FELIS_BACKUP_PVC`
|
||||||
|
are supplied (same gate as restore); otherwise `API.Backuper` is nil and the endpoint
|
||||||
|
returns 503 `backup_unavailable`, so the authorization boundary is exercised before
|
||||||
|
the Job executor is deployable.
|
||||||
|
|
||||||
|
## Files
|
||||||
|
|
||||||
|
| File | Change |
|
||||||
|
|---|---|
|
||||||
|
| `internal/backupjob/jobspec.go` | **new** — `BackupJob` renderer + `BackupJobName`; weak SA, token off, hardened container, world RO / backup RW, config-Secret mount |
|
||||||
|
| `internal/backupjob/backup.go` | **new** — `Backuper` (idempotent enqueue) + `Config`/`withDefaults` |
|
||||||
|
| `internal/backupjob/k8sjobs.go` | **new** — controller-runtime `CreateBackupJob` (AlreadyExists → idempotent) |
|
||||||
|
| `internal/backupjob/jobspec_test.go` | **new** — freezes the Job's security shape incl. the deliberate config-Secret mount |
|
||||||
|
| `internal/backupjob/backup_test.go` | **new** — asserts each `Backup` call mints a unique Job name (repeat-tap must not silently no-op) |
|
||||||
|
| `cmd/felis/backup.go` | **new** — `felis backup` in-Pod entrypoint: archive + self-record + orphan-cleanup |
|
||||||
|
| `cmd/felis/run.go` | dispatch `case "backup"` + usage line |
|
||||||
|
| `internal/api/backuper.go` | **new** — the narrow `Backuper` port |
|
||||||
|
| `internal/api/handlers_backups.go` | **+`handleBackupNow`** |
|
||||||
|
| `internal/api/api.go` | `Backuper` field + `POST /servers/{name}/backup` route |
|
||||||
|
| `internal/api/handlers_backup_now_test.go` | **new** — `fakeBackuper` + handler subtests |
|
||||||
|
| `internal/api/backuper_wire_test.go` | **new** — compile-time `Backuper = (*backupjob.Backuper)(nil)` |
|
||||||
|
| `cmd/felis/api.go` | wire `backuper` under the `FELIS_IMAGE`+`FELIS_BACKUP_PVC` gate; `backupConfig` helper |
|
||||||
|
| `docs/openapi.yaml` | document the `backupNow` operation |
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
WSL oracle (go1.26.4, authoritative for Go):
|
||||||
|
|
||||||
|
```
|
||||||
|
go build ./... && go vet ./... && go test ./... → ALL GREEN
|
||||||
|
```
|
||||||
|
|
||||||
|
The `internal/api` OpenAPI served-route contract test (`TestOpenAPIMatchesServedRoutes`)
|
||||||
|
initially failed — the new route was served but undocumented — and passes after adding
|
||||||
|
the `backupNow` operation to `docs/openapi.yaml`. `internal/backupjob` and the new
|
||||||
|
handler subtests pass. The controller-runtime `K8sJobs` binding is integration-only
|
||||||
|
(needs a live cluster) and is exercised only by the interface conformance test.
|
||||||
|
|
||||||
|
## Self-review outcome
|
||||||
|
|
||||||
|
- **ponytail (over-engineering):** the backup Job is a near-mirror of the restore Job,
|
||||||
|
not a shared parameterization — deliberate, because its security shape differs (it
|
||||||
|
holds DB creds) and must be asserted independently, not hidden behind a shared knob.
|
||||||
|
No speculative config; `Config.withDefaults` fills only real deployment values.
|
||||||
|
- **correctness:** the RWO stopped-gate and the self-recording atomicity were traced to
|
||||||
|
the reaper and restore before writing; the former-owner asymmetry vs restore is
|
||||||
|
justified above.
|
||||||
@@ -23,7 +23,9 @@ not yet committed.
|
|||||||
|
|
||||||
## Pending (built + verified, not yet committed)
|
## Pending (built + verified, not yet committed)
|
||||||
|
|
||||||
_None — the break-glass halt (`c2ee21a`) and `/felis migrate` (`c1aa38b`) landed in the ledger below._
|
| Change | Detail doc | Status |
|
||||||
|
|---|---|---|
|
||||||
|
| On-demand world backup — felis-api `POST /servers/{name}/backup` + backup-Job executor (§B4 "Sync", half 1 of 2) | [2026-07-07-on-demand-world-backup.md](2026-07-07-on-demand-world-backup.md) | oracle-green; commit `pending` |
|
||||||
|
|
||||||
## Committed change ledger
|
## Committed change ledger
|
||||||
|
|
||||||
|
|||||||
@@ -2584,6 +2584,50 @@ paths:
|
|||||||
'503':
|
'503':
|
||||||
$ref: '#/components/responses/ServiceUnavailable'
|
$ref: '#/components/responses/ServiceUnavailable'
|
||||||
|
|
||||||
|
/api/v1/servers/{name}/backup:
|
||||||
|
post:
|
||||||
|
tags: [backups]
|
||||||
|
operationId: backupNow
|
||||||
|
summary: Back up a server's world on demand (owner-or-admin; server must be stopped).
|
||||||
|
description: >-
|
||||||
|
Snapshots the server's world into the archive store as a first-class
|
||||||
|
world_backups row (reason "manual"), restorable later like an inactivity
|
||||||
|
backup. The world PVC is RWO and held by a running server, so the server must
|
||||||
|
be fully stopped first (409 not_stopped otherwise). The backup runs
|
||||||
|
asynchronously as a Job, so success is 202 (backing_up).
|
||||||
|
x-felis-face: [external]
|
||||||
|
x-felis-tier: app
|
||||||
|
security: [{ accessJWT: [] }]
|
||||||
|
parameters:
|
||||||
|
- { name: name, in: path, required: true, schema: { type: string } }
|
||||||
|
responses:
|
||||||
|
'202':
|
||||||
|
description: Backup started.
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema:
|
||||||
|
type: object
|
||||||
|
required: [name, status]
|
||||||
|
properties:
|
||||||
|
name: { type: string }
|
||||||
|
status: { type: string, const: backing_up }
|
||||||
|
'401':
|
||||||
|
$ref: '#/components/responses/Unauthorized'
|
||||||
|
'403':
|
||||||
|
$ref: '#/components/responses/Forbidden'
|
||||||
|
'404':
|
||||||
|
description: Unknown server.
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'409':
|
||||||
|
description: Server is not stopped (its world PVC is still mounted).
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'503':
|
||||||
|
$ref: '#/components/responses/ServiceUnavailable'
|
||||||
|
|
||||||
# ------------------------------------------------------ users (admin tier) ----
|
# ------------------------------------------------------ users (admin tier) ----
|
||||||
/api/v1/users:
|
/api/v1/users:
|
||||||
get:
|
get:
|
||||||
|
|||||||
@@ -58,6 +58,11 @@ type API struct {
|
|||||||
// authorization paths do not need it (they read the Repo), only the kick-off.
|
// authorization paths do not need it (they read the Repo), only the kick-off.
|
||||||
Restorer Restorer
|
Restorer Restorer
|
||||||
|
|
||||||
|
// Backuper starts an on-demand world backup (spec §18/§19 WorldArchiver). Like
|
||||||
|
// Restorer it is optional: when nil the backup route reports 503, so the backup
|
||||||
|
// authorization boundary is exercised before the backup-Job executor is wired.
|
||||||
|
Backuper Backuper
|
||||||
|
|
||||||
// Submissions is the user-modpack approval lane (a user-directed extension over
|
// Submissions is the user-modpack approval lane (a user-directed extension over
|
||||||
// the §16 build subsystem; see internal/submit). It is optional: when
|
// the §16 build subsystem; see internal/submit). It is optional: when
|
||||||
// nil the /me/submissions and /submissions routes report 503 rather than 404, so
|
// nil the /me/submissions and /submissions routes report 503 rather than 404, so
|
||||||
@@ -346,6 +351,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
|||||||
// neither sits behind adminOnly.
|
// neither sits behind adminOnly.
|
||||||
{Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups},
|
{Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups},
|
||||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup},
|
{Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup},
|
||||||
|
{Method: "POST", Pattern: "/api/v1/servers/{name}/backup", h: a.handleBackupNow},
|
||||||
// Account linking (spec §10), web side: /start reports link status (it is the
|
// Account linking (spec §10), web side: /start reports link status (it is the
|
||||||
// pointer handleClaim's 412 emits), /verify consumes the in-game code and binds
|
// pointer handleClaim's 412 emits), /verify consumes the in-game code and binds
|
||||||
// the account. App-tier, not admin — linking your own account is an ordinary
|
// the account. App-tier, not admin — linking your own account is an ordinary
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import "context"
|
||||||
|
|
||||||
|
// Backuper starts an on-demand world backup (spec §18/§19 WorldArchiver, run on
|
||||||
|
// demand rather than on the reaper's daily schedule) — the "back up before I touch
|
||||||
|
// it" lever behind POST /api/v1/servers/{name}/backup. Like Restorer it only STARTS
|
||||||
|
// the work: tarring the world PVC into the archive store is pod-filesystem work
|
||||||
|
// felis-api cannot do in-process — the world PVC is RWO and owned by the operator's
|
||||||
|
// StatefulSet, so the API has nothing to mount at request time. The production
|
||||||
|
// implementation therefore hands off to a backup Job (internal/backupjob), which
|
||||||
|
// self-records the world_backups row like the reaper. The call returns once the
|
||||||
|
// backup is enqueued, so the handler answers 202 (backing_up).
|
||||||
|
//
|
||||||
|
// formerOwner is the current owner recorded on the backup row so it can later be
|
||||||
|
// restored (empty when an admin backs up an unowned server). It returns ErrNotFound
|
||||||
|
// if the server is unknown to the execution backend; any other error maps to 500.
|
||||||
|
//
|
||||||
|
// It is an interface so the handler is tested against a fake; the production executor
|
||||||
|
// (internal/backupjob.Backuper) is integration-only, and until it is wired the
|
||||||
|
// API.Backuper is nil so POST /servers/{name}/backup reports 503 — the backup
|
||||||
|
// authorization boundary is exercised without shipping a stub that cannot run.
|
||||||
|
type Backuper interface {
|
||||||
|
Backup(ctx context.Context, serverName, formerOwner string) error
|
||||||
|
}
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"felis.lolicon.best/internal/backupjob"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Compile-time proof that the production backup executor satisfies the API's
|
||||||
|
// Backuper interface, mirroring restore_wire_test. The dependency points one way
|
||||||
|
// only: api defines the narrow Backuper port and never imports backupjob in
|
||||||
|
// production (handlers_backups.go depends on the interface); this test is the single
|
||||||
|
// place the concrete type and the port are pinned together.
|
||||||
|
var _ Backuper = (*backupjob.Backuper)(nil)
|
||||||
@@ -0,0 +1,178 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
)
|
||||||
|
|
||||||
|
// fakeBackuper records the on-demand backup kick-off the handler makes. It mirrors
|
||||||
|
// fakeRestorer: the handler only STARTS the work, so the fake just captures the
|
||||||
|
// arguments and returns a canned error.
|
||||||
|
type fakeBackuper struct {
|
||||||
|
err error
|
||||||
|
calls int
|
||||||
|
gotName string
|
||||||
|
gotFormerOwn string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeBackuper) Backup(_ context.Context, name, formerOwner string) error {
|
||||||
|
f.calls++
|
||||||
|
f.gotName, f.gotFormerOwn = name, formerOwner
|
||||||
|
return f.err
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBackupNow exercises POST /api/v1/servers/{name}/backup (spec §18/§19 WorldArchiver,
|
||||||
|
// on demand). The default server is stopped and owned by owner1 with a wired
|
||||||
|
// fakeBackuper. The focus is the owner-or-admin gate, the stopped gate (the RWO world
|
||||||
|
// PVC must be free), the 404/503 surface, that the recorded former owner is the
|
||||||
|
// server's CURRENT owner, and that backup is an async 202 kick-off.
|
||||||
|
func TestBackupNow(t *testing.T) {
|
||||||
|
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||||
|
|
||||||
|
mk := func() (*API, *fakeRepo, *fakeCluster, *fakeBackuper) {
|
||||||
|
repo := newFakeRepo()
|
||||||
|
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||||
|
cl := newFakeCluster()
|
||||||
|
cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped",
|
||||||
|
Ready: false, DesiredState: string(v1alpha1.DesiredStopped)}
|
||||||
|
backuper := &fakeBackuper{}
|
||||||
|
api := newTestAPI(repo, cl)
|
||||||
|
api.Backuper = backuper
|
||||||
|
return api, repo, cl, backuper
|
||||||
|
}
|
||||||
|
|
||||||
|
const path = "/api/v1/servers/survival/backup"
|
||||||
|
|
||||||
|
t.Run("owner backs up -> 202 + backuper(currentOwner) + audit", func(t *testing.T) {
|
||||||
|
api, repo, _, backuper := mk()
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
var resp struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
Status string `json:"status"`
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
|
||||||
|
t.Fatalf("body not JSON: %v (%s)", err, w.Body.String())
|
||||||
|
}
|
||||||
|
if resp.Name != "survival" || resp.Status != "backing_up" {
|
||||||
|
t.Fatalf("unexpected response %+v", resp)
|
||||||
|
}
|
||||||
|
if backuper.calls != 1 || backuper.gotName != "survival" || backuper.gotFormerOwn != "owner1" {
|
||||||
|
t.Fatalf("backuper saw (calls=%d,%q,%q), want (1,survival,owner1)",
|
||||||
|
backuper.calls, backuper.gotName, backuper.gotFormerOwn)
|
||||||
|
}
|
||||||
|
if len(repo.audits) != 1 || repo.audits[0].Action != "backup.create" || repo.audits[0].Actor != "[email protected]" {
|
||||||
|
t.Fatalf("audit not written as expected: %+v", repo.audits)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("admin backs up an unowned server -> 202, empty former owner recorded", func(t *testing.T) {
|
||||||
|
api, repo, _, backuper := mk()
|
||||||
|
repo.byName["survival"].OwnerID = "" // released world
|
||||||
|
api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "[email protected]",
|
||||||
|
Role: "admin", ViaAdminAccess: true}}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if backuper.calls != 1 || backuper.gotFormerOwn != "" {
|
||||||
|
t.Fatalf("backuper saw (calls=%d, former=%q), want (1, empty)", backuper.calls, backuper.gotFormerOwn)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("non-owner -> 403, no backup", func(t *testing.T) {
|
||||||
|
api, _, _, backuper := mk()
|
||||||
|
api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusForbidden {
|
||||||
|
t.Fatalf("code = %d, want 403", w.Code)
|
||||||
|
}
|
||||||
|
if backuper.calls != 0 {
|
||||||
|
t.Fatal("a forbidden caller must not start a backup")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("unknown server -> 404", func(t *testing.T) {
|
||||||
|
api, _, _, _ := mk()
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/missing/backup", "", nil)
|
||||||
|
if w.Code != http.StatusNotFound {
|
||||||
|
t.Fatalf("code = %d, want 404", w.Code)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("running server -> 409 not_stopped, no backup", func(t *testing.T) {
|
||||||
|
api, _, cl, backuper := mk()
|
||||||
|
cl.byName["survival"].Ready = true
|
||||||
|
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" {
|
||||||
|
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if backuper.calls != 0 {
|
||||||
|
t.Fatal("a running server holds the RWO world PVC — backup must be refused")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("starting server -> 409 not_stopped", func(t *testing.T) {
|
||||||
|
api, _, cl, _ := mk()
|
||||||
|
cl.byName["survival"].Ready = false
|
||||||
|
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" {
|
||||||
|
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("nil Backuper -> 503 backup_unavailable", func(t *testing.T) {
|
||||||
|
api, _, _, _ := mk()
|
||||||
|
api.Backuper = nil
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "backup_unavailable" {
|
||||||
|
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("backuper failure -> 500, not audited", func(t *testing.T) {
|
||||||
|
api, repo, _, backuper := mk()
|
||||||
|
backuper.err = errors.New("kick-off failed")
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusInternalServerError {
|
||||||
|
t.Fatalf("code = %d, want 500 (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if len(repo.audits) != 0 {
|
||||||
|
t.Fatalf("a failed backup must not be audited: %+v", repo.audits)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("backuper reports unknown server -> 404", func(t *testing.T) {
|
||||||
|
api, _, _, backuper := mk()
|
||||||
|
backuper.err = ErrNotFound
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusNotFound {
|
||||||
|
t.Fatalf("code = %d, want 404", w.Code)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("invalid server name -> 400", func(t *testing.T) {
|
||||||
|
api, _, _, _ := mk()
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/X/backup", "", nil)
|
||||||
|
if w.Code != http.StatusBadRequest {
|
||||||
|
t.Fatalf("code = %d, want 400", w.Code)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
@@ -176,3 +176,77 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
|||||||
"backup_id": backup.ID,
|
"backup_id": backup.ID,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleBackupNow starts an on-demand backup of a server's world (POST
|
||||||
|
// /api/v1/servers/{name}/backup; spec §18/§19 WorldArchiver, run on demand). It is
|
||||||
|
// the "back up before I touch it" lever the break-glass console and the owner both
|
||||||
|
// reach for. Authorization mirrors handleRestoreBackup's front half — the shared
|
||||||
|
// "who may act on this server's world" gate — but stops short of restore's backup
|
||||||
|
// resolution and former-owner-match, because a backup is initiated by the CURRENT
|
||||||
|
// owner and records their ownership; there is no prior owner's data to leak:
|
||||||
|
//
|
||||||
|
// ① name validation
|
||||||
|
// ② ServerByName — an unknown server is 404
|
||||||
|
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
|
||||||
|
// released world can still be snapshotted by an operator before disposal)
|
||||||
|
// ④ stopped gate: the world PVC is RWO and held by a running server, so a backup
|
||||||
|
// Job cannot double-mount it — refuse unless the server is fully stopped. This
|
||||||
|
// also guarantees a quiescent, non-torn archive.
|
||||||
|
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
|
||||||
|
// means "enqueued" and the handler answers 202.
|
||||||
|
//
|
||||||
|
// The former owner recorded on the backup is the server's current OwnerID (empty for
|
||||||
|
// an unowned server backed up by an admin), so the resulting world_backups row is
|
||||||
|
// restorable by that owner exactly like an inactivity backup.
|
||||||
|
func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
|
||||||
|
p := principalFromContext(r.Context())
|
||||||
|
name := r.PathValue("name")
|
||||||
|
if err := naming.ValidateServerName(name); err != nil {
|
||||||
|
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
rec, err := a.Repo.ServerByName(r.Context(), name)
|
||||||
|
if err != nil {
|
||||||
|
a.writeLookupError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !a.isOwnerOrAdmin(p, rec) {
|
||||||
|
writeError(w, r, errForbidden)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Stopped gate: the world PVC is RWO and held by a running server, so a backup
|
||||||
|
// Job cannot double-mount it (mirrors the restore gate). Ready means it is up;
|
||||||
|
// any desiredState other than Stopped means it owns the RWO volume.
|
||||||
|
info, err := a.Cluster.GetServer(r.Context(), name)
|
||||||
|
if err != nil {
|
||||||
|
a.writeLookupError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
|
||||||
|
writeError(w, r, newError(http.StatusConflict, "not_stopped",
|
||||||
|
"stop the server before backing up its world"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Backuper is optional: when unwired the endpoint reports 503 rather than
|
||||||
|
// panicking, so the authorization boundary above is exercised even before the
|
||||||
|
// backup-Job executor is wired (see Backuper).
|
||||||
|
if a.Backuper == nil {
|
||||||
|
writeError(w, r, newError(http.StatusServiceUnavailable, "backup_unavailable",
|
||||||
|
"backup subsystem is not configured"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
|
||||||
|
a.writeLookupError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
a.audit(r, p.Email, "backup.create", name)
|
||||||
|
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||||
|
"name": name,
|
||||||
|
"status": "backing_up",
|
||||||
|
})
|
||||||
|
}
|
||||||
@@ -0,0 +1,235 @@
|
|||||||
|
// Package backupjob implements the on-demand world-backup executor (spec §18/§19
|
||||||
|
// WorldArchiver, run on demand rather than on the reaper's daily schedule). It is
|
||||||
|
// the "back up before I touch it" lever behind POST /api/v1/servers/{name}/backup:
|
||||||
|
// an owner or admin stops a server, then snapshots its world into the archive store
|
||||||
|
// as a first-class world_backups row — restorable by internal/restore and expired
|
||||||
|
// by the reaper's retention pass, so it never leaks as an orphan archive.
|
||||||
|
//
|
||||||
|
// felis-api cannot archive a world in-process: the world PVC is RWO and owned by
|
||||||
|
// the operator's StatefulSet, so the API has nothing to mount at request time (the
|
||||||
|
// same constraint that makes internal/restore a Job). This package is the executor
|
||||||
|
// it hands off to — a one-shot Kubernetes Job in the minecraft namespace that
|
||||||
|
// mounts the target world PVC (read-only) and the backup PVC (read-write), then
|
||||||
|
// runs `felis backup` (cmd/felis) to tar the world into the archive store AND
|
||||||
|
// record the world_backups row.
|
||||||
|
//
|
||||||
|
// Trust model. The backup Pod mirrors internal/restore's weak-SA isolation (a weak
|
||||||
|
// SA with its token un-mounted, so it cannot reach the K8s API) with ONE deliberate
|
||||||
|
// departure, reviewed in jobspec.go: it DOES mount the felis config Secret so it can
|
||||||
|
// self-record its backup row atomically with the archive, exactly like the reaper —
|
||||||
|
// the only other component holding both a world mount and the database. A restore
|
||||||
|
// Pod must stay DB-blind because it processes a poisoned archive; a backup Pod only
|
||||||
|
// reads a world the operator already owns and tars it, so that threat does not apply.
|
||||||
|
//
|
||||||
|
// The Backuper depends on the Jobs interface, so the orchestration (idempotent
|
||||||
|
// enqueue, error mapping) is unit-tested against an in-memory fake; the
|
||||||
|
// controller-runtime implementation (k8sjobs.go) compiles here but is exercised only
|
||||||
|
// by integration tests against a live cluster.
|
||||||
|
package backupjob
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"crypto/rand"
|
||||||
|
"encoding/hex"
|
||||||
|
"errors"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/naming"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ErrAlreadyExists is returned by a Jobs implementation when a backup Job for a
|
||||||
|
// server already exists (a backup is already in flight). The Backuper treats it as
|
||||||
|
// success — see Backup.
|
||||||
|
var ErrAlreadyExists = errors.New("backup: job already exists")
|
||||||
|
|
||||||
|
// Jobs is the cluster-side backup lifecycle the Backuper depends on. It is an
|
||||||
|
// interface so the orchestration is tested against a fake; the controller-runtime
|
||||||
|
// implementation (K8sJobs) is integration-tested only — it requires a live cluster.
|
||||||
|
type Jobs interface {
|
||||||
|
// CreateBackupJob renders and applies the backup Job for p. It returns
|
||||||
|
// ErrAlreadyExists if a Job of the same (deterministic) name already exists.
|
||||||
|
CreateBackupJob(ctx context.Context, p JobParams) error
|
||||||
|
}
|
||||||
|
|
||||||
|
// Config parameterises the backup executor. Deployment-specific values that have no
|
||||||
|
// safe default — the felis Image to run and the BackupPVC to mount — are supplied
|
||||||
|
// by the caller (cmd/felis sources them from the environment); when either is empty
|
||||||
|
// the caller leaves the API's Backuper nil so the endpoint reports 503 rather than
|
||||||
|
// enqueuing a Job that cannot run.
|
||||||
|
type Config struct {
|
||||||
|
// Namespace is where the world PVCs live and the backup Job runs (the minecraft
|
||||||
|
// namespace), co-located with the world it snapshots.
|
||||||
|
Namespace string
|
||||||
|
// ServiceAccount is the weak SA the backup Pod runs as. It reuses felis-restore
|
||||||
|
// (bare, no Role/RoleBinding): a backup Pod needs no K8s API access, only
|
||||||
|
// filesystem access to the two PVCs and — via the mounted config Secret, not the
|
||||||
|
// SA — the database.
|
||||||
|
ServiceAccount string
|
||||||
|
// Image is the felis binary image; the Job runs `felis backup` from it.
|
||||||
|
Image string
|
||||||
|
// BackupPVC is the name of the backup PVC the archive is written into (the same
|
||||||
|
// PVC the reaper writes to and restore reads from).
|
||||||
|
BackupPVC string
|
||||||
|
// ConfigSecret is the felis config Secret (felis.toml, carrying the DB URL) the
|
||||||
|
// backup Pod mounts to self-record its world_backups row. Defaults to the name
|
||||||
|
// the control-plane manifests use.
|
||||||
|
ConfigSecret string
|
||||||
|
// ConfigMount is the in-Pod mount path of ConfigSecret (holds felis.toml).
|
||||||
|
ConfigMount string
|
||||||
|
// BackupRoot is the in-Pod mount path of BackupPVC. It MUST equal cfg.Archive.
|
||||||
|
// LocalPath — the path the reaper wrote archives under and restore mounts to
|
||||||
|
// resolve them — because tarLocal archive refs are absolute.
|
||||||
|
BackupRoot string
|
||||||
|
// WorldsRoot is the in-Pod mount path of the world PVC being archived.
|
||||||
|
WorldsRoot string
|
||||||
|
// Deadline caps the backup Pod's wall-clock (activeDeadlineSeconds).
|
||||||
|
Deadline time.Duration
|
||||||
|
// CPULimit / MemLimit cap the backup container.
|
||||||
|
CPULimit string
|
||||||
|
MemLimit string
|
||||||
|
// RunAsUser / RunAsGroup / FSGroup are the Pod's runtime identity. FSGroup MUST
|
||||||
|
// match the operator StatefulSet's runtime group so the read-only world mount is
|
||||||
|
// readable by this Pod's uid.
|
||||||
|
RunAsUser int64
|
||||||
|
RunAsGroup int64
|
||||||
|
FSGroup int64
|
||||||
|
// TTLAfterFinished is how long a finished backup Job lingers before the Job
|
||||||
|
// controller garbage-collects it; it also bounds the window in which a re-backup
|
||||||
|
// sees a stale completed Job as ErrAlreadyExists.
|
||||||
|
TTLAfterFinished time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
// defaults applied when a Config field is left zero. Image and BackupPVC have no
|
||||||
|
// default on purpose — see Config.
|
||||||
|
const (
|
||||||
|
defaultNamespace = "minecraft"
|
||||||
|
defaultServiceAccount = "felis-restore"
|
||||||
|
defaultConfigSecret = "felis-config"
|
||||||
|
defaultConfigMount = "/etc/felis"
|
||||||
|
defaultBackupRoot = "/backups"
|
||||||
|
defaultWorldsRoot = "/world"
|
||||||
|
defaultDeadline = 30 * time.Minute
|
||||||
|
defaultCPULimit = "1"
|
||||||
|
defaultMemLimit = "1Gi"
|
||||||
|
defaultRunAsID = int64(1000)
|
||||||
|
defaultTTL = 10 * time.Minute
|
||||||
|
)
|
||||||
|
|
||||||
|
// withDefaults returns a copy of c with zero fields filled, so a partially
|
||||||
|
// configured Config (or the zero value, in tests) is always usable.
|
||||||
|
func (c Config) withDefaults() Config {
|
||||||
|
if c.Namespace == "" {
|
||||||
|
c.Namespace = defaultNamespace
|
||||||
|
}
|
||||||
|
if c.ServiceAccount == "" {
|
||||||
|
c.ServiceAccount = defaultServiceAccount
|
||||||
|
}
|
||||||
|
if c.ConfigSecret == "" {
|
||||||
|
c.ConfigSecret = defaultConfigSecret
|
||||||
|
}
|
||||||
|
if c.ConfigMount == "" {
|
||||||
|
c.ConfigMount = defaultConfigMount
|
||||||
|
}
|
||||||
|
if c.BackupRoot == "" {
|
||||||
|
c.BackupRoot = defaultBackupRoot
|
||||||
|
}
|
||||||
|
if c.WorldsRoot == "" {
|
||||||
|
c.WorldsRoot = defaultWorldsRoot
|
||||||
|
}
|
||||||
|
if c.Deadline <= 0 {
|
||||||
|
c.Deadline = defaultDeadline
|
||||||
|
}
|
||||||
|
if c.CPULimit == "" {
|
||||||
|
c.CPULimit = defaultCPULimit
|
||||||
|
}
|
||||||
|
if c.MemLimit == "" {
|
||||||
|
c.MemLimit = defaultMemLimit
|
||||||
|
}
|
||||||
|
if c.RunAsUser == 0 {
|
||||||
|
c.RunAsUser = defaultRunAsID
|
||||||
|
}
|
||||||
|
if c.RunAsGroup == 0 {
|
||||||
|
c.RunAsGroup = defaultRunAsID
|
||||||
|
}
|
||||||
|
if c.FSGroup == 0 {
|
||||||
|
c.FSGroup = defaultRunAsID
|
||||||
|
}
|
||||||
|
if c.TTLAfterFinished <= 0 {
|
||||||
|
c.TTLAfterFinished = defaultTTL
|
||||||
|
}
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
// Backuper is the production internal/api.Backuper (the compile-time proof of that
|
||||||
|
// is in internal/api's wire test, which imports this package; this package never
|
||||||
|
// imports api). It holds no mutable state.
|
||||||
|
type Backuper struct {
|
||||||
|
Jobs Jobs
|
||||||
|
Config Config
|
||||||
|
}
|
||||||
|
|
||||||
|
// Backup enqueues a backup Job that tars serverName's world into the archive store
|
||||||
|
// and records the world_backups row. formerOwner is stamped on that row so the owner
|
||||||
|
// can later restore it (empty for an admin backing up an unowned server). It returns
|
||||||
|
// once the Job is created — the archive runs in the Pod — so the handler's 202
|
||||||
|
// ("backing_up") is honest.
|
||||||
|
//
|
||||||
|
// Each call gets a fresh, unique Job name (BackupJobName + random suffix), so an
|
||||||
|
// on-demand backup requested again after a previous one — the console "立即备份"
|
||||||
|
// repeat-tap case — always produces a new archive rather than colliding with a
|
||||||
|
// just-finished Job still inside its TTL window. ErrAlreadyExists is kept only as a
|
||||||
|
// defensive no-op against the astronomically unlikely suffix collision.
|
||||||
|
//
|
||||||
|
// ponytail: unique names mean two truly simultaneous taps can schedule two backup
|
||||||
|
// Pods; both mount the world PVC read-only so neither corrupts anything, and if they
|
||||||
|
// land on different nodes the RWO attach fails one cleanly. Add single-flight-on-
|
||||||
|
// running only if a real double-tap storm ever shows up.
|
||||||
|
func (b *Backuper) Backup(ctx context.Context, serverName, formerOwner string) error {
|
||||||
|
if err := b.Jobs.CreateBackupJob(ctx, b.jobParams(serverName, formerOwner)); err != nil {
|
||||||
|
if errors.Is(err, ErrAlreadyExists) {
|
||||||
|
return nil // suffix collision — treat as enqueued
|
||||||
|
}
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// jobNameSuffix is a short random hex tag that makes each backup Job name unique.
|
||||||
|
// 32 bits is ample: collisions only matter within a single Job's TTL window across
|
||||||
|
// a handful of manual backups.
|
||||||
|
func jobNameSuffix() string {
|
||||||
|
var b [4]byte
|
||||||
|
if _, err := rand.Read(b[:]); err != nil {
|
||||||
|
// ponytail: crypto/rand only fails if the OS RNG is gone — unrecoverable.
|
||||||
|
panic("backupjob: crypto/rand: " + err.Error())
|
||||||
|
}
|
||||||
|
return hex.EncodeToString(b[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
// jobParams projects the server, former owner, and config onto the inputs jobspec.go
|
||||||
|
// renders. The world PVC name is derived from the single shared naming convention
|
||||||
|
// (naming.WorldPVCName), the same one the operator created it under.
|
||||||
|
func (b *Backuper) jobParams(serverName, formerOwner string) JobParams {
|
||||||
|
cfg := b.Config.withDefaults()
|
||||||
|
return JobParams{
|
||||||
|
Server: serverName,
|
||||||
|
JobName: BackupJobName(serverName) + "-" + jobNameSuffix(),
|
||||||
|
FormerOwner: formerOwner,
|
||||||
|
WorldPVC: naming.WorldPVCName(serverName),
|
||||||
|
BackupPVC: cfg.BackupPVC,
|
||||||
|
Namespace: cfg.Namespace,
|
||||||
|
ServiceAccount: cfg.ServiceAccount,
|
||||||
|
Image: cfg.Image,
|
||||||
|
ConfigSecret: cfg.ConfigSecret,
|
||||||
|
ConfigMount: cfg.ConfigMount,
|
||||||
|
BackupRoot: cfg.BackupRoot,
|
||||||
|
WorldsRoot: cfg.WorldsRoot,
|
||||||
|
Deadline: cfg.Deadline,
|
||||||
|
CPULimit: cfg.CPULimit,
|
||||||
|
MemLimit: cfg.MemLimit,
|
||||||
|
RunAsUser: cfg.RunAsUser,
|
||||||
|
RunAsGroup: cfg.RunAsGroup,
|
||||||
|
FSGroup: cfg.FSGroup,
|
||||||
|
TTLAfterFinished: cfg.TTLAfterFinished,
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
package backupjob
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// captureJobs records every JobParams it is handed so the test can inspect the
|
||||||
|
// names the Backuper minted.
|
||||||
|
type captureJobs struct{ got []JobParams }
|
||||||
|
|
||||||
|
func (c *captureJobs) CreateBackupJob(_ context.Context, p JobParams) error {
|
||||||
|
c.got = append(c.got, p)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// The core fix: a repeat on-demand backup of the same server must not collide with
|
||||||
|
// the previous Job's name (which would silently no-op the retry within the TTL
|
||||||
|
// window). Two Backup calls must mint two distinct Job names, both under the
|
||||||
|
// server's base prefix.
|
||||||
|
func TestBackupMintsUniqueJobNamePerCall(t *testing.T) {
|
||||||
|
jobs := &captureJobs{}
|
||||||
|
b := &Backuper{Jobs: jobs, Config: Config{Image: "img", BackupPVC: "pvc"}}
|
||||||
|
|
||||||
|
for i := 0; i < 2; i++ {
|
||||||
|
if err := b.Backup(context.Background(), "survival", "usr-1"); err != nil {
|
||||||
|
t.Fatalf("Backup: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(jobs.got) != 2 {
|
||||||
|
t.Fatalf("created %d jobs, want 2", len(jobs.got))
|
||||||
|
}
|
||||||
|
base := BackupJobName("survival")
|
||||||
|
for _, p := range jobs.got {
|
||||||
|
if !strings.HasPrefix(p.JobName, base+"-") {
|
||||||
|
t.Errorf("JobName %q must extend the base %q", p.JobName, base)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if jobs.got[0].JobName == jobs.got[1].JobName {
|
||||||
|
t.Errorf("two backups reused the Job name %q — retry would silently no-op", jobs.got[0].JobName)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,242 @@
|
|||||||
|
package backupjob
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
batchv1 "k8s.io/api/batch/v1"
|
||||||
|
corev1 "k8s.io/api/core/v1"
|
||||||
|
"k8s.io/apimachinery/pkg/api/resource"
|
||||||
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Label keys applied to backup objects, mirroring internal/restore so the two
|
||||||
|
// executors are observable the same way.
|
||||||
|
const (
|
||||||
|
LabelManagedBy = "app.kubernetes.io/managed-by"
|
||||||
|
LabelComponent = "app.kubernetes.io/component"
|
||||||
|
LabelServer = "felis.lolicon.best/server"
|
||||||
|
|
||||||
|
managedByValue = "felis-backup"
|
||||||
|
componentValue = "world-backup"
|
||||||
|
|
||||||
|
worldVolume = "world"
|
||||||
|
backupVolume = "backup"
|
||||||
|
configVolume = "config"
|
||||||
|
tmpVolume = "tmp"
|
||||||
|
felisBinaryPath = "/usr/local/bin/felis"
|
||||||
|
)
|
||||||
|
|
||||||
|
// JobParams are the rendered inputs to a backup Job, derived from a server +
|
||||||
|
// former owner + Config by the Backuper. jobspec is a pure function of them so
|
||||||
|
// the security-critical Job shape is unit-tested without a cluster.
|
||||||
|
type JobParams struct {
|
||||||
|
Server string
|
||||||
|
// JobName is the unique Job object name for this invocation. The Backuper sets a
|
||||||
|
// fresh per-call name (BackupJobName(server) + random suffix) so every on-demand
|
||||||
|
// request produces its own archive — a deterministic name would collide with a
|
||||||
|
// just-finished Job still inside its TTL window and silently no-op the retry.
|
||||||
|
// Empty falls back to the deterministic BackupJobName (tests, and any direct call).
|
||||||
|
JobName string
|
||||||
|
FormerOwner string // recorded on the world_backups row so the owner can later restore it
|
||||||
|
WorldPVC string
|
||||||
|
BackupPVC string
|
||||||
|
Namespace string
|
||||||
|
ServiceAccount string
|
||||||
|
Image string
|
||||||
|
ConfigSecret string // the felis config Secret mounted for the DB URL (self-recording)
|
||||||
|
ConfigMount string // where the config Secret is mounted (holds felis.toml)
|
||||||
|
BackupRoot string
|
||||||
|
WorldsRoot string
|
||||||
|
Deadline time.Duration
|
||||||
|
CPULimit string
|
||||||
|
MemLimit string
|
||||||
|
RunAsUser int64
|
||||||
|
RunAsGroup int64
|
||||||
|
FSGroup int64
|
||||||
|
|
||||||
|
TTLAfterFinished time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
// BackupJobName is the base Job name for a server's on-demand backup — the prefix
|
||||||
|
// the Backuper extends with a per-invocation random suffix (JobParams.JobName) so
|
||||||
|
// repeat backups do not collide. It is distinct from the restore Job name so a
|
||||||
|
// backup and a restore of the same server never collide.
|
||||||
|
func BackupJobName(server string) string { return "backup-" + server }
|
||||||
|
|
||||||
|
func backupLabels(p JobParams) map[string]string {
|
||||||
|
return map[string]string{
|
||||||
|
LabelManagedBy: managedByValue,
|
||||||
|
LabelComponent: componentValue,
|
||||||
|
LabelServer: p.Server,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// BackupJob renders the on-demand world-backup Job (spec §18/§19 WorldArchiver,
|
||||||
|
// run on demand rather than on the reaper's daily schedule). Its isolation mirrors
|
||||||
|
// the restore Job (weak SA, non-root, read-only root fs, drop ALL, one-shot with a
|
||||||
|
// deadline) with ONE deliberate, security-reviewed departure asserted by
|
||||||
|
// jobspec_test.go:
|
||||||
|
//
|
||||||
|
// - It DOES mount the felis config Secret (read-only), because a backup must
|
||||||
|
// self-record its world_backups row — archive-then-insert atomically, exactly
|
||||||
|
// like the reaper, which is the only other component holding both a world mount
|
||||||
|
// and the database. A restore Pod must never touch the DB because it processes a
|
||||||
|
// potentially poisoned archive; a backup Pod only READS a world the operator
|
||||||
|
// already owns and tars it (bytes, never executed), so the poisoned-input threat
|
||||||
|
// that forbids restore's DB access does not apply. Its blast radius is the DB
|
||||||
|
// plus the two PVCs, matching the reaper's trust for a strict subset of the
|
||||||
|
// reaper's operations (it never deletes a PVC and never calls the K8s API — its
|
||||||
|
// SA token stays un-mounted).
|
||||||
|
// - The world PVC is mounted READ-ONLY (backup only reads it; the RWO volume must
|
||||||
|
// be free, which the handler's stopped-gate guarantees), and the backup PVC
|
||||||
|
// READ-WRITE (the archive is written into it) — the mirror image of restore.
|
||||||
|
//
|
||||||
|
// The container runs `/usr/local/bin/felis backup` (cmd/felis), which tars the
|
||||||
|
// world at WorldsRoot into BackupRoot and inserts the world_backups row. BackupRoot
|
||||||
|
// MUST equal the reaper's [archive] local_path so the recorded absolute ref
|
||||||
|
// resolves the same way a later restore Job mounts it.
|
||||||
|
func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||||
|
if p.Image == "" {
|
||||||
|
return nil, fmt.Errorf("backup: image is empty")
|
||||||
|
}
|
||||||
|
if p.WorldPVC == "" || p.BackupPVC == "" {
|
||||||
|
return nil, fmt.Errorf("backup: world and backup PVC names are required")
|
||||||
|
}
|
||||||
|
if p.ConfigSecret == "" {
|
||||||
|
return nil, fmt.Errorf("backup: config secret name is required")
|
||||||
|
}
|
||||||
|
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
deadline := int64(p.Deadline / time.Second)
|
||||||
|
if deadline <= 0 {
|
||||||
|
deadline = int64(defaultDeadline / time.Second)
|
||||||
|
}
|
||||||
|
ttl := int32(p.TTLAfterFinished / time.Second)
|
||||||
|
if ttl <= 0 {
|
||||||
|
ttl = int32(defaultTTL / time.Second)
|
||||||
|
}
|
||||||
|
|
||||||
|
args := []string{
|
||||||
|
"--config", p.ConfigMount + "/felis.toml",
|
||||||
|
"--server", p.Server,
|
||||||
|
"--worlds-root", p.WorldsRoot,
|
||||||
|
}
|
||||||
|
// FormerOwner is optional: an admin backing up an unowned (released) server
|
||||||
|
// records an empty former_owner, exactly as the reaper does for an unowned reap.
|
||||||
|
if p.FormerOwner != "" {
|
||||||
|
args = append(args, "--former-owner", p.FormerOwner)
|
||||||
|
}
|
||||||
|
|
||||||
|
container := corev1.Container{
|
||||||
|
Name: "backup",
|
||||||
|
Image: p.Image,
|
||||||
|
Command: []string{felisBinaryPath, "backup"},
|
||||||
|
Args: args,
|
||||||
|
VolumeMounts: []corev1.VolumeMount{
|
||||||
|
// The world is only read to tar it; mounting it read-only means the
|
||||||
|
// backup process can never mutate the world it snapshots.
|
||||||
|
{Name: worldVolume, MountPath: p.WorldsRoot, ReadOnly: true},
|
||||||
|
{Name: backupVolume, MountPath: p.BackupRoot},
|
||||||
|
{Name: configVolume, MountPath: p.ConfigMount, ReadOnly: true},
|
||||||
|
{Name: tmpVolume, MountPath: "/tmp"},
|
||||||
|
},
|
||||||
|
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||||
|
SecurityContext: &corev1.SecurityContext{
|
||||||
|
Privileged: boolPtr(false),
|
||||||
|
AllowPrivilegeEscalation: boolPtr(false),
|
||||||
|
ReadOnlyRootFilesystem: boolPtr(true),
|
||||||
|
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
name := p.JobName
|
||||||
|
if name == "" {
|
||||||
|
name = BackupJobName(p.Server)
|
||||||
|
}
|
||||||
|
job := &batchv1.Job{
|
||||||
|
ObjectMeta: metav1.ObjectMeta{
|
||||||
|
Name: name,
|
||||||
|
Namespace: p.Namespace,
|
||||||
|
Labels: backupLabels(p),
|
||||||
|
},
|
||||||
|
Spec: batchv1.JobSpec{
|
||||||
|
// One shot: a wedged archive must not loop. The TTL GCs the finished Job
|
||||||
|
// so a later backup of the same server is not blocked forever by a stale
|
||||||
|
// completed Job.
|
||||||
|
BackoffLimit: int32Ptr(0),
|
||||||
|
ActiveDeadlineSeconds: int64Ptr(deadline),
|
||||||
|
TTLSecondsAfterFinished: int32Ptr(ttl),
|
||||||
|
Template: corev1.PodTemplateSpec{
|
||||||
|
ObjectMeta: metav1.ObjectMeta{Labels: backupLabels(p)},
|
||||||
|
Spec: corev1.PodSpec{
|
||||||
|
RestartPolicy: corev1.RestartPolicyNever,
|
||||||
|
ServiceAccountName: p.ServiceAccount,
|
||||||
|
AutomountServiceAccountToken: boolPtr(false),
|
||||||
|
SecurityContext: &corev1.PodSecurityContext{
|
||||||
|
RunAsNonRoot: boolPtr(true),
|
||||||
|
RunAsUser: int64Ptr(p.RunAsUser),
|
||||||
|
RunAsGroup: int64Ptr(p.RunAsGroup),
|
||||||
|
FSGroup: int64Ptr(p.FSGroup),
|
||||||
|
},
|
||||||
|
Containers: []corev1.Container{container},
|
||||||
|
Volumes: []corev1.Volume{
|
||||||
|
{
|
||||||
|
Name: worldVolume,
|
||||||
|
VolumeSource: corev1.VolumeSource{
|
||||||
|
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
||||||
|
ClaimName: p.WorldPVC,
|
||||||
|
ReadOnly: true,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
Name: backupVolume,
|
||||||
|
VolumeSource: corev1.VolumeSource{
|
||||||
|
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
||||||
|
ClaimName: p.BackupPVC,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
Name: configVolume,
|
||||||
|
VolumeSource: corev1.VolumeSource{
|
||||||
|
Secret: &corev1.SecretVolumeSource{SecretName: p.ConfigSecret},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return job, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
||||||
|
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||||
|
if cpu == "" {
|
||||||
|
cpu = defaultCPULimit
|
||||||
|
}
|
||||||
|
if mem == "" {
|
||||||
|
mem = defaultMemLimit
|
||||||
|
}
|
||||||
|
cpuQty, err := resource.ParseQuantity(cpu)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("backup: invalid cpu limit %q: %w", cpu, err)
|
||||||
|
}
|
||||||
|
memQty, err := resource.ParseQuantity(mem)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("backup: invalid memory limit %q: %w", mem, err)
|
||||||
|
}
|
||||||
|
return corev1.ResourceList{
|
||||||
|
corev1.ResourceCPU: cpuQty,
|
||||||
|
corev1.ResourceMemory: memQty,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func boolPtr(b bool) *bool { return &b }
|
||||||
|
func int32Ptr(i int32) *int32 { return &i }
|
||||||
|
func int64Ptr(i int64) *int64 { return &i }
|
||||||
@@ -0,0 +1,220 @@
|
|||||||
|
package backupjob
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
corev1 "k8s.io/api/core/v1"
|
||||||
|
)
|
||||||
|
|
||||||
|
func sampleJobParams() JobParams {
|
||||||
|
return JobParams{
|
||||||
|
Server: "survival",
|
||||||
|
FormerOwner: "usr-abc",
|
||||||
|
WorldPVC: "world-survival-0",
|
||||||
|
BackupPVC: "felis-backups",
|
||||||
|
Namespace: defaultNamespace,
|
||||||
|
ServiceAccount: defaultServiceAccount,
|
||||||
|
Image: "registry.felis.svc:5000/felis:1.0",
|
||||||
|
ConfigSecret: defaultConfigSecret,
|
||||||
|
ConfigMount: defaultConfigMount,
|
||||||
|
BackupRoot: "/backups",
|
||||||
|
WorldsRoot: "/world",
|
||||||
|
Deadline: 30 * time.Minute,
|
||||||
|
CPULimit: "1",
|
||||||
|
MemLimit: "1Gi",
|
||||||
|
RunAsUser: 1000,
|
||||||
|
RunAsGroup: 1000,
|
||||||
|
FSGroup: 1000,
|
||||||
|
TTLAfterFinished: 10 * time.Minute,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The backup Pod runs under the weak felis-restore SA — never the felis-api
|
||||||
|
// identity — with its token un-mounted, so it cannot reach the K8s API. Its DB
|
||||||
|
// access comes from the mounted config Secret, not from any SA permission.
|
||||||
|
func TestBackupJobRunsUnderWeakSAWithNoAPIToken(t *testing.T) {
|
||||||
|
job, err := BackupJob(sampleJobParams())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob: %v", err)
|
||||||
|
}
|
||||||
|
sa := job.Spec.Template.Spec.ServiceAccountName
|
||||||
|
if sa != defaultServiceAccount {
|
||||||
|
t.Errorf("service account = %q, want %q", sa, defaultServiceAccount)
|
||||||
|
}
|
||||||
|
if sa == "felis-api" {
|
||||||
|
t.Fatal("backup Pod must NOT run as the felis-api SA")
|
||||||
|
}
|
||||||
|
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
|
||||||
|
t.Error("AutomountServiceAccountToken must be explicitly false")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The deliberate departure from restore's zero-secret isolation: a backup Pod
|
||||||
|
// mounts EXACTLY the two PVCs (world read-only, backup read-write) PLUS the config
|
||||||
|
// Secret read-only (so it can self-record its world_backups row like the reaper) —
|
||||||
|
// and nothing else. This test freezes that exact volume set so a future edit that
|
||||||
|
// widens it (e.g. a second Secret, or a writable world mount) fails loudly.
|
||||||
|
func TestBackupJobMountsTwoPVCsPlusConfigSecretOnly(t *testing.T) {
|
||||||
|
job, err := BackupJob(sampleJobParams())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob: %v", err)
|
||||||
|
}
|
||||||
|
spec := job.Spec.Template.Spec
|
||||||
|
|
||||||
|
var secretVols, pvcVols int
|
||||||
|
for _, v := range spec.Volumes {
|
||||||
|
switch {
|
||||||
|
case v.Secret != nil:
|
||||||
|
secretVols++
|
||||||
|
if v.Secret.SecretName != defaultConfigSecret {
|
||||||
|
t.Errorf("secret volume = %q, want the config secret %q", v.Secret.SecretName, defaultConfigSecret)
|
||||||
|
}
|
||||||
|
case v.PersistentVolumeClaim != nil:
|
||||||
|
pvcVols++
|
||||||
|
case v.EmptyDir != nil:
|
||||||
|
// the /tmp scratch dir under the read-only root fs — allowed
|
||||||
|
default:
|
||||||
|
t.Errorf("unexpected volume %q: a backup Pod mounts only the two PVCs, the config Secret, and a /tmp emptyDir", v.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if secretVols != 1 {
|
||||||
|
t.Errorf("secret volumes = %d, want exactly 1 (the config Secret)", secretVols)
|
||||||
|
}
|
||||||
|
if pvcVols != 2 {
|
||||||
|
t.Errorf("PVC volumes = %d, want exactly 2 (world + backup)", pvcVols)
|
||||||
|
}
|
||||||
|
|
||||||
|
// World read-only (backup never mutates the world), backup read-write (the
|
||||||
|
// archive is written into it) — the mirror image of the restore Job.
|
||||||
|
world := mountByName(t, spec.Containers[0].VolumeMounts, worldVolume)
|
||||||
|
if !world.ReadOnly {
|
||||||
|
t.Error("world mount must be read-only — a backup only reads the world")
|
||||||
|
}
|
||||||
|
back := mountByName(t, spec.Containers[0].VolumeMounts, backupVolume)
|
||||||
|
if back.ReadOnly {
|
||||||
|
t.Error("backup mount must be read-write — the archive is written into it")
|
||||||
|
}
|
||||||
|
cfg := mountByName(t, spec.Containers[0].VolumeMounts, configVolume)
|
||||||
|
if !cfg.ReadOnly {
|
||||||
|
t.Error("config Secret mount must be read-only")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The world PVC volume itself is also declared read-only so the RWO claim is
|
||||||
|
// requested read-only (defense in depth beyond the mount flag).
|
||||||
|
for _, v := range spec.Volumes {
|
||||||
|
if v.PersistentVolumeClaim != nil && v.PersistentVolumeClaim.ClaimName == "world-survival-0" && !v.PersistentVolumeClaim.ReadOnly {
|
||||||
|
t.Error("world PVC volume source must be read-only")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The backup container is hardened exactly like the restore/build Job containers:
|
||||||
|
// no privilege, no escalation, read-only root fs, drop ALL capabilities.
|
||||||
|
func TestBackupJobContainerIsHardened(t *testing.T) {
|
||||||
|
job, err := BackupJob(sampleJobParams())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob: %v", err)
|
||||||
|
}
|
||||||
|
sc := job.Spec.Template.Spec.Containers[0].SecurityContext
|
||||||
|
if sc == nil {
|
||||||
|
t.Fatal("container SecurityContext is nil")
|
||||||
|
}
|
||||||
|
if sc.Privileged == nil || *sc.Privileged {
|
||||||
|
t.Error("Privileged must be false")
|
||||||
|
}
|
||||||
|
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
|
||||||
|
t.Error("AllowPrivilegeEscalation must be false")
|
||||||
|
}
|
||||||
|
if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
|
||||||
|
t.Error("ReadOnlyRootFilesystem must be true")
|
||||||
|
}
|
||||||
|
if sc.Capabilities == nil || len(sc.Capabilities.Drop) != 1 || sc.Capabilities.Drop[0] != "ALL" {
|
||||||
|
t.Error("capabilities must drop ALL")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// One-shot: a wedged archive must not loop, and a deadline caps it.
|
||||||
|
func TestBackupJobIsOneShotWithDeadline(t *testing.T) {
|
||||||
|
job, err := BackupJob(sampleJobParams())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob: %v", err)
|
||||||
|
}
|
||||||
|
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
|
||||||
|
t.Error("BackoffLimit must be 0 (no retry loop)")
|
||||||
|
}
|
||||||
|
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds <= 0 {
|
||||||
|
t.Error("ActiveDeadlineSeconds must be set")
|
||||||
|
}
|
||||||
|
if job.Spec.TTLSecondsAfterFinished == nil {
|
||||||
|
t.Error("TTLSecondsAfterFinished must be set so the finished Job is GC'd")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The command carries the former owner so the recorded backup can be restored by
|
||||||
|
// its owner; an empty former owner (admin backing up an unowned server) omits it.
|
||||||
|
func TestBackupJobArgsCarryServerAndOwner(t *testing.T) {
|
||||||
|
job, err := BackupJob(sampleJobParams())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob: %v", err)
|
||||||
|
}
|
||||||
|
args := job.Spec.Template.Spec.Containers[0].Args
|
||||||
|
if !argsContain(args, "--server", "survival") {
|
||||||
|
t.Errorf("args missing --server survival: %v", args)
|
||||||
|
}
|
||||||
|
if !argsContain(args, "--former-owner", "usr-abc") {
|
||||||
|
t.Errorf("args missing --former-owner usr-abc: %v", args)
|
||||||
|
}
|
||||||
|
|
||||||
|
p := sampleJobParams()
|
||||||
|
p.FormerOwner = ""
|
||||||
|
unowned, err := BackupJob(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob(unowned): %v", err)
|
||||||
|
}
|
||||||
|
for _, a := range unowned.Spec.Template.Spec.Containers[0].Args {
|
||||||
|
if a == "--former-owner" {
|
||||||
|
t.Error("--former-owner must be omitted when there is no former owner")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
mut func(*JobParams)
|
||||||
|
}{
|
||||||
|
{"no image", func(p *JobParams) { p.Image = "" }},
|
||||||
|
{"no world pvc", func(p *JobParams) { p.WorldPVC = "" }},
|
||||||
|
{"no backup pvc", func(p *JobParams) { p.BackupPVC = "" }},
|
||||||
|
{"no config secret", func(p *JobParams) { p.ConfigSecret = "" }},
|
||||||
|
} {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
p := sampleJobParams()
|
||||||
|
tc.mut(&p)
|
||||||
|
if _, err := BackupJob(p); err == nil {
|
||||||
|
t.Errorf("BackupJob(%s) = nil error, want a validation error", tc.name)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func mountByName(t *testing.T, mounts []corev1.VolumeMount, name string) corev1.VolumeMount {
|
||||||
|
t.Helper()
|
||||||
|
for _, m := range mounts {
|
||||||
|
if m.Name == name {
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("volume mount %q not found", name)
|
||||||
|
return corev1.VolumeMount{}
|
||||||
|
}
|
||||||
|
|
||||||
|
func argsContain(args []string, flag, val string) bool {
|
||||||
|
for i := 0; i+1 < len(args); i++ {
|
||||||
|
if args[i] == flag && args[i+1] == val {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
package backupjob
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
|
||||||
|
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||||
|
)
|
||||||
|
|
||||||
|
// K8sJobs is the production Jobs backed by a controller-runtime client. It creates
|
||||||
|
// the on-demand world-backup Job — nothing more: the backup Job is one-shot and
|
||||||
|
// self-cleaning (ttlSecondsAfterFinished), so there is no phase or cancel seam, and
|
||||||
|
// thus no config to hold. Every backup parameter arrives in the JobParams the
|
||||||
|
// Backuper builds from its own (defaulted) Config. The cluster-bootstrap objects
|
||||||
|
// (the weak felis-restore SA it reuses) are installed once by the deployment
|
||||||
|
// manifests, not per backup, so this binding never creates them. It is
|
||||||
|
// integration-tested against a live cluster, not the hermetic unit suite.
|
||||||
|
type K8sJobs struct {
|
||||||
|
c client.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewK8sJobs builds a Jobs over c.
|
||||||
|
func NewK8sJobs(c client.Client) *K8sJobs {
|
||||||
|
return &K8sJobs{c: c}
|
||||||
|
}
|
||||||
|
|
||||||
|
// CreateBackupJob renders and applies the backup Job. Its name is a deterministic
|
||||||
|
// function of the server (BackupJobName), so a concurrent backup of the same server
|
||||||
|
// collides on Create; that collision is mapped to ErrAlreadyExists, which the
|
||||||
|
// Backuper treats as success (idempotent enqueue).
|
||||||
|
func (k *K8sJobs) CreateBackupJob(ctx context.Context, p JobParams) error {
|
||||||
|
job, err := BackupJob(p)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := k.c.Create(ctx, job); err != nil {
|
||||||
|
if apierrors.IsAlreadyExists(err) {
|
||||||
|
return ErrAlreadyExists
|
||||||
|
}
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
Reference in new issue
Block a user