fix(backup): 手动备份按服冷却、每服保留上限与独立保留期,容量驱逐不再删除回收世界的唯一副本

This commit is contained in:
Lemon-miaow committed 2026-09-24 19:58:59 +08:00
1 parent f69cdec9d4
commit 9bda3a52fa
30 files changed
+801 -42

No files matched your search

+14
View File
@@ -24,6 +24,7 @@ import (
"felis.lolicon.best/internal/panel" "felis.lolicon.best/internal/panel"
"felis.lolicon.best/internal/passkey" "felis.lolicon.best/internal/passkey"
"felis.lolicon.best/internal/platform" "felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/reaper"
"felis.lolicon.best/internal/restore" "felis.lolicon.best/internal/restore"
"felis.lolicon.best/internal/store" "felis.lolicon.best/internal/store"
"felis.lolicon.best/internal/submit" "felis.lolicon.best/internal/submit"
@@ -262,6 +263,15 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// agree on what local auth knows. // agree on what local auth knows.
repo := api.NewPGRepo(drv.DB()) repo := api.NewPGRepo(drv.DB())
// The owner's on-demand backup levers come from [archive], the same keys the
// backup Job and the reaper read. A malformed key leaves the defaults in
// place here; the reaper Job fails on it and names it.
rcfg, err := reaperConfig(cfg)
if err != nil {
fmt.Fprintf(stderr, "felis api: %v; using the default backup limits\n", err)
rcfg = reaper.DefaultConfig()
}
a := &api.API{ a := &api.API{
Repo: repo, Repo: repo,
Cluster: api.NewK8sCluster(cl, cfg.K8s.Namespace), Cluster: api.NewK8sCluster(cl, cfg.K8s.Namespace),
@@ -296,6 +306,10 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
AdminHostname: cfg.Auth.AdminHostname, AdminHostname: cfg.Auth.AdminHostname,
PanelHostname: cfg.Auth.PanelHostname, PanelHostname: cfg.Auth.PanelHostname,
WakeCooldown: 30 * time.Second, WakeCooldown: 30 * time.Second,
// An owner may start one backup per server per manual_cooldown, and none
// while the store is at max_local_bytes (data-durability-9).
BackupCooldown: rcfg.ManualCooldown,
BackupStoreCap: rcfg.MaxLocalBytes,
// The user-modpack lane's per-user throttles: a create spaces out // The user-modpack lane's per-user throttles: a create spaces out
// review-queue rows, an upload spaces out (up to 1 GiB) context streams. // review-queue rows, an upload spaces out (up to 1 GiB) context streams.
// Separate keys, so the normal create→upload sequence stays immediate. // Separate keys, so the normal create→upload sequence stays immediate.
+37 -4
View File
@@ -1,6 +1,7 @@
package main package main
import ( import (
"context"
"crypto/rand" "crypto/rand"
"encoding/hex" "encoding/hex"
"flag" "flag"
@@ -52,8 +53,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store) fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
return 1 return 1
} }
// Reuse the reaper's retention derivation so an on-demand backup expires on the // The [archive] parse the reaper uses; an on-demand backup takes its
// same clock as an inactivity backup — one retention policy, not two. // manual_retention and manual_keep.
rcfg, err := reaperConfig(cfg) rcfg, err := reaperConfig(cfg)
if err != nil { if err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err) fmt.Fprintf(stderr, "felis backup: %v\n", err)
@@ -72,6 +73,13 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
ctx := ctrl.SetupSignalHandler() ctx := ctrl.SetupSignalHandler()
// The archive store shares the node's disk with every world and the
// database: an owner's backup must not be what tips it into eviction.
if err := backup.CheckRoom(cfg.Archive.LocalPath, *worldsRoot, backup.MinFreeAfter); err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server)) ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
if err != nil { if err != nil {
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err) fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
@@ -92,9 +100,10 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
BackupRef: string(ref), BackupRef: string(ref),
SizeBytes: size, SizeBytes: size,
Reason: "manual", Reason: "manual",
ExpiresAt: time.Now().Add(rcfg.Retention), ExpiresAt: time.Now().Add(rcfg.ManualRetention),
} }
if err := reaper.NewPGStore(drv.DB()).InsertBackup(ctx, rec); err != nil { st := reaper.NewPGStore(drv.DB())
if err := st.InsertBackup(ctx, rec); err != nil {
// The archive is written but unrecorded — an orphan the retention pass would // The archive is written but unrecorded — an orphan the retention pass would
// never expire. Delete it so a failed backup leaves no leaked bytes, mirroring // never expire. Delete it so a failed backup leaves no leaked bytes, mirroring
// the reaper's archive-then-record atomicity. // the reaper's archive-then-record atomicity.
@@ -107,9 +116,33 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
} }
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID) fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
pruneManualBackups(ctx, st, archiver, *server, rcfg.ManualKeep, stdout, stderr)
return 0 return 0
} }
// pruneManualBackups keeps server's newest keep on-demand backups and removes
// the rest, oldest first, so repeated backups of one world cannot fill the
// shared archive store. The new backup is already recorded; a removal that
// fails is reported and retried after the next backup.
func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server string, keep int, stdout, stderr io.Writer) {
excess, err := st.ExcessManualBackups(ctx, server, keep)
if err != nil {
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
return
}
for _, b := range excess {
if err := archiver.Delete(ctx, backup.ArchiveRef(b.BackupRef)); err != nil {
fmt.Fprintf(stderr, "felis backup: remove older backup %s: %v\n", b.ID, err)
continue
}
if err := st.MarkBackupDeleted(ctx, b.ID, time.Now()); err != nil {
fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
continue
}
fmt.Fprintf(stdout, "felis backup: removed older backup %s of %s (keeping the newest %d)\n", b.ID, server, keep)
}
}
// newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex // newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex
// scheme so a manual and an inactivity backup are indistinguishable downstream. // scheme so a manual and an inactivity backup are indistinguishable downstream.
func newBackupID() string { func newBackupID() string {
+23 -2
View File
@@ -183,8 +183,9 @@ func (w *mailWarner) Warn(ctx context.Context, ownerID, server, remaining string
} }
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d // reaperConfig derives the reaper's retention windows from felis.toml. The 15d
// idle deadline is fixed by §18; only the warning offsets, retention, and the // idle deadline is fixed by §18; only the warning offsets, retention, the
// store soft-cap are configurable (§24). // store soft-cap and the on-demand backup bounds are configurable (§24). The
// backup Job and felis-api read the manual_* bounds through it too.
func reaperConfig(cfg *config.Config) (reaper.Config, error) { func reaperConfig(cfg *config.Config) (reaper.Config, error) {
rc := reaper.DefaultConfig() rc := reaper.DefaultConfig()
if v := cfg.Archive.Retention; v != "" { if v := cfg.Archive.Retention; v != "" {
@@ -212,6 +213,26 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
} }
rc.MaxLocalBytes = b rc.MaxLocalBytes = b
} }
if v := cfg.Archive.ManualRetention; v != "" {
d, err := parseSpanDuration(v)
if err != nil || d <= 0 {
return rc, fmt.Errorf("[archive] manual_retention %q: want a positive span such as 30d", v)
}
rc.ManualRetention = d
}
switch n := cfg.Archive.ManualKeep; {
case n < 0:
return rc, fmt.Errorf("[archive] manual_keep %d: want 1 or more", n)
case n > 0:
rc.ManualKeep = n
}
if v := cfg.Archive.ManualCooldown; v != "" {
d, err := parseSpanDuration(v)
if err != nil || d < 0 {
return rc, fmt.Errorf("[archive] manual_cooldown %q: want a span such as 10m (0s for none)", v)
}
rc.ManualCooldown = d
}
rc.RequireOffsite = cfg.Offsite.Enabled() rc.RequireOffsite = cfg.Offsite.Enabled()
return rc, nil return rc, nil
} }
+34
View File
@@ -8,11 +8,13 @@ import (
"path/filepath" "path/filepath"
"strings" "strings"
"testing" "testing"
"time"
corev1 "k8s.io/api/core/v1" corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client/fake" "sigs.k8s.io/controller-runtime/pkg/client/fake"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/reaper" "felis.lolicon.best/internal/reaper"
) )
@@ -46,6 +48,38 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
// fail-closed miss. The stock local-path arm is derived from the live PVC's // fail-closed miss. The stock local-path arm is derived from the live PVC's
// volumeName — a name-based guess (glob) could tar a stale deleted PV's bytes and // volumeName — a name-based guess (glob) could tar a stale deleted PV's bytes and
// then delete the current world, which is why it is read from the API instead. // then delete the current world, which is why it is read from the API instead.
// TestReaperConfigManualKeys: the on-demand backup keys default to 30 days,
// five per server and a ten-minute cooldown, accept overrides, and refuse
// values that would keep nothing or throttle backwards.
func TestReaperConfigManualKeys(t *testing.T) {
rc, err := reaperConfig(&config.Config{})
if err != nil {
t.Fatal(err)
}
if rc.ManualRetention != 30*reaper.Day || rc.ManualKeep != 5 || rc.ManualCooldown != 10*time.Minute {
t.Fatalf("defaults = %v / %d / %v", rc.ManualRetention, rc.ManualKeep, rc.ManualCooldown)
}
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{
ManualRetention: "7d", ManualKeep: 2, ManualCooldown: "0s"}})
if err != nil {
t.Fatal(err)
}
if rc.ManualRetention != 7*reaper.Day || rc.ManualKeep != 2 || rc.ManualCooldown != 0 {
t.Fatalf("overrides = %v / %d / %v", rc.ManualRetention, rc.ManualKeep, rc.ManualCooldown)
}
for _, bad := range []config.ArchiveConfig{
{ManualRetention: "0d"},
{ManualRetention: "soon"},
{ManualKeep: -1},
{ManualCooldown: "-5m"},
{ManualCooldown: "often"},
} {
if _, err := reaperConfig(&config.Config{Archive: bad}); err == nil {
t.Errorf("%+v was accepted", bad)
}
}
}
func TestResolveWorldDir(t *testing.T) { func TestResolveWorldDir(t *testing.T) {
ctx := context.Background() ctx := context.Background()
root := t.TempDir() root := t.TempDir()
+9 -8
View File
@@ -2314,13 +2314,14 @@ persisted_auth_source_blocks() {
} }
# persisted_archive_block echoes the operator-owned [archive] keys an earlier run # persisted_archive_block echoes the operator-owned [archive] keys an earlier run
# left behind — the retention window, the pre-reap warn offsets, the local cap — # left behind — the retention window, the pre-reap warn offsets, the local cap,
# so a re-run does not silently revert them to the built-ins the reaper carries # the on-demand backup retention/count/cooldown — so a re-run does not silently
# (felis reaper reads these from the config Secret at run time; defaults: 90d # revert them to the built-ins (felis reaper, felis backup and felis api read
# retention, 3d/1d warnings, no cap). store and local_path are NOT carried: this # these from the config Secret; defaults: 90d retention, 3d/1d warnings, no cap,
# script owns them (FELIS_ARCHIVE_LOCAL_PATH must equal the mount). Same # manual backups kept 30d, 5 per server, one per 10m). store and local_path are
# first-readable-file rule as persisted_smtp_block; warn_before must be a # NOT carried: this script owns them (FELIS_ARCHIVE_LOCAL_PATH must equal the
# single-line TOML array (the shape every writer here emits). # mount). Same first-readable-file rule as persisted_smtp_block; warn_before must
# be a single-line TOML array (the shape every writer here emits).
persisted_archive_block() { persisted_archive_block() {
local f out local f out
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
@@ -2328,7 +2329,7 @@ persisted_archive_block() {
out="$(awk ' out="$(awk '
/^[[:space:]]*\[/ { sect = $0; next } /^[[:space:]]*\[/ { sect = $0; next }
sect ~ /^[[:space:]]*\[archive\][[:space:]]*$/ && sect ~ /^[[:space:]]*\[archive\][[:space:]]*$/ &&
/^[[:space:]]*(retention|warn_before|max_local_bytes)[[:space:]]*=/ { print } /^[[:space:]]*(retention|warn_before|max_local_bytes|manual_retention|manual_keep|manual_cooldown)[[:space:]]*=/ { print }
' "$f")" ' "$f")"
[ -n "$out" ] || continue [ -n "$out" ] || continue
printf '%s\n' "$out" printf '%s\n' "$out"
+4
View File
@@ -1046,6 +1046,8 @@ region = "us-east-1"
store = "tarLocal" store = "tarLocal"
local_path = "/stale/path" local_path = "/stale/path"
retention = "30d" retention = "30d"
manual_keep = 3
manual_cooldown = "1h"
[offsite] [offsite]
endpoint = "https://objects.example" endpoint = "https://objects.example"
@@ -1076,6 +1078,8 @@ expect "a re-run carries the [registry.s3] uploads subtable" "[registry.s3]" "$o
expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "$out" expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "$out"
expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out" expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out"
expect "a re-run carries the archive retention window" 'retention = "30d"' "$out" expect "a re-run carries the archive retention window" 'retention = "30d"' "$out"
expect "a re-run carries the on-demand backup count" 'manual_keep = 3' "$out"
expect "a re-run carries the on-demand backup cooldown" 'manual_cooldown = "1h"' "$out"
expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out" expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out"
expect "a re-run keeps the off-site bucket, set apart from the next section" '[offsite] expect "a re-run keeps the off-site bucket, set apart from the next section" '[offsite]
endpoint = "https://objects.example" endpoint = "https://objects.example"
+29 -3
View File
@@ -624,7 +624,8 @@ Registry manifest rendering is [GO-TESTED]; actual serving is
The reaper is a **run-once daily CronJob batch**, not an operator controller. It The reaper is a **run-once daily CronJob batch**, not an operator controller. It
reaps a world only when `now - last_active_at > 15d` (`inactive_15d`); the 15-day reaps a world only when `now - last_active_at > 15d` (`inactive_15d`); the 15-day
deadline is **hard-fixed in code** (only `warn_before` / `retention` / deadline is **hard-fixed in code** (only `warn_before` / `retention` /
`max_local_bytes` are configurable from `felis.toml [archive]`). `max_local_bytes` and the on-demand backup keys `manual_retention` /
`manual_keep` / `manual_cooldown` are configurable from `felis.toml [archive]`).
### What a "backup" contains ### What a "backup" contains
@@ -641,8 +642,13 @@ world growth) — do not size the archive PVC as if only world data were stored.
The reap sequence (all [GO-TESTED] hermetically) preserves the world unless a The reap sequence (all [GO-TESTED] hermetically) preserves the world unless a
**confirmed, DB-recorded backup exists**: **confirmed, DB-recorded backup exists**:
1. `ensureCapacity` (only if `max_local_bytes > 0`) → store full ⇒ world 1. `ensureCapacity` (only if `max_local_bytes > 0`) frees room by evicting
**preserved** (not deleted). owners' on-demand backups first, oldest first, then reaper archives whose
off-site copy is confirmed. The only copy of a reaped world is never
evicted: it stays until `retention` expires it. Still full ⇒ world
**preserved** (not deleted), counted in `store_full=`. [GO-TESTED:
`TestCapacityEvictionOrderSparesSoleCopies`; PG-TESTED:
`TestManualBackupRationing`]
2. `Archiver.Archive` fails ⇒ world **preserved**, PVC untouched. 2. `Archiver.Archive` fails ⇒ world **preserved**, PVC untouched.
3. `InsertBackup` (DB) fails ⇒ the orphan archive is deleted, PVC **untouched**. 3. `InsertBackup` (DB) fails ⇒ the orphan archive is deleted, PVC **untouched**.
4. With an `[offsite]` bucket configured (§16), the archive must also be in the 4. With an `[offsite]` bucket configured (§16), the archive must also be in the
@@ -689,6 +695,26 @@ reports `the world reaper has not succeeded for …`. [GO-TESTED:
`TestReportReaperRunFailsTheJob`, `TestExpiryFailureFailsTheRun`, `TestReportReaperRunFailsTheJob`, `TestExpiryFailureFailsTheRun`,
`TestCapacityStillFullSkipsReap`.] `TestCapacityStillFullSkipsReap`.]
### On-demand backups ("Back up now")
An owner's "Back up now" writes a `manual` backup into the same store and onto
the same disk as the worlds and the database, so it is rationed:
| Key | Default | Effect |
|---|---|---|
| `manual_retention` | `30d` | when a manual backup expires (reaper archives use `retention`) |
| `manual_keep` | `5` | manual backups kept per server; after each backup the Job removes older ones and prints `removed older backup …` |
| `manual_cooldown` | `10m` | one owner-started backup per server per window; the next one gets `429 backup_cooldown` with `Retry-After` (`0s` disables) |
While the present backups add up to `max_local_bytes` or more, owners get
`507 backup_store_full`. Admins and the break-glass console are exempt from the
cooldown and the cap. Whoever starts it, the backup Job refuses to write an
archive that would leave less than 10% of the archive filesystem free: it fails
with `not enough free disk for the archive`. A failed backup or restore shows
the error its container exited on under Recent operations on the server's
backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`,
`TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`]
### Exemptions (world never reaped) ### Exemptions (world never reaped)
- `spec.reaperExempt=true` → skipped entirely (system servers). [GO-TESTED - `spec.reaperExempt=true` → skipped entirely (system servers). [GO-TESTED
+8
View File
@@ -128,6 +128,14 @@ type API struct {
// on the wake lever). Zero disables throttling. // on the wake lever). Zero disables throttling.
WakeCooldown time.Duration WakeCooldown time.Duration
// BackupCooldown spaces out an owner's on-demand backups of one server, and
// BackupStoreCap refuses them once the present backups reach [archive]
// max_local_bytes (data-durability-9): each archive lands on the node disk
// the worlds and the database share. Admins and the break-glass console are
// exempt. Zero disables each lever.
BackupCooldown time.Duration
BackupStoreCap int64
// SubmitCreateCooldown / SubmitUploadCooldown throttle the user-modpack // SubmitCreateCooldown / SubmitUploadCooldown throttle the user-modpack
// submission lane per user: create bounds how quickly review-queue rows can // submission lane per user: create bounds how quickly review-queue rows can
// appear, upload bounds how often a user may stream a (up to 1 GiB) build // appear, upload bounds how often a user may stream a (up to 1 GiB) build
+26
View File
@@ -48,6 +48,9 @@ type fakeRepo struct {
resourceUpdates map[string]ResourceSpec resourceUpdates map[string]ResourceSpec
audits []AuditEntry audits []AuditEntry
failAudit error // Audit fails with it (a store outage) failAudit error // Audit fails with it (a store outage)
// backupRequested mirrors the newest backup.create audit row per server,
// stamped by Audit with the wall clock (LastBackupRequest).
backupRequested map[string]time.Time
joins []string joins []string
// create-server seeding (spec §15) // create-server seeding (spec §15)
seeded map[string]bool // name -> servers row exists seeded map[string]bool // name -> servers row exists
@@ -750,9 +753,32 @@ func (f *fakeRepo) Audit(_ context.Context, e AuditEntry) error {
return f.failAudit return f.failAudit
} }
f.audits = append(f.audits, e) f.audits = append(f.audits, e)
if e.Action == "backup.create" {
if f.backupRequested == nil {
f.backupRequested = map[string]time.Time{}
}
f.backupRequested[e.ServerName] = time.Now()
}
return nil return nil
} }
func (f *fakeRepo) LastBackupRequest(_ context.Context, serverName string, since time.Time) (time.Time, error) {
if at, ok := f.backupRequested[serverName]; ok && !at.Before(since) {
return at, nil
}
return time.Time{}, nil
}
func (f *fakeRepo) BackupStoreBytes(context.Context) (int64, error) {
var n int64
for _, b := range f.backups {
if b.view.Status == "present" {
n += b.view.SizeBytes
}
}
return n, nil
}
// AllBackups / BackupsForUser / LatestBackup mirror the PG queries' contract so // AllBackups / BackupsForUser / LatestBackup mirror the PG queries' contract so
// the hermetic tests can't pass against a too-lenient fake: only status='present' // the hermetic tests can't pass against a too-lenient fake: only status='present'
// rows are visible, the user scope is the former_owner column, and LatestBackup // rows are visible, the user scope is the former_owner column, and LatestBackup
+74
View File
@@ -5,7 +5,9 @@ import (
"encoding/json" "encoding/json"
"errors" "errors"
"net/http" "net/http"
"strconv"
"testing" "testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1" "felis.lolicon.best/internal/apis/felis/v1alpha1"
) )
@@ -191,6 +193,78 @@ func TestBackupNow(t *testing.T) {
t.Fatalf("code = %d, want 400", w.Code) t.Fatalf("code = %d, want 400", w.Code)
} }
}) })
// data-durability-9: an owner's backups are rationed per server; an admin's
// are not.
t.Run("owner inside the cooldown -> 429 backup_cooldown with Retry-After", func(t *testing.T) {
api, _, _, backuper := mk()
api.BackupCooldown = 10 * time.Minute
api.Now = time.Now // the fake stamps backup.create audits with the wall clock
api.External = staticExternal{p: owner}
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("first backup: code = %d (%s)", w.Code, w.Body.String())
}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "backup_cooldown" {
t.Fatalf("second backup: code = %d body %s", w.Code, w.Body.String())
}
if ra, _ := strconv.Atoi(w.Header().Get("Retry-After")); ra < 590 || ra > 600 {
t.Fatalf("Retry-After = %q, want about 600", w.Header().Get("Retry-After"))
}
if backuper.calls != 1 {
t.Fatalf("backuper called %d times, want 1", backuper.calls)
}
})
t.Run("cooldown elapsed -> 202", func(t *testing.T) {
api, repo, _, _ := mk()
api.BackupCooldown = 10 * time.Minute
api.Now = time.Now
repo.backupRequested = map[string]time.Time{"survival": time.Now().Add(-11 * time.Minute)}
api.External = staticExternal{p: owner}
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
}
})
t.Run("admin bypasses the cooldown and the store cap", func(t *testing.T) {
api, repo, _, backuper := mk()
api.BackupCooldown = 10 * time.Minute
api.BackupStoreCap = 100
api.Now = time.Now
repo.backupRequested = map[string]time.Time{"survival": time.Now()}
repo.backups = []fakeBackup{{view: BackupView{ID: "b1", ServerName: "other", Status: "present", SizeBytes: 500}}}
api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "[email protected]",
Role: "admin", ViaAdminAccess: true}}
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
}
if backuper.calls != 1 {
t.Fatal("the admin's backup did not start")
}
})
t.Run("owner with the store at its cap -> 507 backup_store_full", func(t *testing.T) {
api, repo, _, backuper := mk()
api.BackupStoreCap = 1000
repo.backups = []fakeBackup{
{view: BackupView{ID: "b1", ServerName: "other", Status: "present", SizeBytes: 600}},
{view: BackupView{ID: "b2", ServerName: "survival", Status: "present", SizeBytes: 400}},
{view: BackupView{ID: "b3", ServerName: "survival", Status: "deleted", SizeBytes: 9000}},
}
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusInsufficientStorage || decodeErr(t, w) != "backup_store_full" {
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
}
if backuper.calls != 0 {
t.Fatal("a full store still started a backup")
}
repo.backups[0].view.Status = "deleted"
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("below the cap: code = %d (%s)", w.Code, w.Body.String())
}
})
} }
// TestInternalBackup exercises POST /api/v1/internal/servers/{name}/backup, the // TestInternalBackup exercises POST /api/v1/internal/servers/{name}/backup, the
+41
View File
@@ -1,9 +1,11 @@
package api package api
import ( import (
"context"
"errors" "errors"
"net/http" "net/http"
"strings" "strings"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1" "felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance" "felis.lolicon.best/internal/maintenance"
@@ -255,10 +257,49 @@ func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
writeError(w, r, errForbidden) writeError(w, r, errForbidden)
return return
} }
if !p.IsAdmin() {
if err := a.backupAllowance(r.Context(), name); err != nil {
writeError(w, r, err)
return
}
}
a.enqueueBackup(w, r, name, rec, auditActor(p), "external") a.enqueueBackup(w, r, name, rec, auditActor(p), "external")
} }
// backupAllowance rations an owner's on-demand backups (data-durability-9):
// one per BackupCooldown per server, and none while the present backups fill
// BackupStoreCap. The owner's older backups are pruned by the Job itself
// ([archive] manual_keep), so these two gates bound the rate and the total.
func (a *API) backupAllowance(ctx context.Context, name string) error {
if a.BackupCooldown > 0 {
now := a.now()
last, err := a.Repo.LastBackupRequest(ctx, name, now.Add(-a.BackupCooldown))
if err != nil {
return err
}
if !last.IsZero() {
wait := last.Add(a.BackupCooldown).Sub(now)
if wait > 0 {
return newError(http.StatusTooManyRequests, "backup_cooldown",
"a backup of this server was started %s ago; the next one can start in %s",
now.Sub(last).Round(time.Second), wait.Round(time.Second)).retryAfter(wait)
}
}
}
if a.BackupStoreCap > 0 {
used, err := a.Repo.BackupStoreBytes(ctx)
if err != nil {
return err
}
if used >= a.BackupStoreCap {
return newError(http.StatusInsufficientStorage, "backup_store_full",
"the backup store is full; ask an administrator to free space")
}
}
return nil
}
// handleInternalBackup is the internal-face backup trigger. The break-glass console // handleInternalBackup is the internal-face backup trigger. The break-glass console
// (root on the node, holding the service token) POSTs here to snapshot a stopped // (root on the node, holding the service token) POSTs here to snapshot a stopped
// world while felis-api is alive — it goes through the API rather than direct-to-CRD // world while felis-api is alive — it goes through the API rather than direct-to-CRD
+58
View File
@@ -11,6 +11,9 @@ import (
batchv1 "k8s.io/api/batch/v1" batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1" corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
) )
type fakeJobStatus struct { type fakeJobStatus struct {
@@ -145,3 +148,58 @@ func TestJobToAsyncJob(t *testing.T) {
t.Fatal("foreign job must be dropped") t.Fatal("foreign job must be dropped")
} }
} }
// TestLatestJobsExplainsFailures: a failed Job reports the error its container
// exited on (the last line of the terminated message), newest pod first, and
// keeps the condition text when no pod explains it.
func TestLatestJobsExplainsFailures(t *testing.T) {
scheme := runtime.NewScheme()
if err := clientgoscheme.AddToScheme(scheme); err != nil {
t.Fatal(err)
}
labels := func(job string) map[string]string {
return map[string]string{jobServerLabel: "survival", jobManagedByLabel: jobManagedByBackup, "job-name": job}
}
at := func(min int) metav1.Time { return metav1.NewTime(time.Date(2026, 9, 24, 10, min, 0, 0, time.UTC)) }
failedJob := func(name string, min int) *batchv1.Job {
start := at(min)
return &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: "minecraft", Labels: labels(name)},
Status: batchv1.JobStatus{StartTime: &start, Conditions: []batchv1.JobCondition{{
Type: batchv1.JobFailed, Status: corev1.ConditionTrue,
Reason: "BackoffLimitExceeded", Message: "Job has reached the specified backoff limit",
}}},
}
}
pod := func(name, job string, min int, exit int32, msg string) *corev1.Pod {
return &corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: "minecraft", Labels: labels(job), CreationTimestamp: at(min)},
Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{
Name: "backup", State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: exit, Message: msg}},
}}},
}
}
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
failedJob("backup-survival-a", 1),
failedJob("backup-survival-b", 2),
pod("a-1", "backup-survival-a", 1, 1, "felis backup: first try\n"),
pod("a-2", "backup-survival-a", 3, 1,
"archiving survival\nfelis backup: backup: not enough free disk for the archive: the world is 2.0 GiB\n"),
pod("b-1", "backup-survival-b", 2, 0, "done"),
).Build()
jobs, err := NewK8sJobStatus(c, "minecraft").LatestJobs(context.Background(), "survival")
if err != nil {
t.Fatal(err)
}
got := map[string]string{}
for _, j := range jobs {
got[j.Name] = j.Message
}
if want := "felis backup: backup: not enough free disk for the archive: the world is 2.0 GiB"; got["backup-survival-a"] != want {
t.Errorf("a: message = %q, want %q", got["backup-survival-a"], want)
}
if want := "Job has reached the specified backoff limit"; got["backup-survival-b"] != want {
t.Errorf("b: message = %q, want the condition text", got["backup-survival-b"])
}
}
+59
View File
@@ -3,6 +3,8 @@ package api
import ( import (
"context" "context"
"sort" "sort"
"strings"
"time"
batchv1 "k8s.io/api/batch/v1" batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1" corev1 "k8s.io/api/core/v1"
@@ -53,9 +55,66 @@ func (k *K8sJobStatus) LatestJobs(ctx context.Context, serverName string) ([]Asy
if len(out) > 20 { if len(out) > 20 {
out = out[:20] out = out[:20]
} }
k.explainFailures(ctx, serverName, out)
return out, nil return out, nil
} }
// explainFailures replaces a failed Job's condition text ("Job has reached the
// specified backoff limit") with the error its container exited on. The
// executors set TerminationMessagePolicy FallbackToLogsOnError, so the
// terminated state carries the tail of the log, whose last line is the
// "felis backup: …" / "felis restore: …" error. The pods carry the Job's
// labels, so one list covers every Job of the server; a pod already collected
// by the Job TTL, or a list error, leaves the condition text in place.
func (k *K8sJobStatus) explainFailures(ctx context.Context, serverName string, jobs []AsyncJob) {
failed := map[string]int{}
for i, j := range jobs {
if j.State == "failed" {
failed[j.Name] = i
}
}
if len(failed) == 0 {
return
}
var pods corev1.PodList
if err := k.c.List(ctx, &pods, client.InNamespace(k.namespace),
client.MatchingLabels{jobServerLabel: serverName}); err != nil {
return
}
newest := map[string]time.Time{}
for i := range pods.Items {
pod := &pods.Items[i]
idx, ok := failed[pod.Labels["job-name"]]
if !ok {
continue
}
msg := lastTerminationLine(pod)
if msg == "" || !pod.CreationTimestamp.Time.After(newest[pod.Labels["job-name"]]) {
continue
}
newest[pod.Labels["job-name"]] = pod.CreationTimestamp.Time
jobs[idx].Message = msg
}
}
// lastTerminationLine returns the last non-empty line of the pod's terminated
// container message, capped for display.
func lastTerminationLine(pod *corev1.Pod) string {
for _, cs := range pod.Status.ContainerStatuses {
t := cs.State.Terminated
if t == nil || t.ExitCode == 0 {
continue
}
lines := strings.Split(strings.TrimSpace(t.Message), "\n")
line := strings.TrimSpace(lines[len(lines)-1])
if len(line) > 400 {
line = line[:400] + "…"
}
return line
}
return ""
}
// jobToAsyncJob projects one Job onto its kind/state/message. Complete condition → // jobToAsyncJob projects one Job onto its kind/state/message. Complete condition →
// succeeded, Failed → failed with its reason (Job conditions carry the generic // succeeded, Failed → failed with its reason (Job conditions carry the generic
// "backoff limit exceeded" text; the pod log holds the underlying error), anything // "backoff limit exceeded" text; the pod log holds the underlying error), anything
+22
View File
@@ -747,6 +747,28 @@ func (p *PGRepo) BackupByID(ctx context.Context, id string) (*BackupRecord, erro
return &b, nil return &b, nil
} }
// LastBackupRequest reads the newest backup.create audit row for the server
// since the given time; the created_at index bounds the scan to that window.
func (p *PGRepo) LastBackupRequest(ctx context.Context, serverName string, since time.Time) (time.Time, error) {
var at sql.NullTime
err := p.db.QueryRowContext(ctx,
`SELECT max(created_at) FROM audit_logs
WHERE created_at >= $2 AND action = 'backup.create' AND server_name = $1`,
serverName, since).Scan(&at)
if err != nil {
return time.Time{}, err
}
return at.Time, nil
}
// BackupStoreBytes sums size_bytes over the present world backups.
func (p *PGRepo) BackupStoreBytes(ctx context.Context) (int64, error) {
var n int64
err := p.db.QueryRowContext(ctx,
`SELECT COALESCE(sum(size_bytes), 0) FROM world_backups WHERE status = 'present'`).Scan(&n)
return n, err
}
func (p *PGRepo) Audit(ctx context.Context, e AuditEntry) error { func (p *PGRepo) Audit(ctx context.Context, e AuditEntry) error {
// A nil Payload must land as SQL NULL, not the text "null"; a non-nil Payload is // A nil Payload must land as SQL NULL, not the text "null"; a non-nil Payload is
// passed as a JSON text the jsonb column parses (same idiom as reaper.PGStore). // passed as a JSON text the jsonb column parses (same idiom as reaper.PGStore).
+8
View File
@@ -297,6 +297,14 @@ type Repo interface {
// none matches. Like LatestBackup the returned BackupRecord carries the // none matches. Like LatestBackup the returned BackupRecord carries the
// server-side backup_ref the restore path needs; the client never sees it. // server-side backup_ref the restore path needs; the client never sees it.
BackupByID(ctx context.Context, id string) (*BackupRecord, error) BackupByID(ctx context.Context, id string) (*BackupRecord, error)
// LastBackupRequest returns when an on-demand backup of the server was last
// accepted (its newest backup.create audit row) at or after since, or the zero
// time when there was none. The since bound keeps the lookup inside the
// cooldown window the caller enforces.
LastBackupRequest(ctx context.Context, serverName string, since time.Time) (time.Time, error)
// BackupStoreBytes sums the sizes of every present world backup, the figure
// [archive] max_local_bytes caps.
BackupStoreBytes(ctx context.Context) (int64, error)
// SeedServer inserts the business-layer rows for a newly created server (spec // SeedServer inserts the business-layer rows for a newly created server (spec
// §15): a servers row (owner_id NULL — claimed later, spec §9.3) and its // §15): a servers row (owner_id NULL — claimed later, spec §9.3) and its
// subdomain alias, both idempotent. The resource cache (cpuMilli, memoryMB, // subdomain alias, both idempotent. The resource cache (cpuMilli, memoryMB,
+93
View File
@@ -0,0 +1,93 @@
package backup
import (
"errors"
"fmt"
"io/fs"
"os"
"path/filepath"
"syscall"
)
// MinFreeAfter is the share of the archive filesystem an on-demand backup must
// leave free. The archive store sits on the node's disk beside the worlds and
// the database; below about a tenth free the kubelet starts evicting pods
// (docs/troubleshooting.md §13b), so a backup that would cross it is refused.
const MinFreeAfter = 0.10
// ErrNoRoom is returned by CheckRoom when the archive would leave too little
// free space.
var ErrNoRoom = errors.New("backup: not enough free disk for the archive")
// CheckRoom refuses an archive of srcDir into archiveDir that could push the
// archive filesystem below minFree free. The world's uncompressed size stands
// in for the archive's, which gzip only makes smaller. archiveDir need not
// exist yet; its nearest existing parent is measured.
func CheckRoom(archiveDir, srcDir string, minFree float64) error {
dir := archiveDir
for {
if _, err := os.Stat(dir); err == nil {
break
}
parent := filepath.Dir(dir)
if parent == dir {
return fmt.Errorf("backup: no existing directory above %s", archiveDir)
}
dir = parent
}
var st syscall.Statfs_t
if err := syscall.Statfs(dir, &st); err != nil {
return fmt.Errorf("backup: measure %s: %w", dir, err)
}
bsize := uint64(st.Bsize) // uint32 on darwin
total := uint64(st.Blocks) * bsize
avail := uint64(st.Bavail) * bsize
if total == 0 {
return nil
}
need, err := treeBytes(srcDir)
if err != nil {
return err
}
floor := uint64(float64(total) * minFree)
if avail < need || avail-need < floor {
return fmt.Errorf("%w: the world is %s and %s is free of %s, which would leave less than %.0f%% free",
ErrNoRoom, byteSize(need), byteSize(avail), byteSize(total), minFree*100)
}
return nil
}
// treeBytes sums the sizes of the regular files under dir.
func treeBytes(dir string) (uint64, error) {
var n uint64
err := filepath.WalkDir(dir, func(_ string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if d.Type().IsRegular() {
info, err := d.Info()
if err != nil {
return err
}
n += uint64(info.Size())
}
return nil
})
if err != nil {
return 0, fmt.Errorf("backup: measure the world: %w", err)
}
return n, nil
}
func byteSize(b uint64) string {
const unit = 1024
if b < unit {
return fmt.Sprintf("%d B", b)
}
div, exp := uint64(unit), 0
for n := b / unit; n >= unit; n /= unit {
div *= unit
exp++
}
return fmt.Sprintf("%.1f %ciB", float64(b)/float64(div), "KMGTPE"[exp])
}
+26
View File
@@ -0,0 +1,26 @@
package backup
import (
"errors"
"os"
"path/filepath"
"testing"
)
func TestCheckRoom(t *testing.T) {
src := t.TempDir()
if err := os.WriteFile(filepath.Join(src, "level.dat"), make([]byte, 4096), 0o600); err != nil {
t.Fatal(err)
}
archives := filepath.Join(t.TempDir(), "not", "made", "yet")
if err := CheckRoom(archives, src, 0); err != nil {
t.Fatalf("no floor: %v", err)
}
// No disk is ever entirely free, so a 100% floor always refuses.
if err := CheckRoom(archives, src, 1); !errors.Is(err, ErrNoRoom) {
t.Fatalf("full floor: err = %v, want ErrNoRoom", err)
}
if err := CheckRoom(archives, filepath.Join(src, "absent"), 0); err == nil {
t.Fatal("a missing world measured as empty")
}
}
+4
View File
@@ -159,6 +159,10 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
}, },
} }
// The exit error reaches GET /servers/{name}/jobs through the terminated
// state (api.K8sJobStatus), in place of the Job's generic backoff text.
container.TerminationMessagePolicy = corev1.TerminationMessageFallbackToLogsOnError
name := p.JobName name := p.JobName
if name == "" { if name == "" {
name = BackupJobName(p.Server) name = BackupJobName(p.Server)
+13 -6
View File
@@ -220,13 +220,20 @@ type RegistryS3Config struct {
} }
// ArchiveConfig is the [archive] table plus its [archive.s3] subtable (spec §19). // ArchiveConfig is the [archive] table plus its [archive.s3] subtable (spec §19).
// The manual_* keys bound the owners' on-demand backups, which share the
// archive store with the reaper's: how long each is kept (default 30d), how
// many per server (default 5, the oldest go first), and how soon an owner may
// ask for the next one (default 10m). Empty or zero means the default.
type ArchiveConfig struct { type ArchiveConfig struct {
Store string `toml:"store"` Store string `toml:"store"`
LocalPath string `toml:"local_path"` LocalPath string `toml:"local_path"`
Retention string `toml:"retention"` Retention string `toml:"retention"`
WarnBefore []string `toml:"warn_before"` WarnBefore []string `toml:"warn_before"`
MaxLocalBytes string `toml:"max_local_bytes"` MaxLocalBytes string `toml:"max_local_bytes"`
S3 ArchiveS3Config `toml:"s3"` ManualRetention string `toml:"manual_retention"`
ManualKeep int `toml:"manual_keep"`
ManualCooldown string `toml:"manual_cooldown"`
S3 ArchiveS3Config `toml:"s3"`
} }
// ArchiveS3Config is the [archive.s3] subtable. // ArchiveS3Config is the [archive.s3] subtable.
+94
View File
@@ -5,9 +5,11 @@ package pgint
import ( import (
"context" "context"
"database/sql" "database/sql"
"strings"
"testing" "testing"
"time" "time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/backup" "felis.lolicon.best/internal/backup"
"felis.lolicon.best/internal/reaper" "felis.lolicon.best/internal/reaper"
) )
@@ -139,3 +141,95 @@ func TestReclaimRestartsReaperClock(t *testing.T) {
t.Fatalf("owner = %v, newest backup = %q; want released and %q", owner, newest, ar.archived[0]) t.Fatalf("owner = %v, newest backup = %q; want released and %q", owner, newest, ar.archived[0])
} }
} }
// TestManualBackupRationing pins the SQL behind data-durability-9: keep-N
// pruning picks a server's oldest on-demand backups, capacity eviction takes
// on-demand backups before copied reaper archives and never offers the only
// copy of a reaped world, and the API's cooldown and store-size reads see the
// rows the rest of the platform writes.
func TestManualBackupRationing(t *testing.T) {
ctx := context.Background()
st := reaper.NewPGStore(db)
name := "ration-" + suffix(t)
if _, err := db.ExecContext(ctx,
`INSERT INTO servers (name, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, 100, 128, 1)`,
name); err != nil {
t.Fatalf("seed server: %v", err)
}
storeBefore, err := repo.BackupStoreBytes(ctx)
if err != nil {
t.Fatalf("BackupStoreBytes: %v", err)
}
now := time.Now()
insert := func(id, reason string, created time.Time, offsite bool, size int64) {
t.Helper()
var offsiteAt any
if offsite {
offsiteAt = created
}
if _, err := db.ExecContext(ctx,
`INSERT INTO world_backups (id, server_name, backup_ref, size_bytes, reason, status, created_at, expires_at, offsite_at)
VALUES ($1, $2, $3, $4, $5, 'present', $6, $7, $8)`,
id, name, "/archives/"+id+".tar.gz", size, reason, created, created.Add(90*reaper.Day), offsiteAt); err != nil {
t.Fatalf("seed %s: %v", id, err)
}
}
sfx := suffix(t)
var manual []string
for i := 0; i < 7; i++ {
id := "bk-m" + string(rune('0'+i)) + "-" + sfx
manual = append(manual, id)
insert(id, "manual", now.Add(time.Duration(i-7)*time.Hour), false, 10)
}
sole := "bk-sole-" + sfx
copied := "bk-copied-" + sfx
insert(sole, "inactive_15d", now.Add(-100*reaper.Day), false, 1000)
insert(copied, "inactive_15d", now.Add(-50*reaper.Day), true, 100)
excess, err := st.ExcessManualBackups(ctx, name, 5)
if err != nil {
t.Fatalf("ExcessManualBackups: %v", err)
}
if len(excess) != 2 || excess[0].ID != manual[0] || excess[1].ID != manual[1] {
t.Fatalf("excess = %+v; want the two oldest manual backups, oldest first", excess)
}
all, err := st.EvictableBackups(ctx)
if err != nil {
t.Fatalf("EvictableBackups: %v", err)
}
var got []string
for _, b := range all {
if b.ServerName == name {
got = append(got, b.ID)
}
}
want := append(append([]string{}, manual...), copied)
if strings.Join(got, ",") != strings.Join(want, ",") {
t.Fatalf("eviction order = %v\nwant %v (manual oldest first, then the copied archive, never the sole copy)", got, want)
}
storeAfter, err := repo.BackupStoreBytes(ctx)
if err != nil {
t.Fatalf("BackupStoreBytes: %v", err)
}
if storeAfter-storeBefore != 7*10+1000+100 {
t.Fatalf("store grew by %d, want %d", storeAfter-storeBefore, 7*10+1000+100)
}
if at, err := repo.LastBackupRequest(ctx, name, now.Add(-10*time.Minute)); err != nil || !at.IsZero() {
t.Fatalf("before any request: LastBackupRequest = (%v, %v)", at, err)
}
if err := repo.Audit(ctx, api.AuditEntry{Actor: "[email protected]", Source: "external",
Action: "backup.create", ServerName: name}); err != nil {
t.Fatalf("Audit: %v", err)
}
at, err := repo.LastBackupRequest(ctx, name, time.Now().Add(-10*time.Minute))
if err != nil || at.IsZero() || time.Since(at) > time.Minute {
t.Fatalf("after a request: LastBackupRequest = (%v, %v)", at, err)
}
if at, err := repo.LastBackupRequest(ctx, name, time.Now().Add(time.Minute)); err != nil || !at.IsZero() {
t.Fatalf("a request before since still counted: (%v, %v)", at, err)
}
}
+19 -5
View File
@@ -106,14 +106,28 @@ func (s *PGStore) PresentBackupBytes(ctx context.Context) (int64, error) {
return n, err return n, err
} }
func (s *PGStore) OldestPresentBackups(ctx context.Context) ([]StoredBackup, error) { func (s *PGStore) EvictableBackups(ctx context.Context) ([]StoredBackup, error) {
const q = `SELECT id, server_name, backup_ref, size_bytes FROM world_backups const q = `SELECT id, server_name, backup_ref, size_bytes, reason FROM world_backups
WHERE status = 'present' ORDER BY created_at ASC` WHERE status = 'present' AND (reason <> 'inactive_15d' OR offsite_at IS NOT NULL)
ORDER BY reason = 'inactive_15d', created_at ASC`
return s.queryBackups(ctx, q) return s.queryBackups(ctx, q)
} }
// ExcessManualBackups lists server's present on-demand backups beyond the
// newest keep, oldest first: what the backup Job removes after adding one.
func (s *PGStore) ExcessManualBackups(ctx context.Context, server string, keep int) ([]StoredBackup, error) {
const q = `SELECT id, server_name, backup_ref, size_bytes, reason FROM world_backups
WHERE server_name = $1 AND status = 'present' AND reason = 'manual'
ORDER BY created_at DESC OFFSET $2`
out, err := s.queryBackups(ctx, q, server, keep)
for i, j := 0, len(out)-1; i < j; i, j = i+1, j-1 {
out[i], out[j] = out[j], out[i]
}
return out, err
}
func (s *PGStore) ListExpiredBackups(ctx context.Context, now time.Time) ([]StoredBackup, error) { func (s *PGStore) ListExpiredBackups(ctx context.Context, now time.Time) ([]StoredBackup, error) {
const q = `SELECT id, server_name, backup_ref, size_bytes FROM world_backups const q = `SELECT id, server_name, backup_ref, size_bytes, reason FROM world_backups
WHERE status = 'present' AND expires_at < $1 ORDER BY expires_at ASC` WHERE status = 'present' AND expires_at < $1 ORDER BY expires_at ASC`
return s.queryBackups(ctx, q, now) return s.queryBackups(ctx, q, now)
} }
@@ -127,7 +141,7 @@ func (s *PGStore) queryBackups(ctx context.Context, q string, args ...any) ([]St
var out []StoredBackup var out []StoredBackup
for rows.Next() { for rows.Next() {
var b StoredBackup var b StoredBackup
if err := rows.Scan(&b.ID, &b.ServerName, &b.BackupRef, &b.SizeBytes); err != nil { if err := rows.Scan(&b.ID, &b.ServerName, &b.BackupRef, &b.SizeBytes, &b.Reason); err != nil {
return nil, err return nil, err
} }
out = append(out, b) out = append(out, b)
+28 -5
View File
@@ -81,6 +81,16 @@ type Config struct {
WarnBefore []time.Duration // §24: warn these long before the deadline (default 3d, 1d) WarnBefore []time.Duration // §24: warn these long before the deadline (default 3d, 1d)
Retention time.Duration // §18: keep a backup this long after deletion (default 3mo≈90d) Retention time.Duration // §18: keep a backup this long after deletion (default 3mo≈90d)
MaxLocalBytes int64 // §26: backup store soft cap; 0 = unlimited MaxLocalBytes int64 // §26: backup store soft cap; 0 = unlimited
// ManualRetention is how long an owner's on-demand backup is kept; it is
// a restore point for a world that still exists, so it goes sooner than a
// reaped world's only archive (default 30d).
ManualRetention time.Duration
// ManualKeep caps the on-demand backups kept per server; the backup Job
// removes the oldest beyond it (default 5).
ManualKeep int
// ManualCooldown is the shortest gap between two owner-requested backups
// of one server (default 10m); operators are not held to it.
ManualCooldown time.Duration
// RequireOffsite holds each deletion until the world's archive has its // RequireOffsite holds each deletion until the world's archive has its
// off-site copy ([offsite] configured; internal/offsite records the copy). // off-site copy ([offsite] configured; internal/offsite records the copy).
// The archive is written on the run that finds the world idle, and the // The archive is written on the run that finds the world idle, and the
@@ -96,6 +106,10 @@ func DefaultConfig() Config {
WarnBefore: []time.Duration{3 * Day, 1 * Day}, WarnBefore: []time.Duration{3 * Day, 1 * Day},
Retention: 90 * Day, Retention: 90 * Day,
MaxLocalBytes: 0, MaxLocalBytes: 0,
ManualRetention: 30 * Day,
ManualKeep: 5,
ManualCooldown: 10 * time.Minute,
} }
} }
@@ -151,6 +165,7 @@ type StoredBackup struct {
ServerName string ServerName string
BackupRef string BackupRef string
SizeBytes int64 SizeBytes int64
Reason string
} }
// AuditRecord is a reaper-sourced audit_logs entry. The PG binding fills // AuditRecord is a reaper-sourced audit_logs entry. The PG binding fills
@@ -190,9 +205,12 @@ type Store interface {
// PresentBackupBytes is the total size of status=present backups (§26 cap). // PresentBackupBytes is the total size of status=present backups (§26 cap).
PresentBackupBytes(ctx context.Context) (int64, error) PresentBackupBytes(ctx context.Context) (int64, error)
// OldestPresentBackups lists status=present backups oldest-first, for // EvictableBackups lists the status=present backups that may go before
// early eviction when the store is full. // their expiry when the store is full, in eviction order: on-demand
OldestPresentBackups(ctx context.Context) ([]StoredBackup, error) // backups first, then reaper archives that have an off-site copy, oldest
// first within each. A reaper archive without an off-site copy is the only
// copy of a deleted world and is never listed.
EvictableBackups(ctx context.Context) ([]StoredBackup, error)
// ListExpiredBackups lists status=present backups whose expires_at < now. // ListExpiredBackups lists status=present backups whose expires_at < now.
ListExpiredBackups(ctx context.Context, now time.Time) ([]StoredBackup, error) ListExpiredBackups(ctx context.Context, now time.Time) ([]StoredBackup, error)
@@ -450,10 +468,10 @@ func (r *Reaper) ensureCapacity(ctx context.Context, now time.Time, sum *Summary
if used < r.Cfg.MaxLocalBytes { if used < r.Cfg.MaxLocalBytes {
return true, nil return true, nil
} }
r.log().Warn("reaper: backup store at capacity, evicting oldest backups early", r.log().Warn("reaper: backup store at capacity, evicting on-demand and off-site-copied backups early",
"used", used, "max", r.Cfg.MaxLocalBytes) "used", used, "max", r.Cfg.MaxLocalBytes)
old, err := r.Store.OldestPresentBackups(ctx) old, err := r.Store.EvictableBackups(ctx)
if err != nil { if err != nil {
return false, err return false, err
} }
@@ -474,6 +492,11 @@ func (r *Reaper) ensureCapacity(ctx context.Context, now time.Time, sum *Summary
} }
used -= b.SizeBytes used -= b.SizeBytes
sum.EvictedEarly++ sum.EvictedEarly++
r.log().Warn("reaper: backup evicted early", "id", b.ID, "server", b.ServerName, "reason", b.Reason, "bytes", b.SizeBytes)
}
if used >= r.Cfg.MaxLocalBytes {
r.log().Error("reaper: backup store still full; what remains are the only copies of reaped worlds, kept until they expire",
"used", used, "max", r.Cfg.MaxLocalBytes)
} }
return used < r.Cfg.MaxLocalBytes, nil return used < r.Cfg.MaxLocalBytes, nil
} }
+56 -7
View File
@@ -193,17 +193,22 @@ func (s *fakeStore) PresentBackupBytes(context.Context) (int64, error) {
return total, nil return total, nil
} }
func (s *fakeStore) OldestPresentBackups(context.Context) ([]StoredBackup, error) { func (s *fakeStore) EvictableBackups(context.Context) ([]StoredBackup, error) {
var ps []*fakeBackup var ps []*fakeBackup
for _, b := range s.backups { for _, b := range s.backups {
if b.status == "present" { if b.status == "present" && (b.reason != ReasonInactive || b.offsite) {
ps = append(ps, b) ps = append(ps, b)
} }
} }
sort.Slice(ps, func(i, j int) bool { return ps[i].createdAt.Before(ps[j].createdAt) }) sort.SliceStable(ps, func(i, j int) bool {
if ri, rj := ps[i].reason == ReasonInactive, ps[j].reason == ReasonInactive; ri != rj {
return rj
}
return ps[i].createdAt.Before(ps[j].createdAt)
})
out := make([]StoredBackup, 0, len(ps)) out := make([]StoredBackup, 0, len(ps))
for _, b := range ps { for _, b := range ps {
out = append(out, StoredBackup{ID: b.id, ServerName: b.server, BackupRef: b.ref, SizeBytes: b.size}) out = append(out, StoredBackup{ID: b.id, ServerName: b.server, BackupRef: b.ref, SizeBytes: b.size, Reason: b.reason})
} }
return out, nil return out, nil
} }
@@ -623,8 +628,8 @@ func TestCapacityEvictsOldestThenReaps(t *testing.T) {
Candidate{Name: "epsilon", OwnerID: "user-5", LastActiveAt: idleBy(20 * Day)}) Candidate{Name: "epsilon", OwnerID: "user-5", LastActiveAt: idleBy(20 * Day)})
// Two present backups of 75 each = 150 > 100. Oldest must be evicted first. // Two present backups of 75 each = 150 > 100. Oldest must be evicted first.
st.backups = []*fakeBackup{ st.backups = []*fakeBackup{
{id: "old", server: "zzz", ref: "ref-old", size: 75, status: "present", createdAt: idleBy(40 * Day), expires: testNow.Add(30 * Day)}, {id: "old", server: "zzz", ref: "ref-old", reason: "manual", size: 75, status: "present", createdAt: idleBy(40 * Day), expires: testNow.Add(30 * Day)},
{id: "new", server: "yyy", ref: "ref-new", size: 75, status: "present", createdAt: idleBy(5 * Day), expires: testNow.Add(60 * Day)}, {id: "new", server: "yyy", ref: "ref-new", reason: "manual", size: 75, status: "present", createdAt: idleBy(5 * Day), expires: testNow.Add(60 * Day)},
} }
sum := mustRun(t, r) sum := mustRun(t, r)
@@ -669,7 +674,7 @@ func TestCapacityStillFullSkipsReap(t *testing.T) {
r, st, cl, ar := newReaper(cfg, r, st, cl, ar := newReaper(cfg,
Candidate{Name: "zeta", OwnerID: "user-6", LastActiveAt: idleBy(20 * Day)}) Candidate{Name: "zeta", OwnerID: "user-6", LastActiveAt: idleBy(20 * Day)})
st.backups = []*fakeBackup{ st.backups = []*fakeBackup{
{id: "stuck", server: "zzz", ref: "ref-stuck", size: 150, status: "present", createdAt: idleBy(40 * Day), expires: testNow.Add(30 * Day)}, {id: "stuck", server: "zzz", ref: "ref-stuck", reason: "manual", size: 150, status: "present", createdAt: idleBy(40 * Day), expires: testNow.Add(30 * Day)},
} }
// The archive backend can't delete, so eviction cannot free space. // The archive backend can't delete, so eviction cannot free space.
r.Archiver.(*fakeArchiver).deleteErr = errors.New("evict unavailable") r.Archiver.(*fakeArchiver).deleteErr = errors.New("evict unavailable")
@@ -689,6 +694,50 @@ func TestCapacityStillFullSkipsReap(t *testing.T) {
} }
} }
// Eviction order: on-demand backups go first, then reaper archives that have an
// off-site copy; the only copy of a reaped world is never evicted early, even
// when that leaves the store full and the idle world waits.
func TestCapacityEvictionOrderSparesSoleCopies(t *testing.T) {
cfg := DefaultConfig()
cfg.MaxLocalBytes = 100
r, st, _, _ := newReaper(cfg,
Candidate{Name: "theta", OwnerID: "user-8", LastActiveAt: idleBy(20 * Day)})
st.backups = []*fakeBackup{
{id: "sole", server: "gone1", ref: "ref-sole", reason: ReasonInactive, size: 60, status: "present", createdAt: idleBy(80 * Day), expires: testNow.Add(10 * Day)},
{id: "copied", server: "gone2", ref: "ref-copied", reason: ReasonInactive, offsite: true, size: 30, status: "present", createdAt: idleBy(70 * Day), expires: testNow.Add(20 * Day)},
{id: "man", server: "live", ref: "ref-man", reason: "manual", size: 30, status: "present", createdAt: idleBy(2 * Day), expires: testNow.Add(28 * Day)},
}
sum := mustRun(t, r)
status := map[string]string{}
for _, b := range st.backups {
status[b.id] = b.status
}
// 120 over a cap of 100: the on-demand backup (30) alone brings it to 90.
if status["man"] != "deleted" || status["copied"] != "present" || status["sole"] != "present" {
t.Fatalf("evicted %v; want only the on-demand backup", status)
}
if sum.EvictedEarly != 1 || sum.WorldsReaped != 1 {
t.Fatalf("summary = %+v, want 1 evicted, 1 reaped", sum)
}
// Over a cap of 50 with only reaper archives left: the copied one goes,
// the sole copy stays, the store is still full, and the idle world waits.
cfg.MaxLocalBytes = 50
r2, st2, cl2, _ := newReaper(cfg,
Candidate{Name: "iota", OwnerID: "user-9", LastActiveAt: idleBy(20 * Day)})
st2.backups = []*fakeBackup{
{id: "sole", server: "gone1", ref: "ref-sole", reason: ReasonInactive, size: 60, status: "present", createdAt: idleBy(80 * Day), expires: testNow.Add(10 * Day)},
{id: "copied", server: "gone2", ref: "ref-copied", reason: ReasonInactive, offsite: true, size: 30, status: "present", createdAt: idleBy(70 * Day), expires: testNow.Add(20 * Day)},
}
sum2 := mustRun(t, r2)
if st2.backups[0].status != "present" || st2.backups[1].status != "deleted" {
t.Fatalf("sole=%s copied=%s, want present/deleted", st2.backups[0].status, st2.backups[1].status)
}
if sum2.StoreFull != 1 || sum2.WorldsReaped != 0 || cl2.deletePVCCalls != 0 {
t.Fatalf("summary = %+v, deletes = %d: the world should wait while the store is full", sum2, cl2.deletePVCCalls)
}
}
// Retention pass: backups past expires_at are deleted from the backend and // Retention pass: backups past expires_at are deleted from the backend and
// marked deleted; unexpired backups are untouched. // marked deleted; unexpired backups are untouched.
func TestExpiredBackupsDeleted(t *testing.T) { func TestExpiredBackupsDeleted(t *testing.T) {
+4
View File
@@ -138,6 +138,10 @@ func RestoreJob(p JobParams) (*batchv1.Job, error) {
}, },
} }
// The exit error reaches GET /servers/{name}/jobs through the terminated
// state (api.K8sJobStatus), in place of the Job's generic backoff text.
container.TerminationMessagePolicy = corev1.TerminationMessageFallbackToLogsOnError
job := &batchv1.Job{ job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{ ObjectMeta: metav1.ObjectMeta{
Name: RestoreJobName(p.Server), Name: RestoreJobName(p.Server),
+1 -1
View File
@@ -5,7 +5,7 @@
"not_yours_title": "No permission to access backups", "not_yours_title": "No permission to access backups",
"not_yours_body": "Only the owner or an admin can view and restore this server's backups.", "not_yours_body": "Only the owner or an admin can view and restore this server's backups.",
"latest_title": "Latest backup", "latest_title": "Latest backup",
"history_note": "Historical backups can be used for restore before they expire. They are automatically cleaned up when they expire.", "history_note": "Backups can be restored until they expire, then they are cleaned up automatically. Each server keeps only its most recent manual backups; older manual backups are removed once a new one finishes.",
"reason_inactive": "Idle archive", "reason_inactive": "Idle archive",
"reason_manual": "Manual backup", "reason_manual": "Manual backup",
"reason_label": "Reason: {{reason}}", "reason_label": "Reason: {{reason}}",
@@ -19,6 +19,8 @@
"not_stopped": "Stop the server completely before restoring — a restore overwrites the live world volume.", "not_stopped": "Stop the server completely before restoring — a restore overwrites the live world volume.",
"maintenance_in_progress": "This server's world is busy with a restore, backup or file write — try again once it finishes, usually within a minute or two.", "maintenance_in_progress": "This server's world is busy with a restore, backup or file write — try again once it finishes, usually within a minute or two.",
"no_world_volume": "This server has no world volume yet — start it once so it is created, then retry.", "no_world_volume": "This server has no world volume yet — start it once so it is created, then retry.",
"backup_cooldown": "This server was backed up moments ago, and manual backups have a cooldown — try again in a few minutes.",
"backup_store_full": "The backup store is full, so manual backups are paused — ask an administrator to free space.",
"restore_unavailable": "Restore isn't available right now — try again later.", "restore_unavailable": "Restore isn't available right now — try again later.",
"session_expired": "Your session expired — please sign in again.", "session_expired": "Your session expired — please sign in again.",
"forbidden": "You are not allowed to do that.", "forbidden": "You are not allowed to do that.",
+1 -1
View File
@@ -5,7 +5,7 @@
"not_yours_title": "无权访问备份", "not_yours_title": "无权访问备份",
"not_yours_body": "只有所有者或管理员才能查看并恢复该服务器的备份。", "not_yours_body": "只有所有者或管理员才能查看并恢复该服务器的备份。",
"latest_title": "最新备份", "latest_title": "最新备份",
"history_note": "历史备份在过期前均可用于恢复。到期后系统会自动清理,无需手动删除或管理。", "history_note": "历史备份在过期前均可用于恢复,到期后自动清理。手动备份每台服务器只保留最近几份,新备份完成后会自动删除更早的手动备份。",
"reason_inactive": "闲置自动回收", "reason_inactive": "闲置自动回收",
"reason_manual": "手动备份", "reason_manual": "手动备份",
"reason_label": "原因:{{reason}}", "reason_label": "原因:{{reason}}",
@@ -19,6 +19,8 @@
"not_stopped": "回档会覆盖世界的实时存储卷,请先把服务器完全停止再回档。", "not_stopped": "回档会覆盖世界的实时存储卷,请先把服务器完全停止再回档。",
"maintenance_in_progress": "这台服务器的世界正在回档、备份或写入文件——等它完成后再试,通常不超过一两分钟。", "maintenance_in_progress": "这台服务器的世界正在回档、备份或写入文件——等它完成后再试,通常不超过一两分钟。",
"no_world_volume": "这台服务器还没有世界卷——先启动一次让它创建,然后再试。", "no_world_volume": "这台服务器还没有世界卷——先启动一次让它创建,然后再试。",
"backup_cooldown": "这台服务器刚备份过,手动备份之间有冷却时间——请过几分钟再试。",
"backup_store_full": "备份存储已满,暂时无法手动备份——请联系管理员清理空间。",
"restore_unavailable": "回档功能当前不可用,请稍后再试。", "restore_unavailable": "回档功能当前不可用,请稍后再试。",
"session_expired": "会话已过期——请重新登录。", "session_expired": "会话已过期——请重新登录。",
"forbidden": "你无权执行此操作。", "forbidden": "你无权执行此操作。",
+6
View File
@@ -390,6 +390,12 @@ describe("api access-control wire shapes", () => {
expect(humanizeError({ code: "not_running" })).toMatch(/running|wake/i); expect(humanizeError({ code: "not_running" })).toMatch(/running|wake/i);
expect(humanizeError({ code: "console_unavailable" })).toMatch(/console/i); expect(humanizeError({ code: "console_unavailable" })).toMatch(/console/i);
}); });
it("maps the backup rationing codes to their own copy", async () => {
const { humanizeError } = await import("./api");
expect(humanizeError({ code: "backup_cooldown" })).toMatch(/cooldown/i);
expect(humanizeError({ code: "backup_store_full" })).toMatch(/store is full/i);
});
}); });
describe("image whitelist and builds wire shapes", () => { describe("image whitelist and builds wire shapes", () => {
+6
View File
@@ -739,6 +739,12 @@ export function humanizeError(e: unknown): string {
return t("maintenance_in_progress"); return t("maintenance_in_progress");
case "no_world_volume": case "no_world_volume":
return t("no_world_volume"); return t("no_world_volume");
// On-demand backup rationing (data-durability-9): one per server per
// cooldown, none while the shared backup store is at its cap.
case "backup_cooldown":
return t("backup_cooldown");
case "backup_store_full":
return t("backup_store_full");
case "restore_unavailable": case "restore_unavailable":
return t("restore_unavailable"); return t("restore_unavailable");
// Server create/edit (spec §22): the portability regex + reservation list are // Server create/edit (spec §22): the portability regex + reservation list are