feat(backup): 玩过的世界每天在停服后自动打一个定时恢复点,按服按 owner 保留 7 个、90 天过期
This commit is contained in:
29 files changed
+1019
-50
No files matched your search
@@ -440,6 +440,13 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
// reconciles it, but this loop converges builds nobody is polling.
|
// reconciles it, but this loop converges builds nobody is polling.
|
||||||
go reconcileBuilds(ctx, builder, stderr)
|
go reconcileBuilds(ctx, builder, stderr)
|
||||||
go settleRestoreChains(ctx, a, stderr)
|
go settleRestoreChains(ctx, a, stderr)
|
||||||
|
// A daily restore point of every world played since its last one, taken
|
||||||
|
// once the server stops ([archive] scheduled_every; 0s turns it off).
|
||||||
|
if backuper != nil && rcfg.ScheduledEvery > 0 {
|
||||||
|
go scheduleBackups(ctx, &api.BackupScheduler{API: a, Store: repo, Jobs: jobStatus, Every: rcfg.ScheduledEvery}, stderr)
|
||||||
|
} else {
|
||||||
|
fmt.Fprintln(stderr, "felis api: scheduled backups off (needs the backup executor and [archive] scheduled_every above 0s)")
|
||||||
|
}
|
||||||
|
|
||||||
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
|
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
|
||||||
go pruner.Loop(ctx, registryPruneInterval)
|
go pruner.Loop(ctx, registryPruneInterval)
|
||||||
@@ -703,6 +710,25 @@ func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// scheduleBackups starts the scheduled backups (api.BackupScheduler). Each tick
|
||||||
|
// starts at most one, so the interval also spaces the worlds that stopped at
|
||||||
|
// the same time: a world that stops waits at most this long for its point to
|
||||||
|
// start once the Jobs ahead of it are done.
|
||||||
|
func scheduleBackups(ctx context.Context, s *api.BackupScheduler, stderr io.Writer) {
|
||||||
|
t := time.NewTicker(2 * time.Minute)
|
||||||
|
defer t.Stop()
|
||||||
|
for {
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return
|
||||||
|
case <-t.C:
|
||||||
|
if err := s.Tick(ctx); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis api: scheduled backups: %v\n", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
|
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
|
||||||
// submissions rejected more than submit.RejectedContextRetention ago, and the
|
// submissions rejected more than submit.RejectedContextRetention ago, and the
|
||||||
// chunked uploads left untouched for submit.StalePartRetention. Without it a
|
// chunked uploads left untouched for submit.StalePartRetention. Without it a
|
||||||
|
|||||||
+34
-18
@@ -38,7 +38,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
server := fs.String("server", "", "server name whose world is being backed up")
|
server := fs.String("server", "", "server name whose world is being backed up")
|
||||||
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
|
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
|
||||||
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
|
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
|
||||||
reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
|
reason := fs.String("reason", reasonManual, "world_backups reason: manual, pre_restore for the safety snapshot in front of a restore, or scheduled for felis-api's daily restore point")
|
||||||
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
|
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return 2
|
return 2
|
||||||
@@ -47,9 +47,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
fmt.Fprintln(stderr, "felis backup: --server is required")
|
fmt.Fprintln(stderr, "felis backup: --server is required")
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
|
if _, _, ok := backupPolicy(*reason, reaper.DefaultConfig()); !ok {
|
||||||
if !ok {
|
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual, %s or %s)\n", *reason, backupjob.ReasonPreRestore, backupjob.ReasonScheduled)
|
||||||
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
|
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -62,16 +61,14 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
|
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
// The [archive] parse the reaper uses; an on-demand backup takes its
|
// The [archive] parse the reaper uses: it holds each reason's keep and
|
||||||
// manual_retention and manual_keep.
|
// retention.
|
||||||
rcfg, err := reaperConfig(cfg)
|
rcfg, err := reaperConfig(cfg)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if keep < 0 {
|
keep, retention, _ := backupPolicy(*reason, rcfg)
|
||||||
keep = rcfg.ManualKeep
|
|
||||||
}
|
|
||||||
|
|
||||||
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
|
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
|
||||||
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
|
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
|
||||||
@@ -117,7 +114,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
BackupRef: string(ref),
|
BackupRef: string(ref),
|
||||||
SizeBytes: size,
|
SizeBytes: size,
|
||||||
Reason: *reason,
|
Reason: *reason,
|
||||||
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
|
ExpiresAt: time.Now().Add(retention),
|
||||||
|
|
||||||
SHA256: a.SHA256,
|
SHA256: a.SHA256,
|
||||||
SkippedEntries: len(a.Skipped),
|
SkippedEntries: len(a.Skipped),
|
||||||
@@ -136,7 +133,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
||||||
pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
|
pruneBackups(ctx, st, archiver, *server, *formerOwner, *reason, keep, *protect, stdout, stderr)
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -148,13 +145,32 @@ const (
|
|||||||
preRestoreKeep = 3
|
preRestoreKeep = 3
|
||||||
)
|
)
|
||||||
|
|
||||||
// pruneBackups keeps server's newest keep backups of this reason and removes the
|
// backupPolicy is how many backups of one reason a server keeps and how long
|
||||||
// rest, oldest first, so repeated backups of one world cannot fill the shared
|
// each lives: an owner's own backups and the safety snapshots in front of a
|
||||||
// archive store. protect is never removed: it is the backup a chained restore is
|
// restore by [archive] manual_keep / manual_retention (the snapshots capped at
|
||||||
// about to extract. The new backup is already recorded; a removal that fails is
|
// preRestoreKeep), felis-api's daily restore points by scheduled_keep /
|
||||||
// reported and retried after the next backup.
|
// scheduled_retention, so neither kind crowds out the other. ok is false for a
|
||||||
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
|
// reason this command does not record.
|
||||||
excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
|
func backupPolicy(reason string, rcfg reaper.Config) (keep int, retention time.Duration, ok bool) {
|
||||||
|
switch reason {
|
||||||
|
case reasonManual:
|
||||||
|
return rcfg.ManualKeep, rcfg.ManualRetention, true
|
||||||
|
case backupjob.ReasonPreRestore:
|
||||||
|
return preRestoreKeep, rcfg.ManualRetention, true
|
||||||
|
case backupjob.ReasonScheduled:
|
||||||
|
return rcfg.ScheduledKeep, rcfg.ScheduledRetention, true
|
||||||
|
}
|
||||||
|
return 0, 0, false
|
||||||
|
}
|
||||||
|
|
||||||
|
// pruneBackups keeps the newest keep backups of this reason that owner holds of
|
||||||
|
// server and removes the rest, oldest first, so repeated backups of one world
|
||||||
|
// cannot fill the shared archive store and a new owner's backups never remove a
|
||||||
|
// previous owner's. protect is never removed: it is the backup a chained restore
|
||||||
|
// is about to extract. The new backup is already recorded; a removal that fails
|
||||||
|
// is reported and retried after the next backup.
|
||||||
|
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, owner, reason string, keep int, protect string, stdout, stderr io.Writer) {
|
||||||
|
excess, err := st.ExcessBackups(ctx, server, owner, reason, keep, protect)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
|
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
|
||||||
return
|
return
|
||||||
|
|||||||
@@ -4,6 +4,9 @@ import (
|
|||||||
"bytes"
|
"bytes"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/reaper"
|
||||||
)
|
)
|
||||||
|
|
||||||
// The reason decides which backups the new one's prune may remove, so an
|
// The reason decides which backups the new one's prune may remove, so an
|
||||||
@@ -17,3 +20,29 @@ func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
|
|||||||
t.Fatalf("stderr = %q", stderr.String())
|
t.Fatalf("stderr = %q", stderr.String())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Each reason is pruned and expired by its own [archive] keys: a daily
|
||||||
|
// restore point must never count against, or take the lifetime of, the
|
||||||
|
// backups an owner asked for.
|
||||||
|
func TestBackupPolicyPerReason(t *testing.T) {
|
||||||
|
rcfg := reaper.DefaultConfig()
|
||||||
|
rcfg.ManualKeep, rcfg.ManualRetention = 5, 30*reaper.Day
|
||||||
|
rcfg.ScheduledKeep, rcfg.ScheduledRetention = 7, 90*reaper.Day
|
||||||
|
for _, tc := range []struct {
|
||||||
|
reason string
|
||||||
|
keep int
|
||||||
|
retention time.Duration
|
||||||
|
}{
|
||||||
|
{"manual", 5, 30 * reaper.Day},
|
||||||
|
{"pre_restore", preRestoreKeep, 30 * reaper.Day},
|
||||||
|
{"scheduled", 7, 90 * reaper.Day},
|
||||||
|
} {
|
||||||
|
keep, retention, ok := backupPolicy(tc.reason, rcfg)
|
||||||
|
if !ok || keep != tc.keep || retention != tc.retention {
|
||||||
|
t.Errorf("backupPolicy(%q) = (%d, %v, %v); want (%d, %v, true)", tc.reason, keep, retention, ok, tc.keep, tc.retention)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if _, _, ok := backupPolicy("inactive_15d", rcfg); ok {
|
||||||
|
t.Error("backupPolicy accepted inactive_15d; the reaper records those itself")
|
||||||
|
}
|
||||||
|
}
|
||||||
+23
-2
@@ -199,8 +199,9 @@ func (w *mailWarner) Warn(ctx context.Context, ownerID, server, remaining string
|
|||||||
|
|
||||||
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
|
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
|
||||||
// idle deadline is fixed by §18; only the warning offsets, retention, the
|
// idle deadline is fixed by §18; only the warning offsets, retention, the
|
||||||
// store soft-cap and the on-demand backup bounds are configurable (§24). The
|
// store soft-cap, the on-demand backup bounds and the scheduled restore points
|
||||||
// backup Job and felis-api read the manual_* bounds through it too.
|
// are configurable (§24). The backup Job and felis-api read the manual_* and
|
||||||
|
// scheduled_* keys through it too.
|
||||||
func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
||||||
rc := reaper.DefaultConfig()
|
rc := reaper.DefaultConfig()
|
||||||
if v := cfg.Archive.Retention; v != "" {
|
if v := cfg.Archive.Retention; v != "" {
|
||||||
@@ -248,6 +249,26 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
|||||||
}
|
}
|
||||||
rc.ManualCooldown = d
|
rc.ManualCooldown = d
|
||||||
}
|
}
|
||||||
|
if v := cfg.Archive.ScheduledEvery; v != "" {
|
||||||
|
d, err := parseSpanDuration(v)
|
||||||
|
if err != nil || d < 0 {
|
||||||
|
return rc, fmt.Errorf("[archive] scheduled_every %q: want a span such as 1d (0s for none)", v)
|
||||||
|
}
|
||||||
|
rc.ScheduledEvery = d
|
||||||
|
}
|
||||||
|
switch n := cfg.Archive.ScheduledKeep; {
|
||||||
|
case n < 0:
|
||||||
|
return rc, fmt.Errorf("[archive] scheduled_keep %d: want 1 or more", n)
|
||||||
|
case n > 0:
|
||||||
|
rc.ScheduledKeep = n
|
||||||
|
}
|
||||||
|
if v := cfg.Archive.ScheduledRetention; v != "" {
|
||||||
|
d, err := parseSpanDuration(v)
|
||||||
|
if err != nil || d <= 0 {
|
||||||
|
return rc, fmt.Errorf("[archive] scheduled_retention %q: want a positive span such as 90d", v)
|
||||||
|
}
|
||||||
|
rc.ScheduledRetention = d
|
||||||
|
}
|
||||||
rc.RequireOffsite = cfg.Offsite.Enabled()
|
rc.RequireOffsite = cfg.Offsite.Enabled()
|
||||||
return rc, nil
|
return rc, nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -85,6 +85,42 @@ func TestReaperConfigManualKeys(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestReaperConfigScheduledKeys: the scheduled restore points default to one a
|
||||||
|
// day, seven per server and the reaper's 90 days, accept overrides ("0s" turns
|
||||||
|
// them off), and refuse values that would keep nothing or run backwards.
|
||||||
|
func TestReaperConfigScheduledKeys(t *testing.T) {
|
||||||
|
rc, err := reaperConfig(&config.Config{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if rc.ScheduledEvery != reaper.Day || rc.ScheduledKeep != 7 || rc.ScheduledRetention != 90*reaper.Day {
|
||||||
|
t.Fatalf("defaults = %v / %d / %v", rc.ScheduledEvery, rc.ScheduledKeep, rc.ScheduledRetention)
|
||||||
|
}
|
||||||
|
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{
|
||||||
|
ScheduledEvery: "12h", ScheduledKeep: 3, ScheduledRetention: "14d"}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if rc.ScheduledEvery != 12*time.Hour || rc.ScheduledKeep != 3 || rc.ScheduledRetention != 14*reaper.Day {
|
||||||
|
t.Fatalf("overrides = %v / %d / %v", rc.ScheduledEvery, rc.ScheduledKeep, rc.ScheduledRetention)
|
||||||
|
}
|
||||||
|
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{ScheduledEvery: "0s"}})
|
||||||
|
if err != nil || rc.ScheduledEvery != 0 {
|
||||||
|
t.Fatalf("scheduled_every 0s = %v, %v; want off", rc.ScheduledEvery, err)
|
||||||
|
}
|
||||||
|
for _, bad := range []config.ArchiveConfig{
|
||||||
|
{ScheduledEvery: "-1h"},
|
||||||
|
{ScheduledEvery: "daily"},
|
||||||
|
{ScheduledKeep: -1},
|
||||||
|
{ScheduledRetention: "0d"},
|
||||||
|
{ScheduledRetention: "forever"},
|
||||||
|
} {
|
||||||
|
if _, err := reaperConfig(&config.Config{Archive: bad}); err == nil {
|
||||||
|
t.Errorf("%+v was accepted", bad)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestResolveWorldDir(t *testing.T) {
|
func TestResolveWorldDir(t *testing.T) {
|
||||||
ctx := context.Background()
|
ctx := context.Background()
|
||||||
root := t.TempDir()
|
root := t.TempDir()
|
||||||
|
|||||||
+10
-8
@@ -4196,13 +4196,15 @@ persisted_auth_source_blocks() {
|
|||||||
|
|
||||||
# persisted_archive_block echoes the operator-owned [archive] keys an earlier run
|
# persisted_archive_block echoes the operator-owned [archive] keys an earlier run
|
||||||
# left behind — the retention window, the pre-reap warn offsets, the local cap,
|
# left behind — the retention window, the pre-reap warn offsets, the local cap,
|
||||||
# the on-demand backup retention/count/cooldown — so a re-run does not silently
|
# the on-demand backup retention/count/cooldown, the scheduled backup
|
||||||
# revert them to the built-ins (felis reaper, felis backup and felis api read
|
# period/count/retention — so a re-run does not silently revert them to the
|
||||||
# these from the config Secret; defaults: 90d retention, 3d/1d warnings, no cap,
|
# built-ins (felis reaper, felis backup and felis api read these from the config
|
||||||
# manual backups kept 30d, 5 per server, one per 10m). store and local_path are
|
# Secret; defaults: 90d retention, 3d/1d warnings, no cap, manual backups kept
|
||||||
# NOT carried: this script owns them (FELIS_ARCHIVE_LOCAL_PATH must equal the
|
# 30d, 5 per server, one per 10m, a scheduled backup a day kept 7 per server for
|
||||||
# mount). Same first-readable-file rule as persisted_smtp_block; warn_before must
|
# 90d). store and local_path are NOT carried: this script owns them
|
||||||
# be a single-line TOML array (the shape every writer here emits).
|
# (FELIS_ARCHIVE_LOCAL_PATH must equal the mount). Same first-readable-file rule
|
||||||
|
# as persisted_smtp_block; warn_before must be a single-line TOML array (the
|
||||||
|
# shape every writer here emits).
|
||||||
persisted_archive_block() {
|
persisted_archive_block() {
|
||||||
local f out
|
local f out
|
||||||
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
|
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
|
||||||
@@ -4210,7 +4212,7 @@ persisted_archive_block() {
|
|||||||
out="$(awk '
|
out="$(awk '
|
||||||
/^[[:space:]]*\[/ { sect = $0; next }
|
/^[[:space:]]*\[/ { sect = $0; next }
|
||||||
sect ~ /^[[:space:]]*\[archive\][[:space:]]*$/ &&
|
sect ~ /^[[:space:]]*\[archive\][[:space:]]*$/ &&
|
||||||
/^[[:space:]]*(retention|warn_before|max_local_bytes|manual_retention|manual_keep|manual_cooldown)[[:space:]]*=/ { print }
|
/^[[:space:]]*(retention|warn_before|max_local_bytes|manual_retention|manual_keep|manual_cooldown|scheduled_every|scheduled_keep|scheduled_retention)[[:space:]]*=/ { print }
|
||||||
' "$f")"
|
' "$f")"
|
||||||
[ -n "$out" ] || continue
|
[ -n "$out" ] || continue
|
||||||
printf '%s\n' "$out"
|
printf '%s\n' "$out"
|
||||||
|
|||||||
@@ -1452,6 +1452,9 @@ local_path = "/stale/path"
|
|||||||
retention = "30d"
|
retention = "30d"
|
||||||
manual_keep = 3
|
manual_keep = 3
|
||||||
manual_cooldown = "1h"
|
manual_cooldown = "1h"
|
||||||
|
scheduled_every = "12h"
|
||||||
|
scheduled_keep = 14
|
||||||
|
scheduled_retention = "45d"
|
||||||
|
|
||||||
[offsite]
|
[offsite]
|
||||||
endpoint = "https://objects.example"
|
endpoint = "https://objects.example"
|
||||||
@@ -1503,6 +1506,9 @@ expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out"
|
|||||||
expect "a re-run carries the archive retention window" 'retention = "30d"' "$out"
|
expect "a re-run carries the archive retention window" 'retention = "30d"' "$out"
|
||||||
expect "a re-run carries the on-demand backup count" 'manual_keep = 3' "$out"
|
expect "a re-run carries the on-demand backup count" 'manual_keep = 3' "$out"
|
||||||
expect "a re-run carries the on-demand backup cooldown" 'manual_cooldown = "1h"' "$out"
|
expect "a re-run carries the on-demand backup cooldown" 'manual_cooldown = "1h"' "$out"
|
||||||
|
expect "a re-run carries the scheduled backup period" 'scheduled_every = "12h"' "$out"
|
||||||
|
expect "a re-run carries the scheduled backup count" 'scheduled_keep = 14' "$out"
|
||||||
|
expect "a re-run carries the scheduled backup retention" 'scheduled_retention = "45d"' "$out"
|
||||||
expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out"
|
expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out"
|
||||||
expect "the panel learns the public game port" 'game_port = 25570' "$out"
|
expect "the panel learns the public game port" 'game_port = 25570' "$out"
|
||||||
expect "a re-run keeps the off-site bucket, set apart from the next section" '[offsite]
|
expect "a re-run keeps the off-site bucket, set apart from the next section" '[offsite]
|
||||||
|
|||||||
+6
-1
@@ -592,7 +592,7 @@ components:
|
|||||||
size_bytes: { type: integer, format: int64 }
|
size_bytes: { type: integer, format: int64 }
|
||||||
reason:
|
reason:
|
||||||
type: string
|
type: string
|
||||||
description: inactive_15d (idle reclaim), manual (on demand) or pre_restore (the safety snapshot in front of a restore).
|
description: inactive_15d (idle reclaim), manual (on demand), pre_restore (the safety snapshot in front of a restore) or scheduled (the daily restore point felis-api takes of a world played since its last one, once the server stops).
|
||||||
status: { type: string }
|
status: { type: string }
|
||||||
created_at: { type: string, format: date-time }
|
created_at: { type: string, format: date-time }
|
||||||
expires_at: { type: string, format: date-time }
|
expires_at: { type: string, format: date-time }
|
||||||
@@ -3695,6 +3695,11 @@ paths:
|
|||||||
restore_backup_id:
|
restore_backup_id:
|
||||||
type: string
|
type: string
|
||||||
description: The backup the chained restore extracts.
|
description: The backup the chained restore extracts.
|
||||||
|
scheduled:
|
||||||
|
type: boolean
|
||||||
|
description: >-
|
||||||
|
A backup felis-api took on its own: the daily restore
|
||||||
|
point of a played world. Omitted when false.
|
||||||
'401':
|
'401':
|
||||||
$ref: '#/components/responses/Unauthorized'
|
$ref: '#/components/responses/Unauthorized'
|
||||||
'403':
|
'403':
|
||||||
|
|||||||
+51
-2
@@ -892,8 +892,10 @@ re-run turns world reaping on. [GO-TESTED: `TestRunRetentionTouchesNoWorld`,
|
|||||||
With a worlds root the reaper
|
With a worlds root the reaper
|
||||||
reaps a world only when `now - last_active_at > 15d` (`inactive_15d`); the 15-day
|
reaps a world only when `now - last_active_at > 15d` (`inactive_15d`); the 15-day
|
||||||
deadline is **hard-fixed in code** (only `warn_before` / `retention` /
|
deadline is **hard-fixed in code** (only `warn_before` / `retention` /
|
||||||
`max_local_bytes` and the on-demand backup keys `manual_retention` /
|
`max_local_bytes`, the on-demand backup keys `manual_retention` /
|
||||||
`manual_keep` / `manual_cooldown` are configurable from `felis.toml [archive]`).
|
`manual_keep` / `manual_cooldown` and the scheduled backup keys
|
||||||
|
`scheduled_every` / `scheduled_keep` / `scheduled_retention` are configurable
|
||||||
|
from `felis.toml [archive]`).
|
||||||
|
|
||||||
### What a "backup" contains
|
### What a "backup" contains
|
||||||
|
|
||||||
@@ -1022,6 +1024,53 @@ the error its container exited on under Recent operations on the server's
|
|||||||
backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`,
|
backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`,
|
||||||
`TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`]
|
`TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`]
|
||||||
|
|
||||||
|
### Scheduled backups (daily restore points)
|
||||||
|
|
||||||
|
A world played every day never idles 15 days, so the reaper never archives it.
|
||||||
|
felis-api therefore takes a `scheduled` backup of every owned world that
|
||||||
|
somebody joined since its owner's last intact scheduled backup, once that
|
||||||
|
backup is `scheduled_every` old. This works without a worlds root: it is the
|
||||||
|
same backup Job "Back up now" starts, so it needs only `FELIS_IMAGE` and
|
||||||
|
`FELIS_BACKUP_PVC` (felis-api logs `scheduled backups off` at start when
|
||||||
|
either is missing or `scheduled_every` is `0s`).
|
||||||
|
|
||||||
|
| Key | Default | Effect |
|
||||||
|
|---|---|---|
|
||||||
|
| `scheduled_every` | `1d` | how old a world's newest scheduled backup must be before it gets the next one (`0s` turns scheduled backups off) |
|
||||||
|
| `scheduled_keep` | `7` | scheduled backups kept per server and owner; the Job removes older ones like `manual_keep` |
|
||||||
|
| `scheduled_retention` | `90d` | when a scheduled backup expires |
|
||||||
|
|
||||||
|
How it behaves:
|
||||||
|
|
||||||
|
- **Only a stopped server is backed up.** The Job mounts the world volume,
|
||||||
|
which a running server holds. Idle auto-stop brings a played world down
|
||||||
|
minutes after its last player leaves, so the point normally lands the same
|
||||||
|
day. A server with auto-stop turned off gets its point the next time it
|
||||||
|
stops; one that never stops never gets one.
|
||||||
|
- **One Job at a time, cluster-wide.** felis-api checks every 2 minutes and
|
||||||
|
starts one scheduled backup only while no backup or restore Job is running,
|
||||||
|
worlds without a point first. It holds the world like any backup, so a player
|
||||||
|
who wakes the server during those minutes sees the maintenance message and can
|
||||||
|
retry once it finishes.
|
||||||
|
- **Paused while the store is full.** While the present backups add up to
|
||||||
|
`max_local_bytes`, no scheduled backup starts (felis-api logs
|
||||||
|
`scheduled backups paused` and `resumed` once each). The backup Job's 10%
|
||||||
|
free-disk check applies too.
|
||||||
|
- **A failed backup is retried** after a quarter of `scheduled_every` (6 hours
|
||||||
|
by default); the failure shows under Recent operations as "Scheduled backup".
|
||||||
|
- **Counted per owner.** A backup the world's previous owner took before it was
|
||||||
|
reaped and claimed again is no restore point of the new owner's world, and the
|
||||||
|
new owner's backups never prune it. The keep-N prune of every reason is scoped
|
||||||
|
to the backup's owner the same way.
|
||||||
|
- Scheduled backups write the audit action `backup.scheduled` (actor
|
||||||
|
`scheduler`) and never start the owner's `manual_cooldown`. They are copied
|
||||||
|
offsite and evicted by `max_local_bytes` like manual backups.
|
||||||
|
|
||||||
|
[GO-TESTED: `TestBackupScheduler`, `TestK8sScheduledBackupJobs`,
|
||||||
|
`TestBackupJobRecordsAScheduledBackup`, `TestBackupPolicyPerReason`,
|
||||||
|
`TestReaperConfigScheduledKeys`; PG-TESTED: `TestScheduledBackupCandidates`,
|
||||||
|
`TestExcessBackupsPerOwner`]
|
||||||
|
|
||||||
### Exemptions (world never reaped)
|
### Exemptions (world never reaped)
|
||||||
|
|
||||||
- `spec.reaperExempt=true` → skipped entirely (system servers). [GO-TESTED
|
- `spec.reaperExempt=true` → skipped entirely (system servers). [GO-TESTED
|
||||||
|
|||||||
@@ -26,6 +26,9 @@ type AsyncJob struct {
|
|||||||
ThenRestore string `json:"then_restore,omitempty"`
|
ThenRestore string `json:"then_restore,omitempty"`
|
||||||
ThenRestoreReason string `json:"then_restore_reason,omitempty"`
|
ThenRestoreReason string `json:"then_restore_reason,omitempty"`
|
||||||
RestoreBackupID string `json:"restore_backup_id,omitempty"`
|
RestoreBackupID string `json:"restore_backup_id,omitempty"`
|
||||||
|
// Scheduled marks a backup felis-api took on its own (BackupScheduler), so
|
||||||
|
// the owner can tell it from one somebody asked for.
|
||||||
|
Scheduled bool `json:"scheduled,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// JobStatusReader reads the newest backup/restore Jobs for a server, newest
|
// JobStatusReader reads the newest backup/restore Jobs for a server, newest
|
||||||
|
|||||||
@@ -25,6 +25,11 @@ const (
|
|||||||
|
|
||||||
jobManagedByBackup = "felis-backup"
|
jobManagedByBackup = "felis-backup"
|
||||||
jobManagedByRestore = "felis-restore"
|
jobManagedByRestore = "felis-restore"
|
||||||
|
|
||||||
|
// jobBackupReasonLabel marks the backup Job of a scheduled restore point
|
||||||
|
// (backupjob.LabelReason).
|
||||||
|
jobBackupReasonLabel = "felis.lolicon.best/backup-reason"
|
||||||
|
jobBackupReasonScheduled = "scheduled"
|
||||||
)
|
)
|
||||||
|
|
||||||
// K8sJobStatus reads the async Jobs the executors created, by the server label
|
// K8sJobStatus reads the async Jobs the executors created, by the server label
|
||||||
@@ -137,9 +142,27 @@ func jobToAsyncJob(j *batchv1.Job) (AsyncJob, bool) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
aj.Scheduled = aj.Kind == "backup" && j.Labels[jobBackupReasonLabel] == jobBackupReasonScheduled
|
||||||
return aj, true
|
return aj, true
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// RunningWorldJobs counts the backup and restore Jobs of every server that
|
||||||
|
// have yet to finish (BackupScheduler waits for them).
|
||||||
|
func (k *K8sJobStatus) RunningWorldJobs(ctx context.Context) (int, error) {
|
||||||
|
var list batchv1.JobList
|
||||||
|
if err := k.c.List(ctx, &list, client.InNamespace(k.namespace), client.HasLabels{jobManagedByLabel}); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := range list.Items {
|
||||||
|
j := &list.Items[i]
|
||||||
|
if _, ok := jobOutcome(j); ok && !maintenance.JobFinished(j) {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, nil
|
||||||
|
}
|
||||||
|
|
||||||
// PendingRestoreChains lists the safety snapshots felis-api has yet to settle,
|
// PendingRestoreChains lists the safety snapshots felis-api has yet to settle,
|
||||||
// across every server (the label selector keeps it to them).
|
// across every server (the label selector keeps it to them).
|
||||||
func (k *K8sJobStatus) PendingRestoreChains(ctx context.Context) ([]RestoreChain, error) {
|
func (k *K8sJobStatus) PendingRestoreChains(ctx context.Context) ([]RestoreChain, error) {
|
||||||
|
|||||||
@@ -782,6 +782,39 @@ func (p *PGRepo) BackupStoreBytes(ctx context.Context) (int64, error) {
|
|||||||
return n, err
|
return n, err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ScheduledBackupCandidates is the ScheduleStore behind BackupScheduler. A
|
||||||
|
// world's point is its current owner's newest intact scheduled backup taken
|
||||||
|
// since they claimed it: a previous owner's backups say nothing about the world
|
||||||
|
// the new owner has built, and a corrupt one restores nothing. last_active_at
|
||||||
|
// moves on every join, so a world nobody joined since its point is skipped.
|
||||||
|
func (p *PGRepo) ScheduledBackupCandidates(ctx context.Context, before time.Time) ([]ScheduledCandidate, error) {
|
||||||
|
rows, err := p.db.QueryContext(ctx,
|
||||||
|
`SELECT s.name, s.owner_id FROM servers s
|
||||||
|
LEFT JOIN LATERAL (
|
||||||
|
SELECT max(b.created_at) AS at FROM world_backups b
|
||||||
|
WHERE b.server_name = s.name AND b.reason = 'scheduled' AND b.status = 'present'
|
||||||
|
AND b.corrupt_at IS NULL AND b.former_owner = s.owner_id
|
||||||
|
AND b.created_at >= COALESCE(s.claimed_at, '-infinity')
|
||||||
|
) pt ON true
|
||||||
|
WHERE s.deleted_at IS NULL AND s.owner_id IS NOT NULL
|
||||||
|
AND s.last_active_at > COALESCE(pt.at, '-infinity')
|
||||||
|
AND COALESCE(pt.at, '-infinity') < $1
|
||||||
|
ORDER BY pt.at NULLS FIRST, s.name`, before)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
var out []ScheduledCandidate
|
||||||
|
for rows.Next() {
|
||||||
|
var c ScheduledCandidate
|
||||||
|
if err := rows.Scan(&c.Name, &c.OwnerID); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
out = append(out, c)
|
||||||
|
}
|
||||||
|
return out, rows.Err()
|
||||||
|
}
|
||||||
|
|
||||||
func (p *PGRepo) Audit(ctx context.Context, e AuditEntry) error {
|
func (p *PGRepo) Audit(ctx context.Context, e AuditEntry) error {
|
||||||
// A nil Payload must land as SQL NULL, not the text "null"; a non-nil Payload is
|
// A nil Payload must land as SQL NULL, not the text "null"; a non-nil Payload is
|
||||||
// passed as a JSON text the jsonb column parses (same idiom as reaper.PGStore).
|
// passed as a JSON text the jsonb column parses (same idiom as reaper.PGStore).
|
||||||
|
|||||||
@@ -0,0 +1,170 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/maintenance"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Scheduled backups. A world played every day never idles long enough for the
|
||||||
|
// reaper to archive it, and an owner's own backups are only the ones they
|
||||||
|
// remembered to take, so felis-api takes a daily restore point of every world
|
||||||
|
// played since its last one. The backup Job mounts the world volume, which a
|
||||||
|
// running server holds, so the point is taken once the server is stopped: idle
|
||||||
|
// auto-stop brings a played world down minutes after its last player leaves,
|
||||||
|
// and the point lands the same day. A server kept running around the clock gets
|
||||||
|
// one the next time it stops.
|
||||||
|
//
|
||||||
|
// The point is an ordinary world_backups row of reason scheduled: listed and
|
||||||
|
// restorable by its owner, copied offsite like any other, pruned to [archive]
|
||||||
|
// scheduled_keep per owner and expired after scheduled_retention, so it never
|
||||||
|
// takes the place of a backup the owner asked for.
|
||||||
|
|
||||||
|
// ScheduledBackuper enqueues the backup Job of a scheduled restore point
|
||||||
|
// (backupjob.Backuper).
|
||||||
|
type ScheduledBackuper interface {
|
||||||
|
BackupScheduled(ctx context.Context, serverName, formerOwner string) error
|
||||||
|
}
|
||||||
|
|
||||||
|
// ScheduledCandidate is an owned world due a scheduled backup.
|
||||||
|
type ScheduledCandidate struct {
|
||||||
|
Name string
|
||||||
|
OwnerID string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ScheduleStore lists the worlds due a scheduled backup: owned, joined since
|
||||||
|
// their owner's newest intact scheduled backup, and without one taken after
|
||||||
|
// before. Worlds never given one come first, then the longest waiting.
|
||||||
|
type ScheduleStore interface {
|
||||||
|
ScheduledBackupCandidates(ctx context.Context, before time.Time) ([]ScheduledCandidate, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// WorldJobCounter counts the backup and restore Jobs still running.
|
||||||
|
type WorldJobCounter interface {
|
||||||
|
RunningWorldJobs(ctx context.Context) (int, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Audit identity of a scheduled backup. The action is its own so the owner's
|
||||||
|
// manual cooldown (LastBackupRequest, backup.create) never counts it.
|
||||||
|
const (
|
||||||
|
scheduledBackupActor = "scheduler"
|
||||||
|
scheduledBackupAction = "backup.scheduled"
|
||||||
|
)
|
||||||
|
|
||||||
|
// BackupScheduler takes the scheduled backups. Tick launches at most one
|
||||||
|
// backup Job and only while no backup or restore Job is running, so the points
|
||||||
|
// queue behind each other and behind the owners' own operations instead of
|
||||||
|
// loading the node with archives all at once.
|
||||||
|
type BackupScheduler struct {
|
||||||
|
API *API
|
||||||
|
Store ScheduleStore
|
||||||
|
Jobs WorldJobCounter
|
||||||
|
// Every is how often a played world gets a point ([archive]
|
||||||
|
// scheduled_every). Zero or less turns the scheduler off.
|
||||||
|
Every time.Duration
|
||||||
|
|
||||||
|
// tried is when each world last had a Job launched or was found without a
|
||||||
|
// volume, so one whose backup keeps failing is retried every Every/4
|
||||||
|
// instead of every tick.
|
||||||
|
tried map[string]time.Time
|
||||||
|
// full remembers that the store was at max_local_bytes, so the pause and
|
||||||
|
// the resume are logged once each.
|
||||||
|
full bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// Tick launches the backup of the world waiting longest, if any may run now.
|
||||||
|
func (s *BackupScheduler) Tick(ctx context.Context) error {
|
||||||
|
b, ok := s.API.Backuper.(ScheduledBackuper)
|
||||||
|
if !ok || s.Every <= 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
now := s.API.now()
|
||||||
|
|
||||||
|
running, err := s.Jobs.RunningWorldJobs(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("count running backup and restore jobs: %w", err)
|
||||||
|
}
|
||||||
|
if running > 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// A full store is the reaper's to evict; a scheduled point added now would
|
||||||
|
// only push out an older backup of somebody else's.
|
||||||
|
if limit := s.API.BackupStoreCap; limit > 0 {
|
||||||
|
used, err := s.API.Repo.BackupStoreBytes(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read the backup store size: %w", err)
|
||||||
|
}
|
||||||
|
full := used >= limit
|
||||||
|
if full != s.full {
|
||||||
|
if full {
|
||||||
|
log.Printf("api: scheduled backups paused: the backup store holds %d of its %d bytes ([archive] max_local_bytes)", used, limit)
|
||||||
|
} else {
|
||||||
|
log.Printf("api: scheduled backups resumed: the backup store is below [archive] max_local_bytes")
|
||||||
|
}
|
||||||
|
s.full = full
|
||||||
|
}
|
||||||
|
if full {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
due, err := s.Store.ScheduledBackupCandidates(ctx, now.Add(-s.Every))
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("list the worlds due a scheduled backup: %w", err)
|
||||||
|
}
|
||||||
|
if s.tried == nil {
|
||||||
|
s.tried = map[string]time.Time{}
|
||||||
|
}
|
||||||
|
for name, at := range s.tried {
|
||||||
|
if now.Sub(at) >= s.Every {
|
||||||
|
delete(s.tried, name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, c := range due {
|
||||||
|
if at, ok := s.tried[c.Name]; ok && now.Sub(at) < s.Every/4 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// A server never started has no world yet, and one whose volume is gone
|
||||||
|
// has nothing left to save; the Job would sit Pending on the claim.
|
||||||
|
exists, err := s.API.Cluster.WorldVolumeExists(ctx, c.Name)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("look up the world volume of %s: %w", c.Name, err)
|
||||||
|
}
|
||||||
|
if !exists {
|
||||||
|
s.tried[c.Name] = now
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
var busy *MaintenanceBusyError
|
||||||
|
switch err := s.API.Cluster.AcquireMaintenance(ctx, c.Name, maintenance.KindBackup); {
|
||||||
|
case errors.Is(err, ErrNotStopped), errors.As(err, &busy), errors.Is(err, ErrMaintenanceInProgress):
|
||||||
|
continue // running, or somebody else has the world: next tick
|
||||||
|
case errors.Is(err, ErrNotFound):
|
||||||
|
s.tried[c.Name] = now
|
||||||
|
continue
|
||||||
|
case err != nil:
|
||||||
|
return fmt.Errorf("lock the world of %s: %w", c.Name, err)
|
||||||
|
}
|
||||||
|
err = b.BackupScheduled(ctx, c.Name, c.OwnerID)
|
||||||
|
// Once the Job exists it holds the world; the annotation only covered
|
||||||
|
// the gap.
|
||||||
|
if rerr := s.API.Cluster.ReleaseMaintenance(context.WithoutCancel(ctx), c.Name); rerr != nil {
|
||||||
|
log.Printf("api: release the maintenance lock on %s: %v (it lapses after %s)", c.Name, rerr, maintenance.Grace)
|
||||||
|
}
|
||||||
|
s.tried[c.Name] = now
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("start the scheduled backup of %s: %w", c.Name, err)
|
||||||
|
}
|
||||||
|
log.Printf("api: started the scheduled backup of %s", c.Name)
|
||||||
|
s.API.writeAudit(ctx, AuditEntry{
|
||||||
|
Actor: scheduledBackupActor, Source: scheduledBackupActor,
|
||||||
|
Action: scheduledBackupAction, ServerName: c.Name,
|
||||||
|
})
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,263 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/backupjob"
|
||||||
|
"felis.lolicon.best/internal/maintenance"
|
||||||
|
batchv1 "k8s.io/api/batch/v1"
|
||||||
|
corev1 "k8s.io/api/core/v1"
|
||||||
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||||
|
"k8s.io/apimachinery/pkg/runtime"
|
||||||
|
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||||
|
)
|
||||||
|
|
||||||
|
var _ ScheduledBackuper = (*backupjob.Backuper)(nil)
|
||||||
|
|
||||||
|
// fakeScheduledBackuper is a Backuper that can also take a scheduled backup.
|
||||||
|
type fakeScheduledBackuper struct {
|
||||||
|
fakeBackuper
|
||||||
|
err error
|
||||||
|
scheduled []ScheduledCandidate
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeScheduledBackuper) BackupScheduled(_ context.Context, name, formerOwner string) error {
|
||||||
|
f.scheduled = append(f.scheduled, ScheduledCandidate{Name: name, OwnerID: formerOwner})
|
||||||
|
return f.err
|
||||||
|
}
|
||||||
|
|
||||||
|
type fakeScheduleStore struct {
|
||||||
|
due []ScheduledCandidate
|
||||||
|
gotBefore time.Time
|
||||||
|
calls int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeScheduleStore) ScheduledBackupCandidates(_ context.Context, before time.Time) ([]ScheduledCandidate, error) {
|
||||||
|
f.calls++
|
||||||
|
f.gotBefore = before
|
||||||
|
return f.due, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type fakeWorldJobs struct {
|
||||||
|
running int
|
||||||
|
err error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeWorldJobs) RunningWorldJobs(context.Context) (int, error) { return f.running, f.err }
|
||||||
|
|
||||||
|
func TestBackupScheduler(t *testing.T) {
|
||||||
|
const every = 24 * time.Hour
|
||||||
|
type rig struct {
|
||||||
|
a *API
|
||||||
|
repo *fakeRepo
|
||||||
|
cl *fakeCluster
|
||||||
|
b *fakeScheduledBackuper
|
||||||
|
store *fakeScheduleStore
|
||||||
|
jobs *fakeWorldJobs
|
||||||
|
s *BackupScheduler
|
||||||
|
}
|
||||||
|
mk := func() rig {
|
||||||
|
repo := newFakeRepo()
|
||||||
|
cl := newFakeCluster()
|
||||||
|
for _, n := range []string{"alpha", "bravo"} {
|
||||||
|
cl.byName[n] = &ServerInfo{Name: n, Phase: "Stopped"}
|
||||||
|
}
|
||||||
|
a := newTestAPI(repo, cl)
|
||||||
|
b := &fakeScheduledBackuper{}
|
||||||
|
a.Backuper = b
|
||||||
|
store := &fakeScheduleStore{due: []ScheduledCandidate{{"alpha", "usr-a"}, {"bravo", "usr-b"}}}
|
||||||
|
jobs := &fakeWorldJobs{}
|
||||||
|
return rig{a, repo, cl, b, store, jobs, &BackupScheduler{API: a, Store: store, Jobs: jobs, Every: every}}
|
||||||
|
}
|
||||||
|
tick := func(t *testing.T, r rig) {
|
||||||
|
t.Helper()
|
||||||
|
if err := r.s.Tick(context.Background()); err != nil {
|
||||||
|
t.Fatalf("Tick: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
launched := func(r rig) []ScheduledCandidate { return r.b.scheduled }
|
||||||
|
|
||||||
|
t.Run("backs up the first due world as its owner, one per tick", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
tick(t, r)
|
||||||
|
if got := launched(r); len(got) != 1 || got[0] != (ScheduledCandidate{"alpha", "usr-a"}) {
|
||||||
|
t.Fatalf("launched = %+v; want alpha as usr-a only", got)
|
||||||
|
}
|
||||||
|
if want := r.a.now().Add(-every); !r.store.gotBefore.Equal(want) {
|
||||||
|
t.Fatalf("asked for points before %v; want %v", r.store.gotBefore, want)
|
||||||
|
}
|
||||||
|
if len(r.cl.acquired) != 1 || r.cl.acquired[0] != "alpha:"+maintenance.KindBackup ||
|
||||||
|
len(r.cl.released) != 1 || r.cl.released[0] != "alpha" {
|
||||||
|
t.Fatalf("lock: acquired %v released %v; want alpha held as a backup and let go", r.cl.acquired, r.cl.released)
|
||||||
|
}
|
||||||
|
if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "backup.scheduled" || r.repo.audits[0].ServerName != "alpha" {
|
||||||
|
t.Fatalf("audits = %+v; want one backup.scheduled of alpha", r.repo.audits)
|
||||||
|
}
|
||||||
|
if at, _ := r.repo.LastBackupRequest(context.Background(), "alpha", time.Time{}); !at.IsZero() {
|
||||||
|
t.Fatal("the scheduled backup started the owner's manual cooldown")
|
||||||
|
}
|
||||||
|
if r.b.calls != 0 {
|
||||||
|
t.Fatal("the scheduled backup went through the manual Backup")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("a running or busy world waits for the next tick", func(t *testing.T) {
|
||||||
|
for _, err := range []error{ErrNotStopped, &MaintenanceBusyError{Kind: maintenance.KindRestore}, ErrMaintenanceInProgress} {
|
||||||
|
r := mk()
|
||||||
|
r.cl.maintErr["alpha"] = err
|
||||||
|
tick(t, r)
|
||||||
|
if got := launched(r); len(got) != 1 || got[0].Name != "bravo" {
|
||||||
|
t.Fatalf("%v: launched = %+v; want bravo", err, got)
|
||||||
|
}
|
||||||
|
delete(r.cl.maintErr, "alpha")
|
||||||
|
r.b.scheduled = nil
|
||||||
|
tick(t, r)
|
||||||
|
if got := launched(r); len(got) != 1 || got[0].Name != "alpha" {
|
||||||
|
t.Fatalf("%v: once alpha is free, launched = %+v; want alpha", err, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("waits while a backup or restore job runs", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.jobs.running = 1
|
||||||
|
tick(t, r)
|
||||||
|
if len(launched(r)) != 0 || len(r.cl.acquired) != 0 {
|
||||||
|
t.Fatalf("launched %+v, acquired %v beside a running job", launched(r), r.cl.acquired)
|
||||||
|
}
|
||||||
|
r.jobs.running = 0
|
||||||
|
tick(t, r)
|
||||||
|
if len(launched(r)) != 1 {
|
||||||
|
t.Fatalf("launched = %+v once the job finished", launched(r))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("pauses while the store is full", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.a.BackupStoreCap = 100
|
||||||
|
r.repo.backups = []fakeBackup{{view: BackupView{ID: "bk1", ServerName: "bravo", Status: "present", SizeBytes: 100}}}
|
||||||
|
tick(t, r)
|
||||||
|
if len(launched(r)) != 0 {
|
||||||
|
t.Fatalf("launched = %+v into a full store", launched(r))
|
||||||
|
}
|
||||||
|
r.repo.backups[0].view.SizeBytes = 99
|
||||||
|
tick(t, r)
|
||||||
|
if len(launched(r)) != 1 {
|
||||||
|
t.Fatalf("launched = %+v below the cap", launched(r))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("a world whose backup failed is retried after a quarter period", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.store.due = r.store.due[:1]
|
||||||
|
r.b.err = errors.New("apiserver down")
|
||||||
|
if err := r.s.Tick(context.Background()); err == nil {
|
||||||
|
t.Fatal("Tick hid the failed launch")
|
||||||
|
}
|
||||||
|
if len(r.cl.released) != 1 {
|
||||||
|
t.Fatalf("released = %v; the lock must go when the launch fails", r.cl.released)
|
||||||
|
}
|
||||||
|
r.b.err = nil
|
||||||
|
start := r.a.now()
|
||||||
|
r.a.Now = func() time.Time { return start.Add(every/4 - time.Minute) }
|
||||||
|
tick(t, r)
|
||||||
|
if len(launched(r)) != 1 {
|
||||||
|
t.Fatalf("launched = %+v; retried before a quarter period", launched(r))
|
||||||
|
}
|
||||||
|
r.a.Now = func() time.Time { return start.Add(every / 4) }
|
||||||
|
tick(t, r)
|
||||||
|
if len(launched(r)) != 2 {
|
||||||
|
t.Fatalf("launched = %+v; not retried after a quarter period", launched(r))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("a world without a volume is passed over", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.cl.noWorld["alpha"] = true
|
||||||
|
tick(t, r)
|
||||||
|
if got := launched(r); len(got) != 1 || got[0].Name != "bravo" || len(r.cl.acquired) != 1 {
|
||||||
|
t.Fatalf("launched = %+v, acquired %v; want bravo alone", got, r.cl.acquired)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("a world deleted since the listing is passed over", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.cl.maintErr["alpha"] = ErrNotFound
|
||||||
|
tick(t, r)
|
||||||
|
if got := launched(r); len(got) != 1 || got[0].Name != "bravo" {
|
||||||
|
t.Fatalf("launched = %+v; want bravo", got)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("a lock failure stops the tick", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.cl.maintErr["alpha"] = errors.New("conflict storm")
|
||||||
|
if err := r.s.Tick(context.Background()); err == nil || len(launched(r)) != 0 {
|
||||||
|
t.Fatalf("Tick = %v, launched %+v; want the error and nothing started", err, launched(r))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("off without a scheduling backuper or a period", func(t *testing.T) {
|
||||||
|
r := mk()
|
||||||
|
r.a.Backuper = &fakeBackuper{}
|
||||||
|
tick(t, r)
|
||||||
|
r2 := mk()
|
||||||
|
r2.s.Every = 0
|
||||||
|
tick(t, r2)
|
||||||
|
if r.store.calls != 0 || r2.store.calls != 0 || len(r2.b.scheduled) != 0 {
|
||||||
|
t.Fatal("the scheduler ran while off")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// The jobs route marks the executor's scheduled backups, and the scheduler
|
||||||
|
// counts every unfinished backup and restore Job and nothing else.
|
||||||
|
func TestK8sScheduledBackupJobs(t *testing.T) {
|
||||||
|
scheme := runtime.NewScheme()
|
||||||
|
if err := clientgoscheme.AddToScheme(scheme); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
params := func(name string, scheduled bool) backupjob.JobParams {
|
||||||
|
return backupjob.JobParams{
|
||||||
|
Server: "survival", JobName: name, WorldPVC: "world-survival-0", BackupPVC: "felis-backups",
|
||||||
|
Namespace: "minecraft", Image: "felis:1", ConfigSecret: "felis-config", ConfigMount: "/etc/felis",
|
||||||
|
Scheduled: scheduled,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
scheduled, err := backupjob.BackupJob(params("backup-survival-aa", true))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
plain, err := backupjob.BackupJob(params("backup-survival-bb", false))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
plain.Status.Conditions = []batchv1.JobCondition{{Type: batchv1.JobComplete, Status: corev1.ConditionTrue}}
|
||||||
|
restoring := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "restore-other-cc",
|
||||||
|
Labels: map[string]string{jobServerLabel: "other", jobManagedByLabel: jobManagedByRestore}}}
|
||||||
|
foreign := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "files-survival-dd",
|
||||||
|
Labels: map[string]string{jobServerLabel: "survival", jobManagedByLabel: "felis-files"}}}
|
||||||
|
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(scheduled, plain, restoring, foreign).
|
||||||
|
WithStatusSubresource(&batchv1.Job{}).Build()
|
||||||
|
k := NewK8sJobStatus(c, "minecraft")
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
if n, err := k.RunningWorldJobs(ctx); err != nil || n != 2 {
|
||||||
|
t.Fatalf("RunningWorldJobs = %d, %v; want the scheduled backup and the restore", n, err)
|
||||||
|
}
|
||||||
|
jobs, err := k.LatestJobs(ctx, "survival")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
marks := map[string]bool{}
|
||||||
|
for _, j := range jobs {
|
||||||
|
marks[j.Name] = j.Scheduled
|
||||||
|
}
|
||||||
|
if len(marks) != 2 || !marks["backup-survival-aa"] || marks["backup-survival-bb"] {
|
||||||
|
t.Fatalf("scheduled marks = %v; want only backup-survival-aa", marks)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -205,6 +205,21 @@ func (b *Backuper) BackupThenRestore(ctx context.Context, serverName, formerOwne
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// BackupScheduled enqueues the daily restore point of a world (api.BackupScheduler):
|
||||||
|
// a backup Job like Backup's, recorded as a scheduled backup and pruned to
|
||||||
|
// [archive] scheduled_keep, so the owner's own backups keep their count.
|
||||||
|
func (b *Backuper) BackupScheduled(ctx context.Context, serverName, formerOwner string) error {
|
||||||
|
p := b.jobParams(serverName, formerOwner)
|
||||||
|
p.Scheduled = true
|
||||||
|
if err := b.Jobs.CreateBackupJob(ctx, p); err != nil {
|
||||||
|
if errors.Is(err, ErrAlreadyExists) {
|
||||||
|
return nil // suffix collision — treat as enqueued
|
||||||
|
}
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// jobNameSuffix is a short random hex tag that makes each backup Job name unique.
|
// jobNameSuffix is a short random hex tag that makes each backup Job name unique.
|
||||||
// 32 bits is ample: collisions only matter within a single Job's TTL window across
|
// 32 bits is ample: collisions only matter within a single Job's TTL window across
|
||||||
// a handful of manual backups.
|
// a handful of manual backups.
|
||||||
|
|||||||
@@ -59,3 +59,28 @@ func TestBackupThenRestoreChainsTheRestore(t *testing.T) {
|
|||||||
t.Errorf("JobName %q", p.JobName)
|
t.Errorf("JobName %q", p.JobName)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestBackupScheduledMarksTheJob(t *testing.T) {
|
||||||
|
jobs := &captureJobs{}
|
||||||
|
b := &Backuper{Jobs: jobs, Config: Config{Image: "img", BackupPVC: "pvc"}}
|
||||||
|
if err := b.BackupScheduled(context.Background(), "survival", "usr-1"); err != nil {
|
||||||
|
t.Fatalf("BackupScheduled: %v", err)
|
||||||
|
}
|
||||||
|
if len(jobs.got) != 1 {
|
||||||
|
t.Fatalf("created %d jobs, want 1", len(jobs.got))
|
||||||
|
}
|
||||||
|
p := jobs.got[0]
|
||||||
|
if !p.Scheduled || p.FormerOwner != "usr-1" || p.RestoreRef != "" {
|
||||||
|
t.Errorf("params = %+v", p)
|
||||||
|
}
|
||||||
|
if !strings.HasPrefix(p.JobName, BackupJobName("survival")+"-") {
|
||||||
|
t.Errorf("JobName %q", p.JobName)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := b.Backup(context.Background(), "survival", "usr-1"); err != nil {
|
||||||
|
t.Fatalf("Backup: %v", err)
|
||||||
|
}
|
||||||
|
if jobs.got[1].Scheduled {
|
||||||
|
t.Error("an on-demand backup is marked scheduled")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -29,6 +29,12 @@ const (
|
|||||||
|
|
||||||
// ReasonPreRestore is the world_backups reason of a safety snapshot.
|
// ReasonPreRestore is the world_backups reason of a safety snapshot.
|
||||||
ReasonPreRestore = "pre_restore"
|
ReasonPreRestore = "pre_restore"
|
||||||
|
// ReasonScheduled is the world_backups reason of the daily restore point
|
||||||
|
// felis-api takes of a world played since its last one.
|
||||||
|
ReasonScheduled = "scheduled"
|
||||||
|
// LabelReason marks a scheduled backup Job, so the jobs route can tell it
|
||||||
|
// from one somebody asked for.
|
||||||
|
LabelReason = "felis.lolicon.best/backup-reason"
|
||||||
|
|
||||||
worldVolume = "world"
|
worldVolume = "world"
|
||||||
backupVolume = "backup"
|
backupVolume = "backup"
|
||||||
@@ -72,6 +78,10 @@ type JobParams struct {
|
|||||||
// the restore will extract.
|
// the restore will extract.
|
||||||
RestoreRef string
|
RestoreRef string
|
||||||
RestoreBackupID string
|
RestoreBackupID string
|
||||||
|
// Scheduled records the backup as ReasonScheduled, pruned to its own
|
||||||
|
// [archive] scheduled_keep, and labels the Job LabelReason. It never carries
|
||||||
|
// a restore.
|
||||||
|
Scheduled bool
|
||||||
|
|
||||||
TTLAfterFinished time.Duration
|
TTLAfterFinished time.Duration
|
||||||
}
|
}
|
||||||
@@ -134,6 +144,9 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
|||||||
if p.RestoreRef != "" && p.RestoreBackupID == "" {
|
if p.RestoreRef != "" && p.RestoreBackupID == "" {
|
||||||
return nil, fmt.Errorf("backup: a chained restore needs the backup id")
|
return nil, fmt.Errorf("backup: a chained restore needs the backup id")
|
||||||
}
|
}
|
||||||
|
if p.RestoreRef != "" && p.Scheduled {
|
||||||
|
return nil, fmt.Errorf("backup: a scheduled backup carries no restore")
|
||||||
|
}
|
||||||
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -160,6 +173,9 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
|||||||
if p.RestoreRef != "" {
|
if p.RestoreRef != "" {
|
||||||
args = append(args, "--reason", ReasonPreRestore, "--protect", p.RestoreBackupID)
|
args = append(args, "--reason", ReasonPreRestore, "--protect", p.RestoreBackupID)
|
||||||
}
|
}
|
||||||
|
if p.Scheduled {
|
||||||
|
args = append(args, "--reason", ReasonScheduled)
|
||||||
|
}
|
||||||
|
|
||||||
container := corev1.Container{
|
container := corev1.Container{
|
||||||
Name: "backup",
|
Name: "backup",
|
||||||
@@ -212,6 +228,9 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
|||||||
annotationRestoreBackupID: p.RestoreBackupID,
|
annotationRestoreBackupID: p.RestoreBackupID,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if p.Scheduled {
|
||||||
|
meta.Labels[LabelReason] = ReasonScheduled
|
||||||
|
}
|
||||||
job := &batchv1.Job{
|
job := &batchv1.Job{
|
||||||
ObjectMeta: meta,
|
ObjectMeta: meta,
|
||||||
Spec: batchv1.JobSpec{
|
Spec: batchv1.JobSpec{
|
||||||
|
|||||||
@@ -237,6 +237,41 @@ func TestBackupJobCarriesTheRestoreChain(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A scheduled backup records itself as scheduled (so the Job prunes it to its
|
||||||
|
// own keep, not the owner's manual one) and says so on the Job for the jobs
|
||||||
|
// route; it protects nothing and carries no chain.
|
||||||
|
func TestBackupJobRecordsAScheduledBackup(t *testing.T) {
|
||||||
|
p := sampleJobParams()
|
||||||
|
p.Scheduled = true
|
||||||
|
job, err := BackupJob(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob: %v", err)
|
||||||
|
}
|
||||||
|
args := job.Spec.Template.Spec.Containers[0].Args
|
||||||
|
if !argsContain(args, "--reason", ReasonScheduled) {
|
||||||
|
t.Errorf("args = %v, want --reason %s", args, ReasonScheduled)
|
||||||
|
}
|
||||||
|
for _, a := range args {
|
||||||
|
if a == "--protect" {
|
||||||
|
t.Errorf("a scheduled backup passes --protect: %v", args)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if job.Labels[LabelReason] != ReasonScheduled {
|
||||||
|
t.Errorf("job labels = %v, want %s=%s", job.Labels, LabelReason, ReasonScheduled)
|
||||||
|
}
|
||||||
|
if _, ok := job.Labels[labelThenRestore]; ok {
|
||||||
|
t.Errorf("a scheduled backup carries a chain: %v", job.Labels)
|
||||||
|
}
|
||||||
|
|
||||||
|
plain, err := BackupJob(sampleJobParams())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("BackupJob(plain): %v", err)
|
||||||
|
}
|
||||||
|
if _, ok := plain.Labels[LabelReason]; ok {
|
||||||
|
t.Errorf("a plain backup is labelled scheduled: %v", plain.Labels)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
||||||
for _, tc := range []struct {
|
for _, tc := range []struct {
|
||||||
name string
|
name string
|
||||||
@@ -247,6 +282,9 @@ func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
|||||||
{"no backup pvc", func(p *JobParams) { p.BackupPVC = "" }},
|
{"no backup pvc", func(p *JobParams) { p.BackupPVC = "" }},
|
||||||
{"no config secret", func(p *JobParams) { p.ConfigSecret = "" }},
|
{"no config secret", func(p *JobParams) { p.ConfigSecret = "" }},
|
||||||
{"chain without backup id", func(p *JobParams) { p.RestoreRef = "/backups/a.tar.gz" }},
|
{"chain without backup id", func(p *JobParams) { p.RestoreRef = "/backups/a.tar.gz" }},
|
||||||
|
{"scheduled with a chain", func(p *JobParams) {
|
||||||
|
p.Scheduled, p.RestoreRef, p.RestoreBackupID = true, "/backups/a.tar.gz", "bk-1"
|
||||||
|
}},
|
||||||
} {
|
} {
|
||||||
t.Run(tc.name, func(t *testing.T) {
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
p := sampleJobParams()
|
p := sampleJobParams()
|
||||||
|
|||||||
@@ -287,7 +287,10 @@ type RegistryS3Config struct {
|
|||||||
// The manual_* keys bound the owners' on-demand backups, which share the
|
// The manual_* keys bound the owners' on-demand backups, which share the
|
||||||
// archive store with the reaper's: how long each is kept (default 30d), how
|
// archive store with the reaper's: how long each is kept (default 30d), how
|
||||||
// many per server (default 5, the oldest go first), and how soon an owner may
|
// many per server (default 5, the oldest go first), and how soon an owner may
|
||||||
// ask for the next one (default 10m). Empty or zero means the default.
|
// ask for the next one (default 10m). The scheduled_* keys shape the daily
|
||||||
|
// restore points felis-api takes of played worlds: how far apart (default 1d,
|
||||||
|
// "0s" turns them off), how many per server (default 7) and how long each is
|
||||||
|
// kept (default 90d). Empty or zero means the default.
|
||||||
type ArchiveConfig struct {
|
type ArchiveConfig struct {
|
||||||
Store string `toml:"store"`
|
Store string `toml:"store"`
|
||||||
LocalPath string `toml:"local_path"`
|
LocalPath string `toml:"local_path"`
|
||||||
@@ -297,6 +300,9 @@ type ArchiveConfig struct {
|
|||||||
ManualRetention string `toml:"manual_retention"`
|
ManualRetention string `toml:"manual_retention"`
|
||||||
ManualKeep int `toml:"manual_keep"`
|
ManualKeep int `toml:"manual_keep"`
|
||||||
ManualCooldown string `toml:"manual_cooldown"`
|
ManualCooldown string `toml:"manual_cooldown"`
|
||||||
|
ScheduledEvery string `toml:"scheduled_every"`
|
||||||
|
ScheduledKeep int `toml:"scheduled_keep"`
|
||||||
|
ScheduledRetention string `toml:"scheduled_retention"`
|
||||||
S3 ArchiveS3Config `toml:"s3"`
|
S3 ArchiveS3Config `toml:"s3"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -192,7 +192,7 @@ func TestManualBackupRationing(t *testing.T) {
|
|||||||
insert(sole, "inactive_15d", now.Add(-100*reaper.Day), false, 1000)
|
insert(sole, "inactive_15d", now.Add(-100*reaper.Day), false, 1000)
|
||||||
insert(copied, "inactive_15d", now.Add(-50*reaper.Day), true, 100)
|
insert(copied, "inactive_15d", now.Add(-50*reaper.Day), true, 100)
|
||||||
|
|
||||||
excess, err := st.ExcessBackups(ctx, name, "manual", 5, "")
|
excess, err := st.ExcessBackups(ctx, name, "", "manual", 5, "")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("ExcessBackups: %v", err)
|
t.Fatalf("ExcessBackups: %v", err)
|
||||||
}
|
}
|
||||||
@@ -200,14 +200,14 @@ func TestManualBackupRationing(t *testing.T) {
|
|||||||
t.Fatalf("excess = %+v; want the two oldest manual backups, oldest first", excess)
|
t.Fatalf("excess = %+v; want the two oldest manual backups, oldest first", excess)
|
||||||
}
|
}
|
||||||
// The backup a chained restore will extract is never pruned.
|
// The backup a chained restore will extract is never pruned.
|
||||||
excess, err = st.ExcessBackups(ctx, name, "manual", 5, manual[0])
|
excess, err = st.ExcessBackups(ctx, name, "", "manual", 5, manual[0])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("ExcessBackups(protect): %v", err)
|
t.Fatalf("ExcessBackups(protect): %v", err)
|
||||||
}
|
}
|
||||||
if len(excess) != 1 || excess[0].ID != manual[1] {
|
if len(excess) != 1 || excess[0].ID != manual[1] {
|
||||||
t.Fatalf("excess with %s protected = %+v; want only %s", manual[0], excess, manual[1])
|
t.Fatalf("excess with %s protected = %+v; want only %s", manual[0], excess, manual[1])
|
||||||
}
|
}
|
||||||
if excess, err = st.ExcessBackups(ctx, name, "pre_restore", 0, ""); err != nil || len(excess) != 0 {
|
if excess, err = st.ExcessBackups(ctx, name, "", "pre_restore", 0, ""); err != nil || len(excess) != 0 {
|
||||||
t.Fatalf("pre_restore excess = %+v, %v; the manual ones are not its to prune", excess, err)
|
t.Fatalf("pre_restore excess = %+v, %v; the manual ones are not its to prune", excess, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -366,7 +366,7 @@ func TestBackupReadBack(t *testing.T) {
|
|||||||
t.Fatalf("AllBackups listed %d of the 2 backups", seen)
|
t.Fatalf("AllBackups listed %d of the 2 backups", seen)
|
||||||
}
|
}
|
||||||
|
|
||||||
if excess, err := st.ExcessBackups(ctx, name, "inactive_15d", 1, ""); err != nil || len(excess) != 1 || excess[0].ID != newer {
|
if excess, err := st.ExcessBackups(ctx, name, "", "inactive_15d", 1, ""); err != nil || len(excess) != 1 || excess[0].ID != newer {
|
||||||
t.Fatalf("ExcessBackups(keep 1) = (%+v, %v); want the corrupt %s pruned first", excess, err, newer)
|
t.Fatalf("ExcessBackups(keep 1) = (%+v, %v); want the corrupt %s pruned first", excess, err, newer)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,140 @@
|
|||||||
|
//go:build pgint
|
||||||
|
|
||||||
|
package pgint
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/reaper"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestScheduledBackupCandidates pins the SQL behind felis-api's scheduled
|
||||||
|
// backups: an owned world joined since its owner's newest intact scheduled
|
||||||
|
// backup is due once that backup is older than the period, worlds without one
|
||||||
|
// first. A previous owner's, a corrupt, a deleted, a pre-claim or a manual
|
||||||
|
// backup is no restore point of the current owner's world.
|
||||||
|
func TestScheduledBackupCandidates(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
u := newUser(t, "user", "sched")
|
||||||
|
prev := newUser(t, "user", "sched-prev")
|
||||||
|
now := time.Now().UTC()
|
||||||
|
sfx := suffix(t)
|
||||||
|
server := func(tag, owner string, claimed, active time.Time, deleted bool) string {
|
||||||
|
t.Helper()
|
||||||
|
name := "sch-" + tag + "-" + sfx
|
||||||
|
var deletedAt any
|
||||||
|
if deleted {
|
||||||
|
deletedAt = now
|
||||||
|
}
|
||||||
|
if _, err := db.ExecContext(ctx,
|
||||||
|
`INSERT INTO servers (name, owner_id, claimed_at, last_active_at, deleted_at, cached_cpu_milli, cached_memory_mb, cached_storage_mb)
|
||||||
|
VALUES ($1, NULLIF($2, ''), $3, $4, $5, 100, 128, 1)`,
|
||||||
|
name, owner, claimed, active, deletedAt); err != nil {
|
||||||
|
t.Fatalf("seed server %s: %v", name, err)
|
||||||
|
}
|
||||||
|
return name
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
backup := func(server, owner, reason, status string, created time.Time, corrupt bool) {
|
||||||
|
t.Helper()
|
||||||
|
n++
|
||||||
|
id := "bk-sch-" + sfx + "-" + string(rune('a'+n))
|
||||||
|
var corruptAt any
|
||||||
|
if corrupt {
|
||||||
|
corruptAt = created
|
||||||
|
}
|
||||||
|
if _, err := db.ExecContext(ctx,
|
||||||
|
`INSERT INTO world_backups (id, server_name, former_owner, backup_ref, size_bytes, reason, status, created_at, expires_at, corrupt_at)
|
||||||
|
VALUES ($1, $2, NULLIF($3, ''), $4, 1, $5, $6, $7, $8, $9)`,
|
||||||
|
id, server, owner, "/archives/"+id+".tar.gz", reason, status, created, created.Add(90*reaper.Day), corruptAt); err != nil {
|
||||||
|
t.Fatalf("seed backup of %s: %v", server, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
h := time.Hour
|
||||||
|
claimed := now.Add(-10 * reaper.Day)
|
||||||
|
|
||||||
|
fresh := server("fresh", u.ID, claimed, now.Add(-h), false)
|
||||||
|
stale := server("stale", u.ID, claimed, now.Add(-h), false)
|
||||||
|
backup(stale, u.ID, "scheduled", "present", now.Add(-30*h), false)
|
||||||
|
recent := server("recent", u.ID, claimed, now.Add(-h), false)
|
||||||
|
backup(recent, u.ID, "scheduled", "present", now.Add(-2*h), false)
|
||||||
|
idle := server("idle", u.ID, claimed, now.Add(-40*h), false)
|
||||||
|
backup(idle, u.ID, "scheduled", "present", now.Add(-30*h), false)
|
||||||
|
prevOwner := server("prevowner", u.ID, claimed, now.Add(-h), false)
|
||||||
|
backup(prevOwner, prev.ID, "scheduled", "present", now.Add(-2*h), false)
|
||||||
|
preClaim := server("preclaim", u.ID, now.Add(-2*h), now.Add(-2*h), false)
|
||||||
|
backup(preClaim, u.ID, "scheduled", "present", now.Add(-3*h), false)
|
||||||
|
corrupt := server("corrupt", u.ID, claimed, now.Add(-h), false)
|
||||||
|
backup(corrupt, u.ID, "scheduled", "present", now.Add(-30*time.Minute), true)
|
||||||
|
deletedBk := server("deletedbk", u.ID, claimed, now.Add(-h), false)
|
||||||
|
backup(deletedBk, u.ID, "scheduled", "deleted", now.Add(-30*time.Minute), false)
|
||||||
|
manual := server("manual", u.ID, claimed, now.Add(-h), false)
|
||||||
|
backup(manual, u.ID, "manual", "present", now.Add(-30*time.Minute), false)
|
||||||
|
server("unowned", "", claimed, now.Add(-h), false)
|
||||||
|
server("gone", u.ID, claimed, now.Add(-h), true)
|
||||||
|
|
||||||
|
got, err := repo.ScheduledBackupCandidates(ctx, now.Add(-24*h))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ScheduledBackupCandidates: %v", err)
|
||||||
|
}
|
||||||
|
var mine []string
|
||||||
|
for _, c := range got {
|
||||||
|
if strings.HasSuffix(c.Name, sfx) {
|
||||||
|
if c.OwnerID != u.ID {
|
||||||
|
t.Fatalf("candidate %+v; want owner %s", c, u.ID)
|
||||||
|
}
|
||||||
|
mine = append(mine, c.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []string{corrupt, deletedBk, fresh, manual, preClaim, prevOwner, stale}
|
||||||
|
if strings.Join(mine, " ") != strings.Join(want, " ") {
|
||||||
|
t.Fatalf("due = %v\nwant %v (worlds without a point by name, then the longest waiting)", mine, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestExcessBackupsPerOwner pins that the backup Job's keep-N prune counts only
|
||||||
|
// the new backup's owner's backups: the ones a previous owner took of the same
|
||||||
|
// world stay theirs to restore until their own retention runs out.
|
||||||
|
func TestExcessBackupsPerOwner(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
st := reaper.NewPGStore(db)
|
||||||
|
u := newUser(t, "user", "keep")
|
||||||
|
prev := newUser(t, "user", "keep-prev")
|
||||||
|
sfx := suffix(t)
|
||||||
|
name := "keep-" + sfx
|
||||||
|
if _, err := db.ExecContext(ctx,
|
||||||
|
`INSERT INTO servers (name, owner_id, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, $2, 100, 128, 1)`,
|
||||||
|
name, u.ID); err != nil {
|
||||||
|
t.Fatalf("seed server: %v", err)
|
||||||
|
}
|
||||||
|
now := time.Now().UTC()
|
||||||
|
var ids []string
|
||||||
|
for i, owner := range []string{prev.ID, prev.ID, prev.ID, u.ID, u.ID} {
|
||||||
|
id := "bk-keep-" + sfx + "-" + string(rune('0'+i))
|
||||||
|
ids = append(ids, id)
|
||||||
|
created := now.Add(time.Duration(i-5) * time.Hour)
|
||||||
|
if _, err := db.ExecContext(ctx,
|
||||||
|
`INSERT INTO world_backups (id, server_name, former_owner, backup_ref, size_bytes, reason, status, created_at, expires_at)
|
||||||
|
VALUES ($1, $2, $3, $4, 1, 'scheduled', 'present', $5, $6)`,
|
||||||
|
id, name, owner, "/archives/"+id+".tar.gz", created, created.Add(90*reaper.Day)); err != nil {
|
||||||
|
t.Fatalf("seed %s: %v", id, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
excess, err := st.ExcessBackups(ctx, name, u.ID, "scheduled", 1, "")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ExcessBackups: %v", err)
|
||||||
|
}
|
||||||
|
if len(excess) != 1 || excess[0].ID != ids[3] {
|
||||||
|
t.Fatalf("excess for the owner = %+v; want only %s", excess, ids[3])
|
||||||
|
}
|
||||||
|
excess, err = st.ExcessBackups(ctx, name, prev.ID, "scheduled", 1, "")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ExcessBackups(previous owner): %v", err)
|
||||||
|
}
|
||||||
|
if len(excess) != 2 || excess[0].ID != ids[0] || excess[1].ID != ids[1] {
|
||||||
|
t.Fatalf("excess for the previous owner = %+v; want %s, %s", excess, ids[0], ids[1])
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -123,15 +123,17 @@ func (s *PGStore) EvictableBackups(ctx context.Context) ([]StoredBackup, error)
|
|||||||
return s.queryBackups(ctx, q)
|
return s.queryBackups(ctx, q)
|
||||||
}
|
}
|
||||||
|
|
||||||
// ExcessBackups lists server's present backups of one reason beyond the newest
|
// ExcessBackups lists the present backups of one reason that owner holds of
|
||||||
// keep, oldest first: what the backup Job removes after adding one. protect, when
|
// server beyond the newest keep, oldest first: what the backup Job removes after
|
||||||
// set, is a backup id left out of the list whatever its age (the one a chained
|
// adding one. Another owner's backups of the same server (one it held before the
|
||||||
// restore is about to extract).
|
// world was reaped and claimed again) are theirs to restore and never counted.
|
||||||
func (s *PGStore) ExcessBackups(ctx context.Context, server, reason string, keep int, protect string) ([]StoredBackup, error) {
|
// protect, when set, is a backup id left out of the list whatever its age (the
|
||||||
|
// one a chained restore is about to extract).
|
||||||
|
func (s *PGStore) ExcessBackups(ctx context.Context, server, owner, reason string, keep int, protect string) ([]StoredBackup, error) {
|
||||||
const q = `SELECT id, server_name, backup_ref, size_bytes, reason, COALESCE(sha256, '') FROM world_backups
|
const q = `SELECT id, server_name, backup_ref, size_bytes, reason, COALESCE(sha256, '') FROM world_backups
|
||||||
WHERE server_name = $1 AND status = 'present' AND reason = $2 AND id <> $4
|
WHERE server_name = $1 AND COALESCE(former_owner, '') = $5 AND status = 'present' AND reason = $2 AND id <> $4
|
||||||
ORDER BY corrupt_at IS NULL DESC, created_at DESC OFFSET $3`
|
ORDER BY corrupt_at IS NULL DESC, created_at DESC OFFSET $3`
|
||||||
out, err := s.queryBackups(ctx, q, server, reason, keep, protect)
|
out, err := s.queryBackups(ctx, q, server, reason, keep, protect, owner)
|
||||||
for i, j := 0, len(out)-1; i < j; i, j = i+1, j-1 {
|
for i, j := 0, len(out)-1; i < j; i, j = i+1, j-1 {
|
||||||
out[i], out[j] = out[j], out[i]
|
out[i], out[j] = out[j], out[i]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -96,6 +96,14 @@ type Config struct {
|
|||||||
// ManualCooldown is the shortest gap between two owner-requested backups
|
// ManualCooldown is the shortest gap between two owner-requested backups
|
||||||
// of one server (default 10m); operators are not held to it.
|
// of one server (default 10m); operators are not held to it.
|
||||||
ManualCooldown time.Duration
|
ManualCooldown time.Duration
|
||||||
|
// ScheduledEvery is the spacing of the restore points felis-api takes of
|
||||||
|
// a world that has been played since its last one (default 1d; 0 turns
|
||||||
|
// them off). ScheduledKeep caps how many a server keeps (default 7), and
|
||||||
|
// ScheduledRetention is how long each is kept (default 90d, the reaper's
|
||||||
|
// retention): the newest ones are all a quiet world has.
|
||||||
|
ScheduledEvery time.Duration
|
||||||
|
ScheduledKeep int
|
||||||
|
ScheduledRetention time.Duration
|
||||||
// RequireOffsite holds each deletion until the world's archive has its
|
// RequireOffsite holds each deletion until the world's archive has its
|
||||||
// off-site copy ([offsite] configured; internal/offsite records the copy).
|
// off-site copy ([offsite] configured; internal/offsite records the copy).
|
||||||
// The archive is written on the run that finds the world idle, and the
|
// The archive is written on the run that finds the world idle, and the
|
||||||
@@ -124,6 +132,10 @@ func DefaultConfig() Config {
|
|||||||
ManualKeep: 5,
|
ManualKeep: 5,
|
||||||
ManualCooldown: 10 * time.Minute,
|
ManualCooldown: 10 * time.Minute,
|
||||||
|
|
||||||
|
ScheduledEvery: Day,
|
||||||
|
ScheduledKeep: 7,
|
||||||
|
ScheduledRetention: 90 * Day,
|
||||||
|
|
||||||
VerifyEvery: 7 * Day,
|
VerifyEvery: 7 * Day,
|
||||||
VerifyPerRun: 10,
|
VerifyPerRun: 10,
|
||||||
PartialAfter: 6 * time.Hour,
|
PartialAfter: 6 * time.Hour,
|
||||||
|
|||||||
@@ -1,14 +1,15 @@
|
|||||||
{
|
{
|
||||||
"title": "Backups & restore",
|
"title": "Backups & restore",
|
||||||
"subtitle": "A backup is saved automatically when a world is archived after long inactivity. You can use these backups to restore the world — the server must be stopped first.",
|
"subtitle": "A world that was played is backed up automatically once a day after its server stops, and again before it is archived after long inactivity. You can use these backups to restore the world — the server must be stopped first.",
|
||||||
"back_to_console": "Back to console",
|
"back_to_console": "Back to console",
|
||||||
"not_yours_title": "No permission to access backups",
|
"not_yours_title": "No permission to access backups",
|
||||||
"not_yours_body": "Only the owner or an admin can view and restore this server's backups.",
|
"not_yours_body": "Only the owner or an admin can view and restore this server's backups.",
|
||||||
"latest_title": "Latest backup",
|
"latest_title": "Latest backup",
|
||||||
"history_note": "Backups can be restored until they expire, then they are cleaned up automatically. Each server keeps only its most recent manual backups (older ones are removed once a new one finishes) and its last 3 pre-restore snapshots.",
|
"history_note": "Backups can be restored until they expire, then they are cleaned up automatically. Each server keeps only its most recent manual backups and its most recent scheduled backups (older ones of the same kind are removed once a new one finishes), and its last 3 pre-restore snapshots.",
|
||||||
"reason_inactive": "Idle archive",
|
"reason_inactive": "Idle archive",
|
||||||
"reason_manual": "Manual backup",
|
"reason_manual": "Manual backup",
|
||||||
"reason_pre_restore": "Pre-restore snapshot",
|
"reason_pre_restore": "Pre-restore snapshot",
|
||||||
|
"reason_scheduled": "Scheduled backup",
|
||||||
"reason_label": "Reason: {{reason}}",
|
"reason_label": "Reason: {{reason}}",
|
||||||
"backup_now": "Back up now",
|
"backup_now": "Back up now",
|
||||||
"backup_in_progress": "Starting backup…",
|
"backup_in_progress": "Starting backup…",
|
||||||
@@ -18,12 +19,13 @@
|
|||||||
"expired": "Expired",
|
"expired": "Expired",
|
||||||
"former_owner": "Former owner: {{owner}}",
|
"former_owner": "Former owner: {{owner}}",
|
||||||
"empty_title": "No backups for this server",
|
"empty_title": "No backups for this server",
|
||||||
"empty_hint": "A backup is saved when a world is archived after long inactivity — or create one now with “Back up now”.",
|
"empty_hint": "A scheduled backup is saved once the server has run and stopped — or create one now with “Back up now”.",
|
||||||
"jobs_title": "Recent operations",
|
"jobs_title": "Recent operations",
|
||||||
"jobs_empty": "No recent backup or restore operations.",
|
"jobs_empty": "No recent backup or restore operations.",
|
||||||
"job_backup": "Backup",
|
"job_backup": "Backup",
|
||||||
"job_restore": "Restore",
|
"job_restore": "Restore",
|
||||||
"job_pre_restore": "Pre-restore snapshot",
|
"job_pre_restore": "Pre-restore snapshot",
|
||||||
|
"job_scheduled": "Scheduled backup",
|
||||||
"job_running": "Running",
|
"job_running": "Running",
|
||||||
"job_succeeded": "Succeeded",
|
"job_succeeded": "Succeeded",
|
||||||
"job_failed": "Failed",
|
"job_failed": "Failed",
|
||||||
|
|||||||
@@ -1,14 +1,15 @@
|
|||||||
{
|
{
|
||||||
"title": "备份与恢复",
|
"title": "备份与恢复",
|
||||||
"subtitle": "服务器长期闲置并被自动回收前,系统会自动创建备份。可以使用备份随时恢复世界,恢复前需要先停止服务器。",
|
"subtitle": "玩过的世界会在服务器停止后每天自动备份一次,长期闲置被自动回收前也会自动备份。可以使用备份随时恢复世界,恢复前需要先停止服务器。",
|
||||||
"back_to_console": "返回控制台",
|
"back_to_console": "返回控制台",
|
||||||
"not_yours_title": "无权访问备份",
|
"not_yours_title": "无权访问备份",
|
||||||
"not_yours_body": "只有所有者或管理员才能查看并恢复该服务器的备份。",
|
"not_yours_body": "只有所有者或管理员才能查看并恢复该服务器的备份。",
|
||||||
"latest_title": "最新备份",
|
"latest_title": "最新备份",
|
||||||
"history_note": "历史备份在过期前均可用于恢复,到期后自动清理。手动备份每台服务器只保留最近几份,新备份完成后会自动删除更早的手动备份;恢复前快照保留最近 3 份。",
|
"history_note": "历史备份在过期前均可用于恢复,到期后自动清理。手动备份和定时备份每台服务器各只保留最近几份,新备份完成后会自动删除更早的同类备份;恢复前快照保留最近 3 份。",
|
||||||
"reason_inactive": "闲置自动回收",
|
"reason_inactive": "闲置自动回收",
|
||||||
"reason_manual": "手动备份",
|
"reason_manual": "手动备份",
|
||||||
"reason_pre_restore": "恢复前快照",
|
"reason_pre_restore": "恢复前快照",
|
||||||
|
"reason_scheduled": "定时备份",
|
||||||
"reason_label": "原因:{{reason}}",
|
"reason_label": "原因:{{reason}}",
|
||||||
"backup_now": "立即备份",
|
"backup_now": "立即备份",
|
||||||
"backup_in_progress": "正在启动备份……",
|
"backup_in_progress": "正在启动备份……",
|
||||||
@@ -18,12 +19,13 @@
|
|||||||
"expired": "已过期",
|
"expired": "已过期",
|
||||||
"former_owner": "原世界所有者:{{owner}}",
|
"former_owner": "原世界所有者:{{owner}}",
|
||||||
"empty_title": "暂无备份",
|
"empty_title": "暂无备份",
|
||||||
"empty_hint": "服务器长期闲置被自动回收时会生成备份,也可以点“立即备份”马上创建。",
|
"empty_hint": "服务器运行过并停止后会自动生成定时备份,也可以点“立即备份”马上创建。",
|
||||||
"jobs_title": "最近操作",
|
"jobs_title": "最近操作",
|
||||||
"jobs_empty": "暂无备份或恢复操作记录。",
|
"jobs_empty": "暂无备份或恢复操作记录。",
|
||||||
"job_backup": "备份",
|
"job_backup": "备份",
|
||||||
"job_restore": "恢复",
|
"job_restore": "恢复",
|
||||||
"job_pre_restore": "恢复前快照",
|
"job_pre_restore": "恢复前快照",
|
||||||
|
"job_scheduled": "定时备份",
|
||||||
"job_running": "进行中",
|
"job_running": "进行中",
|
||||||
"job_succeeded": "成功",
|
"job_succeeded": "成功",
|
||||||
"job_failed": "失败",
|
"job_failed": "失败",
|
||||||
|
|||||||
@@ -2420,7 +2420,7 @@ export interface components {
|
|||||||
former_owner?: string;
|
former_owner?: string;
|
||||||
/** Format: int64 */
|
/** Format: int64 */
|
||||||
size_bytes: number;
|
size_bytes: number;
|
||||||
/** @description inactive_15d (idle reclaim), manual (on demand) or pre_restore (the safety snapshot in front of a restore). */
|
/** @description inactive_15d (idle reclaim), manual (on demand), pre_restore (the safety snapshot in front of a restore) or scheduled (the daily restore point felis-api takes of a world played since its last one, once the server stops). */
|
||||||
reason: string;
|
reason: string;
|
||||||
status: string;
|
status: string;
|
||||||
/** Format: date-time */
|
/** Format: date-time */
|
||||||
@@ -5548,6 +5548,8 @@ export interface operations {
|
|||||||
then_restore_reason?: "snapshot_failed" | "not_configured" | "server_gone" | "server_started" | "restore_busy";
|
then_restore_reason?: "snapshot_failed" | "not_configured" | "server_gone" | "server_started" | "restore_busy";
|
||||||
/** @description The backup the chained restore extracts. */
|
/** @description The backup the chained restore extracts. */
|
||||||
restore_backup_id?: string;
|
restore_backup_id?: string;
|
||||||
|
/** @description A backup felis-api took on its own: the daily restore point of a played world. Omitted when false. */
|
||||||
|
scheduled?: boolean;
|
||||||
}[];
|
}[];
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -214,6 +214,9 @@ export interface ServerJob {
|
|||||||
then_restore?: string;
|
then_restore?: string;
|
||||||
then_restore_reason?: string;
|
then_restore_reason?: string;
|
||||||
restore_backup_id?: string;
|
restore_backup_id?: string;
|
||||||
|
/** A backup felis-api took on its own: the daily restore point of a world
|
||||||
|
* played since its last one, taken once the server stops. */
|
||||||
|
scheduled?: boolean;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** WhitelistImage is one row of GET /images (the create-form dropdown source). */
|
/** WhitelistImage is one row of GET /images (the create-form dropdown source). */
|
||||||
|
|||||||
@@ -82,6 +82,23 @@ describe("ServerBackups", () => {
|
|||||||
expect(screen.queryByText("Latest backup")).toBeNull();
|
expect(screen.queryByText("Latest backup")).toBeNull();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("names the daily restore points felis-api takes, in the list and in recent operations", async () => {
|
||||||
|
calls.listBackups.mockResolvedValue({
|
||||||
|
backups: [{ ...backup("bk-sched", 3), reason: "scheduled" }, backup("bk-manual", 5)],
|
||||||
|
total: 2,
|
||||||
|
});
|
||||||
|
calls.serverJobs.mockResolvedValue([
|
||||||
|
{ name: "backup-survival-aa", kind: "backup", state: "succeeded", scheduled: true },
|
||||||
|
{ name: "backup-survival-bb", kind: "backup", state: "succeeded" },
|
||||||
|
]);
|
||||||
|
renderPage();
|
||||||
|
const sched = await screen.findByText("3 hours ago");
|
||||||
|
expect(sched.closest("tr")?.textContent).toContain("Scheduled backup");
|
||||||
|
expect(screen.getByText("5 hours ago").closest("tr")?.textContent).toContain("Manual backup");
|
||||||
|
const ops = await screen.findAllByText(/^(Scheduled backup|Backup)$/, { selector: "li span" });
|
||||||
|
expect(ops.map((el) => el.textContent)).toEqual(["Scheduled backup", "Backup"]);
|
||||||
|
});
|
||||||
|
|
||||||
it("shows no pager when the server's backups fit on one page", async () => {
|
it("shows no pager when the server's backups fit on one page", async () => {
|
||||||
calls.listBackups.mockResolvedValue({ backups: [backup("bk-1", 3)], total: 1 });
|
calls.listBackups.mockResolvedValue({ backups: [backup("bk-1", 3)], total: 1 });
|
||||||
renderPage();
|
renderPage();
|
||||||
|
|||||||
@@ -75,6 +75,8 @@ function BackupRow({
|
|||||||
? t("reason_manual")
|
? t("reason_manual")
|
||||||
: b.reason === "pre_restore"
|
: b.reason === "pre_restore"
|
||||||
? t("reason_pre_restore")
|
? t("reason_pre_restore")
|
||||||
|
: b.reason === "scheduled"
|
||||||
|
? t("reason_scheduled")
|
||||||
: t("reason_label", { reason: b.reason });
|
: t("reason_label", { reason: b.reason });
|
||||||
|
|
||||||
// Below md the row stops being a table row: the when/why block takes the full
|
// Below md the row stops being a table row: the when/why block takes the full
|
||||||
@@ -684,6 +686,8 @@ export function ServerBackups() {
|
|||||||
: j.kind === "backup"
|
: j.kind === "backup"
|
||||||
? j.then_restore
|
? j.then_restore
|
||||||
? t("job_pre_restore")
|
? t("job_pre_restore")
|
||||||
|
: j.scheduled
|
||||||
|
? t("job_scheduled")
|
||||||
: t("job_backup")
|
: t("job_backup")
|
||||||
: j.kind}
|
: j.kind}
|
||||||
</span>
|
</span>
|
||||||
|
|||||||
Reference in new issue
Block a user