feat(restore): 恢复前默认为当前世界做安全快照并串接恢复 Job,快照失败则不恢复;并发恢复另一备份返回 409

This commit is contained in:
Lemon-miaow committed 2026-09-25 02:52:18 +08:00
1 parent d44243c0ac
commit 489eff4494
33 files changed
+1260 -105

No files matched your search

+34 -10
View File
@@ -287,6 +287,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
}
cluster := api.NewK8sCluster(cl, cfg.K8s.Namespace)
jobStatus := api.NewK8sJobStatus(cl, cfg.K8s.Namespace)
a := &api.API{
Repo: repo,
Cluster: cluster,
@@ -294,16 +295,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
Logs: api.NewK8sLogStreamer(clientset, cfg.K8s.Namespace),
// Build-log stream (spec §16) is scoped to the BUILD namespace — the same
// value the Builder renders Jobs into — so it follows where build Pods run.
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
Internal: api.BearerTokenAuth{Token: token},
Builder: builder,
Images: imagePinner(cfg.Registry.URL),
Restorer: restorer,
Backuper: backuper,
JobStatus: api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
Files: files,
Submissions: submissions,
Mailer: mailer,
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
Internal: api.BearerTokenAuth{Token: token},
Builder: builder,
Images: imagePinner(cfg.Registry.URL),
Restorer: restorer,
Backuper: backuper,
JobStatus: jobStatus,
// A restore starts with a safety snapshot; settleRestoreChains starts the
// restore behind each one.
RestoreChains: jobStatus,
Files: files,
Submissions: submissions,
Mailer: mailer,
// The external face is fronted by SessionAuth: it prefers a local session
// cookie (minted by the passwordless doors) and otherwise delegates to the
// Cloudflare-Access JWT verifier, so both auth models coexist on one face. The
@@ -410,6 +414,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// and advance any whose Job has reached a terminal phase. GET on a build also
// reconciles it, but this loop converges builds nobody is polling.
go reconcileBuilds(ctx, builder, stderr)
go settleRestoreChains(ctx, a, stderr)
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
go pruner.Loop(ctx, registryPruneInterval)
@@ -650,6 +655,25 @@ func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
}
}
// settleRestoreChains starts the restore behind each safety snapshot that has
// finished (and gives up the one behind a snapshot that failed). The world stays
// locked in between, so the interval is how long a finished snapshot keeps the
// server down before its restore begins.
func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
t := time.NewTicker(5 * time.Second)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
if err := a.SettleRestoreChains(ctx); err != nil {
fmt.Fprintf(stderr, "felis api: restore chains: %v\n", err)
}
}
}
}
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
// submissions rejected more than submit.RejectedContextRetention ago. Without it a
// rejected modpack keeps its bytes on the uploads store (and against its
+29 -9
View File
@@ -10,6 +10,7 @@ import (
"time"
"felis.lolicon.best/internal/backup"
"felis.lolicon.best/internal/backupjob"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/reaper"
@@ -36,6 +37,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
server := fs.String("server", "", "server name whose world is being backed up")
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
if err := fs.Parse(args); err != nil {
return 2
}
@@ -43,6 +46,11 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
fmt.Fprintln(stderr, "felis backup: --server is required")
return 2
}
keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
if !ok {
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
return 2
}
cfg, err := config.Load(*cfgPath)
if err != nil {
@@ -60,6 +68,9 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
if keep < 0 {
keep = rcfg.ManualKeep
}
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
@@ -99,7 +110,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
FormerOwner: *formerOwner,
BackupRef: string(ref),
SizeBytes: size,
Reason: "manual",
Reason: *reason,
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
}
st := reaper.NewPGStore(drv.DB())
@@ -116,16 +127,25 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
}
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
pruneManualBackups(ctx, st, archiver, *server, rcfg.ManualKeep, stdout, stderr)
pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
return 0
}
// pruneManualBackups keeps server's newest keep on-demand backups and removes
// the rest, oldest first, so repeated backups of one world cannot fill the
// shared archive store. The new backup is already recorded; a removal that
// fails is reported and retried after the next backup.
func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server string, keep int, stdout, stderr io.Writer) {
excess, err := st.ExcessManualBackups(ctx, server, keep)
const (
reasonManual = "manual"
// preRestoreKeep is how many safety snapshots a server keeps: enough to walk
// back a couple of restores in a row, without every restore adding a world's
// worth of bytes for the full manual retention.
preRestoreKeep = 3
)
// pruneBackups keeps server's newest keep backups of this reason and removes the
// rest, oldest first, so repeated backups of one world cannot fill the shared
// archive store. protect is never removed: it is the backup a chained restore is
// about to extract. The new backup is already recorded; a removal that fails is
// reported and retried after the next backup.
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
if err != nil {
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
return
@@ -139,7 +159,7 @@ func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup
fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
continue
}
fmt.Fprintf(stdout, "felis backup: removed older backup %s of %s (keeping the newest %d)\n", b.ID, server, keep)
fmt.Fprintf(stdout, "felis backup: removed older %s backup %s of %s (keeping the newest %d)\n", reason, b.ID, server, keep)
}
}
+19
View File
@@ -0,0 +1,19 @@
package main
import (
"bytes"
"strings"
"testing"
)
// The reason decides which backups the new one's prune may remove, so an
// unknown one is refused before anything is archived.
func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
var stderr bytes.Buffer
if code := cmdBackup([]string{"--server", "survival", "--reason", "inactive_15d"}, &bytes.Buffer{}, &stderr); code != 2 {
t.Fatalf("exit = %d, want 2 (%s)", code, stderr.String())
}
if !strings.Contains(stderr.String(), "unknown --reason") {
t.Fatalf("stderr = %q", stderr.String())
}
}