Files
Felis/cmd/felis/backup.go

201 lines
8.1 KiB
Go

package main
import (
"context"
"crypto/rand"
"encoding/hex"
"flag"
"fmt"
"io"
"strings"
"time"
"felis.lolicon.best/internal/backup"
"felis.lolicon.best/internal/backupjob"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/reaper"
ctrl "sigs.k8s.io/controller-runtime"
)
// cmdBackup is the in-Pod entrypoint the on-demand backup Job runs. internal/backupjob
// renders a Pod whose command is `/usr/local/bin/felis backup`. It tars the mounted
// world into the archive store AND records the world_backups row, then exits — it is
// NOT a user-facing command and is never invoked by hand.
//
// Unlike `felis restore`, this command DOES hold database credentials (via the mounted
// config Secret) and calls config.Load: a backup must record its row atomically with
// the archive, exactly like the reaper — otherwise a completed archive would leak as an
// orphan file the retention pass never expires. The security review for that departure
// lives in internal/backupjob/jobspec.go. The world is mounted directly at --worlds-root
// (single-PVC mount, like restore), so the archiver's resolver returns that root for any
// PVC; the archive is written into the backup PVC at cfg.Archive.LocalPath.
func cmdBackup(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("backup", flag.ContinueOnError)
fs.SetOutput(stderr)
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
server := fs.String("server", "", "server name whose world is being backed up")
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
reason := fs.String("reason", reasonManual, "world_backups reason: manual, pre_restore for the safety snapshot in front of a restore, or scheduled for felis-api's daily restore point")
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
if err := fs.Parse(args); err != nil {
return 2
}
if *server == "" {
fmt.Fprintln(stderr, "felis backup: --server is required")
return 2
}
if _, _, ok := backupPolicy(*reason, reaper.DefaultConfig()); !ok {
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual, %s or %s)\n", *reason, backupjob.ReasonPreRestore, backupjob.ReasonScheduled)
return 2
}
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
if cfg.Archive.Store != "tarLocal" {
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
return 1
}
// The [archive] parse the reaper uses: it holds each reason's keep and
// retention.
rcfg, err := reaperConfig(cfg)
if err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
keep, retention, _ := backupPolicy(*reason, rcfg)
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
// writes archives with.
archiver := &backup.TarLocal{
BackupRoot: cfg.Archive.LocalPath,
Resolve: func(string) (string, error) {
return *worldsRoot, nil
},
}
ctx := ctrl.SetupSignalHandler()
// The archive store shares the node's disk with every world and the
// database: an owner's backup must not be what tips it into eviction.
if err := backup.CheckRoom(cfg.Archive.LocalPath, *worldsRoot, backup.MinFreeAfter); err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
a, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
if err != nil {
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
return 1
}
ref, size := a.Ref, a.Size
if len(a.Skipped) > 0 {
fmt.Fprintf(stderr, "felis backup: %d entries are not plain files or directories and are not in the archive: %s\n",
len(a.Skipped), strings.Join(a.Skipped[:min(len(a.Skipped), 10)], ", "))
}
drv, err := openPodStore(ctx, cfg.Database.URL, "backup", stderr)
if err != nil {
fmt.Fprintf(stderr, "felis backup: open database: %v\n", err)
return 1
}
defer drv.Close()
rec := reaper.BackupRecord{
ID: newBackupID(),
ServerName: *server,
FormerOwner: *formerOwner,
BackupRef: string(ref),
SizeBytes: size,
Reason: *reason,
ExpiresAt: time.Now().Add(retention),
SHA256: a.SHA256,
SkippedEntries: len(a.Skipped),
}
st := reaper.NewPGStore(drv.DB())
if err := st.InsertBackup(ctx, rec); err != nil {
// The archive is written but unrecorded — an orphan the retention pass would
// never expire. Delete it so a failed backup leaves no leaked bytes, mirroring
// the reaper's archive-then-record atomicity.
if delErr := archiver.Delete(ctx, ref); delErr != nil {
fmt.Fprintf(stderr, "felis backup: record failed (%v) AND orphan archive %s could not be removed: %v\n", err, ref, delErr)
return 1
}
fmt.Fprintf(stderr, "felis backup: record failed, orphan archive removed: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
pruneBackups(ctx, st, archiver, *server, *formerOwner, *reason, keep, *protect, stdout, stderr)
return 0
}
const (
reasonManual = "manual"
// preRestoreKeep is how many safety snapshots a server keeps: enough to walk
// back a couple of restores in a row, without every restore adding a world's
// worth of bytes for the full manual retention.
preRestoreKeep = 3
)
// backupPolicy is how many backups of one reason a server keeps and how long
// each lives: an owner's own backups and the safety snapshots in front of a
// restore by [archive] manual_keep / manual_retention (the snapshots capped at
// preRestoreKeep), felis-api's daily restore points by scheduled_keep /
// scheduled_retention, so neither kind crowds out the other. ok is false for a
// reason this command does not record.
func backupPolicy(reason string, rcfg reaper.Config) (keep int, retention time.Duration, ok bool) {
switch reason {
case reasonManual:
return rcfg.ManualKeep, rcfg.ManualRetention, true
case backupjob.ReasonPreRestore:
return preRestoreKeep, rcfg.ManualRetention, true
case backupjob.ReasonScheduled:
return rcfg.ScheduledKeep, rcfg.ScheduledRetention, true
}
return 0, 0, false
}
// pruneBackups keeps the newest keep backups of this reason that owner holds of
// server and removes the rest, oldest first, so repeated backups of one world
// cannot fill the shared archive store and a new owner's backups never remove a
// previous owner's. protect is never removed: it is the backup a chained restore
// is about to extract. The new backup is already recorded; a removal that fails
// is reported and retried after the next backup.
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, owner, reason string, keep int, protect string, stdout, stderr io.Writer) {
excess, err := st.ExcessBackups(ctx, server, owner, reason, keep, protect)
if err != nil {
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
return
}
for _, b := range excess {
if err := archiver.Delete(ctx, backup.ArchiveRef(b.BackupRef)); err != nil {
fmt.Fprintf(stderr, "felis backup: remove older backup %s: %v\n", b.ID, err)
continue
}
if err := st.MarkBackupDeleted(ctx, b.ID, time.Now()); err != nil {
fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
continue
}
fmt.Fprintf(stdout, "felis backup: removed older %s backup %s of %s (keeping the newest %d)\n", reason, b.ID, server, keep)
}
}
// newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex
// scheme so a manual and an inactivity backup are indistinguishable downstream.
func newBackupID() string {
var b [16]byte
if _, err := rand.Read(b[:]); err != nil {
// crypto/rand failure is fatal and unrecoverable; a time-based fallback would
// be a weaker ID for no benefit. A panic is the honest failure here.
panic("felis backup: crypto/rand: " + err.Error())
}
return "bk-" + hex.EncodeToString(b[:])
}