feat(backup-now): 一次备份全部世界的命令,补扩盘/数据盘与计划内迁机指南

This commit is contained in:
Lemon-miaow committed 2026-09-27 12:08:03 +08:00
1 parent 64476b0171
commit 28fd20c791
6 files changed
+1208 -2

No files matched your search

+453 -2
View File
@@ -4,15 +4,28 @@ import (
"bytes"
"context"
"encoding/json"
"errors"
"flag"
"fmt"
"io"
"net/http"
"os"
"os/signal"
"sort"
"strings"
"syscall"
"time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/operator"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
@@ -22,7 +35,9 @@ import (
// (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console
// cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth)
// while the API is alive, and the API renders the Job and audits the action. This file
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue.
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. The
// `felis backup-now` command (cmdBackupNow, below) takes the same route for every
// user server in turn.
// backupNowOutcome is the durable result of a backup request, re-printed after the TUI
// alt-screen tears down.
@@ -75,7 +90,7 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o
resp, err := hc.Do(req)
if err != nil {
return backupNowOutcome{}, fmt.Errorf("felis-api unreachable (a backup needs it alive): %w", err)
return backupNowOutcome{}, fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
}
defer resp.Body.Close()
@@ -145,3 +160,439 @@ func backupPickable(servers []haltableServer) []haltableServer {
}
return out
}
// cmdBackupNow is `felis backup-now`: the world of every user server (or of the
// named ones) archived now, one at a time, through the internal backup route the
// console's Sync uses. A world lives only in its volume and the off-site copy holds
// only its archives, so this is the lever in front of a planned move to another
// host, a disk swap or anything else that could lose a volume.
func cmdBackupNow(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("backup-now", flag.ContinueOnError)
fs.SetOutput(stderr)
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
yes := fs.Bool("yes", false, "back up; without it the plan is printed and nothing changes")
stop := fs.Bool("stop", false, "stop the running servers first: their players are disconnected and the servers stay stopped")
fs.Usage = func() {
fmt.Fprintln(stderr, "Usage: felis backup-now [-yes] [-stop] [server ...]")
fmt.Fprintln(stderr)
fmt.Fprintln(stderr, "Archives the world of every user server, or of the named ones, one at a time, and waits for each archive.")
fmt.Fprintln(stderr, "A running server is skipped unless -stop is given. Without -yes it prints what it would do.")
fs.PrintDefaults()
}
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if os.Geteuid() != 0 {
fmt.Fprintln(stderr, "felis backup-now: refused — it reads the cluster's ops token, so it must run as root (try: sudo felis backup-now)")
return 1
}
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
return 1
}
cl, err := buildSystemServerClient()
if err != nil {
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
return 1
}
ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer cancel()
ns := cfg.K8s.Namespace
osUser := accountableOSUser()
jobs := api.NewK8sJobStatus(cl, ns)
var baseURL, token string
var repo ownerStore
var drv *store.PostgresDriver
defer func() {
if drv != nil {
_ = drv.Close()
}
}()
hc := &http.Client{Timeout: 10 * time.Second}
r := backupNowRun{
out: stdout,
errw: stderr,
ns: ns,
list: func(ctx context.Context) ([]backupNowWorld, error) { return listBackupNowWorlds(ctx, cl, ns) },
stopped: func(ctx context.Context, name string) (bool, error) {
var ms v1alpha1.MinecraftServer
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &ms); err != nil {
return false, err
}
return backupNowStopped(ctx, cl, &ms)
},
halt: func(ctx context.Context, name string) error {
// The database is opened only once a server is to be stopped: the halt
// is audited like the console's, and a run with nothing running needs
// no more than the API.
if repo == nil {
d, err := store.Open(ctx, cfg.Database.URL)
if err != nil {
return fmt.Errorf("open the database for the audit log: %w", err)
}
drv, repo = d, api.NewPGRepo(d.DB())
}
out, err := performHalt(ctx, cl, repo, ns, name, osUser)
if err != nil {
return err
}
if out.auditErr != nil {
fmt.Fprintf(stderr, "felis backup-now: the audit row for stopping %s was not written: %v\n", name, out.auditErr)
}
return nil
},
request: func(ctx context.Context, name string) error {
if baseURL == "" {
var err error
if baseURL, token, err = resolveInternalAPI(ctx, cl, platform.DefaultControlNamespace); err != nil {
return fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
}
}
_, err := requestBackup(ctx, hc, baseURL, token, name, osUser)
return err
},
jobs: jobs.LatestJobs,
now: time.Now,
sleep: func(ctx context.Context, d time.Duration) { sleepCtx(ctx, d) },
}
return r.run(ctx, fs.Args(), *yes, *stop)
}
// sleepCtx waits d or until ctx ends.
func sleepCtx(ctx context.Context, d time.Duration) {
t := time.NewTimer(d)
defer t.Stop()
select {
case <-ctx.Done():
case <-t.C:
}
}
// backupNowWorld is one user server as backup-now sees it.
type backupNowWorld struct {
name string
phase string // the observed phase, or the desired state before the operator reconciled it
stopped bool // the backup route's stopped gate admits it
hasWorld bool // its world volume exists
}
// listBackupNowWorlds lists the user servers of namespace in the API server's
// order (by name), each with what the backup route checks. System servers are
// left out: they have no row in the servers table, so the route refuses them
// (backupPickable).
func listBackupNowWorlds(ctx context.Context, cl client.Client, namespace string) ([]backupNowWorld, error) {
var list v1alpha1.MinecraftServerList
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
return nil, err
}
var out []backupNowWorld
for i := range list.Items {
ms := &list.Items[i]
if isSystemServer(ms.Name) {
continue
}
stopped, err := backupNowStopped(ctx, cl, ms)
if err != nil {
return nil, err
}
var pvc corev1.PersistentVolumeClaim
err = cl.Get(ctx, types.NamespacedName{Namespace: namespace, Name: naming.WorldPVCName(ms.Name)}, &pvc)
if err != nil && !apierrors.IsNotFound(err) {
return nil, fmt.Errorf("look up the world volume of %s: %w", ms.Name, err)
}
phase := string(ms.Status.Phase)
if phase == "" {
phase = string(ms.Spec.DesiredState)
}
out = append(out, backupNowWorld{name: ms.Name, phase: phase, stopped: stopped, hasWorld: err == nil})
}
return out, nil
}
// backupNowStopped is the backup route's stopped gate (api.enqueueBackup and
// K8sCluster.AcquireMaintenance together): desired Stopped, not ready, phase
// Stopped and no game pod left.
func backupNowStopped(ctx context.Context, cl client.Client, ms *v1alpha1.MinecraftServer) (bool, error) {
if ms.Spec.DesiredState != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
return false, nil
}
var pods corev1.PodList
if err := cl.List(ctx, &pods, client.InNamespace(ms.Namespace), client.MatchingLabels{
v1alpha1.LabelServer: ms.Name, v1alpha1.LabelComponent: operator.ComponentValue,
}); err != nil {
return false, fmt.Errorf("look up the pod of %s: %w", ms.Name, err)
}
return len(pods.Items) == 0, nil
}
// errBackupAPIUnreachable is a backup request that never reached felis-api. It
// ends a backup-now run: every later world would fail the same way, and stopping
// servers for backups that cannot be taken only takes them away from players.
var errBackupAPIUnreachable = errors.New("felis-api unreachable (a backup needs it alive)")
// Polling of backup-now. A graceful stop saves the world first; the backup Job's
// own deadline (backupjob, 30 minutes) ends a Job that hangs, so its wait needs no
// cap of its own.
const (
backupNowPoll = 2 * time.Second
backupNowStopWait = 10 * time.Minute
backupNowJobAppear = time.Minute
)
// backupNowRun is backup-now over seams, so the plan and the run are tested
// without a cluster or felis-api.
type backupNowRun struct {
out, errw io.Writer
ns string // where the servers and their Jobs live, for the kubectl hints
list func(ctx context.Context) ([]backupNowWorld, error)
stopped func(ctx context.Context, name string) (bool, error)
halt func(ctx context.Context, name string) error
request func(ctx context.Context, name string) error
jobs func(ctx context.Context, name string) ([]api.AsyncJob, error)
now func() time.Time
sleep func(ctx context.Context, d time.Duration)
}
// pickBackupNowWorlds narrows worlds to names, in the order given, or keeps them
// all when names is empty.
func pickBackupNowWorlds(worlds []backupNowWorld, names []string) ([]backupNowWorld, error) {
if len(names) == 0 {
return worlds, nil
}
byName := make(map[string]backupNowWorld, len(worlds))
for _, w := range worlds {
byName[w.name] = w
}
var out []backupNowWorld
seen := map[string]bool{}
for _, n := range names {
if seen[n] {
continue
}
seen[n] = true
w, ok := byName[n]
switch {
case ok:
out = append(out, w)
case isSystemServer(n):
return nil, fmt.Errorf("%s is a system server: its world is rebuilt by felis setup and has no backups", n)
default:
return nil, fmt.Errorf("no server named %q", n)
}
}
return out, nil
}
// backupNowAction is what the plan does with a world.
func backupNowAction(w backupNowWorld, stop bool) string {
switch {
case w.stopped && !w.hasWorld:
return "skip: no world volume (never started, nothing to save)"
case w.stopped:
return "back up"
case stop:
return "stop, then back up"
default:
return "skip: running (stop it first, or pass -stop)"
}
}
func (r *backupNowRun) run(ctx context.Context, names []string, yes, stop bool) int {
all, err := r.list(ctx)
if err != nil {
fmt.Fprintf(r.errw, "felis backup-now: list the servers: %v\n", err)
return 1
}
worlds, err := pickBackupNowWorlds(all, names)
if err != nil {
fmt.Fprintf(r.errw, "felis backup-now: %v\n", err)
return 2
}
if len(worlds) == 0 {
fmt.Fprintln(r.out, "felis backup-now: there are no user servers")
return 0
}
// The stopped worlds go first, so a felis-api that cannot take a backup is
// found before any server is stopped for one.
sort.SliceStable(worlds, func(i, j int) bool { return worlds[i].stopped && !worlds[j].stopped })
width := 0
for _, w := range worlds {
width = max(width, len(w.name))
}
fmt.Fprintf(r.out, "felis backup-now: %d server(s), backed up one at a time:\n", len(worlds))
stopping, work := false, 0
for _, w := range worlds {
fmt.Fprintf(r.out, " %-*s %-8s %s\n", width, w.name, w.phase, backupNowAction(w, stop))
stopping = stopping || (!w.stopped && stop)
if (w.stopped && w.hasWorld) || (!w.stopped && stop) {
work++
}
}
if work > 0 {
fmt.Fprintln(r.out, "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.")
}
if stopping {
fmt.Fprintln(r.out, "Stopping disconnects the players on those servers, and they stay stopped afterwards.")
}
if !yes {
if work == 0 {
fmt.Fprintln(r.out, "Nothing to back up.")
} else {
fmt.Fprintln(r.out, "Nothing changed. Run again with -yes to back them up.")
}
return 0
}
var done, failed, running, empty int
var leftStopped []string
for _, w := range worlds {
if err := ctx.Err(); err != nil {
break
}
switch {
case w.stopped && !w.hasWorld:
empty++
continue
case !w.stopped && !stop:
running++
continue
}
if !w.stopped {
halted, err := r.stopWorld(ctx, w.name)
if halted {
leftStopped = append(leftStopped, w.name)
}
if err != nil {
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
failed++
continue
}
}
err := r.backUp(ctx, w.name)
if errors.Is(err, errBackupAPIUnreachable) {
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
fmt.Fprintln(r.out, "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).")
failed++
break
}
if err != nil {
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
failed++
continue
}
done++
}
interrupted := ctx.Err() != nil
if interrupted {
fmt.Fprintln(r.out, "Interrupted: a backup Job already started runs to its end.")
}
parts := []string{fmt.Sprintf("%d backed up", done)}
if failed > 0 {
parts = append(parts, fmt.Sprintf("%d failed", failed))
}
if running > 0 {
parts = append(parts, fmt.Sprintf("%d skipped (running)", running))
}
if empty > 0 {
parts = append(parts, fmt.Sprintf("%d without a world", empty))
}
fmt.Fprintf(r.out, "%s.\n", strings.Join(parts, ", "))
if len(leftStopped) > 0 {
fmt.Fprintf(r.out, "Left stopped: %s. Start them from the panel when you are done.\n", strings.Join(leftStopped, ", "))
}
if done > 0 {
fmt.Fprintln(r.out, "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.")
}
if failed > 0 || running > 0 || interrupted {
return 1
}
return 0
}
// stopWorld stops one server and waits until the backup route would admit it.
// halted is whether the stop was asked for: the server then stays stopped, even
// when it takes longer than the wait.
func (r *backupNowRun) stopWorld(ctx context.Context, name string) (halted bool, err error) {
fmt.Fprintf(r.out, " %s: stopping\n", name)
if err := r.halt(ctx, name); err != nil {
return false, err
}
start := r.now()
for {
ok, err := r.stopped(ctx, name)
if err != nil {
return true, err
}
if ok {
fmt.Fprintf(r.out, " %s: stopped after %s\n", name, r.now().Sub(start).Round(time.Second))
return true, nil
}
if r.now().Sub(start) >= backupNowStopWait {
return true, fmt.Errorf("did not stop within %s (kubectl -n %s describe minecraftserver %s)", backupNowStopWait, r.ns, name)
}
if err := ctx.Err(); err != nil {
return true, err
}
r.sleep(ctx, backupNowPoll)
}
}
// backUp requests one world's backup and waits for its Job to finish. The Job is
// the one of this server that was not there before the request.
func (r *backupNowRun) backUp(ctx context.Context, name string) error {
before, err := r.jobs(ctx, name)
if err != nil {
return fmt.Errorf("list its backup Jobs: %w", err)
}
known := make(map[string]bool, len(before))
for _, j := range before {
known[j.Name] = true
}
if err := r.request(ctx, name); err != nil {
return err
}
fmt.Fprintf(r.out, " %s: backing up\n", name)
start := r.now()
seen := ""
for {
jobs, err := r.jobs(ctx, name)
if err != nil {
return fmt.Errorf("list its backup Jobs: %w", err)
}
var job *api.AsyncJob
for i := range jobs {
if jobs[i].Kind == "backup" && (jobs[i].Name == seen || seen == "" && !known[jobs[i].Name]) {
job = &jobs[i]
break
}
}
switch {
case job == nil && seen != "":
return fmt.Errorf("its backup Job %s was deleted before it finished", seen)
case job == nil && r.now().Sub(start) >= backupNowJobAppear:
return fmt.Errorf("felis-api accepted the backup, but no backup Job appeared within %s (kubectl -n %s get jobs)", backupNowJobAppear, r.ns)
case job != nil && job.State == "succeeded":
fmt.Fprintf(r.out, " %s: archived in %s\n", name, r.now().Sub(start).Round(time.Second))
return nil
case job != nil && job.State == "failed":
msg := job.Message
if msg == "" {
msg = "the backup Job failed"
}
return fmt.Errorf("%s (kubectl -n %s logs job/%s)", msg, r.ns, job.Name)
case job != nil:
seen = job.Name
}
if err := ctx.Err(); err != nil {
return err
}
r.sleep(ctx, backupNowPoll)
}
}
+563
View File
@@ -0,0 +1,563 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"reflect"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/naming"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// bnRig drives backupNowRun against scripted servers and Jobs on a fake clock that
// moves only when the run sleeps.
type bnRig struct {
out, errw bytes.Buffer
clock time.Time
events []string
worlds []backupNowWorld
// stopAfter is how many polls a halted server takes to stop; -1 never does.
stopAfter map[string]int
polls map[string]int
reqErr map[string]error
// states is what the server's new backup Job reports on each poll after the
// request, the last one repeating; "" is no Job.
states map[string][]string
messages map[string]string
jobPolls map[string]int
// stopErr fails a server's stop polls; jobsFail fails the nth (1-based) Job
// list of a server.
stopErr map[string]error
jobsFail map[string]int
jobCalls map[string]int
// ctx is the run's context, and onAct runs after each halt and request.
ctx context.Context
onAct func(event string)
}
func newBNRig(worlds ...backupNowWorld) *bnRig {
return &bnRig{
clock: time.Unix(1_800_000_000, 0),
worlds: worlds,
stopAfter: map[string]int{},
polls: map[string]int{},
reqErr: map[string]error{},
states: map[string][]string{},
messages: map[string]string{},
jobPolls: map[string]int{},
stopErr: map[string]error{},
jobsFail: map[string]int{},
jobCalls: map[string]int{},
ctx: context.Background(),
onAct: func(string) {},
}
}
func (g *bnRig) run(names []string, yes, stop bool) int {
r := backupNowRun{
out: &g.out, errw: &g.errw, ns: "minecraft",
list: func(context.Context) ([]backupNowWorld, error) {
return append([]backupNowWorld(nil), g.worlds...), nil
},
stopped: func(_ context.Context, name string) (bool, error) {
g.polls[name]++
if err := g.stopErr[name]; err != nil {
return false, err
}
n := g.stopAfter[name]
return n >= 0 && g.polls[name] > n, nil
},
halt: func(_ context.Context, name string) error {
g.events = append(g.events, "halt "+name)
g.onAct("halt " + name)
return nil
},
request: func(_ context.Context, name string) error {
g.events = append(g.events, "request "+name)
g.onAct("request " + name)
if err := g.reqErr[name]; err != nil {
return err
}
g.jobPolls[name] = 0
return nil
},
jobs: func(_ context.Context, name string) ([]api.AsyncJob, error) {
// Every server has an older finished backup, and a restore Job that
// shows up with the new backup: neither is the Job to wait for.
g.jobCalls[name]++
if g.jobCalls[name] == g.jobsFail[name] {
return nil, errors.New("the apiserver is gone")
}
out := []api.AsyncJob{{Name: "backup-" + name + "-old", Kind: "backup", State: "succeeded"}}
n, requested := g.jobPolls[name]
if !requested {
return out, nil
}
g.jobPolls[name] = n + 1
states := g.states[name]
if len(states) == 0 {
states = []string{"succeeded"}
}
state := states[min(n, len(states)-1)]
if state == "" {
return out, nil
}
return append([]api.AsyncJob{
{Name: "restore-" + name + "-x", Kind: "restore", State: "succeeded"},
{Name: "backup-" + name + "-new", Kind: "backup", State: state, Message: g.messages[name]},
}, out...), nil
},
now: func() time.Time { return g.clock },
sleep: func(_ context.Context, d time.Duration) { g.clock = g.clock.Add(d) },
}
return r.run(g.ctx, names, yes, stop)
}
func stoppedWorld(name string) backupNowWorld {
return backupNowWorld{name: name, phase: "Stopped", stopped: true, hasWorld: true}
}
func runningWorld(name string) backupNowWorld {
return backupNowWorld{name: name, phase: "Running", hasWorld: true}
}
func emptyWorld(name string) backupNowWorld {
return backupNowWorld{name: name, phase: "Stopped", stopped: true}
}
const (
bnManualKeep = "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.\n"
bnStopping = "Stopping disconnects the players on those servers, and they stay stopped afterwards.\n"
bnOffsite = "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.\n"
)
func TestBackupNowPlanChangesNothing(t *testing.T) {
for _, tc := range []struct {
stop bool
want string
}{
{false, "felis backup-now: 4 server(s), backed up one at a time:\n" +
" alpha Stopped back up\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running skip: running (stop it first, or pass -stop)\n" +
" delta Starting skip: running (stop it first, or pass -stop)\n" +
bnManualKeep +
"Nothing changed. Run again with -yes to back them up.\n"},
{true, "felis backup-now: 4 server(s), backed up one at a time:\n" +
" alpha Stopped back up\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running stop, then back up\n" +
" delta Starting stop, then back up\n" +
bnManualKeep + bnStopping +
"Nothing changed. Run again with -yes to back them up.\n"},
} {
t.Run(fmt.Sprintf("stop=%v", tc.stop), func(t *testing.T) {
delta := runningWorld("delta")
delta.phase = "Starting"
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), delta)
if code := g.run(nil, false, tc.stop); code != 0 {
t.Fatalf("exit = %d, want 0; stderr %q", code, g.errw.String())
}
if g.out.String() != tc.want {
t.Fatalf("plan =\n%s\nwant\n%s", g.out.String(), tc.want)
}
if len(g.events) != 0 || len(g.polls) != 0 {
t.Fatalf("the plan acted: events %v, polls %v", g.events, g.polls)
}
})
}
}
// A plan that saves nothing says so, without the manual_keep warning; a running
// server counts as something to save once -stop is given.
func TestBackupNowPlanWithNothingToSave(t *testing.T) {
for _, tc := range []struct {
stop bool
want string
}{
{false, "felis backup-now: 2 server(s), backed up one at a time:\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running skip: running (stop it first, or pass -stop)\n" +
"Nothing to back up.\n"},
{true, "felis backup-now: 2 server(s), backed up one at a time:\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running stop, then back up\n" +
bnManualKeep + bnStopping +
"Nothing changed. Run again with -yes to back them up.\n"},
} {
g := newBNRig(runningWorld("bravo"), emptyWorld("charlie"))
if code := g.run(nil, false, tc.stop); code != 0 {
t.Fatalf("stop=%v: exit = %d, want 0", tc.stop, code)
}
if g.out.String() != tc.want {
t.Fatalf("stop=%v: plan =\n%s\nwant\n%s", tc.stop, g.out.String(), tc.want)
}
}
}
func TestBackupNowBacksUpEachWorldInTurn(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), stoppedWorld("delta"))
g.states["alpha"] = []string{"", "running", "succeeded"}
g.states["delta"] = []string{"running", "failed"}
g.messages["delta"] = "felis backup: not enough free disk for the archive"
g.stopAfter["bravo"] = 2
g.states["bravo"] = []string{"running", "running", "running", "succeeded"}
code := g.run(nil, true, true)
if code != 1 {
t.Fatalf("exit = %d, want 1 (delta failed)", code)
}
if want := []string{"request alpha", "request delta", "halt bravo", "request bravo"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
_, run, _ := strings.Cut(g.out.String(), bnStopping+"")
want := " alpha: backing up\n" +
" alpha: archived in 4s\n" +
" delta: backing up\n" +
" delta: failed: felis backup: not enough free disk for the archive (kubectl -n minecraft logs job/backup-delta-new)\n" +
" bravo: stopping\n" +
" bravo: stopped after 4s\n" +
" bravo: backing up\n" +
" bravo: archived in 6s\n" +
"2 backed up, 1 failed, 1 without a world.\n" +
"Left stopped: bravo. Start them from the panel when you are done.\n" +
bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
}
func TestBackupNowSkipsRunningServersWithoutStop(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"))
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1 (bravo was not backed up)", code)
}
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
if len(g.polls) != 0 {
t.Fatalf("polled a server it did not stop: %v", g.polls)
}
_, run, _ := strings.Cut(g.out.String(), "Nothing changed")
if run != "" {
t.Fatalf("-yes printed the plan's closing line")
}
_, run, _ = strings.Cut(g.out.String(), bnManualKeep)
want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 skipped (running).\n" + bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
}
func TestBackupNowExitsCleanWhenEverythingIsSaved(t *testing.T) {
// -stop with nothing running stops nothing and says nothing about stopping.
g := newBNRig(stoppedWorld("alpha"), emptyWorld("charlie"))
if code := g.run(nil, true, true); code != 0 {
t.Fatalf("exit = %d, want 0; output\n%s", code, g.out.String())
}
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
if want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 without a world.\n" + bnOffsite; run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
g = newBNRig(emptyWorld("charlie"))
if code := g.run(nil, true, false); code != 0 {
t.Fatalf("exit = %d, want 0 for a server with nothing to save", code)
}
// Nothing to archive: no manual_keep warning, and no off-site hint.
if want := "felis backup-now: 1 server(s), backed up one at a time:\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
"0 backed up, 1 without a world.\n"; g.out.String() != want {
t.Fatalf("output =\n%s\nwant\n%s", g.out.String(), want)
}
if len(g.events) != 0 {
t.Fatalf("events = %v, want none", g.events)
}
}
func TestBackupNowStopsAtAnUnreachableAPI(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), runningWorld("charlie"))
g.reqErr["alpha"] = fmt.Errorf("%w: dial tcp 10.43.0.9:8081: connect: connection refused", errBackupAPIUnreachable)
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v: nothing after the API proved unreachable, and no server stopped", g.events, want)
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
want := " alpha: failed: felis-api unreachable (a backup needs it alive): dial tcp 10.43.0.9:8081: connect: connection refused\n" +
"Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).\n" +
"0 backed up, 1 failed.\n"
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
// Any other refusal is that world's alone: the run goes on.
g = newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
g.reqErr["alpha"] = errors.New("felis-api: the world is being restored")
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"request alpha", "request bravo"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
}
func TestBackupNowGivesUpOnAServerThatDoesNotStop(t *testing.T) {
g := newBNRig(runningWorld("bravo"), runningWorld("echo"))
g.stopAfter["bravo"] = -1
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"halt bravo", "halt echo", "request echo"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
// One poll at the start and one per 2s sleep up to the 10-minute mark.
if g.polls["bravo"] != 301 {
t.Fatalf("bravo polled %d times, want 301", g.polls["bravo"])
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
want := " bravo: stopping\n" +
" bravo: failed: did not stop within 10m0s (kubectl -n minecraft describe minecraftserver bravo)\n" +
" echo: stopping\n" +
" echo: stopped after 0s\n" +
" echo: backing up\n" +
" echo: archived in 0s\n" +
"1 backed up, 1 failed.\n" +
"Left stopped: bravo, echo. Start them from the panel when you are done.\n" +
bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
}
func TestBackupNowReportsAJobThatNeverRuns(t *testing.T) {
for _, tc := range []struct {
name string
states []string
polls int
want string
}{
{"never appears", []string{""}, 31,
"felis-api accepted the backup, but no backup Job appeared within 1m0s (kubectl -n minecraft get jobs)"},
{"deleted while running", []string{"running", "running", ""}, 3,
"its backup Job backup-alpha-new was deleted before it finished"},
{"fails without a message", []string{"failed"}, 1,
"the backup Job failed (kubectl -n minecraft logs job/backup-alpha-new)"},
} {
t.Run(tc.name, func(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"))
g.states["alpha"] = tc.states
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if !strings.Contains(g.out.String(), " alpha: failed: "+tc.want+"\n") {
t.Fatalf("output =\n%s\nwant the line %q", g.out.String(), tc.want)
}
if g.jobPolls["alpha"] != tc.polls {
t.Fatalf("polled the Jobs %d times after the request, want %d", g.jobPolls["alpha"], tc.polls)
}
})
}
}
func TestBackupNowNamedServers(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), stoppedWorld("delta"))
if code := g.run([]string{"delta", "alpha", "delta"}, true, false); code != 0 {
t.Fatalf("exit = %d, want 0", code)
}
if want := []string{"request delta", "request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
if !strings.HasPrefix(g.out.String(), "felis backup-now: 2 server(s), backed up one at a time:\n delta Stopped back up\n alpha Stopped back up\n") {
t.Fatalf("plan =\n%s", g.out.String())
}
for _, tc := range []struct{ name, want string }{
{"login", "felis backup-now: login is a system server: its world is rebuilt by felis setup and has no backups\n"},
{"lobby", "felis backup-now: lobby is a system server: its world is rebuilt by felis setup and has no backups\n"},
{"nope", "felis backup-now: no server named \"nope\"\n"},
} {
g := newBNRig(stoppedWorld("alpha"))
if code := g.run([]string{"alpha", tc.name}, true, false); code != 2 {
t.Fatalf("%s: exit = %d, want 2", tc.name, code)
}
if g.errw.String() != tc.want {
t.Fatalf("%s: stderr = %q, want %q", tc.name, g.errw.String(), tc.want)
}
if len(g.events) != 0 || g.out.Len() != 0 {
t.Fatalf("%s: acted on a bad name: events %v, output %q", tc.name, g.events, g.out.String())
}
}
g = newBNRig()
if code := g.run(nil, true, false); code != 0 || g.out.String() != "felis backup-now: there are no user servers\n" {
t.Fatalf("empty fleet: exit %d, output %q", code, g.out.String())
}
}
// The world list mirrors the backup route's own gate, so the plan says exactly what
// the route would refuse.
func TestListBackupNowWorlds(t *testing.T) {
withStatus := func(ms *v1alpha1.MinecraftServer, ready bool) *v1alpha1.MinecraftServer {
ms.Status.Ready = ready
return ms
}
pvc := func(server, ns string) *corev1.PersistentVolumeClaim {
return &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: naming.WorldPVCName(server), Namespace: ns}}
}
pod := func(name, ns string, labels map[string]string) *corev1.Pod {
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: labels}}
}
gameLabels := func(server string) map[string]string {
return map[string]string{v1alpha1.LabelServer: server, v1alpha1.LabelComponent: "server"}
}
other := mcServer("zulu", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
other.Namespace = "elsewhere"
hotel := mcServer("hotel", v1alpha1.DesiredStopped, "")
objs := []client.Object{
mcServer("alpha", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("alpha", haltNS),
// A backup Job's pod carries the server label with its own component, and
// a game pod of the same name in another namespace is somebody else's.
pod("backup-alpha-1-x", haltNS, map[string]string{v1alpha1.LabelServer: "alpha", "app.kubernetes.io/component": "world-backup"}),
pod("alpha-0", "elsewhere", gameLabels("alpha")),
withStatus(mcServer("bravo", v1alpha1.DesiredRunning, v1alpha1.PhaseRunning), true), pvc("bravo", haltNS), pod("bravo-0", haltNS, gameLabels("bravo")),
mcServer("charlie", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped),
mcServer("delta", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("delta", haltNS), pod("delta-0", haltNS, gameLabels("delta")),
mcServer("echo", v1alpha1.DesiredStopped, v1alpha1.PhaseStopping), pvc("echo", haltNS),
mcServer("foxtrot", v1alpha1.DesiredRunning, v1alpha1.PhaseStopped), pvc("foxtrot", haltNS),
withStatus(mcServer("golf", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), true), pvc("golf", haltNS),
hotel, pvc("hotel", haltNS),
mcServer("login", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("login", haltNS),
mcServer("lobby", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("lobby", haltNS),
other, pvc("zulu", "elsewhere"),
}
got, err := listBackupNowWorlds(context.Background(), haltClient(t, objs...), haltNS)
if err != nil {
t.Fatal(err)
}
want := []backupNowWorld{
{name: "alpha", phase: "Stopped", stopped: true, hasWorld: true},
{name: "bravo", phase: "Running", hasWorld: true},
{name: "charlie", phase: "Stopped", stopped: true},
{name: "delta", phase: "Stopped", hasWorld: true}, // its pod is still going
{name: "echo", phase: "Stopping", hasWorld: true}, // still stopping
{name: "foxtrot", phase: "Stopped", hasWorld: true}, // asked to start
{name: "golf", phase: "Stopped", hasWorld: true}, // still reports ready
{name: "hotel", phase: "Stopped", hasWorld: true}, // not reconciled yet: the desired state shows
}
if !reflect.DeepEqual(got, want) {
t.Fatalf("worlds =\n%+v\nwant\n%+v", got, want)
}
}
func TestBackupNowStopsWhenInterrupted(t *testing.T) {
t.Run("between worlds", func(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
g.ctx = ctx
g.onAct = func(e string) {
if e == "request alpha" {
cancel()
}
}
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
want := " alpha: backing up\n alpha: archived in 0s\n" +
"Interrupted: a backup Job already started runs to its end.\n" +
"1 backed up.\n" + bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
})
t.Run("while a Job runs", func(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"))
g.states["alpha"] = []string{"running"}
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
g.ctx = ctx
g.onAct = func(string) { cancel() }
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if g.jobPolls["alpha"] != 1 {
t.Fatalf("polled the Job %d times after the interrupt, want 1", g.jobPolls["alpha"])
}
if !strings.Contains(g.out.String(), " alpha: failed: context canceled\nInterrupted: a backup Job already started runs to its end.\n0 backed up, 1 failed.\n") {
t.Fatalf("output =\n%s", g.out.String())
}
})
t.Run("while a server stops", func(t *testing.T) {
g := newBNRig(runningWorld("bravo"))
g.stopAfter["bravo"] = -1
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
g.ctx = ctx
g.onAct = func(string) { cancel() }
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if g.polls["bravo"] != 1 {
t.Fatalf("polled bravo %d times after the interrupt, want 1", g.polls["bravo"])
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
want := " bravo: stopping\n bravo: failed: context canceled\n" +
"Interrupted: a backup Job already started runs to its end.\n" +
"0 backed up, 1 failed.\n" +
"Left stopped: bravo. Start them from the panel when you are done.\n"
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
})
}
func TestBackupNowReportsClusterErrors(t *testing.T) {
g := newBNRig(runningWorld("bravo"))
g.stopErr["bravo"] = errors.New("the apiserver is gone")
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
// The stop was asked for, so the server stays stopped whatever the poll said.
want := " bravo: stopping\n bravo: failed: the apiserver is gone\n0 backed up, 1 failed.\n" +
"Left stopped: bravo. Start them from the panel when you are done.\n"
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
for _, tc := range []struct {
call int
events []string
}{
{1, nil}, // before the request: nothing is asked for
{2, []string{"request alpha"}}, // the first poll after it
} {
g := newBNRig(stoppedWorld("alpha"))
g.jobsFail["alpha"] = tc.call
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("call %d: exit = %d, want 1", tc.call, code)
}
if !reflect.DeepEqual(g.events, tc.events) {
t.Fatalf("call %d: events = %v, want %v", tc.call, g.events, tc.events)
}
if !strings.Contains(g.out.String(), " alpha: failed: list its backup Jobs: the apiserver is gone\n") {
t.Fatalf("call %d: output =\n%s", tc.call, g.out.String())
}
}
}
+5
View File
@@ -3,6 +3,7 @@ package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"io"
"net/http"
@@ -152,6 +153,10 @@ func TestRequestBackup(t *testing.T) {
if err == nil || !strings.Contains(err.Error(), "unreachable") {
t.Fatalf("err = %v, want an 'unreachable' transport error", err)
}
// backup-now ends its run on this error alone.
if !errors.Is(err, errBackupAPIUnreachable) {
t.Fatalf("err = %v, want it to wrap errBackupAPIUnreachable", err)
}
})
}
+2
View File
@@ -20,6 +20,7 @@ Commands:
reaper Run the world reaper / backup batch
restore Extract a world archive into a world volume (internal Job entrypoint)
backup Archive a world into the backup store and record it (internal Job entrypoint)
backup-now Archive every user server's world now, one at a time (or the named ones; -stop stops running ones first; prints the plan, -yes applies; requires root/sudo)
files List/read/write one file in a stopped server's world (internal Job entrypoint)
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
@@ -60,6 +61,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
"reaper": cmdReaper,
"restore": cmdRestore,
"backup": cmdBackup,
"backup-now": cmdBackupNow,
"files": cmdFiles,
"egress-gate": cmdEgressGate,
"fetch-context": cmdFetchContext,
+135
View File
@@ -262,6 +262,95 @@ its threshold, and §13b covers a full disk. On a host that built its images,
`docker builder prune -af` (with Docker started) reclaims the build cache when space is
short; the next upgrade rebuilds it.
### Growing the disk
Everything above shares the root filesystem, so more room means a bigger root
filesystem. It grows in place, with everything running: enlarge the virtual disk at the
provider, then the partition and the filesystem on it.
```bash
sudo felis backup-now -yes # a mistyped partition number is how a resize loses a disk
lsblk -f # which disk and partition hold /, and whether LVM sits on it
sudo growpart /dev/vda 3 # cloud-utils-growpart (RHEL) / cloud-guest-utils (Debian, Ubuntu)
# LVM (the RHEL-family default):
sudo pvresize /dev/vda3
sudo lvextend -r -l +100%FREE /dev/<vg>/root # -r grows the filesystem with it
# no LVM:
sudo xfs_growfs / # xfs
sudo resize2fs /dev/vda3 # ext4
df -h /
```
`felis backup-now` (troubleshooting.md §10) archives every stopped world; add `-stop` to
include the running ones.
### Moving the data to its own disk [VM-VERIFIED]
The bulk lives under `/var/lib/rancher/k3s`: the worlds, the world archives, the registry
and the images. On a disk of its own it grows without touching the system, and a full
world store leaves the root filesystem alone. The database and its bundles
(`/var/lib/felis`) are small and stay on the root disk. The move takes the platform down
for the copy plus a minute or two: the drill copied 4.2 GB in 18 s, and felis-api answered
`/readyz` 14 s after k3s started on the new disk.
1. Attach the disk and put a filesystem on it (the whole disk; `lsblk` shows it empty):
```bash
sudo mkfs.xfs /dev/vdb
U=$(sudo blkid -s UUID -o value /dev/vdb)
```
2. Archive every world, stopping the servers so each one saves, and keep the watchdog
quiet for the next hour (the marker the installer writes: no mail, no failure pings
to the heartbeat, until the time in it):
```bash
sudo felis backup-now -yes -stop
sudo install -d -m 0755 /run/felis
echo $(( $(date +%s) + 3600 )) | sudo tee /run/felis/watchdog-quiet-until
```
3. Stop k3s and copy:
```bash
sudo systemctl stop k3s
sudo /usr/local/bin/k3s-killall.sh # the containers k3s leaves running, and their mounts
sudo mkdir -p /mnt/felis-data
sudo mount UUID=$U /mnt/felis-data
sudo rsync -aHAX --numeric-ids /var/lib/rancher/k3s/ /mnt/felis-data/
sudo umount /mnt/felis-data
```
`-X` carries the SELinux labels k3s set itself. Leave `restorecon` out: it would reset
runc and the CNI binaries from `container_runtime_exec_t` to the policy default.
4. Mount it in place, and tie k3s to the mount:
```bash
sudo mv /var/lib/rancher/k3s /var/lib/rancher/k3s.old
sudo mkdir /var/lib/rancher/k3s
echo "UUID=$U /var/lib/rancher/k3s xfs defaults,nofail 0 0" | sudo tee -a /etc/fstab
sudo mkdir -p /etc/systemd/system/k3s.service.d
printf '[Unit]\nRequiresMountsFor=/var/lib/rancher/k3s\n' | sudo tee /etc/systemd/system/k3s.service.d/data-disk.conf
sudo systemctl daemon-reload
sudo mount /var/lib/rancher/k3s
sudo systemctl start k3s
```
The drop-in is what keeps the data safe: k3s started on the empty mount point creates
a new, empty cluster there. With it, a disk that does not come up fails the start with
`A dependency job for k3s.service failed`, and `nofail` keeps the host booting so you
can reach it. In the drill a detached disk left k3s inactive and the mount point empty;
reattached, `systemctl start k3s` mounted it and started.
5. Check that `sudo k3s kubectl -n felis get pods` shows every pod ready and
`findmnt /var/lib/rancher/k3s` names the new disk, then start the servers from the
panel and `sudo rm /run/felis/watchdog-quiet-until`. Once the host has run a day,
`sudo rm -rf /var/lib/rancher/k3s.old` frees the root disk.
The watchdog already watches `/var/lib/rancher/k3s` as a filesystem of its own (its
`-disk-paths`), so the new disk's fill level is mailed like the root's.
## 3. Uninstall
`deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove
@@ -604,6 +693,52 @@ production install:
non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
or rely on the watchdog's mail.
### Moving to another host (planned)
A planned move is the rebuild of troubleshooting.md §16, with the old host still there to
hand over a copy that misses nothing. It needs the off-site bucket: that is how the world
archives reach the new host (§16 step 7). The platform is down from step 1 until the new
host serves.
1. **On the old host**, stop everything that changes a world, then send the last copy:
```bash
sudo install -d -m 0755 /run/felis
echo $(( $(date +%s) + 4 * 3600 )) | sudo tee /run/felis/watchdog-quiet-until
sudo systemctl stop felis-velocity # no joins, so no server wakes
sudo felis backup-now -yes -stop # every world archived; the servers stay stopped
sudo k3s kubectl -n felis scale deploy/felis-operator --replicas=0 # nothing starts a server from here on
sudo felis db backup # a bundle that lists those archives
sudo systemctl start felis-offsite.service
sudo felis offsite status # again until nothing waits
```
The order matters. The new host fetches the archives its restored database lists, so
the bundle comes after the last archive. The operator goes after `backup-now`, which
needs it to stop the servers. The quiet marker keeps the watchdog from mailing the
owners about the stopped proxy and operator for the next 4 hours.
2. **On the new host**, follow troubleshooting.md §16 "Rebuild on a new host" from step 1;
`fetch-db latest` picks the bundle the old host just sent. Step 8 (`felis offsite
take-over`) makes the new host the one that writes the bucket, and from then on the
old host copies nothing more. Steps 10 and 11 move the names and the tunnel.
3. **Check the new host** before announcing it: sign in with an email code, restore one
world and join it, and see `sudo felis offsite status` show a recent `last success` and
no stand-by notice.
4. **Retire the old host.** It holds the last copy of every world outside the bucket, so
keep it powered off with its disk for a few days first, disabled so a boot brings
nothing up:
```bash
sudo systemctl disable k3s felis-velocity felis-watchdog.timer felis-offsite.timer \
felis-db-backup.timer felis-update-check.timer felis-build-tools.timer
sudo poweroff
```
Then uninstall it (§3) or wipe it.
Each step is covered where it is documented (backup-now in troubleshooting.md §10, the
rebuild in §16); the sequence as a whole has not been rehearsed as one move.
## 6. Changing the root domain [VM-VERIFIED] [GO-TESTED] [SH-TESTED]
The root domain is written into more places than the installer's config: the panel
+50
View File
@@ -1080,6 +1080,56 @@ the error its container exited on under Recent operations on the server's
backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`,
`TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`]
### Every world at once: `felis backup-now`
A world lives only in its volume, and the off-site copy holds only its archives.
Before anything that could lose a volume (growing the disk, moving the data to its
own disk, moving to another host: operations.md §2 and §5), archive every world
from the node:
```bash
sudo felis backup-now # the plan; nothing changes
sudo felis backup-now -yes # archive every stopped world, one at a time
sudo felis backup-now -yes -stop # stop the running servers first
sudo felis backup-now -yes alpha bravo # only these servers
```
It runs as root because it reads the ops token (`felis/felis-ops-token`, §6) and
asks felis-api's internal face for each backup. Each archive is an ordinary manual
backup (the same Job, the same 10% free-disk check, the audit action
`backup.create` with the source `internal:ops` and the sudo user as actor), exempt
from the owner cooldown and the `max_local_bytes` cap like the break-glass console.
- **The plan** lists every user server with its phase and what the run does with
it: `back up`, `stop, then back up`, `skip: running (stop it first, or pass
-stop)`, or `skip: no world volume (never started, nothing to save)`. With
nothing to archive it ends `Nothing to back up.`
- **Each archive counts against `manual_keep`**: a server that already holds that
many manual backups loses its oldest, and the plan says so. Raise
`[archive] manual_keep` first when those older backups matter.
- **Stopped servers go first**, so a felis-api that cannot take a backup is found
before anything is stopped for one. The run waits for each Job (`alpha: archived
in 42s`) before starting the next.
- **`-stop` disconnects the players** and leaves those servers stopped (`Left
stopped: …`; start them from the panel). Each stop writes `break_glass.halt` to
the audit log. A server still up after 10 minutes counts as failed, and the run
moves on.
- **An unreachable felis-api ends the run** (`Stopped: nothing more can be backed
up until felis-api answers`). Ctrl-C ends it after the current step; a backup Job
already started runs to its end.
- **The archives stay on the node** until the hourly off-site copy. After a run
that archived something the command prints `sudo systemctl start
felis-offsite.service`, which sends them now; `sudo felis offsite status` shows
what still waits.
It exits 0 when every world with a volume was archived, 1 when a backup failed, a
running server was skipped (no `-stop`) or the run was interrupted, and 2 for a
name that is no user server. [GO-TESTED: `TestBackupNowPlanChangesNothing`,
`TestBackupNowPlanWithNothingToSave`, `TestBackupNowBacksUpEachWorldInTurn`,
`TestBackupNowSkipsRunningServersWithoutStop`,
`TestBackupNowStopsAtAnUnreachableAPI`, `TestBackupNowStopsWhenInterrupted`,
`TestBackupNowNamedServers`]
### Scheduled backups (daily restore points)
A world played every day never idles 15 days, so the reaper never archives it.