From 28fd20c791c0472da000ad43516be3f2a7ecdec1 Mon Sep 17 00:00:00 2001 From: Lemon-miaow Date: Sun, 27 Sep 2026 12:08:03 +0800 Subject: [PATCH] =?UTF-8?q?feat(backup-now):=20=E4=B8=80=E6=AC=A1=E5=A4=87?= =?UTF-8?q?=E4=BB=BD=E5=85=A8=E9=83=A8=E4=B8=96=E7=95=8C=E7=9A=84=E5=91=BD?= =?UTF-8?q?=E4=BB=A4=EF=BC=8C=E8=A1=A5=E6=89=A9=E7=9B=98/=E6=95=B0?= =?UTF-8?q?=E6=8D=AE=E7=9B=98=E4=B8=8E=E8=AE=A1=E5=88=92=E5=86=85=E8=BF=81?= =?UTF-8?q?=E6=9C=BA=E6=8C=87=E5=8D=97?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- cmd/felis/backupnow.go | 455 +++++++++++++++++++++++++- cmd/felis/backupnow_cmd_test.go | 563 ++++++++++++++++++++++++++++++++ cmd/felis/backupnow_test.go | 5 + cmd/felis/run.go | 2 + docs/operations.md | 135 ++++++++ docs/troubleshooting.md | 50 +++ 6 files changed, 1208 insertions(+), 2 deletions(-) create mode 100644 cmd/felis/backupnow_cmd_test.go diff --git a/cmd/felis/backupnow.go b/cmd/felis/backupnow.go index 0b3539c..513fe6d 100644 --- a/cmd/felis/backupnow.go +++ b/cmd/felis/backupnow.go @@ -4,15 +4,28 @@ import ( "bytes" "context" "encoding/json" + "errors" + "flag" "fmt" "io" "net/http" + "os" + "os/signal" + "sort" + "strings" + "syscall" "time" + "felis.lolicon.best/internal/api" + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/config" "felis.lolicon.best/internal/naming" + "felis.lolicon.best/internal/operator" "felis.lolicon.best/internal/platform" + "felis.lolicon.best/internal/store" corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/types" "sigs.k8s.io/controller-runtime/pkg/client" ) @@ -22,7 +35,9 @@ import ( // (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console // cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth) // while the API is alive, and the API renders the Job and audits the action. This file -// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. +// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. The +// `felis backup-now` command (cmdBackupNow, below) takes the same route for every +// user server in turn. // backupNowOutcome is the durable result of a backup request, re-printed after the TUI // alt-screen tears down. @@ -75,7 +90,7 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o resp, err := hc.Do(req) if err != nil { - return backupNowOutcome{}, fmt.Errorf("felis-api unreachable (a backup needs it alive): %w", err) + return backupNowOutcome{}, fmt.Errorf("%w: %w", errBackupAPIUnreachable, err) } defer resp.Body.Close() @@ -145,3 +160,439 @@ func backupPickable(servers []haltableServer) []haltableServer { } return out } + +// cmdBackupNow is `felis backup-now`: the world of every user server (or of the +// named ones) archived now, one at a time, through the internal backup route the +// console's Sync uses. A world lives only in its volume and the off-site copy holds +// only its archives, so this is the lever in front of a planned move to another +// host, a disk swap or anything else that could lose a volume. +func cmdBackupNow(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("backup-now", flag.ContinueOnError) + fs.SetOutput(stderr) + cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml") + yes := fs.Bool("yes", false, "back up; without it the plan is printed and nothing changes") + stop := fs.Bool("stop", false, "stop the running servers first: their players are disconnected and the servers stay stopped") + fs.Usage = func() { + fmt.Fprintln(stderr, "Usage: felis backup-now [-yes] [-stop] [server ...]") + fmt.Fprintln(stderr) + fmt.Fprintln(stderr, "Archives the world of every user server, or of the named ones, one at a time, and waits for each archive.") + fmt.Fprintln(stderr, "A running server is skipped unless -stop is given. Without -yes it prints what it would do.") + fs.PrintDefaults() + } + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if os.Geteuid() != 0 { + fmt.Fprintln(stderr, "felis backup-now: refused — it reads the cluster's ops token, so it must run as root (try: sudo felis backup-now)") + return 1 + } + cfg, err := config.Load(*cfgPath) + if err != nil { + fmt.Fprintf(stderr, "felis backup-now: %v\n", err) + return 1 + } + cl, err := buildSystemServerClient() + if err != nil { + fmt.Fprintf(stderr, "felis backup-now: %v\n", err) + return 1 + } + ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer cancel() + + ns := cfg.K8s.Namespace + osUser := accountableOSUser() + jobs := api.NewK8sJobStatus(cl, ns) + var baseURL, token string + var repo ownerStore + var drv *store.PostgresDriver + defer func() { + if drv != nil { + _ = drv.Close() + } + }() + hc := &http.Client{Timeout: 10 * time.Second} + r := backupNowRun{ + out: stdout, + errw: stderr, + ns: ns, + list: func(ctx context.Context) ([]backupNowWorld, error) { return listBackupNowWorlds(ctx, cl, ns) }, + stopped: func(ctx context.Context, name string) (bool, error) { + var ms v1alpha1.MinecraftServer + if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &ms); err != nil { + return false, err + } + return backupNowStopped(ctx, cl, &ms) + }, + halt: func(ctx context.Context, name string) error { + // The database is opened only once a server is to be stopped: the halt + // is audited like the console's, and a run with nothing running needs + // no more than the API. + if repo == nil { + d, err := store.Open(ctx, cfg.Database.URL) + if err != nil { + return fmt.Errorf("open the database for the audit log: %w", err) + } + drv, repo = d, api.NewPGRepo(d.DB()) + } + out, err := performHalt(ctx, cl, repo, ns, name, osUser) + if err != nil { + return err + } + if out.auditErr != nil { + fmt.Fprintf(stderr, "felis backup-now: the audit row for stopping %s was not written: %v\n", name, out.auditErr) + } + return nil + }, + request: func(ctx context.Context, name string) error { + if baseURL == "" { + var err error + if baseURL, token, err = resolveInternalAPI(ctx, cl, platform.DefaultControlNamespace); err != nil { + return fmt.Errorf("%w: %w", errBackupAPIUnreachable, err) + } + } + _, err := requestBackup(ctx, hc, baseURL, token, name, osUser) + return err + }, + jobs: jobs.LatestJobs, + now: time.Now, + sleep: func(ctx context.Context, d time.Duration) { sleepCtx(ctx, d) }, + } + return r.run(ctx, fs.Args(), *yes, *stop) +} + +// sleepCtx waits d or until ctx ends. +func sleepCtx(ctx context.Context, d time.Duration) { + t := time.NewTimer(d) + defer t.Stop() + select { + case <-ctx.Done(): + case <-t.C: + } +} + +// backupNowWorld is one user server as backup-now sees it. +type backupNowWorld struct { + name string + phase string // the observed phase, or the desired state before the operator reconciled it + stopped bool // the backup route's stopped gate admits it + hasWorld bool // its world volume exists +} + +// listBackupNowWorlds lists the user servers of namespace in the API server's +// order (by name), each with what the backup route checks. System servers are +// left out: they have no row in the servers table, so the route refuses them +// (backupPickable). +func listBackupNowWorlds(ctx context.Context, cl client.Client, namespace string) ([]backupNowWorld, error) { + var list v1alpha1.MinecraftServerList + if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil { + return nil, err + } + var out []backupNowWorld + for i := range list.Items { + ms := &list.Items[i] + if isSystemServer(ms.Name) { + continue + } + stopped, err := backupNowStopped(ctx, cl, ms) + if err != nil { + return nil, err + } + var pvc corev1.PersistentVolumeClaim + err = cl.Get(ctx, types.NamespacedName{Namespace: namespace, Name: naming.WorldPVCName(ms.Name)}, &pvc) + if err != nil && !apierrors.IsNotFound(err) { + return nil, fmt.Errorf("look up the world volume of %s: %w", ms.Name, err) + } + phase := string(ms.Status.Phase) + if phase == "" { + phase = string(ms.Spec.DesiredState) + } + out = append(out, backupNowWorld{name: ms.Name, phase: phase, stopped: stopped, hasWorld: err == nil}) + } + return out, nil +} + +// backupNowStopped is the backup route's stopped gate (api.enqueueBackup and +// K8sCluster.AcquireMaintenance together): desired Stopped, not ready, phase +// Stopped and no game pod left. +func backupNowStopped(ctx context.Context, cl client.Client, ms *v1alpha1.MinecraftServer) (bool, error) { + if ms.Spec.DesiredState != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped { + return false, nil + } + var pods corev1.PodList + if err := cl.List(ctx, &pods, client.InNamespace(ms.Namespace), client.MatchingLabels{ + v1alpha1.LabelServer: ms.Name, v1alpha1.LabelComponent: operator.ComponentValue, + }); err != nil { + return false, fmt.Errorf("look up the pod of %s: %w", ms.Name, err) + } + return len(pods.Items) == 0, nil +} + +// errBackupAPIUnreachable is a backup request that never reached felis-api. It +// ends a backup-now run: every later world would fail the same way, and stopping +// servers for backups that cannot be taken only takes them away from players. +var errBackupAPIUnreachable = errors.New("felis-api unreachable (a backup needs it alive)") + +// Polling of backup-now. A graceful stop saves the world first; the backup Job's +// own deadline (backupjob, 30 minutes) ends a Job that hangs, so its wait needs no +// cap of its own. +const ( + backupNowPoll = 2 * time.Second + backupNowStopWait = 10 * time.Minute + backupNowJobAppear = time.Minute +) + +// backupNowRun is backup-now over seams, so the plan and the run are tested +// without a cluster or felis-api. +type backupNowRun struct { + out, errw io.Writer + ns string // where the servers and their Jobs live, for the kubectl hints + list func(ctx context.Context) ([]backupNowWorld, error) + stopped func(ctx context.Context, name string) (bool, error) + halt func(ctx context.Context, name string) error + request func(ctx context.Context, name string) error + jobs func(ctx context.Context, name string) ([]api.AsyncJob, error) + now func() time.Time + sleep func(ctx context.Context, d time.Duration) +} + +// pickBackupNowWorlds narrows worlds to names, in the order given, or keeps them +// all when names is empty. +func pickBackupNowWorlds(worlds []backupNowWorld, names []string) ([]backupNowWorld, error) { + if len(names) == 0 { + return worlds, nil + } + byName := make(map[string]backupNowWorld, len(worlds)) + for _, w := range worlds { + byName[w.name] = w + } + var out []backupNowWorld + seen := map[string]bool{} + for _, n := range names { + if seen[n] { + continue + } + seen[n] = true + w, ok := byName[n] + switch { + case ok: + out = append(out, w) + case isSystemServer(n): + return nil, fmt.Errorf("%s is a system server: its world is rebuilt by felis setup and has no backups", n) + default: + return nil, fmt.Errorf("no server named %q", n) + } + } + return out, nil +} + +// backupNowAction is what the plan does with a world. +func backupNowAction(w backupNowWorld, stop bool) string { + switch { + case w.stopped && !w.hasWorld: + return "skip: no world volume (never started, nothing to save)" + case w.stopped: + return "back up" + case stop: + return "stop, then back up" + default: + return "skip: running (stop it first, or pass -stop)" + } +} + +func (r *backupNowRun) run(ctx context.Context, names []string, yes, stop bool) int { + all, err := r.list(ctx) + if err != nil { + fmt.Fprintf(r.errw, "felis backup-now: list the servers: %v\n", err) + return 1 + } + worlds, err := pickBackupNowWorlds(all, names) + if err != nil { + fmt.Fprintf(r.errw, "felis backup-now: %v\n", err) + return 2 + } + if len(worlds) == 0 { + fmt.Fprintln(r.out, "felis backup-now: there are no user servers") + return 0 + } + + // The stopped worlds go first, so a felis-api that cannot take a backup is + // found before any server is stopped for one. + sort.SliceStable(worlds, func(i, j int) bool { return worlds[i].stopped && !worlds[j].stopped }) + width := 0 + for _, w := range worlds { + width = max(width, len(w.name)) + } + fmt.Fprintf(r.out, "felis backup-now: %d server(s), backed up one at a time:\n", len(worlds)) + stopping, work := false, 0 + for _, w := range worlds { + fmt.Fprintf(r.out, " %-*s %-8s %s\n", width, w.name, w.phase, backupNowAction(w, stop)) + stopping = stopping || (!w.stopped && stop) + if (w.stopped && w.hasWorld) || (!w.stopped && stop) { + work++ + } + } + if work > 0 { + fmt.Fprintln(r.out, "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.") + } + if stopping { + fmt.Fprintln(r.out, "Stopping disconnects the players on those servers, and they stay stopped afterwards.") + } + if !yes { + if work == 0 { + fmt.Fprintln(r.out, "Nothing to back up.") + } else { + fmt.Fprintln(r.out, "Nothing changed. Run again with -yes to back them up.") + } + return 0 + } + + var done, failed, running, empty int + var leftStopped []string + for _, w := range worlds { + if err := ctx.Err(); err != nil { + break + } + switch { + case w.stopped && !w.hasWorld: + empty++ + continue + case !w.stopped && !stop: + running++ + continue + } + if !w.stopped { + halted, err := r.stopWorld(ctx, w.name) + if halted { + leftStopped = append(leftStopped, w.name) + } + if err != nil { + fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err) + failed++ + continue + } + } + err := r.backUp(ctx, w.name) + if errors.Is(err, errBackupAPIUnreachable) { + fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err) + fmt.Fprintln(r.out, "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).") + failed++ + break + } + if err != nil { + fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err) + failed++ + continue + } + done++ + } + + interrupted := ctx.Err() != nil + if interrupted { + fmt.Fprintln(r.out, "Interrupted: a backup Job already started runs to its end.") + } + parts := []string{fmt.Sprintf("%d backed up", done)} + if failed > 0 { + parts = append(parts, fmt.Sprintf("%d failed", failed)) + } + if running > 0 { + parts = append(parts, fmt.Sprintf("%d skipped (running)", running)) + } + if empty > 0 { + parts = append(parts, fmt.Sprintf("%d without a world", empty)) + } + fmt.Fprintf(r.out, "%s.\n", strings.Join(parts, ", ")) + if len(leftStopped) > 0 { + fmt.Fprintf(r.out, "Left stopped: %s. Start them from the panel when you are done.\n", strings.Join(leftStopped, ", ")) + } + if done > 0 { + fmt.Fprintln(r.out, "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.") + } + if failed > 0 || running > 0 || interrupted { + return 1 + } + return 0 +} + +// stopWorld stops one server and waits until the backup route would admit it. +// halted is whether the stop was asked for: the server then stays stopped, even +// when it takes longer than the wait. +func (r *backupNowRun) stopWorld(ctx context.Context, name string) (halted bool, err error) { + fmt.Fprintf(r.out, " %s: stopping\n", name) + if err := r.halt(ctx, name); err != nil { + return false, err + } + start := r.now() + for { + ok, err := r.stopped(ctx, name) + if err != nil { + return true, err + } + if ok { + fmt.Fprintf(r.out, " %s: stopped after %s\n", name, r.now().Sub(start).Round(time.Second)) + return true, nil + } + if r.now().Sub(start) >= backupNowStopWait { + return true, fmt.Errorf("did not stop within %s (kubectl -n %s describe minecraftserver %s)", backupNowStopWait, r.ns, name) + } + if err := ctx.Err(); err != nil { + return true, err + } + r.sleep(ctx, backupNowPoll) + } +} + +// backUp requests one world's backup and waits for its Job to finish. The Job is +// the one of this server that was not there before the request. +func (r *backupNowRun) backUp(ctx context.Context, name string) error { + before, err := r.jobs(ctx, name) + if err != nil { + return fmt.Errorf("list its backup Jobs: %w", err) + } + known := make(map[string]bool, len(before)) + for _, j := range before { + known[j.Name] = true + } + if err := r.request(ctx, name); err != nil { + return err + } + fmt.Fprintf(r.out, " %s: backing up\n", name) + start := r.now() + seen := "" + for { + jobs, err := r.jobs(ctx, name) + if err != nil { + return fmt.Errorf("list its backup Jobs: %w", err) + } + var job *api.AsyncJob + for i := range jobs { + if jobs[i].Kind == "backup" && (jobs[i].Name == seen || seen == "" && !known[jobs[i].Name]) { + job = &jobs[i] + break + } + } + switch { + case job == nil && seen != "": + return fmt.Errorf("its backup Job %s was deleted before it finished", seen) + case job == nil && r.now().Sub(start) >= backupNowJobAppear: + return fmt.Errorf("felis-api accepted the backup, but no backup Job appeared within %s (kubectl -n %s get jobs)", backupNowJobAppear, r.ns) + case job != nil && job.State == "succeeded": + fmt.Fprintf(r.out, " %s: archived in %s\n", name, r.now().Sub(start).Round(time.Second)) + return nil + case job != nil && job.State == "failed": + msg := job.Message + if msg == "" { + msg = "the backup Job failed" + } + return fmt.Errorf("%s (kubectl -n %s logs job/%s)", msg, r.ns, job.Name) + case job != nil: + seen = job.Name + } + if err := ctx.Err(); err != nil { + return err + } + r.sleep(ctx, backupNowPoll) + } +} diff --git a/cmd/felis/backupnow_cmd_test.go b/cmd/felis/backupnow_cmd_test.go new file mode 100644 index 0000000..1a9508f --- /dev/null +++ b/cmd/felis/backupnow_cmd_test.go @@ -0,0 +1,563 @@ +package main + +import ( + "bytes" + "context" + "errors" + "fmt" + "reflect" + "strings" + "testing" + "time" + + "felis.lolicon.best/internal/api" + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/naming" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +// bnRig drives backupNowRun against scripted servers and Jobs on a fake clock that +// moves only when the run sleeps. +type bnRig struct { + out, errw bytes.Buffer + clock time.Time + events []string + worlds []backupNowWorld + // stopAfter is how many polls a halted server takes to stop; -1 never does. + stopAfter map[string]int + polls map[string]int + reqErr map[string]error + // states is what the server's new backup Job reports on each poll after the + // request, the last one repeating; "" is no Job. + states map[string][]string + messages map[string]string + jobPolls map[string]int + // stopErr fails a server's stop polls; jobsFail fails the nth (1-based) Job + // list of a server. + stopErr map[string]error + jobsFail map[string]int + jobCalls map[string]int + // ctx is the run's context, and onAct runs after each halt and request. + ctx context.Context + onAct func(event string) +} + +func newBNRig(worlds ...backupNowWorld) *bnRig { + return &bnRig{ + clock: time.Unix(1_800_000_000, 0), + worlds: worlds, + stopAfter: map[string]int{}, + polls: map[string]int{}, + reqErr: map[string]error{}, + states: map[string][]string{}, + messages: map[string]string{}, + jobPolls: map[string]int{}, + stopErr: map[string]error{}, + jobsFail: map[string]int{}, + jobCalls: map[string]int{}, + ctx: context.Background(), + onAct: func(string) {}, + } +} + +func (g *bnRig) run(names []string, yes, stop bool) int { + r := backupNowRun{ + out: &g.out, errw: &g.errw, ns: "minecraft", + list: func(context.Context) ([]backupNowWorld, error) { + return append([]backupNowWorld(nil), g.worlds...), nil + }, + stopped: func(_ context.Context, name string) (bool, error) { + g.polls[name]++ + if err := g.stopErr[name]; err != nil { + return false, err + } + n := g.stopAfter[name] + return n >= 0 && g.polls[name] > n, nil + }, + halt: func(_ context.Context, name string) error { + g.events = append(g.events, "halt "+name) + g.onAct("halt " + name) + return nil + }, + request: func(_ context.Context, name string) error { + g.events = append(g.events, "request "+name) + g.onAct("request " + name) + if err := g.reqErr[name]; err != nil { + return err + } + g.jobPolls[name] = 0 + return nil + }, + jobs: func(_ context.Context, name string) ([]api.AsyncJob, error) { + // Every server has an older finished backup, and a restore Job that + // shows up with the new backup: neither is the Job to wait for. + g.jobCalls[name]++ + if g.jobCalls[name] == g.jobsFail[name] { + return nil, errors.New("the apiserver is gone") + } + out := []api.AsyncJob{{Name: "backup-" + name + "-old", Kind: "backup", State: "succeeded"}} + n, requested := g.jobPolls[name] + if !requested { + return out, nil + } + g.jobPolls[name] = n + 1 + states := g.states[name] + if len(states) == 0 { + states = []string{"succeeded"} + } + state := states[min(n, len(states)-1)] + if state == "" { + return out, nil + } + return append([]api.AsyncJob{ + {Name: "restore-" + name + "-x", Kind: "restore", State: "succeeded"}, + {Name: "backup-" + name + "-new", Kind: "backup", State: state, Message: g.messages[name]}, + }, out...), nil + }, + now: func() time.Time { return g.clock }, + sleep: func(_ context.Context, d time.Duration) { g.clock = g.clock.Add(d) }, + } + return r.run(g.ctx, names, yes, stop) +} + +func stoppedWorld(name string) backupNowWorld { + return backupNowWorld{name: name, phase: "Stopped", stopped: true, hasWorld: true} +} + +func runningWorld(name string) backupNowWorld { + return backupNowWorld{name: name, phase: "Running", hasWorld: true} +} + +func emptyWorld(name string) backupNowWorld { + return backupNowWorld{name: name, phase: "Stopped", stopped: true} +} + +const ( + bnManualKeep = "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.\n" + bnStopping = "Stopping disconnects the players on those servers, and they stay stopped afterwards.\n" + bnOffsite = "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.\n" +) + +func TestBackupNowPlanChangesNothing(t *testing.T) { + for _, tc := range []struct { + stop bool + want string + }{ + {false, "felis backup-now: 4 server(s), backed up one at a time:\n" + + " alpha Stopped back up\n" + + " charlie Stopped skip: no world volume (never started, nothing to save)\n" + + " bravo Running skip: running (stop it first, or pass -stop)\n" + + " delta Starting skip: running (stop it first, or pass -stop)\n" + + bnManualKeep + + "Nothing changed. Run again with -yes to back them up.\n"}, + {true, "felis backup-now: 4 server(s), backed up one at a time:\n" + + " alpha Stopped back up\n" + + " charlie Stopped skip: no world volume (never started, nothing to save)\n" + + " bravo Running stop, then back up\n" + + " delta Starting stop, then back up\n" + + bnManualKeep + bnStopping + + "Nothing changed. Run again with -yes to back them up.\n"}, + } { + t.Run(fmt.Sprintf("stop=%v", tc.stop), func(t *testing.T) { + delta := runningWorld("delta") + delta.phase = "Starting" + g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), delta) + if code := g.run(nil, false, tc.stop); code != 0 { + t.Fatalf("exit = %d, want 0; stderr %q", code, g.errw.String()) + } + if g.out.String() != tc.want { + t.Fatalf("plan =\n%s\nwant\n%s", g.out.String(), tc.want) + } + if len(g.events) != 0 || len(g.polls) != 0 { + t.Fatalf("the plan acted: events %v, polls %v", g.events, g.polls) + } + }) + } +} + +// A plan that saves nothing says so, without the manual_keep warning; a running +// server counts as something to save once -stop is given. +func TestBackupNowPlanWithNothingToSave(t *testing.T) { + for _, tc := range []struct { + stop bool + want string + }{ + {false, "felis backup-now: 2 server(s), backed up one at a time:\n" + + " charlie Stopped skip: no world volume (never started, nothing to save)\n" + + " bravo Running skip: running (stop it first, or pass -stop)\n" + + "Nothing to back up.\n"}, + {true, "felis backup-now: 2 server(s), backed up one at a time:\n" + + " charlie Stopped skip: no world volume (never started, nothing to save)\n" + + " bravo Running stop, then back up\n" + + bnManualKeep + bnStopping + + "Nothing changed. Run again with -yes to back them up.\n"}, + } { + g := newBNRig(runningWorld("bravo"), emptyWorld("charlie")) + if code := g.run(nil, false, tc.stop); code != 0 { + t.Fatalf("stop=%v: exit = %d, want 0", tc.stop, code) + } + if g.out.String() != tc.want { + t.Fatalf("stop=%v: plan =\n%s\nwant\n%s", tc.stop, g.out.String(), tc.want) + } + } +} + +func TestBackupNowBacksUpEachWorldInTurn(t *testing.T) { + g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), stoppedWorld("delta")) + g.states["alpha"] = []string{"", "running", "succeeded"} + g.states["delta"] = []string{"running", "failed"} + g.messages["delta"] = "felis backup: not enough free disk for the archive" + g.stopAfter["bravo"] = 2 + g.states["bravo"] = []string{"running", "running", "running", "succeeded"} + + code := g.run(nil, true, true) + if code != 1 { + t.Fatalf("exit = %d, want 1 (delta failed)", code) + } + if want := []string{"request alpha", "request delta", "halt bravo", "request bravo"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v", g.events, want) + } + _, run, _ := strings.Cut(g.out.String(), bnStopping+"") + want := " alpha: backing up\n" + + " alpha: archived in 4s\n" + + " delta: backing up\n" + + " delta: failed: felis backup: not enough free disk for the archive (kubectl -n minecraft logs job/backup-delta-new)\n" + + " bravo: stopping\n" + + " bravo: stopped after 4s\n" + + " bravo: backing up\n" + + " bravo: archived in 6s\n" + + "2 backed up, 1 failed, 1 without a world.\n" + + "Left stopped: bravo. Start them from the panel when you are done.\n" + + bnOffsite + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } +} + +func TestBackupNowSkipsRunningServersWithoutStop(t *testing.T) { + g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo")) + if code := g.run(nil, true, false); code != 1 { + t.Fatalf("exit = %d, want 1 (bravo was not backed up)", code) + } + if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v", g.events, want) + } + if len(g.polls) != 0 { + t.Fatalf("polled a server it did not stop: %v", g.polls) + } + _, run, _ := strings.Cut(g.out.String(), "Nothing changed") + if run != "" { + t.Fatalf("-yes printed the plan's closing line") + } + _, run, _ = strings.Cut(g.out.String(), bnManualKeep) + want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 skipped (running).\n" + bnOffsite + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } +} + +func TestBackupNowExitsCleanWhenEverythingIsSaved(t *testing.T) { + // -stop with nothing running stops nothing and says nothing about stopping. + g := newBNRig(stoppedWorld("alpha"), emptyWorld("charlie")) + if code := g.run(nil, true, true); code != 0 { + t.Fatalf("exit = %d, want 0; output\n%s", code, g.out.String()) + } + _, run, _ := strings.Cut(g.out.String(), bnManualKeep) + if want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 without a world.\n" + bnOffsite; run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } + + g = newBNRig(emptyWorld("charlie")) + if code := g.run(nil, true, false); code != 0 { + t.Fatalf("exit = %d, want 0 for a server with nothing to save", code) + } + // Nothing to archive: no manual_keep warning, and no off-site hint. + if want := "felis backup-now: 1 server(s), backed up one at a time:\n" + + " charlie Stopped skip: no world volume (never started, nothing to save)\n" + + "0 backed up, 1 without a world.\n"; g.out.String() != want { + t.Fatalf("output =\n%s\nwant\n%s", g.out.String(), want) + } + if len(g.events) != 0 { + t.Fatalf("events = %v, want none", g.events) + } +} + +func TestBackupNowStopsAtAnUnreachableAPI(t *testing.T) { + g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), runningWorld("charlie")) + g.reqErr["alpha"] = fmt.Errorf("%w: dial tcp 10.43.0.9:8081: connect: connection refused", errBackupAPIUnreachable) + if code := g.run(nil, true, true); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v: nothing after the API proved unreachable, and no server stopped", g.events, want) + } + _, run, _ := strings.Cut(g.out.String(), bnStopping) + want := " alpha: failed: felis-api unreachable (a backup needs it alive): dial tcp 10.43.0.9:8081: connect: connection refused\n" + + "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).\n" + + "0 backed up, 1 failed.\n" + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } + + // Any other refusal is that world's alone: the run goes on. + g = newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo")) + g.reqErr["alpha"] = errors.New("felis-api: the world is being restored") + if code := g.run(nil, true, false); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if want := []string{"request alpha", "request bravo"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v", g.events, want) + } +} + +func TestBackupNowGivesUpOnAServerThatDoesNotStop(t *testing.T) { + g := newBNRig(runningWorld("bravo"), runningWorld("echo")) + g.stopAfter["bravo"] = -1 + if code := g.run(nil, true, true); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if want := []string{"halt bravo", "halt echo", "request echo"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v", g.events, want) + } + // One poll at the start and one per 2s sleep up to the 10-minute mark. + if g.polls["bravo"] != 301 { + t.Fatalf("bravo polled %d times, want 301", g.polls["bravo"]) + } + _, run, _ := strings.Cut(g.out.String(), bnStopping) + want := " bravo: stopping\n" + + " bravo: failed: did not stop within 10m0s (kubectl -n minecraft describe minecraftserver bravo)\n" + + " echo: stopping\n" + + " echo: stopped after 0s\n" + + " echo: backing up\n" + + " echo: archived in 0s\n" + + "1 backed up, 1 failed.\n" + + "Left stopped: bravo, echo. Start them from the panel when you are done.\n" + + bnOffsite + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } +} + +func TestBackupNowReportsAJobThatNeverRuns(t *testing.T) { + for _, tc := range []struct { + name string + states []string + polls int + want string + }{ + {"never appears", []string{""}, 31, + "felis-api accepted the backup, but no backup Job appeared within 1m0s (kubectl -n minecraft get jobs)"}, + {"deleted while running", []string{"running", "running", ""}, 3, + "its backup Job backup-alpha-new was deleted before it finished"}, + {"fails without a message", []string{"failed"}, 1, + "the backup Job failed (kubectl -n minecraft logs job/backup-alpha-new)"}, + } { + t.Run(tc.name, func(t *testing.T) { + g := newBNRig(stoppedWorld("alpha")) + g.states["alpha"] = tc.states + if code := g.run(nil, true, false); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if !strings.Contains(g.out.String(), " alpha: failed: "+tc.want+"\n") { + t.Fatalf("output =\n%s\nwant the line %q", g.out.String(), tc.want) + } + if g.jobPolls["alpha"] != tc.polls { + t.Fatalf("polled the Jobs %d times after the request, want %d", g.jobPolls["alpha"], tc.polls) + } + }) + } +} + +func TestBackupNowNamedServers(t *testing.T) { + g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), stoppedWorld("delta")) + if code := g.run([]string{"delta", "alpha", "delta"}, true, false); code != 0 { + t.Fatalf("exit = %d, want 0", code) + } + if want := []string{"request delta", "request alpha"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v", g.events, want) + } + if !strings.HasPrefix(g.out.String(), "felis backup-now: 2 server(s), backed up one at a time:\n delta Stopped back up\n alpha Stopped back up\n") { + t.Fatalf("plan =\n%s", g.out.String()) + } + + for _, tc := range []struct{ name, want string }{ + {"login", "felis backup-now: login is a system server: its world is rebuilt by felis setup and has no backups\n"}, + {"lobby", "felis backup-now: lobby is a system server: its world is rebuilt by felis setup and has no backups\n"}, + {"nope", "felis backup-now: no server named \"nope\"\n"}, + } { + g := newBNRig(stoppedWorld("alpha")) + if code := g.run([]string{"alpha", tc.name}, true, false); code != 2 { + t.Fatalf("%s: exit = %d, want 2", tc.name, code) + } + if g.errw.String() != tc.want { + t.Fatalf("%s: stderr = %q, want %q", tc.name, g.errw.String(), tc.want) + } + if len(g.events) != 0 || g.out.Len() != 0 { + t.Fatalf("%s: acted on a bad name: events %v, output %q", tc.name, g.events, g.out.String()) + } + } + + g = newBNRig() + if code := g.run(nil, true, false); code != 0 || g.out.String() != "felis backup-now: there are no user servers\n" { + t.Fatalf("empty fleet: exit %d, output %q", code, g.out.String()) + } +} + +// The world list mirrors the backup route's own gate, so the plan says exactly what +// the route would refuse. +func TestListBackupNowWorlds(t *testing.T) { + withStatus := func(ms *v1alpha1.MinecraftServer, ready bool) *v1alpha1.MinecraftServer { + ms.Status.Ready = ready + return ms + } + pvc := func(server, ns string) *corev1.PersistentVolumeClaim { + return &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: naming.WorldPVCName(server), Namespace: ns}} + } + pod := func(name, ns string, labels map[string]string) *corev1.Pod { + return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: labels}} + } + gameLabels := func(server string) map[string]string { + return map[string]string{v1alpha1.LabelServer: server, v1alpha1.LabelComponent: "server"} + } + other := mcServer("zulu", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped) + other.Namespace = "elsewhere" + hotel := mcServer("hotel", v1alpha1.DesiredStopped, "") + objs := []client.Object{ + mcServer("alpha", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("alpha", haltNS), + // A backup Job's pod carries the server label with its own component, and + // a game pod of the same name in another namespace is somebody else's. + pod("backup-alpha-1-x", haltNS, map[string]string{v1alpha1.LabelServer: "alpha", "app.kubernetes.io/component": "world-backup"}), + pod("alpha-0", "elsewhere", gameLabels("alpha")), + withStatus(mcServer("bravo", v1alpha1.DesiredRunning, v1alpha1.PhaseRunning), true), pvc("bravo", haltNS), pod("bravo-0", haltNS, gameLabels("bravo")), + mcServer("charlie", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), + mcServer("delta", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("delta", haltNS), pod("delta-0", haltNS, gameLabels("delta")), + mcServer("echo", v1alpha1.DesiredStopped, v1alpha1.PhaseStopping), pvc("echo", haltNS), + mcServer("foxtrot", v1alpha1.DesiredRunning, v1alpha1.PhaseStopped), pvc("foxtrot", haltNS), + withStatus(mcServer("golf", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), true), pvc("golf", haltNS), + hotel, pvc("hotel", haltNS), + mcServer("login", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("login", haltNS), + mcServer("lobby", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("lobby", haltNS), + other, pvc("zulu", "elsewhere"), + } + got, err := listBackupNowWorlds(context.Background(), haltClient(t, objs...), haltNS) + if err != nil { + t.Fatal(err) + } + want := []backupNowWorld{ + {name: "alpha", phase: "Stopped", stopped: true, hasWorld: true}, + {name: "bravo", phase: "Running", hasWorld: true}, + {name: "charlie", phase: "Stopped", stopped: true}, + {name: "delta", phase: "Stopped", hasWorld: true}, // its pod is still going + {name: "echo", phase: "Stopping", hasWorld: true}, // still stopping + {name: "foxtrot", phase: "Stopped", hasWorld: true}, // asked to start + {name: "golf", phase: "Stopped", hasWorld: true}, // still reports ready + {name: "hotel", phase: "Stopped", hasWorld: true}, // not reconciled yet: the desired state shows + } + if !reflect.DeepEqual(got, want) { + t.Fatalf("worlds =\n%+v\nwant\n%+v", got, want) + } +} + +func TestBackupNowStopsWhenInterrupted(t *testing.T) { + t.Run("between worlds", func(t *testing.T) { + g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo")) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + g.ctx = ctx + g.onAct = func(e string) { + if e == "request alpha" { + cancel() + } + } + if code := g.run(nil, true, false); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) { + t.Fatalf("events = %v, want %v", g.events, want) + } + _, run, _ := strings.Cut(g.out.String(), bnManualKeep) + want := " alpha: backing up\n alpha: archived in 0s\n" + + "Interrupted: a backup Job already started runs to its end.\n" + + "1 backed up.\n" + bnOffsite + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } + }) + t.Run("while a Job runs", func(t *testing.T) { + g := newBNRig(stoppedWorld("alpha")) + g.states["alpha"] = []string{"running"} + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + g.ctx = ctx + g.onAct = func(string) { cancel() } + if code := g.run(nil, true, false); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if g.jobPolls["alpha"] != 1 { + t.Fatalf("polled the Job %d times after the interrupt, want 1", g.jobPolls["alpha"]) + } + if !strings.Contains(g.out.String(), " alpha: failed: context canceled\nInterrupted: a backup Job already started runs to its end.\n0 backed up, 1 failed.\n") { + t.Fatalf("output =\n%s", g.out.String()) + } + }) + t.Run("while a server stops", func(t *testing.T) { + g := newBNRig(runningWorld("bravo")) + g.stopAfter["bravo"] = -1 + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + g.ctx = ctx + g.onAct = func(string) { cancel() } + if code := g.run(nil, true, true); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + if g.polls["bravo"] != 1 { + t.Fatalf("polled bravo %d times after the interrupt, want 1", g.polls["bravo"]) + } + _, run, _ := strings.Cut(g.out.String(), bnStopping) + want := " bravo: stopping\n bravo: failed: context canceled\n" + + "Interrupted: a backup Job already started runs to its end.\n" + + "0 backed up, 1 failed.\n" + + "Left stopped: bravo. Start them from the panel when you are done.\n" + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } + }) +} + +func TestBackupNowReportsClusterErrors(t *testing.T) { + g := newBNRig(runningWorld("bravo")) + g.stopErr["bravo"] = errors.New("the apiserver is gone") + if code := g.run(nil, true, true); code != 1 { + t.Fatalf("exit = %d, want 1", code) + } + _, run, _ := strings.Cut(g.out.String(), bnStopping) + // The stop was asked for, so the server stays stopped whatever the poll said. + want := " bravo: stopping\n bravo: failed: the apiserver is gone\n0 backed up, 1 failed.\n" + + "Left stopped: bravo. Start them from the panel when you are done.\n" + if run != want { + t.Fatalf("run =\n%s\nwant\n%s", run, want) + } + + for _, tc := range []struct { + call int + events []string + }{ + {1, nil}, // before the request: nothing is asked for + {2, []string{"request alpha"}}, // the first poll after it + } { + g := newBNRig(stoppedWorld("alpha")) + g.jobsFail["alpha"] = tc.call + if code := g.run(nil, true, false); code != 1 { + t.Fatalf("call %d: exit = %d, want 1", tc.call, code) + } + if !reflect.DeepEqual(g.events, tc.events) { + t.Fatalf("call %d: events = %v, want %v", tc.call, g.events, tc.events) + } + if !strings.Contains(g.out.String(), " alpha: failed: list its backup Jobs: the apiserver is gone\n") { + t.Fatalf("call %d: output =\n%s", tc.call, g.out.String()) + } + } +} diff --git a/cmd/felis/backupnow_test.go b/cmd/felis/backupnow_test.go index a3c51da..0600915 100644 --- a/cmd/felis/backupnow_test.go +++ b/cmd/felis/backupnow_test.go @@ -3,6 +3,7 @@ package main import ( "context" "encoding/json" + "errors" "fmt" "io" "net/http" @@ -152,6 +153,10 @@ func TestRequestBackup(t *testing.T) { if err == nil || !strings.Contains(err.Error(), "unreachable") { t.Fatalf("err = %v, want an 'unreachable' transport error", err) } + // backup-now ends its run on this error alone. + if !errors.Is(err, errBackupAPIUnreachable) { + t.Fatalf("err = %v, want it to wrap errBackupAPIUnreachable", err) + } }) } diff --git a/cmd/felis/run.go b/cmd/felis/run.go index 966ec6a..ecac0e7 100644 --- a/cmd/felis/run.go +++ b/cmd/felis/run.go @@ -20,6 +20,7 @@ Commands: reaper Run the world reaper / backup batch restore Extract a world archive into a world volume (internal Job entrypoint) backup Archive a world into the backup store and record it (internal Job entrypoint) + backup-now Archive every user server's world now, one at a time (or the named ones; -stop stops running ones first; prints the plan, -yes applies; requires root/sudo) files List/read/write one file in a stopped server's world (internal Job entrypoint) egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint) fetch-context Fetch and extract a submission's build context (internal Job entrypoint) @@ -60,6 +61,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{ "reaper": cmdReaper, "restore": cmdRestore, "backup": cmdBackup, + "backup-now": cmdBackupNow, "files": cmdFiles, "egress-gate": cmdEgressGate, "fetch-context": cmdFetchContext, diff --git a/docs/operations.md b/docs/operations.md index bb3cdf4..93f21b9 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -262,6 +262,95 @@ its threshold, and §13b covers a full disk. On a host that built its images, `docker builder prune -af` (with Docker started) reclaims the build cache when space is short; the next upgrade rebuilds it. +### Growing the disk + +Everything above shares the root filesystem, so more room means a bigger root +filesystem. It grows in place, with everything running: enlarge the virtual disk at the +provider, then the partition and the filesystem on it. + +```bash +sudo felis backup-now -yes # a mistyped partition number is how a resize loses a disk +lsblk -f # which disk and partition hold /, and whether LVM sits on it +sudo growpart /dev/vda 3 # cloud-utils-growpart (RHEL) / cloud-guest-utils (Debian, Ubuntu) +# LVM (the RHEL-family default): +sudo pvresize /dev/vda3 +sudo lvextend -r -l +100%FREE /dev//root # -r grows the filesystem with it +# no LVM: +sudo xfs_growfs / # xfs +sudo resize2fs /dev/vda3 # ext4 +df -h / +``` + +`felis backup-now` (troubleshooting.md §10) archives every stopped world; add `-stop` to +include the running ones. + +### Moving the data to its own disk [VM-VERIFIED] + +The bulk lives under `/var/lib/rancher/k3s`: the worlds, the world archives, the registry +and the images. On a disk of its own it grows without touching the system, and a full +world store leaves the root filesystem alone. The database and its bundles +(`/var/lib/felis`) are small and stay on the root disk. The move takes the platform down +for the copy plus a minute or two: the drill copied 4.2 GB in 18 s, and felis-api answered +`/readyz` 14 s after k3s started on the new disk. + +1. Attach the disk and put a filesystem on it (the whole disk; `lsblk` shows it empty): + + ```bash + sudo mkfs.xfs /dev/vdb + U=$(sudo blkid -s UUID -o value /dev/vdb) + ``` + +2. Archive every world, stopping the servers so each one saves, and keep the watchdog + quiet for the next hour (the marker the installer writes: no mail, no failure pings + to the heartbeat, until the time in it): + + ```bash + sudo felis backup-now -yes -stop + sudo install -d -m 0755 /run/felis + echo $(( $(date +%s) + 3600 )) | sudo tee /run/felis/watchdog-quiet-until + ``` + +3. Stop k3s and copy: + + ```bash + sudo systemctl stop k3s + sudo /usr/local/bin/k3s-killall.sh # the containers k3s leaves running, and their mounts + sudo mkdir -p /mnt/felis-data + sudo mount UUID=$U /mnt/felis-data + sudo rsync -aHAX --numeric-ids /var/lib/rancher/k3s/ /mnt/felis-data/ + sudo umount /mnt/felis-data + ``` + + `-X` carries the SELinux labels k3s set itself. Leave `restorecon` out: it would reset + runc and the CNI binaries from `container_runtime_exec_t` to the policy default. + +4. Mount it in place, and tie k3s to the mount: + + ```bash + sudo mv /var/lib/rancher/k3s /var/lib/rancher/k3s.old + sudo mkdir /var/lib/rancher/k3s + echo "UUID=$U /var/lib/rancher/k3s xfs defaults,nofail 0 0" | sudo tee -a /etc/fstab + sudo mkdir -p /etc/systemd/system/k3s.service.d + printf '[Unit]\nRequiresMountsFor=/var/lib/rancher/k3s\n' | sudo tee /etc/systemd/system/k3s.service.d/data-disk.conf + sudo systemctl daemon-reload + sudo mount /var/lib/rancher/k3s + sudo systemctl start k3s + ``` + + The drop-in is what keeps the data safe: k3s started on the empty mount point creates + a new, empty cluster there. With it, a disk that does not come up fails the start with + `A dependency job for k3s.service failed`, and `nofail` keeps the host booting so you + can reach it. In the drill a detached disk left k3s inactive and the mount point empty; + reattached, `systemctl start k3s` mounted it and started. + +5. Check that `sudo k3s kubectl -n felis get pods` shows every pod ready and + `findmnt /var/lib/rancher/k3s` names the new disk, then start the servers from the + panel and `sudo rm /run/felis/watchdog-quiet-until`. Once the host has run a day, + `sudo rm -rf /var/lib/rancher/k3s.old` frees the root disk. + +The watchdog already watches `/var/lib/rancher/k3s` as a filesystem of its own (its +`-disk-paths`), so the new disk's fill level is mailed like the root's. + ## 3. Uninstall `deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove @@ -604,6 +693,52 @@ production install: non-zero when the copy or the newest bundle is stale; wire them into your monitoring, or rely on the watchdog's mail. +### Moving to another host (planned) + +A planned move is the rebuild of troubleshooting.md §16, with the old host still there to +hand over a copy that misses nothing. It needs the off-site bucket: that is how the world +archives reach the new host (§16 step 7). The platform is down from step 1 until the new +host serves. + +1. **On the old host**, stop everything that changes a world, then send the last copy: + + ```bash + sudo install -d -m 0755 /run/felis + echo $(( $(date +%s) + 4 * 3600 )) | sudo tee /run/felis/watchdog-quiet-until + sudo systemctl stop felis-velocity # no joins, so no server wakes + sudo felis backup-now -yes -stop # every world archived; the servers stay stopped + sudo k3s kubectl -n felis scale deploy/felis-operator --replicas=0 # nothing starts a server from here on + sudo felis db backup # a bundle that lists those archives + sudo systemctl start felis-offsite.service + sudo felis offsite status # again until nothing waits + ``` + + The order matters. The new host fetches the archives its restored database lists, so + the bundle comes after the last archive. The operator goes after `backup-now`, which + needs it to stop the servers. The quiet marker keeps the watchdog from mailing the + owners about the stopped proxy and operator for the next 4 hours. +2. **On the new host**, follow troubleshooting.md §16 "Rebuild on a new host" from step 1; + `fetch-db latest` picks the bundle the old host just sent. Step 8 (`felis offsite + take-over`) makes the new host the one that writes the bucket, and from then on the + old host copies nothing more. Steps 10 and 11 move the names and the tunnel. +3. **Check the new host** before announcing it: sign in with an email code, restore one + world and join it, and see `sudo felis offsite status` show a recent `last success` and + no stand-by notice. +4. **Retire the old host.** It holds the last copy of every world outside the bucket, so + keep it powered off with its disk for a few days first, disabled so a boot brings + nothing up: + + ```bash + sudo systemctl disable k3s felis-velocity felis-watchdog.timer felis-offsite.timer \ + felis-db-backup.timer felis-update-check.timer felis-build-tools.timer + sudo poweroff + ``` + + Then uninstall it (§3) or wipe it. + +Each step is covered where it is documented (backup-now in troubleshooting.md §10, the +rebuild in §16); the sequence as a whole has not been rehearsed as one move. + ## 6. Changing the root domain [VM-VERIFIED] [GO-TESTED] [SH-TESTED] The root domain is written into more places than the installer's config: the panel diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 5b85771..efad008 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -1080,6 +1080,56 @@ the error its container exited on under Recent operations on the server's backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`, `TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`] +### Every world at once: `felis backup-now` + +A world lives only in its volume, and the off-site copy holds only its archives. +Before anything that could lose a volume (growing the disk, moving the data to its +own disk, moving to another host: operations.md §2 and §5), archive every world +from the node: + +```bash +sudo felis backup-now # the plan; nothing changes +sudo felis backup-now -yes # archive every stopped world, one at a time +sudo felis backup-now -yes -stop # stop the running servers first +sudo felis backup-now -yes alpha bravo # only these servers +``` + +It runs as root because it reads the ops token (`felis/felis-ops-token`, §6) and +asks felis-api's internal face for each backup. Each archive is an ordinary manual +backup (the same Job, the same 10% free-disk check, the audit action +`backup.create` with the source `internal:ops` and the sudo user as actor), exempt +from the owner cooldown and the `max_local_bytes` cap like the break-glass console. + +- **The plan** lists every user server with its phase and what the run does with + it: `back up`, `stop, then back up`, `skip: running (stop it first, or pass + -stop)`, or `skip: no world volume (never started, nothing to save)`. With + nothing to archive it ends `Nothing to back up.` +- **Each archive counts against `manual_keep`**: a server that already holds that + many manual backups loses its oldest, and the plan says so. Raise + `[archive] manual_keep` first when those older backups matter. +- **Stopped servers go first**, so a felis-api that cannot take a backup is found + before anything is stopped for one. The run waits for each Job (`alpha: archived + in 42s`) before starting the next. +- **`-stop` disconnects the players** and leaves those servers stopped (`Left + stopped: …`; start them from the panel). Each stop writes `break_glass.halt` to + the audit log. A server still up after 10 minutes counts as failed, and the run + moves on. +- **An unreachable felis-api ends the run** (`Stopped: nothing more can be backed + up until felis-api answers`). Ctrl-C ends it after the current step; a backup Job + already started runs to its end. +- **The archives stay on the node** until the hourly off-site copy. After a run + that archived something the command prints `sudo systemctl start + felis-offsite.service`, which sends them now; `sudo felis offsite status` shows + what still waits. + +It exits 0 when every world with a volume was archived, 1 when a backup failed, a +running server was skipped (no `-stop`) or the run was interrupted, and 2 for a +name that is no user server. [GO-TESTED: `TestBackupNowPlanChangesNothing`, +`TestBackupNowPlanWithNothingToSave`, `TestBackupNowBacksUpEachWorldInTurn`, +`TestBackupNowSkipsRunningServersWithoutStop`, +`TestBackupNowStopsAtAnUnreachableAPI`, `TestBackupNowStopsWhenInterrupted`, +`TestBackupNowNamedServers`] + ### Scheduled backups (daily restore points) A world played every day never idles 15 days, so the reaper never archives it.