Unverified Commit 28fd20c7 authored by Lemon-miaow's avatar Lemon-miaow
Browse files

feat(backup-now): 一次备份全部世界的命令,补扩盘/数据盘与计划内迁机指南

parent 64476b01
Loading
Loading
Loading
Loading
+453 −2
Changes for cmd/felis/backupnow.go: 453 added lines, 2 removed lines.
Original line number Diff line number Diff line
@@ -4,15 +4,28 @@ import (
	"bytes"
	"context"
	"encoding/json"
	"errors"
	"flag"
	"fmt"
	"io"
	"net/http"
	"os"
	"os/signal"
	"sort"
	"strings"
	"syscall"
	"time"

	"felis.lolicon.best/internal/api"
	"felis.lolicon.best/internal/apis/felis/v1alpha1"
	"felis.lolicon.best/internal/config"
	"felis.lolicon.best/internal/naming"
	"felis.lolicon.best/internal/operator"
	"felis.lolicon.best/internal/platform"
	"felis.lolicon.best/internal/store"

	corev1 "k8s.io/api/core/v1"
	apierrors "k8s.io/apimachinery/pkg/api/errors"
	"k8s.io/apimachinery/pkg/types"
	"sigs.k8s.io/controller-runtime/pkg/client"
)
@@ -22,7 +35,9 @@ import (
// (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console
// cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth)
// while the API is alive, and the API renders the Job and audits the action. This file
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue.
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. The
// `felis backup-now` command (cmdBackupNow, below) takes the same route for every
// user server in turn.

// backupNowOutcome is the durable result of a backup request, re-printed after the TUI
// alt-screen tears down.
@@ -75,7 +90,7 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o

	resp, err := hc.Do(req)
	if err != nil {
		return backupNowOutcome{}, fmt.Errorf("felis-api unreachable (a backup needs it alive): %w", err)
		return backupNowOutcome{}, fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
	}
	defer resp.Body.Close()

@@ -145,3 +160,439 @@ func backupPickable(servers []haltableServer) []haltableServer {
	}
	return out
}

// cmdBackupNow is `felis backup-now`: the world of every user server (or of the
// named ones) archived now, one at a time, through the internal backup route the
// console's Sync uses. A world lives only in its volume and the off-site copy holds
// only its archives, so this is the lever in front of a planned move to another
// host, a disk swap or anything else that could lose a volume.
func cmdBackupNow(args []string, stdout, stderr io.Writer) int {
	fs := flag.NewFlagSet("backup-now", flag.ContinueOnError)
	fs.SetOutput(stderr)
	cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
	yes := fs.Bool("yes", false, "back up; without it the plan is printed and nothing changes")
	stop := fs.Bool("stop", false, "stop the running servers first: their players are disconnected and the servers stay stopped")
	fs.Usage = func() {
		fmt.Fprintln(stderr, "Usage: felis backup-now [-yes] [-stop] [server ...]")
		fmt.Fprintln(stderr)
		fmt.Fprintln(stderr, "Archives the world of every user server, or of the named ones, one at a time, and waits for each archive.")
		fmt.Fprintln(stderr, "A running server is skipped unless -stop is given. Without -yes it prints what it would do.")
		fs.PrintDefaults()
	}
	if err := fs.Parse(args); err != nil {
		if errors.Is(err, flag.ErrHelp) {
			return 0
		}
		return 2
	}
	if os.Geteuid() != 0 {
		fmt.Fprintln(stderr, "felis backup-now: refused — it reads the cluster's ops token, so it must run as root (try: sudo felis backup-now)")
		return 1
	}
	cfg, err := config.Load(*cfgPath)
	if err != nil {
		fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
		return 1
	}
	cl, err := buildSystemServerClient()
	if err != nil {
		fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
		return 1
	}
	ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
	defer cancel()

	ns := cfg.K8s.Namespace
	osUser := accountableOSUser()
	jobs := api.NewK8sJobStatus(cl, ns)
	var baseURL, token string
	var repo ownerStore
	var drv *store.PostgresDriver
	defer func() {
		if drv != nil {
			_ = drv.Close()
		}
	}()
	hc := &http.Client{Timeout: 10 * time.Second}
	r := backupNowRun{
		out:  stdout,
		errw: stderr,
		ns:   ns,
		list: func(ctx context.Context) ([]backupNowWorld, error) { return listBackupNowWorlds(ctx, cl, ns) },
		stopped: func(ctx context.Context, name string) (bool, error) {
			var ms v1alpha1.MinecraftServer
			if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &ms); err != nil {
				return false, err
			}
			return backupNowStopped(ctx, cl, &ms)
		},
		halt: func(ctx context.Context, name string) error {
			// The database is opened only once a server is to be stopped: the halt
			// is audited like the console's, and a run with nothing running needs
			// no more than the API.
			if repo == nil {
				d, err := store.Open(ctx, cfg.Database.URL)
				if err != nil {
					return fmt.Errorf("open the database for the audit log: %w", err)
				}
				drv, repo = d, api.NewPGRepo(d.DB())
			}
			out, err := performHalt(ctx, cl, repo, ns, name, osUser)
			if err != nil {
				return err
			}
			if out.auditErr != nil {
				fmt.Fprintf(stderr, "felis backup-now: the audit row for stopping %s was not written: %v\n", name, out.auditErr)
			}
			return nil
		},
		request: func(ctx context.Context, name string) error {
			if baseURL == "" {
				var err error
				if baseURL, token, err = resolveInternalAPI(ctx, cl, platform.DefaultControlNamespace); err != nil {
					return fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
				}
			}
			_, err := requestBackup(ctx, hc, baseURL, token, name, osUser)
			return err
		},
		jobs:  jobs.LatestJobs,
		now:   time.Now,
		sleep: func(ctx context.Context, d time.Duration) { sleepCtx(ctx, d) },
	}
	return r.run(ctx, fs.Args(), *yes, *stop)
}

// sleepCtx waits d or until ctx ends.
func sleepCtx(ctx context.Context, d time.Duration) {
	t := time.NewTimer(d)
	defer t.Stop()
	select {
	case <-ctx.Done():
	case <-t.C:
	}
}

// backupNowWorld is one user server as backup-now sees it.
type backupNowWorld struct {
	name     string
	phase    string // the observed phase, or the desired state before the operator reconciled it
	stopped  bool   // the backup route's stopped gate admits it
	hasWorld bool   // its world volume exists
}

// listBackupNowWorlds lists the user servers of namespace in the API server's
// order (by name), each with what the backup route checks. System servers are
// left out: they have no row in the servers table, so the route refuses them
// (backupPickable).
func listBackupNowWorlds(ctx context.Context, cl client.Client, namespace string) ([]backupNowWorld, error) {
	var list v1alpha1.MinecraftServerList
	if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
		return nil, err
	}
	var out []backupNowWorld
	for i := range list.Items {
		ms := &list.Items[i]
		if isSystemServer(ms.Name) {
			continue
		}
		stopped, err := backupNowStopped(ctx, cl, ms)
		if err != nil {
			return nil, err
		}
		var pvc corev1.PersistentVolumeClaim
		err = cl.Get(ctx, types.NamespacedName{Namespace: namespace, Name: naming.WorldPVCName(ms.Name)}, &pvc)
		if err != nil && !apierrors.IsNotFound(err) {
			return nil, fmt.Errorf("look up the world volume of %s: %w", ms.Name, err)
		}
		phase := string(ms.Status.Phase)
		if phase == "" {
			phase = string(ms.Spec.DesiredState)
		}
		out = append(out, backupNowWorld{name: ms.Name, phase: phase, stopped: stopped, hasWorld: err == nil})
	}
	return out, nil
}

// backupNowStopped is the backup route's stopped gate (api.enqueueBackup and
// K8sCluster.AcquireMaintenance together): desired Stopped, not ready, phase
// Stopped and no game pod left.
func backupNowStopped(ctx context.Context, cl client.Client, ms *v1alpha1.MinecraftServer) (bool, error) {
	if ms.Spec.DesiredState != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
		return false, nil
	}
	var pods corev1.PodList
	if err := cl.List(ctx, &pods, client.InNamespace(ms.Namespace), client.MatchingLabels{
		v1alpha1.LabelServer: ms.Name, v1alpha1.LabelComponent: operator.ComponentValue,
	}); err != nil {
		return false, fmt.Errorf("look up the pod of %s: %w", ms.Name, err)
	}
	return len(pods.Items) == 0, nil
}

// errBackupAPIUnreachable is a backup request that never reached felis-api. It
// ends a backup-now run: every later world would fail the same way, and stopping
// servers for backups that cannot be taken only takes them away from players.
var errBackupAPIUnreachable = errors.New("felis-api unreachable (a backup needs it alive)")

// Polling of backup-now. A graceful stop saves the world first; the backup Job's
// own deadline (backupjob, 30 minutes) ends a Job that hangs, so its wait needs no
// cap of its own.
const (
	backupNowPoll      = 2 * time.Second
	backupNowStopWait  = 10 * time.Minute
	backupNowJobAppear = time.Minute
)

// backupNowRun is backup-now over seams, so the plan and the run are tested
// without a cluster or felis-api.
type backupNowRun struct {
	out, errw io.Writer
	ns        string // where the servers and their Jobs live, for the kubectl hints
	list      func(ctx context.Context) ([]backupNowWorld, error)
	stopped   func(ctx context.Context, name string) (bool, error)
	halt      func(ctx context.Context, name string) error
	request   func(ctx context.Context, name string) error
	jobs      func(ctx context.Context, name string) ([]api.AsyncJob, error)
	now       func() time.Time
	sleep     func(ctx context.Context, d time.Duration)
}

// pickBackupNowWorlds narrows worlds to names, in the order given, or keeps them
// all when names is empty.
func pickBackupNowWorlds(worlds []backupNowWorld, names []string) ([]backupNowWorld, error) {
	if len(names) == 0 {
		return worlds, nil
	}
	byName := make(map[string]backupNowWorld, len(worlds))
	for _, w := range worlds {
		byName[w.name] = w
	}
	var out []backupNowWorld
	seen := map[string]bool{}
	for _, n := range names {
		if seen[n] {
			continue
		}
		seen[n] = true
		w, ok := byName[n]
		switch {
		case ok:
			out = append(out, w)
		case isSystemServer(n):
			return nil, fmt.Errorf("%s is a system server: its world is rebuilt by felis setup and has no backups", n)
		default:
			return nil, fmt.Errorf("no server named %q", n)
		}
	}
	return out, nil
}

// backupNowAction is what the plan does with a world.
func backupNowAction(w backupNowWorld, stop bool) string {
	switch {
	case w.stopped && !w.hasWorld:
		return "skip: no world volume (never started, nothing to save)"
	case w.stopped:
		return "back up"
	case stop:
		return "stop, then back up"
	default:
		return "skip: running (stop it first, or pass -stop)"
	}
}

func (r *backupNowRun) run(ctx context.Context, names []string, yes, stop bool) int {
	all, err := r.list(ctx)
	if err != nil {
		fmt.Fprintf(r.errw, "felis backup-now: list the servers: %v\n", err)
		return 1
	}
	worlds, err := pickBackupNowWorlds(all, names)
	if err != nil {
		fmt.Fprintf(r.errw, "felis backup-now: %v\n", err)
		return 2
	}
	if len(worlds) == 0 {
		fmt.Fprintln(r.out, "felis backup-now: there are no user servers")
		return 0
	}

	// The stopped worlds go first, so a felis-api that cannot take a backup is
	// found before any server is stopped for one.
	sort.SliceStable(worlds, func(i, j int) bool { return worlds[i].stopped && !worlds[j].stopped })
	width := 0
	for _, w := range worlds {
		width = max(width, len(w.name))
	}
	fmt.Fprintf(r.out, "felis backup-now: %d server(s), backed up one at a time:\n", len(worlds))
	stopping, work := false, 0
	for _, w := range worlds {
		fmt.Fprintf(r.out, "  %-*s  %-8s  %s\n", width, w.name, w.phase, backupNowAction(w, stop))
		stopping = stopping || (!w.stopped && stop)
		if (w.stopped && w.hasWorld) || (!w.stopped && stop) {
			work++
		}
	}
	if work > 0 {
		fmt.Fprintln(r.out, "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.")
	}
	if stopping {
		fmt.Fprintln(r.out, "Stopping disconnects the players on those servers, and they stay stopped afterwards.")
	}
	if !yes {
		if work == 0 {
			fmt.Fprintln(r.out, "Nothing to back up.")
		} else {
			fmt.Fprintln(r.out, "Nothing changed. Run again with -yes to back them up.")
		}
		return 0
	}

	var done, failed, running, empty int
	var leftStopped []string
	for _, w := range worlds {
		if err := ctx.Err(); err != nil {
			break
		}
		switch {
		case w.stopped && !w.hasWorld:
			empty++
			continue
		case !w.stopped && !stop:
			running++
			continue
		}
		if !w.stopped {
			halted, err := r.stopWorld(ctx, w.name)
			if halted {
				leftStopped = append(leftStopped, w.name)
			}
			if err != nil {
				fmt.Fprintf(r.out, "  %s: failed: %v\n", w.name, err)
				failed++
				continue
			}
		}
		err := r.backUp(ctx, w.name)
		if errors.Is(err, errBackupAPIUnreachable) {
			fmt.Fprintf(r.out, "  %s: failed: %v\n", w.name, err)
			fmt.Fprintln(r.out, "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).")
			failed++
			break
		}
		if err != nil {
			fmt.Fprintf(r.out, "  %s: failed: %v\n", w.name, err)
			failed++
			continue
		}
		done++
	}

	interrupted := ctx.Err() != nil
	if interrupted {
		fmt.Fprintln(r.out, "Interrupted: a backup Job already started runs to its end.")
	}
	parts := []string{fmt.Sprintf("%d backed up", done)}
	if failed > 0 {
		parts = append(parts, fmt.Sprintf("%d failed", failed))
	}
	if running > 0 {
		parts = append(parts, fmt.Sprintf("%d skipped (running)", running))
	}
	if empty > 0 {
		parts = append(parts, fmt.Sprintf("%d without a world", empty))
	}
	fmt.Fprintf(r.out, "%s.\n", strings.Join(parts, ", "))
	if len(leftStopped) > 0 {
		fmt.Fprintf(r.out, "Left stopped: %s. Start them from the panel when you are done.\n", strings.Join(leftStopped, ", "))
	}
	if done > 0 {
		fmt.Fprintln(r.out, "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.")
	}
	if failed > 0 || running > 0 || interrupted {
		return 1
	}
	return 0
}

// stopWorld stops one server and waits until the backup route would admit it.
// halted is whether the stop was asked for: the server then stays stopped, even
// when it takes longer than the wait.
func (r *backupNowRun) stopWorld(ctx context.Context, name string) (halted bool, err error) {
	fmt.Fprintf(r.out, "  %s: stopping\n", name)
	if err := r.halt(ctx, name); err != nil {
		return false, err
	}
	start := r.now()
	for {
		ok, err := r.stopped(ctx, name)
		if err != nil {
			return true, err
		}
		if ok {
			fmt.Fprintf(r.out, "  %s: stopped after %s\n", name, r.now().Sub(start).Round(time.Second))
			return true, nil
		}
		if r.now().Sub(start) >= backupNowStopWait {
			return true, fmt.Errorf("did not stop within %s (kubectl -n %s describe minecraftserver %s)", backupNowStopWait, r.ns, name)
		}
		if err := ctx.Err(); err != nil {
			return true, err
		}
		r.sleep(ctx, backupNowPoll)
	}
}

// backUp requests one world's backup and waits for its Job to finish. The Job is
// the one of this server that was not there before the request.
func (r *backupNowRun) backUp(ctx context.Context, name string) error {
	before, err := r.jobs(ctx, name)
	if err != nil {
		return fmt.Errorf("list its backup Jobs: %w", err)
	}
	known := make(map[string]bool, len(before))
	for _, j := range before {
		known[j.Name] = true
	}
	if err := r.request(ctx, name); err != nil {
		return err
	}
	fmt.Fprintf(r.out, "  %s: backing up\n", name)
	start := r.now()
	seen := ""
	for {
		jobs, err := r.jobs(ctx, name)
		if err != nil {
			return fmt.Errorf("list its backup Jobs: %w", err)
		}
		var job *api.AsyncJob
		for i := range jobs {
			if jobs[i].Kind == "backup" && (jobs[i].Name == seen || seen == "" && !known[jobs[i].Name]) {
				job = &jobs[i]
				break
			}
		}
		switch {
		case job == nil && seen != "":
			return fmt.Errorf("its backup Job %s was deleted before it finished", seen)
		case job == nil && r.now().Sub(start) >= backupNowJobAppear:
			return fmt.Errorf("felis-api accepted the backup, but no backup Job appeared within %s (kubectl -n %s get jobs)", backupNowJobAppear, r.ns)
		case job != nil && job.State == "succeeded":
			fmt.Fprintf(r.out, "  %s: archived in %s\n", name, r.now().Sub(start).Round(time.Second))
			return nil
		case job != nil && job.State == "failed":
			msg := job.Message
			if msg == "" {
				msg = "the backup Job failed"
			}
			return fmt.Errorf("%s (kubectl -n %s logs job/%s)", msg, r.ns, job.Name)
		case job != nil:
			seen = job.Name
		}
		if err := ctx.Err(); err != nil {
			return err
		}
		r.sleep(ctx, backupNowPoll)
	}
}
+563 −0

File added.

Preview size limit exceeded, changes collapsed.

+5 −0
Changes for cmd/felis/backupnow_test.go: 5 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -3,6 +3,7 @@ package main
import (
	"context"
	"encoding/json"
	"errors"
	"fmt"
	"io"
	"net/http"
@@ -152,6 +153,10 @@ func TestRequestBackup(t *testing.T) {
		if err == nil || !strings.Contains(err.Error(), "unreachable") {
			t.Fatalf("err = %v, want an 'unreachable' transport error", err)
		}
		// backup-now ends its run on this error alone.
		if !errors.Is(err, errBackupAPIUnreachable) {
			t.Fatalf("err = %v, want it to wrap errBackupAPIUnreachable", err)
		}
	})
}

+2 −0
Changes for cmd/felis/run.go: 2 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -20,6 +20,7 @@ Commands:
  reaper            Run the world reaper / backup batch
  restore           Extract a world archive into a world volume (internal Job entrypoint)
  backup            Archive a world into the backup store and record it (internal Job entrypoint)
  backup-now        Archive every user server's world now, one at a time (or the named ones; -stop stops running ones first; prints the plan, -yes applies; requires root/sudo)
  files             List/read/write one file in a stopped server's world (internal Job entrypoint)
  egress-gate       Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
  fetch-context     Fetch and extract a submission's build context (internal Job entrypoint)
@@ -60,6 +61,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
	"reaper":             cmdReaper,
	"restore":            cmdRestore,
	"backup":             cmdBackup,
	"backup-now":         cmdBackupNow,
	"files":              cmdFiles,
	"egress-gate":        cmdEgressGate,
	"fetch-context":      cmdFetchContext,
+135 −0
Changes for docs/operations.md: 135 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -262,6 +262,95 @@ its threshold, and §13b covers a full disk. On a host that built its images,
`docker builder prune -af` (with Docker started) reclaims the build cache when space is
short; the next upgrade rebuilds it.

### Growing the disk

Everything above shares the root filesystem, so more room means a bigger root
filesystem. It grows in place, with everything running: enlarge the virtual disk at the
provider, then the partition and the filesystem on it.

```bash
sudo felis backup-now -yes                      # a mistyped partition number is how a resize loses a disk
lsblk -f                                        # which disk and partition hold /, and whether LVM sits on it
sudo growpart /dev/vda 3                        # cloud-utils-growpart (RHEL) / cloud-guest-utils (Debian, Ubuntu)
# LVM (the RHEL-family default):
sudo pvresize /dev/vda3
sudo lvextend -r -l +100%FREE /dev/<vg>/root    # -r grows the filesystem with it
# no LVM:
sudo xfs_growfs /                               # xfs
sudo resize2fs /dev/vda3                        # ext4
df -h /
```

`felis backup-now` (troubleshooting.md §10) archives every stopped world; add `-stop` to
include the running ones.

### Moving the data to its own disk [VM-VERIFIED]

The bulk lives under `/var/lib/rancher/k3s`: the worlds, the world archives, the registry
and the images. On a disk of its own it grows without touching the system, and a full
world store leaves the root filesystem alone. The database and its bundles
(`/var/lib/felis`) are small and stay on the root disk. The move takes the platform down
for the copy plus a minute or two: the drill copied 4.2 GB in 18 s, and felis-api answered
`/readyz` 14 s after k3s started on the new disk.

1. Attach the disk and put a filesystem on it (the whole disk; `lsblk` shows it empty):

   ```bash
   sudo mkfs.xfs /dev/vdb
   U=$(sudo blkid -s UUID -o value /dev/vdb)
   ```

2. Archive every world, stopping the servers so each one saves, and keep the watchdog
   quiet for the next hour (the marker the installer writes: no mail, no failure pings
   to the heartbeat, until the time in it):

   ```bash
   sudo felis backup-now -yes -stop
   sudo install -d -m 0755 /run/felis
   echo $(( $(date +%s) + 3600 )) | sudo tee /run/felis/watchdog-quiet-until
   ```

3. Stop k3s and copy:

   ```bash
   sudo systemctl stop k3s
   sudo /usr/local/bin/k3s-killall.sh       # the containers k3s leaves running, and their mounts
   sudo mkdir -p /mnt/felis-data
   sudo mount UUID=$U /mnt/felis-data
   sudo rsync -aHAX --numeric-ids /var/lib/rancher/k3s/ /mnt/felis-data/
   sudo umount /mnt/felis-data
   ```

   `-X` carries the SELinux labels k3s set itself. Leave `restorecon` out: it would reset
   runc and the CNI binaries from `container_runtime_exec_t` to the policy default.

4. Mount it in place, and tie k3s to the mount:

   ```bash
   sudo mv /var/lib/rancher/k3s /var/lib/rancher/k3s.old
   sudo mkdir /var/lib/rancher/k3s
   echo "UUID=$U /var/lib/rancher/k3s xfs defaults,nofail 0 0" | sudo tee -a /etc/fstab
   sudo mkdir -p /etc/systemd/system/k3s.service.d
   printf '[Unit]\nRequiresMountsFor=/var/lib/rancher/k3s\n' | sudo tee /etc/systemd/system/k3s.service.d/data-disk.conf
   sudo systemctl daemon-reload
   sudo mount /var/lib/rancher/k3s
   sudo systemctl start k3s
   ```

   The drop-in is what keeps the data safe: k3s started on the empty mount point creates
   a new, empty cluster there. With it, a disk that does not come up fails the start with
   `A dependency job for k3s.service failed`, and `nofail` keeps the host booting so you
   can reach it. In the drill a detached disk left k3s inactive and the mount point empty;
   reattached, `systemctl start k3s` mounted it and started.

5. Check that `sudo k3s kubectl -n felis get pods` shows every pod ready and
   `findmnt /var/lib/rancher/k3s` names the new disk, then start the servers from the
   panel and `sudo rm /run/felis/watchdog-quiet-until`. Once the host has run a day,
   `sudo rm -rf /var/lib/rancher/k3s.old` frees the root disk.

The watchdog already watches `/var/lib/rancher/k3s` as a filesystem of its own (its
`-disk-paths`), so the new disk's fill level is mailed like the root's.

## 3. Uninstall

`deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove
@@ -604,6 +693,52 @@ production install:
  non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
  or rely on the watchdog's mail.

### Moving to another host (planned)

A planned move is the rebuild of troubleshooting.md §16, with the old host still there to
hand over a copy that misses nothing. It needs the off-site bucket: that is how the world
archives reach the new host (§16 step 7). The platform is down from step 1 until the new
host serves.

1. **On the old host**, stop everything that changes a world, then send the last copy:

   ```bash
   sudo install -d -m 0755 /run/felis
   echo $(( $(date +%s) + 4 * 3600 )) | sudo tee /run/felis/watchdog-quiet-until
   sudo systemctl stop felis-velocity                                   # no joins, so no server wakes
   sudo felis backup-now -yes -stop                                     # every world archived; the servers stay stopped
   sudo k3s kubectl -n felis scale deploy/felis-operator --replicas=0   # nothing starts a server from here on
   sudo felis db backup                                                 # a bundle that lists those archives
   sudo systemctl start felis-offsite.service
   sudo felis offsite status                                            # again until nothing waits
   ```

   The order matters. The new host fetches the archives its restored database lists, so
   the bundle comes after the last archive. The operator goes after `backup-now`, which
   needs it to stop the servers. The quiet marker keeps the watchdog from mailing the
   owners about the stopped proxy and operator for the next 4 hours.
2. **On the new host**, follow troubleshooting.md §16 "Rebuild on a new host" from step 1;
   `fetch-db latest` picks the bundle the old host just sent. Step 8 (`felis offsite
   take-over`) makes the new host the one that writes the bucket, and from then on the
   old host copies nothing more. Steps 10 and 11 move the names and the tunnel.
3. **Check the new host** before announcing it: sign in with an email code, restore one
   world and join it, and see `sudo felis offsite status` show a recent `last success` and
   no stand-by notice.
4. **Retire the old host.** It holds the last copy of every world outside the bucket, so
   keep it powered off with its disk for a few days first, disabled so a boot brings
   nothing up:

   ```bash
   sudo systemctl disable k3s felis-velocity felis-watchdog.timer felis-offsite.timer \
       felis-db-backup.timer felis-update-check.timer felis-build-tools.timer
   sudo poweroff
   ```

   Then uninstall it (§3) or wipe it.

Each step is covered where it is documented (backup-now in troubleshooting.md §10, the
rebuild in §16); the sequence as a whole has not been rehearsed as one move.

## 6. Changing the root domain [VM-VERIFIED] [GO-TESTED] [SH-TESTED]

The root domain is written into more places than the installer's config: the panel
Loading