feat(backup-now): 一次备份全部世界的命令,补扩盘/数据盘与计划内迁机指南
This commit is contained in:
6 files changed
+1208
-2
No files matched your search
+453
-2
@@ -4,15 +4,28 @@ import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/signal"
|
||||
"sort"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
@@ -22,7 +35,9 @@ import (
|
||||
// (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console
|
||||
// cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth)
|
||||
// while the API is alive, and the API renders the Job and audits the action. This file
|
||||
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue.
|
||||
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. The
|
||||
// `felis backup-now` command (cmdBackupNow, below) takes the same route for every
|
||||
// user server in turn.
|
||||
|
||||
// backupNowOutcome is the durable result of a backup request, re-printed after the TUI
|
||||
// alt-screen tears down.
|
||||
@@ -75,7 +90,7 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o
|
||||
|
||||
resp, err := hc.Do(req)
|
||||
if err != nil {
|
||||
return backupNowOutcome{}, fmt.Errorf("felis-api unreachable (a backup needs it alive): %w", err)
|
||||
return backupNowOutcome{}, fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
@@ -145,3 +160,439 @@ func backupPickable(servers []haltableServer) []haltableServer {
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// cmdBackupNow is `felis backup-now`: the world of every user server (or of the
|
||||
// named ones) archived now, one at a time, through the internal backup route the
|
||||
// console's Sync uses. A world lives only in its volume and the off-site copy holds
|
||||
// only its archives, so this is the lever in front of a planned move to another
|
||||
// host, a disk swap or anything else that could lose a volume.
|
||||
func cmdBackupNow(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("backup-now", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
|
||||
yes := fs.Bool("yes", false, "back up; without it the plan is printed and nothing changes")
|
||||
stop := fs.Bool("stop", false, "stop the running servers first: their players are disconnected and the servers stay stopped")
|
||||
fs.Usage = func() {
|
||||
fmt.Fprintln(stderr, "Usage: felis backup-now [-yes] [-stop] [server ...]")
|
||||
fmt.Fprintln(stderr)
|
||||
fmt.Fprintln(stderr, "Archives the world of every user server, or of the named ones, one at a time, and waits for each archive.")
|
||||
fmt.Fprintln(stderr, "A running server is skipped unless -stop is given. Without -yes it prints what it would do.")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis backup-now: refused — it reads the cluster's ops token, so it must run as root (try: sudo felis backup-now)")
|
||||
return 1
|
||||
}
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer cancel()
|
||||
|
||||
ns := cfg.K8s.Namespace
|
||||
osUser := accountableOSUser()
|
||||
jobs := api.NewK8sJobStatus(cl, ns)
|
||||
var baseURL, token string
|
||||
var repo ownerStore
|
||||
var drv *store.PostgresDriver
|
||||
defer func() {
|
||||
if drv != nil {
|
||||
_ = drv.Close()
|
||||
}
|
||||
}()
|
||||
hc := &http.Client{Timeout: 10 * time.Second}
|
||||
r := backupNowRun{
|
||||
out: stdout,
|
||||
errw: stderr,
|
||||
ns: ns,
|
||||
list: func(ctx context.Context) ([]backupNowWorld, error) { return listBackupNowWorlds(ctx, cl, ns) },
|
||||
stopped: func(ctx context.Context, name string) (bool, error) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &ms); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return backupNowStopped(ctx, cl, &ms)
|
||||
},
|
||||
halt: func(ctx context.Context, name string) error {
|
||||
// The database is opened only once a server is to be stopped: the halt
|
||||
// is audited like the console's, and a run with nothing running needs
|
||||
// no more than the API.
|
||||
if repo == nil {
|
||||
d, err := store.Open(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("open the database for the audit log: %w", err)
|
||||
}
|
||||
drv, repo = d, api.NewPGRepo(d.DB())
|
||||
}
|
||||
out, err := performHalt(ctx, cl, repo, ns, name, osUser)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if out.auditErr != nil {
|
||||
fmt.Fprintf(stderr, "felis backup-now: the audit row for stopping %s was not written: %v\n", name, out.auditErr)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
request: func(ctx context.Context, name string) error {
|
||||
if baseURL == "" {
|
||||
var err error
|
||||
if baseURL, token, err = resolveInternalAPI(ctx, cl, platform.DefaultControlNamespace); err != nil {
|
||||
return fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
|
||||
}
|
||||
}
|
||||
_, err := requestBackup(ctx, hc, baseURL, token, name, osUser)
|
||||
return err
|
||||
},
|
||||
jobs: jobs.LatestJobs,
|
||||
now: time.Now,
|
||||
sleep: func(ctx context.Context, d time.Duration) { sleepCtx(ctx, d) },
|
||||
}
|
||||
return r.run(ctx, fs.Args(), *yes, *stop)
|
||||
}
|
||||
|
||||
// sleepCtx waits d or until ctx ends.
|
||||
func sleepCtx(ctx context.Context, d time.Duration) {
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
case <-t.C:
|
||||
}
|
||||
}
|
||||
|
||||
// backupNowWorld is one user server as backup-now sees it.
|
||||
type backupNowWorld struct {
|
||||
name string
|
||||
phase string // the observed phase, or the desired state before the operator reconciled it
|
||||
stopped bool // the backup route's stopped gate admits it
|
||||
hasWorld bool // its world volume exists
|
||||
}
|
||||
|
||||
// listBackupNowWorlds lists the user servers of namespace in the API server's
|
||||
// order (by name), each with what the backup route checks. System servers are
|
||||
// left out: they have no row in the servers table, so the route refuses them
|
||||
// (backupPickable).
|
||||
func listBackupNowWorlds(ctx context.Context, cl client.Client, namespace string) ([]backupNowWorld, error) {
|
||||
var list v1alpha1.MinecraftServerList
|
||||
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var out []backupNowWorld
|
||||
for i := range list.Items {
|
||||
ms := &list.Items[i]
|
||||
if isSystemServer(ms.Name) {
|
||||
continue
|
||||
}
|
||||
stopped, err := backupNowStopped(ctx, cl, ms)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var pvc corev1.PersistentVolumeClaim
|
||||
err = cl.Get(ctx, types.NamespacedName{Namespace: namespace, Name: naming.WorldPVCName(ms.Name)}, &pvc)
|
||||
if err != nil && !apierrors.IsNotFound(err) {
|
||||
return nil, fmt.Errorf("look up the world volume of %s: %w", ms.Name, err)
|
||||
}
|
||||
phase := string(ms.Status.Phase)
|
||||
if phase == "" {
|
||||
phase = string(ms.Spec.DesiredState)
|
||||
}
|
||||
out = append(out, backupNowWorld{name: ms.Name, phase: phase, stopped: stopped, hasWorld: err == nil})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// backupNowStopped is the backup route's stopped gate (api.enqueueBackup and
|
||||
// K8sCluster.AcquireMaintenance together): desired Stopped, not ready, phase
|
||||
// Stopped and no game pod left.
|
||||
func backupNowStopped(ctx context.Context, cl client.Client, ms *v1alpha1.MinecraftServer) (bool, error) {
|
||||
if ms.Spec.DesiredState != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
|
||||
return false, nil
|
||||
}
|
||||
var pods corev1.PodList
|
||||
if err := cl.List(ctx, &pods, client.InNamespace(ms.Namespace), client.MatchingLabels{
|
||||
v1alpha1.LabelServer: ms.Name, v1alpha1.LabelComponent: operator.ComponentValue,
|
||||
}); err != nil {
|
||||
return false, fmt.Errorf("look up the pod of %s: %w", ms.Name, err)
|
||||
}
|
||||
return len(pods.Items) == 0, nil
|
||||
}
|
||||
|
||||
// errBackupAPIUnreachable is a backup request that never reached felis-api. It
|
||||
// ends a backup-now run: every later world would fail the same way, and stopping
|
||||
// servers for backups that cannot be taken only takes them away from players.
|
||||
var errBackupAPIUnreachable = errors.New("felis-api unreachable (a backup needs it alive)")
|
||||
|
||||
// Polling of backup-now. A graceful stop saves the world first; the backup Job's
|
||||
// own deadline (backupjob, 30 minutes) ends a Job that hangs, so its wait needs no
|
||||
// cap of its own.
|
||||
const (
|
||||
backupNowPoll = 2 * time.Second
|
||||
backupNowStopWait = 10 * time.Minute
|
||||
backupNowJobAppear = time.Minute
|
||||
)
|
||||
|
||||
// backupNowRun is backup-now over seams, so the plan and the run are tested
|
||||
// without a cluster or felis-api.
|
||||
type backupNowRun struct {
|
||||
out, errw io.Writer
|
||||
ns string // where the servers and their Jobs live, for the kubectl hints
|
||||
list func(ctx context.Context) ([]backupNowWorld, error)
|
||||
stopped func(ctx context.Context, name string) (bool, error)
|
||||
halt func(ctx context.Context, name string) error
|
||||
request func(ctx context.Context, name string) error
|
||||
jobs func(ctx context.Context, name string) ([]api.AsyncJob, error)
|
||||
now func() time.Time
|
||||
sleep func(ctx context.Context, d time.Duration)
|
||||
}
|
||||
|
||||
// pickBackupNowWorlds narrows worlds to names, in the order given, or keeps them
|
||||
// all when names is empty.
|
||||
func pickBackupNowWorlds(worlds []backupNowWorld, names []string) ([]backupNowWorld, error) {
|
||||
if len(names) == 0 {
|
||||
return worlds, nil
|
||||
}
|
||||
byName := make(map[string]backupNowWorld, len(worlds))
|
||||
for _, w := range worlds {
|
||||
byName[w.name] = w
|
||||
}
|
||||
var out []backupNowWorld
|
||||
seen := map[string]bool{}
|
||||
for _, n := range names {
|
||||
if seen[n] {
|
||||
continue
|
||||
}
|
||||
seen[n] = true
|
||||
w, ok := byName[n]
|
||||
switch {
|
||||
case ok:
|
||||
out = append(out, w)
|
||||
case isSystemServer(n):
|
||||
return nil, fmt.Errorf("%s is a system server: its world is rebuilt by felis setup and has no backups", n)
|
||||
default:
|
||||
return nil, fmt.Errorf("no server named %q", n)
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// backupNowAction is what the plan does with a world.
|
||||
func backupNowAction(w backupNowWorld, stop bool) string {
|
||||
switch {
|
||||
case w.stopped && !w.hasWorld:
|
||||
return "skip: no world volume (never started, nothing to save)"
|
||||
case w.stopped:
|
||||
return "back up"
|
||||
case stop:
|
||||
return "stop, then back up"
|
||||
default:
|
||||
return "skip: running (stop it first, or pass -stop)"
|
||||
}
|
||||
}
|
||||
|
||||
func (r *backupNowRun) run(ctx context.Context, names []string, yes, stop bool) int {
|
||||
all, err := r.list(ctx)
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.errw, "felis backup-now: list the servers: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
worlds, err := pickBackupNowWorlds(all, names)
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.errw, "felis backup-now: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
if len(worlds) == 0 {
|
||||
fmt.Fprintln(r.out, "felis backup-now: there are no user servers")
|
||||
return 0
|
||||
}
|
||||
|
||||
// The stopped worlds go first, so a felis-api that cannot take a backup is
|
||||
// found before any server is stopped for one.
|
||||
sort.SliceStable(worlds, func(i, j int) bool { return worlds[i].stopped && !worlds[j].stopped })
|
||||
width := 0
|
||||
for _, w := range worlds {
|
||||
width = max(width, len(w.name))
|
||||
}
|
||||
fmt.Fprintf(r.out, "felis backup-now: %d server(s), backed up one at a time:\n", len(worlds))
|
||||
stopping, work := false, 0
|
||||
for _, w := range worlds {
|
||||
fmt.Fprintf(r.out, " %-*s %-8s %s\n", width, w.name, w.phase, backupNowAction(w, stop))
|
||||
stopping = stopping || (!w.stopped && stop)
|
||||
if (w.stopped && w.hasWorld) || (!w.stopped && stop) {
|
||||
work++
|
||||
}
|
||||
}
|
||||
if work > 0 {
|
||||
fmt.Fprintln(r.out, "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.")
|
||||
}
|
||||
if stopping {
|
||||
fmt.Fprintln(r.out, "Stopping disconnects the players on those servers, and they stay stopped afterwards.")
|
||||
}
|
||||
if !yes {
|
||||
if work == 0 {
|
||||
fmt.Fprintln(r.out, "Nothing to back up.")
|
||||
} else {
|
||||
fmt.Fprintln(r.out, "Nothing changed. Run again with -yes to back them up.")
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
var done, failed, running, empty int
|
||||
var leftStopped []string
|
||||
for _, w := range worlds {
|
||||
if err := ctx.Err(); err != nil {
|
||||
break
|
||||
}
|
||||
switch {
|
||||
case w.stopped && !w.hasWorld:
|
||||
empty++
|
||||
continue
|
||||
case !w.stopped && !stop:
|
||||
running++
|
||||
continue
|
||||
}
|
||||
if !w.stopped {
|
||||
halted, err := r.stopWorld(ctx, w.name)
|
||||
if halted {
|
||||
leftStopped = append(leftStopped, w.name)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
|
||||
failed++
|
||||
continue
|
||||
}
|
||||
}
|
||||
err := r.backUp(ctx, w.name)
|
||||
if errors.Is(err, errBackupAPIUnreachable) {
|
||||
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
|
||||
fmt.Fprintln(r.out, "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).")
|
||||
failed++
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
|
||||
failed++
|
||||
continue
|
||||
}
|
||||
done++
|
||||
}
|
||||
|
||||
interrupted := ctx.Err() != nil
|
||||
if interrupted {
|
||||
fmt.Fprintln(r.out, "Interrupted: a backup Job already started runs to its end.")
|
||||
}
|
||||
parts := []string{fmt.Sprintf("%d backed up", done)}
|
||||
if failed > 0 {
|
||||
parts = append(parts, fmt.Sprintf("%d failed", failed))
|
||||
}
|
||||
if running > 0 {
|
||||
parts = append(parts, fmt.Sprintf("%d skipped (running)", running))
|
||||
}
|
||||
if empty > 0 {
|
||||
parts = append(parts, fmt.Sprintf("%d without a world", empty))
|
||||
}
|
||||
fmt.Fprintf(r.out, "%s.\n", strings.Join(parts, ", "))
|
||||
if len(leftStopped) > 0 {
|
||||
fmt.Fprintf(r.out, "Left stopped: %s. Start them from the panel when you are done.\n", strings.Join(leftStopped, ", "))
|
||||
}
|
||||
if done > 0 {
|
||||
fmt.Fprintln(r.out, "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.")
|
||||
}
|
||||
if failed > 0 || running > 0 || interrupted {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// stopWorld stops one server and waits until the backup route would admit it.
|
||||
// halted is whether the stop was asked for: the server then stays stopped, even
|
||||
// when it takes longer than the wait.
|
||||
func (r *backupNowRun) stopWorld(ctx context.Context, name string) (halted bool, err error) {
|
||||
fmt.Fprintf(r.out, " %s: stopping\n", name)
|
||||
if err := r.halt(ctx, name); err != nil {
|
||||
return false, err
|
||||
}
|
||||
start := r.now()
|
||||
for {
|
||||
ok, err := r.stopped(ctx, name)
|
||||
if err != nil {
|
||||
return true, err
|
||||
}
|
||||
if ok {
|
||||
fmt.Fprintf(r.out, " %s: stopped after %s\n", name, r.now().Sub(start).Round(time.Second))
|
||||
return true, nil
|
||||
}
|
||||
if r.now().Sub(start) >= backupNowStopWait {
|
||||
return true, fmt.Errorf("did not stop within %s (kubectl -n %s describe minecraftserver %s)", backupNowStopWait, r.ns, name)
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return true, err
|
||||
}
|
||||
r.sleep(ctx, backupNowPoll)
|
||||
}
|
||||
}
|
||||
|
||||
// backUp requests one world's backup and waits for its Job to finish. The Job is
|
||||
// the one of this server that was not there before the request.
|
||||
func (r *backupNowRun) backUp(ctx context.Context, name string) error {
|
||||
before, err := r.jobs(ctx, name)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list its backup Jobs: %w", err)
|
||||
}
|
||||
known := make(map[string]bool, len(before))
|
||||
for _, j := range before {
|
||||
known[j.Name] = true
|
||||
}
|
||||
if err := r.request(ctx, name); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(r.out, " %s: backing up\n", name)
|
||||
start := r.now()
|
||||
seen := ""
|
||||
for {
|
||||
jobs, err := r.jobs(ctx, name)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list its backup Jobs: %w", err)
|
||||
}
|
||||
var job *api.AsyncJob
|
||||
for i := range jobs {
|
||||
if jobs[i].Kind == "backup" && (jobs[i].Name == seen || seen == "" && !known[jobs[i].Name]) {
|
||||
job = &jobs[i]
|
||||
break
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case job == nil && seen != "":
|
||||
return fmt.Errorf("its backup Job %s was deleted before it finished", seen)
|
||||
case job == nil && r.now().Sub(start) >= backupNowJobAppear:
|
||||
return fmt.Errorf("felis-api accepted the backup, but no backup Job appeared within %s (kubectl -n %s get jobs)", backupNowJobAppear, r.ns)
|
||||
case job != nil && job.State == "succeeded":
|
||||
fmt.Fprintf(r.out, " %s: archived in %s\n", name, r.now().Sub(start).Round(time.Second))
|
||||
return nil
|
||||
case job != nil && job.State == "failed":
|
||||
msg := job.Message
|
||||
if msg == "" {
|
||||
msg = "the backup Job failed"
|
||||
}
|
||||
return fmt.Errorf("%s (kubectl -n %s logs job/%s)", msg, r.ns, job.Name)
|
||||
case job != nil:
|
||||
seen = job.Name
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
r.sleep(ctx, backupNowPoll)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,563 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// bnRig drives backupNowRun against scripted servers and Jobs on a fake clock that
|
||||
// moves only when the run sleeps.
|
||||
type bnRig struct {
|
||||
out, errw bytes.Buffer
|
||||
clock time.Time
|
||||
events []string
|
||||
worlds []backupNowWorld
|
||||
// stopAfter is how many polls a halted server takes to stop; -1 never does.
|
||||
stopAfter map[string]int
|
||||
polls map[string]int
|
||||
reqErr map[string]error
|
||||
// states is what the server's new backup Job reports on each poll after the
|
||||
// request, the last one repeating; "" is no Job.
|
||||
states map[string][]string
|
||||
messages map[string]string
|
||||
jobPolls map[string]int
|
||||
// stopErr fails a server's stop polls; jobsFail fails the nth (1-based) Job
|
||||
// list of a server.
|
||||
stopErr map[string]error
|
||||
jobsFail map[string]int
|
||||
jobCalls map[string]int
|
||||
// ctx is the run's context, and onAct runs after each halt and request.
|
||||
ctx context.Context
|
||||
onAct func(event string)
|
||||
}
|
||||
|
||||
func newBNRig(worlds ...backupNowWorld) *bnRig {
|
||||
return &bnRig{
|
||||
clock: time.Unix(1_800_000_000, 0),
|
||||
worlds: worlds,
|
||||
stopAfter: map[string]int{},
|
||||
polls: map[string]int{},
|
||||
reqErr: map[string]error{},
|
||||
states: map[string][]string{},
|
||||
messages: map[string]string{},
|
||||
jobPolls: map[string]int{},
|
||||
stopErr: map[string]error{},
|
||||
jobsFail: map[string]int{},
|
||||
jobCalls: map[string]int{},
|
||||
ctx: context.Background(),
|
||||
onAct: func(string) {},
|
||||
}
|
||||
}
|
||||
|
||||
func (g *bnRig) run(names []string, yes, stop bool) int {
|
||||
r := backupNowRun{
|
||||
out: &g.out, errw: &g.errw, ns: "minecraft",
|
||||
list: func(context.Context) ([]backupNowWorld, error) {
|
||||
return append([]backupNowWorld(nil), g.worlds...), nil
|
||||
},
|
||||
stopped: func(_ context.Context, name string) (bool, error) {
|
||||
g.polls[name]++
|
||||
if err := g.stopErr[name]; err != nil {
|
||||
return false, err
|
||||
}
|
||||
n := g.stopAfter[name]
|
||||
return n >= 0 && g.polls[name] > n, nil
|
||||
},
|
||||
halt: func(_ context.Context, name string) error {
|
||||
g.events = append(g.events, "halt "+name)
|
||||
g.onAct("halt " + name)
|
||||
return nil
|
||||
},
|
||||
request: func(_ context.Context, name string) error {
|
||||
g.events = append(g.events, "request "+name)
|
||||
g.onAct("request " + name)
|
||||
if err := g.reqErr[name]; err != nil {
|
||||
return err
|
||||
}
|
||||
g.jobPolls[name] = 0
|
||||
return nil
|
||||
},
|
||||
jobs: func(_ context.Context, name string) ([]api.AsyncJob, error) {
|
||||
// Every server has an older finished backup, and a restore Job that
|
||||
// shows up with the new backup: neither is the Job to wait for.
|
||||
g.jobCalls[name]++
|
||||
if g.jobCalls[name] == g.jobsFail[name] {
|
||||
return nil, errors.New("the apiserver is gone")
|
||||
}
|
||||
out := []api.AsyncJob{{Name: "backup-" + name + "-old", Kind: "backup", State: "succeeded"}}
|
||||
n, requested := g.jobPolls[name]
|
||||
if !requested {
|
||||
return out, nil
|
||||
}
|
||||
g.jobPolls[name] = n + 1
|
||||
states := g.states[name]
|
||||
if len(states) == 0 {
|
||||
states = []string{"succeeded"}
|
||||
}
|
||||
state := states[min(n, len(states)-1)]
|
||||
if state == "" {
|
||||
return out, nil
|
||||
}
|
||||
return append([]api.AsyncJob{
|
||||
{Name: "restore-" + name + "-x", Kind: "restore", State: "succeeded"},
|
||||
{Name: "backup-" + name + "-new", Kind: "backup", State: state, Message: g.messages[name]},
|
||||
}, out...), nil
|
||||
},
|
||||
now: func() time.Time { return g.clock },
|
||||
sleep: func(_ context.Context, d time.Duration) { g.clock = g.clock.Add(d) },
|
||||
}
|
||||
return r.run(g.ctx, names, yes, stop)
|
||||
}
|
||||
|
||||
func stoppedWorld(name string) backupNowWorld {
|
||||
return backupNowWorld{name: name, phase: "Stopped", stopped: true, hasWorld: true}
|
||||
}
|
||||
|
||||
func runningWorld(name string) backupNowWorld {
|
||||
return backupNowWorld{name: name, phase: "Running", hasWorld: true}
|
||||
}
|
||||
|
||||
func emptyWorld(name string) backupNowWorld {
|
||||
return backupNowWorld{name: name, phase: "Stopped", stopped: true}
|
||||
}
|
||||
|
||||
const (
|
||||
bnManualKeep = "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.\n"
|
||||
bnStopping = "Stopping disconnects the players on those servers, and they stay stopped afterwards.\n"
|
||||
bnOffsite = "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.\n"
|
||||
)
|
||||
|
||||
func TestBackupNowPlanChangesNothing(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
stop bool
|
||||
want string
|
||||
}{
|
||||
{false, "felis backup-now: 4 server(s), backed up one at a time:\n" +
|
||||
" alpha Stopped back up\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running skip: running (stop it first, or pass -stop)\n" +
|
||||
" delta Starting skip: running (stop it first, or pass -stop)\n" +
|
||||
bnManualKeep +
|
||||
"Nothing changed. Run again with -yes to back them up.\n"},
|
||||
{true, "felis backup-now: 4 server(s), backed up one at a time:\n" +
|
||||
" alpha Stopped back up\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running stop, then back up\n" +
|
||||
" delta Starting stop, then back up\n" +
|
||||
bnManualKeep + bnStopping +
|
||||
"Nothing changed. Run again with -yes to back them up.\n"},
|
||||
} {
|
||||
t.Run(fmt.Sprintf("stop=%v", tc.stop), func(t *testing.T) {
|
||||
delta := runningWorld("delta")
|
||||
delta.phase = "Starting"
|
||||
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), delta)
|
||||
if code := g.run(nil, false, tc.stop); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0; stderr %q", code, g.errw.String())
|
||||
}
|
||||
if g.out.String() != tc.want {
|
||||
t.Fatalf("plan =\n%s\nwant\n%s", g.out.String(), tc.want)
|
||||
}
|
||||
if len(g.events) != 0 || len(g.polls) != 0 {
|
||||
t.Fatalf("the plan acted: events %v, polls %v", g.events, g.polls)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A plan that saves nothing says so, without the manual_keep warning; a running
|
||||
// server counts as something to save once -stop is given.
|
||||
func TestBackupNowPlanWithNothingToSave(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
stop bool
|
||||
want string
|
||||
}{
|
||||
{false, "felis backup-now: 2 server(s), backed up one at a time:\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running skip: running (stop it first, or pass -stop)\n" +
|
||||
"Nothing to back up.\n"},
|
||||
{true, "felis backup-now: 2 server(s), backed up one at a time:\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running stop, then back up\n" +
|
||||
bnManualKeep + bnStopping +
|
||||
"Nothing changed. Run again with -yes to back them up.\n"},
|
||||
} {
|
||||
g := newBNRig(runningWorld("bravo"), emptyWorld("charlie"))
|
||||
if code := g.run(nil, false, tc.stop); code != 0 {
|
||||
t.Fatalf("stop=%v: exit = %d, want 0", tc.stop, code)
|
||||
}
|
||||
if g.out.String() != tc.want {
|
||||
t.Fatalf("stop=%v: plan =\n%s\nwant\n%s", tc.stop, g.out.String(), tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowBacksUpEachWorldInTurn(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), stoppedWorld("delta"))
|
||||
g.states["alpha"] = []string{"", "running", "succeeded"}
|
||||
g.states["delta"] = []string{"running", "failed"}
|
||||
g.messages["delta"] = "felis backup: not enough free disk for the archive"
|
||||
g.stopAfter["bravo"] = 2
|
||||
g.states["bravo"] = []string{"running", "running", "running", "succeeded"}
|
||||
|
||||
code := g.run(nil, true, true)
|
||||
if code != 1 {
|
||||
t.Fatalf("exit = %d, want 1 (delta failed)", code)
|
||||
}
|
||||
if want := []string{"request alpha", "request delta", "halt bravo", "request bravo"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping+"")
|
||||
want := " alpha: backing up\n" +
|
||||
" alpha: archived in 4s\n" +
|
||||
" delta: backing up\n" +
|
||||
" delta: failed: felis backup: not enough free disk for the archive (kubectl -n minecraft logs job/backup-delta-new)\n" +
|
||||
" bravo: stopping\n" +
|
||||
" bravo: stopped after 4s\n" +
|
||||
" bravo: backing up\n" +
|
||||
" bravo: archived in 6s\n" +
|
||||
"2 backed up, 1 failed, 1 without a world.\n" +
|
||||
"Left stopped: bravo. Start them from the panel when you are done.\n" +
|
||||
bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowSkipsRunningServersWithoutStop(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"))
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1 (bravo was not backed up)", code)
|
||||
}
|
||||
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
if len(g.polls) != 0 {
|
||||
t.Fatalf("polled a server it did not stop: %v", g.polls)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), "Nothing changed")
|
||||
if run != "" {
|
||||
t.Fatalf("-yes printed the plan's closing line")
|
||||
}
|
||||
_, run, _ = strings.Cut(g.out.String(), bnManualKeep)
|
||||
want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 skipped (running).\n" + bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowExitsCleanWhenEverythingIsSaved(t *testing.T) {
|
||||
// -stop with nothing running stops nothing and says nothing about stopping.
|
||||
g := newBNRig(stoppedWorld("alpha"), emptyWorld("charlie"))
|
||||
if code := g.run(nil, true, true); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0; output\n%s", code, g.out.String())
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
|
||||
if want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 without a world.\n" + bnOffsite; run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
|
||||
g = newBNRig(emptyWorld("charlie"))
|
||||
if code := g.run(nil, true, false); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0 for a server with nothing to save", code)
|
||||
}
|
||||
// Nothing to archive: no manual_keep warning, and no off-site hint.
|
||||
if want := "felis backup-now: 1 server(s), backed up one at a time:\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
"0 backed up, 1 without a world.\n"; g.out.String() != want {
|
||||
t.Fatalf("output =\n%s\nwant\n%s", g.out.String(), want)
|
||||
}
|
||||
if len(g.events) != 0 {
|
||||
t.Fatalf("events = %v, want none", g.events)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowStopsAtAnUnreachableAPI(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), runningWorld("charlie"))
|
||||
g.reqErr["alpha"] = fmt.Errorf("%w: dial tcp 10.43.0.9:8081: connect: connection refused", errBackupAPIUnreachable)
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v: nothing after the API proved unreachable, and no server stopped", g.events, want)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
want := " alpha: failed: felis-api unreachable (a backup needs it alive): dial tcp 10.43.0.9:8081: connect: connection refused\n" +
|
||||
"Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).\n" +
|
||||
"0 backed up, 1 failed.\n"
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
|
||||
// Any other refusal is that world's alone: the run goes on.
|
||||
g = newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
|
||||
g.reqErr["alpha"] = errors.New("felis-api: the world is being restored")
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"request alpha", "request bravo"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowGivesUpOnAServerThatDoesNotStop(t *testing.T) {
|
||||
g := newBNRig(runningWorld("bravo"), runningWorld("echo"))
|
||||
g.stopAfter["bravo"] = -1
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"halt bravo", "halt echo", "request echo"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
// One poll at the start and one per 2s sleep up to the 10-minute mark.
|
||||
if g.polls["bravo"] != 301 {
|
||||
t.Fatalf("bravo polled %d times, want 301", g.polls["bravo"])
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
want := " bravo: stopping\n" +
|
||||
" bravo: failed: did not stop within 10m0s (kubectl -n minecraft describe minecraftserver bravo)\n" +
|
||||
" echo: stopping\n" +
|
||||
" echo: stopped after 0s\n" +
|
||||
" echo: backing up\n" +
|
||||
" echo: archived in 0s\n" +
|
||||
"1 backed up, 1 failed.\n" +
|
||||
"Left stopped: bravo, echo. Start them from the panel when you are done.\n" +
|
||||
bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowReportsAJobThatNeverRuns(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
states []string
|
||||
polls int
|
||||
want string
|
||||
}{
|
||||
{"never appears", []string{""}, 31,
|
||||
"felis-api accepted the backup, but no backup Job appeared within 1m0s (kubectl -n minecraft get jobs)"},
|
||||
{"deleted while running", []string{"running", "running", ""}, 3,
|
||||
"its backup Job backup-alpha-new was deleted before it finished"},
|
||||
{"fails without a message", []string{"failed"}, 1,
|
||||
"the backup Job failed (kubectl -n minecraft logs job/backup-alpha-new)"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
g.states["alpha"] = tc.states
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if !strings.Contains(g.out.String(), " alpha: failed: "+tc.want+"\n") {
|
||||
t.Fatalf("output =\n%s\nwant the line %q", g.out.String(), tc.want)
|
||||
}
|
||||
if g.jobPolls["alpha"] != tc.polls {
|
||||
t.Fatalf("polled the Jobs %d times after the request, want %d", g.jobPolls["alpha"], tc.polls)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowNamedServers(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), stoppedWorld("delta"))
|
||||
if code := g.run([]string{"delta", "alpha", "delta"}, true, false); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0", code)
|
||||
}
|
||||
if want := []string{"request delta", "request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
if !strings.HasPrefix(g.out.String(), "felis backup-now: 2 server(s), backed up one at a time:\n delta Stopped back up\n alpha Stopped back up\n") {
|
||||
t.Fatalf("plan =\n%s", g.out.String())
|
||||
}
|
||||
|
||||
for _, tc := range []struct{ name, want string }{
|
||||
{"login", "felis backup-now: login is a system server: its world is rebuilt by felis setup and has no backups\n"},
|
||||
{"lobby", "felis backup-now: lobby is a system server: its world is rebuilt by felis setup and has no backups\n"},
|
||||
{"nope", "felis backup-now: no server named \"nope\"\n"},
|
||||
} {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
if code := g.run([]string{"alpha", tc.name}, true, false); code != 2 {
|
||||
t.Fatalf("%s: exit = %d, want 2", tc.name, code)
|
||||
}
|
||||
if g.errw.String() != tc.want {
|
||||
t.Fatalf("%s: stderr = %q, want %q", tc.name, g.errw.String(), tc.want)
|
||||
}
|
||||
if len(g.events) != 0 || g.out.Len() != 0 {
|
||||
t.Fatalf("%s: acted on a bad name: events %v, output %q", tc.name, g.events, g.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
g = newBNRig()
|
||||
if code := g.run(nil, true, false); code != 0 || g.out.String() != "felis backup-now: there are no user servers\n" {
|
||||
t.Fatalf("empty fleet: exit %d, output %q", code, g.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The world list mirrors the backup route's own gate, so the plan says exactly what
|
||||
// the route would refuse.
|
||||
func TestListBackupNowWorlds(t *testing.T) {
|
||||
withStatus := func(ms *v1alpha1.MinecraftServer, ready bool) *v1alpha1.MinecraftServer {
|
||||
ms.Status.Ready = ready
|
||||
return ms
|
||||
}
|
||||
pvc := func(server, ns string) *corev1.PersistentVolumeClaim {
|
||||
return &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: naming.WorldPVCName(server), Namespace: ns}}
|
||||
}
|
||||
pod := func(name, ns string, labels map[string]string) *corev1.Pod {
|
||||
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: labels}}
|
||||
}
|
||||
gameLabels := func(server string) map[string]string {
|
||||
return map[string]string{v1alpha1.LabelServer: server, v1alpha1.LabelComponent: "server"}
|
||||
}
|
||||
other := mcServer("zulu", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
|
||||
other.Namespace = "elsewhere"
|
||||
hotel := mcServer("hotel", v1alpha1.DesiredStopped, "")
|
||||
objs := []client.Object{
|
||||
mcServer("alpha", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("alpha", haltNS),
|
||||
// A backup Job's pod carries the server label with its own component, and
|
||||
// a game pod of the same name in another namespace is somebody else's.
|
||||
pod("backup-alpha-1-x", haltNS, map[string]string{v1alpha1.LabelServer: "alpha", "app.kubernetes.io/component": "world-backup"}),
|
||||
pod("alpha-0", "elsewhere", gameLabels("alpha")),
|
||||
withStatus(mcServer("bravo", v1alpha1.DesiredRunning, v1alpha1.PhaseRunning), true), pvc("bravo", haltNS), pod("bravo-0", haltNS, gameLabels("bravo")),
|
||||
mcServer("charlie", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped),
|
||||
mcServer("delta", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("delta", haltNS), pod("delta-0", haltNS, gameLabels("delta")),
|
||||
mcServer("echo", v1alpha1.DesiredStopped, v1alpha1.PhaseStopping), pvc("echo", haltNS),
|
||||
mcServer("foxtrot", v1alpha1.DesiredRunning, v1alpha1.PhaseStopped), pvc("foxtrot", haltNS),
|
||||
withStatus(mcServer("golf", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), true), pvc("golf", haltNS),
|
||||
hotel, pvc("hotel", haltNS),
|
||||
mcServer("login", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("login", haltNS),
|
||||
mcServer("lobby", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("lobby", haltNS),
|
||||
other, pvc("zulu", "elsewhere"),
|
||||
}
|
||||
got, err := listBackupNowWorlds(context.Background(), haltClient(t, objs...), haltNS)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := []backupNowWorld{
|
||||
{name: "alpha", phase: "Stopped", stopped: true, hasWorld: true},
|
||||
{name: "bravo", phase: "Running", hasWorld: true},
|
||||
{name: "charlie", phase: "Stopped", stopped: true},
|
||||
{name: "delta", phase: "Stopped", hasWorld: true}, // its pod is still going
|
||||
{name: "echo", phase: "Stopping", hasWorld: true}, // still stopping
|
||||
{name: "foxtrot", phase: "Stopped", hasWorld: true}, // asked to start
|
||||
{name: "golf", phase: "Stopped", hasWorld: true}, // still reports ready
|
||||
{name: "hotel", phase: "Stopped", hasWorld: true}, // not reconciled yet: the desired state shows
|
||||
}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("worlds =\n%+v\nwant\n%+v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowStopsWhenInterrupted(t *testing.T) {
|
||||
t.Run("between worlds", func(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
g.ctx = ctx
|
||||
g.onAct = func(e string) {
|
||||
if e == "request alpha" {
|
||||
cancel()
|
||||
}
|
||||
}
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
|
||||
want := " alpha: backing up\n alpha: archived in 0s\n" +
|
||||
"Interrupted: a backup Job already started runs to its end.\n" +
|
||||
"1 backed up.\n" + bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
})
|
||||
t.Run("while a Job runs", func(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
g.states["alpha"] = []string{"running"}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
g.ctx = ctx
|
||||
g.onAct = func(string) { cancel() }
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if g.jobPolls["alpha"] != 1 {
|
||||
t.Fatalf("polled the Job %d times after the interrupt, want 1", g.jobPolls["alpha"])
|
||||
}
|
||||
if !strings.Contains(g.out.String(), " alpha: failed: context canceled\nInterrupted: a backup Job already started runs to its end.\n0 backed up, 1 failed.\n") {
|
||||
t.Fatalf("output =\n%s", g.out.String())
|
||||
}
|
||||
})
|
||||
t.Run("while a server stops", func(t *testing.T) {
|
||||
g := newBNRig(runningWorld("bravo"))
|
||||
g.stopAfter["bravo"] = -1
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
g.ctx = ctx
|
||||
g.onAct = func(string) { cancel() }
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if g.polls["bravo"] != 1 {
|
||||
t.Fatalf("polled bravo %d times after the interrupt, want 1", g.polls["bravo"])
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
want := " bravo: stopping\n bravo: failed: context canceled\n" +
|
||||
"Interrupted: a backup Job already started runs to its end.\n" +
|
||||
"0 backed up, 1 failed.\n" +
|
||||
"Left stopped: bravo. Start them from the panel when you are done.\n"
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestBackupNowReportsClusterErrors(t *testing.T) {
|
||||
g := newBNRig(runningWorld("bravo"))
|
||||
g.stopErr["bravo"] = errors.New("the apiserver is gone")
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
// The stop was asked for, so the server stays stopped whatever the poll said.
|
||||
want := " bravo: stopping\n bravo: failed: the apiserver is gone\n0 backed up, 1 failed.\n" +
|
||||
"Left stopped: bravo. Start them from the panel when you are done.\n"
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
|
||||
for _, tc := range []struct {
|
||||
call int
|
||||
events []string
|
||||
}{
|
||||
{1, nil}, // before the request: nothing is asked for
|
||||
{2, []string{"request alpha"}}, // the first poll after it
|
||||
} {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
g.jobsFail["alpha"] = tc.call
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("call %d: exit = %d, want 1", tc.call, code)
|
||||
}
|
||||
if !reflect.DeepEqual(g.events, tc.events) {
|
||||
t.Fatalf("call %d: events = %v, want %v", tc.call, g.events, tc.events)
|
||||
}
|
||||
if !strings.Contains(g.out.String(), " alpha: failed: list its backup Jobs: the apiserver is gone\n") {
|
||||
t.Fatalf("call %d: output =\n%s", tc.call, g.out.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package main
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
@@ -152,6 +153,10 @@ func TestRequestBackup(t *testing.T) {
|
||||
if err == nil || !strings.Contains(err.Error(), "unreachable") {
|
||||
t.Fatalf("err = %v, want an 'unreachable' transport error", err)
|
||||
}
|
||||
// backup-now ends its run on this error alone.
|
||||
if !errors.Is(err, errBackupAPIUnreachable) {
|
||||
t.Fatalf("err = %v, want it to wrap errBackupAPIUnreachable", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -20,6 +20,7 @@ Commands:
|
||||
reaper Run the world reaper / backup batch
|
||||
restore Extract a world archive into a world volume (internal Job entrypoint)
|
||||
backup Archive a world into the backup store and record it (internal Job entrypoint)
|
||||
backup-now Archive every user server's world now, one at a time (or the named ones; -stop stops running ones first; prints the plan, -yes applies; requires root/sudo)
|
||||
files List/read/write one file in a stopped server's world (internal Job entrypoint)
|
||||
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
|
||||
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
|
||||
@@ -60,6 +61,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||
"reaper": cmdReaper,
|
||||
"restore": cmdRestore,
|
||||
"backup": cmdBackup,
|
||||
"backup-now": cmdBackupNow,
|
||||
"files": cmdFiles,
|
||||
"egress-gate": cmdEgressGate,
|
||||
"fetch-context": cmdFetchContext,
|
||||
|
||||
@@ -262,6 +262,95 @@ its threshold, and §13b covers a full disk. On a host that built its images,
|
||||
`docker builder prune -af` (with Docker started) reclaims the build cache when space is
|
||||
short; the next upgrade rebuilds it.
|
||||
|
||||
### Growing the disk
|
||||
|
||||
Everything above shares the root filesystem, so more room means a bigger root
|
||||
filesystem. It grows in place, with everything running: enlarge the virtual disk at the
|
||||
provider, then the partition and the filesystem on it.
|
||||
|
||||
```bash
|
||||
sudo felis backup-now -yes # a mistyped partition number is how a resize loses a disk
|
||||
lsblk -f # which disk and partition hold /, and whether LVM sits on it
|
||||
sudo growpart /dev/vda 3 # cloud-utils-growpart (RHEL) / cloud-guest-utils (Debian, Ubuntu)
|
||||
# LVM (the RHEL-family default):
|
||||
sudo pvresize /dev/vda3
|
||||
sudo lvextend -r -l +100%FREE /dev/<vg>/root # -r grows the filesystem with it
|
||||
# no LVM:
|
||||
sudo xfs_growfs / # xfs
|
||||
sudo resize2fs /dev/vda3 # ext4
|
||||
df -h /
|
||||
```
|
||||
|
||||
`felis backup-now` (troubleshooting.md §10) archives every stopped world; add `-stop` to
|
||||
include the running ones.
|
||||
|
||||
### Moving the data to its own disk [VM-VERIFIED]
|
||||
|
||||
The bulk lives under `/var/lib/rancher/k3s`: the worlds, the world archives, the registry
|
||||
and the images. On a disk of its own it grows without touching the system, and a full
|
||||
world store leaves the root filesystem alone. The database and its bundles
|
||||
(`/var/lib/felis`) are small and stay on the root disk. The move takes the platform down
|
||||
for the copy plus a minute or two: the drill copied 4.2 GB in 18 s, and felis-api answered
|
||||
`/readyz` 14 s after k3s started on the new disk.
|
||||
|
||||
1. Attach the disk and put a filesystem on it (the whole disk; `lsblk` shows it empty):
|
||||
|
||||
```bash
|
||||
sudo mkfs.xfs /dev/vdb
|
||||
U=$(sudo blkid -s UUID -o value /dev/vdb)
|
||||
```
|
||||
|
||||
2. Archive every world, stopping the servers so each one saves, and keep the watchdog
|
||||
quiet for the next hour (the marker the installer writes: no mail, no failure pings
|
||||
to the heartbeat, until the time in it):
|
||||
|
||||
```bash
|
||||
sudo felis backup-now -yes -stop
|
||||
sudo install -d -m 0755 /run/felis
|
||||
echo $(( $(date +%s) + 3600 )) | sudo tee /run/felis/watchdog-quiet-until
|
||||
```
|
||||
|
||||
3. Stop k3s and copy:
|
||||
|
||||
```bash
|
||||
sudo systemctl stop k3s
|
||||
sudo /usr/local/bin/k3s-killall.sh # the containers k3s leaves running, and their mounts
|
||||
sudo mkdir -p /mnt/felis-data
|
||||
sudo mount UUID=$U /mnt/felis-data
|
||||
sudo rsync -aHAX --numeric-ids /var/lib/rancher/k3s/ /mnt/felis-data/
|
||||
sudo umount /mnt/felis-data
|
||||
```
|
||||
|
||||
`-X` carries the SELinux labels k3s set itself. Leave `restorecon` out: it would reset
|
||||
runc and the CNI binaries from `container_runtime_exec_t` to the policy default.
|
||||
|
||||
4. Mount it in place, and tie k3s to the mount:
|
||||
|
||||
```bash
|
||||
sudo mv /var/lib/rancher/k3s /var/lib/rancher/k3s.old
|
||||
sudo mkdir /var/lib/rancher/k3s
|
||||
echo "UUID=$U /var/lib/rancher/k3s xfs defaults,nofail 0 0" | sudo tee -a /etc/fstab
|
||||
sudo mkdir -p /etc/systemd/system/k3s.service.d
|
||||
printf '[Unit]\nRequiresMountsFor=/var/lib/rancher/k3s\n' | sudo tee /etc/systemd/system/k3s.service.d/data-disk.conf
|
||||
sudo systemctl daemon-reload
|
||||
sudo mount /var/lib/rancher/k3s
|
||||
sudo systemctl start k3s
|
||||
```
|
||||
|
||||
The drop-in is what keeps the data safe: k3s started on the empty mount point creates
|
||||
a new, empty cluster there. With it, a disk that does not come up fails the start with
|
||||
`A dependency job for k3s.service failed`, and `nofail` keeps the host booting so you
|
||||
can reach it. In the drill a detached disk left k3s inactive and the mount point empty;
|
||||
reattached, `systemctl start k3s` mounted it and started.
|
||||
|
||||
5. Check that `sudo k3s kubectl -n felis get pods` shows every pod ready and
|
||||
`findmnt /var/lib/rancher/k3s` names the new disk, then start the servers from the
|
||||
panel and `sudo rm /run/felis/watchdog-quiet-until`. Once the host has run a day,
|
||||
`sudo rm -rf /var/lib/rancher/k3s.old` frees the root disk.
|
||||
|
||||
The watchdog already watches `/var/lib/rancher/k3s` as a filesystem of its own (its
|
||||
`-disk-paths`), so the new disk's fill level is mailed like the root's.
|
||||
|
||||
## 3. Uninstall
|
||||
|
||||
`deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove
|
||||
@@ -604,6 +693,52 @@ production install:
|
||||
non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
|
||||
or rely on the watchdog's mail.
|
||||
|
||||
### Moving to another host (planned)
|
||||
|
||||
A planned move is the rebuild of troubleshooting.md §16, with the old host still there to
|
||||
hand over a copy that misses nothing. It needs the off-site bucket: that is how the world
|
||||
archives reach the new host (§16 step 7). The platform is down from step 1 until the new
|
||||
host serves.
|
||||
|
||||
1. **On the old host**, stop everything that changes a world, then send the last copy:
|
||||
|
||||
```bash
|
||||
sudo install -d -m 0755 /run/felis
|
||||
echo $(( $(date +%s) + 4 * 3600 )) | sudo tee /run/felis/watchdog-quiet-until
|
||||
sudo systemctl stop felis-velocity # no joins, so no server wakes
|
||||
sudo felis backup-now -yes -stop # every world archived; the servers stay stopped
|
||||
sudo k3s kubectl -n felis scale deploy/felis-operator --replicas=0 # nothing starts a server from here on
|
||||
sudo felis db backup # a bundle that lists those archives
|
||||
sudo systemctl start felis-offsite.service
|
||||
sudo felis offsite status # again until nothing waits
|
||||
```
|
||||
|
||||
The order matters. The new host fetches the archives its restored database lists, so
|
||||
the bundle comes after the last archive. The operator goes after `backup-now`, which
|
||||
needs it to stop the servers. The quiet marker keeps the watchdog from mailing the
|
||||
owners about the stopped proxy and operator for the next 4 hours.
|
||||
2. **On the new host**, follow troubleshooting.md §16 "Rebuild on a new host" from step 1;
|
||||
`fetch-db latest` picks the bundle the old host just sent. Step 8 (`felis offsite
|
||||
take-over`) makes the new host the one that writes the bucket, and from then on the
|
||||
old host copies nothing more. Steps 10 and 11 move the names and the tunnel.
|
||||
3. **Check the new host** before announcing it: sign in with an email code, restore one
|
||||
world and join it, and see `sudo felis offsite status` show a recent `last success` and
|
||||
no stand-by notice.
|
||||
4. **Retire the old host.** It holds the last copy of every world outside the bucket, so
|
||||
keep it powered off with its disk for a few days first, disabled so a boot brings
|
||||
nothing up:
|
||||
|
||||
```bash
|
||||
sudo systemctl disable k3s felis-velocity felis-watchdog.timer felis-offsite.timer \
|
||||
felis-db-backup.timer felis-update-check.timer felis-build-tools.timer
|
||||
sudo poweroff
|
||||
```
|
||||
|
||||
Then uninstall it (§3) or wipe it.
|
||||
|
||||
Each step is covered where it is documented (backup-now in troubleshooting.md §10, the
|
||||
rebuild in §16); the sequence as a whole has not been rehearsed as one move.
|
||||
|
||||
## 6. Changing the root domain [VM-VERIFIED] [GO-TESTED] [SH-TESTED]
|
||||
|
||||
The root domain is written into more places than the installer's config: the panel
|
||||
|
||||
@@ -1080,6 +1080,56 @@ the error its container exited on under Recent operations on the server's
|
||||
backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`,
|
||||
`TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`]
|
||||
|
||||
### Every world at once: `felis backup-now`
|
||||
|
||||
A world lives only in its volume, and the off-site copy holds only its archives.
|
||||
Before anything that could lose a volume (growing the disk, moving the data to its
|
||||
own disk, moving to another host: operations.md §2 and §5), archive every world
|
||||
from the node:
|
||||
|
||||
```bash
|
||||
sudo felis backup-now # the plan; nothing changes
|
||||
sudo felis backup-now -yes # archive every stopped world, one at a time
|
||||
sudo felis backup-now -yes -stop # stop the running servers first
|
||||
sudo felis backup-now -yes alpha bravo # only these servers
|
||||
```
|
||||
|
||||
It runs as root because it reads the ops token (`felis/felis-ops-token`, §6) and
|
||||
asks felis-api's internal face for each backup. Each archive is an ordinary manual
|
||||
backup (the same Job, the same 10% free-disk check, the audit action
|
||||
`backup.create` with the source `internal:ops` and the sudo user as actor), exempt
|
||||
from the owner cooldown and the `max_local_bytes` cap like the break-glass console.
|
||||
|
||||
- **The plan** lists every user server with its phase and what the run does with
|
||||
it: `back up`, `stop, then back up`, `skip: running (stop it first, or pass
|
||||
-stop)`, or `skip: no world volume (never started, nothing to save)`. With
|
||||
nothing to archive it ends `Nothing to back up.`
|
||||
- **Each archive counts against `manual_keep`**: a server that already holds that
|
||||
many manual backups loses its oldest, and the plan says so. Raise
|
||||
`[archive] manual_keep` first when those older backups matter.
|
||||
- **Stopped servers go first**, so a felis-api that cannot take a backup is found
|
||||
before anything is stopped for one. The run waits for each Job (`alpha: archived
|
||||
in 42s`) before starting the next.
|
||||
- **`-stop` disconnects the players** and leaves those servers stopped (`Left
|
||||
stopped: …`; start them from the panel). Each stop writes `break_glass.halt` to
|
||||
the audit log. A server still up after 10 minutes counts as failed, and the run
|
||||
moves on.
|
||||
- **An unreachable felis-api ends the run** (`Stopped: nothing more can be backed
|
||||
up until felis-api answers`). Ctrl-C ends it after the current step; a backup Job
|
||||
already started runs to its end.
|
||||
- **The archives stay on the node** until the hourly off-site copy. After a run
|
||||
that archived something the command prints `sudo systemctl start
|
||||
felis-offsite.service`, which sends them now; `sudo felis offsite status` shows
|
||||
what still waits.
|
||||
|
||||
It exits 0 when every world with a volume was archived, 1 when a backup failed, a
|
||||
running server was skipped (no `-stop`) or the run was interrupted, and 2 for a
|
||||
name that is no user server. [GO-TESTED: `TestBackupNowPlanChangesNothing`,
|
||||
`TestBackupNowPlanWithNothingToSave`, `TestBackupNowBacksUpEachWorldInTurn`,
|
||||
`TestBackupNowSkipsRunningServersWithoutStop`,
|
||||
`TestBackupNowStopsAtAnUnreachableAPI`, `TestBackupNowStopsWhenInterrupted`,
|
||||
`TestBackupNowNamedServers`]
|
||||
|
||||
### Scheduled backups (daily restore points)
|
||||
|
||||
A world played every day never idles 15 days, so the reaper never archives it.
|
||||
|
||||
Reference in new issue
Block a user