fix(reaper): 回收前先停服并持有世界维护锁再归档,锁丢失不删卷;无世界卷的无主服务器只重置时钟不再重复回收

This commit is contained in:
Lemon-miaow committed 2026-09-25 03:31:50 +08:00
1 parent 89a0134707
commit 7766e8efa4
14 files changed
+751 -51

No files matched your search

+2 -2
View File
@@ -138,8 +138,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
// FelisWorldJobFailed rule) reach the operator: a world that cannot be archived
// is kept, and without this nobody would learn that it is never reaped.
func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int {
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.Warned, sum.Skipped, sum.StoreFull,
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed,
sum.Verified, sum.Corrupt, sum.VerifyFailed, sum.Swept, sum.OrphanArchives)
if !sum.Failed() {
+1
View File
@@ -27,6 +27,7 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
want int
}{
{"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0},
{"waiting for a stop", reaper.Summary{Evaluated: 3, AwaitingStop: 1}, 0},
{"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1},
{"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1},
{"expiry failed", reaper.Summary{Evaluated: 3, ExpireFailed: 2}, 1},
+2
View File
@@ -37,6 +37,8 @@ func maintenanceLabel(kind string) string {
return "a backup"
case maintenance.KindFileWrite:
return "a file write"
case maintenance.KindReap:
return "the idle-world reaper"
}
return "another operation"
}
+6 -1
View File
@@ -18,7 +18,8 @@
// conflict and re-checks.
//
// A lock older than Grace with no Job behind it is stale (felis-api died between
// the two writes) and holds nothing.
// the two writes) and holds nothing. The reaper is the one holder without a Job:
// it keeps its lock fresh by rewriting it while it archives and reclaims a world.
//
// A restore that starts with a safety snapshot is two Jobs in a row: the backup
// Job carries the restore to run after it (LabelThenRestore), and felis-api
@@ -83,6 +84,10 @@ const (
KindRestore = "restore"
KindBackup = "backup"
KindFileWrite = "file-write"
// KindReap is the reaper archiving an idle world and reclaiming its volume.
// It runs no Job: the reaper holds the Annotation itself and rewrites it
// well inside Grace for as long as it works on the world.
KindReap = "reap"
)
// FilesModeWrite is the LabelFilesMode value of a file write.
+38 -1
View File
@@ -33,7 +33,11 @@ func (c *reclaimCluster) DeletePVC(_ context.Context, pvc string) error {
return nil
}
func (c *reclaimCluster) Stop(context.Context, string) error { return nil }
func (c *reclaimCluster) HoldWorld(ctx context.Context, _ string) (context.Context, func(), error) {
return ctx, func() {}, nil
}
func (c *reclaimCluster) WorldExists(context.Context, string) (bool, error) { return true, nil }
type reclaimArchiver struct{ archived []string }
@@ -387,3 +391,36 @@ func TestBackupReadBack(t *testing.T) {
t.Fatalf("live refs after deleting %s still claim its archive", newer)
}
}
// TestRestartClock: an idle server with no world to reclaim gets its clock and
// warnings reset, and keeps its owner and resource cache (data-durability-19).
func TestRestartClock(t *testing.T) {
ctx := context.Background()
st := reaper.NewPGStore(db)
name := "clock-" + suffix(t)
old := time.Now().Add(-20 * reaper.Day).UTC().Truncate(time.Microsecond)
if _, err := db.ExecContext(ctx,
`INSERT INTO servers (name, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, 100, 128, 1)`,
name); err != nil {
t.Fatalf("seed server: %v", err)
}
if _, err := db.ExecContext(ctx,
`UPDATE servers SET last_active_at = $2, warned_3d_at = $2, warned_1d_at = $2 WHERE name = $1`, name, old); err != nil {
t.Fatalf("age server: %v", err)
}
at := time.Now().UTC().Truncate(time.Microsecond)
if err := st.RestartClock(ctx, name, at); err != nil {
t.Fatalf("RestartClock: %v", err)
}
var last time.Time
var w3, w1 sql.NullTime
var cpu int
if err := db.QueryRowContext(ctx,
`SELECT last_active_at, warned_3d_at, warned_1d_at, cached_cpu_milli FROM servers WHERE name = $1`, name).
Scan(&last, &w3, &w1, &cpu); err != nil {
t.Fatalf("read back: %v", err)
}
if !last.Equal(at) || w3.Valid || w1.Valid || cpu != 100 {
t.Fatalf("after RestartClock: last_active_at=%v warned=%v/%v cpu=%d; want %v, cleared, cache kept", last, w3, w1, cpu, at)
}
}
+10 -3
View File
@@ -175,9 +175,14 @@ func OperatorRole(p Params) *rbacv1.Role {
// ReaperRole grants felis-reaper its two destructive, disjoint powers
// (internal/reaper.k8scluster): patch a MinecraftServer to Stop it and delete its
// world PVC. Candidate servers come from the Postgres store, not a cluster List,
// so no list/watch is needed; the reaper uses a direct client. It can read+patch
// minecraftservers but cannot create them, and holds no power over StatefulSets,
// Services, or Secrets — those belong to the operator and api.
// so minecraftservers need no list/watch; the reaper uses a direct client. It can
// read+patch minecraftservers but cannot create them, and holds no power over
// StatefulSets, Services, or Secrets — those belong to the operator and api.
//
// The same patch holds the world maintenance lock while a world is archived and
// reclaimed (internal/maintenance). Taking it needs the two reads felis-api makes
// before a restore: list pods (is the game pod gone) and list jobs (does a
// restore, backup or file write hold the world). Both are list-only.
//
// persistentvolumeclaims also carries get: resolving where a world lives
// (cmd/felis/reaper.resolveWorldDir) reads the PVC's volumeName to derive the
@@ -195,6 +200,8 @@ func ReaperRole(p Params) *rbacv1.Role {
return role(p.MinecraftNamespace, "felis-reaper", ComponentReaper, []rbacv1.PolicyRule{
rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "patch"}),
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get", "delete"}),
rule([]string{groupCore}, []string{"pods"}, []string{"list"}),
rule([]string{groupBatch}, []string{"jobs"}, []string{"list"}),
})
}
+12 -1
View File
@@ -221,12 +221,23 @@ func TestReaperRole_ScopeExact(t *testing.T) {
t.Errorf("reaper must NOT touch %s/%s (operator/api territory)", res.group, res.name)
}
}
// The reaper never lists from the cluster (candidates come from Postgres).
// The reaper never lists servers from the cluster (candidates come from Postgres).
for _, v := range []string{"list", "watch"} {
if hasRule(rp, groupFelis, "minecraftservers", v) {
t.Errorf("reaper must NOT %s minecraftservers (candidates come from the store)", v)
}
}
// Taking the world lock reads the game pod and the maintenance Jobs, list only.
if !hasRule(rp, groupCore, "pods", "list") || !hasRule(rp, groupBatch, "jobs", "list") {
t.Error("reaper must list pods and jobs to take the world maintenance lock")
}
for _, res := range []struct{ group, name string }{{groupCore, "pods"}, {groupBatch, "jobs"}} {
for _, v := range []string{"get", "watch", "create", "update", "patch", "delete"} {
if hasRule(rp, res.group, res.name, v) {
t.Errorf("reaper must NOT %s %s (list-only)", v, res.name)
}
}
}
}
// TestReaperRBAC_GatedOnRetention locks the reaper identity to its consumer: the
+184 -15
View File
@@ -2,13 +2,20 @@ package reaper
import (
"context"
"errors"
"fmt"
"strings"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
"felis.lolicon.best/internal/naming"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"k8s.io/client-go/util/retry"
"sigs.k8s.io/controller-runtime/pkg/client"
)
@@ -20,13 +27,20 @@ func WorldPVCName(server string) string {
return naming.WorldPVCName(server)
}
// gamePodComponent is the operator's component label on a game server's pod
// (internal/operator.ComponentValue; k8scluster_test pins the two).
const gamePodComponent = "server"
// K8sCluster is the production Cluster backed by a controller-runtime client
// (spec §4, §18). It reads spec.reaperExempt, deletes the world PVC, and flips
// spec.desiredState to Stopped — nothing else. It is integration-tested against
// a live cluster, not the hermetic reaper_test.go suite.
// (spec §4, §18). It reads spec.reaperExempt, stops a server and holds its world
// volume through the maintenance lock, and deletes the world PVC — nothing else.
type K8sCluster struct {
c client.Client
namespace string
now func() time.Time
// beat is how often a held lock is rewritten; maintenance.Grace/4 unless a
// test shortens it.
beat time.Duration
}
// NewK8sCluster builds a Cluster over c, scoped to namespace.
@@ -34,17 +48,177 @@ func NewK8sCluster(c client.Client, namespace string) *K8sCluster {
return &K8sCluster{c: c, namespace: namespace}
}
func (k *K8sCluster) clock() time.Time {
if k.now != nil {
return k.now()
}
return time.Now()
}
func (k *K8sCluster) Inspect(ctx context.Context, name string) (ServerCRD, error) {
var ms v1alpha1.MinecraftServer
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil {
if apierrors.IsNotFound(err) {
return ServerCRD{}, ErrNotFound
}
if err := k.get(ctx, name, &ms); err != nil {
return ServerCRD{}, err
}
return ServerCRD{Exempt: ms.Spec.ReaperExempt, PVC: WorldPVCName(name)}, nil
}
// HoldWorld implements Cluster. The lock is the same Annotation felis-api
// writes before a restore, backup or file-write Job (internal/maintenance): a
// wake through felis-api, the operator scaling the server up, and another
// world operation all refuse while it is fresh. With no Job behind it, it
// holds for maintenance.Grace after each write, so it is rewritten every beat;
// the held context is cancelled once a rewrite has failed for Grace/2, well
// before any reader could take the lock as lapsed.
func (k *K8sCluster) HoldWorld(ctx context.Context, name string) (context.Context, func(), error) {
if err := k.lock(ctx, name); err != nil {
return nil, nil, err
}
held, cancel := context.WithCancelCause(ctx)
done := make(chan struct{})
go k.keep(held, cancel, name, done)
release := func() {
cancel(nil)
<-done
rctx, stop := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second)
defer stop()
_ = k.unlock(rctx, name)
}
return held, release, nil
}
// lock stops the server if it is still meant to run and takes the reap lock
// once it is fully down and nothing else holds its world.
func (k *K8sCluster) lock(ctx context.Context, name string) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer
if err := k.get(ctx, name, &ms); err != nil {
return err
}
if ms.Spec.DesiredState != "" && ms.Spec.DesiredState != v1alpha1.DesiredStopped {
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
ms.Spec.DesiredState = v1alpha1.DesiredStopped
if err := k.c.Patch(ctx, &ms, patch); err != nil {
return err
}
return fmt.Errorf("%w: it was still up and has been told to stop", ErrNotQuiet)
}
if ms.Status.Ready || (ms.Status.Phase != "" && ms.Status.Phase != v1alpha1.PhaseStopped) {
return fmt.Errorf("%w: it is still stopping (phase %s)", ErrNotQuiet, ms.Status.Phase)
}
var pods corev1.PodList
if err := k.c.List(ctx, &pods, client.InNamespace(k.namespace), client.MatchingLabels{
v1alpha1.LabelServer: name, v1alpha1.LabelComponent: gamePodComponent,
}); err != nil {
return err
}
if len(pods.Items) > 0 {
return fmt.Errorf("%w: its game pod is still shutting down", ErrNotQuiet)
}
var jobs batchv1.JobList
if err := k.c.List(ctx, &jobs, client.InNamespace(k.namespace),
client.MatchingLabels{maintenance.LabelServer: name}); err != nil {
return err
}
if kind, held := maintenance.Holder(name, ms.Annotations, jobs.Items, k.clock()); held {
return fmt.Errorf("%w: a %s holds its world", ErrNotQuiet, kind)
}
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
if ms.Annotations == nil {
ms.Annotations = map[string]string{}
}
ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindReap, k.clock())
return k.c.Patch(ctx, &ms, patch)
})
}
// errLockLost reports that the lock on the object is no longer the reaper's.
var errLockLost = errors.New("the reap lock is gone from the server")
// keep rewrites the lock every beat until held ends, and ends held itself once
// it has not managed to for Grace/2.
func (k *K8sCluster) keep(held context.Context, cancel context.CancelCauseFunc, name string, done chan<- struct{}) {
defer close(done)
beat := k.beat
if beat <= 0 {
beat = maintenance.Grace / 4
}
t := time.NewTicker(beat)
defer t.Stop()
last := k.clock()
for {
select {
case <-held.Done():
return
case <-t.C:
}
err := k.refresh(held, name)
if err == nil {
last = k.clock()
continue
}
if held.Err() != nil {
return
}
if errors.Is(err, errLockLost) || k.clock().Sub(last) >= maintenance.Grace/2 {
cancel(fmt.Errorf("reaper: lost the world lock on %s: %w", name, err))
return
}
}
}
func (k *K8sCluster) refresh(ctx context.Context, name string) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer
if err := k.get(ctx, name, &ms); err != nil {
return err
}
if !reapLock(ms.Annotations) {
return errLockLost
}
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindReap, k.clock())
return k.c.Patch(ctx, &ms, patch)
})
}
// unlock drops the lock if it is still the reaper's; a lock left behind lapses
// after maintenance.Grace on its own.
func (k *K8sCluster) unlock(ctx context.Context, name string) error {
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
var ms v1alpha1.MinecraftServer
if err := k.get(ctx, name, &ms); err != nil {
if errors.Is(err, ErrNotFound) {
return nil
}
return err
}
if !reapLock(ms.Annotations) {
return nil
}
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
delete(ms.Annotations, maintenance.Annotation)
return k.c.Patch(ctx, &ms, patch)
})
}
func reapLock(annotations map[string]string) bool {
return strings.HasPrefix(annotations[maintenance.Annotation], maintenance.KindReap+"@")
}
// WorldExists reports whether the world PersistentVolumeClaim exists.
func (k *K8sCluster) WorldExists(ctx context.Context, pvc string) (bool, error) {
var claim corev1.PersistentVolumeClaim
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: pvc}, &claim)
switch {
case apierrors.IsNotFound(err):
return false, nil
case err != nil:
return false, err
}
return true, nil
}
// DeletePVC deletes the world PersistentVolumeClaim. A missing PVC is not an
// error: the reap is idempotent and a re-run after a partial failure must still
// converge.
@@ -58,17 +232,12 @@ func (k *K8sCluster) DeletePVC(ctx context.Context, pvc string) error {
return nil
}
// Stop sets spec.desiredState=Stopped with a merge patch so a concurrent status
// write by the operator is never clobbered (spec §9.1).
func (k *K8sCluster) Stop(ctx context.Context, name string) error {
var ms v1alpha1.MinecraftServer
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil {
func (k *K8sCluster) get(ctx context.Context, name string, ms *v1alpha1.MinecraftServer) error {
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, ms); err != nil {
if apierrors.IsNotFound(err) {
return ErrNotFound
}
return err
}
patch := client.MergeFrom(ms.DeepCopy())
ms.Spec.DesiredState = v1alpha1.DesiredStopped
return k.c.Patch(ctx, &ms, patch)
return nil
}
+218
View File
@@ -0,0 +1,218 @@
package reaper
import (
"context"
"errors"
"strings"
"sync"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
"felis.lolicon.best/internal/operator"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
"k8s.io/apimachinery/pkg/types"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
// HoldWorld against a fake API server, which honours resourceVersion on the
// optimistic-lock patches the lock is written with.
func holdServer(desired v1alpha1.DesiredState, phase v1alpha1.Phase) *v1alpha1.MinecraftServer {
return &v1alpha1.MinecraftServer{
ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"},
Spec: v1alpha1.MinecraftServerSpec{DesiredState: desired},
Status: v1alpha1.MinecraftServerStatus{Phase: phase},
}
}
// fakeClock is a clock the tests and the heartbeat goroutine share.
type fakeClock struct {
mu sync.Mutex
t time.Time
}
func (c *fakeClock) now() time.Time { c.mu.Lock(); defer c.mu.Unlock(); return c.t }
func (c *fakeClock) add(d time.Duration) {
c.mu.Lock()
defer c.mu.Unlock()
c.t = c.t.Add(d)
}
func holdCluster(t *testing.T, objs ...client.Object) (*K8sCluster, client.Client, *fakeClock) {
t.Helper()
scheme := runtime.NewScheme()
if err := clientgoscheme.AddToScheme(scheme); err != nil {
t.Fatalf("scheme: %v", err)
}
if err := v1alpha1.AddToScheme(scheme); err != nil {
t.Fatalf("scheme: %v", err)
}
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(objs...).
WithStatusSubresource(&v1alpha1.MinecraftServer{}).Build()
clk := &fakeClock{t: time.Date(2026, 9, 24, 12, 0, 0, 0, time.UTC)}
k := NewK8sCluster(c, "minecraft")
k.now = clk.now
k.beat = time.Hour
return k, c, clk
}
func serverState(t *testing.T, c client.Client) (map[string]string, v1alpha1.DesiredState) {
t.Helper()
var ms v1alpha1.MinecraftServer
if err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival"}, &ms); err != nil {
t.Fatalf("get: %v", err)
}
return ms.Annotations, ms.Spec.DesiredState
}
func TestHoldWorldStopsARunningServer(t *testing.T) {
k, c, _ := holdCluster(t, holdServer(v1alpha1.DesiredRunning, v1alpha1.PhaseRunning))
_, _, err := k.HoldWorld(context.Background(), "survival")
if !errors.Is(err, ErrNotQuiet) {
t.Fatalf("HoldWorld = %v, want ErrNotQuiet", err)
}
ann, desired := serverState(t, c)
if desired != v1alpha1.DesiredStopped || ann[maintenance.Annotation] != "" {
t.Fatalf("desired=%q annotations=%v: want told to stop and no lock yet", desired, ann)
}
}
func TestHoldWorldWaitsUntilQuiet(t *testing.T) {
ready := holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
ready.Status.Ready = true
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "survival-0", Namespace: "minecraft",
Labels: map[string]string{v1alpha1.LabelServer: "survival", v1alpha1.LabelComponent: gamePodComponent}}}
job := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Name: "restore-survival", Namespace: "minecraft",
Labels: map[string]string{maintenance.LabelServer: "survival", maintenance.LabelManagedBy: "felis-restore"}}}
locked := holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
locked.Annotations = map[string]string{
maintenance.Annotation: maintenance.LockValue(maintenance.KindBackup, time.Date(2026, 9, 24, 11, 59, 30, 0, time.UTC)),
}
for _, tc := range []struct {
name string
objs []client.Object
why string
}{
{"stopping", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopping)}, "still stopping"},
{"ready", []client.Object{ready}, "still stopping"},
{"game pod", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pod}, "game pod"},
{"restore job", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), job}, "a restore"},
{"admission lock", []client.Object{locked}, "a backup"},
} {
t.Run(tc.name, func(t *testing.T) {
k, c, _ := holdCluster(t, tc.objs...)
before, _ := serverState(t, c)
_, _, err := k.HoldWorld(context.Background(), "survival")
if !errors.Is(err, ErrNotQuiet) || !strings.Contains(err.Error(), tc.why) {
t.Fatalf("HoldWorld = %v, want ErrNotQuiet naming %q", err, tc.why)
}
if after, _ := serverState(t, c); after[maintenance.Annotation] != before[maintenance.Annotation] {
t.Fatalf("lock changed to %q", after[maintenance.Annotation])
}
})
}
}
// Once held, every other holder check sees the reaper, and release lets go.
func TestHoldWorldLocksAndReleases(t *testing.T) {
k, c, clk := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped))
held, release, err := k.HoldWorld(context.Background(), "survival")
if err != nil {
t.Fatalf("HoldWorld: %v", err)
}
ann, _ := serverState(t, c)
if kind, ok := maintenance.Holder("survival", ann, nil, clk.now()); !ok || kind != maintenance.KindReap {
t.Fatalf("Holder = (%q, %v) with %v, want the reaper", kind, ok, ann)
}
release()
if held.Err() == nil {
t.Fatal("held context still live after release")
}
if ann, _ := serverState(t, c); ann[maintenance.Annotation] != "" {
t.Fatalf("lock left behind: %v", ann)
}
}
// The lock is rewritten while held, so it never lapses under a long archive.
func TestHoldWorldKeepsLockFresh(t *testing.T) {
k, c, clk := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped))
k.beat = 5 * time.Millisecond
held, release, err := k.HoldWorld(context.Background(), "survival")
if err != nil {
t.Fatalf("HoldWorld: %v", err)
}
defer release()
clk.add(10 * maintenance.Grace)
want := maintenance.LockValue(maintenance.KindReap, clk.now())
deadline := time.Now().Add(5 * time.Second)
for {
if ann, _ := serverState(t, c); ann[maintenance.Annotation] == want {
break
}
if time.Now().After(deadline) {
ann, _ := serverState(t, c)
t.Fatalf("lock = %q, want rewritten to %q", ann[maintenance.Annotation], want)
}
time.Sleep(5 * time.Millisecond)
}
if held.Err() != nil {
t.Fatalf("held ended: %v", context.Cause(held))
}
}
// A lock someone else removed (or replaced) ends the hold: the reaper must not
// delete a world it no longer holds.
func TestHoldWorldEndsWhenLockIsLost(t *testing.T) {
k, c, _ := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped))
k.beat = 5 * time.Millisecond
held, release, err := k.HoldWorld(context.Background(), "survival")
if err != nil {
t.Fatalf("HoldWorld: %v", err)
}
var ms v1alpha1.MinecraftServer
if err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival"}, &ms); err != nil {
t.Fatalf("get: %v", err)
}
patch := client.MergeFrom(ms.DeepCopy())
ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindRestore, time.Now())
if err := c.Patch(context.Background(), &ms, patch); err != nil {
t.Fatalf("patch: %v", err)
}
select {
case <-held.Done():
case <-time.After(5 * time.Second):
t.Fatal("held context still live after the lock was taken over")
}
if !errors.Is(context.Cause(held), errLockLost) {
t.Fatalf("cause = %v, want errLockLost", context.Cause(held))
}
release()
if ann, _ := serverState(t, c); !strings.HasPrefix(ann[maintenance.Annotation], maintenance.KindRestore+"@") {
t.Fatalf("release removed another holder's lock: %v", ann)
}
}
func TestWorldExists(t *testing.T) {
pvc := &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: "world-survival-0", Namespace: "minecraft"}}
k, _, _ := holdCluster(t, pvc)
for name, want := range map[string]bool{"world-survival-0": true, "world-other-0": false} {
if got, err := k.WorldExists(context.Background(), name); err != nil || got != want {
t.Errorf("WorldExists(%s) = (%v, %v), want %v", name, got, err, want)
}
}
}
// gamePodComponent is a copy of the operator's pod label value; a drift would
// let the reaper archive a world beside a server still saving it.
func TestGamePodComponentMatchesOperator(t *testing.T) {
if gamePodComponent != operator.ComponentValue {
t.Fatalf("gamePodComponent = %q, operator labels its pods %q", gamePodComponent, operator.ComponentValue)
}
}
+7
View File
@@ -90,6 +90,13 @@ func (s *PGStore) ReleaseWorld(ctx context.Context, name string, at time.Time) e
return err
}
func (s *PGStore) RestartClock(ctx context.Context, name string, at time.Time) error {
const q = `UPDATE servers SET last_active_at = $2, warned_3d_at = NULL, warned_1d_at = NULL
WHERE name = $1 AND deleted_at IS NULL`
_, err := s.db.ExecContext(ctx, q, name, at)
return err
}
func (s *PGStore) MarkWarned(ctx context.Context, name string, tier Tier, at time.Time) error {
// The column is one of two fixed identifiers, never user input.
col := "warned_3d_at"
+92 -14
View File
@@ -55,6 +55,11 @@ const (
// delete world data it cannot first inspect for the exemption flag.
var ErrNotFound = errors.New("reaper: server not found")
// ErrNotQuiet is returned by Cluster.HoldWorld while the server is not fully
// down yet, or another operation (a restore, backup or file write) holds its
// world. The world is left alone and the server is tried again on the next run.
var ErrNotQuiet = errors.New("reaper: world not quiet")
// errStoreFull is an internal sentinel: the backup store is at capacity and
// could not be freed, so the world is preserved rather than deleted without a
// backup (red line ④). It is never returned to callers.
@@ -219,6 +224,10 @@ type Store interface {
// row (red line ②).
ReleaseWorld(ctx context.Context, name string, at time.Time) error
// RestartClock sets last_active_at→at and clears warned_* on a server with
// no world to reclaim, so it is not found idle again every run.
RestartClock(ctx context.Context, name string, at time.Time) error
// MarkWarned stamps the warned_3d_at / warned_1d_at column for tier.
MarkWarned(ctx context.Context, name string, tier Tier, at time.Time) error
@@ -255,17 +264,25 @@ type Store interface {
Audit(ctx context.Context, rec AuditRecord) error
}
// Cluster is the lifecycle (Kubernetes) face: read the CRD, delete the world
// PVC, and flip desiredState to Stopped. These are the only cluster operations
// §18 performs.
// Cluster is the lifecycle (Kubernetes) face: read the CRD, stop the server
// and hold its world, and delete the world PVC. These are the only cluster
// operations §18 performs.
type Cluster interface {
// Inspect returns the exemption flag and world PVC name for a server, or
// ErrNotFound if the CRD is gone.
Inspect(ctx context.Context, name string) (ServerCRD, error)
// HoldWorld keeps everything else off the server's world until release is
// called: it sets desiredState=Stopped if the server is still meant to run,
// and once the server is fully down (not ready, phase Stopped, no game pod)
// and no restore, backup or file write holds the world, it takes the world
// maintenance lock (internal/maintenance, KindReap) and keeps it fresh.
// Until then it returns ErrNotQuiet. The returned context ends if the lock
// cannot be kept; the world must not be read or deleted on it after that.
HoldWorld(ctx context.Context, name string) (held context.Context, release func(), err error)
// WorldExists reports whether the world PersistentVolumeClaim exists.
WorldExists(ctx context.Context, pvc string) (bool, error)
// DeletePVC deletes the world PersistentVolumeClaim.
DeletePVC(ctx context.Context, pvc string) error
// Stop sets spec.desiredState=Stopped.
Stop(ctx context.Context, name string) error
}
// Warner delivers an impending-reap notice. It is optional and best-effort: a
@@ -309,6 +326,10 @@ type Summary struct {
// AwaitingOffsite are idle worlds that are archived and kept until the
// archive's off-site copy lands.
AwaitingOffsite int
// AwaitingStop are idle servers left for the next run because they were not
// fully down yet (the run told a running one to stop) or another operation
// held their world.
AwaitingStop int
// Verified are archives read back in full and found matching; Corrupt are
// the ones that were not (marked, and never reused or restored from);
// VerifyFailed are the ones that could not be read back at all this run.
@@ -379,6 +400,11 @@ func (r *Reaper) RunOnce(ctx context.Context) (Summary, error) {
for _, c := range cands {
sum.Evaluated++
if err := r.evaluate(ctx, now, offs, c, &sum); err != nil {
if errors.Is(err, ErrNotQuiet) {
sum.AwaitingStop++
r.log().Warn("reaper: idle world not quiet, retrying next run", "server", c.Name, "why", err)
continue
}
r.log().Error("reaper: skipping server", "server", c.Name, "err", err)
sum.Skipped++
if errors.Is(err, errStoreFull) {
@@ -419,9 +445,27 @@ func (r *Reaper) evaluate(ctx context.Context, now time.Time, offs []time.Durati
return nil
}
// reap archives the world, records the backup, and only then deletes the PVC,
// releases ownership, and stops the server — the strict ordering of red line ④.
// reap stops the server and holds its world, archives the world, records the
// backup, and only then deletes the PVC and releases ownership — the strict
// ordering of red line ④. Nothing can start the server or touch its world while
// it is held, so the archive is of a world at rest and the PVC deleted is the
// one archived.
func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd ServerCRD, sum *Summary) error {
held, release, err := r.Cluster.HoldWorld(ctx, c.Name)
if err != nil {
return fmt.Errorf("hold world: %w", err)
}
defer release()
ctx = held
exists, err := r.Cluster.WorldExists(ctx, crd.PVC)
if err != nil {
return fmt.Errorf("look up world volume: %w", err)
}
if !exists {
return r.reapNoWorld(ctx, now, c, sum)
}
// §26 soft cap: free space before adding a backup. If the store cannot be
// brought under cap, preserve the world rather than delete it unbacked.
if r.Cfg.MaxLocalBytes > 0 {
@@ -497,32 +541,66 @@ func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd Serve
}
// World is safely archived and recorded — now (and only now) delete it.
// The lock must still be held: a lapsed one could have let the server start.
if err := context.Cause(ctx); err != nil {
return fmt.Errorf("world no longer held: %w", err)
}
if err := r.Cluster.DeletePVC(ctx, crd.PVC); err != nil {
// The backup row persists; next run's FreshBackup reuses it and retries
// the delete, so no duplicate archive is created.
return fmt.Errorf("delete pvc: %w", err)
}
return r.finishReap(ctx, now, c, ref, sum)
}
// finishReap releases a world whose PVC is gone and records the reap.
func (r *Reaper) finishReap(ctx context.Context, now time.Time, c Candidate, ref string, sum *Summary) error {
if err := r.Store.ReleaseWorld(ctx, c.Name, now); err != nil {
return fmt.Errorf("release world: %w", err)
}
if err := r.Cluster.Stop(ctx, c.Name); err != nil {
// The world is already deleted and ownership released; the desiredState
// flip is cosmetic by comparison. Log, but the reap stands.
r.log().Error("reaper: set desiredState=Stopped failed", "server", c.Name, "err", err)
}
if err := r.Store.Audit(ctx, AuditRecord{Action: ActionReapWorld, ServerName: c.Name, FormerOwner: c.OwnerID}); err != nil {
r.log().Error("reaper: audit reap_world failed", "server", c.Name, "err", err)
}
sum.WorldsReaped++
// felis_reaper_worlds_deleted_total (spec §23) advances in lockstep with the
// per-run Summary tally — incremented here, at the one point a world's PVC has
// actually been deleted, not at evaluation time.
// per-run Summary tally — incremented once per world whose PVC went, not at
// evaluation time.
metrics.ReaperWorldsDeletedTotal.Inc()
r.log().Info("reaper: world reaped", "server", c.Name, "former_owner", c.OwnerID, "backup_ref", ref)
return nil
}
// reapNoWorld handles an idle server whose world PVC does not exist. With an
// archive of the current world on record, an earlier run deleted the PVC and
// stopped before releasing it, so the reap is finished now. Otherwise there was
// no world to reclaim: an owner who never started the server gives it up
// (nothing to back up), and an unowned one — typically a world reaped earlier —
// only has its clock restarted, so it is not reaped over and over.
func (r *Reaper) reapNoWorld(ctx context.Context, now time.Time, c Candidate, sum *Summary) error {
fresh, ok, err := r.Store.FreshBackup(ctx, c.Name, c.LastActiveAt)
if err != nil {
return fmt.Errorf("lookup fresh backup: %w", err)
}
if ok {
return r.finishReap(ctx, now, c, fresh.Ref, sum)
}
if c.OwnerID == "" {
if err := r.Store.RestartClock(ctx, c.Name, now); err != nil {
return fmt.Errorf("restart clock: %w", err)
}
return nil
}
if err := r.Store.ReleaseWorld(ctx, c.Name, now); err != nil {
return fmt.Errorf("release world: %w", err)
}
if err := r.Store.Audit(ctx, AuditRecord{Action: ActionReapWorld, ServerName: c.Name, FormerOwner: c.OwnerID}); err != nil {
r.log().Error("reaper: audit reap_world failed", "server", c.Name, "err", err)
}
r.log().Info("reaper: idle server released; it had no world to archive", "server", c.Name, "former_owner", c.OwnerID)
return nil
}
// ensureCapacity frees the backup store down under MaxLocalBytes by evicting the
// oldest present backups early. Early eviction is destructive (it removes
// not-yet-expired backups), so each eviction is alerted and audited. It returns
+177 -12
View File
@@ -24,7 +24,7 @@ var testNow = time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC)
func idleBy(d time.Duration) time.Time { return testNow.Add(-d) }
// recorder captures the cross-fake call order so a test can assert the strict
// archive→insert→deletePVC→release→stop→audit sequence of red line ④.
// hold→archive→insert→deletePVC→release→audit→unhold sequence of red line ④.
type recorder struct{ events []string }
func (r *recorder) add(e string) { r.events = append(r.events, e) }
@@ -33,6 +33,7 @@ func (r *recorder) add(e string) { r.events = append(r.events, e) }
type fakeArchiver struct {
rec *recorder
onArchive func()
archiveErr error
deleteErr error
archives int
@@ -47,6 +48,9 @@ func (f *fakeArchiver) Archive(_ context.Context, server, _ string) (backup.Arch
f.archives++
f.seq++
f.rec.add("archive")
if f.onArchive != nil {
f.onArchive()
}
ref := fmt.Sprintf("ref-%s-%d", server, f.seq)
return backup.Archived{Ref: backup.ArchiveRef(ref), Size: 10, SHA256: "sha-" + ref}, nil
}
@@ -104,7 +108,39 @@ type fakeCluster struct {
deletePVCErr error
deletePVCCalls int
deletedPVCs []string
stopped []string
notQuiet map[string]bool // HoldWorld refuses these with ErrNotQuiet
holdErr error // HoldWorld fails with this
noWorld map[string]bool // PVCs that do not exist
held map[string]bool // servers held right now
holds []string
lost context.CancelCauseFunc
}
func (c *fakeCluster) HoldWorld(ctx context.Context, name string) (context.Context, func(), error) {
if c.holdErr != nil {
return nil, nil, c.holdErr
}
if c.notQuiet[name] {
return nil, nil, fmt.Errorf("%w: still stopping", ErrNotQuiet)
}
if c.held == nil {
c.held = map[string]bool{}
}
c.held[name] = true
c.holds = append(c.holds, name)
c.rec.add("hold")
held, cancel := context.WithCancelCause(ctx)
c.lost = cancel
return held, func() {
cancel(nil)
delete(c.held, name)
c.rec.add("unhold")
}, nil
}
func (c *fakeCluster) WorldExists(_ context.Context, pvc string) (bool, error) {
return !c.noWorld[pvc], nil
}
func (c *fakeCluster) Inspect(_ context.Context, name string) (ServerCRD, error) {
@@ -118,22 +154,19 @@ func (c *fakeCluster) Inspect(_ context.Context, name string) (ServerCRD, error)
return crd, nil
}
func (c *fakeCluster) DeletePVC(_ context.Context, pvc string) error {
func (c *fakeCluster) DeletePVC(ctx context.Context, pvc string) error {
c.deletePVCCalls++
if c.deletePVCErr != nil {
return c.deletePVCErr
}
if len(c.held) == 0 || ctx.Err() != nil {
return fmt.Errorf("deleted %s without holding its world", pvc)
}
c.deletedPVCs = append(c.deletedPVCs, pvc)
c.rec.add("deletePVC")
return nil
}
func (c *fakeCluster) Stop(_ context.Context, name string) error {
c.stopped = append(c.stopped, name)
c.rec.add("stop")
return nil
}
// ---- fake Store -----------------------------------------------------------
type fakeBackup struct {
@@ -212,6 +245,15 @@ func (s *fakeStore) ReleaseWorld(_ context.Context, name string, at time.Time) e
return nil
}
func (s *fakeStore) RestartClock(_ context.Context, name string, at time.Time) error {
c := s.byName[name]
c.LastActiveAt = at
c.Warned3dAt = time.Time{}
c.Warned1dAt = time.Time{}
s.rec.add("restartClock")
return nil
}
func (s *fakeStore) MarkWarned(_ context.Context, name string, tier Tier, at time.Time) error {
c := s.byName[name]
if tier == Tier1d {
@@ -405,7 +447,7 @@ func TestReapIdleWorldFullSequence(t *testing.T) {
if sum.WorldsReaped != 1 {
t.Fatalf("WorldsReaped = %d, want 1", sum.WorldsReaped)
}
want := []string{"archive", "insert", "deletePVC", "release", "stop", "audit:" + ActionReapWorld}
want := []string{"hold", "archive", "insert", "deletePVC", "release", "audit:" + ActionReapWorld, "unhold"}
if !reflect.DeepEqual(st.rec.events, want) {
t.Fatalf("call order = %v, want %v", st.rec.events, want)
}
@@ -936,7 +978,7 @@ func TestReapReadsBackReusedArchive(t *testing.T) {
if sum.WorldsReaped != 1 || ar.archives != 0 || len(cl.deletedPVCs) != 1 {
t.Fatalf("summary %+v archives=%d deleted=%v: want the checked archive reused", sum, ar.archives, cl.deletedPVCs)
}
if st.rec.events[0] != "verify" || st.rec.events[1] != "deletePVC" {
if st.rec.events[1] != "verify" || st.rec.events[2] != "deletePVC" {
t.Errorf("events = %v, want the read-back before the delete", st.rec.events)
}
if !st.backups[0].verifiedAt.Equal(testNow) || !reflect.DeepEqual(ca.verified, []string{"ref-old"}) {
@@ -958,7 +1000,7 @@ func TestReapReplacesCorruptArchive(t *testing.T) {
if sum.WorldsReaped != 1 || sum.Corrupt != 1 || !sum.Failed() {
t.Fatalf("summary = %+v, want reaped with one corrupt archive reported", sum)
}
want := []string{"verify", "corrupt", "archive", "insert", "deletePVC"}
want := []string{"hold", "verify", "corrupt", "archive", "insert", "deletePVC"}
if !reflect.DeepEqual(st.rec.events[:len(want)], want) {
t.Errorf("events = %v, want %v first", st.rec.events, want)
}
@@ -1057,3 +1099,126 @@ func TestSweepPass(t *testing.T) {
t.Errorf("sweep ran without the recorded archives: %+v", sum)
}
}
// data-durability-8: a server still up when its world goes idle is told to stop
// and left for the next run, never archived while it may be writing; once it is
// down the next run reaps it.
func TestReapWaitsForServerToStop(t *testing.T) {
r, st, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "busy", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
cl.notQuiet = map[string]bool{"busy": true}
sum := mustRun(t, r)
if sum.AwaitingStop != 1 || sum.Skipped != 0 || sum.Failed() || ar.archives != 0 || cl.deletePVCCalls != 0 {
t.Fatalf("summary %+v archives=%d deletes=%d: want the world left alone for the next run", sum, ar.archives, cl.deletePVCCalls)
}
if st.byName["busy"].OwnerID != "user-1" || len(st.audits) != 0 {
t.Fatalf("a server not yet down was released: %+v", st.byName["busy"])
}
cl.notQuiet = nil
if sum := mustRun(t, r); sum.WorldsReaped != 1 || sum.AwaitingStop != 0 {
t.Fatalf("second run = %+v, want reaped once the server is down", sum)
}
}
// A hold that cannot be taken for any other reason is a failure of the run.
func TestReapHoldErrorSkips(t *testing.T) {
r, _, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "flaky", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
cl.holdErr = errors.New("apiserver timeout")
sum := mustRun(t, r)
if sum.Skipped != 1 || !sum.Failed() || ar.archives != 0 {
t.Fatalf("summary %+v archives=%d: want the server skipped and the run failed", sum, ar.archives)
}
}
// The world is held from before the archive until after the reap is recorded,
// and let go on every path, a failed archive included.
func TestReapReleasesHoldOnEveryPath(t *testing.T) {
r, st, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "hotel", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
ar.archiveErr = errors.New("disk full")
mustRun(t, r)
if len(cl.held) != 0 || st.rec.events[len(st.rec.events)-1] != "unhold" {
t.Fatalf("hold not released after a failed archive: events=%v held=%v", st.rec.events, cl.held)
}
}
// A lock lost while the archive is written (felis-api could then start the
// server) keeps the PVC; the archive is recorded and reused next run.
func TestReapLostHoldKeepsWorld(t *testing.T) {
r, st, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "india", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
ar.onArchive = func() { cl.lost(errors.New("lock rewrite failing")) }
sum := mustRun(t, r)
if sum.WorldsReaped != 0 || sum.Skipped != 1 || cl.deletePVCCalls != 0 {
t.Fatalf("summary %+v deletes=%d: want the world kept once the hold was lost", sum, cl.deletePVCCalls)
}
if st.byName["india"].OwnerID != "user-1" {
t.Fatal("ownership released after the hold was lost")
}
ar.onArchive = nil
if sum := mustRun(t, r); sum.WorldsReaped != 1 || ar.archives != 1 {
t.Fatalf("retry = %+v archives=%d, want the recorded archive reused", sum, ar.archives)
}
}
// data-durability-19: an unowned server with no world (reaped before) is not
// reaped again when its clock runs out; the clock restarts, with no audit and
// no count.
func TestReapUnownedWithoutWorldRestartsClock(t *testing.T) {
r, st, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "juliet", LastActiveAt: idleBy(16 * Day)})
cl.noWorld = map[string]bool{"world-juliet-0": true}
sum := mustRun(t, r)
if sum.WorldsReaped != 0 || sum.Skipped != 0 || sum.Failed() || ar.archives != 0 || cl.deletePVCCalls != 0 || len(st.audits) != 0 {
t.Fatalf("summary %+v archives=%d deletes=%d audits=%v: want only the clock restarted",
sum, ar.archives, cl.deletePVCCalls, st.audits)
}
if !st.byName["juliet"].LastActiveAt.Equal(testNow) {
t.Fatalf("last_active_at = %v, want restarted at %v", st.byName["juliet"].LastActiveAt, testNow)
}
if sum := mustRun(t, r); sum.WorldsReaped != 0 || len(st.audits) != 0 {
t.Fatalf("second run = %+v audits=%v", sum, st.audits)
}
}
// An owner who never started the server gives it up at the deadline like any
// other; there is nothing to archive, and no world was deleted to count.
func TestReapOwnedWithoutWorldReleases(t *testing.T) {
r, st, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "kappa", OwnerID: "user-4", LastActiveAt: idleBy(16 * Day)})
cl.noWorld = map[string]bool{"world-kappa-0": true}
sum := mustRun(t, r)
if sum.WorldsReaped != 0 || sum.Skipped != 0 || ar.archives != 0 || cl.deletePVCCalls != 0 {
t.Fatalf("summary %+v archives=%d deletes=%d", sum, ar.archives, cl.deletePVCCalls)
}
if st.byName["kappa"].OwnerID != "" || len(st.audits) != 1 || st.audits[0].FormerOwner != "user-4" {
t.Fatalf("owner %q audits %+v: want released and audited", st.byName["kappa"].OwnerID, st.audits)
}
}
// A run that deleted the PVC and failed before releasing the world is finished
// by the next: released, audited and counted once, without a new archive.
func TestReapFinishesInterruptedReap(t *testing.T) {
r, st, cl, ar := newReaper(DefaultConfig(),
Candidate{Name: "lambda", OwnerID: "user-5", LastActiveAt: idleBy(20 * Day)})
cl.noWorld = map[string]bool{"world-lambda-0": true}
st.backups = []*fakeBackup{{id: "prev", server: "lambda", ref: "ref-prev", reason: ReasonInactive, size: 5,
status: "present", createdAt: idleBy(Day), expires: testNow.Add(89 * Day)}}
sum := mustRun(t, r)
if sum.WorldsReaped != 1 || ar.archives != 0 || cl.deletePVCCalls != 0 {
t.Fatalf("summary %+v archives=%d deletes=%d: want the reap finished", sum, ar.archives, cl.deletePVCCalls)
}
if st.byName["lambda"].OwnerID != "" || len(st.audits) != 1 {
t.Fatalf("owner %q audits %+v", st.byName["lambda"].OwnerID, st.audits)
}
}
+1 -1
View File
@@ -18,7 +18,7 @@
"no_backup": "There's no restorable backup for this server yet.",
"backup_corrupt": "This backup failed its read-back check and can't be restored intact — pick another backup.",
"not_stopped": "Stop the server completely before restoring — a restore overwrites the live world volume.",
"maintenance_in_progress": "This server's world is busy with a restore, backup or file write — try again once it finishes, usually within a minute or two.",
"maintenance_in_progress": "This server's world is busy with a restore, backup, file write or idle reclaim — try again once it finishes. A restore, backup or file write usually takes a minute or two; an idle reclaim can take longer on a large world.",
"no_world_volume": "This server has no world volume yet — start it once so it is created, then retry.",
"backup_cooldown": "This server was backed up moments ago, and manual backups have a cooldown — try again in a few minutes.",
"backup_store_full": "The backup store is full, so manual backups are paused — ask an administrator to free space.",
+1 -1
View File
@@ -18,7 +18,7 @@
"no_backup": "这台服务器暂时没有可回档的备份。",
"backup_corrupt": "这份备份回读校验未通过,已无法完整恢复——请选择另一份备份。",
"not_stopped": "回档会覆盖世界的实时存储卷,请先把服务器完全停止再回档。",
"maintenance_in_progress": "这台服务器的世界正在回档、备份或写入文件——等它完成后再试,通常不超过一两分钟。",
"maintenance_in_progress": "这台服务器的世界正在回档、备份、写入文件或闲置回收——等它完成后再试。回档、备份和写文件通常一两分钟,闲置回收视世界大小可能更久。",
"no_world_volume": "这台服务器还没有世界卷——先启动一次让它创建,然后再试。",
"backup_cooldown": "这台服务器刚备份过,手动备份之间有冷却时间——请过几分钟再试。",
"backup_store_full": "备份存储已满,暂时无法手动备份——请联系管理员清理空间。",