diff --git a/cmd/felis/reaper.go b/cmd/felis/reaper.go index f82aa30..f9fde57 100644 --- a/cmd/felis/reaper.go +++ b/cmd/felis/reaper.go @@ -138,8 +138,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int { // FelisWorldJobFailed rule) reach the operator: a world that cannot be archived // is kept, and without this nobody would learn that it is never reaped. func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int { - fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n", - sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.Warned, sum.Skipped, sum.StoreFull, + fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n", + sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull, sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed, sum.Verified, sum.Corrupt, sum.VerifyFailed, sum.Swept, sum.OrphanArchives) if !sum.Failed() { diff --git a/cmd/felis/reaper_test.go b/cmd/felis/reaper_test.go index e4c2c12..e8caa00 100644 --- a/cmd/felis/reaper_test.go +++ b/cmd/felis/reaper_test.go @@ -27,6 +27,7 @@ func TestReportReaperRunFailsTheJob(t *testing.T) { want int }{ {"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0}, + {"waiting for a stop", reaper.Summary{Evaluated: 3, AwaitingStop: 1}, 0}, {"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1}, {"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1}, {"expiry failed", reaper.Summary{Evaluated: 3, ExpireFailed: 2}, 1}, diff --git a/internal/api/maintenance.go b/internal/api/maintenance.go index 52e32af..eea54f6 100644 --- a/internal/api/maintenance.go +++ b/internal/api/maintenance.go @@ -37,6 +37,8 @@ func maintenanceLabel(kind string) string { return "a backup" case maintenance.KindFileWrite: return "a file write" + case maintenance.KindReap: + return "the idle-world reaper" } return "another operation" } diff --git a/internal/maintenance/maintenance.go b/internal/maintenance/maintenance.go index 7b13838..913bb5b 100644 --- a/internal/maintenance/maintenance.go +++ b/internal/maintenance/maintenance.go @@ -18,7 +18,8 @@ // conflict and re-checks. // // A lock older than Grace with no Job behind it is stale (felis-api died between -// the two writes) and holds nothing. +// the two writes) and holds nothing. The reaper is the one holder without a Job: +// it keeps its lock fresh by rewriting it while it archives and reclaims a world. // // A restore that starts with a safety snapshot is two Jobs in a row: the backup // Job carries the restore to run after it (LabelThenRestore), and felis-api @@ -83,6 +84,10 @@ const ( KindRestore = "restore" KindBackup = "backup" KindFileWrite = "file-write" + // KindReap is the reaper archiving an idle world and reclaiming its volume. + // It runs no Job: the reaper holds the Annotation itself and rewrites it + // well inside Grace for as long as it works on the world. + KindReap = "reap" ) // FilesModeWrite is the LabelFilesMode value of a file write. diff --git a/internal/pgint/reaper_test.go b/internal/pgint/reaper_test.go index 88e3db8..590dbfd 100644 --- a/internal/pgint/reaper_test.go +++ b/internal/pgint/reaper_test.go @@ -33,7 +33,11 @@ func (c *reclaimCluster) DeletePVC(_ context.Context, pvc string) error { return nil } -func (c *reclaimCluster) Stop(context.Context, string) error { return nil } +func (c *reclaimCluster) HoldWorld(ctx context.Context, _ string) (context.Context, func(), error) { + return ctx, func() {}, nil +} + +func (c *reclaimCluster) WorldExists(context.Context, string) (bool, error) { return true, nil } type reclaimArchiver struct{ archived []string } @@ -387,3 +391,36 @@ func TestBackupReadBack(t *testing.T) { t.Fatalf("live refs after deleting %s still claim its archive", newer) } } + +// TestRestartClock: an idle server with no world to reclaim gets its clock and +// warnings reset, and keeps its owner and resource cache (data-durability-19). +func TestRestartClock(t *testing.T) { + ctx := context.Background() + st := reaper.NewPGStore(db) + name := "clock-" + suffix(t) + old := time.Now().Add(-20 * reaper.Day).UTC().Truncate(time.Microsecond) + if _, err := db.ExecContext(ctx, + `INSERT INTO servers (name, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, 100, 128, 1)`, + name); err != nil { + t.Fatalf("seed server: %v", err) + } + if _, err := db.ExecContext(ctx, + `UPDATE servers SET last_active_at = $2, warned_3d_at = $2, warned_1d_at = $2 WHERE name = $1`, name, old); err != nil { + t.Fatalf("age server: %v", err) + } + at := time.Now().UTC().Truncate(time.Microsecond) + if err := st.RestartClock(ctx, name, at); err != nil { + t.Fatalf("RestartClock: %v", err) + } + var last time.Time + var w3, w1 sql.NullTime + var cpu int + if err := db.QueryRowContext(ctx, + `SELECT last_active_at, warned_3d_at, warned_1d_at, cached_cpu_milli FROM servers WHERE name = $1`, name). + Scan(&last, &w3, &w1, &cpu); err != nil { + t.Fatalf("read back: %v", err) + } + if !last.Equal(at) || w3.Valid || w1.Valid || cpu != 100 { + t.Fatalf("after RestartClock: last_active_at=%v warned=%v/%v cpu=%d; want %v, cleared, cache kept", last, w3, w1, cpu, at) + } +} diff --git a/internal/platform/rbac.go b/internal/platform/rbac.go index 2b1d1ee..838c04e 100644 --- a/internal/platform/rbac.go +++ b/internal/platform/rbac.go @@ -175,9 +175,14 @@ func OperatorRole(p Params) *rbacv1.Role { // ReaperRole grants felis-reaper its two destructive, disjoint powers // (internal/reaper.k8scluster): patch a MinecraftServer to Stop it and delete its // world PVC. Candidate servers come from the Postgres store, not a cluster List, -// so no list/watch is needed; the reaper uses a direct client. It can read+patch -// minecraftservers but cannot create them, and holds no power over StatefulSets, -// Services, or Secrets — those belong to the operator and api. +// so minecraftservers need no list/watch; the reaper uses a direct client. It can +// read+patch minecraftservers but cannot create them, and holds no power over +// StatefulSets, Services, or Secrets — those belong to the operator and api. +// +// The same patch holds the world maintenance lock while a world is archived and +// reclaimed (internal/maintenance). Taking it needs the two reads felis-api makes +// before a restore: list pods (is the game pod gone) and list jobs (does a +// restore, backup or file write hold the world). Both are list-only. // // persistentvolumeclaims also carries get: resolving where a world lives // (cmd/felis/reaper.resolveWorldDir) reads the PVC's volumeName to derive the @@ -195,6 +200,8 @@ func ReaperRole(p Params) *rbacv1.Role { return role(p.MinecraftNamespace, "felis-reaper", ComponentReaper, []rbacv1.PolicyRule{ rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "patch"}), rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get", "delete"}), + rule([]string{groupCore}, []string{"pods"}, []string{"list"}), + rule([]string{groupBatch}, []string{"jobs"}, []string{"list"}), }) } diff --git a/internal/platform/rbac_test.go b/internal/platform/rbac_test.go index 178ddf7..7f8b6cb 100644 --- a/internal/platform/rbac_test.go +++ b/internal/platform/rbac_test.go @@ -221,12 +221,23 @@ func TestReaperRole_ScopeExact(t *testing.T) { t.Errorf("reaper must NOT touch %s/%s (operator/api territory)", res.group, res.name) } } - // The reaper never lists from the cluster (candidates come from Postgres). + // The reaper never lists servers from the cluster (candidates come from Postgres). for _, v := range []string{"list", "watch"} { if hasRule(rp, groupFelis, "minecraftservers", v) { t.Errorf("reaper must NOT %s minecraftservers (candidates come from the store)", v) } } + // Taking the world lock reads the game pod and the maintenance Jobs, list only. + if !hasRule(rp, groupCore, "pods", "list") || !hasRule(rp, groupBatch, "jobs", "list") { + t.Error("reaper must list pods and jobs to take the world maintenance lock") + } + for _, res := range []struct{ group, name string }{{groupCore, "pods"}, {groupBatch, "jobs"}} { + for _, v := range []string{"get", "watch", "create", "update", "patch", "delete"} { + if hasRule(rp, res.group, res.name, v) { + t.Errorf("reaper must NOT %s %s (list-only)", v, res.name) + } + } + } } // TestReaperRBAC_GatedOnRetention locks the reaper identity to its consumer: the diff --git a/internal/reaper/k8scluster.go b/internal/reaper/k8scluster.go index 1fe7cda..b5f0c31 100644 --- a/internal/reaper/k8scluster.go +++ b/internal/reaper/k8scluster.go @@ -2,13 +2,20 @@ package reaper import ( "context" + "errors" + "fmt" + "strings" + "time" "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/maintenance" "felis.lolicon.best/internal/naming" + batchv1 "k8s.io/api/batch/v1" corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/util/retry" "sigs.k8s.io/controller-runtime/pkg/client" ) @@ -20,13 +27,20 @@ func WorldPVCName(server string) string { return naming.WorldPVCName(server) } +// gamePodComponent is the operator's component label on a game server's pod +// (internal/operator.ComponentValue; k8scluster_test pins the two). +const gamePodComponent = "server" + // K8sCluster is the production Cluster backed by a controller-runtime client -// (spec §4, §18). It reads spec.reaperExempt, deletes the world PVC, and flips -// spec.desiredState to Stopped — nothing else. It is integration-tested against -// a live cluster, not the hermetic reaper_test.go suite. +// (spec §4, §18). It reads spec.reaperExempt, stops a server and holds its world +// volume through the maintenance lock, and deletes the world PVC — nothing else. type K8sCluster struct { c client.Client namespace string + now func() time.Time + // beat is how often a held lock is rewritten; maintenance.Grace/4 unless a + // test shortens it. + beat time.Duration } // NewK8sCluster builds a Cluster over c, scoped to namespace. @@ -34,17 +48,177 @@ func NewK8sCluster(c client.Client, namespace string) *K8sCluster { return &K8sCluster{c: c, namespace: namespace} } +func (k *K8sCluster) clock() time.Time { + if k.now != nil { + return k.now() + } + return time.Now() +} + func (k *K8sCluster) Inspect(ctx context.Context, name string) (ServerCRD, error) { var ms v1alpha1.MinecraftServer - if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil { - if apierrors.IsNotFound(err) { - return ServerCRD{}, ErrNotFound - } + if err := k.get(ctx, name, &ms); err != nil { return ServerCRD{}, err } return ServerCRD{Exempt: ms.Spec.ReaperExempt, PVC: WorldPVCName(name)}, nil } +// HoldWorld implements Cluster. The lock is the same Annotation felis-api +// writes before a restore, backup or file-write Job (internal/maintenance): a +// wake through felis-api, the operator scaling the server up, and another +// world operation all refuse while it is fresh. With no Job behind it, it +// holds for maintenance.Grace after each write, so it is rewritten every beat; +// the held context is cancelled once a rewrite has failed for Grace/2, well +// before any reader could take the lock as lapsed. +func (k *K8sCluster) HoldWorld(ctx context.Context, name string) (context.Context, func(), error) { + if err := k.lock(ctx, name); err != nil { + return nil, nil, err + } + held, cancel := context.WithCancelCause(ctx) + done := make(chan struct{}) + go k.keep(held, cancel, name, done) + release := func() { + cancel(nil) + <-done + rctx, stop := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second) + defer stop() + _ = k.unlock(rctx, name) + } + return held, release, nil +} + +// lock stops the server if it is still meant to run and takes the reap lock +// once it is fully down and nothing else holds its world. +func (k *K8sCluster) lock(ctx context.Context, name string) error { + return retry.RetryOnConflict(retry.DefaultRetry, func() error { + var ms v1alpha1.MinecraftServer + if err := k.get(ctx, name, &ms); err != nil { + return err + } + if ms.Spec.DesiredState != "" && ms.Spec.DesiredState != v1alpha1.DesiredStopped { + patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{}) + ms.Spec.DesiredState = v1alpha1.DesiredStopped + if err := k.c.Patch(ctx, &ms, patch); err != nil { + return err + } + return fmt.Errorf("%w: it was still up and has been told to stop", ErrNotQuiet) + } + if ms.Status.Ready || (ms.Status.Phase != "" && ms.Status.Phase != v1alpha1.PhaseStopped) { + return fmt.Errorf("%w: it is still stopping (phase %s)", ErrNotQuiet, ms.Status.Phase) + } + var pods corev1.PodList + if err := k.c.List(ctx, &pods, client.InNamespace(k.namespace), client.MatchingLabels{ + v1alpha1.LabelServer: name, v1alpha1.LabelComponent: gamePodComponent, + }); err != nil { + return err + } + if len(pods.Items) > 0 { + return fmt.Errorf("%w: its game pod is still shutting down", ErrNotQuiet) + } + var jobs batchv1.JobList + if err := k.c.List(ctx, &jobs, client.InNamespace(k.namespace), + client.MatchingLabels{maintenance.LabelServer: name}); err != nil { + return err + } + if kind, held := maintenance.Holder(name, ms.Annotations, jobs.Items, k.clock()); held { + return fmt.Errorf("%w: a %s holds its world", ErrNotQuiet, kind) + } + patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{}) + if ms.Annotations == nil { + ms.Annotations = map[string]string{} + } + ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindReap, k.clock()) + return k.c.Patch(ctx, &ms, patch) + }) +} + +// errLockLost reports that the lock on the object is no longer the reaper's. +var errLockLost = errors.New("the reap lock is gone from the server") + +// keep rewrites the lock every beat until held ends, and ends held itself once +// it has not managed to for Grace/2. +func (k *K8sCluster) keep(held context.Context, cancel context.CancelCauseFunc, name string, done chan<- struct{}) { + defer close(done) + beat := k.beat + if beat <= 0 { + beat = maintenance.Grace / 4 + } + t := time.NewTicker(beat) + defer t.Stop() + last := k.clock() + for { + select { + case <-held.Done(): + return + case <-t.C: + } + err := k.refresh(held, name) + if err == nil { + last = k.clock() + continue + } + if held.Err() != nil { + return + } + if errors.Is(err, errLockLost) || k.clock().Sub(last) >= maintenance.Grace/2 { + cancel(fmt.Errorf("reaper: lost the world lock on %s: %w", name, err)) + return + } + } +} + +func (k *K8sCluster) refresh(ctx context.Context, name string) error { + return retry.RetryOnConflict(retry.DefaultRetry, func() error { + var ms v1alpha1.MinecraftServer + if err := k.get(ctx, name, &ms); err != nil { + return err + } + if !reapLock(ms.Annotations) { + return errLockLost + } + patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{}) + ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindReap, k.clock()) + return k.c.Patch(ctx, &ms, patch) + }) +} + +// unlock drops the lock if it is still the reaper's; a lock left behind lapses +// after maintenance.Grace on its own. +func (k *K8sCluster) unlock(ctx context.Context, name string) error { + return retry.RetryOnConflict(retry.DefaultRetry, func() error { + var ms v1alpha1.MinecraftServer + if err := k.get(ctx, name, &ms); err != nil { + if errors.Is(err, ErrNotFound) { + return nil + } + return err + } + if !reapLock(ms.Annotations) { + return nil + } + patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{}) + delete(ms.Annotations, maintenance.Annotation) + return k.c.Patch(ctx, &ms, patch) + }) +} + +func reapLock(annotations map[string]string) bool { + return strings.HasPrefix(annotations[maintenance.Annotation], maintenance.KindReap+"@") +} + +// WorldExists reports whether the world PersistentVolumeClaim exists. +func (k *K8sCluster) WorldExists(ctx context.Context, pvc string) (bool, error) { + var claim corev1.PersistentVolumeClaim + err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: pvc}, &claim) + switch { + case apierrors.IsNotFound(err): + return false, nil + case err != nil: + return false, err + } + return true, nil +} + // DeletePVC deletes the world PersistentVolumeClaim. A missing PVC is not an // error: the reap is idempotent and a re-run after a partial failure must still // converge. @@ -58,17 +232,12 @@ func (k *K8sCluster) DeletePVC(ctx context.Context, pvc string) error { return nil } -// Stop sets spec.desiredState=Stopped with a merge patch so a concurrent status -// write by the operator is never clobbered (spec §9.1). -func (k *K8sCluster) Stop(ctx context.Context, name string) error { - var ms v1alpha1.MinecraftServer - if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil { +func (k *K8sCluster) get(ctx context.Context, name string, ms *v1alpha1.MinecraftServer) error { + if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, ms); err != nil { if apierrors.IsNotFound(err) { return ErrNotFound } return err } - patch := client.MergeFrom(ms.DeepCopy()) - ms.Spec.DesiredState = v1alpha1.DesiredStopped - return k.c.Patch(ctx, &ms, patch) + return nil } diff --git a/internal/reaper/k8scluster_test.go b/internal/reaper/k8scluster_test.go new file mode 100644 index 0000000..e6b61bb --- /dev/null +++ b/internal/reaper/k8scluster_test.go @@ -0,0 +1,218 @@ +package reaper + +import ( + "context" + "errors" + "strings" + "sync" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/maintenance" + "felis.lolicon.best/internal/operator" + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/types" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" +) + +// HoldWorld against a fake API server, which honours resourceVersion on the +// optimistic-lock patches the lock is written with. + +func holdServer(desired v1alpha1.DesiredState, phase v1alpha1.Phase) *v1alpha1.MinecraftServer { + return &v1alpha1.MinecraftServer{ + ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"}, + Spec: v1alpha1.MinecraftServerSpec{DesiredState: desired}, + Status: v1alpha1.MinecraftServerStatus{Phase: phase}, + } +} + +// fakeClock is a clock the tests and the heartbeat goroutine share. +type fakeClock struct { + mu sync.Mutex + t time.Time +} + +func (c *fakeClock) now() time.Time { c.mu.Lock(); defer c.mu.Unlock(); return c.t } +func (c *fakeClock) add(d time.Duration) { + c.mu.Lock() + defer c.mu.Unlock() + c.t = c.t.Add(d) +} + +func holdCluster(t *testing.T, objs ...client.Object) (*K8sCluster, client.Client, *fakeClock) { + t.Helper() + scheme := runtime.NewScheme() + if err := clientgoscheme.AddToScheme(scheme); err != nil { + t.Fatalf("scheme: %v", err) + } + if err := v1alpha1.AddToScheme(scheme); err != nil { + t.Fatalf("scheme: %v", err) + } + c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(objs...). + WithStatusSubresource(&v1alpha1.MinecraftServer{}).Build() + clk := &fakeClock{t: time.Date(2026, 9, 24, 12, 0, 0, 0, time.UTC)} + k := NewK8sCluster(c, "minecraft") + k.now = clk.now + k.beat = time.Hour + return k, c, clk +} + +func serverState(t *testing.T, c client.Client) (map[string]string, v1alpha1.DesiredState) { + t.Helper() + var ms v1alpha1.MinecraftServer + if err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival"}, &ms); err != nil { + t.Fatalf("get: %v", err) + } + return ms.Annotations, ms.Spec.DesiredState +} + +func TestHoldWorldStopsARunningServer(t *testing.T) { + k, c, _ := holdCluster(t, holdServer(v1alpha1.DesiredRunning, v1alpha1.PhaseRunning)) + _, _, err := k.HoldWorld(context.Background(), "survival") + if !errors.Is(err, ErrNotQuiet) { + t.Fatalf("HoldWorld = %v, want ErrNotQuiet", err) + } + ann, desired := serverState(t, c) + if desired != v1alpha1.DesiredStopped || ann[maintenance.Annotation] != "" { + t.Fatalf("desired=%q annotations=%v: want told to stop and no lock yet", desired, ann) + } +} + +func TestHoldWorldWaitsUntilQuiet(t *testing.T) { + ready := holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped) + ready.Status.Ready = true + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "survival-0", Namespace: "minecraft", + Labels: map[string]string{v1alpha1.LabelServer: "survival", v1alpha1.LabelComponent: gamePodComponent}}} + job := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Name: "restore-survival", Namespace: "minecraft", + Labels: map[string]string{maintenance.LabelServer: "survival", maintenance.LabelManagedBy: "felis-restore"}}} + locked := holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped) + locked.Annotations = map[string]string{ + maintenance.Annotation: maintenance.LockValue(maintenance.KindBackup, time.Date(2026, 9, 24, 11, 59, 30, 0, time.UTC)), + } + for _, tc := range []struct { + name string + objs []client.Object + why string + }{ + {"stopping", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopping)}, "still stopping"}, + {"ready", []client.Object{ready}, "still stopping"}, + {"game pod", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pod}, "game pod"}, + {"restore job", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), job}, "a restore"}, + {"admission lock", []client.Object{locked}, "a backup"}, + } { + t.Run(tc.name, func(t *testing.T) { + k, c, _ := holdCluster(t, tc.objs...) + before, _ := serverState(t, c) + _, _, err := k.HoldWorld(context.Background(), "survival") + if !errors.Is(err, ErrNotQuiet) || !strings.Contains(err.Error(), tc.why) { + t.Fatalf("HoldWorld = %v, want ErrNotQuiet naming %q", err, tc.why) + } + if after, _ := serverState(t, c); after[maintenance.Annotation] != before[maintenance.Annotation] { + t.Fatalf("lock changed to %q", after[maintenance.Annotation]) + } + }) + } +} + +// Once held, every other holder check sees the reaper, and release lets go. +func TestHoldWorldLocksAndReleases(t *testing.T) { + k, c, clk := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)) + held, release, err := k.HoldWorld(context.Background(), "survival") + if err != nil { + t.Fatalf("HoldWorld: %v", err) + } + ann, _ := serverState(t, c) + if kind, ok := maintenance.Holder("survival", ann, nil, clk.now()); !ok || kind != maintenance.KindReap { + t.Fatalf("Holder = (%q, %v) with %v, want the reaper", kind, ok, ann) + } + release() + if held.Err() == nil { + t.Fatal("held context still live after release") + } + if ann, _ := serverState(t, c); ann[maintenance.Annotation] != "" { + t.Fatalf("lock left behind: %v", ann) + } +} + +// The lock is rewritten while held, so it never lapses under a long archive. +func TestHoldWorldKeepsLockFresh(t *testing.T) { + k, c, clk := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)) + k.beat = 5 * time.Millisecond + held, release, err := k.HoldWorld(context.Background(), "survival") + if err != nil { + t.Fatalf("HoldWorld: %v", err) + } + defer release() + clk.add(10 * maintenance.Grace) + want := maintenance.LockValue(maintenance.KindReap, clk.now()) + deadline := time.Now().Add(5 * time.Second) + for { + if ann, _ := serverState(t, c); ann[maintenance.Annotation] == want { + break + } + if time.Now().After(deadline) { + ann, _ := serverState(t, c) + t.Fatalf("lock = %q, want rewritten to %q", ann[maintenance.Annotation], want) + } + time.Sleep(5 * time.Millisecond) + } + if held.Err() != nil { + t.Fatalf("held ended: %v", context.Cause(held)) + } +} + +// A lock someone else removed (or replaced) ends the hold: the reaper must not +// delete a world it no longer holds. +func TestHoldWorldEndsWhenLockIsLost(t *testing.T) { + k, c, _ := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)) + k.beat = 5 * time.Millisecond + held, release, err := k.HoldWorld(context.Background(), "survival") + if err != nil { + t.Fatalf("HoldWorld: %v", err) + } + var ms v1alpha1.MinecraftServer + if err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival"}, &ms); err != nil { + t.Fatalf("get: %v", err) + } + patch := client.MergeFrom(ms.DeepCopy()) + ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindRestore, time.Now()) + if err := c.Patch(context.Background(), &ms, patch); err != nil { + t.Fatalf("patch: %v", err) + } + select { + case <-held.Done(): + case <-time.After(5 * time.Second): + t.Fatal("held context still live after the lock was taken over") + } + if !errors.Is(context.Cause(held), errLockLost) { + t.Fatalf("cause = %v, want errLockLost", context.Cause(held)) + } + release() + if ann, _ := serverState(t, c); !strings.HasPrefix(ann[maintenance.Annotation], maintenance.KindRestore+"@") { + t.Fatalf("release removed another holder's lock: %v", ann) + } +} + +func TestWorldExists(t *testing.T) { + pvc := &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: "world-survival-0", Namespace: "minecraft"}} + k, _, _ := holdCluster(t, pvc) + for name, want := range map[string]bool{"world-survival-0": true, "world-other-0": false} { + if got, err := k.WorldExists(context.Background(), name); err != nil || got != want { + t.Errorf("WorldExists(%s) = (%v, %v), want %v", name, got, err, want) + } + } +} + +// gamePodComponent is a copy of the operator's pod label value; a drift would +// let the reaper archive a world beside a server still saving it. +func TestGamePodComponentMatchesOperator(t *testing.T) { + if gamePodComponent != operator.ComponentValue { + t.Fatalf("gamePodComponent = %q, operator labels its pods %q", gamePodComponent, operator.ComponentValue) + } +} diff --git a/internal/reaper/pgstore.go b/internal/reaper/pgstore.go index 64c38b3..93b25bf 100644 --- a/internal/reaper/pgstore.go +++ b/internal/reaper/pgstore.go @@ -90,6 +90,13 @@ func (s *PGStore) ReleaseWorld(ctx context.Context, name string, at time.Time) e return err } +func (s *PGStore) RestartClock(ctx context.Context, name string, at time.Time) error { + const q = `UPDATE servers SET last_active_at = $2, warned_3d_at = NULL, warned_1d_at = NULL + WHERE name = $1 AND deleted_at IS NULL` + _, err := s.db.ExecContext(ctx, q, name, at) + return err +} + func (s *PGStore) MarkWarned(ctx context.Context, name string, tier Tier, at time.Time) error { // The column is one of two fixed identifiers, never user input. col := "warned_3d_at" diff --git a/internal/reaper/reaper.go b/internal/reaper/reaper.go index dce4f4d..4dd5633 100644 --- a/internal/reaper/reaper.go +++ b/internal/reaper/reaper.go @@ -55,6 +55,11 @@ const ( // delete world data it cannot first inspect for the exemption flag. var ErrNotFound = errors.New("reaper: server not found") +// ErrNotQuiet is returned by Cluster.HoldWorld while the server is not fully +// down yet, or another operation (a restore, backup or file write) holds its +// world. The world is left alone and the server is tried again on the next run. +var ErrNotQuiet = errors.New("reaper: world not quiet") + // errStoreFull is an internal sentinel: the backup store is at capacity and // could not be freed, so the world is preserved rather than deleted without a // backup (red line ④). It is never returned to callers. @@ -219,6 +224,10 @@ type Store interface { // row (red line ②). ReleaseWorld(ctx context.Context, name string, at time.Time) error + // RestartClock sets last_active_at→at and clears warned_* on a server with + // no world to reclaim, so it is not found idle again every run. + RestartClock(ctx context.Context, name string, at time.Time) error + // MarkWarned stamps the warned_3d_at / warned_1d_at column for tier. MarkWarned(ctx context.Context, name string, tier Tier, at time.Time) error @@ -255,17 +264,25 @@ type Store interface { Audit(ctx context.Context, rec AuditRecord) error } -// Cluster is the lifecycle (Kubernetes) face: read the CRD, delete the world -// PVC, and flip desiredState to Stopped. These are the only cluster operations -// §18 performs. +// Cluster is the lifecycle (Kubernetes) face: read the CRD, stop the server +// and hold its world, and delete the world PVC. These are the only cluster +// operations §18 performs. type Cluster interface { // Inspect returns the exemption flag and world PVC name for a server, or // ErrNotFound if the CRD is gone. Inspect(ctx context.Context, name string) (ServerCRD, error) + // HoldWorld keeps everything else off the server's world until release is + // called: it sets desiredState=Stopped if the server is still meant to run, + // and once the server is fully down (not ready, phase Stopped, no game pod) + // and no restore, backup or file write holds the world, it takes the world + // maintenance lock (internal/maintenance, KindReap) and keeps it fresh. + // Until then it returns ErrNotQuiet. The returned context ends if the lock + // cannot be kept; the world must not be read or deleted on it after that. + HoldWorld(ctx context.Context, name string) (held context.Context, release func(), err error) + // WorldExists reports whether the world PersistentVolumeClaim exists. + WorldExists(ctx context.Context, pvc string) (bool, error) // DeletePVC deletes the world PersistentVolumeClaim. DeletePVC(ctx context.Context, pvc string) error - // Stop sets spec.desiredState=Stopped. - Stop(ctx context.Context, name string) error } // Warner delivers an impending-reap notice. It is optional and best-effort: a @@ -309,6 +326,10 @@ type Summary struct { // AwaitingOffsite are idle worlds that are archived and kept until the // archive's off-site copy lands. AwaitingOffsite int + // AwaitingStop are idle servers left for the next run because they were not + // fully down yet (the run told a running one to stop) or another operation + // held their world. + AwaitingStop int // Verified are archives read back in full and found matching; Corrupt are // the ones that were not (marked, and never reused or restored from); // VerifyFailed are the ones that could not be read back at all this run. @@ -379,6 +400,11 @@ func (r *Reaper) RunOnce(ctx context.Context) (Summary, error) { for _, c := range cands { sum.Evaluated++ if err := r.evaluate(ctx, now, offs, c, &sum); err != nil { + if errors.Is(err, ErrNotQuiet) { + sum.AwaitingStop++ + r.log().Warn("reaper: idle world not quiet, retrying next run", "server", c.Name, "why", err) + continue + } r.log().Error("reaper: skipping server", "server", c.Name, "err", err) sum.Skipped++ if errors.Is(err, errStoreFull) { @@ -419,9 +445,27 @@ func (r *Reaper) evaluate(ctx context.Context, now time.Time, offs []time.Durati return nil } -// reap archives the world, records the backup, and only then deletes the PVC, -// releases ownership, and stops the server — the strict ordering of red line ④. +// reap stops the server and holds its world, archives the world, records the +// backup, and only then deletes the PVC and releases ownership — the strict +// ordering of red line ④. Nothing can start the server or touch its world while +// it is held, so the archive is of a world at rest and the PVC deleted is the +// one archived. func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd ServerCRD, sum *Summary) error { + held, release, err := r.Cluster.HoldWorld(ctx, c.Name) + if err != nil { + return fmt.Errorf("hold world: %w", err) + } + defer release() + ctx = held + + exists, err := r.Cluster.WorldExists(ctx, crd.PVC) + if err != nil { + return fmt.Errorf("look up world volume: %w", err) + } + if !exists { + return r.reapNoWorld(ctx, now, c, sum) + } + // §26 soft cap: free space before adding a backup. If the store cannot be // brought under cap, preserve the world rather than delete it unbacked. if r.Cfg.MaxLocalBytes > 0 { @@ -497,32 +541,66 @@ func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd Serve } // World is safely archived and recorded — now (and only now) delete it. + // The lock must still be held: a lapsed one could have let the server start. + if err := context.Cause(ctx); err != nil { + return fmt.Errorf("world no longer held: %w", err) + } if err := r.Cluster.DeletePVC(ctx, crd.PVC); err != nil { // The backup row persists; next run's FreshBackup reuses it and retries // the delete, so no duplicate archive is created. return fmt.Errorf("delete pvc: %w", err) } + return r.finishReap(ctx, now, c, ref, sum) +} + +// finishReap releases a world whose PVC is gone and records the reap. +func (r *Reaper) finishReap(ctx context.Context, now time.Time, c Candidate, ref string, sum *Summary) error { if err := r.Store.ReleaseWorld(ctx, c.Name, now); err != nil { return fmt.Errorf("release world: %w", err) } - if err := r.Cluster.Stop(ctx, c.Name); err != nil { - // The world is already deleted and ownership released; the desiredState - // flip is cosmetic by comparison. Log, but the reap stands. - r.log().Error("reaper: set desiredState=Stopped failed", "server", c.Name, "err", err) - } if err := r.Store.Audit(ctx, AuditRecord{Action: ActionReapWorld, ServerName: c.Name, FormerOwner: c.OwnerID}); err != nil { r.log().Error("reaper: audit reap_world failed", "server", c.Name, "err", err) } sum.WorldsReaped++ // felis_reaper_worlds_deleted_total (spec §23) advances in lockstep with the - // per-run Summary tally — incremented here, at the one point a world's PVC has - // actually been deleted, not at evaluation time. + // per-run Summary tally — incremented once per world whose PVC went, not at + // evaluation time. metrics.ReaperWorldsDeletedTotal.Inc() r.log().Info("reaper: world reaped", "server", c.Name, "former_owner", c.OwnerID, "backup_ref", ref) return nil } +// reapNoWorld handles an idle server whose world PVC does not exist. With an +// archive of the current world on record, an earlier run deleted the PVC and +// stopped before releasing it, so the reap is finished now. Otherwise there was +// no world to reclaim: an owner who never started the server gives it up +// (nothing to back up), and an unowned one — typically a world reaped earlier — +// only has its clock restarted, so it is not reaped over and over. +func (r *Reaper) reapNoWorld(ctx context.Context, now time.Time, c Candidate, sum *Summary) error { + fresh, ok, err := r.Store.FreshBackup(ctx, c.Name, c.LastActiveAt) + if err != nil { + return fmt.Errorf("lookup fresh backup: %w", err) + } + if ok { + return r.finishReap(ctx, now, c, fresh.Ref, sum) + } + if c.OwnerID == "" { + if err := r.Store.RestartClock(ctx, c.Name, now); err != nil { + return fmt.Errorf("restart clock: %w", err) + } + return nil + } + if err := r.Store.ReleaseWorld(ctx, c.Name, now); err != nil { + return fmt.Errorf("release world: %w", err) + } + if err := r.Store.Audit(ctx, AuditRecord{Action: ActionReapWorld, ServerName: c.Name, FormerOwner: c.OwnerID}); err != nil { + r.log().Error("reaper: audit reap_world failed", "server", c.Name, "err", err) + } + r.log().Info("reaper: idle server released; it had no world to archive", "server", c.Name, "former_owner", c.OwnerID) + return nil +} + // ensureCapacity frees the backup store down under MaxLocalBytes by evicting the // oldest present backups early. Early eviction is destructive (it removes // not-yet-expired backups), so each eviction is alerted and audited. It returns diff --git a/internal/reaper/reaper_test.go b/internal/reaper/reaper_test.go index 2405daa..dc2aa61 100644 --- a/internal/reaper/reaper_test.go +++ b/internal/reaper/reaper_test.go @@ -24,7 +24,7 @@ var testNow = time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) func idleBy(d time.Duration) time.Time { return testNow.Add(-d) } // recorder captures the cross-fake call order so a test can assert the strict -// archive→insert→deletePVC→release→stop→audit sequence of red line ④. +// hold→archive→insert→deletePVC→release→audit→unhold sequence of red line ④. type recorder struct{ events []string } func (r *recorder) add(e string) { r.events = append(r.events, e) } @@ -33,6 +33,7 @@ func (r *recorder) add(e string) { r.events = append(r.events, e) } type fakeArchiver struct { rec *recorder + onArchive func() archiveErr error deleteErr error archives int @@ -47,6 +48,9 @@ func (f *fakeArchiver) Archive(_ context.Context, server, _ string) (backup.Arch f.archives++ f.seq++ f.rec.add("archive") + if f.onArchive != nil { + f.onArchive() + } ref := fmt.Sprintf("ref-%s-%d", server, f.seq) return backup.Archived{Ref: backup.ArchiveRef(ref), Size: 10, SHA256: "sha-" + ref}, nil } @@ -104,7 +108,39 @@ type fakeCluster struct { deletePVCErr error deletePVCCalls int deletedPVCs []string - stopped []string + + notQuiet map[string]bool // HoldWorld refuses these with ErrNotQuiet + holdErr error // HoldWorld fails with this + noWorld map[string]bool // PVCs that do not exist + held map[string]bool // servers held right now + holds []string + lost context.CancelCauseFunc +} + +func (c *fakeCluster) HoldWorld(ctx context.Context, name string) (context.Context, func(), error) { + if c.holdErr != nil { + return nil, nil, c.holdErr + } + if c.notQuiet[name] { + return nil, nil, fmt.Errorf("%w: still stopping", ErrNotQuiet) + } + if c.held == nil { + c.held = map[string]bool{} + } + c.held[name] = true + c.holds = append(c.holds, name) + c.rec.add("hold") + held, cancel := context.WithCancelCause(ctx) + c.lost = cancel + return held, func() { + cancel(nil) + delete(c.held, name) + c.rec.add("unhold") + }, nil +} + +func (c *fakeCluster) WorldExists(_ context.Context, pvc string) (bool, error) { + return !c.noWorld[pvc], nil } func (c *fakeCluster) Inspect(_ context.Context, name string) (ServerCRD, error) { @@ -118,22 +154,19 @@ func (c *fakeCluster) Inspect(_ context.Context, name string) (ServerCRD, error) return crd, nil } -func (c *fakeCluster) DeletePVC(_ context.Context, pvc string) error { +func (c *fakeCluster) DeletePVC(ctx context.Context, pvc string) error { c.deletePVCCalls++ if c.deletePVCErr != nil { return c.deletePVCErr } + if len(c.held) == 0 || ctx.Err() != nil { + return fmt.Errorf("deleted %s without holding its world", pvc) + } c.deletedPVCs = append(c.deletedPVCs, pvc) c.rec.add("deletePVC") return nil } -func (c *fakeCluster) Stop(_ context.Context, name string) error { - c.stopped = append(c.stopped, name) - c.rec.add("stop") - return nil -} - // ---- fake Store ----------------------------------------------------------- type fakeBackup struct { @@ -212,6 +245,15 @@ func (s *fakeStore) ReleaseWorld(_ context.Context, name string, at time.Time) e return nil } +func (s *fakeStore) RestartClock(_ context.Context, name string, at time.Time) error { + c := s.byName[name] + c.LastActiveAt = at + c.Warned3dAt = time.Time{} + c.Warned1dAt = time.Time{} + s.rec.add("restartClock") + return nil +} + func (s *fakeStore) MarkWarned(_ context.Context, name string, tier Tier, at time.Time) error { c := s.byName[name] if tier == Tier1d { @@ -405,7 +447,7 @@ func TestReapIdleWorldFullSequence(t *testing.T) { if sum.WorldsReaped != 1 { t.Fatalf("WorldsReaped = %d, want 1", sum.WorldsReaped) } - want := []string{"archive", "insert", "deletePVC", "release", "stop", "audit:" + ActionReapWorld} + want := []string{"hold", "archive", "insert", "deletePVC", "release", "audit:" + ActionReapWorld, "unhold"} if !reflect.DeepEqual(st.rec.events, want) { t.Fatalf("call order = %v, want %v", st.rec.events, want) } @@ -936,7 +978,7 @@ func TestReapReadsBackReusedArchive(t *testing.T) { if sum.WorldsReaped != 1 || ar.archives != 0 || len(cl.deletedPVCs) != 1 { t.Fatalf("summary %+v archives=%d deleted=%v: want the checked archive reused", sum, ar.archives, cl.deletedPVCs) } - if st.rec.events[0] != "verify" || st.rec.events[1] != "deletePVC" { + if st.rec.events[1] != "verify" || st.rec.events[2] != "deletePVC" { t.Errorf("events = %v, want the read-back before the delete", st.rec.events) } if !st.backups[0].verifiedAt.Equal(testNow) || !reflect.DeepEqual(ca.verified, []string{"ref-old"}) { @@ -958,7 +1000,7 @@ func TestReapReplacesCorruptArchive(t *testing.T) { if sum.WorldsReaped != 1 || sum.Corrupt != 1 || !sum.Failed() { t.Fatalf("summary = %+v, want reaped with one corrupt archive reported", sum) } - want := []string{"verify", "corrupt", "archive", "insert", "deletePVC"} + want := []string{"hold", "verify", "corrupt", "archive", "insert", "deletePVC"} if !reflect.DeepEqual(st.rec.events[:len(want)], want) { t.Errorf("events = %v, want %v first", st.rec.events, want) } @@ -1057,3 +1099,126 @@ func TestSweepPass(t *testing.T) { t.Errorf("sweep ran without the recorded archives: %+v", sum) } } + +// data-durability-8: a server still up when its world goes idle is told to stop +// and left for the next run, never archived while it may be writing; once it is +// down the next run reaps it. +func TestReapWaitsForServerToStop(t *testing.T) { + r, st, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "busy", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)}) + cl.notQuiet = map[string]bool{"busy": true} + + sum := mustRun(t, r) + if sum.AwaitingStop != 1 || sum.Skipped != 0 || sum.Failed() || ar.archives != 0 || cl.deletePVCCalls != 0 { + t.Fatalf("summary %+v archives=%d deletes=%d: want the world left alone for the next run", sum, ar.archives, cl.deletePVCCalls) + } + if st.byName["busy"].OwnerID != "user-1" || len(st.audits) != 0 { + t.Fatalf("a server not yet down was released: %+v", st.byName["busy"]) + } + + cl.notQuiet = nil + if sum := mustRun(t, r); sum.WorldsReaped != 1 || sum.AwaitingStop != 0 { + t.Fatalf("second run = %+v, want reaped once the server is down", sum) + } +} + +// A hold that cannot be taken for any other reason is a failure of the run. +func TestReapHoldErrorSkips(t *testing.T) { + r, _, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "flaky", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)}) + cl.holdErr = errors.New("apiserver timeout") + + sum := mustRun(t, r) + if sum.Skipped != 1 || !sum.Failed() || ar.archives != 0 { + t.Fatalf("summary %+v archives=%d: want the server skipped and the run failed", sum, ar.archives) + } +} + +// The world is held from before the archive until after the reap is recorded, +// and let go on every path, a failed archive included. +func TestReapReleasesHoldOnEveryPath(t *testing.T) { + r, st, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "hotel", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)}) + ar.archiveErr = errors.New("disk full") + + mustRun(t, r) + if len(cl.held) != 0 || st.rec.events[len(st.rec.events)-1] != "unhold" { + t.Fatalf("hold not released after a failed archive: events=%v held=%v", st.rec.events, cl.held) + } +} + +// A lock lost while the archive is written (felis-api could then start the +// server) keeps the PVC; the archive is recorded and reused next run. +func TestReapLostHoldKeepsWorld(t *testing.T) { + r, st, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "india", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)}) + ar.onArchive = func() { cl.lost(errors.New("lock rewrite failing")) } + + sum := mustRun(t, r) + if sum.WorldsReaped != 0 || sum.Skipped != 1 || cl.deletePVCCalls != 0 { + t.Fatalf("summary %+v deletes=%d: want the world kept once the hold was lost", sum, cl.deletePVCCalls) + } + if st.byName["india"].OwnerID != "user-1" { + t.Fatal("ownership released after the hold was lost") + } + + ar.onArchive = nil + if sum := mustRun(t, r); sum.WorldsReaped != 1 || ar.archives != 1 { + t.Fatalf("retry = %+v archives=%d, want the recorded archive reused", sum, ar.archives) + } +} + +// data-durability-19: an unowned server with no world (reaped before) is not +// reaped again when its clock runs out; the clock restarts, with no audit and +// no count. +func TestReapUnownedWithoutWorldRestartsClock(t *testing.T) { + r, st, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "juliet", LastActiveAt: idleBy(16 * Day)}) + cl.noWorld = map[string]bool{"world-juliet-0": true} + + sum := mustRun(t, r) + if sum.WorldsReaped != 0 || sum.Skipped != 0 || sum.Failed() || ar.archives != 0 || cl.deletePVCCalls != 0 || len(st.audits) != 0 { + t.Fatalf("summary %+v archives=%d deletes=%d audits=%v: want only the clock restarted", + sum, ar.archives, cl.deletePVCCalls, st.audits) + } + if !st.byName["juliet"].LastActiveAt.Equal(testNow) { + t.Fatalf("last_active_at = %v, want restarted at %v", st.byName["juliet"].LastActiveAt, testNow) + } + if sum := mustRun(t, r); sum.WorldsReaped != 0 || len(st.audits) != 0 { + t.Fatalf("second run = %+v audits=%v", sum, st.audits) + } +} + +// An owner who never started the server gives it up at the deadline like any +// other; there is nothing to archive, and no world was deleted to count. +func TestReapOwnedWithoutWorldReleases(t *testing.T) { + r, st, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "kappa", OwnerID: "user-4", LastActiveAt: idleBy(16 * Day)}) + cl.noWorld = map[string]bool{"world-kappa-0": true} + + sum := mustRun(t, r) + if sum.WorldsReaped != 0 || sum.Skipped != 0 || ar.archives != 0 || cl.deletePVCCalls != 0 { + t.Fatalf("summary %+v archives=%d deletes=%d", sum, ar.archives, cl.deletePVCCalls) + } + if st.byName["kappa"].OwnerID != "" || len(st.audits) != 1 || st.audits[0].FormerOwner != "user-4" { + t.Fatalf("owner %q audits %+v: want released and audited", st.byName["kappa"].OwnerID, st.audits) + } +} + +// A run that deleted the PVC and failed before releasing the world is finished +// by the next: released, audited and counted once, without a new archive. +func TestReapFinishesInterruptedReap(t *testing.T) { + r, st, cl, ar := newReaper(DefaultConfig(), + Candidate{Name: "lambda", OwnerID: "user-5", LastActiveAt: idleBy(20 * Day)}) + cl.noWorld = map[string]bool{"world-lambda-0": true} + st.backups = []*fakeBackup{{id: "prev", server: "lambda", ref: "ref-prev", reason: ReasonInactive, size: 5, + status: "present", createdAt: idleBy(Day), expires: testNow.Add(89 * Day)}} + + sum := mustRun(t, r) + if sum.WorldsReaped != 1 || ar.archives != 0 || cl.deletePVCCalls != 0 { + t.Fatalf("summary %+v archives=%d deletes=%d: want the reap finished", sum, ar.archives, cl.deletePVCCalls) + } + if st.byName["lambda"].OwnerID != "" || len(st.audits) != 1 { + t.Fatalf("owner %q audits %+v", st.byName["lambda"].OwnerID, st.audits) + } +} diff --git a/panel/src/i18n/resources/en-US/errors.json b/panel/src/i18n/resources/en-US/errors.json index e8fdba0..e0e466d 100644 --- a/panel/src/i18n/resources/en-US/errors.json +++ b/panel/src/i18n/resources/en-US/errors.json @@ -18,7 +18,7 @@ "no_backup": "There's no restorable backup for this server yet.", "backup_corrupt": "This backup failed its read-back check and can't be restored intact — pick another backup.", "not_stopped": "Stop the server completely before restoring — a restore overwrites the live world volume.", - "maintenance_in_progress": "This server's world is busy with a restore, backup or file write — try again once it finishes, usually within a minute or two.", + "maintenance_in_progress": "This server's world is busy with a restore, backup, file write or idle reclaim — try again once it finishes. A restore, backup or file write usually takes a minute or two; an idle reclaim can take longer on a large world.", "no_world_volume": "This server has no world volume yet — start it once so it is created, then retry.", "backup_cooldown": "This server was backed up moments ago, and manual backups have a cooldown — try again in a few minutes.", "backup_store_full": "The backup store is full, so manual backups are paused — ask an administrator to free space.", diff --git a/panel/src/i18n/resources/zh-CN/errors.json b/panel/src/i18n/resources/zh-CN/errors.json index 3e16492..9e0b68d 100644 --- a/panel/src/i18n/resources/zh-CN/errors.json +++ b/panel/src/i18n/resources/zh-CN/errors.json @@ -18,7 +18,7 @@ "no_backup": "这台服务器暂时没有可回档的备份。", "backup_corrupt": "这份备份回读校验未通过,已无法完整恢复——请选择另一份备份。", "not_stopped": "回档会覆盖世界的实时存储卷,请先把服务器完全停止再回档。", - "maintenance_in_progress": "这台服务器的世界正在回档、备份或写入文件——等它完成后再试,通常不超过一两分钟。", + "maintenance_in_progress": "这台服务器的世界正在回档、备份、写入文件或闲置回收——等它完成后再试。回档、备份和写文件通常一两分钟,闲置回收视世界大小可能更久。", "no_world_volume": "这台服务器还没有世界卷——先启动一次让它创建,然后再试。", "backup_cooldown": "这台服务器刚备份过,手动备份之间有冷却时间——请过几分钟再试。", "backup_store_full": "备份存储已满,暂时无法手动备份——请联系管理员清理空间。",