fix(reaper): 回收前先停服并持有世界维护锁再归档,锁丢失不删卷;无世界卷的无主服务器只重置时钟不再重复回收
This commit is contained in:
14 files changed
+751
-51
No files matched your search
+2
-2
@@ -138,8 +138,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
// FelisWorldJobFailed rule) reach the operator: a world that cannot be archived
|
||||
// is kept, and without this nobody would learn that it is never reaped.
|
||||
func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
|
||||
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
|
||||
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed,
|
||||
sum.Verified, sum.Corrupt, sum.VerifyFailed, sum.Swept, sum.OrphanArchives)
|
||||
if !sum.Failed() {
|
||||
|
||||
@@ -27,6 +27,7 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
|
||||
want int
|
||||
}{
|
||||
{"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0},
|
||||
{"waiting for a stop", reaper.Summary{Evaluated: 3, AwaitingStop: 1}, 0},
|
||||
{"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1},
|
||||
{"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1},
|
||||
{"expiry failed", reaper.Summary{Evaluated: 3, ExpireFailed: 2}, 1},
|
||||
|
||||
@@ -37,6 +37,8 @@ func maintenanceLabel(kind string) string {
|
||||
return "a backup"
|
||||
case maintenance.KindFileWrite:
|
||||
return "a file write"
|
||||
case maintenance.KindReap:
|
||||
return "the idle-world reaper"
|
||||
}
|
||||
return "another operation"
|
||||
}
|
||||
|
||||
@@ -18,7 +18,8 @@
|
||||
// conflict and re-checks.
|
||||
//
|
||||
// A lock older than Grace with no Job behind it is stale (felis-api died between
|
||||
// the two writes) and holds nothing.
|
||||
// the two writes) and holds nothing. The reaper is the one holder without a Job:
|
||||
// it keeps its lock fresh by rewriting it while it archives and reclaims a world.
|
||||
//
|
||||
// A restore that starts with a safety snapshot is two Jobs in a row: the backup
|
||||
// Job carries the restore to run after it (LabelThenRestore), and felis-api
|
||||
@@ -83,6 +84,10 @@ const (
|
||||
KindRestore = "restore"
|
||||
KindBackup = "backup"
|
||||
KindFileWrite = "file-write"
|
||||
// KindReap is the reaper archiving an idle world and reclaiming its volume.
|
||||
// It runs no Job: the reaper holds the Annotation itself and rewrites it
|
||||
// well inside Grace for as long as it works on the world.
|
||||
KindReap = "reap"
|
||||
)
|
||||
|
||||
// FilesModeWrite is the LabelFilesMode value of a file write.
|
||||
|
||||
@@ -33,7 +33,11 @@ func (c *reclaimCluster) DeletePVC(_ context.Context, pvc string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *reclaimCluster) Stop(context.Context, string) error { return nil }
|
||||
func (c *reclaimCluster) HoldWorld(ctx context.Context, _ string) (context.Context, func(), error) {
|
||||
return ctx, func() {}, nil
|
||||
}
|
||||
|
||||
func (c *reclaimCluster) WorldExists(context.Context, string) (bool, error) { return true, nil }
|
||||
|
||||
type reclaimArchiver struct{ archived []string }
|
||||
|
||||
@@ -387,3 +391,36 @@ func TestBackupReadBack(t *testing.T) {
|
||||
t.Fatalf("live refs after deleting %s still claim its archive", newer)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestartClock: an idle server with no world to reclaim gets its clock and
|
||||
// warnings reset, and keeps its owner and resource cache (data-durability-19).
|
||||
func TestRestartClock(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
st := reaper.NewPGStore(db)
|
||||
name := "clock-" + suffix(t)
|
||||
old := time.Now().Add(-20 * reaper.Day).UTC().Truncate(time.Microsecond)
|
||||
if _, err := db.ExecContext(ctx,
|
||||
`INSERT INTO servers (name, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, 100, 128, 1)`,
|
||||
name); err != nil {
|
||||
t.Fatalf("seed server: %v", err)
|
||||
}
|
||||
if _, err := db.ExecContext(ctx,
|
||||
`UPDATE servers SET last_active_at = $2, warned_3d_at = $2, warned_1d_at = $2 WHERE name = $1`, name, old); err != nil {
|
||||
t.Fatalf("age server: %v", err)
|
||||
}
|
||||
at := time.Now().UTC().Truncate(time.Microsecond)
|
||||
if err := st.RestartClock(ctx, name, at); err != nil {
|
||||
t.Fatalf("RestartClock: %v", err)
|
||||
}
|
||||
var last time.Time
|
||||
var w3, w1 sql.NullTime
|
||||
var cpu int
|
||||
if err := db.QueryRowContext(ctx,
|
||||
`SELECT last_active_at, warned_3d_at, warned_1d_at, cached_cpu_milli FROM servers WHERE name = $1`, name).
|
||||
Scan(&last, &w3, &w1, &cpu); err != nil {
|
||||
t.Fatalf("read back: %v", err)
|
||||
}
|
||||
if !last.Equal(at) || w3.Valid || w1.Valid || cpu != 100 {
|
||||
t.Fatalf("after RestartClock: last_active_at=%v warned=%v/%v cpu=%d; want %v, cleared, cache kept", last, w3, w1, cpu, at)
|
||||
}
|
||||
}
|
||||
@@ -175,9 +175,14 @@ func OperatorRole(p Params) *rbacv1.Role {
|
||||
// ReaperRole grants felis-reaper its two destructive, disjoint powers
|
||||
// (internal/reaper.k8scluster): patch a MinecraftServer to Stop it and delete its
|
||||
// world PVC. Candidate servers come from the Postgres store, not a cluster List,
|
||||
// so no list/watch is needed; the reaper uses a direct client. It can read+patch
|
||||
// minecraftservers but cannot create them, and holds no power over StatefulSets,
|
||||
// Services, or Secrets — those belong to the operator and api.
|
||||
// so minecraftservers need no list/watch; the reaper uses a direct client. It can
|
||||
// read+patch minecraftservers but cannot create them, and holds no power over
|
||||
// StatefulSets, Services, or Secrets — those belong to the operator and api.
|
||||
//
|
||||
// The same patch holds the world maintenance lock while a world is archived and
|
||||
// reclaimed (internal/maintenance). Taking it needs the two reads felis-api makes
|
||||
// before a restore: list pods (is the game pod gone) and list jobs (does a
|
||||
// restore, backup or file write hold the world). Both are list-only.
|
||||
//
|
||||
// persistentvolumeclaims also carries get: resolving where a world lives
|
||||
// (cmd/felis/reaper.resolveWorldDir) reads the PVC's volumeName to derive the
|
||||
@@ -195,6 +200,8 @@ func ReaperRole(p Params) *rbacv1.Role {
|
||||
return role(p.MinecraftNamespace, "felis-reaper", ComponentReaper, []rbacv1.PolicyRule{
|
||||
rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "patch"}),
|
||||
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get", "delete"}),
|
||||
rule([]string{groupCore}, []string{"pods"}, []string{"list"}),
|
||||
rule([]string{groupBatch}, []string{"jobs"}, []string{"list"}),
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -221,12 +221,23 @@ func TestReaperRole_ScopeExact(t *testing.T) {
|
||||
t.Errorf("reaper must NOT touch %s/%s (operator/api territory)", res.group, res.name)
|
||||
}
|
||||
}
|
||||
// The reaper never lists from the cluster (candidates come from Postgres).
|
||||
// The reaper never lists servers from the cluster (candidates come from Postgres).
|
||||
for _, v := range []string{"list", "watch"} {
|
||||
if hasRule(rp, groupFelis, "minecraftservers", v) {
|
||||
t.Errorf("reaper must NOT %s minecraftservers (candidates come from the store)", v)
|
||||
}
|
||||
}
|
||||
// Taking the world lock reads the game pod and the maintenance Jobs, list only.
|
||||
if !hasRule(rp, groupCore, "pods", "list") || !hasRule(rp, groupBatch, "jobs", "list") {
|
||||
t.Error("reaper must list pods and jobs to take the world maintenance lock")
|
||||
}
|
||||
for _, res := range []struct{ group, name string }{{groupCore, "pods"}, {groupBatch, "jobs"}} {
|
||||
for _, v := range []string{"get", "watch", "create", "update", "patch", "delete"} {
|
||||
if hasRule(rp, res.group, res.name, v) {
|
||||
t.Errorf("reaper must NOT %s %s (list-only)", v, res.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestReaperRBAC_GatedOnRetention locks the reaper identity to its consumer: the
|
||||
|
||||
+184
-15
@@ -2,13 +2,20 @@ package reaper
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"k8s.io/client-go/util/retry"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
@@ -20,13 +27,20 @@ func WorldPVCName(server string) string {
|
||||
return naming.WorldPVCName(server)
|
||||
}
|
||||
|
||||
// gamePodComponent is the operator's component label on a game server's pod
|
||||
// (internal/operator.ComponentValue; k8scluster_test pins the two).
|
||||
const gamePodComponent = "server"
|
||||
|
||||
// K8sCluster is the production Cluster backed by a controller-runtime client
|
||||
// (spec §4, §18). It reads spec.reaperExempt, deletes the world PVC, and flips
|
||||
// spec.desiredState to Stopped — nothing else. It is integration-tested against
|
||||
// a live cluster, not the hermetic reaper_test.go suite.
|
||||
// (spec §4, §18). It reads spec.reaperExempt, stops a server and holds its world
|
||||
// volume through the maintenance lock, and deletes the world PVC — nothing else.
|
||||
type K8sCluster struct {
|
||||
c client.Client
|
||||
namespace string
|
||||
now func() time.Time
|
||||
// beat is how often a held lock is rewritten; maintenance.Grace/4 unless a
|
||||
// test shortens it.
|
||||
beat time.Duration
|
||||
}
|
||||
|
||||
// NewK8sCluster builds a Cluster over c, scoped to namespace.
|
||||
@@ -34,17 +48,177 @@ func NewK8sCluster(c client.Client, namespace string) *K8sCluster {
|
||||
return &K8sCluster{c: c, namespace: namespace}
|
||||
}
|
||||
|
||||
func (k *K8sCluster) clock() time.Time {
|
||||
if k.now != nil {
|
||||
return k.now()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
func (k *K8sCluster) Inspect(ctx context.Context, name string) (ServerCRD, error) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return ServerCRD{}, ErrNotFound
|
||||
}
|
||||
if err := k.get(ctx, name, &ms); err != nil {
|
||||
return ServerCRD{}, err
|
||||
}
|
||||
return ServerCRD{Exempt: ms.Spec.ReaperExempt, PVC: WorldPVCName(name)}, nil
|
||||
}
|
||||
|
||||
// HoldWorld implements Cluster. The lock is the same Annotation felis-api
|
||||
// writes before a restore, backup or file-write Job (internal/maintenance): a
|
||||
// wake through felis-api, the operator scaling the server up, and another
|
||||
// world operation all refuse while it is fresh. With no Job behind it, it
|
||||
// holds for maintenance.Grace after each write, so it is rewritten every beat;
|
||||
// the held context is cancelled once a rewrite has failed for Grace/2, well
|
||||
// before any reader could take the lock as lapsed.
|
||||
func (k *K8sCluster) HoldWorld(ctx context.Context, name string) (context.Context, func(), error) {
|
||||
if err := k.lock(ctx, name); err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
held, cancel := context.WithCancelCause(ctx)
|
||||
done := make(chan struct{})
|
||||
go k.keep(held, cancel, name, done)
|
||||
release := func() {
|
||||
cancel(nil)
|
||||
<-done
|
||||
rctx, stop := context.WithTimeout(context.WithoutCancel(ctx), 10*time.Second)
|
||||
defer stop()
|
||||
_ = k.unlock(rctx, name)
|
||||
}
|
||||
return held, release, nil
|
||||
}
|
||||
|
||||
// lock stops the server if it is still meant to run and takes the reap lock
|
||||
// once it is fully down and nothing else holds its world.
|
||||
func (k *K8sCluster) lock(ctx context.Context, name string) error {
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.get(ctx, name, &ms); err != nil {
|
||||
return err
|
||||
}
|
||||
if ms.Spec.DesiredState != "" && ms.Spec.DesiredState != v1alpha1.DesiredStopped {
|
||||
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
|
||||
ms.Spec.DesiredState = v1alpha1.DesiredStopped
|
||||
if err := k.c.Patch(ctx, &ms, patch); err != nil {
|
||||
return err
|
||||
}
|
||||
return fmt.Errorf("%w: it was still up and has been told to stop", ErrNotQuiet)
|
||||
}
|
||||
if ms.Status.Ready || (ms.Status.Phase != "" && ms.Status.Phase != v1alpha1.PhaseStopped) {
|
||||
return fmt.Errorf("%w: it is still stopping (phase %s)", ErrNotQuiet, ms.Status.Phase)
|
||||
}
|
||||
var pods corev1.PodList
|
||||
if err := k.c.List(ctx, &pods, client.InNamespace(k.namespace), client.MatchingLabels{
|
||||
v1alpha1.LabelServer: name, v1alpha1.LabelComponent: gamePodComponent,
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
if len(pods.Items) > 0 {
|
||||
return fmt.Errorf("%w: its game pod is still shutting down", ErrNotQuiet)
|
||||
}
|
||||
var jobs batchv1.JobList
|
||||
if err := k.c.List(ctx, &jobs, client.InNamespace(k.namespace),
|
||||
client.MatchingLabels{maintenance.LabelServer: name}); err != nil {
|
||||
return err
|
||||
}
|
||||
if kind, held := maintenance.Holder(name, ms.Annotations, jobs.Items, k.clock()); held {
|
||||
return fmt.Errorf("%w: a %s holds its world", ErrNotQuiet, kind)
|
||||
}
|
||||
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
|
||||
if ms.Annotations == nil {
|
||||
ms.Annotations = map[string]string{}
|
||||
}
|
||||
ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindReap, k.clock())
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
})
|
||||
}
|
||||
|
||||
// errLockLost reports that the lock on the object is no longer the reaper's.
|
||||
var errLockLost = errors.New("the reap lock is gone from the server")
|
||||
|
||||
// keep rewrites the lock every beat until held ends, and ends held itself once
|
||||
// it has not managed to for Grace/2.
|
||||
func (k *K8sCluster) keep(held context.Context, cancel context.CancelCauseFunc, name string, done chan<- struct{}) {
|
||||
defer close(done)
|
||||
beat := k.beat
|
||||
if beat <= 0 {
|
||||
beat = maintenance.Grace / 4
|
||||
}
|
||||
t := time.NewTicker(beat)
|
||||
defer t.Stop()
|
||||
last := k.clock()
|
||||
for {
|
||||
select {
|
||||
case <-held.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
}
|
||||
err := k.refresh(held, name)
|
||||
if err == nil {
|
||||
last = k.clock()
|
||||
continue
|
||||
}
|
||||
if held.Err() != nil {
|
||||
return
|
||||
}
|
||||
if errors.Is(err, errLockLost) || k.clock().Sub(last) >= maintenance.Grace/2 {
|
||||
cancel(fmt.Errorf("reaper: lost the world lock on %s: %w", name, err))
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (k *K8sCluster) refresh(ctx context.Context, name string) error {
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.get(ctx, name, &ms); err != nil {
|
||||
return err
|
||||
}
|
||||
if !reapLock(ms.Annotations) {
|
||||
return errLockLost
|
||||
}
|
||||
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
|
||||
ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindReap, k.clock())
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
})
|
||||
}
|
||||
|
||||
// unlock drops the lock if it is still the reaper's; a lock left behind lapses
|
||||
// after maintenance.Grace on its own.
|
||||
func (k *K8sCluster) unlock(ctx context.Context, name string) error {
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.get(ctx, name, &ms); err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
if !reapLock(ms.Annotations) {
|
||||
return nil
|
||||
}
|
||||
patch := client.MergeFromWithOptions(ms.DeepCopy(), client.MergeFromWithOptimisticLock{})
|
||||
delete(ms.Annotations, maintenance.Annotation)
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
})
|
||||
}
|
||||
|
||||
func reapLock(annotations map[string]string) bool {
|
||||
return strings.HasPrefix(annotations[maintenance.Annotation], maintenance.KindReap+"@")
|
||||
}
|
||||
|
||||
// WorldExists reports whether the world PersistentVolumeClaim exists.
|
||||
func (k *K8sCluster) WorldExists(ctx context.Context, pvc string) (bool, error) {
|
||||
var claim corev1.PersistentVolumeClaim
|
||||
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: pvc}, &claim)
|
||||
switch {
|
||||
case apierrors.IsNotFound(err):
|
||||
return false, nil
|
||||
case err != nil:
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// DeletePVC deletes the world PersistentVolumeClaim. A missing PVC is not an
|
||||
// error: the reap is idempotent and a re-run after a partial failure must still
|
||||
// converge.
|
||||
@@ -58,17 +232,12 @@ func (k *K8sCluster) DeletePVC(ctx context.Context, pvc string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Stop sets spec.desiredState=Stopped with a merge patch so a concurrent status
|
||||
// write by the operator is never clobbered (spec §9.1).
|
||||
func (k *K8sCluster) Stop(ctx context.Context, name string) error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil {
|
||||
func (k *K8sCluster) get(ctx context.Context, name string, ms *v1alpha1.MinecraftServer) error {
|
||||
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, ms); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return ErrNotFound
|
||||
}
|
||||
return err
|
||||
}
|
||||
patch := client.MergeFrom(ms.DeepCopy())
|
||||
ms.Spec.DesiredState = v1alpha1.DesiredStopped
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
package reaper
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
// HoldWorld against a fake API server, which honours resourceVersion on the
|
||||
// optimistic-lock patches the lock is written with.
|
||||
|
||||
func holdServer(desired v1alpha1.DesiredState, phase v1alpha1.Phase) *v1alpha1.MinecraftServer {
|
||||
return &v1alpha1.MinecraftServer{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"},
|
||||
Spec: v1alpha1.MinecraftServerSpec{DesiredState: desired},
|
||||
Status: v1alpha1.MinecraftServerStatus{Phase: phase},
|
||||
}
|
||||
}
|
||||
|
||||
// fakeClock is a clock the tests and the heartbeat goroutine share.
|
||||
type fakeClock struct {
|
||||
mu sync.Mutex
|
||||
t time.Time
|
||||
}
|
||||
|
||||
func (c *fakeClock) now() time.Time { c.mu.Lock(); defer c.mu.Unlock(); return c.t }
|
||||
func (c *fakeClock) add(d time.Duration) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.t = c.t.Add(d)
|
||||
}
|
||||
|
||||
func holdCluster(t *testing.T, objs ...client.Object) (*K8sCluster, client.Client, *fakeClock) {
|
||||
t.Helper()
|
||||
scheme := runtime.NewScheme()
|
||||
if err := clientgoscheme.AddToScheme(scheme); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
if err := v1alpha1.AddToScheme(scheme); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(objs...).
|
||||
WithStatusSubresource(&v1alpha1.MinecraftServer{}).Build()
|
||||
clk := &fakeClock{t: time.Date(2026, 9, 24, 12, 0, 0, 0, time.UTC)}
|
||||
k := NewK8sCluster(c, "minecraft")
|
||||
k.now = clk.now
|
||||
k.beat = time.Hour
|
||||
return k, c, clk
|
||||
}
|
||||
|
||||
func serverState(t *testing.T, c client.Client) (map[string]string, v1alpha1.DesiredState) {
|
||||
t.Helper()
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival"}, &ms); err != nil {
|
||||
t.Fatalf("get: %v", err)
|
||||
}
|
||||
return ms.Annotations, ms.Spec.DesiredState
|
||||
}
|
||||
|
||||
func TestHoldWorldStopsARunningServer(t *testing.T) {
|
||||
k, c, _ := holdCluster(t, holdServer(v1alpha1.DesiredRunning, v1alpha1.PhaseRunning))
|
||||
_, _, err := k.HoldWorld(context.Background(), "survival")
|
||||
if !errors.Is(err, ErrNotQuiet) {
|
||||
t.Fatalf("HoldWorld = %v, want ErrNotQuiet", err)
|
||||
}
|
||||
ann, desired := serverState(t, c)
|
||||
if desired != v1alpha1.DesiredStopped || ann[maintenance.Annotation] != "" {
|
||||
t.Fatalf("desired=%q annotations=%v: want told to stop and no lock yet", desired, ann)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHoldWorldWaitsUntilQuiet(t *testing.T) {
|
||||
ready := holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
|
||||
ready.Status.Ready = true
|
||||
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "survival-0", Namespace: "minecraft",
|
||||
Labels: map[string]string{v1alpha1.LabelServer: "survival", v1alpha1.LabelComponent: gamePodComponent}}}
|
||||
job := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Name: "restore-survival", Namespace: "minecraft",
|
||||
Labels: map[string]string{maintenance.LabelServer: "survival", maintenance.LabelManagedBy: "felis-restore"}}}
|
||||
locked := holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
|
||||
locked.Annotations = map[string]string{
|
||||
maintenance.Annotation: maintenance.LockValue(maintenance.KindBackup, time.Date(2026, 9, 24, 11, 59, 30, 0, time.UTC)),
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
objs []client.Object
|
||||
why string
|
||||
}{
|
||||
{"stopping", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopping)}, "still stopping"},
|
||||
{"ready", []client.Object{ready}, "still stopping"},
|
||||
{"game pod", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pod}, "game pod"},
|
||||
{"restore job", []client.Object{holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), job}, "a restore"},
|
||||
{"admission lock", []client.Object{locked}, "a backup"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
k, c, _ := holdCluster(t, tc.objs...)
|
||||
before, _ := serverState(t, c)
|
||||
_, _, err := k.HoldWorld(context.Background(), "survival")
|
||||
if !errors.Is(err, ErrNotQuiet) || !strings.Contains(err.Error(), tc.why) {
|
||||
t.Fatalf("HoldWorld = %v, want ErrNotQuiet naming %q", err, tc.why)
|
||||
}
|
||||
if after, _ := serverState(t, c); after[maintenance.Annotation] != before[maintenance.Annotation] {
|
||||
t.Fatalf("lock changed to %q", after[maintenance.Annotation])
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Once held, every other holder check sees the reaper, and release lets go.
|
||||
func TestHoldWorldLocksAndReleases(t *testing.T) {
|
||||
k, c, clk := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped))
|
||||
held, release, err := k.HoldWorld(context.Background(), "survival")
|
||||
if err != nil {
|
||||
t.Fatalf("HoldWorld: %v", err)
|
||||
}
|
||||
ann, _ := serverState(t, c)
|
||||
if kind, ok := maintenance.Holder("survival", ann, nil, clk.now()); !ok || kind != maintenance.KindReap {
|
||||
t.Fatalf("Holder = (%q, %v) with %v, want the reaper", kind, ok, ann)
|
||||
}
|
||||
release()
|
||||
if held.Err() == nil {
|
||||
t.Fatal("held context still live after release")
|
||||
}
|
||||
if ann, _ := serverState(t, c); ann[maintenance.Annotation] != "" {
|
||||
t.Fatalf("lock left behind: %v", ann)
|
||||
}
|
||||
}
|
||||
|
||||
// The lock is rewritten while held, so it never lapses under a long archive.
|
||||
func TestHoldWorldKeepsLockFresh(t *testing.T) {
|
||||
k, c, clk := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped))
|
||||
k.beat = 5 * time.Millisecond
|
||||
held, release, err := k.HoldWorld(context.Background(), "survival")
|
||||
if err != nil {
|
||||
t.Fatalf("HoldWorld: %v", err)
|
||||
}
|
||||
defer release()
|
||||
clk.add(10 * maintenance.Grace)
|
||||
want := maintenance.LockValue(maintenance.KindReap, clk.now())
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for {
|
||||
if ann, _ := serverState(t, c); ann[maintenance.Annotation] == want {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
ann, _ := serverState(t, c)
|
||||
t.Fatalf("lock = %q, want rewritten to %q", ann[maintenance.Annotation], want)
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
if held.Err() != nil {
|
||||
t.Fatalf("held ended: %v", context.Cause(held))
|
||||
}
|
||||
}
|
||||
|
||||
// A lock someone else removed (or replaced) ends the hold: the reaper must not
|
||||
// delete a world it no longer holds.
|
||||
func TestHoldWorldEndsWhenLockIsLost(t *testing.T) {
|
||||
k, c, _ := holdCluster(t, holdServer(v1alpha1.DesiredStopped, v1alpha1.PhaseStopped))
|
||||
k.beat = 5 * time.Millisecond
|
||||
held, release, err := k.HoldWorld(context.Background(), "survival")
|
||||
if err != nil {
|
||||
t.Fatalf("HoldWorld: %v", err)
|
||||
}
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := c.Get(context.Background(), types.NamespacedName{Namespace: "minecraft", Name: "survival"}, &ms); err != nil {
|
||||
t.Fatalf("get: %v", err)
|
||||
}
|
||||
patch := client.MergeFrom(ms.DeepCopy())
|
||||
ms.Annotations[maintenance.Annotation] = maintenance.LockValue(maintenance.KindRestore, time.Now())
|
||||
if err := c.Patch(context.Background(), &ms, patch); err != nil {
|
||||
t.Fatalf("patch: %v", err)
|
||||
}
|
||||
select {
|
||||
case <-held.Done():
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("held context still live after the lock was taken over")
|
||||
}
|
||||
if !errors.Is(context.Cause(held), errLockLost) {
|
||||
t.Fatalf("cause = %v, want errLockLost", context.Cause(held))
|
||||
}
|
||||
release()
|
||||
if ann, _ := serverState(t, c); !strings.HasPrefix(ann[maintenance.Annotation], maintenance.KindRestore+"@") {
|
||||
t.Fatalf("release removed another holder's lock: %v", ann)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWorldExists(t *testing.T) {
|
||||
pvc := &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: "world-survival-0", Namespace: "minecraft"}}
|
||||
k, _, _ := holdCluster(t, pvc)
|
||||
for name, want := range map[string]bool{"world-survival-0": true, "world-other-0": false} {
|
||||
if got, err := k.WorldExists(context.Background(), name); err != nil || got != want {
|
||||
t.Errorf("WorldExists(%s) = (%v, %v), want %v", name, got, err, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// gamePodComponent is a copy of the operator's pod label value; a drift would
|
||||
// let the reaper archive a world beside a server still saving it.
|
||||
func TestGamePodComponentMatchesOperator(t *testing.T) {
|
||||
if gamePodComponent != operator.ComponentValue {
|
||||
t.Fatalf("gamePodComponent = %q, operator labels its pods %q", gamePodComponent, operator.ComponentValue)
|
||||
}
|
||||
}
|
||||
@@ -90,6 +90,13 @@ func (s *PGStore) ReleaseWorld(ctx context.Context, name string, at time.Time) e
|
||||
return err
|
||||
}
|
||||
|
||||
func (s *PGStore) RestartClock(ctx context.Context, name string, at time.Time) error {
|
||||
const q = `UPDATE servers SET last_active_at = $2, warned_3d_at = NULL, warned_1d_at = NULL
|
||||
WHERE name = $1 AND deleted_at IS NULL`
|
||||
_, err := s.db.ExecContext(ctx, q, name, at)
|
||||
return err
|
||||
}
|
||||
|
||||
func (s *PGStore) MarkWarned(ctx context.Context, name string, tier Tier, at time.Time) error {
|
||||
// The column is one of two fixed identifiers, never user input.
|
||||
col := "warned_3d_at"
|
||||
|
||||
+92
-14
@@ -55,6 +55,11 @@ const (
|
||||
// delete world data it cannot first inspect for the exemption flag.
|
||||
var ErrNotFound = errors.New("reaper: server not found")
|
||||
|
||||
// ErrNotQuiet is returned by Cluster.HoldWorld while the server is not fully
|
||||
// down yet, or another operation (a restore, backup or file write) holds its
|
||||
// world. The world is left alone and the server is tried again on the next run.
|
||||
var ErrNotQuiet = errors.New("reaper: world not quiet")
|
||||
|
||||
// errStoreFull is an internal sentinel: the backup store is at capacity and
|
||||
// could not be freed, so the world is preserved rather than deleted without a
|
||||
// backup (red line ④). It is never returned to callers.
|
||||
@@ -219,6 +224,10 @@ type Store interface {
|
||||
// row (red line ②).
|
||||
ReleaseWorld(ctx context.Context, name string, at time.Time) error
|
||||
|
||||
// RestartClock sets last_active_at→at and clears warned_* on a server with
|
||||
// no world to reclaim, so it is not found idle again every run.
|
||||
RestartClock(ctx context.Context, name string, at time.Time) error
|
||||
|
||||
// MarkWarned stamps the warned_3d_at / warned_1d_at column for tier.
|
||||
MarkWarned(ctx context.Context, name string, tier Tier, at time.Time) error
|
||||
|
||||
@@ -255,17 +264,25 @@ type Store interface {
|
||||
Audit(ctx context.Context, rec AuditRecord) error
|
||||
}
|
||||
|
||||
// Cluster is the lifecycle (Kubernetes) face: read the CRD, delete the world
|
||||
// PVC, and flip desiredState to Stopped. These are the only cluster operations
|
||||
// §18 performs.
|
||||
// Cluster is the lifecycle (Kubernetes) face: read the CRD, stop the server
|
||||
// and hold its world, and delete the world PVC. These are the only cluster
|
||||
// operations §18 performs.
|
||||
type Cluster interface {
|
||||
// Inspect returns the exemption flag and world PVC name for a server, or
|
||||
// ErrNotFound if the CRD is gone.
|
||||
Inspect(ctx context.Context, name string) (ServerCRD, error)
|
||||
// HoldWorld keeps everything else off the server's world until release is
|
||||
// called: it sets desiredState=Stopped if the server is still meant to run,
|
||||
// and once the server is fully down (not ready, phase Stopped, no game pod)
|
||||
// and no restore, backup or file write holds the world, it takes the world
|
||||
// maintenance lock (internal/maintenance, KindReap) and keeps it fresh.
|
||||
// Until then it returns ErrNotQuiet. The returned context ends if the lock
|
||||
// cannot be kept; the world must not be read or deleted on it after that.
|
||||
HoldWorld(ctx context.Context, name string) (held context.Context, release func(), err error)
|
||||
// WorldExists reports whether the world PersistentVolumeClaim exists.
|
||||
WorldExists(ctx context.Context, pvc string) (bool, error)
|
||||
// DeletePVC deletes the world PersistentVolumeClaim.
|
||||
DeletePVC(ctx context.Context, pvc string) error
|
||||
// Stop sets spec.desiredState=Stopped.
|
||||
Stop(ctx context.Context, name string) error
|
||||
}
|
||||
|
||||
// Warner delivers an impending-reap notice. It is optional and best-effort: a
|
||||
@@ -309,6 +326,10 @@ type Summary struct {
|
||||
// AwaitingOffsite are idle worlds that are archived and kept until the
|
||||
// archive's off-site copy lands.
|
||||
AwaitingOffsite int
|
||||
// AwaitingStop are idle servers left for the next run because they were not
|
||||
// fully down yet (the run told a running one to stop) or another operation
|
||||
// held their world.
|
||||
AwaitingStop int
|
||||
// Verified are archives read back in full and found matching; Corrupt are
|
||||
// the ones that were not (marked, and never reused or restored from);
|
||||
// VerifyFailed are the ones that could not be read back at all this run.
|
||||
@@ -379,6 +400,11 @@ func (r *Reaper) RunOnce(ctx context.Context) (Summary, error) {
|
||||
for _, c := range cands {
|
||||
sum.Evaluated++
|
||||
if err := r.evaluate(ctx, now, offs, c, &sum); err != nil {
|
||||
if errors.Is(err, ErrNotQuiet) {
|
||||
sum.AwaitingStop++
|
||||
r.log().Warn("reaper: idle world not quiet, retrying next run", "server", c.Name, "why", err)
|
||||
continue
|
||||
}
|
||||
r.log().Error("reaper: skipping server", "server", c.Name, "err", err)
|
||||
sum.Skipped++
|
||||
if errors.Is(err, errStoreFull) {
|
||||
@@ -419,9 +445,27 @@ func (r *Reaper) evaluate(ctx context.Context, now time.Time, offs []time.Durati
|
||||
return nil
|
||||
}
|
||||
|
||||
// reap archives the world, records the backup, and only then deletes the PVC,
|
||||
// releases ownership, and stops the server — the strict ordering of red line ④.
|
||||
// reap stops the server and holds its world, archives the world, records the
|
||||
// backup, and only then deletes the PVC and releases ownership — the strict
|
||||
// ordering of red line ④. Nothing can start the server or touch its world while
|
||||
// it is held, so the archive is of a world at rest and the PVC deleted is the
|
||||
// one archived.
|
||||
func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd ServerCRD, sum *Summary) error {
|
||||
held, release, err := r.Cluster.HoldWorld(ctx, c.Name)
|
||||
if err != nil {
|
||||
return fmt.Errorf("hold world: %w", err)
|
||||
}
|
||||
defer release()
|
||||
ctx = held
|
||||
|
||||
exists, err := r.Cluster.WorldExists(ctx, crd.PVC)
|
||||
if err != nil {
|
||||
return fmt.Errorf("look up world volume: %w", err)
|
||||
}
|
||||
if !exists {
|
||||
return r.reapNoWorld(ctx, now, c, sum)
|
||||
}
|
||||
|
||||
// §26 soft cap: free space before adding a backup. If the store cannot be
|
||||
// brought under cap, preserve the world rather than delete it unbacked.
|
||||
if r.Cfg.MaxLocalBytes > 0 {
|
||||
@@ -497,32 +541,66 @@ func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd Serve
|
||||
}
|
||||
|
||||
// World is safely archived and recorded — now (and only now) delete it.
|
||||
// The lock must still be held: a lapsed one could have let the server start.
|
||||
if err := context.Cause(ctx); err != nil {
|
||||
return fmt.Errorf("world no longer held: %w", err)
|
||||
}
|
||||
if err := r.Cluster.DeletePVC(ctx, crd.PVC); err != nil {
|
||||
// The backup row persists; next run's FreshBackup reuses it and retries
|
||||
// the delete, so no duplicate archive is created.
|
||||
return fmt.Errorf("delete pvc: %w", err)
|
||||
}
|
||||
return r.finishReap(ctx, now, c, ref, sum)
|
||||
}
|
||||
|
||||
// finishReap releases a world whose PVC is gone and records the reap.
|
||||
func (r *Reaper) finishReap(ctx context.Context, now time.Time, c Candidate, ref string, sum *Summary) error {
|
||||
if err := r.Store.ReleaseWorld(ctx, c.Name, now); err != nil {
|
||||
return fmt.Errorf("release world: %w", err)
|
||||
}
|
||||
if err := r.Cluster.Stop(ctx, c.Name); err != nil {
|
||||
// The world is already deleted and ownership released; the desiredState
|
||||
// flip is cosmetic by comparison. Log, but the reap stands.
|
||||
r.log().Error("reaper: set desiredState=Stopped failed", "server", c.Name, "err", err)
|
||||
}
|
||||
if err := r.Store.Audit(ctx, AuditRecord{Action: ActionReapWorld, ServerName: c.Name, FormerOwner: c.OwnerID}); err != nil {
|
||||
r.log().Error("reaper: audit reap_world failed", "server", c.Name, "err", err)
|
||||
}
|
||||
|
||||
sum.WorldsReaped++
|
||||
// felis_reaper_worlds_deleted_total (spec §23) advances in lockstep with the
|
||||
// per-run Summary tally — incremented here, at the one point a world's PVC has
|
||||
// actually been deleted, not at evaluation time.
|
||||
// per-run Summary tally — incremented once per world whose PVC went, not at
|
||||
// evaluation time.
|
||||
metrics.ReaperWorldsDeletedTotal.Inc()
|
||||
r.log().Info("reaper: world reaped", "server", c.Name, "former_owner", c.OwnerID, "backup_ref", ref)
|
||||
return nil
|
||||
}
|
||||
|
||||
// reapNoWorld handles an idle server whose world PVC does not exist. With an
|
||||
// archive of the current world on record, an earlier run deleted the PVC and
|
||||
// stopped before releasing it, so the reap is finished now. Otherwise there was
|
||||
// no world to reclaim: an owner who never started the server gives it up
|
||||
// (nothing to back up), and an unowned one — typically a world reaped earlier —
|
||||
// only has its clock restarted, so it is not reaped over and over.
|
||||
func (r *Reaper) reapNoWorld(ctx context.Context, now time.Time, c Candidate, sum *Summary) error {
|
||||
fresh, ok, err := r.Store.FreshBackup(ctx, c.Name, c.LastActiveAt)
|
||||
if err != nil {
|
||||
return fmt.Errorf("lookup fresh backup: %w", err)
|
||||
}
|
||||
if ok {
|
||||
return r.finishReap(ctx, now, c, fresh.Ref, sum)
|
||||
}
|
||||
if c.OwnerID == "" {
|
||||
if err := r.Store.RestartClock(ctx, c.Name, now); err != nil {
|
||||
return fmt.Errorf("restart clock: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := r.Store.ReleaseWorld(ctx, c.Name, now); err != nil {
|
||||
return fmt.Errorf("release world: %w", err)
|
||||
}
|
||||
if err := r.Store.Audit(ctx, AuditRecord{Action: ActionReapWorld, ServerName: c.Name, FormerOwner: c.OwnerID}); err != nil {
|
||||
r.log().Error("reaper: audit reap_world failed", "server", c.Name, "err", err)
|
||||
}
|
||||
r.log().Info("reaper: idle server released; it had no world to archive", "server", c.Name, "former_owner", c.OwnerID)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ensureCapacity frees the backup store down under MaxLocalBytes by evicting the
|
||||
// oldest present backups early. Early eviction is destructive (it removes
|
||||
// not-yet-expired backups), so each eviction is alerted and audited. It returns
|
||||
|
||||
+177
-12
@@ -24,7 +24,7 @@ var testNow = time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC)
|
||||
func idleBy(d time.Duration) time.Time { return testNow.Add(-d) }
|
||||
|
||||
// recorder captures the cross-fake call order so a test can assert the strict
|
||||
// archive→insert→deletePVC→release→stop→audit sequence of red line ④.
|
||||
// hold→archive→insert→deletePVC→release→audit→unhold sequence of red line ④.
|
||||
type recorder struct{ events []string }
|
||||
|
||||
func (r *recorder) add(e string) { r.events = append(r.events, e) }
|
||||
@@ -33,6 +33,7 @@ func (r *recorder) add(e string) { r.events = append(r.events, e) }
|
||||
|
||||
type fakeArchiver struct {
|
||||
rec *recorder
|
||||
onArchive func()
|
||||
archiveErr error
|
||||
deleteErr error
|
||||
archives int
|
||||
@@ -47,6 +48,9 @@ func (f *fakeArchiver) Archive(_ context.Context, server, _ string) (backup.Arch
|
||||
f.archives++
|
||||
f.seq++
|
||||
f.rec.add("archive")
|
||||
if f.onArchive != nil {
|
||||
f.onArchive()
|
||||
}
|
||||
ref := fmt.Sprintf("ref-%s-%d", server, f.seq)
|
||||
return backup.Archived{Ref: backup.ArchiveRef(ref), Size: 10, SHA256: "sha-" + ref}, nil
|
||||
}
|
||||
@@ -104,7 +108,39 @@ type fakeCluster struct {
|
||||
deletePVCErr error
|
||||
deletePVCCalls int
|
||||
deletedPVCs []string
|
||||
stopped []string
|
||||
|
||||
notQuiet map[string]bool // HoldWorld refuses these with ErrNotQuiet
|
||||
holdErr error // HoldWorld fails with this
|
||||
noWorld map[string]bool // PVCs that do not exist
|
||||
held map[string]bool // servers held right now
|
||||
holds []string
|
||||
lost context.CancelCauseFunc
|
||||
}
|
||||
|
||||
func (c *fakeCluster) HoldWorld(ctx context.Context, name string) (context.Context, func(), error) {
|
||||
if c.holdErr != nil {
|
||||
return nil, nil, c.holdErr
|
||||
}
|
||||
if c.notQuiet[name] {
|
||||
return nil, nil, fmt.Errorf("%w: still stopping", ErrNotQuiet)
|
||||
}
|
||||
if c.held == nil {
|
||||
c.held = map[string]bool{}
|
||||
}
|
||||
c.held[name] = true
|
||||
c.holds = append(c.holds, name)
|
||||
c.rec.add("hold")
|
||||
held, cancel := context.WithCancelCause(ctx)
|
||||
c.lost = cancel
|
||||
return held, func() {
|
||||
cancel(nil)
|
||||
delete(c.held, name)
|
||||
c.rec.add("unhold")
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (c *fakeCluster) WorldExists(_ context.Context, pvc string) (bool, error) {
|
||||
return !c.noWorld[pvc], nil
|
||||
}
|
||||
|
||||
func (c *fakeCluster) Inspect(_ context.Context, name string) (ServerCRD, error) {
|
||||
@@ -118,22 +154,19 @@ func (c *fakeCluster) Inspect(_ context.Context, name string) (ServerCRD, error)
|
||||
return crd, nil
|
||||
}
|
||||
|
||||
func (c *fakeCluster) DeletePVC(_ context.Context, pvc string) error {
|
||||
func (c *fakeCluster) DeletePVC(ctx context.Context, pvc string) error {
|
||||
c.deletePVCCalls++
|
||||
if c.deletePVCErr != nil {
|
||||
return c.deletePVCErr
|
||||
}
|
||||
if len(c.held) == 0 || ctx.Err() != nil {
|
||||
return fmt.Errorf("deleted %s without holding its world", pvc)
|
||||
}
|
||||
c.deletedPVCs = append(c.deletedPVCs, pvc)
|
||||
c.rec.add("deletePVC")
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *fakeCluster) Stop(_ context.Context, name string) error {
|
||||
c.stopped = append(c.stopped, name)
|
||||
c.rec.add("stop")
|
||||
return nil
|
||||
}
|
||||
|
||||
// ---- fake Store -----------------------------------------------------------
|
||||
|
||||
type fakeBackup struct {
|
||||
@@ -212,6 +245,15 @@ func (s *fakeStore) ReleaseWorld(_ context.Context, name string, at time.Time) e
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *fakeStore) RestartClock(_ context.Context, name string, at time.Time) error {
|
||||
c := s.byName[name]
|
||||
c.LastActiveAt = at
|
||||
c.Warned3dAt = time.Time{}
|
||||
c.Warned1dAt = time.Time{}
|
||||
s.rec.add("restartClock")
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *fakeStore) MarkWarned(_ context.Context, name string, tier Tier, at time.Time) error {
|
||||
c := s.byName[name]
|
||||
if tier == Tier1d {
|
||||
@@ -405,7 +447,7 @@ func TestReapIdleWorldFullSequence(t *testing.T) {
|
||||
if sum.WorldsReaped != 1 {
|
||||
t.Fatalf("WorldsReaped = %d, want 1", sum.WorldsReaped)
|
||||
}
|
||||
want := []string{"archive", "insert", "deletePVC", "release", "stop", "audit:" + ActionReapWorld}
|
||||
want := []string{"hold", "archive", "insert", "deletePVC", "release", "audit:" + ActionReapWorld, "unhold"}
|
||||
if !reflect.DeepEqual(st.rec.events, want) {
|
||||
t.Fatalf("call order = %v, want %v", st.rec.events, want)
|
||||
}
|
||||
@@ -936,7 +978,7 @@ func TestReapReadsBackReusedArchive(t *testing.T) {
|
||||
if sum.WorldsReaped != 1 || ar.archives != 0 || len(cl.deletedPVCs) != 1 {
|
||||
t.Fatalf("summary %+v archives=%d deleted=%v: want the checked archive reused", sum, ar.archives, cl.deletedPVCs)
|
||||
}
|
||||
if st.rec.events[0] != "verify" || st.rec.events[1] != "deletePVC" {
|
||||
if st.rec.events[1] != "verify" || st.rec.events[2] != "deletePVC" {
|
||||
t.Errorf("events = %v, want the read-back before the delete", st.rec.events)
|
||||
}
|
||||
if !st.backups[0].verifiedAt.Equal(testNow) || !reflect.DeepEqual(ca.verified, []string{"ref-old"}) {
|
||||
@@ -958,7 +1000,7 @@ func TestReapReplacesCorruptArchive(t *testing.T) {
|
||||
if sum.WorldsReaped != 1 || sum.Corrupt != 1 || !sum.Failed() {
|
||||
t.Fatalf("summary = %+v, want reaped with one corrupt archive reported", sum)
|
||||
}
|
||||
want := []string{"verify", "corrupt", "archive", "insert", "deletePVC"}
|
||||
want := []string{"hold", "verify", "corrupt", "archive", "insert", "deletePVC"}
|
||||
if !reflect.DeepEqual(st.rec.events[:len(want)], want) {
|
||||
t.Errorf("events = %v, want %v first", st.rec.events, want)
|
||||
}
|
||||
@@ -1057,3 +1099,126 @@ func TestSweepPass(t *testing.T) {
|
||||
t.Errorf("sweep ran without the recorded archives: %+v", sum)
|
||||
}
|
||||
}
|
||||
|
||||
// data-durability-8: a server still up when its world goes idle is told to stop
|
||||
// and left for the next run, never archived while it may be writing; once it is
|
||||
// down the next run reaps it.
|
||||
func TestReapWaitsForServerToStop(t *testing.T) {
|
||||
r, st, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "busy", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
|
||||
cl.notQuiet = map[string]bool{"busy": true}
|
||||
|
||||
sum := mustRun(t, r)
|
||||
if sum.AwaitingStop != 1 || sum.Skipped != 0 || sum.Failed() || ar.archives != 0 || cl.deletePVCCalls != 0 {
|
||||
t.Fatalf("summary %+v archives=%d deletes=%d: want the world left alone for the next run", sum, ar.archives, cl.deletePVCCalls)
|
||||
}
|
||||
if st.byName["busy"].OwnerID != "user-1" || len(st.audits) != 0 {
|
||||
t.Fatalf("a server not yet down was released: %+v", st.byName["busy"])
|
||||
}
|
||||
|
||||
cl.notQuiet = nil
|
||||
if sum := mustRun(t, r); sum.WorldsReaped != 1 || sum.AwaitingStop != 0 {
|
||||
t.Fatalf("second run = %+v, want reaped once the server is down", sum)
|
||||
}
|
||||
}
|
||||
|
||||
// A hold that cannot be taken for any other reason is a failure of the run.
|
||||
func TestReapHoldErrorSkips(t *testing.T) {
|
||||
r, _, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "flaky", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
|
||||
cl.holdErr = errors.New("apiserver timeout")
|
||||
|
||||
sum := mustRun(t, r)
|
||||
if sum.Skipped != 1 || !sum.Failed() || ar.archives != 0 {
|
||||
t.Fatalf("summary %+v archives=%d: want the server skipped and the run failed", sum, ar.archives)
|
||||
}
|
||||
}
|
||||
|
||||
// The world is held from before the archive until after the reap is recorded,
|
||||
// and let go on every path, a failed archive included.
|
||||
func TestReapReleasesHoldOnEveryPath(t *testing.T) {
|
||||
r, st, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "hotel", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
|
||||
ar.archiveErr = errors.New("disk full")
|
||||
|
||||
mustRun(t, r)
|
||||
if len(cl.held) != 0 || st.rec.events[len(st.rec.events)-1] != "unhold" {
|
||||
t.Fatalf("hold not released after a failed archive: events=%v held=%v", st.rec.events, cl.held)
|
||||
}
|
||||
}
|
||||
|
||||
// A lock lost while the archive is written (felis-api could then start the
|
||||
// server) keeps the PVC; the archive is recorded and reused next run.
|
||||
func TestReapLostHoldKeepsWorld(t *testing.T) {
|
||||
r, st, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "india", OwnerID: "user-1", LastActiveAt: idleBy(20 * Day)})
|
||||
ar.onArchive = func() { cl.lost(errors.New("lock rewrite failing")) }
|
||||
|
||||
sum := mustRun(t, r)
|
||||
if sum.WorldsReaped != 0 || sum.Skipped != 1 || cl.deletePVCCalls != 0 {
|
||||
t.Fatalf("summary %+v deletes=%d: want the world kept once the hold was lost", sum, cl.deletePVCCalls)
|
||||
}
|
||||
if st.byName["india"].OwnerID != "user-1" {
|
||||
t.Fatal("ownership released after the hold was lost")
|
||||
}
|
||||
|
||||
ar.onArchive = nil
|
||||
if sum := mustRun(t, r); sum.WorldsReaped != 1 || ar.archives != 1 {
|
||||
t.Fatalf("retry = %+v archives=%d, want the recorded archive reused", sum, ar.archives)
|
||||
}
|
||||
}
|
||||
|
||||
// data-durability-19: an unowned server with no world (reaped before) is not
|
||||
// reaped again when its clock runs out; the clock restarts, with no audit and
|
||||
// no count.
|
||||
func TestReapUnownedWithoutWorldRestartsClock(t *testing.T) {
|
||||
r, st, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "juliet", LastActiveAt: idleBy(16 * Day)})
|
||||
cl.noWorld = map[string]bool{"world-juliet-0": true}
|
||||
|
||||
sum := mustRun(t, r)
|
||||
if sum.WorldsReaped != 0 || sum.Skipped != 0 || sum.Failed() || ar.archives != 0 || cl.deletePVCCalls != 0 || len(st.audits) != 0 {
|
||||
t.Fatalf("summary %+v archives=%d deletes=%d audits=%v: want only the clock restarted",
|
||||
sum, ar.archives, cl.deletePVCCalls, st.audits)
|
||||
}
|
||||
if !st.byName["juliet"].LastActiveAt.Equal(testNow) {
|
||||
t.Fatalf("last_active_at = %v, want restarted at %v", st.byName["juliet"].LastActiveAt, testNow)
|
||||
}
|
||||
if sum := mustRun(t, r); sum.WorldsReaped != 0 || len(st.audits) != 0 {
|
||||
t.Fatalf("second run = %+v audits=%v", sum, st.audits)
|
||||
}
|
||||
}
|
||||
|
||||
// An owner who never started the server gives it up at the deadline like any
|
||||
// other; there is nothing to archive, and no world was deleted to count.
|
||||
func TestReapOwnedWithoutWorldReleases(t *testing.T) {
|
||||
r, st, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "kappa", OwnerID: "user-4", LastActiveAt: idleBy(16 * Day)})
|
||||
cl.noWorld = map[string]bool{"world-kappa-0": true}
|
||||
|
||||
sum := mustRun(t, r)
|
||||
if sum.WorldsReaped != 0 || sum.Skipped != 0 || ar.archives != 0 || cl.deletePVCCalls != 0 {
|
||||
t.Fatalf("summary %+v archives=%d deletes=%d", sum, ar.archives, cl.deletePVCCalls)
|
||||
}
|
||||
if st.byName["kappa"].OwnerID != "" || len(st.audits) != 1 || st.audits[0].FormerOwner != "user-4" {
|
||||
t.Fatalf("owner %q audits %+v: want released and audited", st.byName["kappa"].OwnerID, st.audits)
|
||||
}
|
||||
}
|
||||
|
||||
// A run that deleted the PVC and failed before releasing the world is finished
|
||||
// by the next: released, audited and counted once, without a new archive.
|
||||
func TestReapFinishesInterruptedReap(t *testing.T) {
|
||||
r, st, cl, ar := newReaper(DefaultConfig(),
|
||||
Candidate{Name: "lambda", OwnerID: "user-5", LastActiveAt: idleBy(20 * Day)})
|
||||
cl.noWorld = map[string]bool{"world-lambda-0": true}
|
||||
st.backups = []*fakeBackup{{id: "prev", server: "lambda", ref: "ref-prev", reason: ReasonInactive, size: 5,
|
||||
status: "present", createdAt: idleBy(Day), expires: testNow.Add(89 * Day)}}
|
||||
|
||||
sum := mustRun(t, r)
|
||||
if sum.WorldsReaped != 1 || ar.archives != 0 || cl.deletePVCCalls != 0 {
|
||||
t.Fatalf("summary %+v archives=%d deletes=%d: want the reap finished", sum, ar.archives, cl.deletePVCCalls)
|
||||
}
|
||||
if st.byName["lambda"].OwnerID != "" || len(st.audits) != 1 {
|
||||
t.Fatalf("owner %q audits %+v", st.byName["lambda"].OwnerID, st.audits)
|
||||
}
|
||||
}
|
||||
@@ -18,7 +18,7 @@
|
||||
"no_backup": "There's no restorable backup for this server yet.",
|
||||
"backup_corrupt": "This backup failed its read-back check and can't be restored intact — pick another backup.",
|
||||
"not_stopped": "Stop the server completely before restoring — a restore overwrites the live world volume.",
|
||||
"maintenance_in_progress": "This server's world is busy with a restore, backup or file write — try again once it finishes, usually within a minute or two.",
|
||||
"maintenance_in_progress": "This server's world is busy with a restore, backup, file write or idle reclaim — try again once it finishes. A restore, backup or file write usually takes a minute or two; an idle reclaim can take longer on a large world.",
|
||||
"no_world_volume": "This server has no world volume yet — start it once so it is created, then retry.",
|
||||
"backup_cooldown": "This server was backed up moments ago, and manual backups have a cooldown — try again in a few minutes.",
|
||||
"backup_store_full": "The backup store is full, so manual backups are paused — ask an administrator to free space.",
|
||||
|
||||
@@ -18,7 +18,7 @@
|
||||
"no_backup": "这台服务器暂时没有可回档的备份。",
|
||||
"backup_corrupt": "这份备份回读校验未通过,已无法完整恢复——请选择另一份备份。",
|
||||
"not_stopped": "回档会覆盖世界的实时存储卷,请先把服务器完全停止再回档。",
|
||||
"maintenance_in_progress": "这台服务器的世界正在回档、备份或写入文件——等它完成后再试,通常不超过一两分钟。",
|
||||
"maintenance_in_progress": "这台服务器的世界正在回档、备份、写入文件或闲置回收——等它完成后再试。回档、备份和写文件通常一两分钟,闲置回收视世界大小可能更久。",
|
||||
"no_world_volume": "这台服务器还没有世界卷——先启动一次让它创建,然后再试。",
|
||||
"backup_cooldown": "这台服务器刚备份过,手动备份之间有冷却时间——请过几分钟再试。",
|
||||
"backup_store_full": "备份存储已满,暂时无法手动备份——请联系管理员清理空间。",
|
||||
|
||||
Reference in new issue
Block a user