fix(backup): 手动备份按服冷却、每服保留上限与独立保留期,容量驱逐不再删除回收世界的唯一副本

This commit is contained in:
Lemon-miaow committed 2026-09-24 19:58:59 +08:00
1 parent f69cdec9d4
commit 9bda3a52fa
30 files changed
+801 -42

No files matched your search

+8
View File
@@ -128,6 +128,14 @@ type API struct {
// on the wake lever). Zero disables throttling.
WakeCooldown time.Duration
// BackupCooldown spaces out an owner's on-demand backups of one server, and
// BackupStoreCap refuses them once the present backups reach [archive]
// max_local_bytes (data-durability-9): each archive lands on the node disk
// the worlds and the database share. Admins and the break-glass console are
// exempt. Zero disables each lever.
BackupCooldown time.Duration
BackupStoreCap int64
// SubmitCreateCooldown / SubmitUploadCooldown throttle the user-modpack
// submission lane per user: create bounds how quickly review-queue rows can
// appear, upload bounds how often a user may stream a (up to 1 GiB) build
+26
View File
@@ -48,6 +48,9 @@ type fakeRepo struct {
resourceUpdates map[string]ResourceSpec
audits []AuditEntry
failAudit error // Audit fails with it (a store outage)
// backupRequested mirrors the newest backup.create audit row per server,
// stamped by Audit with the wall clock (LastBackupRequest).
backupRequested map[string]time.Time
joins []string
// create-server seeding (spec §15)
seeded map[string]bool // name -> servers row exists
@@ -750,9 +753,32 @@ func (f *fakeRepo) Audit(_ context.Context, e AuditEntry) error {
return f.failAudit
}
f.audits = append(f.audits, e)
if e.Action == "backup.create" {
if f.backupRequested == nil {
f.backupRequested = map[string]time.Time{}
}
f.backupRequested[e.ServerName] = time.Now()
}
return nil
}
func (f *fakeRepo) LastBackupRequest(_ context.Context, serverName string, since time.Time) (time.Time, error) {
if at, ok := f.backupRequested[serverName]; ok && !at.Before(since) {
return at, nil
}
return time.Time{}, nil
}
func (f *fakeRepo) BackupStoreBytes(context.Context) (int64, error) {
var n int64
for _, b := range f.backups {
if b.view.Status == "present" {
n += b.view.SizeBytes
}
}
return n, nil
}
// AllBackups / BackupsForUser / LatestBackup mirror the PG queries' contract so
// the hermetic tests can't pass against a too-lenient fake: only status='present'
// rows are visible, the user scope is the former_owner column, and LatestBackup
+74
View File
@@ -5,7 +5,9 @@ import (
"encoding/json"
"errors"
"net/http"
"strconv"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
)
@@ -191,6 +193,78 @@ func TestBackupNow(t *testing.T) {
t.Fatalf("code = %d, want 400", w.Code)
}
})
// data-durability-9: an owner's backups are rationed per server; an admin's
// are not.
t.Run("owner inside the cooldown -> 429 backup_cooldown with Retry-After", func(t *testing.T) {
api, _, _, backuper := mk()
api.BackupCooldown = 10 * time.Minute
api.Now = time.Now // the fake stamps backup.create audits with the wall clock
api.External = staticExternal{p: owner}
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("first backup: code = %d (%s)", w.Code, w.Body.String())
}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "backup_cooldown" {
t.Fatalf("second backup: code = %d body %s", w.Code, w.Body.String())
}
if ra, _ := strconv.Atoi(w.Header().Get("Retry-After")); ra < 590 || ra > 600 {
t.Fatalf("Retry-After = %q, want about 600", w.Header().Get("Retry-After"))
}
if backuper.calls != 1 {
t.Fatalf("backuper called %d times, want 1", backuper.calls)
}
})
t.Run("cooldown elapsed -> 202", func(t *testing.T) {
api, repo, _, _ := mk()
api.BackupCooldown = 10 * time.Minute
api.Now = time.Now
repo.backupRequested = map[string]time.Time{"survival": time.Now().Add(-11 * time.Minute)}
api.External = staticExternal{p: owner}
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
}
})
t.Run("admin bypasses the cooldown and the store cap", func(t *testing.T) {
api, repo, _, backuper := mk()
api.BackupCooldown = 10 * time.Minute
api.BackupStoreCap = 100
api.Now = time.Now
repo.backupRequested = map[string]time.Time{"survival": time.Now()}
repo.backups = []fakeBackup{{view: BackupView{ID: "b1", ServerName: "other", Status: "present", SizeBytes: 500}}}
api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "[email protected]",
Role: "admin", ViaAdminAccess: true}}
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
}
if backuper.calls != 1 {
t.Fatal("the admin's backup did not start")
}
})
t.Run("owner with the store at its cap -> 507 backup_store_full", func(t *testing.T) {
api, repo, _, backuper := mk()
api.BackupStoreCap = 1000
repo.backups = []fakeBackup{
{view: BackupView{ID: "b1", ServerName: "other", Status: "present", SizeBytes: 600}},
{view: BackupView{ID: "b2", ServerName: "survival", Status: "present", SizeBytes: 400}},
{view: BackupView{ID: "b3", ServerName: "survival", Status: "deleted", SizeBytes: 9000}},
}
api.External = staticExternal{p: owner}
w := do(api.ExternalHandler(), "POST", path, "", nil)
if w.Code != http.StatusInsufficientStorage || decodeErr(t, w) != "backup_store_full" {
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
}
if backuper.calls != 0 {
t.Fatal("a full store still started a backup")
}
repo.backups[0].view.Status = "deleted"
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
t.Fatalf("below the cap: code = %d (%s)", w.Code, w.Body.String())
}
})
}
// TestInternalBackup exercises POST /api/v1/internal/servers/{name}/backup, the
+41
View File
@@ -1,9 +1,11 @@
package api
import (
"context"
"errors"
"net/http"
"strings"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
@@ -255,10 +257,49 @@ func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
writeError(w, r, errForbidden)
return
}
if !p.IsAdmin() {
if err := a.backupAllowance(r.Context(), name); err != nil {
writeError(w, r, err)
return
}
}
a.enqueueBackup(w, r, name, rec, auditActor(p), "external")
}
// backupAllowance rations an owner's on-demand backups (data-durability-9):
// one per BackupCooldown per server, and none while the present backups fill
// BackupStoreCap. The owner's older backups are pruned by the Job itself
// ([archive] manual_keep), so these two gates bound the rate and the total.
func (a *API) backupAllowance(ctx context.Context, name string) error {
if a.BackupCooldown > 0 {
now := a.now()
last, err := a.Repo.LastBackupRequest(ctx, name, now.Add(-a.BackupCooldown))
if err != nil {
return err
}
if !last.IsZero() {
wait := last.Add(a.BackupCooldown).Sub(now)
if wait > 0 {
return newError(http.StatusTooManyRequests, "backup_cooldown",
"a backup of this server was started %s ago; the next one can start in %s",
now.Sub(last).Round(time.Second), wait.Round(time.Second)).retryAfter(wait)
}
}
}
if a.BackupStoreCap > 0 {
used, err := a.Repo.BackupStoreBytes(ctx)
if err != nil {
return err
}
if used >= a.BackupStoreCap {
return newError(http.StatusInsufficientStorage, "backup_store_full",
"the backup store is full; ask an administrator to free space")
}
}
return nil
}
// handleInternalBackup is the internal-face backup trigger. The break-glass console
// (root on the node, holding the service token) POSTs here to snapshot a stopped
// world while felis-api is alive — it goes through the API rather than direct-to-CRD
+58
View File
@@ -11,6 +11,9 @@ import (
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
type fakeJobStatus struct {
@@ -145,3 +148,58 @@ func TestJobToAsyncJob(t *testing.T) {
t.Fatal("foreign job must be dropped")
}
}
// TestLatestJobsExplainsFailures: a failed Job reports the error its container
// exited on (the last line of the terminated message), newest pod first, and
// keeps the condition text when no pod explains it.
func TestLatestJobsExplainsFailures(t *testing.T) {
scheme := runtime.NewScheme()
if err := clientgoscheme.AddToScheme(scheme); err != nil {
t.Fatal(err)
}
labels := func(job string) map[string]string {
return map[string]string{jobServerLabel: "survival", jobManagedByLabel: jobManagedByBackup, "job-name": job}
}
at := func(min int) metav1.Time { return metav1.NewTime(time.Date(2026, 9, 24, 10, min, 0, 0, time.UTC)) }
failedJob := func(name string, min int) *batchv1.Job {
start := at(min)
return &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: "minecraft", Labels: labels(name)},
Status: batchv1.JobStatus{StartTime: &start, Conditions: []batchv1.JobCondition{{
Type: batchv1.JobFailed, Status: corev1.ConditionTrue,
Reason: "BackoffLimitExceeded", Message: "Job has reached the specified backoff limit",
}}},
}
}
pod := func(name, job string, min int, exit int32, msg string) *corev1.Pod {
return &corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: "minecraft", Labels: labels(job), CreationTimestamp: at(min)},
Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{
Name: "backup", State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: exit, Message: msg}},
}}},
}
}
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
failedJob("backup-survival-a", 1),
failedJob("backup-survival-b", 2),
pod("a-1", "backup-survival-a", 1, 1, "felis backup: first try\n"),
pod("a-2", "backup-survival-a", 3, 1,
"archiving survival\nfelis backup: backup: not enough free disk for the archive: the world is 2.0 GiB\n"),
pod("b-1", "backup-survival-b", 2, 0, "done"),
).Build()
jobs, err := NewK8sJobStatus(c, "minecraft").LatestJobs(context.Background(), "survival")
if err != nil {
t.Fatal(err)
}
got := map[string]string{}
for _, j := range jobs {
got[j.Name] = j.Message
}
if want := "felis backup: backup: not enough free disk for the archive: the world is 2.0 GiB"; got["backup-survival-a"] != want {
t.Errorf("a: message = %q, want %q", got["backup-survival-a"], want)
}
if want := "Job has reached the specified backoff limit"; got["backup-survival-b"] != want {
t.Errorf("b: message = %q, want the condition text", got["backup-survival-b"])
}
}
+59
View File
@@ -3,6 +3,8 @@ package api
import (
"context"
"sort"
"strings"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
@@ -53,9 +55,66 @@ func (k *K8sJobStatus) LatestJobs(ctx context.Context, serverName string) ([]Asy
if len(out) > 20 {
out = out[:20]
}
k.explainFailures(ctx, serverName, out)
return out, nil
}
// explainFailures replaces a failed Job's condition text ("Job has reached the
// specified backoff limit") with the error its container exited on. The
// executors set TerminationMessagePolicy FallbackToLogsOnError, so the
// terminated state carries the tail of the log, whose last line is the
// "felis backup: …" / "felis restore: …" error. The pods carry the Job's
// labels, so one list covers every Job of the server; a pod already collected
// by the Job TTL, or a list error, leaves the condition text in place.
func (k *K8sJobStatus) explainFailures(ctx context.Context, serverName string, jobs []AsyncJob) {
failed := map[string]int{}
for i, j := range jobs {
if j.State == "failed" {
failed[j.Name] = i
}
}
if len(failed) == 0 {
return
}
var pods corev1.PodList
if err := k.c.List(ctx, &pods, client.InNamespace(k.namespace),
client.MatchingLabels{jobServerLabel: serverName}); err != nil {
return
}
newest := map[string]time.Time{}
for i := range pods.Items {
pod := &pods.Items[i]
idx, ok := failed[pod.Labels["job-name"]]
if !ok {
continue
}
msg := lastTerminationLine(pod)
if msg == "" || !pod.CreationTimestamp.Time.After(newest[pod.Labels["job-name"]]) {
continue
}
newest[pod.Labels["job-name"]] = pod.CreationTimestamp.Time
jobs[idx].Message = msg
}
}
// lastTerminationLine returns the last non-empty line of the pod's terminated
// container message, capped for display.
func lastTerminationLine(pod *corev1.Pod) string {
for _, cs := range pod.Status.ContainerStatuses {
t := cs.State.Terminated
if t == nil || t.ExitCode == 0 {
continue
}
lines := strings.Split(strings.TrimSpace(t.Message), "\n")
line := strings.TrimSpace(lines[len(lines)-1])
if len(line) > 400 {
line = line[:400] + "…"
}
return line
}
return ""
}
// jobToAsyncJob projects one Job onto its kind/state/message. Complete condition →
// succeeded, Failed → failed with its reason (Job conditions carry the generic
// "backoff limit exceeded" text; the pod log holds the underlying error), anything
+22
View File
@@ -747,6 +747,28 @@ func (p *PGRepo) BackupByID(ctx context.Context, id string) (*BackupRecord, erro
return &b, nil
}
// LastBackupRequest reads the newest backup.create audit row for the server
// since the given time; the created_at index bounds the scan to that window.
func (p *PGRepo) LastBackupRequest(ctx context.Context, serverName string, since time.Time) (time.Time, error) {
var at sql.NullTime
err := p.db.QueryRowContext(ctx,
`SELECT max(created_at) FROM audit_logs
WHERE created_at >= $2 AND action = 'backup.create' AND server_name = $1`,
serverName, since).Scan(&at)
if err != nil {
return time.Time{}, err
}
return at.Time, nil
}
// BackupStoreBytes sums size_bytes over the present world backups.
func (p *PGRepo) BackupStoreBytes(ctx context.Context) (int64, error) {
var n int64
err := p.db.QueryRowContext(ctx,
`SELECT COALESCE(sum(size_bytes), 0) FROM world_backups WHERE status = 'present'`).Scan(&n)
return n, err
}
func (p *PGRepo) Audit(ctx context.Context, e AuditEntry) error {
// A nil Payload must land as SQL NULL, not the text "null"; a non-nil Payload is
// passed as a JSON text the jsonb column parses (same idiom as reaper.PGStore).
+8
View File
@@ -297,6 +297,14 @@ type Repo interface {
// none matches. Like LatestBackup the returned BackupRecord carries the
// server-side backup_ref the restore path needs; the client never sees it.
BackupByID(ctx context.Context, id string) (*BackupRecord, error)
// LastBackupRequest returns when an on-demand backup of the server was last
// accepted (its newest backup.create audit row) at or after since, or the zero
// time when there was none. The since bound keeps the lookup inside the
// cooldown window the caller enforces.
LastBackupRequest(ctx context.Context, serverName string, since time.Time) (time.Time, error)
// BackupStoreBytes sums the sizes of every present world backup, the figure
// [archive] max_local_bytes caps.
BackupStoreBytes(ctx context.Context) (int64, error)
// SeedServer inserts the business-layer rows for a newly created server (spec
// §15): a servers row (owner_id NULL — claimed later, spec §9.3) and its
// subdomain alias, both idempotent. The resource cache (cpuMilli, memoryMB,