feat(restore): 恢复前默认为当前世界做安全快照并串接恢复 Job,快照失败则不恢复;并发恢复另一备份返回 409
This commit is contained in:
33 files changed
+1260
-105
No files matched your search
+34
-10
@@ -287,6 +287,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
cluster := api.NewK8sCluster(cl, cfg.K8s.Namespace)
|
||||
jobStatus := api.NewK8sJobStatus(cl, cfg.K8s.Namespace)
|
||||
a := &api.API{
|
||||
Repo: repo,
|
||||
Cluster: cluster,
|
||||
@@ -294,16 +295,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
Logs: api.NewK8sLogStreamer(clientset, cfg.K8s.Namespace),
|
||||
// Build-log stream (spec §16) is scoped to the BUILD namespace — the same
|
||||
// value the Builder renders Jobs into — so it follows where build Pods run.
|
||||
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
|
||||
Internal: api.BearerTokenAuth{Token: token},
|
||||
Builder: builder,
|
||||
Images: imagePinner(cfg.Registry.URL),
|
||||
Restorer: restorer,
|
||||
Backuper: backuper,
|
||||
JobStatus: api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
|
||||
Files: files,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
|
||||
Internal: api.BearerTokenAuth{Token: token},
|
||||
Builder: builder,
|
||||
Images: imagePinner(cfg.Registry.URL),
|
||||
Restorer: restorer,
|
||||
Backuper: backuper,
|
||||
JobStatus: jobStatus,
|
||||
// A restore starts with a safety snapshot; settleRestoreChains starts the
|
||||
// restore behind each one.
|
||||
RestoreChains: jobStatus,
|
||||
Files: files,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
// The external face is fronted by SessionAuth: it prefers a local session
|
||||
// cookie (minted by the passwordless doors) and otherwise delegates to the
|
||||
// Cloudflare-Access JWT verifier, so both auth models coexist on one face. The
|
||||
@@ -410,6 +414,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// and advance any whose Job has reached a terminal phase. GET on a build also
|
||||
// reconciles it, but this loop converges builds nobody is polling.
|
||||
go reconcileBuilds(ctx, builder, stderr)
|
||||
go settleRestoreChains(ctx, a, stderr)
|
||||
|
||||
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
|
||||
go pruner.Loop(ctx, registryPruneInterval)
|
||||
@@ -650,6 +655,25 @@ func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
|
||||
}
|
||||
}
|
||||
|
||||
// settleRestoreChains starts the restore behind each safety snapshot that has
|
||||
// finished (and gives up the one behind a snapshot that failed). The world stays
|
||||
// locked in between, so the interval is how long a finished snapshot keeps the
|
||||
// server down before its restore begins.
|
||||
func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
|
||||
t := time.NewTicker(5 * time.Second)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
if err := a.SettleRestoreChains(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: restore chains: %v\n", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
|
||||
// submissions rejected more than submit.RejectedContextRetention ago. Without it a
|
||||
// rejected modpack keeps its bytes on the uploads store (and against its
|
||||
|
||||
+29
-9
@@ -10,6 +10,7 @@ import (
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/backup"
|
||||
"felis.lolicon.best/internal/backupjob"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
@@ -36,6 +37,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
server := fs.String("server", "", "server name whose world is being backed up")
|
||||
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
|
||||
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
|
||||
reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
|
||||
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -43,6 +46,11 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintln(stderr, "felis backup: --server is required")
|
||||
return 2
|
||||
}
|
||||
keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
|
||||
if !ok {
|
||||
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
|
||||
return 2
|
||||
}
|
||||
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
if err != nil {
|
||||
@@ -60,6 +68,9 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if keep < 0 {
|
||||
keep = rcfg.ManualKeep
|
||||
}
|
||||
|
||||
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
|
||||
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
|
||||
@@ -99,7 +110,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
FormerOwner: *formerOwner,
|
||||
BackupRef: string(ref),
|
||||
SizeBytes: size,
|
||||
Reason: "manual",
|
||||
Reason: *reason,
|
||||
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
|
||||
}
|
||||
st := reaper.NewPGStore(drv.DB())
|
||||
@@ -116,16 +127,25 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
||||
pruneManualBackups(ctx, st, archiver, *server, rcfg.ManualKeep, stdout, stderr)
|
||||
pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
|
||||
return 0
|
||||
}
|
||||
|
||||
// pruneManualBackups keeps server's newest keep on-demand backups and removes
|
||||
// the rest, oldest first, so repeated backups of one world cannot fill the
|
||||
// shared archive store. The new backup is already recorded; a removal that
|
||||
// fails is reported and retried after the next backup.
|
||||
func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server string, keep int, stdout, stderr io.Writer) {
|
||||
excess, err := st.ExcessManualBackups(ctx, server, keep)
|
||||
const (
|
||||
reasonManual = "manual"
|
||||
// preRestoreKeep is how many safety snapshots a server keeps: enough to walk
|
||||
// back a couple of restores in a row, without every restore adding a world's
|
||||
// worth of bytes for the full manual retention.
|
||||
preRestoreKeep = 3
|
||||
)
|
||||
|
||||
// pruneBackups keeps server's newest keep backups of this reason and removes the
|
||||
// rest, oldest first, so repeated backups of one world cannot fill the shared
|
||||
// archive store. protect is never removed: it is the backup a chained restore is
|
||||
// about to extract. The new backup is already recorded; a removal that fails is
|
||||
// reported and retried after the next backup.
|
||||
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
|
||||
excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
|
||||
return
|
||||
@@ -139,7 +159,7 @@ func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup
|
||||
fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis backup: removed older backup %s of %s (keeping the newest %d)\n", b.ID, server, keep)
|
||||
fmt.Fprintf(stdout, "felis backup: removed older %s backup %s of %s (keeping the newest %d)\n", reason, b.ID, server, keep)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The reason decides which backups the new one's prune may remove, so an
|
||||
// unknown one is refused before anything is archived.
|
||||
func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
|
||||
var stderr bytes.Buffer
|
||||
if code := cmdBackup([]string{"--server", "survival", "--reason", "inactive_15d"}, &bytes.Buffer{}, &stderr); code != 2 {
|
||||
t.Fatalf("exit = %d, want 2 (%s)", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "unknown --reason") {
|
||||
t.Fatalf("stderr = %q", stderr.String())
|
||||
}
|
||||
}
|
||||
+37
-4
@@ -362,7 +362,9 @@ components:
|
||||
type: string
|
||||
description: Present only when the world had an owner at backup time.
|
||||
size_bytes: { type: integer, format: int64 }
|
||||
reason: { type: string }
|
||||
reason:
|
||||
type: string
|
||||
description: inactive_15d (idle reclaim), manual (on demand) or pre_restore (the safety snapshot in front of a restore).
|
||||
status: { type: string }
|
||||
created_at: { type: string, format: date-time }
|
||||
expires_at: { type: string, format: date-time }
|
||||
@@ -2961,6 +2963,15 @@ paths:
|
||||
tags: [backups]
|
||||
operationId: restoreBackup
|
||||
summary: Restore a world from a backup (owner-or-admin plus a former-owner match).
|
||||
description: >-
|
||||
By default the restore starts with a safety snapshot: a backup of the
|
||||
data volume as it is now (reason "pre_restore", the newest 3 kept per
|
||||
server), and the restore Job starts only once that backup has
|
||||
succeeded. If the snapshot fails the restore is given up and the world
|
||||
is left as it was. GET /servers/{name}/jobs shows the snapshot as a
|
||||
backup job whose then_restore says what became of the restore. The
|
||||
world stays locked from the request until the restore Job finishes.
|
||||
Pass safety_snapshot false to restore straight away.
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: app
|
||||
security: [{ accessJWT: [] }]
|
||||
@@ -2976,18 +2987,25 @@ paths:
|
||||
backup_id:
|
||||
type: string
|
||||
description: Which backup to restore; defaults to the latest for the server.
|
||||
safety_snapshot:
|
||||
type: boolean
|
||||
default: true
|
||||
description: Back up the current world before overwriting it.
|
||||
responses:
|
||||
'202':
|
||||
description: Restore started.
|
||||
description: Restore started (after the safety snapshot when safety_snapshot is true).
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
type: object
|
||||
required: [name, status, backup_id]
|
||||
required: [name, status, backup_id, safety_snapshot]
|
||||
properties:
|
||||
name: { type: string }
|
||||
status: { type: string, const: restoring }
|
||||
backup_id: { type: string }
|
||||
safety_snapshot:
|
||||
type: boolean
|
||||
description: Whether a safety snapshot runs first. False when the request turned it off, or when this install cannot take one.
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
@@ -2998,7 +3016,7 @@ paths:
|
||||
application/json:
|
||||
schema: { $ref: '#/components/schemas/Error' }
|
||||
'409':
|
||||
description: Server is not stopped (not_stopped), or a restore, backup or file write already holds its world volume (maintenance_in_progress).
|
||||
description: Server is not stopped (not_stopped), a restore, backup or file write already holds its world volume (maintenance_in_progress), or a restore of another backup is still running (restore_in_progress).
|
||||
content:
|
||||
application/json:
|
||||
schema: { $ref: '#/components/schemas/Error' }
|
||||
@@ -3088,6 +3106,21 @@ paths:
|
||||
message: { type: string }
|
||||
started_at: { type: string, format: date-time }
|
||||
finished_at: { type: string, format: date-time }
|
||||
then_restore:
|
||||
type: string
|
||||
enum: [pending, started, abandoned]
|
||||
description: >-
|
||||
Set on a restore's safety snapshot (a backup job):
|
||||
pending until the restore behind it starts, or
|
||||
abandoned with then_restore_reason saying why (its
|
||||
English wording is in message).
|
||||
then_restore_reason:
|
||||
type: string
|
||||
enum: [snapshot_failed, not_configured, server_gone, server_started, restore_busy]
|
||||
description: Why an abandoned chain was given up.
|
||||
restore_backup_id:
|
||||
type: string
|
||||
description: The backup the chained restore extracts.
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
|
||||
@@ -75,6 +75,12 @@ type API struct {
|
||||
// endpoints only answer 202). Optional: nil → that route reports 503.
|
||||
JobStatus JobStatusReader
|
||||
|
||||
// RestoreChains finds and settles the safety snapshots that run in front of a
|
||||
// restore (restorechain.go). A restore takes a snapshot first only when this
|
||||
// is set, the Backuper can chain a restore and the Restorer is wired, since
|
||||
// something has to start the restore once the snapshot is done.
|
||||
RestoreChains RestoreChains
|
||||
|
||||
// Files is the server file editor (list / read / write a file in a stopped
|
||||
// server's world volume — the "one wrong line in server.properties" repair).
|
||||
// Like Restorer and Backuper it is optional: when nil the file routes report
|
||||
|
||||
@@ -23,3 +23,13 @@ import "context"
|
||||
type Backuper interface {
|
||||
Backup(ctx context.Context, serverName, formerOwner string) error
|
||||
}
|
||||
|
||||
// RestoreSnapshotter is the Backuper's safety-snapshot lever: it enqueues a backup
|
||||
// of the world as it is now, labelled with the restore to run once that backup
|
||||
// has succeeded (backupID, backupRef). RestoreChains settles it: it starts the
|
||||
// restore after a successful snapshot and gives the restore up after a failed
|
||||
// one, so a restore never overwrites a world that has no copy. Until it is
|
||||
// settled the snapshot holds the world volume as a restore (internal/maintenance).
|
||||
type RestoreSnapshotter interface {
|
||||
BackupThenRestore(ctx context.Context, serverName, formerOwner, backupID, backupRef string) error
|
||||
}
|
||||
@@ -76,8 +76,14 @@ func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
|
||||
// makes that atomic against a wake and refuses a second restore, backup or
|
||||
// file write on the same world with 409 maintenance_in_progress until the
|
||||
// restore Job finishes.
|
||||
// ⑧ hand off to the Restorer. Restore is asynchronous (a restore Job, like an
|
||||
// image build Job), so success means "enqueued" and the handler answers 202.
|
||||
// ⑧ hand off. By default the restore starts with a safety snapshot of the world
|
||||
// as it is (restorechain.go): a pre_restore backup Job that the restore Job
|
||||
// follows once it succeeds, so a wrong pick can be walked back from the
|
||||
// backup list. "safety_snapshot": false in the body restores straight away.
|
||||
// Either way the work is asynchronous (Jobs, like an image build), so
|
||||
// success means "enqueued" and the handler answers 202, saying in
|
||||
// safety_snapshot which of the two it did. A restore of another backup still
|
||||
// running on the world is 409 restore_in_progress.
|
||||
//
|
||||
// The opaque backup_ref is resolved server-side from the backup and handed to the
|
||||
// Restorer directly; the client never names a backup by handle (spec §286
|
||||
@@ -104,8 +110,10 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
// Optional backup_id in the JSON body; absent → LatestBackup (backward compat).
|
||||
// safety_snapshot defaults to true.
|
||||
var body struct {
|
||||
BackupID string `json:"backup_id"`
|
||||
BackupID string `json:"backup_id"`
|
||||
SafetySnapshot *bool `json:"safety_snapshot"`
|
||||
}
|
||||
if strings.HasPrefix(r.Header.Get("Content-Type"), "application/json") {
|
||||
if err := decodeJSON(w, r, &body); err != nil {
|
||||
@@ -204,7 +212,23 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
defer release()
|
||||
|
||||
if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil {
|
||||
snapshotter, snapshot := a.snapshotFirst()
|
||||
if body.SafetySnapshot != nil && !*body.SafetySnapshot {
|
||||
snapshot = false
|
||||
}
|
||||
if snapshot {
|
||||
// The snapshot records the current owner, like an on-demand backup, so
|
||||
// the way back is theirs to take.
|
||||
err = snapshotter.BackupThenRestore(r.Context(), name, rec.OwnerID, backup.ID, backup.BackupRef)
|
||||
} else {
|
||||
err = a.Restorer.Restore(r.Context(), name, backup.BackupRef)
|
||||
}
|
||||
if err != nil {
|
||||
if isRestoreInProgress(err) {
|
||||
writeError(w, r, newError(http.StatusConflict, "restore_in_progress",
|
||||
"a restore of another backup is still running on this server's world; retry once it finishes"))
|
||||
return
|
||||
}
|
||||
// ErrNotFound (server vanished from the execution backend) → 404; else 500.
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
@@ -212,9 +236,10 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
a.audit(r, "backup.restore", name)
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||
"name": name,
|
||||
"status": "restoring",
|
||||
"backup_id": backup.ID,
|
||||
"name": name,
|
||||
"status": "restoring",
|
||||
"backup_id": backup.ID,
|
||||
"safety_snapshot": snapshot,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -19,6 +19,13 @@ type AsyncJob struct {
|
||||
Message string `json:"message,omitempty"`
|
||||
StartedAt time.Time `json:"started_at,omitzero"`
|
||||
FinishedAt time.Time `json:"finished_at,omitzero"`
|
||||
// ThenRestore is set on a restore's safety snapshot (a backup Job): "pending"
|
||||
// until the restore behind it starts ("started") or is given up
|
||||
// ("abandoned", with a ChainAbandon* code in ThenRestoreReason and its
|
||||
// English in Message). RestoreBackupID is the backup that restore extracts.
|
||||
ThenRestore string `json:"then_restore,omitempty"`
|
||||
ThenRestoreReason string `json:"then_restore_reason,omitempty"`
|
||||
RestoreBackupID string `json:"restore_backup_id,omitempty"`
|
||||
}
|
||||
|
||||
// JobStatusReader reads the newest backup/restore Jobs for a server, newest
|
||||
|
||||
@@ -2,12 +2,16 @@ package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
@@ -115,11 +119,83 @@ func lastTerminationLine(pod *corev1.Pod) string {
|
||||
return ""
|
||||
}
|
||||
|
||||
// jobToAsyncJob projects one Job onto its kind/state/message. Complete condition →
|
||||
// jobToAsyncJob projects one Job onto the AsyncJob the jobs route answers with.
|
||||
func jobToAsyncJob(j *batchv1.Job) (AsyncJob, bool) {
|
||||
aj, ok := jobOutcome(j)
|
||||
if !ok {
|
||||
return aj, false
|
||||
}
|
||||
// A safety snapshot says what became of the restore behind it; a chain given
|
||||
// up after a snapshot that succeeded explains itself in the message.
|
||||
if state := j.Labels[maintenance.LabelThenRestore]; state != "" && aj.Kind == "backup" {
|
||||
aj.ThenRestore = state
|
||||
aj.RestoreBackupID = j.Annotations[maintenance.AnnotationRestoreBackupID]
|
||||
if reason := j.Annotations[maintenance.AnnotationThenRestoreReason]; reason != "" {
|
||||
aj.ThenRestoreReason = reason
|
||||
if aj.Message == "" {
|
||||
aj.Message = chainAbandonMessage(reason)
|
||||
}
|
||||
}
|
||||
}
|
||||
return aj, true
|
||||
}
|
||||
|
||||
// PendingRestoreChains lists the safety snapshots felis-api has yet to settle,
|
||||
// across every server (the label selector keeps it to them).
|
||||
func (k *K8sJobStatus) PendingRestoreChains(ctx context.Context) ([]RestoreChain, error) {
|
||||
var list batchv1.JobList
|
||||
if err := k.c.List(ctx, &list, client.InNamespace(k.namespace), client.MatchingLabels{
|
||||
jobManagedByLabel: jobManagedByBackup,
|
||||
maintenance.LabelThenRestore: maintenance.ThenRestorePending,
|
||||
}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]RestoreChain, 0, len(list.Items))
|
||||
for i := range list.Items {
|
||||
j := &list.Items[i]
|
||||
snapshot := ChainSnapshotRunning
|
||||
for _, c := range j.Status.Conditions {
|
||||
if c.Status != corev1.ConditionTrue {
|
||||
continue
|
||||
}
|
||||
switch c.Type {
|
||||
case batchv1.JobComplete, batchv1.JobSuccessCriteriaMet:
|
||||
snapshot = ChainSnapshotSucceeded
|
||||
case batchv1.JobFailed, batchv1.JobFailureTarget:
|
||||
snapshot = ChainSnapshotFailed
|
||||
}
|
||||
}
|
||||
out = append(out, RestoreChain{
|
||||
Job: j.Name,
|
||||
Server: j.Labels[jobServerLabel],
|
||||
BackupID: j.Annotations[maintenance.AnnotationRestoreBackupID],
|
||||
BackupRef: j.Annotations[maintenance.AnnotationRestoreRef],
|
||||
Snapshot: snapshot,
|
||||
})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// SettleRestoreChain records how a chain was settled on its snapshot Job, which
|
||||
// releases the world volume the chain was holding.
|
||||
func (k *K8sJobStatus) SettleRestoreChain(ctx context.Context, job, state, reason string) error {
|
||||
meta := map[string]any{"labels": map[string]string{maintenance.LabelThenRestore: state}}
|
||||
if reason != "" {
|
||||
meta["annotations"] = map[string]string{maintenance.AnnotationThenRestoreReason: reason}
|
||||
}
|
||||
patch, err := json.Marshal(map[string]any{"metadata": meta})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: k.namespace, Name: job}}
|
||||
return k.c.Patch(ctx, obj, client.RawPatch(types.MergePatchType, patch))
|
||||
}
|
||||
|
||||
// jobOutcome projects one Job onto its kind/state/message. Complete condition →
|
||||
// succeeded, Failed → failed with its reason (Job conditions carry the generic
|
||||
// "backoff limit exceeded" text; the pod log holds the underlying error), anything
|
||||
// else is still running.
|
||||
func jobToAsyncJob(j *batchv1.Job) (AsyncJob, bool) {
|
||||
func jobOutcome(j *batchv1.Job) (AsyncJob, bool) {
|
||||
kind := ""
|
||||
switch j.Labels[jobManagedByLabel] {
|
||||
case jobManagedByBackup:
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
)
|
||||
|
||||
// A restore overwrites the world, so by default it starts with a safety snapshot:
|
||||
// POST /servers/{name}/restore-backup enqueues a backup Job carrying the restore
|
||||
// (RestoreSnapshotter), and SettleRestoreChains starts that restore once the
|
||||
// snapshot has succeeded. The snapshot is an ordinary pre_restore row in the
|
||||
// backup list, which is the way back from a wrong pick.
|
||||
//
|
||||
// The chain lives on the backup Job (labels and annotations), so a felis-api
|
||||
// restart picks it up where it was. While it is pending the Job holds the world
|
||||
// volume as a restore, which keeps the server down in the gap between the two
|
||||
// Jobs; if felis-api never settles it, the lock goes with the Job at its TTL and
|
||||
// the world is left as it was.
|
||||
|
||||
// Snapshot Job states a RestoreChain reports.
|
||||
const (
|
||||
ChainSnapshotRunning = "running"
|
||||
ChainSnapshotSucceeded = "succeeded"
|
||||
ChainSnapshotFailed = "failed"
|
||||
)
|
||||
|
||||
// RestoreChain is one safety snapshot whose restore is still pending.
|
||||
type RestoreChain struct {
|
||||
Job string // the snapshot's backup Job
|
||||
Server string
|
||||
BackupID string // the backup the restore extracts
|
||||
BackupRef string
|
||||
Snapshot string // ChainSnapshotRunning / Succeeded / Failed
|
||||
}
|
||||
|
||||
// Why a chain was abandoned: the code recorded on the snapshot Job and reported
|
||||
// by the jobs route as then_restore_reason, for the panel to word in its own
|
||||
// language. chainAbandonText is the English the log and the job's message carry.
|
||||
const (
|
||||
ChainAbandonSnapshotFailed = "snapshot_failed"
|
||||
ChainAbandonNotConfigured = "not_configured"
|
||||
ChainAbandonServerGone = "server_gone"
|
||||
ChainAbandonServerStarted = "server_started"
|
||||
ChainAbandonRestoreBusy = "restore_busy"
|
||||
)
|
||||
|
||||
var chainAbandonText = map[string]string{
|
||||
ChainAbandonSnapshotFailed: "the safety backup failed, so the world was left as it was",
|
||||
ChainAbandonNotConfigured: "restore is not configured",
|
||||
ChainAbandonServerGone: "the server no longer exists",
|
||||
ChainAbandonServerStarted: "the server was started before the restore could run",
|
||||
ChainAbandonRestoreBusy: "another restore is running on this world",
|
||||
}
|
||||
|
||||
// chainAbandonMessage words a reason code; an unknown code (a newer felis-api
|
||||
// wrote it) is shown as is.
|
||||
func chainAbandonMessage(reason string) string {
|
||||
if s, ok := chainAbandonText[reason]; ok {
|
||||
return s
|
||||
}
|
||||
return reason
|
||||
}
|
||||
|
||||
// RestoreChains reads the pending chains and records how each was settled
|
||||
// (maintenance.ThenRestoreStarted or ThenRestoreAbandoned, with the reason code
|
||||
// of a chain given up).
|
||||
type RestoreChains interface {
|
||||
PendingRestoreChains(ctx context.Context) ([]RestoreChain, error)
|
||||
SettleRestoreChain(ctx context.Context, job, state, reason string) error
|
||||
}
|
||||
|
||||
// snapshotFirst reports whether a restore can start with a safety snapshot:
|
||||
// the Backuper can carry a restore and something settles the chain.
|
||||
func (a *API) snapshotFirst() (RestoreSnapshotter, bool) {
|
||||
if a.RestoreChains == nil || a.Restorer == nil {
|
||||
return nil, false
|
||||
}
|
||||
s, ok := a.Backuper.(RestoreSnapshotter)
|
||||
return s, ok
|
||||
}
|
||||
|
||||
// SettleRestoreChains advances every pending chain once: it starts the restore
|
||||
// behind a snapshot that succeeded and gives up the one behind a snapshot that
|
||||
// failed. A chain whose restore could not be created this time stays pending and
|
||||
// is retried on the next call. cmd/felis runs it on a short interval.
|
||||
func (a *API) SettleRestoreChains(ctx context.Context) error {
|
||||
if a.RestoreChains == nil {
|
||||
return nil
|
||||
}
|
||||
chains, err := a.RestoreChains.PendingRestoreChains(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var errs []error
|
||||
for _, c := range chains {
|
||||
state, reason, err := a.settleRestoreChain(ctx, c)
|
||||
if err != nil {
|
||||
errs = append(errs, fmt.Errorf("%s: %w", c.Job, err))
|
||||
continue
|
||||
}
|
||||
if state == "" {
|
||||
continue
|
||||
}
|
||||
if err := a.RestoreChains.SettleRestoreChain(ctx, c.Job, state, reason); err != nil {
|
||||
errs = append(errs, fmt.Errorf("%s: settle as %s: %w", c.Job, state, err))
|
||||
continue
|
||||
}
|
||||
if state == maintenance.ThenRestoreStarted {
|
||||
log.Printf("api: safety snapshot %s done; restoring %s from backup %s", c.Job, c.Server, c.BackupID)
|
||||
} else {
|
||||
log.Printf("api: restore of %s from backup %s abandoned: %s", c.Server, c.BackupID, chainAbandonMessage(reason))
|
||||
}
|
||||
}
|
||||
return errors.Join(errs...)
|
||||
}
|
||||
|
||||
// settleRestoreChain decides one chain: the state to record ("" to leave it
|
||||
// pending) and, for an abandoned chain, the reason code.
|
||||
func (a *API) settleRestoreChain(ctx context.Context, c RestoreChain) (string, string, error) {
|
||||
switch c.Snapshot {
|
||||
case ChainSnapshotFailed:
|
||||
return maintenance.ThenRestoreAbandoned, ChainAbandonSnapshotFailed, nil
|
||||
case ChainSnapshotSucceeded:
|
||||
default:
|
||||
return "", "", nil
|
||||
}
|
||||
if a.Restorer == nil {
|
||||
return maintenance.ThenRestoreAbandoned, ChainAbandonNotConfigured, nil
|
||||
}
|
||||
// The pending chain holds the world volume, so nothing should have started
|
||||
// the server; a start that got past it anyway must not have its world
|
||||
// replaced underneath.
|
||||
info, err := a.Cluster.GetServer(ctx, c.Server)
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
return maintenance.ThenRestoreAbandoned, ChainAbandonServerGone, nil
|
||||
}
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
|
||||
return maintenance.ThenRestoreAbandoned, ChainAbandonServerStarted, nil
|
||||
}
|
||||
if err := a.Restorer.Restore(ctx, c.Server, c.BackupRef); err != nil {
|
||||
if isRestoreInProgress(err) {
|
||||
return maintenance.ThenRestoreAbandoned, ChainAbandonRestoreBusy, nil
|
||||
}
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
return maintenance.ThenRestoreAbandoned, ChainAbandonServerGone, nil
|
||||
}
|
||||
return "", "", err
|
||||
}
|
||||
return maintenance.ThenRestoreStarted, "", nil
|
||||
}
|
||||
@@ -0,0 +1,310 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/backupjob"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/restore"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
// fakeSnapshotter is a Backuper that can also chain a restore.
|
||||
type fakeSnapshotter struct {
|
||||
fakeBackuper
|
||||
err error
|
||||
chained []RestoreChain
|
||||
}
|
||||
|
||||
func (f *fakeSnapshotter) BackupThenRestore(_ context.Context, name, formerOwner, backupID, backupRef string) error {
|
||||
f.gotFormerOwn = formerOwner
|
||||
f.chained = append(f.chained, RestoreChain{Server: name, BackupID: backupID, BackupRef: backupRef})
|
||||
return f.err
|
||||
}
|
||||
|
||||
type settled struct{ job, state, note string }
|
||||
|
||||
type fakeChains struct {
|
||||
pending []RestoreChain
|
||||
listErr error
|
||||
settleErr error
|
||||
settled []settled
|
||||
}
|
||||
|
||||
func (f *fakeChains) PendingRestoreChains(context.Context) ([]RestoreChain, error) {
|
||||
return f.pending, f.listErr
|
||||
}
|
||||
|
||||
func (f *fakeChains) SettleRestoreChain(_ context.Context, job, state, note string) error {
|
||||
if f.settleErr != nil {
|
||||
return f.settleErr
|
||||
}
|
||||
f.settled = append(f.settled, settled{job, state, note})
|
||||
return nil
|
||||
}
|
||||
|
||||
// otherRestore mirrors the executor's conflict error, which api only knows by
|
||||
// its method.
|
||||
type otherRestore struct{}
|
||||
|
||||
func (otherRestore) Error() string { return "another restore" }
|
||||
func (otherRestore) RestoreInProgress() bool { return true }
|
||||
|
||||
var _ RestoreSnapshotter = (*backupjob.Backuper)(nil)
|
||||
|
||||
func TestIsRestoreInProgressMatchesTheExecutor(t *testing.T) {
|
||||
if !isRestoreInProgress(restore.ErrOtherRestoreRunning) {
|
||||
t.Fatal("restore.ErrOtherRestoreRunning is not recognised")
|
||||
}
|
||||
if isRestoreInProgress(restore.ErrAlreadyExists) || isRestoreInProgress(errors.New("x")) {
|
||||
t.Fatal("an unrelated error reads as a restore in progress")
|
||||
}
|
||||
}
|
||||
|
||||
// With the chain wired, a restore starts with a safety snapshot of the world as
|
||||
// it is: the Backuper gets the restore to carry, the Restorer is not called
|
||||
// yet, and the 202 says so. safety_snapshot:false restores straight away.
|
||||
func TestRestoreStartsWithSafetySnapshot(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
mk := func() (*API, *fakeSnapshotter, *fakeRestorer) {
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
repo.backups = []fakeBackup{{view: BackupView{ID: "bk1", ServerName: "survival", FormerOwner: "owner1",
|
||||
Status: "present", Reason: "manual", CreatedAt: time.Unix(1_699_000_000, 0)}, ref: "ref-1"}}
|
||||
cl := newFakeCluster()
|
||||
cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped", DesiredState: string(v1alpha1.DesiredStopped)}
|
||||
a := newTestAPI(repo, cl)
|
||||
snap, restorer := &fakeSnapshotter{}, &fakeRestorer{}
|
||||
a.Backuper, a.Restorer, a.RestoreChains = snap, restorer, &fakeChains{}
|
||||
a.External = staticExternal{p: owner}
|
||||
return a, snap, restorer
|
||||
}
|
||||
const path = "/api/v1/servers/survival/restore-backup"
|
||||
decode := func(t *testing.T, body []byte) map[string]any {
|
||||
t.Helper()
|
||||
var m map[string]any
|
||||
if err := json.Unmarshal(body, &m); err != nil {
|
||||
t.Fatalf("body: %v", err)
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
t.Run("default -> snapshot first", func(t *testing.T) {
|
||||
a, snap, restorer := mk()
|
||||
w := do(a.ExternalHandler(), "POST", path, `{"backup_id":"bk1"}`, jsonHeader)
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if m := decode(t, w.Body.Bytes()); m["safety_snapshot"] != true || m["status"] != "restoring" || m["backup_id"] != "bk1" {
|
||||
t.Fatalf("response = %v", m)
|
||||
}
|
||||
if len(snap.chained) != 1 || snap.chained[0] != (RestoreChain{Server: "survival", BackupID: "bk1", BackupRef: "ref-1"}) || snap.gotFormerOwn != "owner1" {
|
||||
t.Fatalf("chained = %+v (owner %q)", snap.chained, snap.gotFormerOwn)
|
||||
}
|
||||
if restorer.calls != 0 {
|
||||
t.Fatal("the restore ran before its snapshot")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("safety_snapshot false -> restore now", func(t *testing.T) {
|
||||
a, snap, restorer := mk()
|
||||
w := do(a.ExternalHandler(), "POST", path, `{"safety_snapshot":false}`, jsonHeader)
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if m := decode(t, w.Body.Bytes()); m["safety_snapshot"] != false {
|
||||
t.Fatalf("response = %v", m)
|
||||
}
|
||||
if restorer.calls != 1 || len(snap.chained) != 0 {
|
||||
t.Fatalf("restorer calls %d, chained %d", restorer.calls, len(snap.chained))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("chain not wired -> restore now", func(t *testing.T) {
|
||||
a, snap, restorer := mk()
|
||||
a.RestoreChains = nil
|
||||
if w := do(a.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if restorer.calls != 1 || len(snap.chained) != 0 {
|
||||
t.Fatalf("restorer calls %d, chained %d", restorer.calls, len(snap.chained))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("another restore running -> 409 restore_in_progress", func(t *testing.T) {
|
||||
a, _, restorer := mk()
|
||||
restorer.err = otherRestore{}
|
||||
w := do(a.ExternalHandler(), "POST", path, `{"safety_snapshot":false}`, jsonHeader)
|
||||
if w.Code != http.StatusConflict || errCode(w.Body.Bytes()) != "restore_in_progress" {
|
||||
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestSettleRestoreChains(t *testing.T) {
|
||||
mk := func(chains ...RestoreChain) (*API, *fakeChains, *fakeRestorer, *fakeCluster) {
|
||||
cl := newFakeCluster()
|
||||
cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped", DesiredState: string(v1alpha1.DesiredStopped)}
|
||||
a := newTestAPI(newFakeRepo(), cl)
|
||||
fc, restorer := &fakeChains{pending: chains}, &fakeRestorer{}
|
||||
a.RestoreChains, a.Restorer = fc, restorer
|
||||
return a, fc, restorer, cl
|
||||
}
|
||||
chain := func(snapshot string) RestoreChain {
|
||||
return RestoreChain{Job: "backup-survival-1", Server: "survival", BackupID: "bk1", BackupRef: "ref-1", Snapshot: snapshot}
|
||||
}
|
||||
ctx := context.Background()
|
||||
|
||||
t.Run("snapshot succeeded -> restore started", func(t *testing.T) {
|
||||
a, fc, restorer, _ := mk(chain(ChainSnapshotSucceeded))
|
||||
if err := a.SettleRestoreChains(ctx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if restorer.calls != 1 || restorer.gotName != "survival" || restorer.gotRef != "ref-1" {
|
||||
t.Fatalf("restorer = %+v", restorer)
|
||||
}
|
||||
if len(fc.settled) != 1 || fc.settled[0] != (settled{"backup-survival-1", maintenance.ThenRestoreStarted, ""}) {
|
||||
t.Fatalf("settled = %+v", fc.settled)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("snapshot running -> left alone", func(t *testing.T) {
|
||||
a, fc, restorer, _ := mk(chain(ChainSnapshotRunning))
|
||||
if err := a.SettleRestoreChains(ctx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if restorer.calls != 0 || len(fc.settled) != 0 {
|
||||
t.Fatalf("restorer %d, settled %+v", restorer.calls, fc.settled)
|
||||
}
|
||||
})
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
setup func(*fakeRestorer, *fakeCluster)
|
||||
snap string
|
||||
reason string
|
||||
}{
|
||||
{"snapshot failed", func(*fakeRestorer, *fakeCluster) {}, ChainSnapshotFailed, ChainAbandonSnapshotFailed},
|
||||
{"server started meanwhile", func(_ *fakeRestorer, cl *fakeCluster) {
|
||||
cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning)
|
||||
}, ChainSnapshotSucceeded, ChainAbandonServerStarted},
|
||||
{"server gone", func(_ *fakeRestorer, cl *fakeCluster) { delete(cl.byName, "survival") }, ChainSnapshotSucceeded, ChainAbandonServerGone},
|
||||
{"another restore running", func(r *fakeRestorer, _ *fakeCluster) { r.err = otherRestore{} }, ChainSnapshotSucceeded, ChainAbandonRestoreBusy},
|
||||
} {
|
||||
t.Run(tc.name+" -> abandoned", func(t *testing.T) {
|
||||
a, fc, restorer, cl := mk(chain(tc.snap))
|
||||
tc.setup(restorer, cl)
|
||||
if err := a.SettleRestoreChains(ctx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(fc.settled) != 1 || fc.settled[0].state != maintenance.ThenRestoreAbandoned || fc.settled[0].note != tc.reason {
|
||||
t.Fatalf("settled = %+v", fc.settled)
|
||||
}
|
||||
if tc.snap == ChainSnapshotFailed && restorer.calls != 0 {
|
||||
t.Fatal("restored after a failed snapshot")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("restore error -> stays pending, retried", func(t *testing.T) {
|
||||
a, fc, restorer, _ := mk(chain(ChainSnapshotSucceeded))
|
||||
restorer.err = errors.New("apiserver hiccup")
|
||||
if err := a.SettleRestoreChains(ctx); err == nil {
|
||||
t.Fatal("the failure was swallowed")
|
||||
}
|
||||
if len(fc.settled) != 0 {
|
||||
t.Fatalf("settled = %+v", fc.settled)
|
||||
}
|
||||
restorer.err = nil
|
||||
if err := a.SettleRestoreChains(ctx); err != nil || len(fc.settled) != 1 || fc.settled[0].state != maintenance.ThenRestoreStarted {
|
||||
t.Fatalf("retry: err %v, settled %+v", err, fc.settled)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The cluster side: the chain the backupjob executor renders is found by the
|
||||
// label selector, its snapshot outcome read from the Job conditions, and
|
||||
// settling it releases the world volume.
|
||||
func TestK8sRestoreChains(t *testing.T) {
|
||||
scheme := runtime.NewScheme()
|
||||
if err := clientgoscheme.AddToScheme(scheme); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
chain, err := backupjob.BackupJob(backupjob.JobParams{
|
||||
Server: "survival", JobName: "backup-survival-aa", WorldPVC: "world-survival-0", BackupPVC: "felis-backups",
|
||||
Namespace: "minecraft", Image: "felis:1", ConfigSecret: "felis-config", ConfigMount: "/etc/felis",
|
||||
RestoreRef: "/backups/survival/a.tar.gz", RestoreBackupID: "bk-1",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
chain.Status.Conditions = []batchv1.JobCondition{{Type: batchv1.JobComplete, Status: corev1.ConditionTrue}}
|
||||
plain, err := backupjob.BackupJob(backupjob.JobParams{
|
||||
Server: "survival", JobName: "backup-survival-bb", WorldPVC: "world-survival-0", BackupPVC: "felis-backups",
|
||||
Namespace: "minecraft", Image: "felis:1", ConfigSecret: "felis-config", ConfigMount: "/etc/felis",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
plain.Status.Conditions = chain.Status.Conditions
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(chain, plain).WithStatusSubresource(&batchv1.Job{}).Build()
|
||||
k := NewK8sJobStatus(c, "minecraft")
|
||||
ctx := context.Background()
|
||||
|
||||
got, err := k.PendingRestoreChains(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := RestoreChain{Job: "backup-survival-aa", Server: "survival", BackupID: "bk-1",
|
||||
BackupRef: "/backups/survival/a.tar.gz", Snapshot: ChainSnapshotSucceeded}
|
||||
if len(got) != 1 || got[0] != want {
|
||||
t.Fatalf("pending = %+v, want [%+v]", got, want)
|
||||
}
|
||||
|
||||
var jobs batchv1.JobList
|
||||
if err := c.List(ctx, &jobs); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if kind, held := maintenance.Holder("survival", nil, jobs.Items, time.Now()); !held || kind != maintenance.KindRestore {
|
||||
t.Fatalf("pending chain: Holder = %q, %v", kind, held)
|
||||
}
|
||||
|
||||
if err := k.SettleRestoreChain(ctx, "backup-survival-aa", maintenance.ThenRestoreAbandoned, ChainAbandonServerStarted); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var settledJob batchv1.Job
|
||||
if err := c.Get(ctx, types.NamespacedName{Namespace: "minecraft", Name: "backup-survival-aa"}, &settledJob); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if settledJob.Labels[maintenance.LabelThenRestore] != maintenance.ThenRestoreAbandoned ||
|
||||
settledJob.Annotations[maintenance.AnnotationThenRestoreReason] != ChainAbandonServerStarted ||
|
||||
settledJob.Annotations[maintenance.AnnotationRestoreRef] == "" {
|
||||
t.Fatalf("settled job meta: labels %v annotations %v", settledJob.Labels, settledJob.Annotations)
|
||||
}
|
||||
if got, _ := k.PendingRestoreChains(ctx); len(got) != 0 {
|
||||
t.Fatalf("still pending after settling: %+v", got)
|
||||
}
|
||||
if err := c.List(ctx, &jobs); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if kind, held := maintenance.Holder("survival", nil, jobs.Items, time.Now()); held {
|
||||
t.Fatalf("settled chain still holds the volume as %q", kind)
|
||||
}
|
||||
|
||||
aj, ok := jobToAsyncJob(&settledJob)
|
||||
if !ok || aj.ThenRestore != maintenance.ThenRestoreAbandoned || aj.RestoreBackupID != "bk-1" || aj.State != "succeeded" ||
|
||||
aj.ThenRestoreReason != ChainAbandonServerStarted || aj.Message != chainAbandonText[ChainAbandonServerStarted] {
|
||||
t.Fatalf("projection = %+v", aj)
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,9 @@
|
||||
package api
|
||||
|
||||
import "context"
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
)
|
||||
|
||||
// Restorer starts a world restore from a stored backup (spec §466: former_owner
|
||||
// 3mo 内重新 claim → restore PVC). Restore only STARTS the work: recreating the
|
||||
@@ -12,15 +15,23 @@ import "context"
|
||||
// call returns once the restore is enqueued, so the handler answers 202
|
||||
// (restoring), never claiming the world is already back.
|
||||
//
|
||||
// It returns ErrNotFound if the server is unknown to the execution backend; any
|
||||
// other error is an internal failure (the handler maps it to 500).
|
||||
// It returns ErrNotFound if the server is unknown to the execution backend, and
|
||||
// an error whose RestoreInProgress method reports true when a restore of another
|
||||
// backup is still running on the world (409 restore_in_progress; see
|
||||
// isRestoreInProgress). Any other error is an internal failure (500).
|
||||
//
|
||||
// It is an interface so the handler is tested against a fake (api_test.go). The
|
||||
// production executor — a restore Job mirroring internal/build's jobspec + weak-SA
|
||||
// isolation — is integration-only and is a deliberate follow-up: until it is wired
|
||||
// the API.Restorer is nil and POST /servers/{name}/restore-backup reports 503, so
|
||||
// the restore authorization boundary is exercised without shipping a stub that
|
||||
// cannot run in the real cluster topology.
|
||||
// production executor is internal/restore's weak-SA restore Job; when cmd/felis
|
||||
// cannot configure it the API.Restorer is nil and POST
|
||||
// /servers/{name}/restore-backup reports 503.
|
||||
type Restorer interface {
|
||||
Restore(ctx context.Context, serverName, backupRef string) error
|
||||
}
|
||||
|
||||
// isRestoreInProgress reports whether err says another restore is still running
|
||||
// on the world. The executor cannot import this package, so its error is matched
|
||||
// by method rather than by value.
|
||||
func isRestoreInProgress(err error) bool {
|
||||
var busy interface{ RestoreInProgress() bool }
|
||||
return errors.As(err, &busy) && busy.RestoreInProgress()
|
||||
}
|
||||
@@ -187,6 +187,24 @@ func (b *Backuper) Backup(ctx context.Context, serverName, formerOwner string) e
|
||||
return nil
|
||||
}
|
||||
|
||||
// BackupThenRestore enqueues the safety snapshot in front of a restore: a backup
|
||||
// Job like Backup's, recorded as a pre_restore backup and labelled with the
|
||||
// restore to run once it succeeds (backupID, backupRef). felis-api creates that
|
||||
// restore Job when the snapshot finishes and gives it up if the snapshot fails,
|
||||
// so the world is never overwritten without a way back; the snapshot Job holds
|
||||
// the world volume as a restore until then (internal/maintenance).
|
||||
func (b *Backuper) BackupThenRestore(ctx context.Context, serverName, formerOwner, backupID, backupRef string) error {
|
||||
p := b.jobParams(serverName, formerOwner)
|
||||
p.RestoreRef, p.RestoreBackupID = backupRef, backupID
|
||||
if err := b.Jobs.CreateBackupJob(ctx, p); err != nil {
|
||||
if errors.Is(err, ErrAlreadyExists) {
|
||||
return nil // suffix collision — treat as enqueued
|
||||
}
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// jobNameSuffix is a short random hex tag that makes each backup Job name unique.
|
||||
// 32 bits is ample: collisions only matter within a single Job's TTL window across
|
||||
// a handful of manual backups.
|
||||
|
||||
@@ -41,3 +41,21 @@ func TestBackupMintsUniqueJobNamePerCall(t *testing.T) {
|
||||
t.Errorf("two backups reused the Job name %q — retry would silently no-op", jobs.got[0].JobName)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupThenRestoreChainsTheRestore(t *testing.T) {
|
||||
jobs := &captureJobs{}
|
||||
b := &Backuper{Jobs: jobs, Config: Config{Image: "img", BackupPVC: "pvc"}}
|
||||
if err := b.BackupThenRestore(context.Background(), "survival", "usr-1", "bk-1", "/backups/a.tar.gz"); err != nil {
|
||||
t.Fatalf("BackupThenRestore: %v", err)
|
||||
}
|
||||
if len(jobs.got) != 1 {
|
||||
t.Fatalf("created %d jobs, want 1", len(jobs.got))
|
||||
}
|
||||
p := jobs.got[0]
|
||||
if p.RestoreRef != "/backups/a.tar.gz" || p.RestoreBackupID != "bk-1" || p.FormerOwner != "usr-1" {
|
||||
t.Errorf("params = %+v", p)
|
||||
}
|
||||
if !strings.HasPrefix(p.JobName, BackupJobName("survival")+"-") {
|
||||
t.Errorf("JobName %q", p.JobName)
|
||||
}
|
||||
}
|
||||
@@ -20,6 +20,16 @@ const (
|
||||
managedByValue = "felis-backup"
|
||||
componentValue = "world-backup"
|
||||
|
||||
// The restore chain a safety snapshot carries (internal/maintenance keeps the
|
||||
// canonical copies; maintenance_test pins these against them).
|
||||
labelThenRestore = "felis.lolicon.best/then-restore"
|
||||
thenRestorePending = "pending"
|
||||
annotationRestoreRef = "felis.lolicon.best/restore-ref"
|
||||
annotationRestoreBackupID = "felis.lolicon.best/restore-backup-id"
|
||||
|
||||
// ReasonPreRestore is the world_backups reason of a safety snapshot.
|
||||
ReasonPreRestore = "pre_restore"
|
||||
|
||||
worldVolume = "world"
|
||||
backupVolume = "backup"
|
||||
configVolume = "config"
|
||||
@@ -55,6 +65,14 @@ type JobParams struct {
|
||||
RunAsGroup int64
|
||||
FSGroup int64
|
||||
|
||||
// RestoreRef / RestoreBackupID make this backup the safety snapshot in front
|
||||
// of a restore: the Job is labelled as a pending chain and names the backup
|
||||
// felis-api restores once the snapshot succeeds (internal/maintenance). The
|
||||
// snapshot is recorded as ReasonPreRestore, and its prune spares the backup
|
||||
// the restore will extract.
|
||||
RestoreRef string
|
||||
RestoreBackupID string
|
||||
|
||||
TTLAfterFinished time.Duration
|
||||
}
|
||||
|
||||
@@ -106,6 +124,9 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||
if p.ConfigSecret == "" {
|
||||
return nil, fmt.Errorf("backup: config secret name is required")
|
||||
}
|
||||
if p.RestoreRef != "" && p.RestoreBackupID == "" {
|
||||
return nil, fmt.Errorf("backup: a chained restore needs the backup id")
|
||||
}
|
||||
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -129,6 +150,9 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||
if p.FormerOwner != "" {
|
||||
args = append(args, "--former-owner", p.FormerOwner)
|
||||
}
|
||||
if p.RestoreRef != "" {
|
||||
args = append(args, "--reason", ReasonPreRestore, "--protect", p.RestoreBackupID)
|
||||
}
|
||||
|
||||
container := corev1.Container{
|
||||
Name: "backup",
|
||||
@@ -167,12 +191,22 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||
if name == "" {
|
||||
name = BackupJobName(p.Server)
|
||||
}
|
||||
meta := metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: p.Namespace,
|
||||
Labels: backupLabels(p),
|
||||
}
|
||||
if p.RestoreRef != "" {
|
||||
// On the Job only: felis-api settles the chain by patching this label, and
|
||||
// the pods never need it.
|
||||
meta.Labels[labelThenRestore] = thenRestorePending
|
||||
meta.Annotations = map[string]string{
|
||||
annotationRestoreRef: p.RestoreRef,
|
||||
annotationRestoreBackupID: p.RestoreBackupID,
|
||||
}
|
||||
}
|
||||
job := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: p.Namespace,
|
||||
Labels: backupLabels(p),
|
||||
},
|
||||
ObjectMeta: meta,
|
||||
Spec: batchv1.JobSpec{
|
||||
// One shot: a wedged archive must not loop. The TTL GCs the finished Job
|
||||
// so a later backup of the same server is not blocked forever by a stale
|
||||
|
||||
@@ -199,6 +199,44 @@ func TestBackupJobArgsCarryServerAndOwner(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A safety snapshot records itself as pre_restore, spares the backup the chained
|
||||
// restore extracts from its prune, and carries the chain on the Job (only there:
|
||||
// felis-api settles it by patching the Job's label).
|
||||
func TestBackupJobCarriesTheRestoreChain(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
p.RestoreRef, p.RestoreBackupID = "/backups/survival/a.tar.gz", "bk-1"
|
||||
job, err := BackupJob(p)
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
args := job.Spec.Template.Spec.Containers[0].Args
|
||||
if !argsContain(args, "--reason", ReasonPreRestore) || !argsContain(args, "--protect", "bk-1") {
|
||||
t.Errorf("args = %v, want --reason %s --protect bk-1", args, ReasonPreRestore)
|
||||
}
|
||||
if job.Labels[labelThenRestore] != thenRestorePending {
|
||||
t.Errorf("job labels = %v, want %s=%s", job.Labels, labelThenRestore, thenRestorePending)
|
||||
}
|
||||
if _, ok := job.Spec.Template.Labels[labelThenRestore]; ok {
|
||||
t.Errorf("pod template carries the chain label: %v", job.Spec.Template.Labels)
|
||||
}
|
||||
if job.Annotations[annotationRestoreRef] != p.RestoreRef || job.Annotations[annotationRestoreBackupID] != "bk-1" {
|
||||
t.Errorf("job annotations = %v", job.Annotations)
|
||||
}
|
||||
|
||||
plain, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob(plain): %v", err)
|
||||
}
|
||||
if _, ok := plain.Labels[labelThenRestore]; ok || len(plain.Annotations) != 0 {
|
||||
t.Errorf("a plain backup carries a chain: labels %v annotations %v", plain.Labels, plain.Annotations)
|
||||
}
|
||||
for _, a := range plain.Spec.Template.Spec.Containers[0].Args {
|
||||
if a == "--reason" || a == "--protect" {
|
||||
t.Errorf("a plain backup passes %s: %v", a, plain.Spec.Template.Spec.Containers[0].Args)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
@@ -208,6 +246,7 @@ func TestBackupJobRejectsMissingInputs(t *testing.T) {
|
||||
{"no world pvc", func(p *JobParams) { p.WorldPVC = "" }},
|
||||
{"no backup pvc", func(p *JobParams) { p.BackupPVC = "" }},
|
||||
{"no config secret", func(p *JobParams) { p.ConfigSecret = "" }},
|
||||
{"chain without backup id", func(p *JobParams) { p.RestoreRef = "/backups/a.tar.gz" }},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
|
||||
@@ -20,6 +20,12 @@
|
||||
// A lock older than Grace with no Job behind it is stale (felis-api died between
|
||||
// the two writes) and holds nothing.
|
||||
//
|
||||
// A restore that starts with a safety snapshot is two Jobs in a row: the backup
|
||||
// Job carries the restore to run after it (LabelThenRestore), and felis-api
|
||||
// creates the restore Job once the backup has succeeded. The backup Job keeps
|
||||
// holding the volume, as a restore, from its creation until felis-api has
|
||||
// settled what follows it, so nothing can wake the server between the two Jobs.
|
||||
//
|
||||
// File reads and listings are not holders. They mount the volume read-only for a
|
||||
// second or two, and a server starting beside one cannot hurt either side, so
|
||||
// nobody waits for them.
|
||||
@@ -50,6 +56,26 @@ const (
|
||||
// LabelFilesMode is the file-editor operation (list, read, write) a files Job
|
||||
// performs. Only write holds the volume.
|
||||
LabelFilesMode = "felis.lolicon.best/files-mode"
|
||||
|
||||
// LabelThenRestore marks a backup Job that is the safety snapshot in front of
|
||||
// a restore. Its value is the chain's state: ThenRestorePending until felis-api
|
||||
// settles it, then ThenRestoreStarted or ThenRestoreAbandoned. A label, so the
|
||||
// settling loop finds pending chains with a selector.
|
||||
LabelThenRestore = "felis.lolicon.best/then-restore"
|
||||
// AnnotationRestoreRef / AnnotationRestoreBackupID name the backup the chained
|
||||
// restore extracts: its archive ref and its world_backups id.
|
||||
AnnotationRestoreRef = "felis.lolicon.best/restore-ref"
|
||||
AnnotationRestoreBackupID = "felis.lolicon.best/restore-backup-id"
|
||||
// AnnotationThenRestoreReason is the code saying why a chain was abandoned
|
||||
// (internal/api defines the codes).
|
||||
AnnotationThenRestoreReason = "felis.lolicon.best/then-restore-reason"
|
||||
)
|
||||
|
||||
// States of LabelThenRestore.
|
||||
const (
|
||||
ThenRestorePending = "pending"
|
||||
ThenRestoreStarted = "started"
|
||||
ThenRestoreAbandoned = "abandoned"
|
||||
)
|
||||
|
||||
// Kinds of holder.
|
||||
@@ -82,6 +108,13 @@ func JobKind(j *batchv1.Job) (string, bool) {
|
||||
return "", false
|
||||
}
|
||||
|
||||
// RestorePending reports whether j is a safety-snapshot backup Job whose restore
|
||||
// felis-api has not yet started or abandoned. Such a Job holds the volume as a
|
||||
// restore whether or not it has finished.
|
||||
func RestorePending(j *batchv1.Job) bool {
|
||||
return j.Labels[LabelManagedBy] == "felis-backup" && j.Labels[LabelThenRestore] == ThenRestorePending
|
||||
}
|
||||
|
||||
// JobFinished reports whether a Job has reached a terminal condition. The
|
||||
// success/failure-target conditions count as terminal: the Job controller sets
|
||||
// them the moment the outcome is decided, before it finishes tearing the pods
|
||||
@@ -119,13 +152,19 @@ func parseLock(v string) (string, time.Time, bool) {
|
||||
}
|
||||
|
||||
// Holder reports what, if anything, holds the server's world volume at `now`:
|
||||
// the first unfinished maintenance Job among jobs, else a lock in annotations
|
||||
// younger than Grace. jobs may contain unrelated Jobs; only the server's own
|
||||
// holders count.
|
||||
// the first unfinished maintenance Job among jobs (or a safety snapshot whose
|
||||
// restore is still pending), else a lock in annotations younger than Grace. jobs
|
||||
// may contain unrelated Jobs; only the server's own holders count.
|
||||
func Holder(server string, annotations map[string]string, jobs []batchv1.Job, now time.Time) (string, bool) {
|
||||
for i := range jobs {
|
||||
j := &jobs[i]
|
||||
if j.Labels[LabelServer] != server || JobFinished(j) {
|
||||
if j.Labels[LabelServer] != server {
|
||||
continue
|
||||
}
|
||||
if RestorePending(j) {
|
||||
return KindRestore, true
|
||||
}
|
||||
if JobFinished(j) {
|
||||
continue
|
||||
}
|
||||
if kind, ok := JobKind(j); ok {
|
||||
|
||||
@@ -125,6 +125,52 @@ func TestHolderFromJobs(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A safety snapshot holds the volume as a restore from its creation until
|
||||
// felis-api settles the chain, finished or not, so nothing wakes the server
|
||||
// between the backup Job and the restore Job it is followed by.
|
||||
func TestHolderFromRestoreChain(t *testing.T) {
|
||||
j, err := backupjob.BackupJob(backupjob.JobParams{
|
||||
Server: "survival", WorldPVC: "world-survival-0", BackupPVC: "felis-backups",
|
||||
Namespace: "minecraft", Image: "felis:1", ConfigSecret: "felis-config", ConfigMount: "/etc/felis",
|
||||
RestoreRef: "/backups/survival/a.tar.gz", RestoreBackupID: "bk-1",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
if j.Annotations[AnnotationRestoreRef] != "/backups/survival/a.tar.gz" || j.Annotations[AnnotationRestoreBackupID] != "bk-1" {
|
||||
t.Fatalf("chain annotations = %v", j.Annotations)
|
||||
}
|
||||
if !RestorePending(j) {
|
||||
t.Fatalf("a fresh safety snapshot is not a pending chain: labels %v", j.Labels)
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
job batchv1.Job
|
||||
held bool
|
||||
kind string
|
||||
}{
|
||||
{"running", *j, true, KindRestore},
|
||||
{"succeeded, restore not started yet", finished(*j, batchv1.JobComplete), true, KindRestore},
|
||||
{"failed, not settled yet", finished(*j, batchv1.JobFailed), true, KindRestore},
|
||||
{"restore started", settled(finished(*j, batchv1.JobComplete), ThenRestoreStarted), false, ""},
|
||||
{"abandoned", settled(finished(*j, batchv1.JobFailed), ThenRestoreAbandoned), false, ""},
|
||||
{"abandoned while running", settled(*j, ThenRestoreAbandoned), true, KindBackup},
|
||||
} {
|
||||
kind, held := Holder("survival", nil, []batchv1.Job{tc.job}, now)
|
||||
if held != tc.held || kind != tc.kind {
|
||||
t.Errorf("%s: Holder = %q, %v; want %q, %v", tc.name, kind, held, tc.kind, tc.held)
|
||||
}
|
||||
}
|
||||
if plain := backupJob(t, "survival"); RestorePending(&plain) {
|
||||
t.Fatal("a plain backup is a pending chain")
|
||||
}
|
||||
}
|
||||
|
||||
func settled(j batchv1.Job, state string) batchv1.Job {
|
||||
j.Labels = map[string]string{LabelServer: j.Labels[LabelServer], LabelManagedBy: j.Labels[LabelManagedBy], LabelThenRestore: state}
|
||||
return j
|
||||
}
|
||||
|
||||
func TestHolderFromLock(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
|
||||
@@ -187,13 +187,24 @@ func TestManualBackupRationing(t *testing.T) {
|
||||
insert(sole, "inactive_15d", now.Add(-100*reaper.Day), false, 1000)
|
||||
insert(copied, "inactive_15d", now.Add(-50*reaper.Day), true, 100)
|
||||
|
||||
excess, err := st.ExcessManualBackups(ctx, name, 5)
|
||||
excess, err := st.ExcessBackups(ctx, name, "manual", 5, "")
|
||||
if err != nil {
|
||||
t.Fatalf("ExcessManualBackups: %v", err)
|
||||
t.Fatalf("ExcessBackups: %v", err)
|
||||
}
|
||||
if len(excess) != 2 || excess[0].ID != manual[0] || excess[1].ID != manual[1] {
|
||||
t.Fatalf("excess = %+v; want the two oldest manual backups, oldest first", excess)
|
||||
}
|
||||
// The backup a chained restore will extract is never pruned.
|
||||
excess, err = st.ExcessBackups(ctx, name, "manual", 5, manual[0])
|
||||
if err != nil {
|
||||
t.Fatalf("ExcessBackups(protect): %v", err)
|
||||
}
|
||||
if len(excess) != 1 || excess[0].ID != manual[1] {
|
||||
t.Fatalf("excess with %s protected = %+v; want only %s", manual[0], excess, manual[1])
|
||||
}
|
||||
if excess, err = st.ExcessBackups(ctx, name, "pre_restore", 0, ""); err != nil || len(excess) != 0 {
|
||||
t.Fatalf("pre_restore excess = %+v, %v; the manual ones are not its to prune", excess, err)
|
||||
}
|
||||
|
||||
all, err := st.EvictableBackups(ctx)
|
||||
if err != nil {
|
||||
|
||||
@@ -104,8 +104,10 @@ func APIMinecraftRole(p Params) *rbacv1.Role {
|
||||
// nothing in felis-api lists or deletes PVCs.
|
||||
rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"get"}),
|
||||
// list backs GET /servers/{name}/jobs — the async status outlet reads the
|
||||
// backup/restore Jobs back by the server label (read-only).
|
||||
rule([]string{groupBatch}, []string{"jobs"}, []string{"create", "get", "delete", "list"}),
|
||||
// backup/restore Jobs back by the server label — and finds the pending
|
||||
// restore chains; patch settles a chain by relabelling its safety-snapshot
|
||||
// Job (internal/api restorechain.go).
|
||||
rule([]string{groupBatch}, []string{"jobs"}, []string{"create", "get", "delete", "list", "patch"}),
|
||||
// Read-side console (spec §8 读=pods/log follow): list pods to find the
|
||||
// server's running pod, then read its log subresource. Two separate rules so
|
||||
// the verbs stay tight — list on pods, get on pods/log, and nothing else.
|
||||
|
||||
@@ -73,7 +73,7 @@ func TestAPIRole_CreatesJobsInBothNamespaces(t *testing.T) {
|
||||
if mc.Namespace != "minecraft" {
|
||||
t.Errorf("felis-api minecraft Role namespace = %q, want minecraft", mc.Namespace)
|
||||
}
|
||||
for _, v := range []string{"create", "get", "delete", "list"} {
|
||||
for _, v := range []string{"create", "get", "delete", "list", "patch"} {
|
||||
if !hasRule(mc, "batch", "jobs", v) {
|
||||
t.Errorf("felis-api (minecraft) must have batch/jobs:%s for the restore-Job lifecycle", v)
|
||||
}
|
||||
|
||||
@@ -113,13 +113,15 @@ func (s *PGStore) EvictableBackups(ctx context.Context) ([]StoredBackup, error)
|
||||
return s.queryBackups(ctx, q)
|
||||
}
|
||||
|
||||
// ExcessManualBackups lists server's present on-demand backups beyond the
|
||||
// newest keep, oldest first: what the backup Job removes after adding one.
|
||||
func (s *PGStore) ExcessManualBackups(ctx context.Context, server string, keep int) ([]StoredBackup, error) {
|
||||
// ExcessBackups lists server's present backups of one reason beyond the newest
|
||||
// keep, oldest first: what the backup Job removes after adding one. protect, when
|
||||
// set, is a backup id left out of the list whatever its age (the one a chained
|
||||
// restore is about to extract).
|
||||
func (s *PGStore) ExcessBackups(ctx context.Context, server, reason string, keep int, protect string) ([]StoredBackup, error) {
|
||||
const q = `SELECT id, server_name, backup_ref, size_bytes, reason FROM world_backups
|
||||
WHERE server_name = $1 AND status = 'present' AND reason = 'manual'
|
||||
ORDER BY created_at DESC OFFSET $2`
|
||||
out, err := s.queryBackups(ctx, q, server, keep)
|
||||
WHERE server_name = $1 AND status = 'present' AND reason = $2 AND id <> $4
|
||||
ORDER BY created_at DESC OFFSET $3`
|
||||
out, err := s.queryBackups(ctx, q, server, reason, keep, protect)
|
||||
for i, j := 0, len(out)-1; i < j; i, j = i+1, j-1 {
|
||||
out[i], out[j] = out[j], out[i]
|
||||
}
|
||||
|
||||
@@ -20,6 +20,11 @@ const (
|
||||
managedByValue = "felis-restore"
|
||||
componentValue = "world-restore"
|
||||
|
||||
// AnnotationBackupRef records which archive the Job extracts, so a second
|
||||
// restore that collides with it on the name can tell a duplicate of the same
|
||||
// request from a request for a different backup (K8sJobs.CreateRestoreJob).
|
||||
AnnotationBackupRef = "felis.lolicon.best/backup-ref"
|
||||
|
||||
worldVolume = "world"
|
||||
backupVolume = "backup"
|
||||
felisBinaryPath = "/usr/local/bin/felis"
|
||||
@@ -144,9 +149,10 @@ func RestoreJob(p JobParams) (*batchv1.Job, error) {
|
||||
|
||||
job := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: RestoreJobName(p.Server),
|
||||
Namespace: p.Namespace,
|
||||
Labels: restoreLabels(p),
|
||||
Name: RestoreJobName(p.Server),
|
||||
Namespace: p.Namespace,
|
||||
Labels: restoreLabels(p),
|
||||
Annotations: map[string]string{AnnotationBackupRef: p.BackupRef},
|
||||
},
|
||||
Spec: batchv1.JobSpec{
|
||||
// One shot: a bad archive must not loop. The TTL GCs the finished Job
|
||||
|
||||
@@ -224,6 +224,9 @@ func TestRestoreJobInvokesFelisRestoreWithParams(t *testing.T) {
|
||||
if !argPairPresent(c.Args, "--ref", p.BackupRef) {
|
||||
t.Errorf("args must carry --ref %q, got %v", p.BackupRef, c.Args)
|
||||
}
|
||||
if got := job.Annotations[AnnotationBackupRef]; got != p.BackupRef {
|
||||
t.Errorf("backup-ref annotation = %q, want %q", got, p.BackupRef)
|
||||
}
|
||||
if !argPairPresent(c.Args, "--archive-store", p.ArchiveStore) {
|
||||
t.Errorf("args must carry --archive-store %q, got %v", p.ArchiveStore, c.Args)
|
||||
}
|
||||
|
||||
@@ -6,7 +6,9 @@ import (
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
@@ -37,8 +39,10 @@ func NewK8sJobs(c client.Client) *K8sJobs {
|
||||
// of the same server collides on Create. The collision is answered by the state
|
||||
// of the Job already holding the name:
|
||||
//
|
||||
// - still running (or not yet started): ErrAlreadyExists, which the Restorer
|
||||
// treats as success — the idempotent coalesce.
|
||||
// - still running (or not yet started) on the same archive: ErrAlreadyExists,
|
||||
// which the Restorer treats as success — the idempotent coalesce.
|
||||
// - still running on another archive: ErrOtherRestoreRunning. A Job from
|
||||
// before the ref annotation cannot be compared and coalesces as before.
|
||||
// - finished (succeeded OR failed): the finished Job is deleted and replaced,
|
||||
// so the caller's retry enqueues for real. Without this, the deterministic
|
||||
// name plus the ten-minute TTL would swallow the retry — most importantly
|
||||
@@ -68,9 +72,14 @@ func (k *K8sJobs) CreateRestoreJob(ctx context.Context, p JobParams) error {
|
||||
return getErr
|
||||
}
|
||||
if !restoreJobFinished(&existing) {
|
||||
if ref, ok := existing.Annotations[AnnotationBackupRef]; ok && ref != p.BackupRef {
|
||||
return ErrOtherRestoreRunning
|
||||
}
|
||||
return ErrAlreadyExists
|
||||
}
|
||||
if deleteErr := k.c.Delete(ctx, &existing); deleteErr != nil && !apierrors.IsNotFound(deleteErr) {
|
||||
// Background propagation: a Job deleted with the API's default policy
|
||||
// orphans its pods, which then outlive it for good.
|
||||
if deleteErr := k.c.Delete(ctx, &existing, client.PropagationPolicy(metav1.DeletePropagationBackground)); deleteErr != nil && !apierrors.IsNotFound(deleteErr) {
|
||||
return deleteErr
|
||||
}
|
||||
// The API server keeps the object until its job-tracking finalizer has run,
|
||||
@@ -128,7 +137,21 @@ func (k *K8sJobs) recreate(ctx context.Context, job *batchv1.Job) error {
|
||||
// that is merely created-but-not-started (no active pods yet, no completions)
|
||||
// counts as in flight, not finished, so a duplicate enqueue during startup still
|
||||
// coalesces.
|
||||
//
|
||||
// A terminal condition wins over pods still shutting down: the world-volume lock
|
||||
// (internal/maintenance.JobFinished) already lets the next restore in at that
|
||||
// point, and answering it with the coalesce would be a 202 for a restore that
|
||||
// never runs.
|
||||
func restoreJobFinished(job *batchv1.Job) bool {
|
||||
for _, c := range job.Status.Conditions {
|
||||
if c.Status != corev1.ConditionTrue {
|
||||
continue
|
||||
}
|
||||
switch c.Type {
|
||||
case batchv1.JobComplete, batchv1.JobFailed, batchv1.JobSuccessCriteriaMet, batchv1.JobFailureTarget:
|
||||
return true
|
||||
}
|
||||
}
|
||||
if job.Status.Active > 0 {
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -103,6 +103,12 @@ func TestRestoreJobFinished(t *testing.T) {
|
||||
Conditions: []batchv1.JobCondition{{Type: batchv1.JobFailed, Status: "True"}},
|
||||
}}, true},
|
||||
{"failed-and-some-active", batchv1.Job{Status: batchv1.JobStatus{Active: 1, Failed: 1}}, false},
|
||||
// The failure is decided while the pod is still being torn down: the
|
||||
// world-volume lock already admits the next restore, so this must too.
|
||||
{"failure-target-pod-terminating", batchv1.Job{Status: batchv1.JobStatus{
|
||||
Active: 1,
|
||||
Conditions: []batchv1.JobCondition{{Type: batchv1.JobFailureTarget, Status: "True"}},
|
||||
}}, true},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := restoreJobFinished(&tc.job); got != tc.want {
|
||||
@@ -110,3 +116,33 @@ func TestRestoreJobFinished(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A running restore of a different archive refuses the request instead of
|
||||
// absorbing it: the caller would otherwise get a 202 naming the backup it chose
|
||||
// while another one is extracted.
|
||||
func TestCreateRestoreJobRefusesAnotherArchiveInFlight(t *testing.T) {
|
||||
inFlight := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "restore-survival", Namespace: "minecraft",
|
||||
Annotations: map[string]string{AnnotationBackupRef: "/backups/other.tar.gz"},
|
||||
},
|
||||
Status: batchv1.JobStatus{Active: 1},
|
||||
}
|
||||
c := fake.NewClientBuilder().WithScheme(testScheme(t)).WithObjects(inFlight).Build()
|
||||
|
||||
err := NewK8sJobs(c).CreateRestoreJob(context.Background(), testParams())
|
||||
if !errors.Is(err, ErrOtherRestoreRunning) {
|
||||
t.Fatalf("CreateRestoreJob = %v, want ErrOtherRestoreRunning", err)
|
||||
}
|
||||
if err := (&Restorer{Jobs: NewK8sJobs(c), Config: Config{Image: "felis:test", BackupPVC: "felis-backups"}}).
|
||||
Restore(context.Background(), "survival", "/backups/x.tar.gz"); !errors.Is(err, ErrOtherRestoreRunning) {
|
||||
t.Fatalf("Restore = %v, want ErrOtherRestoreRunning", err)
|
||||
}
|
||||
|
||||
// The same archive still coalesces.
|
||||
p := testParams()
|
||||
p.BackupRef = "/backups/other.tar.gz"
|
||||
if err := NewK8sJobs(c).CreateRestoreJob(context.Background(), p); !errors.Is(err, ErrAlreadyExists) {
|
||||
t.Fatalf("same archive: CreateRestoreJob = %v, want ErrAlreadyExists", err)
|
||||
}
|
||||
}
|
||||
+30
-10
@@ -39,6 +39,22 @@ import (
|
||||
// as success — see Restore.
|
||||
var ErrAlreadyExists = errors.New("restore: job already exists")
|
||||
|
||||
// ErrOtherRestoreRunning is returned when a restore of a DIFFERENT backup is
|
||||
// still running on the server. The caller's request did not take effect; the
|
||||
// API answers 409 restore_in_progress rather than a 202 that names the backup
|
||||
// it asked for.
|
||||
var ErrOtherRestoreRunning error = otherRestoreRunning{}
|
||||
|
||||
type otherRestoreRunning struct{}
|
||||
|
||||
func (otherRestoreRunning) Error() string {
|
||||
return "restore: a restore of another backup is still running"
|
||||
}
|
||||
|
||||
// RestoreInProgress marks the error for internal/api, which cannot import this
|
||||
// package and recognises it by the method.
|
||||
func (otherRestoreRunning) RestoreInProgress() bool { return true }
|
||||
|
||||
// Jobs is the cluster-side restore lifecycle the Restorer depends on. It is an
|
||||
// interface so the orchestration is tested against a fake; the controller-runtime
|
||||
// implementation (K8sJobs) is integration-tested only — it requires a live
|
||||
@@ -47,7 +63,9 @@ var ErrAlreadyExists = errors.New("restore: job already exists")
|
||||
// whole contract, exactly matching the asynchronous 202 the handler answers.
|
||||
type Jobs interface {
|
||||
// CreateRestoreJob renders and applies the restore Job for p. It returns
|
||||
// ErrAlreadyExists if a Job of the same (deterministic) name already exists.
|
||||
// ErrAlreadyExists if an unfinished Job of the same (deterministic) name is
|
||||
// already restoring p.BackupRef, and ErrOtherRestoreRunning if it is
|
||||
// restoring another archive.
|
||||
CreateRestoreJob(ctx context.Context, p JobParams) error
|
||||
}
|
||||
|
||||
@@ -162,16 +180,18 @@ type Restorer struct {
|
||||
// serverName's world PVC. It returns once the Job is created — the extraction
|
||||
// runs in the Pod — so the handler's 202 ("restoring") is honest.
|
||||
//
|
||||
// It is idempotent: a duplicate enqueue while a restore Job for this server is
|
||||
// still running is treated as success rather than surfaced as an error.
|
||||
// It is idempotent: a duplicate enqueue of the same archive while a restore Job
|
||||
// for this server is still running is treated as success rather than surfaced
|
||||
// as an error.
|
||||
//
|
||||
// The coalescing key is the Job name (RestoreJobName), which depends only on the
|
||||
// server, NOT on backupRef — so a second request that arrives while one is in
|
||||
// flight is absorbed regardless of the ref it carries, and if the two refs
|
||||
// differ the second is silently dropped (the in-flight restore wins). That is
|
||||
// acceptable here: restore runs only for a Stopped server (handler gate ⑥) and
|
||||
// the handler always passes the latest backup, which for a stopped server does
|
||||
// not change, so concurrent requests carry the same ref in practice.
|
||||
// The Job name (RestoreJobName) depends only on the server, and the handler lets
|
||||
// the caller pick any of the server's backups, so a collision can carry a
|
||||
// different ref. The world-volume lock refuses a second restore while the first
|
||||
// Job runs, which makes this rare (it takes the lock seeing the Job finished
|
||||
// while its pods are still going), but when it happens the running Job's ref
|
||||
// annotation decides: the same archive coalesces, another archive is
|
||||
// ErrOtherRestoreRunning, so no caller is told "restoring" for a backup that is
|
||||
// not the one being extracted.
|
||||
//
|
||||
// A FINISHED Job — succeeded or failed — does not absorb the next request: its
|
||||
// deterministic name is replaced so the retry enqueues for real (see
|
||||
|
||||
@@ -5,9 +5,10 @@
|
||||
"not_yours_title": "No permission to access backups",
|
||||
"not_yours_body": "Only the owner or an admin can view and restore this server's backups.",
|
||||
"latest_title": "Latest backup",
|
||||
"history_note": "Backups can be restored until they expire, then they are cleaned up automatically. Each server keeps only its most recent manual backups; older manual backups are removed once a new one finishes.",
|
||||
"history_note": "Backups can be restored until they expire, then they are cleaned up automatically. Each server keeps only its most recent manual backups (older ones are removed once a new one finishes) and its last 3 pre-restore snapshots.",
|
||||
"reason_inactive": "Idle archive",
|
||||
"reason_manual": "Manual backup",
|
||||
"reason_pre_restore": "Pre-restore snapshot",
|
||||
"reason_label": "Reason: {{reason}}",
|
||||
"backup_now": "Back up now",
|
||||
"backup_in_progress": "Starting backup…",
|
||||
@@ -22,16 +23,34 @@
|
||||
"jobs_empty": "No recent backup or restore operations.",
|
||||
"job_backup": "Backup",
|
||||
"job_restore": "Restore",
|
||||
"job_pre_restore": "Pre-restore snapshot",
|
||||
"job_running": "Running",
|
||||
"job_succeeded": "Succeeded",
|
||||
"job_failed": "Failed",
|
||||
"chain_pending": "Restores the chosen backup once done",
|
||||
"chain_starting": "Snapshot done, starting the restore…",
|
||||
"chain_started": "Restore started",
|
||||
"chain_abandoned": "Not restored; the world was left as it was",
|
||||
"chain_abandoned_because": "Not restored: {{reason}}",
|
||||
"chain_abandoned_snapshot_failed": "Snapshot failed, so the world was left as it was",
|
||||
"chain_abandoned_not_configured": "Restore is not configured; nothing was restored",
|
||||
"chain_abandoned_server_gone": "The server no longer exists; nothing was restored",
|
||||
"chain_abandoned_server_started": "The server was started before the restore could run",
|
||||
"chain_abandoned_restore_busy": "Another restore is running on this world",
|
||||
"restore_btn": "Restore this backup",
|
||||
"restore_btn_short": "Restore",
|
||||
"restore_confirm": "Restoring a backup will stop the server (all online players will be disconnected) and completely overwrite the current world with this backup. This operation is irreversible. Do you want to continue?",
|
||||
"restore_confirm": "Restoring a backup will stop the server (all online players will be disconnected) and completely overwrite the current world with this backup. Without a safety snapshot this cannot be undone. Do you want to continue?",
|
||||
"restore_confirm_safe": "Restoring a backup will stop the server (all online players will be disconnected), take a safety snapshot of the current world, and overwrite the world with this backup once the snapshot succeeds. If you picked the wrong backup, restore the pre-restore snapshot to get it back. Continue?",
|
||||
"safety_snapshot_label": "Take a safety snapshot of the current world first (recommended)",
|
||||
"safety_snapshot_hint": "The restore starts once the snapshot succeeds; if it fails, nothing is restored and the world stays as it is.",
|
||||
"safety_snapshot_off_hint": "No snapshot: the current world is overwritten right away and cannot be recovered.",
|
||||
"restore_confirm_yes": "Confirm restore",
|
||||
"restore_confirm_yes_unsafe": "Overwrite and restore",
|
||||
"cancel": "Cancel",
|
||||
"restore_started": "Restore started — the world is being rebuilt from this backup. Wake the server once it finishes to see the restored world.",
|
||||
"restore_snapshot_started": "Taking a safety snapshot of the current world — the chosen backup is restored automatically once it finishes; follow it under recent operations below. Wake the server once the restore finishes.",
|
||||
"restore_started_short": "Restore started",
|
||||
"restore_snapshot_started_short": "Snapshot, then restore",
|
||||
"stopping": "Stopping the server…",
|
||||
"restoring": "Restoring…",
|
||||
"stop_timeout": "The server did not stop in time. Close this window and try again shortly.",
|
||||
|
||||
@@ -5,9 +5,10 @@
|
||||
"not_yours_title": "无权访问备份",
|
||||
"not_yours_body": "只有所有者或管理员才能查看并恢复该服务器的备份。",
|
||||
"latest_title": "最新备份",
|
||||
"history_note": "历史备份在过期前均可用于恢复,到期后自动清理。手动备份每台服务器只保留最近几份,新备份完成后会自动删除更早的手动备份。",
|
||||
"history_note": "历史备份在过期前均可用于恢复,到期后自动清理。手动备份每台服务器只保留最近几份,新备份完成后会自动删除更早的手动备份;恢复前快照保留最近 3 份。",
|
||||
"reason_inactive": "闲置自动回收",
|
||||
"reason_manual": "手动备份",
|
||||
"reason_pre_restore": "恢复前快照",
|
||||
"reason_label": "原因:{{reason}}",
|
||||
"backup_now": "立即备份",
|
||||
"backup_in_progress": "正在启动备份……",
|
||||
@@ -22,16 +23,34 @@
|
||||
"jobs_empty": "暂无备份或恢复操作记录。",
|
||||
"job_backup": "备份",
|
||||
"job_restore": "恢复",
|
||||
"job_pre_restore": "恢复前快照",
|
||||
"job_running": "进行中",
|
||||
"job_succeeded": "成功",
|
||||
"job_failed": "失败",
|
||||
"chain_pending": "完成后自动恢复所选备份",
|
||||
"chain_starting": "快照完成,正在开始恢复……",
|
||||
"chain_started": "已开始恢复",
|
||||
"chain_abandoned": "未恢复,世界保持原样",
|
||||
"chain_abandoned_because": "未恢复:{{reason}}",
|
||||
"chain_abandoned_snapshot_failed": "快照失败,未恢复,世界保持原样",
|
||||
"chain_abandoned_not_configured": "恢复功能未配置,未恢复",
|
||||
"chain_abandoned_server_gone": "服务器已不存在,未恢复",
|
||||
"chain_abandoned_server_started": "恢复开始前服务器被启动,未恢复",
|
||||
"chain_abandoned_restore_busy": "该世界正在进行另一次恢复,未恢复",
|
||||
"restore_btn": "恢复此备份",
|
||||
"restore_btn_short": "恢复",
|
||||
"restore_confirm": "恢复备份将停止服务器(所有在线玩家将被断开),并用该备份完全覆盖当前世界的全部数据。此操作不可逆,请确认是否继续?",
|
||||
"restore_confirm": "恢复备份将停止服务器(所有在线玩家将被断开),并用该备份完全覆盖当前世界的全部数据。未勾选安全快照时此操作不可逆,请确认是否继续?",
|
||||
"restore_confirm_safe": "恢复备份将停止服务器(所有在线玩家将被断开),先为当前世界做一份安全快照,快照成功后再用该备份覆盖世界。选错了备份,可以用这份“恢复前快照”恢复回来。确认继续?",
|
||||
"safety_snapshot_label": "先为当前世界做安全快照(推荐)",
|
||||
"safety_snapshot_hint": "快照成功后自动开始恢复;快照失败则不恢复,世界保持原样。",
|
||||
"safety_snapshot_off_hint": "不做快照,立即覆盖当前世界,之后无法找回。",
|
||||
"restore_confirm_yes": "确认恢复",
|
||||
"restore_confirm_yes_unsafe": "直接覆盖并恢复",
|
||||
"cancel": "取消",
|
||||
"restore_started": "已开始恢复备份——正在用此备份重建世界。完成后启动服务器即可看到恢复后的世界。",
|
||||
"restore_snapshot_started": "已开始为当前世界做安全快照——快照完成后会自动用所选备份恢复,进度见下方“最近操作”。恢复完成后启动服务器即可看到恢复后的世界。",
|
||||
"restore_started_short": "已启动恢复",
|
||||
"restore_snapshot_started_short": "快照中,随后恢复",
|
||||
"stopping": "正在停止服务器……",
|
||||
"restoring": "正在恢复备份……",
|
||||
"stop_timeout": "服务器停止超时。请关闭此窗口,稍后重试。",
|
||||
|
||||
@@ -634,6 +634,28 @@ describe("image whitelist and builds wire shapes", () => {
|
||||
expect((opts as RequestInit).body).toBeUndefined();
|
||||
});
|
||||
|
||||
it("restoreBackup sends backup_id, and safety_snapshot only when turned off", async () => {
|
||||
const fetchSpy = fakeFetch(
|
||||
{ name: "survival", status: "restoring", backup_id: "bk1", safety_snapshot: true },
|
||||
{ status: 202 },
|
||||
);
|
||||
vi.stubGlobal("fetch", fetchSpy);
|
||||
const res = await api.restoreBackup("survival", "bk1");
|
||||
expect(res.safety_snapshot).toBe(true);
|
||||
const calls = (fetchSpy as unknown as ReturnType<typeof vi.fn>).mock.calls;
|
||||
expect(String(calls[0][0])).toBe("/servers/survival/restore-backup");
|
||||
expect(JSON.parse((calls[0][1] as RequestInit).body as string)).toEqual({ backup_id: "bk1" });
|
||||
|
||||
await api.restoreBackup("survival", "bk1", false);
|
||||
expect(JSON.parse((calls[1][1] as RequestInit).body as string)).toEqual({
|
||||
backup_id: "bk1",
|
||||
safety_snapshot: false,
|
||||
});
|
||||
|
||||
await api.restoreBackup("survival");
|
||||
expect((calls[2][1] as RequestInit).body).toBeUndefined();
|
||||
});
|
||||
|
||||
it("serverJobs GETs /servers/{name}/jobs and unwraps the jobs array", async () => {
|
||||
const job = {
|
||||
name: "felis-backup-survival-123",
|
||||
|
||||
+13
-6
@@ -357,14 +357,21 @@ export const api = {
|
||||
// Preconditions are enforced server-side and surfaced as codes: owner-or-admin +
|
||||
// former-owner match (403), a present backup must exist (404 no_backup), and the
|
||||
// server MUST be fully stopped (409 not_stopped) since the restore writes into
|
||||
// the live world volume. The reply is 202 {name, status:"restoring", backup_id} —
|
||||
// success means the restore Job was enqueued, not that the world is back yet.
|
||||
restoreBackup: (name: string, backupId?: string) =>
|
||||
request<{ name: string; status: string; backup_id: string }>(
|
||||
// the live world volume. By default the backend first backs up the world as it
|
||||
// is (a "pre_restore" backup) and starts the restore only once that succeeded;
|
||||
// safetySnapshot=false skips it. The reply is 202 {name, status:"restoring",
|
||||
// backup_id, safety_snapshot} — success means the work was enqueued, not that
|
||||
// the world is back yet; serverJobs shows the snapshot and the restore.
|
||||
restoreBackup: (name: string, backupId?: string, safetySnapshot = true) => {
|
||||
const body: { backup_id?: string; safety_snapshot?: boolean } = {};
|
||||
if (backupId) body.backup_id = backupId;
|
||||
if (!safetySnapshot) body.safety_snapshot = false;
|
||||
return request<{ name: string; status: string; backup_id: string; safety_snapshot?: boolean }>(
|
||||
"POST",
|
||||
`/servers/${name}/restore-backup`,
|
||||
backupId ? { backup_id: backupId } : undefined,
|
||||
),
|
||||
Object.keys(body).length ? body : undefined,
|
||||
);
|
||||
},
|
||||
|
||||
// backupNow enqueues a manual backup (spec §7 POST backup). Preconditions are
|
||||
// enforced server-side and surfaced as codes: owner-or-admin (403) and the
|
||||
|
||||
@@ -168,6 +168,12 @@ export interface ServerJob {
|
||||
message?: string;
|
||||
started_at?: string;
|
||||
finished_at?: string;
|
||||
/** Set on a restore's safety snapshot (a backup job): "pending" until the
|
||||
* restore behind it starts ("started") or is given up ("abandoned", with a
|
||||
* code in then_restore_reason and its English in message). */
|
||||
then_restore?: string;
|
||||
then_restore_reason?: string;
|
||||
restore_backup_id?: string;
|
||||
}
|
||||
|
||||
/** WhitelistImage is one row of GET /images (the create-form dropdown source). */
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { useEffect, useState } from "react";
|
||||
import { useEffect, useRef, useState } from "react";
|
||||
import { useParams } from "react-router-dom";
|
||||
import {
|
||||
Archive,
|
||||
@@ -7,6 +7,7 @@ import {
|
||||
HardDrive,
|
||||
Loader2,
|
||||
RotateCcw,
|
||||
ShieldCheck,
|
||||
UserMinus,
|
||||
XCircle,
|
||||
} from "lucide-react";
|
||||
@@ -33,7 +34,7 @@ import { useTier } from "@/lib/tier";
|
||||
import { canManage, ownershipPending } from "@/lib/ownership";
|
||||
import { formatBytes, formatRelative, formatAbsolute, isExpired } from "@/lib/format";
|
||||
import { cn } from "@/lib/utils";
|
||||
import type { BackupView } from "@/lib/types";
|
||||
import type { BackupView, ServerJob } from "@/lib/types";
|
||||
|
||||
/** LatestBackupCard renders the most-recent backup as the restore card — the one a
|
||||
* restore actually recovers (the list is created_at-descending and the backend's
|
||||
@@ -67,6 +68,8 @@ function BackupRow({
|
||||
? t("reason_inactive")
|
||||
: b.reason === "manual"
|
||||
? t("reason_manual")
|
||||
: b.reason === "pre_restore"
|
||||
? t("reason_pre_restore")
|
||||
: t("reason_label", { reason: b.reason });
|
||||
|
||||
return (
|
||||
@@ -169,13 +172,53 @@ function JobStateBadge({ state }: { state: string }) {
|
||||
return <span className="text-[10px] font-medium text-muted-foreground">{state}</span>;
|
||||
}
|
||||
|
||||
/** ChainNote says what became of the restore behind a safety snapshot (a backup
|
||||
* job carrying then_restore). A snapshot that failed already reads as a failed
|
||||
* job with its error, so this only covers the chain itself. */
|
||||
function ChainNote({ job }: { job: ServerJob }) {
|
||||
const { t } = useTranslation("backups");
|
||||
if (job.state === "failed") return null;
|
||||
if (job.then_restore === "pending") {
|
||||
return (
|
||||
<span className="text-xs text-muted-foreground">
|
||||
{job.state === "running" ? t("chain_pending") : t("chain_starting")}
|
||||
</span>
|
||||
);
|
||||
}
|
||||
if (job.then_restore === "started") {
|
||||
return <span className="text-xs text-muted-foreground">{t("chain_started")}</span>;
|
||||
}
|
||||
if (job.then_restore === "abandoned") {
|
||||
// Known codes get their own wording; one this panel predates falls back to
|
||||
// the backend's English.
|
||||
const known = ["snapshot_failed", "not_configured", "server_gone", "server_started", "restore_busy"];
|
||||
const reason = job.then_restore_reason;
|
||||
const text =
|
||||
reason && known.includes(reason)
|
||||
? t(`chain_abandoned_${reason}`)
|
||||
: job.message
|
||||
? t("chain_abandoned_because", { reason: job.message })
|
||||
: t("chain_abandoned");
|
||||
return (
|
||||
<span
|
||||
className="max-w-[22rem] truncate text-xs text-amber-600 dark:text-amber-500"
|
||||
title={text}
|
||||
>
|
||||
{text}
|
||||
</span>
|
||||
);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** RestoreControls is the restore ACTION, living only on the latest backup card.
|
||||
* Restore is the panel's one irreversible operation, so it hides behind a single
|
||||
* button that opens a confirm dialog. The dialog states the full cost up front — the
|
||||
* server is stopped (online players drop) and the current world is overwritten by
|
||||
* THIS exact backup (named by relative + absolute time), and it can't be undone — and
|
||||
* then, on confirm, runs the whole chain itself: stop → wait for Stopped → restore.
|
||||
/** RestoreControls is the restore ACTION on each unexpired backup row. Restore
|
||||
* overwrites the world, so it hides behind a single button that opens a confirm
|
||||
* dialog. The dialog states the full cost up front — the server is stopped (online
|
||||
* players drop) and the current world is overwritten by THIS exact backup — and
|
||||
* offers a safety snapshot, on by default: the backend first backs up the world as
|
||||
* it is and restores only once that succeeded, which makes a wrong pick undoable.
|
||||
* Turning it off brings back the irreversible wording. On confirm it runs the whole
|
||||
* chain itself: stop → wait for Stopped → restore.
|
||||
* The backend refuses a restore unless the world volume is free (409 not_stopped), so
|
||||
* stopping here means the user never has to detour to the console and come back. There
|
||||
* is deliberately no type-the-name step: the friction that matters is owning the
|
||||
@@ -217,8 +260,10 @@ function RestoreControls({
|
||||
// drives which progress label shows once the chain reaches a concrete stage.
|
||||
const [submitting, setSubmitting] = useState(false);
|
||||
const [step, setStep] = useState<"stopping" | "restoring" | null>(null);
|
||||
const [done, setDone] = useState(false); // restore enqueued — terminal
|
||||
// restore enqueued — terminal; "snapshot" when the backend backs the world up first
|
||||
const [done, setDone] = useState<"snapshot" | "direct" | null>(null);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [safety, setSafety] = useState(true);
|
||||
|
||||
if (isExpired(backup.expires_at, now)) {
|
||||
if (layout === "row") return null;
|
||||
@@ -234,14 +279,14 @@ function RestoreControls({
|
||||
return (
|
||||
<div className="flex items-center gap-1.5 text-xs text-emerald-500 font-medium">
|
||||
<CheckCircle2 className="h-3.5 w-3.5 shrink-0" />
|
||||
<span>{t("restore_started_short")}</span>
|
||||
<span>{t(done === "snapshot" ? "restore_snapshot_started_short" : "restore_started_short")}</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<div className="mt-4 flex items-start gap-2 border-t border-primary/20 pt-4 text-sm text-emerald-500">
|
||||
<CheckCircle2 className="mt-0.5 h-4 w-4 shrink-0" />
|
||||
<span>{t("restore_started")}</span>
|
||||
<span>{t(done === "snapshot" ? "restore_snapshot_started" : "restore_started")}</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -273,8 +318,8 @@ function RestoreControls({
|
||||
}
|
||||
}
|
||||
setStep("restoring");
|
||||
await api.restoreBackup(serverName, backup.id);
|
||||
setDone(true);
|
||||
const res = await api.restoreBackup(serverName, backup.id, safety);
|
||||
setDone(res.safety_snapshot ? "snapshot" : "direct");
|
||||
setOpen(false);
|
||||
} catch (e) {
|
||||
setError(humanizeError(e));
|
||||
@@ -298,13 +343,38 @@ function RestoreControls({
|
||||
<DialogHeader>
|
||||
<DialogTitle>{t("restore_btn")}</DialogTitle>
|
||||
<DialogDescription>
|
||||
{t("restore_confirm", {
|
||||
{t(safety ? "restore_confirm_safe" : "restore_confirm", {
|
||||
relative: formatRelative(backup.created_at, now, locale),
|
||||
absolute: formatAbsolute(backup.created_at, locale),
|
||||
})}
|
||||
</DialogDescription>
|
||||
</DialogHeader>
|
||||
|
||||
<label
|
||||
className={cn(
|
||||
"flex cursor-pointer items-start gap-3 rounded-md border p-3 text-sm transition-colors",
|
||||
safety ? "border-primary/30 bg-primary/5" : "border-destructive/30 bg-destructive/5",
|
||||
submitting && "cursor-not-allowed opacity-60",
|
||||
)}
|
||||
>
|
||||
<input
|
||||
type="checkbox"
|
||||
className="mt-0.5 h-4 w-4 shrink-0 cursor-pointer accent-primary disabled:cursor-not-allowed"
|
||||
checked={safety}
|
||||
disabled={submitting}
|
||||
onChange={(e) => setSafety(e.target.checked)}
|
||||
/>
|
||||
<span className="flex flex-col gap-1">
|
||||
<span className="flex items-center gap-1.5 font-medium text-foreground">
|
||||
<ShieldCheck className="h-4 w-4 shrink-0 text-primary" />
|
||||
{t("safety_snapshot_label")}
|
||||
</span>
|
||||
<span className="text-xs text-muted-foreground">
|
||||
{t(safety ? "safety_snapshot_hint" : "safety_snapshot_off_hint")}
|
||||
</span>
|
||||
</span>
|
||||
</label>
|
||||
|
||||
{step && (
|
||||
<div className="flex items-center gap-2 text-sm text-muted-foreground">
|
||||
<Loader2 className="h-4 w-4 animate-spin" />
|
||||
@@ -318,7 +388,7 @@ function RestoreControls({
|
||||
onConfirm={confirmRestore}
|
||||
loading={submitting}
|
||||
cancelLabel={t("cancel")}
|
||||
confirmLabel={t("restore_confirm_yes")}
|
||||
confirmLabel={t(safety ? "restore_confirm_yes" : "restore_confirm_yes_unsafe")}
|
||||
/>
|
||||
</DialogContent>
|
||||
);
|
||||
@@ -382,17 +452,30 @@ export function ServerBackups() {
|
||||
|
||||
// The async world-operation history (the backup/restore Jobs behind every 202).
|
||||
// Read only once the viewer is resolved as owner-or-admin (the route 403s
|
||||
// otherwise); while anything is still running it re-reads on an interval so the
|
||||
// enqueue converges to succeeded/failed here instead of only in kubectl.
|
||||
// otherwise); while anything is still running — or a safety snapshot still has
|
||||
// its restore to start — it re-reads on an interval so the enqueue converges to
|
||||
// succeeded/failed here instead of only in kubectl.
|
||||
const jobsQ = useAsync(
|
||||
() => (owned ? api.serverJobs(name) : Promise.resolve([])),
|
||||
[name, owned],
|
||||
);
|
||||
useEffect(() => {
|
||||
if (!(jobsQ.data ?? []).some((j) => j.state === "running")) return;
|
||||
if (!(jobsQ.data ?? []).some((j) => j.state === "running" || j.then_restore === "pending")) return;
|
||||
const id = setInterval(jobsQ.reload, 5000);
|
||||
return () => clearInterval(id);
|
||||
}, [jobsQ.data, jobsQ.reload]);
|
||||
// A backup job that stops running has just added (or failed to add) a row, so
|
||||
// the list is re-read then rather than only on the next visit.
|
||||
const runningBackups = useRef<Set<string>>(new Set());
|
||||
const reloadBackups = backupsQ.reload;
|
||||
useEffect(() => {
|
||||
const now = new Set(
|
||||
(jobsQ.data ?? []).filter((j) => j.kind === "backup" && j.state === "running").map((j) => j.name),
|
||||
);
|
||||
const finished = [...runningBackups.current].some((n) => !now.has(n));
|
||||
runningBackups.current = now;
|
||||
if (finished) reloadBackups();
|
||||
}, [jobsQ.data, reloadBackups]);
|
||||
|
||||
const [backingUp, setBackingUp] = useState(false);
|
||||
const [backupMsg, setBackupMsg] = useState<{ kind: "success" | "error"; text: string } | null>(null);
|
||||
@@ -571,7 +654,9 @@ export function ServerBackups() {
|
||||
{j.kind === "restore"
|
||||
? t("job_restore")
|
||||
: j.kind === "backup"
|
||||
? t("job_backup")
|
||||
? j.then_restore
|
||||
? t("job_pre_restore")
|
||||
: t("job_backup")
|
||||
: j.kind}
|
||||
</span>
|
||||
{at && (
|
||||
@@ -593,6 +678,7 @@ export function ServerBackups() {
|
||||
{j.message}
|
||||
</span>
|
||||
)}
|
||||
{j.then_restore && <ChainNote job={j} />}
|
||||
</div>
|
||||
</li>
|
||||
);
|
||||
|
||||
Reference in new issue
Block a user