Unverified Commit 489eff44 authored by Lemon-miaow's avatar Lemon-miaow
Browse files

feat(restore): 恢复前默认为当前世界做安全快照并串接恢复 Job,快照失败则不恢复;并发恢复另一备份返回 409

parent d44243c0
Loading
Loading
Loading
Loading
+25 −1
Changes for cmd/felis/api.go: 25 added lines, 1 removed line.
Original line number Diff line number Diff line
@@ -287,6 +287,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
	}

	cluster := api.NewK8sCluster(cl, cfg.K8s.Namespace)
	jobStatus := api.NewK8sJobStatus(cl, cfg.K8s.Namespace)
	a := &api.API{
		Repo:    repo,
		Cluster: cluster,
@@ -300,7 +301,10 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
		Images:    imagePinner(cfg.Registry.URL),
		Restorer:  restorer,
		Backuper:  backuper,
		JobStatus:   api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
		JobStatus: jobStatus,
		// A restore starts with a safety snapshot; settleRestoreChains starts the
		// restore behind each one.
		RestoreChains: jobStatus,
		Files:         files,
		Submissions:   submissions,
		Mailer:        mailer,
@@ -410,6 +414,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
	// and advance any whose Job has reached a terminal phase. GET on a build also
	// reconciles it, but this loop converges builds nobody is polling.
	go reconcileBuilds(ctx, builder, stderr)
	go settleRestoreChains(ctx, a, stderr)

	if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
		go pruner.Loop(ctx, registryPruneInterval)
@@ -650,6 +655,25 @@ func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
	}
}

// settleRestoreChains starts the restore behind each safety snapshot that has
// finished (and gives up the one behind a snapshot that failed). The world stays
// locked in between, so the interval is how long a finished snapshot keeps the
// server down before its restore begins.
func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
	t := time.NewTicker(5 * time.Second)
	defer t.Stop()
	for {
		select {
		case <-ctx.Done():
			return
		case <-t.C:
			if err := a.SettleRestoreChains(ctx); err != nil {
				fmt.Fprintf(stderr, "felis api: restore chains: %v\n", err)
			}
		}
	}
}

// reapRejectedContexts deletes, once an hour, the uploaded contexts of
// submissions rejected more than submit.RejectedContextRetention ago. Without it a
// rejected modpack keeps its bytes on the uploads store (and against its
+29 −9
Changes for cmd/felis/backup.go: 29 added lines, 9 removed lines.
Original line number Diff line number Diff line
@@ -10,6 +10,7 @@ import (
	"time"

	"felis.lolicon.best/internal/backup"
	"felis.lolicon.best/internal/backupjob"
	"felis.lolicon.best/internal/config"
	"felis.lolicon.best/internal/naming"
	"felis.lolicon.best/internal/reaper"
@@ -36,6 +37,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
	server := fs.String("server", "", "server name whose world is being backed up")
	formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
	worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
	reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
	protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
	if err := fs.Parse(args); err != nil {
		return 2
	}
@@ -43,6 +46,11 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
		fmt.Fprintln(stderr, "felis backup: --server is required")
		return 2
	}
	keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
	if !ok {
		fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
		return 2
	}

	cfg, err := config.Load(*cfgPath)
	if err != nil {
@@ -60,6 +68,9 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
		fmt.Fprintf(stderr, "felis backup: %v\n", err)
		return 1
	}
	if keep < 0 {
		keep = rcfg.ManualKeep
	}

	// The world PVC is mounted directly at worldsRoot; the resolver returns it for
	// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
@@ -99,7 +110,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
		FormerOwner: *formerOwner,
		BackupRef:   string(ref),
		SizeBytes:   size,
		Reason:      "manual",
		Reason:      *reason,
		ExpiresAt:   time.Now().Add(rcfg.ManualRetention),
	}
	st := reaper.NewPGStore(drv.DB())
@@ -116,16 +127,25 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
	}

	fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
	pruneManualBackups(ctx, st, archiver, *server, rcfg.ManualKeep, stdout, stderr)
	pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
	return 0
}

// pruneManualBackups keeps server's newest keep on-demand backups and removes
// the rest, oldest first, so repeated backups of one world cannot fill the
// shared archive store. The new backup is already recorded; a removal that
// fails is reported and retried after the next backup.
func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server string, keep int, stdout, stderr io.Writer) {
	excess, err := st.ExcessManualBackups(ctx, server, keep)
const (
	reasonManual = "manual"
	// preRestoreKeep is how many safety snapshots a server keeps: enough to walk
	// back a couple of restores in a row, without every restore adding a world's
	// worth of bytes for the full manual retention.
	preRestoreKeep = 3
)

// pruneBackups keeps server's newest keep backups of this reason and removes the
// rest, oldest first, so repeated backups of one world cannot fill the shared
// archive store. protect is never removed: it is the backup a chained restore is
// about to extract. The new backup is already recorded; a removal that fails is
// reported and retried after the next backup.
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
	excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
	if err != nil {
		fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
		return
@@ -139,7 +159,7 @@ func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup
			fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
			continue
		}
		fmt.Fprintf(stdout, "felis backup: removed older backup %s of %s (keeping the newest %d)\n", b.ID, server, keep)
		fmt.Fprintf(stdout, "felis backup: removed older %s backup %s of %s (keeping the newest %d)\n", reason, b.ID, server, keep)
	}
}

+19 −0
Changes for cmd/felis/backup_test.go: 19 added lines, 0 removed lines.
Original line number Diff line number Diff line
package main

import (
	"bytes"
	"strings"
	"testing"
)

// The reason decides which backups the new one's prune may remove, so an
// unknown one is refused before anything is archived.
func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
	var stderr bytes.Buffer
	if code := cmdBackup([]string{"--server", "survival", "--reason", "inactive_15d"}, &bytes.Buffer{}, &stderr); code != 2 {
		t.Fatalf("exit = %d, want 2 (%s)", code, stderr.String())
	}
	if !strings.Contains(stderr.String(), "unknown --reason") {
		t.Fatalf("stderr = %q", stderr.String())
	}
}
+37 −4
Changes for docs/openapi.yaml: 37 added lines, 4 removed lines.
Original line number Diff line number Diff line
@@ -362,7 +362,9 @@ components:
          type: string
          description: Present only when the world had an owner at backup time.
        size_bytes: { type: integer, format: int64 }
        reason: { type: string }
        reason:
          type: string
          description: inactive_15d (idle reclaim), manual (on demand) or pre_restore (the safety snapshot in front of a restore).
        status: { type: string }
        created_at: { type: string, format: date-time }
        expires_at: { type: string, format: date-time }
@@ -2961,6 +2963,15 @@ paths:
      tags: [backups]
      operationId: restoreBackup
      summary: Restore a world from a backup (owner-or-admin plus a former-owner match).
      description: >-
        By default the restore starts with a safety snapshot: a backup of the
        data volume as it is now (reason "pre_restore", the newest 3 kept per
        server), and the restore Job starts only once that backup has
        succeeded. If the snapshot fails the restore is given up and the world
        is left as it was. GET /servers/{name}/jobs shows the snapshot as a
        backup job whose then_restore says what became of the restore. The
        world stays locked from the request until the restore Job finishes.
        Pass safety_snapshot false to restore straight away.
      x-felis-face: [external]
      x-felis-tier: app
      security: [{ accessJWT: [] }]
@@ -2976,18 +2987,25 @@ paths:
                backup_id:
                  type: string
                  description: Which backup to restore; defaults to the latest for the server.
                safety_snapshot:
                  type: boolean
                  default: true
                  description: Back up the current world before overwriting it.
      responses:
        '202':
          description: Restore started.
          description: Restore started (after the safety snapshot when safety_snapshot is true).
          content:
            application/json:
              schema:
                type: object
                required: [name, status, backup_id]
                required: [name, status, backup_id, safety_snapshot]
                properties:
                  name: { type: string }
                  status: { type: string, const: restoring }
                  backup_id: { type: string }
                  safety_snapshot:
                    type: boolean
                    description: Whether a safety snapshot runs first. False when the request turned it off, or when this install cannot take one.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
@@ -2998,7 +3016,7 @@ paths:
            application/json:
              schema: { $ref: '#/components/schemas/Error' }
        '409':
          description: Server is not stopped (not_stopped), or a restore, backup or file write already holds its world volume (maintenance_in_progress).
          description: Server is not stopped (not_stopped), a restore, backup or file write already holds its world volume (maintenance_in_progress), or a restore of another backup is still running (restore_in_progress).
          content:
            application/json:
              schema: { $ref: '#/components/schemas/Error' }
@@ -3088,6 +3106,21 @@ paths:
                        message: { type: string }
                        started_at: { type: string, format: date-time }
                        finished_at: { type: string, format: date-time }
                        then_restore:
                          type: string
                          enum: [pending, started, abandoned]
                          description: >-
                            Set on a restore's safety snapshot (a backup job):
                            pending until the restore behind it starts, or
                            abandoned with then_restore_reason saying why (its
                            English wording is in message).
                        then_restore_reason:
                          type: string
                          enum: [snapshot_failed, not_configured, server_gone, server_started, restore_busy]
                          description: Why an abandoned chain was given up.
                        restore_backup_id:
                          type: string
                          description: The backup the chained restore extracts.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
+6 −0
Changes for internal/api/api.go: 6 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -75,6 +75,12 @@ type API struct {
	// endpoints only answer 202). Optional: nil → that route reports 503.
	JobStatus JobStatusReader

	// RestoreChains finds and settles the safety snapshots that run in front of a
	// restore (restorechain.go). A restore takes a snapshot first only when this
	// is set, the Backuper can chain a restore and the Restorer is wired, since
	// something has to start the restore once the snapshot is done.
	RestoreChains RestoreChains

	// Files is the server file editor (list / read / write a file in a stopped
	// server's world volume — the "one wrong line in server.properties" repair).
	// Like Restorer and Backuper it is optional: when nil the file routes report
Loading