Unverified Commit b6751eac authored by Lemon-miaow's avatar Lemon-miaow
Browse files

feat(backups): 主人和管理员可删除单个备份,归档由 reaper 下一轮删除、异地副本随下次同步删除,恢复可能在读时拒绝

parent db7fbfec
Loading
Loading
Loading
Loading
+40 −0
Changes for docs/openapi.yaml: 40 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -3803,6 +3803,46 @@ paths:
        '401':
          $ref: '#/components/responses/Unauthorized'

  /api/v1/backups/{id}:
    delete:
      tags: [backups]
      operationId: deleteBackup
      summary: Delete one world backup (admin, or the user who owned the world).
      description: >-
        The backup leaves every list, restore and the backup budget at once;
        the reaper's next daily run deletes the archive and the off-site copy's
        next sync removes the bucket's copy. A user gets 404 for a backup
        outside their scope, as their list never shows it. Refused while a
        restore on the backup's server may still read it.
      x-felis-face: [external]
      x-felis-tier: app
      security: [{ sessionCookie: [] }]
      parameters:
        - { name: id, in: path, required: true, schema: { type: string } }
      responses:
        '200':
          description: The backup is deleted; its archive goes at the reaper's next run.
          content:
            application/json:
              schema:
                type: object
                required: [id, status]
                properties:
                  id: { type: string }
                  status: { type: string, const: expired }
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          description: No present backup with this id in the caller's scope (no_backup).
          content:
            application/json:
              schema: { $ref: '#/components/schemas/Error' }
        '409':
          description: A restore running on the backup's server may be reading it (restore_in_progress).
          content:
            application/json:
              schema: { $ref: '#/components/schemas/Error' }

  /api/v1/servers/{name}/restore-backup:
    post:
      tags: [backups]
+26 −0
Changes for docs/troubleshooting.md: 26 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -1082,6 +1082,32 @@ the error its container exited on under Recent operations on the server's
backup page. [GO-TESTED: `TestBackupNow`, `TestCheckRoom`,
`TestReaperConfigManualKeys`, `TestLatestJobsExplainsFailures`]

### Deleting one backup

The delete button on a backup row (`DELETE /api/v1/backups/{id}`) takes that backup
out of every list, every restore and the `max_local_bytes` count at once: its row
turns `expired`, with `expires_at` pulled back to the moment of the delete. An admin
may delete any backup, a user only one of a world they owned (the scope that lists
it); any other id is `404 no_backup`. While a restore on that server may be reading
the archive (a restore Job still running, or a safety snapshot whose restore of this
backup has yet to start) the delete is `409 restore_in_progress`. The audit action
is `backup.delete`, with the backup id, former owner and size in its payload.

The archive stays on the backup volume until the reaper's next daily run, whose
retention pass deletes every `expired` row's archive whatever its `expires_at`; the
off-site copy goes at the sync after the delete. Until that run an admin can take
the delete back, with the id from the audit entry; clearing `offsite_at` has the
next sync copy it off site again in case its copy is already gone:

```sh
sudo k3s kubectl -n felis exec deploy/felis-postgres -c postgres -- psql -U postgres felis -c \
  "UPDATE world_backups SET status = 'present', expires_at = now() + interval '30 days', offsite_at = NULL
   WHERE id = '<backup id>' AND status = 'expired'"
```

[GO-TESTED: `TestDeleteBackup`, `TestExpiredBackupsDeleted`,
`TestSyncExpiresOnlyPastRetention`; PG-TESTED: `TestOwnerDeletedBackup`]

### Every world at once: `felis backup-now`

A world lives only in its volume, and the off-site copy holds only its archives.
+4 −3
Changes for internal/api/api.go: 4 added lines, 3 removed lines.
Original line number Diff line number Diff line
@@ -533,11 +533,12 @@ func (a *API) externalAPIRoutes() []apiRoute {
		// identity (including email_verified) to drive the setup flow.
		{Method: "GET", Pattern: "/api/v1/me", SetupAllowed: true, h: a.handleMe},
		{Method: "GET", Pattern: "/api/v1/me/servers", SetupAllowed: true, h: a.handleMyServers},
		// World backups (spec §7, §466). Both are app-tier: GET /backups is scoped
		// World backups (spec §7, §466). All are app-tier: GET /backups is scoped
		// inside the handler (admin sees all; a user sees only worlds they formerly
		// owned), and restore is gated by owner-or-admin PLUS a former-owner match, so
		// neither sits behind adminOnly.
		// owned), DELETE takes the same scope, and restore is gated by owner-or-admin
		// PLUS a former-owner match, so none sits behind adminOnly.
		{Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups},
		{Method: "DELETE", Pattern: "/api/v1/backups/{id}", h: a.handleDeleteBackup},
		{Method: "GET", Pattern: "/api/v1/servers/{name}/jobs", h: a.handleServerJobs},
		{Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup},
		{Method: "POST", Pattern: "/api/v1/servers/{name}/backup", h: a.handleBackupNow},
+14 −0
Changes for internal/api/api_test.go: 14 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -1058,6 +1058,20 @@ func (f *fakeRepo) BackupByID(_ context.Context, id string) (*BackupRecord, erro
	return nil, ErrNotFound
}

func (f *fakeRepo) ExpireBackup(_ context.Context, id string, at time.Time) error {
	for i := range f.backups {
		b := &f.backups[i]
		if b.view.Status == "present" && b.view.ID == id {
			b.view.Status = "expired"
			if at.Before(b.view.ExpiresAt) {
				b.view.ExpiresAt = at
			}
			return nil
		}
	}
	return ErrNotFound
}

// ---- staff / session auth fakes (spec §B, passwordless) ----
// Each method mirrors the PGRepo contract: a returned StaffUser is copied so a
// test cannot mutate the stored row by reference, SessionUser re-reads the
+69 −0
Changes for internal/api/handlers_backups.go: 69 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -74,6 +74,75 @@ func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
	writeJSON(w, http.StatusOK, map[string]any{"backups": backups, "total": total})
}

// handleDeleteBackup deletes one world backup (DELETE /api/v1/backups/{id}). An
// admin may delete any; a user only one of a world they owned, the scope that
// lists and restores it, and any other id reads as unknown (404), as it does in
// their list. The row turns expired at once, so no list, restore or backup
// budget counts it again; the reaper's next retention pass deletes the archive
// and the off-site copy's next sync the bucket's copy. A reaped world's backup
// is that world's only copy, which the panel says before it asks.
//
// A restore running on the backup's server may be reading the archive, so the
// delete waits for it: a restore Job still running, or a safety snapshot whose
// restore of this backup has yet to start, is 409 restore_in_progress. Without
// a JobStatus reader there is nothing to ask and the delete goes ahead: the
// reaper removes the archive at its next daily run, long after any restore
// admitted before the delete has read it.
func (a *API) handleDeleteBackup(w http.ResponseWriter, r *http.Request) {
	p := principalFromContext(r.Context())
	backup, err := a.Repo.BackupByID(r.Context(), r.PathValue("id"))
	if err == nil && !p.IsAdmin() && (p.UserID == "" || backup.FormerOwner != p.UserID) {
		err = ErrNotFound
	}
	if errors.Is(err, ErrNotFound) {
		writeError(w, r, newError(http.StatusNotFound, "no_backup", "no matching backup exists"))
		return
	}
	if err != nil {
		writeError(w, r, err)
		return
	}
	if a.JobStatus != nil {
		jobs, err := a.JobStatus.LatestJobs(r.Context(), backup.ServerName)
		if err != nil {
			writeError(w, r, err)
			return
		}
		if restoreMayRead(jobs, backup.ID) {
			writeError(w, r, newError(http.StatusConflict, "restore_in_progress",
				"a restore is running on this backup's server and may be reading it; delete it once the restore finishes"))
			return
		}
	}
	if err := a.Repo.ExpireBackup(r.Context(), backup.ID, a.now()); err != nil {
		if errors.Is(err, ErrNotFound) {
			writeError(w, r, newError(http.StatusNotFound, "no_backup", "no matching backup exists"))
			return
		}
		writeError(w, r, err)
		return
	}
	e := AuditEntry{Actor: auditActor(p), ActorUserID: p.UserID, Action: "backup.delete", ServerName: backup.ServerName}
	e.Payload = auditPayload(map[string]any{"backup_id": backup.ID, "former_owner": backup.FormerOwner, "size_bytes": backup.SizeBytes})
	a.auditEntry(r, e)
	writeJSON(w, http.StatusOK, map[string]any{"id": backup.ID, "status": "expired"})
}

// restoreMayRead reports whether one of a server's Jobs may still read backup
// id's archive: any restore Job not yet finished (which archive it extracts is
// not on the Job), or a safety snapshot whose restore of id has not started.
func restoreMayRead(jobs []AsyncJob, id string) bool {
	for _, j := range jobs {
		if j.Kind == "restore" && j.State == "running" {
			return true
		}
		if j.ThenRestore == maintenance.ThenRestorePending && j.RestoreBackupID == id {
			return true
		}
	}
	return false
}

// handleRestoreBackup starts restoring a server's world from a backup (spec §7
// POST /servers/{name}/restore-backup; spec §466). It accepts an optional JSON
// body with a backup_id; when absent it restores the latest backup for the server
Loading