Files
Felis/internal/api/handlers_backups.go
T
flyemoji 7a7c0d53ab feat(api): add on-demand world backup endpoint and Job executor (§B4 Sync)
Add POST /api/v1/servers/{name}/backup: an owner or admin snapshots a
stopped server's world into the archive store on demand, recorded as a
first-class world_backups row (reason `manual`) — restorable by the
existing restore path and expired by the reaper's retention pass, so it
never leaks as an orphan archive. This is the break-glass "Sync" op,
resolved as immediate/on-demand backup.

felis-api cannot archive in-process (the world PVC is RWO, held by the
operator StatefulSet), so the work hands off to a one-shot Kubernetes Job
(new internal/backupjob) that mounts the world PVC read-only and the
backup PVC read-write, plus the felis config Secret so it self-records
its row atomically like the reaper. The Pod mirrors restore's weak-SA
isolation (SA token un-mounted, non-root, read-only rootfs, drop ALL);
the one reviewed departure is that config-Secret mount, frozen by
jobspec_test.go. Handler answers 202 backing_up; gated on the server
being Stopped (RWO world PVC), owner-or-admin, and FELIS_IMAGE +
FELIS_BACKUP_PVC being wired (else 503 backup_unavailable).

Each request mints a unique Job name (backup-<server>-<rand>) so a repeat
on-demand backup produces a fresh archive rather than colliding with a
just-finished Job still inside its TTL window and silently no-op'ing the
retry.
2026-07-07 10:04:30 +09:00

253 lines
9.5 KiB
Go

package api
import (
"errors"
"net/http"
"strings"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/naming"
)
// handleListBackups lists the world backups visible to the caller (spec §7 GET
// /api/v1/backups; world_backups in §22). It is app-tier: an admin sees every
// present backup; a regular user sees only the backups of worlds they formerly
// owned. The scope is decided here from the Principal and enforced by which Repo
// query runs (AllBackups vs BackupsForUser) — there is no client-supplied filter
// a user could widen, so "a user cannot see another's backups" is a property of
// the query, not of request parsing.
func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
var (
backups []BackupView
err error
)
if p.IsAdmin() {
backups, err = a.Repo.AllBackups(r.Context())
} else {
backups, err = a.Repo.BackupsForUser(r.Context(), p.UserID)
}
if err != nil {
writeError(w, r, err)
return
}
if backups == nil {
backups = []BackupView{}
}
writeJSON(w, http.StatusOK, map[string]any{"backups": backups})
}
// handleRestoreBackup starts restoring a server's world from a backup (spec §7
// POST /servers/{name}/restore-backup; spec §466). It accepts an optional JSON
// body with a backup_id; when absent it restores the latest backup for the server
// (backward-compatible default). The authorization is deliberately stricter than
// ordinary owner-or-admin, in this order:
//
// ① name validation
// ② ServerByName — an unknown server is 404
// ③ owner-or-admin, else 403. A released world's server row is unowned
// (owner_id NULL → OwnerID ""), so this also enforces "重新 claim": a former
// owner must re-claim the server before they can restore into it.
// ④ if the optional backup_id is supplied the handler resolves the specific
// backup; otherwise it picks the most recent present backup, else
// 404 no_backup
// ⑤ cross-server guard: a backup requested by id must belong to the server in
// the path — restoring server A's backup onto server B would be a data leak
// ⑥ former-owner match: a non-admin may restore ONLY a world they formerly
// owned. The current-owner gate in ③ is not enough — user B who
// re-claims a released server could otherwise resurrect user A's world (the
// backup still carries former_owner=A), a data leak. Admin skips this check.
// ⑦ stopped gate: the world PVC must be free, so restore is refused unless the
// server is fully stopped.
// ⑧ hand off to the Restorer. Restore is asynchronous (a restore Job, like an
// image build Job), so success means "enqueued" and the handler answers 202.
//
// The opaque backup_ref is resolved server-side from the backup and handed to the
// Restorer directly; the client never names a backup by handle (spec §286
// principle).
func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
// Ownership: owner or admin, mirroring handleStop. An unknown server is 404; an
// unowned (released) server fails the owner check for everyone but admin, which
// is exactly the "must re-claim first" rule.
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
// Optional backup_id in the JSON body; absent → LatestBackup (backward compat).
var body struct {
BackupID string `json:"backup_id"`
}
if strings.HasPrefix(r.Header.Get("Content-Type"), "application/json") {
if err := decodeJSON(w, r, &body); err != nil {
writeError(w, r, err)
return
}
}
// Resolve the backup record. When backup_id is specified the handler resolves
// that exact backup; otherwise it picks the most recent present one.
var backup *BackupRecord
if body.BackupID != "" {
backup, err = a.Repo.BackupByID(r.Context(), body.BackupID)
if err != nil {
if errors.Is(err, ErrNotFound) {
writeError(w, r, newError(http.StatusNotFound, "no_backup",
"no matching backup exists"))
return
}
writeError(w, r, err)
return
}
// Cross-server guard: the backup must belong to the server named in the path.
if backup.ServerName != name {
writeError(w, r, errForbidden)
return
}
} else {
backup, err = a.Repo.LatestBackup(r.Context(), name)
if err != nil {
if errors.Is(err, ErrNotFound) {
writeError(w, r, newError(http.StatusNotFound, "no_backup",
"no restorable backup exists for this server"))
return
}
writeError(w, r, err)
return
}
}
// A non-admin may restore only a world they formerly owned (spec §466). Without
// this a fresh claimant of a released server could resurrect the previous
// owner's world data.
if !p.IsAdmin() && backup.FormerOwner != p.UserID {
writeError(w, r, errForbidden)
return
}
// Stopped gate: the world PVC must be free for the restore to write into it.
// Refuse unless the server is fully stopped — Ready means it is up, and any
// desiredState other than Stopped means it is up or coming up and still owns the
// RWO volume (spec §141 readiness is an RCON probe; DesiredStopped is the
// intent). This yields a specific 409 instead of a restore Job that cannot mount.
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
writeError(w, r, newError(http.StatusConflict, "not_stopped",
"stop the server before restoring a backup"))
return
}
// Restorer is optional: when unwired the endpoint reports 503 rather than
// panicking, so the authorization boundary above is exercised even before the
// restore-Job executor is wired (see Restorer).
if a.Restorer == nil {
writeError(w, r, newError(http.StatusServiceUnavailable, "restore_unavailable",
"restore subsystem is not configured"))
return
}
if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil {
// ErrNotFound (server vanished from the execution backend) → 404; else 500.
a.writeLookupError(w, r, err)
return
}
a.audit(r, p.Email, "backup.restore", name)
writeJSON(w, http.StatusAccepted, map[string]any{
"name": name,
"status": "restoring",
"backup_id": backup.ID,
})
}
// handleBackupNow starts an on-demand backup of a server's world (POST
// /api/v1/servers/{name}/backup; spec §18/§19 WorldArchiver, run on demand). It is
// the "back up before I touch it" lever the break-glass console and the owner both
// reach for. Authorization mirrors handleRestoreBackup's front half — the shared
// "who may act on this server's world" gate — but stops short of restore's backup
// resolution and former-owner-match, because a backup is initiated by the CURRENT
// owner and records their ownership; there is no prior owner's data to leak:
//
// ① name validation
// ② ServerByName — an unknown server is 404
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
// released world can still be snapshotted by an operator before disposal)
// ④ stopped gate: the world PVC is RWO and held by a running server, so a backup
// Job cannot double-mount it — refuse unless the server is fully stopped. This
// also guarantees a quiescent, non-torn archive.
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
// means "enqueued" and the handler answers 202.
//
// The former owner recorded on the backup is the server's current OwnerID (empty for
// an unowned server backed up by an admin), so the resulting world_backups row is
// restorable by that owner exactly like an inactivity backup.
func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
// Stopped gate: the world PVC is RWO and held by a running server, so a backup
// Job cannot double-mount it (mirrors the restore gate). Ready means it is up;
// any desiredState other than Stopped means it owns the RWO volume.
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
writeError(w, r, newError(http.StatusConflict, "not_stopped",
"stop the server before backing up its world"))
return
}
// Backuper is optional: when unwired the endpoint reports 503 rather than
// panicking, so the authorization boundary above is exercised even before the
// backup-Job executor is wired (see Backuper).
if a.Backuper == nil {
writeError(w, r, newError(http.StatusServiceUnavailable, "backup_unavailable",
"backup subsystem is not configured"))
return
}
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
a.writeLookupError(w, r, err)
return
}
a.audit(r, p.Email, "backup.create", name)
writeJSON(w, http.StatusAccepted, map[string]any{
"name": name,
"status": "backing_up",
})
}