Files
Felis/internal/api/handlers_backups.go

414 lines
16 KiB
Go

package api
import (
"context"
"errors"
"net/http"
"strings"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
"felis.lolicon.best/internal/naming"
)
// errNoWorldVolume is the shared 409 for backup and restore when the server's
// world PVC does not exist: the Job would only hang Pending on the missing
// claim — invisible to the caller and to the backups list — so the handlers
// refuse up front. Starting the server once (which creates the claim via the
// StatefulSet volumeClaimTemplate) unlocks both ops.
func errNoWorldVolume() error {
return newError(http.StatusConflict, "no_world_volume",
"this server has no world volume yet — start it once to create it, then retry")
}
// handleListBackups lists the world backups visible to the caller (spec §7 GET
// /api/v1/backups; world_backups in §22). It is app-tier: an admin sees every
// present backup; a regular user sees only the backups of worlds they formerly
// owned. The scope is decided here from the Principal and enforced by which Repo
// query runs (AllBackups vs BackupsForUser) — there is no client-supplied filter
// a user could widen, so "a user cannot see another's backups" is a property of
// the query, not of request parsing.
func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
var (
backups []BackupView
err error
)
if p.IsAdmin() {
backups, err = a.Repo.AllBackups(r.Context())
} else {
backups, err = a.Repo.BackupsForUser(r.Context(), p.UserID)
}
if err != nil {
writeError(w, r, err)
return
}
if backups == nil {
backups = []BackupView{}
}
writeJSON(w, http.StatusOK, map[string]any{"backups": backups})
}
// handleRestoreBackup starts restoring a server's world from a backup (spec §7
// POST /servers/{name}/restore-backup; spec §466). It accepts an optional JSON
// body with a backup_id; when absent it restores the latest backup for the server
// (backward-compatible default). The authorization is deliberately stricter than
// ordinary owner-or-admin, in this order:
//
// ① name validation
// ② ServerByName — an unknown server is 404
// ③ owner-or-admin, else 403. A released world's server row is unowned
// (owner_id NULL → OwnerID ""), so this also enforces "重新 claim": a former
// owner must re-claim the server before they can restore into it.
// ④ if the optional backup_id is supplied the handler resolves the specific
// backup; otherwise it picks the most recent present backup, else
// 404 no_backup
// ⑤ cross-server guard: a backup requested by id must belong to the server in
// the path — restoring server A's backup onto server B would be a data leak
// ⑥ former-owner match: a non-admin may restore ONLY a world they formerly
// owned. The current-owner gate in ③ is not enough — user B who
// re-claims a released server could otherwise resurrect user A's world (the
// backup still carries former_owner=A), a data leak. Admin skips this check.
// ⑦ stopped gate: the world PVC must be free, so restore is refused unless the
// server is fully stopped. The world-volume lock (internal/maintenance) then
// makes that atomic against a wake and refuses a second restore, backup or
// file write on the same world with 409 maintenance_in_progress until the
// restore Job finishes.
// ⑧ hand off to the Restorer. Restore is asynchronous (a restore Job, like an
// image build Job), so success means "enqueued" and the handler answers 202.
//
// The opaque backup_ref is resolved server-side from the backup and handed to the
// Restorer directly; the client never names a backup by handle (spec §286
// principle).
func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
// Ownership: owner or admin, mirroring handleStop. An unknown server is 404; an
// unowned (released) server fails the owner check for everyone but admin, which
// is exactly the "must re-claim first" rule.
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
// Optional backup_id in the JSON body; absent → LatestBackup (backward compat).
var body struct {
BackupID string `json:"backup_id"`
}
if strings.HasPrefix(r.Header.Get("Content-Type"), "application/json") {
if err := decodeJSON(w, r, &body); err != nil {
writeError(w, r, err)
return
}
}
// Resolve the backup record. When backup_id is specified the handler resolves
// that exact backup; otherwise it picks the most recent present one.
var backup *BackupRecord
if body.BackupID != "" {
backup, err = a.Repo.BackupByID(r.Context(), body.BackupID)
if err != nil {
if errors.Is(err, ErrNotFound) {
writeError(w, r, newError(http.StatusNotFound, "no_backup",
"no matching backup exists"))
return
}
writeError(w, r, err)
return
}
// Cross-server guard: the backup must belong to the server named in the path.
if backup.ServerName != name {
writeError(w, r, errForbidden)
return
}
} else {
backup, err = a.Repo.LatestBackup(r.Context(), name)
if err != nil {
if errors.Is(err, ErrNotFound) {
writeError(w, r, newError(http.StatusNotFound, "no_backup",
"no restorable backup exists for this server"))
return
}
writeError(w, r, err)
return
}
}
// A non-admin may restore only a world they formerly owned (spec §466). Without
// this a fresh claimant of a released server could resurrect the previous
// owner's world data.
if !p.IsAdmin() && backup.FormerOwner != p.UserID {
writeError(w, r, errForbidden)
return
}
// Stopped gate: the world PVC must be free for the restore to write into it.
// Refuse unless the server is fully stopped — Ready means it is up, and any
// desiredState other than Stopped means it is up or coming up and still owns the
// RWO volume (spec §141 readiness is an RCON probe; DesiredStopped is the
// intent). This is the early, readable refusal from a snapshot; acquireWorld
// below is the atomic one (RWO is per node, so on a single node a restore Job
// WOULD mount beside a running server).
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
writeError(w, r, newError(http.StatusConflict, "not_stopped",
"stop the server before restoring a backup"))
return
}
// World-volume gate: the restore Job mounts the world PVC read-write to unpack
// the archive into it, so a missing claim means a Pod stuck Pending — a 202
// "restoring" with nothing ever written. Same refusal as the backup face
// (shared errNoWorldVolume): the operator starts the server once to create the
// claim, then restores into it.
if exists, err := a.Cluster.WorldVolumeExists(r.Context(), name); err != nil {
writeError(w, r, err)
return
} else if !exists {
writeError(w, r, errNoWorldVolume())
return
}
// Restorer is optional: when unwired the endpoint reports 503 rather than
// panicking, so the authorization boundary above is exercised even before the
// restore-Job executor is wired (see Restorer).
if a.Restorer == nil {
writeError(w, r, newError(http.StatusServiceUnavailable, "restore_unavailable",
"restore subsystem is not configured"))
return
}
// World-volume lock (internal/maintenance). The stopped gate above reads a
// snapshot; this is the atomic check, and it keeps a wake — the owner's, or a
// player's join through velocity — from booting the server on a half-extracted
// world until the restore Job has finished.
release, ok := a.acquireWorld(w, r, name, maintenance.KindRestore, "stop the server before restoring a backup")
if !ok {
return
}
defer release()
if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil {
// ErrNotFound (server vanished from the execution backend) → 404; else 500.
a.writeLookupError(w, r, err)
return
}
a.audit(r, "backup.restore", name)
writeJSON(w, http.StatusAccepted, map[string]any{
"name": name,
"status": "restoring",
"backup_id": backup.ID,
})
}
// handleBackupNow starts an on-demand backup of a server's world (POST
// /api/v1/servers/{name}/backup; spec §18/§19 WorldArchiver, run on demand). It is
// the "back up before I touch it" lever the break-glass console and the owner both
// reach for. Authorization mirrors handleRestoreBackup's front half — the shared
// "who may act on this server's world" gate — but stops short of restore's backup
// resolution and former-owner-match, because a backup is initiated by the CURRENT
// owner and records their ownership; there is no prior owner's data to leak:
//
// ① name validation
// ② ServerByName — an unknown server is 404
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
// released world can still be snapshotted by an operator before disposal)
// ④ stopped gate: refuse unless the server is fully stopped, so the archive is
// quiescent and non-torn. The world-volume lock (internal/maintenance) keeps it
// that way until the backup Job finishes: a wake meanwhile, or a second
// restore/backup/file write, gets 409 maintenance_in_progress.
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
// means "enqueued" and the handler answers 202.
//
// The former owner recorded on the backup is the server's current OwnerID (empty for
// an unowned server backed up by an admin), so the resulting world_backups row is
// restorable by that owner exactly like an inactivity backup.
func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
if !p.IsAdmin() {
if err := a.backupAllowance(r.Context(), name); err != nil {
writeError(w, r, err)
return
}
}
a.enqueueBackup(w, r, name, rec, auditActor(p), "external")
}
// backupAllowance rations an owner's on-demand backups (data-durability-9):
// one per BackupCooldown per server, and none while the present backups fill
// BackupStoreCap. The owner's older backups are pruned by the Job itself
// ([archive] manual_keep), so these two gates bound the rate and the total.
func (a *API) backupAllowance(ctx context.Context, name string) error {
if a.BackupCooldown > 0 {
now := a.now()
last, err := a.Repo.LastBackupRequest(ctx, name, now.Add(-a.BackupCooldown))
if err != nil {
return err
}
if !last.IsZero() {
wait := last.Add(a.BackupCooldown).Sub(now)
if wait > 0 {
return newError(http.StatusTooManyRequests, "backup_cooldown",
"a backup of this server was started %s ago; the next one can start in %s",
now.Sub(last).Round(time.Second), wait.Round(time.Second)).retryAfter(wait)
}
}
}
if a.BackupStoreCap > 0 {
used, err := a.Repo.BackupStoreBytes(ctx)
if err != nil {
return err
}
if used >= a.BackupStoreCap {
return newError(http.StatusInsufficientStorage, "backup_store_full",
"the backup store is full; ask an administrator to free space")
}
}
return nil
}
// handleInternalBackup is the internal-face backup trigger. The break-glass console
// (root on the node, holding the service token) POSTs here to snapshot a stopped
// world while felis-api is alive — it goes through the API rather than direct-to-CRD
// like halt does, because rendering the backup Job needs deployment coordinates
// (FELIS_IMAGE, FELIS_BACKUP_PVC) that only felis-api holds.
//
// There is no Principal: the service token is a trusted machine caller (auth.go), so
// the requireInternal middleware IS the authorization — the operator already has root
// on the node. It audits the action to "break-glass" so a console-initiated backup is
// distinguishable from an owner's self-service one.
//
// The console passes the OS user at the keyboard in an optional {"os_user":"..."} body,
// which becomes the audit actor (parity with the halt peer's accountability). The body
// is decoded whenever one is present — not gated on Content-Type — so a console that
// forgets the header still records the operator rather than silently attributing to the
// generic "break-glass". Absent/blank falls back to "break-glass".
func (a *API) handleInternalBackup(w http.ResponseWriter, r *http.Request) {
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
actor := "break-glass"
if r.ContentLength != 0 {
var body struct {
OSUser string `json:"os_user"`
}
if err := decodeJSON(w, r, &body); err != nil {
writeError(w, r, err)
return
}
if u := strings.TrimSpace(body.OSUser); u != "" {
actor = u
}
}
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
a.enqueueBackup(w, r, name, rec, actor, "internal")
}
// enqueueBackup is the shared tail of both backup faces: the RWO stopped-gate, the
// optional-Backuper 503, the async hand-off, and the audit + 202. Both faces validate
// the name and resolve rec themselves and differ only in how the caller is authorized
// (Principal vs trusted service token) and the audit actor/source — keeping the
// security-critical stopped-gate single-sourced so the two faces cannot diverge.
func (a *API) enqueueBackup(w http.ResponseWriter, r *http.Request, name string, rec *ServerRecord, actor, source string) {
// Stopped gate (mirrors the restore gate): Ready means it is up; any
// desiredState other than Stopped means it is up or coming up. RWO is per node,
// so on a single node the Job WOULD mount beside a live server and archive a
// torn world; acquireWorld below makes this check atomic.
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
writeError(w, r, newError(http.StatusConflict, "not_stopped",
"stop the server before backing up its world"))
return
}
// World-volume gate: the Job mounts the world PVC by claim name, and a missing
// claim would leave its Pod Pending — a 202 "backing_up" with nothing ever
// recorded anywhere. A never-started or already-reaped server is refused with
// the same specificity as the stopped gate.
if exists, err := a.Cluster.WorldVolumeExists(r.Context(), name); err != nil {
writeError(w, r, err)
return
} else if !exists {
writeError(w, r, errNoWorldVolume())
return
}
// Backuper is optional: when unwired the endpoint reports 503 rather than
// panicking, so the authorization boundary above is exercised even before the
// backup-Job executor is wired (see Backuper).
if a.Backuper == nil {
writeError(w, r, newError(http.StatusServiceUnavailable, "backup_unavailable",
"backup subsystem is not configured"))
return
}
// World-volume lock (internal/maintenance): a server woken mid-backup would
// leave a torn archive that a later restore makes permanent.
release, ok := a.acquireWorld(w, r, name, maintenance.KindBackup, "stop the server before backing up its world")
if !ok {
return
}
defer release()
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
a.writeLookupError(w, r, err)
return
}
e := AuditEntry{Actor: actor, Source: source, Action: "backup.create", ServerName: name}
if p := principalFromContext(r.Context()); p != nil {
e.ActorUserID = p.UserID
}
a.auditEntry(r, e)
writeJSON(w, http.StatusAccepted, map[string]any{
"name": name,
"status": "backing_up",
})
}