diff --git a/docs/openapi.yaml b/docs/openapi.yaml index 94ff71b..d8f009a 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -1017,6 +1017,54 @@ paths: application/json: schema: { $ref: '#/components/schemas/Error' } + /api/v1/internal/servers/{name}/backup: + post: + tags: [account-internal] + operationId: internalBackupNow + summary: Break-glass on-demand world backup (service token; server must be stopped). + description: >- + The break-glass console (root on the node, holding the service token) POSTs + here to snapshot a stopped world while the API is alive — it goes through the + API rather than direct-to-CRD because rendering the backup Job needs + deployment coordinates only felis-api holds. Same RWO stopped-gate and async + 202 as the external backupNow; there is no Principal (trusted machine caller), + and the action is audited to "break-glass". + x-felis-face: [internal] + x-felis-tier: service + security: [{ serviceToken: [] }] + parameters: + - { name: name, in: path, required: true, schema: { type: string } } + responses: + '202': + description: Backup started. + content: + application/json: + schema: + type: object + required: [name, status] + properties: + name: { type: string } + status: { type: string, const: backing_up } + '400': + description: Invalid server name (bad_name). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '401': + $ref: '#/components/responses/Unauthorized' + '404': + description: Unknown server. + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '409': + description: Server is not stopped (its world PVC is still mounted). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '503': + $ref: '#/components/responses/ServiceUnavailable' + # ----------------------------------------------------- external: servers --- /api/v1/servers/{name}/wake: post: diff --git a/internal/api/api.go b/internal/api/api.go index 983f8cb..fb98c2b 100644 --- a/internal/api/api.go +++ b/internal/api/api.go @@ -266,6 +266,11 @@ func (a *API) internalAPIRoutes() []apiRoute { // Principal); the external face carries the start/status/finish the op drives. {Method: "GET", Pattern: "/api/v1/internal/op-login/pending", h: a.handleOpLoginPending}, {Method: "POST", Pattern: "/api/v1/internal/op-login/{id}/approve", h: a.handleOpLoginApprove}, + + // Break-glass backup (spec §B4 "Sync"): the on-node console POSTs here to + // snapshot a stopped world while the API is alive. Service-token auth (no + // Principal); the shared enqueueBackup tail enforces the RWO stopped-gate. + {Method: "POST", Pattern: "/api/v1/internal/servers/{name}/backup", h: a.handleInternalBackup}, } } diff --git a/internal/api/handlers_backup_now_test.go b/internal/api/handlers_backup_now_test.go index e93c747..be148eb 100644 --- a/internal/api/handlers_backup_now_test.go +++ b/internal/api/handlers_backup_now_test.go @@ -176,3 +176,81 @@ func TestBackupNow(t *testing.T) { } }) } + +// TestInternalBackup exercises POST /api/v1/internal/servers/{name}/backup, the +// break-glass console's face. It shares enqueueBackup with the external handler, so +// the stopped-gate / 503 / async-202 behaviour is proven there; here the focus is the +// internal-face difference: no Principal (service-token auth), no owner gate — even a +// server owned by someone else backs up (the on-node operator is trusted) — and the +// audit is attributed to "break-glass"/"internal", not an email/"external". +func TestInternalBackup(t *testing.T) { + mk := func() (*API, *fakeRepo, *fakeCluster, *fakeBackuper) { + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "someone-else"} + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped", + Ready: false, DesiredState: string(v1alpha1.DesiredStopped)} + backuper := &fakeBackuper{} + api := newTestAPI(repo, cl) + api.Backuper = backuper + return api, repo, cl, backuper + } + + const path = "/api/v1/internal/servers/survival/backup" + + t.Run("stopped server -> 202 + backuper(currentOwner) + break-glass audit", func(t *testing.T) { + api, repo, _, backuper := mk() + w := do(api.InternalHandler(), "POST", path, "", jsonHeader) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String()) + } + // No owner gate on the internal face: a server owned by someone else still backs + // up, and the recorded former owner is the server's CURRENT owner. + if backuper.calls != 1 || backuper.gotName != "survival" || backuper.gotFormerOwn != "someone-else" { + t.Fatalf("backuper saw (calls=%d,%q,%q), want (1,survival,someone-else)", + backuper.calls, backuper.gotName, backuper.gotFormerOwn) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "backup.create" || + repo.audits[0].Actor != "break-glass" || repo.audits[0].Source != "internal" { + t.Fatalf("audit not attributed to break-glass/internal: %+v", repo.audits) + } + }) + + t.Run("running server -> 409 not_stopped, no backup", func(t *testing.T) { + api, _, cl, backuper := mk() + cl.byName["survival"].Ready = true + cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning) + w := do(api.InternalHandler(), "POST", path, "", jsonHeader) + if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if backuper.calls != 0 { + t.Fatal("a running server holds the RWO world PVC — backup must be refused") + } + }) + + t.Run("nil Backuper -> 503 backup_unavailable", func(t *testing.T) { + api, _, _, _ := mk() + api.Backuper = nil + w := do(api.InternalHandler(), "POST", path, "", jsonHeader) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "backup_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + + t.Run("unknown server -> 404", func(t *testing.T) { + api, _, _, _ := mk() + w := do(api.InternalHandler(), "POST", "/api/v1/internal/servers/missing/backup", "", jsonHeader) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) + + t.Run("invalid server name -> 400 bad_name", func(t *testing.T) { + api, _, _, _ := mk() + w := do(api.InternalHandler(), "POST", "/api/v1/internal/servers/X/backup", "", jsonHeader) + if w.Code != http.StatusBadRequest || decodeErr(t, w) != "bad_name" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) +} diff --git a/internal/api/handlers_backups.go b/internal/api/handlers_backups.go index 4a283f2..65d98a8 100644 --- a/internal/api/handlers_backups.go +++ b/internal/api/handlers_backups.go @@ -216,6 +216,41 @@ func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) { return } + a.enqueueBackup(w, r, name, rec, p.Email, "external") +} + +// handleInternalBackup is the internal-face backup trigger. The break-glass console +// (root on the node, holding the service token) POSTs here to snapshot a stopped +// world while felis-api is alive — it goes through the API rather than direct-to-CRD +// like halt does, because rendering the backup Job needs deployment coordinates +// (FELIS_IMAGE, FELIS_BACKUP_PVC) that only felis-api holds. +// +// There is no Principal: the service token is a trusted machine caller (auth.go), so +// the requireInternal middleware IS the authorization — the operator already has root +// on the node. It audits the action to "break-glass" so a console-initiated backup is +// distinguishable from an owner's self-service one. +func (a *API) handleInternalBackup(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + + a.enqueueBackup(w, r, name, rec, "break-glass", "internal") +} + +// enqueueBackup is the shared tail of both backup faces: the RWO stopped-gate, the +// optional-Backuper 503, the async hand-off, and the audit + 202. Both faces validate +// the name and resolve rec themselves and differ only in how the caller is authorized +// (Principal vs trusted service token) and the audit actor/source — keeping the +// security-critical stopped-gate single-sourced so the two faces cannot diverge. +func (a *API) enqueueBackup(w http.ResponseWriter, r *http.Request, name string, rec *ServerRecord, actor, source string) { // Stopped gate: the world PVC is RWO and held by a running server, so a backup // Job cannot double-mount it (mirrors the restore gate). Ready means it is up; // any desiredState other than Stopped means it owns the RWO volume. @@ -244,7 +279,10 @@ func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) { return } - a.audit(r, p.Email, "backup.create", name) + _ = a.Repo.Audit(r.Context(), AuditEntry{ + Actor: actor, Source: source, Action: "backup.create", + ServerName: name, RequestID: requestIDFromContext(r.Context()), + }) writeJSON(w, http.StatusAccepted, map[string]any{ "name": name, "status": "backing_up",