Files
Felis/internal/api/restorechain.go
T

159 lines
5.7 KiB
Go

package api
import (
"context"
"errors"
"fmt"
"log"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
)
// A restore overwrites the world, so by default it starts with a safety snapshot:
// POST /servers/{name}/restore-backup enqueues a backup Job carrying the restore
// (RestoreSnapshotter), and SettleRestoreChains starts that restore once the
// snapshot has succeeded. The snapshot is an ordinary pre_restore row in the
// backup list, which is the way back from a wrong pick.
//
// The chain lives on the backup Job (labels and annotations), so a felis-api
// restart picks it up where it was. While it is pending the Job holds the world
// volume as a restore, which keeps the server down in the gap between the two
// Jobs; if felis-api never settles it, the lock goes with the Job at its TTL and
// the world is left as it was.
// Snapshot Job states a RestoreChain reports.
const (
ChainSnapshotRunning = "running"
ChainSnapshotSucceeded = "succeeded"
ChainSnapshotFailed = "failed"
)
// RestoreChain is one safety snapshot whose restore is still pending.
type RestoreChain struct {
Job string // the snapshot's backup Job
Server string
BackupID string // the backup the restore extracts
BackupRef string
Snapshot string // ChainSnapshotRunning / Succeeded / Failed
}
// Why a chain was abandoned: the code recorded on the snapshot Job and reported
// by the jobs route as then_restore_reason, for the panel to word in its own
// language. chainAbandonText is the English the log and the job's message carry.
const (
ChainAbandonSnapshotFailed = "snapshot_failed"
ChainAbandonNotConfigured = "not_configured"
ChainAbandonServerGone = "server_gone"
ChainAbandonServerStarted = "server_started"
ChainAbandonRestoreBusy = "restore_busy"
)
var chainAbandonText = map[string]string{
ChainAbandonSnapshotFailed: "the safety backup failed, so the world was left as it was",
ChainAbandonNotConfigured: "restore is not configured",
ChainAbandonServerGone: "the server no longer exists",
ChainAbandonServerStarted: "the server was started before the restore could run",
ChainAbandonRestoreBusy: "another restore is running on this world",
}
// chainAbandonMessage words a reason code; an unknown code (a newer felis-api
// wrote it) is shown as is.
func chainAbandonMessage(reason string) string {
if s, ok := chainAbandonText[reason]; ok {
return s
}
return reason
}
// RestoreChains reads the pending chains and records how each was settled
// (maintenance.ThenRestoreStarted or ThenRestoreAbandoned, with the reason code
// of a chain given up).
type RestoreChains interface {
PendingRestoreChains(ctx context.Context) ([]RestoreChain, error)
SettleRestoreChain(ctx context.Context, job, state, reason string) error
}
// snapshotFirst reports whether a restore can start with a safety snapshot:
// the Backuper can carry a restore and something settles the chain.
func (a *API) snapshotFirst() (RestoreSnapshotter, bool) {
if a.RestoreChains == nil || a.Restorer == nil {
return nil, false
}
s, ok := a.Backuper.(RestoreSnapshotter)
return s, ok
}
// SettleRestoreChains advances every pending chain once: it starts the restore
// behind a snapshot that succeeded and gives up the one behind a snapshot that
// failed. A chain whose restore could not be created this time stays pending and
// is retried on the next call. cmd/felis runs it on a short interval.
func (a *API) SettleRestoreChains(ctx context.Context) error {
if a.RestoreChains == nil {
return nil
}
chains, err := a.RestoreChains.PendingRestoreChains(ctx)
if err != nil {
return err
}
var errs []error
for _, c := range chains {
state, reason, err := a.settleRestoreChain(ctx, c)
if err != nil {
errs = append(errs, fmt.Errorf("%s: %w", c.Job, err))
continue
}
if state == "" {
continue
}
if err := a.RestoreChains.SettleRestoreChain(ctx, c.Job, state, reason); err != nil {
errs = append(errs, fmt.Errorf("%s: settle as %s: %w", c.Job, state, err))
continue
}
if state == maintenance.ThenRestoreStarted {
log.Printf("api: safety snapshot %s done; restoring %s from backup %s", c.Job, c.Server, c.BackupID)
} else {
log.Printf("api: restore of %s from backup %s abandoned: %s", c.Server, c.BackupID, chainAbandonMessage(reason))
}
}
return errors.Join(errs...)
}
// settleRestoreChain decides one chain: the state to record ("" to leave it
// pending) and, for an abandoned chain, the reason code.
func (a *API) settleRestoreChain(ctx context.Context, c RestoreChain) (string, string, error) {
switch c.Snapshot {
case ChainSnapshotFailed:
return maintenance.ThenRestoreAbandoned, ChainAbandonSnapshotFailed, nil
case ChainSnapshotSucceeded:
default:
return "", "", nil
}
if a.Restorer == nil {
return maintenance.ThenRestoreAbandoned, ChainAbandonNotConfigured, nil
}
// The pending chain holds the world volume, so nothing should have started
// the server; a start that got past it anyway must not have its world
// replaced underneath.
info, err := a.Cluster.GetServer(ctx, c.Server)
if errors.Is(err, ErrNotFound) {
return maintenance.ThenRestoreAbandoned, ChainAbandonServerGone, nil
}
if err != nil {
return "", "", err
}
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
return maintenance.ThenRestoreAbandoned, ChainAbandonServerStarted, nil
}
if err := a.Restorer.Restore(ctx, c.Server, c.BackupRef); err != nil {
if isRestoreInProgress(err) {
return maintenance.ThenRestoreAbandoned, ChainAbandonRestoreBusy, nil
}
if errors.Is(err, ErrNotFound) {
return maintenance.ThenRestoreAbandoned, ChainAbandonServerGone, nil
}
return "", "", err
}
return maintenance.ThenRestoreStarted, "", nil
}