159 lines
5.7 KiB
Go
159 lines
5.7 KiB
Go
package api
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"log"
|
|
|
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
|
"felis.lolicon.best/internal/maintenance"
|
|
)
|
|
|
|
// A restore overwrites the world, so by default it starts with a safety snapshot:
|
|
// POST /servers/{name}/restore-backup enqueues a backup Job carrying the restore
|
|
// (RestoreSnapshotter), and SettleRestoreChains starts that restore once the
|
|
// snapshot has succeeded. The snapshot is an ordinary pre_restore row in the
|
|
// backup list, which is the way back from a wrong pick.
|
|
//
|
|
// The chain lives on the backup Job (labels and annotations), so a felis-api
|
|
// restart picks it up where it was. While it is pending the Job holds the world
|
|
// volume as a restore, which keeps the server down in the gap between the two
|
|
// Jobs; if felis-api never settles it, the lock goes with the Job at its TTL and
|
|
// the world is left as it was.
|
|
|
|
// Snapshot Job states a RestoreChain reports.
|
|
const (
|
|
ChainSnapshotRunning = "running"
|
|
ChainSnapshotSucceeded = "succeeded"
|
|
ChainSnapshotFailed = "failed"
|
|
)
|
|
|
|
// RestoreChain is one safety snapshot whose restore is still pending.
|
|
type RestoreChain struct {
|
|
Job string // the snapshot's backup Job
|
|
Server string
|
|
BackupID string // the backup the restore extracts
|
|
BackupRef string
|
|
Snapshot string // ChainSnapshotRunning / Succeeded / Failed
|
|
}
|
|
|
|
// Why a chain was abandoned: the code recorded on the snapshot Job and reported
|
|
// by the jobs route as then_restore_reason, for the panel to word in its own
|
|
// language. chainAbandonText is the English the log and the job's message carry.
|
|
const (
|
|
ChainAbandonSnapshotFailed = "snapshot_failed"
|
|
ChainAbandonNotConfigured = "not_configured"
|
|
ChainAbandonServerGone = "server_gone"
|
|
ChainAbandonServerStarted = "server_started"
|
|
ChainAbandonRestoreBusy = "restore_busy"
|
|
)
|
|
|
|
var chainAbandonText = map[string]string{
|
|
ChainAbandonSnapshotFailed: "the safety backup failed, so the world was left as it was",
|
|
ChainAbandonNotConfigured: "restore is not configured",
|
|
ChainAbandonServerGone: "the server no longer exists",
|
|
ChainAbandonServerStarted: "the server was started before the restore could run",
|
|
ChainAbandonRestoreBusy: "another restore is running on this world",
|
|
}
|
|
|
|
// chainAbandonMessage words a reason code; an unknown code (a newer felis-api
|
|
// wrote it) is shown as is.
|
|
func chainAbandonMessage(reason string) string {
|
|
if s, ok := chainAbandonText[reason]; ok {
|
|
return s
|
|
}
|
|
return reason
|
|
}
|
|
|
|
// RestoreChains reads the pending chains and records how each was settled
|
|
// (maintenance.ThenRestoreStarted or ThenRestoreAbandoned, with the reason code
|
|
// of a chain given up).
|
|
type RestoreChains interface {
|
|
PendingRestoreChains(ctx context.Context) ([]RestoreChain, error)
|
|
SettleRestoreChain(ctx context.Context, job, state, reason string) error
|
|
}
|
|
|
|
// snapshotFirst reports whether a restore can start with a safety snapshot:
|
|
// the Backuper can carry a restore and something settles the chain.
|
|
func (a *API) snapshotFirst() (RestoreSnapshotter, bool) {
|
|
if a.RestoreChains == nil || a.Restorer == nil {
|
|
return nil, false
|
|
}
|
|
s, ok := a.Backuper.(RestoreSnapshotter)
|
|
return s, ok
|
|
}
|
|
|
|
// SettleRestoreChains advances every pending chain once: it starts the restore
|
|
// behind a snapshot that succeeded and gives up the one behind a snapshot that
|
|
// failed. A chain whose restore could not be created this time stays pending and
|
|
// is retried on the next call. cmd/felis runs it on a short interval.
|
|
func (a *API) SettleRestoreChains(ctx context.Context) error {
|
|
if a.RestoreChains == nil {
|
|
return nil
|
|
}
|
|
chains, err := a.RestoreChains.PendingRestoreChains(ctx)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
var errs []error
|
|
for _, c := range chains {
|
|
state, reason, err := a.settleRestoreChain(ctx, c)
|
|
if err != nil {
|
|
errs = append(errs, fmt.Errorf("%s: %w", c.Job, err))
|
|
continue
|
|
}
|
|
if state == "" {
|
|
continue
|
|
}
|
|
if err := a.RestoreChains.SettleRestoreChain(ctx, c.Job, state, reason); err != nil {
|
|
errs = append(errs, fmt.Errorf("%s: settle as %s: %w", c.Job, state, err))
|
|
continue
|
|
}
|
|
if state == maintenance.ThenRestoreStarted {
|
|
log.Printf("api: safety snapshot %s done; restoring %s from backup %s", c.Job, c.Server, c.BackupID)
|
|
} else {
|
|
log.Printf("api: restore of %s from backup %s abandoned: %s", c.Server, c.BackupID, chainAbandonMessage(reason))
|
|
}
|
|
}
|
|
return errors.Join(errs...)
|
|
}
|
|
|
|
// settleRestoreChain decides one chain: the state to record ("" to leave it
|
|
// pending) and, for an abandoned chain, the reason code.
|
|
func (a *API) settleRestoreChain(ctx context.Context, c RestoreChain) (string, string, error) {
|
|
switch c.Snapshot {
|
|
case ChainSnapshotFailed:
|
|
return maintenance.ThenRestoreAbandoned, ChainAbandonSnapshotFailed, nil
|
|
case ChainSnapshotSucceeded:
|
|
default:
|
|
return "", "", nil
|
|
}
|
|
if a.Restorer == nil {
|
|
return maintenance.ThenRestoreAbandoned, ChainAbandonNotConfigured, nil
|
|
}
|
|
// The pending chain holds the world volume, so nothing should have started
|
|
// the server; a start that got past it anyway must not have its world
|
|
// replaced underneath.
|
|
info, err := a.Cluster.GetServer(ctx, c.Server)
|
|
if errors.Is(err, ErrNotFound) {
|
|
return maintenance.ThenRestoreAbandoned, ChainAbandonServerGone, nil
|
|
}
|
|
if err != nil {
|
|
return "", "", err
|
|
}
|
|
if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) {
|
|
return maintenance.ThenRestoreAbandoned, ChainAbandonServerStarted, nil
|
|
}
|
|
if err := a.Restorer.Restore(ctx, c.Server, c.BackupRef); err != nil {
|
|
if isRestoreInProgress(err) {
|
|
return maintenance.ThenRestoreAbandoned, ChainAbandonRestoreBusy, nil
|
|
}
|
|
if errors.Is(err, ErrNotFound) {
|
|
return maintenance.ThenRestoreAbandoned, ChainAbandonServerGone, nil
|
|
}
|
|
return "", "", err
|
|
}
|
|
return maintenance.ThenRestoreStarted, "", nil
|
|
}
|