Files
Felis/internal/api/scheduledbackups.go
T

171 lines
5.8 KiB
Go

package api
import (
"context"
"errors"
"fmt"
"log"
"time"
"felis.lolicon.best/internal/maintenance"
)
// Scheduled backups. A world played every day never idles long enough for the
// reaper to archive it, and an owner's own backups are only the ones they
// remembered to take, so felis-api takes a daily restore point of every world
// played since its last one. The backup Job mounts the world volume, which a
// running server holds, so the point is taken once the server is stopped: idle
// auto-stop brings a played world down minutes after its last player leaves,
// and the point lands the same day. A server kept running around the clock gets
// one the next time it stops.
//
// The point is an ordinary world_backups row of reason scheduled: listed and
// restorable by its owner, copied offsite like any other, pruned to [archive]
// scheduled_keep per owner and expired after scheduled_retention, so it never
// takes the place of a backup the owner asked for.
// ScheduledBackuper enqueues the backup Job of a scheduled restore point
// (backupjob.Backuper).
type ScheduledBackuper interface {
BackupScheduled(ctx context.Context, serverName, formerOwner string) error
}
// ScheduledCandidate is an owned world due a scheduled backup.
type ScheduledCandidate struct {
Name string
OwnerID string
}
// ScheduleStore lists the worlds due a scheduled backup: owned, joined since
// their owner's newest intact scheduled backup, and without one taken after
// before. Worlds never given one come first, then the longest waiting.
type ScheduleStore interface {
ScheduledBackupCandidates(ctx context.Context, before time.Time) ([]ScheduledCandidate, error)
}
// WorldJobCounter counts the backup and restore Jobs still running.
type WorldJobCounter interface {
RunningWorldJobs(ctx context.Context) (int, error)
}
// Audit identity of a scheduled backup. The action is its own so the owner's
// manual cooldown (LastBackupRequest, backup.create) never counts it.
const (
scheduledBackupActor = "scheduler"
scheduledBackupAction = "backup.scheduled"
)
// BackupScheduler takes the scheduled backups. Tick launches at most one
// backup Job and only while no backup or restore Job is running, so the points
// queue behind each other and behind the owners' own operations instead of
// loading the node with archives all at once.
type BackupScheduler struct {
API *API
Store ScheduleStore
Jobs WorldJobCounter
// Every is how often a played world gets a point ([archive]
// scheduled_every). Zero or less turns the scheduler off.
Every time.Duration
// tried is when each world last had a Job launched or was found without a
// volume, so one whose backup keeps failing is retried every Every/4
// instead of every tick.
tried map[string]time.Time
// full remembers that the store was at max_local_bytes, so the pause and
// the resume are logged once each.
full bool
}
// Tick launches the backup of the world waiting longest, if any may run now.
func (s *BackupScheduler) Tick(ctx context.Context) error {
b, ok := s.API.Backuper.(ScheduledBackuper)
if !ok || s.Every <= 0 {
return nil
}
now := s.API.now()
running, err := s.Jobs.RunningWorldJobs(ctx)
if err != nil {
return fmt.Errorf("count running backup and restore jobs: %w", err)
}
if running > 0 {
return nil
}
// A full store is the reaper's to evict; a scheduled point added now would
// only push out an older backup of somebody else's.
if limit := s.API.BackupStoreCap; limit > 0 {
used, err := s.API.Repo.BackupStoreBytes(ctx)
if err != nil {
return fmt.Errorf("read the backup store size: %w", err)
}
full := used >= limit
if full != s.full {
if full {
log.Printf("api: scheduled backups paused: the backup store holds %d of its %d bytes ([archive] max_local_bytes)", used, limit)
} else {
log.Printf("api: scheduled backups resumed: the backup store is below [archive] max_local_bytes")
}
s.full = full
}
if full {
return nil
}
}
due, err := s.Store.ScheduledBackupCandidates(ctx, now.Add(-s.Every))
if err != nil {
return fmt.Errorf("list the worlds due a scheduled backup: %w", err)
}
if s.tried == nil {
s.tried = map[string]time.Time{}
}
for name, at := range s.tried {
if now.Sub(at) >= s.Every {
delete(s.tried, name)
}
}
for _, c := range due {
if at, ok := s.tried[c.Name]; ok && now.Sub(at) < s.Every/4 {
continue
}
// A server never started has no world yet, and one whose volume is gone
// has nothing left to save; the Job would sit Pending on the claim.
exists, err := s.API.Cluster.WorldVolumeExists(ctx, c.Name)
if err != nil {
return fmt.Errorf("look up the world volume of %s: %w", c.Name, err)
}
if !exists {
s.tried[c.Name] = now
continue
}
var busy *MaintenanceBusyError
switch err := s.API.Cluster.AcquireMaintenance(ctx, c.Name, maintenance.KindBackup); {
case errors.Is(err, ErrNotStopped), errors.As(err, &busy), errors.Is(err, ErrMaintenanceInProgress):
continue // running, or somebody else has the world: next tick
case errors.Is(err, ErrNotFound):
s.tried[c.Name] = now
continue
case err != nil:
return fmt.Errorf("lock the world of %s: %w", c.Name, err)
}
err = b.BackupScheduled(ctx, c.Name, c.OwnerID)
// Once the Job exists it holds the world; the annotation only covered
// the gap.
if rerr := s.API.Cluster.ReleaseMaintenance(context.WithoutCancel(ctx), c.Name); rerr != nil {
log.Printf("api: release the maintenance lock on %s: %v (it lapses after %s)", c.Name, rerr, maintenance.Grace)
}
s.tried[c.Name] = now
if err != nil {
return fmt.Errorf("start the scheduled backup of %s: %w", c.Name, err)
}
log.Printf("api: started the scheduled backup of %s", c.Name)
s.API.writeAudit(ctx, AuditEntry{
Actor: scheduledBackupActor, Source: scheduledBackupActor,
Action: scheduledBackupAction, ServerName: c.Name,
})
return nil
}
return nil
}