330 lines
12 KiB
Go
330 lines
12 KiB
Go
package backupjob
|
|
|
|
import (
|
|
"fmt"
|
|
"time"
|
|
|
|
batchv1 "k8s.io/api/batch/v1"
|
|
corev1 "k8s.io/api/core/v1"
|
|
"k8s.io/apimachinery/pkg/api/resource"
|
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
|
)
|
|
|
|
// Label keys applied to backup objects, mirroring internal/restore so the two
|
|
// executors are observable the same way.
|
|
const (
|
|
LabelManagedBy = "app.kubernetes.io/managed-by"
|
|
LabelComponent = "app.kubernetes.io/component"
|
|
LabelServer = "felis.lolicon.best/server"
|
|
|
|
managedByValue = "felis-backup"
|
|
componentValue = "world-backup"
|
|
|
|
// The restore chain a safety snapshot carries (internal/maintenance keeps the
|
|
// canonical copies; maintenance_test pins these against them).
|
|
labelThenRestore = "felis.lolicon.best/then-restore"
|
|
thenRestorePending = "pending"
|
|
annotationRestoreRef = "felis.lolicon.best/restore-ref"
|
|
annotationRestoreBackupID = "felis.lolicon.best/restore-backup-id"
|
|
|
|
// ReasonPreRestore is the world_backups reason of a safety snapshot.
|
|
ReasonPreRestore = "pre_restore"
|
|
// ReasonScheduled is the world_backups reason of the daily restore point
|
|
// felis-api takes of a world played since its last one.
|
|
ReasonScheduled = "scheduled"
|
|
// LabelReason marks a scheduled backup Job, so the jobs route can tell it
|
|
// from one somebody asked for.
|
|
LabelReason = "felis.lolicon.best/backup-reason"
|
|
|
|
worldVolume = "world"
|
|
backupVolume = "backup"
|
|
configVolume = "config"
|
|
tmpVolume = "tmp"
|
|
felisBinaryPath = "/usr/local/bin/felis"
|
|
)
|
|
|
|
// JobParams are the rendered inputs to a backup Job, derived from a server +
|
|
// former owner + Config by the Backuper. jobspec is a pure function of them so
|
|
// the security-critical Job shape is unit-tested without a cluster.
|
|
type JobParams struct {
|
|
Server string
|
|
// JobName is the unique Job object name for this invocation. The Backuper sets a
|
|
// fresh per-call name (BackupJobName(server) + random suffix) so every on-demand
|
|
// request produces its own archive — a deterministic name would collide with a
|
|
// just-finished Job still inside its TTL window and silently no-op the retry.
|
|
// Empty falls back to the deterministic BackupJobName (tests, and any direct call).
|
|
JobName string
|
|
FormerOwner string // recorded on the world_backups row so the owner can later restore it
|
|
WorldPVC string
|
|
BackupPVC string
|
|
Namespace string
|
|
ServiceAccount string
|
|
Image string
|
|
ConfigSecret string // the felis config Secret mounted for the DB URL (self-recording)
|
|
ConfigMount string // where the config Secret is mounted (holds felis.toml)
|
|
BackupRoot string
|
|
WorldsRoot string
|
|
Deadline time.Duration
|
|
CPULimit string
|
|
MemLimit string
|
|
RunAsUser int64
|
|
RunAsGroup int64
|
|
FSGroup int64
|
|
|
|
// RestoreRef / RestoreBackupID make this backup the safety snapshot in front
|
|
// of a restore: the Job is labelled as a pending chain and names the backup
|
|
// felis-api restores once the snapshot succeeds (internal/maintenance). The
|
|
// snapshot is recorded as ReasonPreRestore, and its prune spares the backup
|
|
// the restore will extract.
|
|
RestoreRef string
|
|
RestoreBackupID string
|
|
// Scheduled records the backup as ReasonScheduled, pruned to its own
|
|
// [archive] scheduled_keep, and labels the Job LabelReason. It never carries
|
|
// a restore.
|
|
Scheduled bool
|
|
|
|
TTLAfterFinished time.Duration
|
|
}
|
|
|
|
// BackupJobName is the base Job name for a server's on-demand backup — the prefix
|
|
// the Backuper extends with a per-invocation random suffix (JobParams.JobName) so
|
|
// repeat backups do not collide. It is distinct from the restore Job name so a
|
|
// backup and a restore of the same server never collide.
|
|
func BackupJobName(server string) string { return "backup-" + server }
|
|
|
|
// PodSelector matches every world-backup Job pod, whichever server it backs up:
|
|
// the peer the database's ingress fence admits (internal/platform/postgres.go).
|
|
func PodSelector() map[string]string {
|
|
return map[string]string{
|
|
LabelManagedBy: managedByValue,
|
|
LabelComponent: componentValue,
|
|
}
|
|
}
|
|
|
|
func backupLabels(p JobParams) map[string]string {
|
|
labels := PodSelector()
|
|
labels[LabelServer] = p.Server
|
|
return labels
|
|
}
|
|
|
|
// BackupJob renders the on-demand world-backup Job (spec §18/§19 WorldArchiver,
|
|
// run on demand rather than on the reaper's daily schedule). Its isolation mirrors
|
|
// the restore Job (weak SA, non-root, read-only root fs, drop ALL, one-shot with a
|
|
// deadline) with ONE deliberate, security-reviewed departure asserted by
|
|
// jobspec_test.go:
|
|
//
|
|
// - It DOES mount the felis config Secret (read-only), because a backup must
|
|
// self-record its world_backups row — archive-then-insert atomically, exactly
|
|
// like the reaper, which is the only other component holding both a world mount
|
|
// and the database. A restore Pod must never touch the DB because it processes a
|
|
// potentially poisoned archive; a backup Pod only READS a world the operator
|
|
// already owns and tars it (bytes, never executed), so the poisoned-input threat
|
|
// that forbids restore's DB access does not apply. Its blast radius is the DB
|
|
// plus the two PVCs, matching the reaper's trust for a strict subset of the
|
|
// reaper's operations (it never deletes a PVC and never calls the K8s API — its
|
|
// SA token stays un-mounted).
|
|
// - The world PVC is mounted READ-ONLY (backup only reads it; the RWO volume must
|
|
// be free, which the handler's stopped-gate guarantees), and the backup PVC
|
|
// READ-WRITE (the archive is written into it) — the mirror image of restore.
|
|
//
|
|
// The container runs `/usr/local/bin/felis backup` (cmd/felis), which tars the
|
|
// world at WorldsRoot into BackupRoot and inserts the world_backups row. BackupRoot
|
|
// MUST equal the reaper's [archive] local_path so the recorded absolute ref
|
|
// resolves the same way a later restore Job mounts it.
|
|
func BackupJob(p JobParams) (*batchv1.Job, error) {
|
|
if p.Image == "" {
|
|
return nil, fmt.Errorf("backup: image is empty")
|
|
}
|
|
if p.WorldPVC == "" || p.BackupPVC == "" {
|
|
return nil, fmt.Errorf("backup: world and backup PVC names are required")
|
|
}
|
|
if p.ConfigSecret == "" {
|
|
return nil, fmt.Errorf("backup: config secret name is required")
|
|
}
|
|
if p.RestoreRef != "" && p.RestoreBackupID == "" {
|
|
return nil, fmt.Errorf("backup: a chained restore needs the backup id")
|
|
}
|
|
if p.RestoreRef != "" && p.Scheduled {
|
|
return nil, fmt.Errorf("backup: a scheduled backup carries no restore")
|
|
}
|
|
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
deadline := int64(p.Deadline / time.Second)
|
|
if deadline <= 0 {
|
|
deadline = int64(defaultDeadline / time.Second)
|
|
}
|
|
ttl := int32(p.TTLAfterFinished / time.Second)
|
|
if ttl <= 0 {
|
|
ttl = int32(defaultTTL / time.Second)
|
|
}
|
|
|
|
args := []string{
|
|
"--config", p.ConfigMount + "/felis.toml",
|
|
"--server", p.Server,
|
|
"--worlds-root", p.WorldsRoot,
|
|
}
|
|
// FormerOwner is optional: an admin backing up an unowned (released) server
|
|
// records an empty former_owner, exactly as the reaper does for an unowned reap.
|
|
if p.FormerOwner != "" {
|
|
args = append(args, "--former-owner", p.FormerOwner)
|
|
}
|
|
if p.RestoreRef != "" {
|
|
args = append(args, "--reason", ReasonPreRestore, "--protect", p.RestoreBackupID)
|
|
}
|
|
if p.Scheduled {
|
|
args = append(args, "--reason", ReasonScheduled)
|
|
}
|
|
|
|
container := corev1.Container{
|
|
Name: "backup",
|
|
Image: p.Image,
|
|
Command: []string{felisBinaryPath, "backup"},
|
|
Args: args,
|
|
VolumeMounts: []corev1.VolumeMount{
|
|
// The world is only read to tar it; mounting it read-only means the
|
|
// backup process can never mutate the world it snapshots.
|
|
{Name: worldVolume, MountPath: p.WorldsRoot, ReadOnly: true},
|
|
{Name: backupVolume, MountPath: p.BackupRoot},
|
|
{Name: configVolume, MountPath: p.ConfigMount, ReadOnly: true},
|
|
{Name: tmpVolume, MountPath: "/tmp"},
|
|
},
|
|
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
|
SecurityContext: &corev1.SecurityContext{
|
|
Privileged: boolPtr(false),
|
|
AllowPrivilegeEscalation: boolPtr(false),
|
|
ReadOnlyRootFilesystem: boolPtr(true),
|
|
// DAC_OVERRIDE is granted on top of dropping ALL: the pod runs as root,
|
|
// but the world may have been written by a game image whose UID is
|
|
// neither root nor ours, and Paper's own files are mode 0600. It is the
|
|
// minimal extra power that makes the archive read every world shape.
|
|
Capabilities: &corev1.Capabilities{
|
|
Drop: []corev1.Capability{"ALL"},
|
|
Add: []corev1.Capability{"DAC_OVERRIDE"},
|
|
},
|
|
},
|
|
}
|
|
|
|
// The exit error reaches GET /servers/{name}/jobs through the terminated
|
|
// state (api.K8sJobStatus), in place of the Job's generic backoff text.
|
|
container.TerminationMessagePolicy = corev1.TerminationMessageFallbackToLogsOnError
|
|
|
|
name := p.JobName
|
|
if name == "" {
|
|
name = BackupJobName(p.Server)
|
|
}
|
|
meta := metav1.ObjectMeta{
|
|
Name: name,
|
|
Namespace: p.Namespace,
|
|
Labels: backupLabels(p),
|
|
}
|
|
if p.RestoreRef != "" {
|
|
// On the Job only: felis-api settles the chain by patching this label, and
|
|
// the pods never need it.
|
|
meta.Labels[labelThenRestore] = thenRestorePending
|
|
meta.Annotations = map[string]string{
|
|
annotationRestoreRef: p.RestoreRef,
|
|
annotationRestoreBackupID: p.RestoreBackupID,
|
|
}
|
|
}
|
|
if p.Scheduled {
|
|
meta.Labels[LabelReason] = ReasonScheduled
|
|
}
|
|
job := &batchv1.Job{
|
|
ObjectMeta: meta,
|
|
Spec: batchv1.JobSpec{
|
|
// One shot: a wedged archive must not loop. The TTL GCs the finished Job
|
|
// so a later backup of the same server is not blocked forever by a stale
|
|
// completed Job.
|
|
BackoffLimit: int32Ptr(0),
|
|
ActiveDeadlineSeconds: int64Ptr(deadline),
|
|
TTLSecondsAfterFinished: int32Ptr(ttl),
|
|
Template: corev1.PodTemplateSpec{
|
|
ObjectMeta: metav1.ObjectMeta{Labels: backupLabels(p)},
|
|
Spec: corev1.PodSpec{
|
|
RestartPolicy: corev1.RestartPolicyNever,
|
|
ServiceAccountName: p.ServiceAccount,
|
|
AutomountServiceAccountToken: boolPtr(false),
|
|
// Root by default (see Config.RunAsUser): the world volume's
|
|
// owner is the game uid, so only an owner-matching or
|
|
// DAC-overriding uid can read it. FSGroup is omitted when unset
|
|
// so a root pod never triggers a volume chgrp.
|
|
SecurityContext: backupPodSecurityContext(p),
|
|
Containers: []corev1.Container{container},
|
|
Volumes: []corev1.Volume{
|
|
{
|
|
Name: worldVolume,
|
|
VolumeSource: corev1.VolumeSource{
|
|
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
|
ClaimName: p.WorldPVC,
|
|
ReadOnly: true,
|
|
},
|
|
},
|
|
},
|
|
{
|
|
Name: backupVolume,
|
|
VolumeSource: corev1.VolumeSource{
|
|
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
|
|
ClaimName: p.BackupPVC,
|
|
},
|
|
},
|
|
},
|
|
{
|
|
Name: configVolume,
|
|
VolumeSource: corev1.VolumeSource{
|
|
Secret: &corev1.SecretVolumeSource{SecretName: p.ConfigSecret},
|
|
},
|
|
},
|
|
{Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}},
|
|
},
|
|
},
|
|
},
|
|
},
|
|
}
|
|
return job, nil
|
|
}
|
|
|
|
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
|
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
|
if cpu == "" {
|
|
cpu = defaultCPULimit
|
|
}
|
|
if mem == "" {
|
|
mem = defaultMemLimit
|
|
}
|
|
cpuQty, err := resource.ParseQuantity(cpu)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("backup: invalid cpu limit %q: %w", cpu, err)
|
|
}
|
|
memQty, err := resource.ParseQuantity(mem)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("backup: invalid memory limit %q: %w", mem, err)
|
|
}
|
|
return corev1.ResourceList{
|
|
corev1.ResourceCPU: cpuQty,
|
|
corev1.ResourceMemory: memQty,
|
|
}, nil
|
|
}
|
|
|
|
func boolPtr(b bool) *bool { return &b }
|
|
func int32Ptr(i int32) *int32 { return &i }
|
|
func int64Ptr(i int64) *int64 { return &i }
|
|
|
|
// backupPodSecurityContext pins the Pod identity. RunAsNonRoot is false because
|
|
// the default identity is root: worlds are owned by the game uid (or root, for a
|
|
// world an older release wrote), and Paper writes mode-0600 files any other
|
|
// non-root reader cannot open. FSGroup stays unset unless configured — a root executor must not
|
|
// needlessly chgrp the world volume.
|
|
func backupPodSecurityContext(p JobParams) *corev1.PodSecurityContext {
|
|
sc := &corev1.PodSecurityContext{
|
|
RunAsNonRoot: boolPtr(false),
|
|
RunAsUser: int64Ptr(p.RunAsUser),
|
|
RunAsGroup: int64Ptr(p.RunAsGroup),
|
|
}
|
|
if p.FSGroup > 0 {
|
|
sc.FSGroup = int64Ptr(p.FSGroup)
|
|
}
|
|
return sc
|
|
}
|