Files
Felis/internal/fileedit/jobspec.go
T

289 lines
11 KiB
Go

package fileedit
import (
"encoding/base64"
"fmt"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
// Label keys applied to file-editor objects, mirroring internal/restore and
// internal/backupjob so all three world-touching executors are observable the same
// way. LabelOpID is the one addition: it is how felis-api finds THIS invocation's
// Pod among any others, which matters here in a way it does not for restore —
// file operations are interactive and repeated, so several may be in flight or
// lingering inside their TTL at once. LabelMode carries the operation so the
// world-volume lock (internal/maintenance) can tell a write, which holds the
// volume, from a read or listing, which does not.
const (
LabelManagedBy = "app.kubernetes.io/managed-by"
LabelComponent = "app.kubernetes.io/component"
LabelServer = "felis.lolicon.best/server"
LabelOpID = "felis.lolicon.best/files-op"
LabelMode = "felis.lolicon.best/files-mode"
managedByValue = "felis-files"
componentValue = "world-files"
worldVolume = "world"
felisBinaryPath = "/usr/local/bin/felis"
containerName = "files"
)
// JobParams are the rendered inputs to a file-editor Job, derived from a server +
// operation + Config by the Editor. jobspec is a pure function of them so the
// security-critical Job shape is unit-tested without a cluster.
type JobParams struct {
Server string
// OpID is the per-invocation identifier that both names the Job and labels its
// Pod. See Editor.run for why every invocation gets a fresh one.
OpID string
Op string
Path string
Content []byte // OpWrite only
// Expect is a write's precondition hash (see Execute); empty writes
// unconditionally.
Expect string
WorldPVC string
Namespace string
ServiceAccount string
Image string
WorldsRoot string
Deadline time.Duration
CPULimit string
MemLimit string
RunAsUser int64
RunAsGroup int64
FSGroup int64
TTLAfterFinished time.Duration
}
// FilesJobName is the Job name for one file operation. Unlike RestoreJobName it is
// NOT a pure function of the server: it carries the per-invocation OpID.
//
// That difference is forced by RBAC and is the single most load-bearing decision in
// this package. felis-api holds `jobs: create` in the minecraft namespace and
// NOTHING else — no jobs:get, no jobs:delete (internal/platform.APIMinecraftRole).
// So it can neither poll a Job nor clean one up; ttlSecondsAfterFinished is the only
// reclamation. With a deterministic name the FIRST file operation would leave a
// completed Job squatting the name for the whole TTL window, and every subsequent
// operation would collide with it — and, unable to delete it, the editor would be
// wedged until the TTL expired. A restore can accept that (it runs once per
// incident); an editor cannot (browse a directory, open a file, save it — three
// operations in as many seconds). internal/backupjob reached the same conclusion for
// the same reason.
func FilesJobName(server, opID string) string { return "files-" + server + "-" + opID }
func filesLabels(p JobParams) map[string]string {
return map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
LabelServer: p.Server,
LabelOpID: p.OpID,
LabelMode: p.Op,
}
}
// FilesJob renders the file-editor Job. Its isolation is the strictest of the three
// world executors — a strict SUBSET of what a restore Pod gets — and every guarantee
// is asserted by jobspec_test.go, because no cluster runs in this environment:
//
// - runs under the weak felis-restore SA (never the felis-api SA) with its token
// auto-mount disabled, so it cannot reach the K8s API (spec §16, §21). It reuses
// that bare, Role-less SA for the same reason internal/backupjob does: this Pod
// needs no K8s API access at all, so a second identity with the same (empty)
// powers would be a manifest to maintain for no isolation gain;
// - mounts EXACTLY ONE volume — the world PVC — and NO Secret, NO ConfigMap, and
// NO backup PVC. It is therefore strictly blinder than the backup Pod, which
// mounts the config Secret to self-record its row: a file-editor Pod has nothing
// to record, so it is handed no database URL and no credential of any kind (the
// four-power red line, spec §22);
// - mounts that one volume READ-ONLY for list and read. Only a write needs to
// mutate the world, so two of the three operations physically cannot — the
// kernel refuses, not merely the code. This is why readOnly is derived from the
// op rather than fixed;
// - runs as a non-root, fixed uid/gid with an fsGroup matching the operator's
// StatefulSet, so a file this Pod writes is owned by the identity the minecraft
// server later runs as — a config file the server cannot read would be worse
// than no edit at all;
// - no privilege, no privilege escalation, read-only root filesystem, drop ALL
// capabilities. The world mount is the only writable path, and only on a write;
// - activeDeadlineSeconds + backoffLimit=0 so a wedged mount cannot loop or hang
// forever; ttlSecondsAfterFinished GCs the finished Job, which — see
// FilesJobName — is the ONLY cleanup available to felis-api.
//
// The container runs `/usr/local/bin/felis files` (cmd/felis), which performs the
// operation under os.Root containment and prints the marked JSON Result line that
// felis-api reads back through pods/log.
func FilesJob(p JobParams) (*batchv1.Job, error) {
if p.Image == "" {
return nil, fmt.Errorf("fileedit: image is empty")
}
if p.WorldPVC == "" {
return nil, fmt.Errorf("fileedit: world PVC name is required")
}
if p.OpID == "" {
return nil, fmt.Errorf("fileedit: op id is required")
}
if p.Op != OpList && p.Op != OpRead && p.Op != OpWrite {
return nil, fmt.Errorf("fileedit: unknown op %q", p.Op)
}
if len(p.Content) > MaxWriteBytes {
return nil, fmt.Errorf("fileedit: content is %d bytes, over the %d limit", len(p.Content), MaxWriteBytes)
}
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
if err != nil {
return nil, err
}
deadline := int64(p.Deadline / time.Second)
if deadline <= 0 {
deadline = int64(defaultDeadline / time.Second)
}
ttl := int32(p.TTLAfterFinished / time.Second)
if ttl <= 0 {
ttl = int32(defaultTTL / time.Second)
}
// Only a write may mutate the world. Mounting read-only for the other two ops
// makes "a listing cannot damage a world" a kernel guarantee rather than a
// code-review one.
readOnlyWorld := p.Op != OpWrite
args := []string{
"--op", p.Op,
"--path", p.Path,
"--worlds-root", p.WorldsRoot,
}
// The expected hash is a digest of content the caller already holds, not a
// secret, so it rides argv; only the content itself needs the env channel.
if p.Op == OpWrite && p.Expect != "" {
args = append(args, "--expect-sha256", p.Expect)
}
container := corev1.Container{
Name: containerName,
Image: p.Image,
Command: []string{felisBinaryPath, "files"},
Args: args,
VolumeMounts: []corev1.VolumeMount{
{Name: worldVolume, MountPath: p.WorldsRoot, ReadOnly: readOnlyWorld},
},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
ReadOnlyRootFilesystem: boolPtr(true),
Capabilities: &corev1.Capabilities{
Drop: []corev1.Capability{"ALL"},
Add: filesCapabilities(p.Op),
},
},
}
// New content rides the Job spec as a base64 env var. felis-api cannot create a
// Secret (it holds secrets:get only), so the spec is the sole channel into the
// Pod; base64 keeps arbitrary bytes — CRLF line endings, a UTF-8 BOM, a binary
// blob — intact through a field that must be a valid string. The env var is set
// ONLY for a write, so a list/read Job spec carries no caller content at all.
if p.Op == OpWrite {
container.Env = []corev1.EnvVar{{
Name: ContentEnv,
Value: base64.StdEncoding.EncodeToString(p.Content),
}}
}
job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{
Name: FilesJobName(p.Server, p.OpID),
Namespace: p.Namespace,
Labels: filesLabels(p),
},
Spec: batchv1.JobSpec{
// One shot: a file operation that failed must surface its failure, not be
// retried behind the caller's back — a retried write is a second write.
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(deadline),
TTLSecondsAfterFinished: int32Ptr(ttl),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: filesLabels(p)},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
SecurityContext: filesPodSecurityContext(p),
Containers: []corev1.Container{container},
Volumes: []corev1.Volume{{
Name: worldVolume,
VolumeSource: corev1.VolumeSource{
PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{
ClaimName: p.WorldPVC,
ReadOnly: readOnlyWorld,
},
},
}},
},
},
},
}
return job, nil
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
cpu = defaultCPULimit
}
if mem == "" {
mem = defaultMemLimit
}
cpuQty, err := resource.ParseQuantity(cpu)
if err != nil {
return nil, fmt.Errorf("fileedit: invalid cpu limit %q: %w", cpu, err)
}
memQty, err := resource.ParseQuantity(mem)
if err != nil {
return nil, fmt.Errorf("fileedit: invalid memory limit %q: %w", mem, err)
}
return corev1.ResourceList{
corev1.ResourceCPU: cpuQty,
corev1.ResourceMemory: memQty,
}, nil
}
func boolPtr(b bool) *bool { return &b }
func int32Ptr(i int32) *int32 { return &i }
func int64Ptr(i int64) *int64 { return &i }
// filesCapabilities is what the root executor keeps after dropping ALL (see
// Config.RunAsUser). DAC_OVERRIDE opens a mode-0600 file (level.dat) the game wrote
// as its own uid, which a fixed non-root uid could not. A write also keeps CHOWN so
// the file it creates can be handed to naming.GameUID (exec.go ownWritten); a list
// or read changes nothing and gets no more than it needs.
func filesCapabilities(op string) []corev1.Capability {
if op == OpWrite {
return []corev1.Capability{"CHOWN", "DAC_OVERRIDE"}
}
return []corev1.Capability{"DAC_OVERRIDE"}
}
// filesPodSecurityContext pins the Pod identity. Root by default: the world volume
// belongs to the game uid (naming.GameUID), and root with DAC_OVERRIDE reaches its
// mode-0600 files as well as any a previous root-run release left behind. FSGroup is
// only rendered when configured so a root executor never chgrps the volume.
func filesPodSecurityContext(p JobParams) *corev1.PodSecurityContext {
sc := &corev1.PodSecurityContext{
RunAsNonRoot: boolPtr(false),
RunAsUser: int64Ptr(p.RunAsUser),
RunAsGroup: int64Ptr(p.RunAsGroup),
}
if p.FSGroup > 0 {
sc.FSGroup = int64Ptr(p.FSGroup)
}
return sc
}