Give an owner a way to repair the one failure no other endpoint covers: a
server that will not boot because a single line of server.properties or a
plugin's YAML is wrong. Until now that needed a human with cluster access.
felis-api cannot touch a world in-process — the world PVC is ReadWriteOnce
and its lifecycle belongs to the operator's StatefulSet — so the work runs
as a one-shot Job, and the server must be stopped first because a running
one holds the volume. That is the same constraint that shapes restore and
backup, and the handlers enforce the stopped gate the same way.
What is different is that the caller wants the OUTPUT, not just the side
effect. The Job prints its result to stdout and felis-api reads it back
through the pods/log subresource, which needs no permission felis-api does
not already hold: jobs:create, pods:list, pods/log:get. No pods/exec, no
pods/portforward, not even pods:get. The price is latency — every operation
is a Pod schedule — which is why this is a repair tool and not a file
manager.
Containment is structural, not textual. Every filesystem access goes through
os.Root, the stdlib's escape-proof directory handle, which resolves each
component against the open root descriptor and refuses "..", absolute paths,
and symlinks leading outside. The string-prefix check used elsewhere is not
reused here: it validates a path as text and then opens it as a path, and a
world directory holds attacker-influenced content, so a symlink swapped in
between those two steps is a live threat rather than a theoretical one.
os.Root has no such window because the check and the open are one operation.
The Job's isolation is a strict subset of a restore Pod's: the weak
felis-restore SA with its token auto-mount disabled, exactly one volume (the
world PVC, mounted read-only for list and read so two of the three
operations cannot mutate anything), no Secret, no ConfigMap, no database
URL, non-root with an fsGroup matching the operator's so a written file is
readable by the server that later mounts it, and backoffLimit 0 so a failed
write is never silently retried as a second write.
Two limits on the surface are worth stating plainly, because the mount is
the server's whole working directory rather than a config subtree:
* A write accepts arbitrary bytes at any path, so an owner can place a
loadable plugin jar. This is deliberate — it is what a hosting panel's
file manager does, scoped to a server the caller already owns and
already drives through /command — but it is the one owner-tier route
that lands executable code in a backend pod, since images are
admin-only and modpack submissions need an admin verdict.
* config/paper-global.yml is refused on read. felis-lobby's entrypoint
writes FELIS_FORWARDING_SECRET into it on every boot, and that value is
identical on every backend, so reading it from a server you own would
hand you the handshake key for everyone else's. It is the only path in
the mount that is not the caller's own data, and therefore the only
denial. The comparison is on the cleaned path, or ./config/... would
walk straight through it.
Writing that file is still allowed: it leaks nothing, and the entrypoint
rewrites it whole on every boot regardless.
The write body's content field is a *[]byte rather than a []byte for the
reason permissionRequest.Value is a *bool — a plain slice makes absent,
null, and empty indistinguishable, so a body of {} would decode to nil and
truncate the target to zero bytes while answering 200, destroying the very
config the caller opened the editor to repair.
260 lines
9.1 KiB
Go
260 lines
9.1 KiB
Go
package fileedit
|
|
|
|
import (
|
|
"encoding/base64"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
corev1 "k8s.io/api/core/v1"
|
|
)
|
|
|
|
func testParams(op string) JobParams {
|
|
return JobParams{
|
|
Server: "survival",
|
|
OpID: "deadbeefcafe0001",
|
|
Op: op,
|
|
Path: "server.properties",
|
|
WorldPVC: "world-survival-0",
|
|
Namespace: "minecraft",
|
|
ServiceAccount: "felis-restore",
|
|
Image: "registry.example/felis:v1",
|
|
WorldsRoot: "/data",
|
|
Deadline: 2 * time.Minute,
|
|
CPULimit: "500m",
|
|
MemLimit: "256Mi",
|
|
RunAsUser: 1000,
|
|
RunAsGroup: 1000,
|
|
FSGroup: 1000,
|
|
TTLAfterFinished: 2 * time.Minute,
|
|
}
|
|
}
|
|
|
|
// TestFilesJobIsolation asserts every isolation guarantee FilesJob documents. No
|
|
// cluster runs in this environment, so this pure-function test IS the enforcement:
|
|
// if someone loosens the Pod spec, this is what catches it.
|
|
func TestFilesJobIsolation(t *testing.T) {
|
|
job, err := FilesJob(testParams(OpRead))
|
|
if err != nil {
|
|
t.Fatalf("FilesJob: %v", err)
|
|
}
|
|
spec := job.Spec.Template.Spec
|
|
|
|
t.Run("runs under the weak SA with its token un-mounted", func(t *testing.T) {
|
|
if spec.ServiceAccountName != "felis-restore" {
|
|
t.Fatalf("SA = %q, want the weak felis-restore", spec.ServiceAccountName)
|
|
}
|
|
if spec.AutomountServiceAccountToken == nil || *spec.AutomountServiceAccountToken {
|
|
t.Fatal("the SA token MUST NOT be auto-mounted — the Pod must not reach the K8s API")
|
|
}
|
|
})
|
|
|
|
// The four-power red line: a file-editor Pod holds no credential of any kind. It
|
|
// is strictly blinder than the backup Pod, which does mount the config Secret.
|
|
t.Run("mounts exactly one volume and no credential", func(t *testing.T) {
|
|
if len(spec.Volumes) != 1 {
|
|
t.Fatalf("volumes = %d, want exactly 1 (the world PVC)", len(spec.Volumes))
|
|
}
|
|
v := spec.Volumes[0]
|
|
if v.PersistentVolumeClaim == nil || v.PersistentVolumeClaim.ClaimName != "world-survival-0" {
|
|
t.Fatalf("the sole volume must be the world PVC, got %+v", v)
|
|
}
|
|
if v.Secret != nil || v.ConfigMap != nil || v.Projected != nil {
|
|
t.Fatalf("no Secret/ConfigMap/Projected volume may be mounted, got %+v", v)
|
|
}
|
|
})
|
|
|
|
t.Run("runs non-root with the operator's runtime identity", func(t *testing.T) {
|
|
sc := spec.SecurityContext
|
|
if sc == nil || sc.RunAsNonRoot == nil || !*sc.RunAsNonRoot {
|
|
t.Fatal("RunAsNonRoot must be true")
|
|
}
|
|
// FSGroup must match the minecraft server's group or a file this Pod writes
|
|
// would be unreadable by the server that later mounts the same volume.
|
|
if sc.RunAsUser == nil || *sc.RunAsUser != 1000 ||
|
|
sc.RunAsGroup == nil || *sc.RunAsGroup != 1000 ||
|
|
sc.FSGroup == nil || *sc.FSGroup != 1000 {
|
|
t.Fatalf("uid/gid/fsGroup must all be 1000, got %+v", sc)
|
|
}
|
|
})
|
|
|
|
t.Run("container drops every privilege", func(t *testing.T) {
|
|
if len(spec.Containers) != 1 {
|
|
t.Fatalf("containers = %d, want 1", len(spec.Containers))
|
|
}
|
|
sc := spec.Containers[0].SecurityContext
|
|
if sc == nil {
|
|
t.Fatal("the container needs a SecurityContext")
|
|
}
|
|
if sc.Privileged == nil || *sc.Privileged {
|
|
t.Fatal("Privileged must be false")
|
|
}
|
|
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
|
|
t.Fatal("AllowPrivilegeEscalation must be false")
|
|
}
|
|
if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
|
|
t.Fatal("ReadOnlyRootFilesystem must be true")
|
|
}
|
|
if sc.Capabilities == nil || len(sc.Capabilities.Drop) != 1 || sc.Capabilities.Drop[0] != "ALL" {
|
|
t.Fatalf("capabilities must drop ALL, got %+v", sc.Capabilities)
|
|
}
|
|
})
|
|
|
|
t.Run("is one-shot, deadlined, and self-collecting", func(t *testing.T) {
|
|
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
|
|
t.Fatal("BackoffLimit must be 0 — a retried write is a second write")
|
|
}
|
|
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds != 120 {
|
|
t.Fatalf("ActiveDeadlineSeconds = %v, want 120", job.Spec.ActiveDeadlineSeconds)
|
|
}
|
|
// The TTL is the ONLY cleanup available: felis-api holds no jobs:delete.
|
|
if job.Spec.TTLSecondsAfterFinished == nil || *job.Spec.TTLSecondsAfterFinished != 120 {
|
|
t.Fatalf("TTLSecondsAfterFinished = %v, want 120", job.Spec.TTLSecondsAfterFinished)
|
|
}
|
|
if spec.RestartPolicy != corev1.RestartPolicyNever {
|
|
t.Fatalf("RestartPolicy = %q, want Never", spec.RestartPolicy)
|
|
}
|
|
})
|
|
|
|
t.Run("runs the files entrypoint with the op as arguments", func(t *testing.T) {
|
|
c := spec.Containers[0]
|
|
if len(c.Command) != 2 || c.Command[0] != felisBinaryPath || c.Command[1] != "files" {
|
|
t.Fatalf("command = %v, want [%s files]", c.Command, felisBinaryPath)
|
|
}
|
|
args := strings.Join(c.Args, " ")
|
|
for _, want := range []string{"--op read", "--path server.properties", "--worlds-root /data"} {
|
|
if !strings.Contains(args, want) {
|
|
t.Fatalf("args %q missing %q", args, want)
|
|
}
|
|
}
|
|
})
|
|
}
|
|
|
|
// TestFilesJobWorldMountIsReadOnlyExceptForWrite pins the guarantee that only a
|
|
// write can mutate a world. For list and read the kernel refuses the write, not
|
|
// merely the code — a defence that survives a bug in the entrypoint.
|
|
func TestFilesJobWorldMountIsReadOnlyExceptForWrite(t *testing.T) {
|
|
cases := []struct {
|
|
op string
|
|
wantReadOnly bool
|
|
}{
|
|
{OpList, true},
|
|
{OpRead, true},
|
|
{OpWrite, false},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.op, func(t *testing.T) {
|
|
job, err := FilesJob(testParams(tc.op))
|
|
if err != nil {
|
|
t.Fatalf("FilesJob: %v", err)
|
|
}
|
|
spec := job.Spec.Template.Spec
|
|
gotMount := spec.Containers[0].VolumeMounts[0].ReadOnly
|
|
gotVol := spec.Volumes[0].PersistentVolumeClaim.ReadOnly
|
|
if gotMount != tc.wantReadOnly || gotVol != tc.wantReadOnly {
|
|
t.Fatalf("op %s: mount.readOnly=%v volume.readOnly=%v, want %v",
|
|
tc.op, gotMount, gotVol, tc.wantReadOnly)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestFilesJobContentEnv pins the write channel: content rides the Job spec
|
|
// base64-encoded, and ONLY for a write — a list or read Job spec must carry no
|
|
// caller content at all.
|
|
func TestFilesJobContentEnv(t *testing.T) {
|
|
t.Run("write carries base64 content", func(t *testing.T) {
|
|
p := testParams(OpWrite)
|
|
p.Content = []byte("motd=hello\n\x00\xff")
|
|
job, err := FilesJob(p)
|
|
if err != nil {
|
|
t.Fatalf("FilesJob: %v", err)
|
|
}
|
|
env := job.Spec.Template.Spec.Containers[0].Env
|
|
if len(env) != 1 || env[0].Name != ContentEnv {
|
|
t.Fatalf("env = %+v, want exactly %s", env, ContentEnv)
|
|
}
|
|
got, err := base64.StdEncoding.DecodeString(env[0].Value)
|
|
if err != nil {
|
|
t.Fatalf("env value is not base64: %v", err)
|
|
}
|
|
if string(got) != string(p.Content) {
|
|
t.Fatalf("decoded %q, want %q — arbitrary bytes must survive", got, p.Content)
|
|
}
|
|
// The content must never leak into argv, which is world-readable on the node.
|
|
if strings.Contains(strings.Join(job.Spec.Template.Spec.Containers[0].Args, " "), "motd=hello") {
|
|
t.Fatal("content must not appear in the container arguments")
|
|
}
|
|
})
|
|
|
|
for _, op := range []string{OpList, OpRead} {
|
|
t.Run(op+" carries no content env", func(t *testing.T) {
|
|
job, err := FilesJob(testParams(op))
|
|
if err != nil {
|
|
t.Fatalf("FilesJob: %v", err)
|
|
}
|
|
if env := job.Spec.Template.Spec.Containers[0].Env; len(env) != 0 {
|
|
t.Fatalf("env = %+v, want none for a %s", env, op)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestFilesJobNameIsPerInvocation is the RBAC-forced property documented on
|
|
// FilesJobName. felis-api holds jobs:create and NOTHING else — no jobs:delete — so
|
|
// a deterministic name would let the first completed Job squat it for a whole TTL
|
|
// window and wedge every subsequent operation. Two operations on the same server
|
|
// must therefore never collide.
|
|
func TestFilesJobNameIsPerInvocation(t *testing.T) {
|
|
a := testParams(OpRead)
|
|
b := testParams(OpRead)
|
|
b.OpID = "deadbeefcafe0002"
|
|
|
|
ja, err := FilesJob(a)
|
|
if err != nil {
|
|
t.Fatalf("FilesJob: %v", err)
|
|
}
|
|
jb, err := FilesJob(b)
|
|
if err != nil {
|
|
t.Fatalf("FilesJob: %v", err)
|
|
}
|
|
if ja.Name == jb.Name {
|
|
t.Fatalf("two operations on one server share the Job name %q — the editor would wedge", ja.Name)
|
|
}
|
|
if !strings.Contains(ja.Name, "survival") || !strings.Contains(ja.Name, a.OpID) {
|
|
t.Fatalf("job name %q should carry the server and the op id", ja.Name)
|
|
}
|
|
// The op id must also label the Pod, or the runner could not select THIS
|
|
// operation's Pod to read its result from.
|
|
if got := ja.Spec.Template.ObjectMeta.Labels[LabelOpID]; got != a.OpID {
|
|
t.Fatalf("pod label %s = %q, want %q", LabelOpID, got, a.OpID)
|
|
}
|
|
}
|
|
|
|
// TestFilesJobRejectsBadParams checks the renderer fails loudly rather than
|
|
// producing a Job that cannot run or that would be refused by etcd.
|
|
func TestFilesJobRejectsBadParams(t *testing.T) {
|
|
cases := []struct {
|
|
name string
|
|
mutate func(*JobParams)
|
|
}{
|
|
{"no image", func(p *JobParams) { p.Image = "" }},
|
|
{"no world PVC", func(p *JobParams) { p.WorldPVC = "" }},
|
|
{"no op id", func(p *JobParams) { p.OpID = "" }},
|
|
{"unknown op", func(p *JobParams) { p.Op = "delete" }},
|
|
{"oversized content", func(p *JobParams) {
|
|
p.Op, p.Content = OpWrite, make([]byte, MaxWriteBytes+1)
|
|
}},
|
|
{"bad cpu limit", func(p *JobParams) { p.CPULimit = "half" }},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
p := testParams(OpRead)
|
|
tc.mutate(&p)
|
|
if _, err := FilesJob(p); err == nil {
|
|
t.Fatal("expected an error")
|
|
}
|
|
})
|
|
}
|
|
}
|