diff --git a/cmd/felis/api.go b/cmd/felis/api.go index c495721..9526e5c 100644 --- a/cmd/felis/api.go +++ b/cmd/felis/api.go @@ -9,6 +9,7 @@ import ( "os" "regexp" "strings" + "sync/atomic" "time" "felis.lolicon.best/internal/api" @@ -152,11 +153,13 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int { // The fetch initContainer runs THIS image's fetch-context entrypoint, so the // build config carries the api's own image (the platform sets FELIS_IMAGE). buildCfg.FelisImage = os.Getenv("FELIS_IMAGE") + buildJobs := build.NewK8sJobs(cl, buildCfg) builder := &build.Builder{ Store: build.NewPGStore(drv.DB()), - Jobs: build.NewK8sJobs(cl, buildCfg), + Jobs: buildJobs, Config: buildCfg, } + go probeBuildUserNamespaces(ctx, buildJobs, buildCfg, stderr) // User-modpack approval lane (user-directed extension over §16; see // internal/submit). An ordinary user may only SUBMIT a @@ -457,6 +460,11 @@ func buildConfig(cfg *config.Config) build.Config { TrivyImage: cfg.Registry.TrivyImage, CPULimit: cfg.Registry.BuildCPULimit, MemLimit: cfg.Registry.BuildMemLimit, + DiskLimit: cfg.Registry.BuildDiskLimit, + // "auto" follows the startup probe (see probeBuildUserNamespaces). + UserNamespaces: cfg.Registry.BuildUserNamespaces, + UserNamespacesProbe: new(atomic.Bool), + RuntimeClass: cfg.Registry.BuildRuntimeClass, // Empty keeps Trivy's own default; an install with builds points this at // the internal DB mirror (see config.RegistryConfig.TrivyDBRepository). TrivyDBRepository: cfg.Registry.TrivyDBRepository, @@ -464,6 +472,29 @@ func buildConfig(cfg *config.Config) build.Config { } } +// probeBuildUserNamespaces settles build_user_namespaces = "auto": one probe +// pod with hostUsers: false tells whether this node's kernel and runtime can run +// build pods in a user namespace. Builds submitted before it answers run without. +func probeBuildUserNamespaces(ctx context.Context, jobs *build.K8sJobs, cfg build.Config, stderr io.Writer) { + if mode := cfg.UserNamespaces; mode != "" && mode != build.UserNamespacesAuto { + return + } + if cfg.FelisImage == "" { + fmt.Fprintln(stderr, "felis api: FELIS_IMAGE unset — build pods run without a user namespace") + return + } + ok, err := jobs.ProbeUserNamespaces(ctx, cfg.FelisImage) + cfg.UserNamespacesProbe.Store(ok) + switch { + case ok: + fmt.Fprintln(stderr, "felis api: build pods run in a user namespace (hostUsers: false)") + case err != nil: + fmt.Fprintf(stderr, "felis api: build pods run without a user namespace: the probe failed: %v\n", err) + default: + fmt.Fprintln(stderr, "felis api: build pods run without a user namespace: this node cannot start a pod with hostUsers: false") + } +} + // internalAPIBaseURL resolves the platform's internal-face base URL: the address // the platform rendered into this pod (felis API base URL env), or — for a // hand-rolled deployment that set none — the platform default control namespace, diff --git a/cmd/felis/egressgate.go b/cmd/felis/egressgate.go new file mode 100644 index 0000000..b92e899 --- /dev/null +++ b/cmd/felis/egressgate.go @@ -0,0 +1,66 @@ +package main + +import ( + "flag" + "fmt" + "io" + "net" + "os" + "time" +) + +// Vars so tests can shrink them. A dial that neither connects nor is refused +// within egressDialTimeout counts as blocked: a policy that drops packets looks +// exactly like that. +var ( + egressDialTimeout = 500 * time.Millisecond + egressPollInterval = 200 * time.Millisecond +) + +// cmdEgressGate is the first initContainer of every build pod. The pod's +// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s: +// a build-labelled pod reached the internet and the Kubernetes API for its first +// ~0.7 s), so the gate dials a destination the policy denies until it stops +// answering, and only then lets the pod's next container, eventually the +// untrusted Dockerfile, start. +// +// The default probe is the Kubernetes API Service, which the kubelet names in +// every pod's environment and the build policy never admits. A probe that still +// answers after --wait means the policy is not enforced at all (a CNI without +// NetworkPolicy support, or k3s run with --disable-network-policy), and the +// build fails closed. +func cmdEgressGate(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError) + fs.SetOutput(stderr) + probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)") + wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused") + if err := fs.Parse(args); err != nil { + return 2 + } + if *probe == "" { + host, port := os.Getenv("KUBERNETES_SERVICE_HOST"), os.Getenv("KUBERNETES_SERVICE_PORT") + if host == "" || port == "" { + fmt.Fprintln(stderr, "felis egress-gate: no --probe and no KUBERNETES_SERVICE_HOST/PORT to default to") + return 2 + } + *probe = net.JoinHostPort(host, port) + } + + start := time.Now() + for { + conn, err := net.DialTimeout("tcp", *probe, egressDialTimeout) + if err != nil { + fmt.Fprintf(stdout, "felis egress-gate: %s is unreachable after %s (%v); the egress lock is in effect\n", + *probe, time.Since(start).Round(time.Millisecond), err) + return 0 + } + _ = conn.Close() + if time.Since(start) >= *wait { + fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+ + "(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n", + *probe, *wait) + return 1 + } + time.Sleep(egressPollInterval) + } +} diff --git a/cmd/felis/egressgate_test.go b/cmd/felis/egressgate_test.go new file mode 100644 index 0000000..f1f2a49 --- /dev/null +++ b/cmd/felis/egressgate_test.go @@ -0,0 +1,98 @@ +package main + +import ( + "bytes" + "net" + "strings" + "testing" + "time" +) + +func shrinkEgressGate(t *testing.T) { + t.Helper() + dial, poll := egressDialTimeout, egressPollInterval + egressDialTimeout, egressPollInterval = 200*time.Millisecond, 10*time.Millisecond + t.Cleanup(func() { egressDialTimeout, egressPollInterval = dial, poll }) +} + +// The gate holds while the probe answers and lets the pod go on once the policy +// lands, which the test plays by closing the listener. +func TestEgressGateWaitsForTheLock(t *testing.T) { + shrinkEgressGate(t) + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + accepted := make(chan struct{}, 100) + go func() { + for { + c, err := ln.Accept() + if err != nil { + return + } + _ = c.Close() + accepted <- struct{}{} + } + }() + go func() { + for i := 0; i < 3; i++ { + <-accepted + } + _ = ln.Close() + }() + var out, errb bytes.Buffer + if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "10s"}, &out, &errb); code != 0 { + t.Fatalf("exit %d: %s", code, errb.String()) + } + if !strings.Contains(out.String(), "egress lock is in effect") { + t.Errorf("stdout = %q", out.String()) + } +} + +// A probe that keeps answering means no policy is enforced: the build must not run. +func TestEgressGateRefusesAnOpenNetwork(t *testing.T) { + shrinkEgressGate(t) + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + defer ln.Close() + go func() { + for { + c, err := ln.Accept() + if err != nil { + return + } + _ = c.Close() + } + }() + var out, errb bytes.Buffer + if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "100ms"}, &out, &errb); code != 1 { + t.Fatalf("exit %d, want 1", code) + } + if !strings.Contains(errb.String(), "not enforced") { + t.Errorf("stderr = %q", errb.String()) + } +} + +func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) { + shrinkEgressGate(t) + t.Setenv("KUBERNETES_SERVICE_HOST", "") + t.Setenv("KUBERNETES_SERVICE_PORT", "") + var out, errb bytes.Buffer + if code := cmdEgressGate(nil, &out, &errb); code != 2 { + t.Fatalf("exit %d without a probe, want 2", code) + } + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + host, port, _ := net.SplitHostPort(ln.Addr().String()) + _ = ln.Close() // closed: the lock reads as in effect at once + t.Setenv("KUBERNETES_SERVICE_HOST", host) + t.Setenv("KUBERNETES_SERVICE_PORT", port) + out.Reset() + if code := cmdEgressGate(nil, &out, &errb); code != 0 || !strings.Contains(out.String(), ln.Addr().String()) { + t.Fatalf("exit %d, stdout %q", code, out.String()) + } +} diff --git a/cmd/felis/fetchcontext.go b/cmd/felis/fetchcontext.go index 9c9cf65..ba83cd3 100644 --- a/cmd/felis/fetchcontext.go +++ b/cmd/felis/fetchcontext.go @@ -136,13 +136,27 @@ func fetchContextWithRetry(ctx context.Context, client *http.Client, url, token } } +// maxContextBytes / maxContextEntries bound what one context may expand to. The +// compressed upload is capped at 1 GiB, but gzip turns that into hundreds of GiB +// or millions of empty files, and the emptyDir's 4 GiB sizeLimit is only +// enforced by the kubelet's periodic sweep, after the disk has filled. The byte +// cap matches that sizeLimit; the entry cap is far above any real modpack (a +// large one is a few thousand files) and far below an inode exhaustion. +// +// Vars, not consts, so tests can shrink them. +var ( + maxContextBytes int64 = 4 << 30 + maxContextEntries = 200_000 +) + // extractTarGz streams a gzip'd tarball into root, creating directories as // needed. Every entry is vetted BEFORE anything is written: a path that is // absolute or escapes root (via ".."), a link (symlink or hardlink), or any // special file kind aborts the whole extraction. Refusing rather than skipping is // deliberate — a context that needs one of those constructs is not a context this // transport carries, and silently dropping entries would build from a corpus the -// submitter did not upload. +// submitter did not upload. The whole extraction is also bounded by +// maxContextBytes and maxContextEntries. func extractTarGz(r io.Reader, root string) error { if err := os.MkdirAll(root, 0o755); err != nil { return fmt.Errorf("create context dir: %w", err) @@ -153,6 +167,8 @@ func extractTarGz(r io.Reader, root string) error { } defer zr.Close() tr := tar.NewReader(zr) + var written int64 + entries := 0 for { hdr, err := tr.Next() if errors.Is(err, io.EOF) { @@ -161,6 +177,9 @@ func extractTarGz(r io.Reader, root string) error { if err != nil { return fmt.Errorf("read context tarball: %w", err) } + if entries++; entries > maxContextEntries { + return fmt.Errorf("the build context has more than %d entries", maxContextEntries) + } name := filepath.Clean(hdr.Name) if name == "." { continue @@ -188,10 +207,16 @@ func extractTarGz(r io.Reader, root string) error { if err != nil { return fmt.Errorf("create %q: %w", name, err) } - if _, err := io.Copy(f, tr); err != nil { + n, err := io.Copy(f, io.LimitReader(tr, maxContextBytes-written+1)) + written += n + if err != nil { _ = f.Close() return fmt.Errorf("write %q: %w", name, err) } + if written > maxContextBytes { + _ = f.Close() + return fmt.Errorf("the build context expands past %d bytes", maxContextBytes) + } if err := f.Close(); err != nil { return fmt.Errorf("close %q: %w", name, err) } diff --git a/cmd/felis/fetchcontext_test.go b/cmd/felis/fetchcontext_test.go index 21ba60c..5d1c23c 100644 --- a/cmd/felis/fetchcontext_test.go +++ b/cmd/felis/fetchcontext_test.go @@ -121,6 +121,30 @@ func TestExtractTarGzRefusesEscapes(t *testing.T) { } } +// A context that expands past the byte or entry cap is refused, however small +// it was compressed: gzip bombs and inode floods stop at the cap. +func TestExtractTarGzCapsExpansion(t *testing.T) { + bytesCap, entriesCap := maxContextBytes, maxContextEntries + t.Cleanup(func() { maxContextBytes, maxContextEntries = bytesCap, entriesCap }) + maxContextBytes, maxContextEntries = 1000, 5 + + fits := tgzBody(t, tarEntry{name: "a", body: strings.Repeat("x", 600)}, tarEntry{name: "b", body: strings.Repeat("y", 400)}) + if err := extractTarGz(bytes.NewReader(fits), t.TempDir()); err != nil { + t.Fatalf("a context exactly at the byte cap: %v", err) + } + big := tgzBody(t, tarEntry{name: "a", body: strings.Repeat("x", 600)}, tarEntry{name: "b", body: strings.Repeat("y", 401)}) + if err := extractTarGz(bytes.NewReader(big), t.TempDir()); err == nil || !strings.Contains(err.Error(), "expands past") { + t.Fatalf("one byte over the cap: err = %v", err) + } + var many []tarEntry + for i := 0; i < 6; i++ { + many = append(many, tarEntry{name: "d" + string(rune('0'+i)) + "/", typ: tar.TypeDir}) + } + if err := extractTarGz(bytes.NewReader(tgzBody(t, many...)), t.TempDir()); err == nil || !strings.Contains(err.Error(), "entries") { + t.Fatalf("six entries over a cap of five: err = %v", err) + } +} + // The command end to end: it dials the URL with the bearer token from the // environment, and refuses to run without it (the internal face would 401 // anyway; failing at parse time is the honest earlier error). diff --git a/cmd/felis/run.go b/cmd/felis/run.go index 11b622d..221180e 100644 --- a/cmd/felis/run.go +++ b/cmd/felis/run.go @@ -21,6 +21,7 @@ Commands: restore Extract a world archive into a world volume (internal Job entrypoint) backup Archive a world into the backup store and record it (internal Job entrypoint) files List/read/write one file in a stopped server's world (internal Job entrypoint) + egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint) fetch-context Fetch and extract a submission's build context (internal Job entrypoint) push-image Push a scanned image tarball to the registry (internal Job entrypoint) registry-gate Authorize registry writes in front of registry:2 (internal sidecar entrypoint) @@ -56,6 +57,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{ "restore": cmdRestore, "backup": cmdBackup, "files": cmdFiles, + "egress-gate": cmdEgressGate, "fetch-context": cmdFetchContext, "push-image": cmdPushImage, "registry-gate": cmdRegistryGate, diff --git a/deploy/bootstrap.sh b/deploy/bootstrap.sh index 2932953..181d5e6 100644 --- a/deploy/bootstrap.sh +++ b/deploy/bootstrap.sh @@ -2352,7 +2352,7 @@ persisted_registry_block() { out="$(awk ' /^[[:space:]]*\[/ { sect = $0; next } sect ~ /^[[:space:]]*\[registry\][[:space:]]*$/ && - /^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|user_uploads_context)[[:space:]]*=/ { print } + /^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|user_uploads_context)[[:space:]]*=/ { print } sect ~ /^[[:space:]]*\[registry\.s3\][[:space:]]*$/ && /^[[:space:]]*[A-Za-z_]+[[:space:]]*=/ { if (!s3hdr) { printf "[registry.s3]\n"; s3hdr = 1 } print diff --git a/deploy/bootstrap_test.sh b/deploy/bootstrap_test.sh index a4066ef..57816f3 100644 --- a/deploy/bootstrap_test.sh +++ b/deploy/bootstrap_test.sh @@ -1037,6 +1037,9 @@ build_namespace = "stale-ns" kaniko_image = "registry.felis.svc:5000/mirror/kaniko-executor:v1.24.0" trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2" trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1" +build_disk_limit = "20Gi" +build_user_namespaces = "off" +build_runtime_class = "gvisor" [registry.s3] endpoint = "https://s3.example" @@ -1074,6 +1077,9 @@ expect "a re-run carries the trivy vulnerability-DB mirror" \ 'trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"' "$out" expect "a re-run carries the trivy java-DB mirror" \ 'trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"' "$out" +expect "a re-run carries the build disk cap" 'build_disk_limit = "20Gi"' "$out" +expect "a re-run carries the build user-namespace mode" 'build_user_namespaces = "off"' "$out" +expect "a re-run carries the build runtime class" 'build_runtime_class = "gvisor"' "$out" expect "a re-run carries the [registry.s3] uploads subtable" "[registry.s3]" "$out" expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "$out" expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out" diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 62d80dc..b44b056 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -454,9 +454,9 @@ internal registry. `writeBuildError` (JobPhase→Failed). [GO-TESTED for the mapping.] The underlying cause — a kaniko build error, the **Trivy CRITICAL-CVE gate** failing the build (spec §16), or the final push — is in the Job's pod logs and is -[INTEGRATION-ONLY]. The pod runs `kaniko` (builds a tarball, never pushes) and -`trivy` (scans that tarball) as init containers, then `push` — so a CVE-rejected -image never reaches the registry. Inspect every step: +[INTEGRATION-ONLY]. The pod runs `egress-gate` (§8f), `context-fetch`, `kaniko` +(builds a tarball, never pushes) and `trivy` (scans that tarball) as init +containers, then `push` — so a CVE-rejected image never reaches the registry. Inspect every step: ``` kubectl logs -n felis-build job/ --all-containers --prefix @@ -490,6 +490,9 @@ trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2" trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1" build_cpu_limit = "2" build_mem_limit = "4Gi" +build_disk_limit = "12Gi" # §8f +build_user_namespaces = "auto" # §8f: auto | on | off +build_runtime_class = "" # §8f: e.g. "gvisor" ``` Mirror the executor images into the registry once. On the node itself, push @@ -560,6 +563,81 @@ artifacts — every real modpack — and that download fails closed too. Mirror above); the Java DB refreshes far less often than the vulnerability DB, so a one-off mirror is usually fine. +### 8f. Build isolation model and residual risk + +A Dockerfile's `RUN` steps execute inside the kaniko container, as root. Kaniko +is a daemonless image builder; it is **not** a sandbox. What stands between an +approved-but-hostile Dockerfile and the node is the pod around it: + +| Layer | What it does | Where | +|---|---|---| +| Admin approval | Nothing builds until an administrator approves the submission | submit lane | +| Weak identity | `felis-build` SA, no Role anywhere, no token mounted | §8b | +| Egress lock | `felis-build-egress`: cluster DNS, the registry, the api internal face, nothing else | §8c | +| Egress gate | first init container; holds the pod until the lock is enforced for it | `felis egress-gate` | +| Capabilities | every container drops ALL; kaniko gets back only CHOWN, DAC_OVERRIDE, FOWNER to unpack base images | jobspec | +| seccomp | the whole pod runs under the runtime's default profile (no `unshare`, `mount`, `keyctl`, `bpf`, …) | jobspec | +| User namespace | with `build_user_namespaces` on, root in the pod is an unprivileged uid on the node | below | +| Sandbox runtime | optional `build_runtime_class` (gVisor, Kata) | below | +| Credentials | the registry credential lives only in the `push` container; the service token only in `context-fetch` | jobspec | +| Resources | CPU, memory and ephemeral-storage limits per container; `activeDeadlineSeconds`; the context extraction stops at 4 GiB or 200 000 entries | jobspec, `felis fetch-context` | +| Namespace backstop | `felis-build-limits` LimitRange gives any container without limits 1 CPU / 1 GiB / 1 GiB disk | bundle | + +**Egress gate.** The CNI programs a new pod's NetworkPolicy a moment after the +pod starts. On k3s (kube-router), a pod in `felis-build` could reach the internet +and the Kubernetes API for its first ~0.7 s. `egress-gate` dials the Kubernetes +API Service, which the build policy never admits, and exits once it stops +answering. The build pod log shows the wait: + +``` +felis egress-gate: 10.43.0.1:443 is unreachable after 612ms (...); the egress lock is in effect +``` + +If the probe still answers after two minutes the gate exits 1 and the build +fails: `the build namespace's NetworkPolicy is not enforced`. The cluster is +running without NetworkPolicy enforcement (a CNI without it, or k3s started with +`--disable-network-policy`); fix the cluster, not the gate. + +**User namespaces (`build_user_namespaces`).** With `hostUsers: false`, uid 0 +in the build pod maps to an unprivileged uid range on the node, so a container +escape lands as nobody. It needs Kubernetes ≥ 1.33, containerd 2.x, and a kernel +with idmapped mounts on the node filesystem (5.19+ upstream; the RHEL/CentOS +Stream 9 kernels carry the backport). `auto`, the default, lets felis-api decide +at startup: it runs one `userns-probe-*` Job in `felis-build` and turns the +feature on only when that pod ran. The api log says which way it went: + +``` +felis api: build pods run in a user namespace (hostUsers: false) +felis api: build pods run without a user namespace: this node cannot start a pod with hostUsers: false +``` + +`on` forces it (builds then fail to start on a node that cannot do it), `off` +never uses it. + +**Sandbox runtime (`build_runtime_class`).** Naming a RuntimeClass runs build +pods under it, for example gVisor (`runsc`) or Kata. The class must exist +(`kubectl get runtimeclass`), and kaniko must work under it: gVisor needs its +default `overlay2` rootfs, and Kata needs nested virtualization on a VM node. +Leave it empty unless you have installed and tested one. + +**Disk (`build_disk_limit`, default `12Gi`).** This caps kaniko's writable layer +(the unpacked base image) and, as the largest limit in the pod, the pod's total +disk: extracted context, unpacked base image and image tarball together. The +kubelet enforces it by eviction on its housekeeping sweep, so a build that runs +past it is killed within seconds. The build then fails with `Evicted` in +`kubectl -n felis-build describe pod`. Raise it for very large modpacks, and +keep the node's free disk above it. + +**Residual risk.** Without a user namespace or a sandbox runtime, the build runs +as root in a container on the same kernel as the game servers and the control +plane. Seccomp and the dropped capabilities remove the common escape primitives. +A kernel vulnerability reachable through the remaining syscalls still reaches +the node, and on a single-node install the node is the whole platform. Admin +approval is the control that remains: read the Dockerfile and download the +context before approving. On a node +where the probe comes back negative, a kernel upgrade that brings idmapped +mounts is the cheapest hardening available. + --- ## 9. Registry push/pull failures (spec §15) diff --git a/internal/build/build.go b/internal/build/build.go index 69f55e3..81d704c 100644 --- a/internal/build/build.go +++ b/internal/build/build.go @@ -37,6 +37,7 @@ import ( "errors" "fmt" "strings" + "sync/atomic" "time" "felis.lolicon.best/internal/metrics" @@ -243,6 +244,37 @@ type Config struct { // CPULimit / MemLimit cap each build container (spec §16: resource limits). CPULimit string MemLimit string + // DiskLimit caps the build pod's ephemeral storage: the extracted context, + // the base image kaniko unpacks and the image tarball together. + DiskLimit string + // UserNamespaces selects hostUsers: false for build pods: UserNamespacesOn, + // UserNamespacesOff, or UserNamespacesAuto (the default), which follows + // UserNamespacesProbe. + UserNamespaces string + // UserNamespacesProbe carries ProbeUserNamespaces' verdict to every copy of + // this Config; nil or false keeps "auto" off. + UserNamespacesProbe *atomic.Bool + // RuntimeClass runs build pods under a sandbox RuntimeClass (gVisor, Kata) + // when set. The class must exist on the cluster. + RuntimeClass string +} + +// Values of Config.UserNamespaces. +const ( + UserNamespacesAuto = "auto" + UserNamespacesOn = "on" + UserNamespacesOff = "off" +) + +// userNamespaces resolves Config.UserNamespaces for one build. +func (c Config) userNamespaces() bool { + switch c.UserNamespaces { + case UserNamespacesOn: + return true + case UserNamespacesOff: + return false + } + return c.UserNamespacesProbe != nil && c.UserNamespacesProbe.Load() } // Defaults applied when a Config field is left zero. @@ -255,6 +287,7 @@ const ( defaultMaxDockerfile = 256 * 1024 // 256 KiB defaultCPULimit = "2" defaultMemLimit = "4Gi" + defaultDiskLimit = "12Gi" ) // withDefaults returns a copy of c with zero fields filled, so a partially @@ -284,6 +317,12 @@ func (c Config) withDefaults() Config { if c.MemLimit == "" { c.MemLimit = defaultMemLimit } + if c.DiskLimit == "" { + c.DiskLimit = defaultDiskLimit + } + if c.UserNamespaces == "" { + c.UserNamespaces = UserNamespacesAuto + } return c } @@ -380,6 +419,9 @@ func (b *Builder) jobParams(bld *Build, cfg Config) JobParams { Deadline: cfg.Deadline, CPULimit: cfg.CPULimit, MemLimit: cfg.MemLimit, + DiskLimit: cfg.DiskLimit, + UserNamespaces: cfg.userNamespaces(), + RuntimeClass: cfg.RuntimeClass, } } diff --git a/internal/build/jobspec.go b/internal/build/jobspec.go index 06c7401..a9338ec 100644 --- a/internal/build/jobspec.go +++ b/internal/build/jobspec.go @@ -33,6 +33,7 @@ const ( // build-log streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) // follows the same container this Job defines — one source of truth for the name. const ( + ContainerGate = "egress-gate" ContainerKaniko = "kaniko" ContainerTrivy = "trivy" ContainerPush = "push" @@ -64,6 +65,23 @@ var imageSizeLimit = resource.MustParse("10Gi") // upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room. var contextSizeLimit = resource.MustParse("4Gi") +// Per-container ephemeral-storage bounds (writable layer + logs; emptyDirs count +// toward the pod as a whole). Kaniko's limit is the operator's disk cap because +// kaniko unpacks the base image into its own root filesystem, which no emptyDir +// bound covers; it is also the largest limit in the pod, so it becomes the +// pod-level cap the kubelet holds context + unpacked rootfs + image tarball to. +// Trivy keeps its vulnerability and Java DBs (about 1.4 GiB live) in its layer. +// The others write nothing but logs. +var ( + gateDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("64Mi")} + fetchDisk = diskBounds{request: resource.MustParse("64Mi"), limit: resource.MustParse("256Mi")} + kanikoDiskRq = resource.MustParse("1Gi") + trivyDisk = diskBounds{request: resource.MustParse("256Mi"), limit: resource.MustParse("4Gi")} + pushDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("256Mi")} +) + +type diskBounds struct{ request, limit resource.Quantity } + // buildJobTTL is how long a finished build Job survives before the Job // controller deletes it — and with it the Pod whose kaniko log is the admin // failure-triage surface (GET /api/v1/images/build/{id}/logs). @@ -105,6 +123,15 @@ type JobParams struct { Deadline time.Duration CPULimit string MemLimit string + // DiskLimit caps kaniko's ephemeral storage, and with it the pod's (see + // kanikoDiskRq). Empty applies defaultDiskLimit. + DiskLimit string + // UserNamespaces runs the pod with hostUsers: false, so root in the build + // containers is an unprivileged uid on the node. It needs a kernel and runtime + // with idmapped mounts; Config.UserNamespaces decides. + UserNamespaces bool + // RuntimeClass, when set, runs the pod under that RuntimeClass (gVisor, Kata). + RuntimeClass string } // BuildJobName is the deterministic Job name for a build id. @@ -127,8 +154,14 @@ func buildLabels(p JobParams) map[string]string { // reach the K8s API (spec §16, §21); // - no privileged container — Kaniko builds the Dockerfile without a daemon, // so docker-in-docker / privileged is never needed (spec §16, §22); -// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so -// a runaway or poisoned build cannot exhaust the cluster (spec §16); +// - the RuntimeDefault seccomp profile on the whole pod, and optionally a user +// namespace (hostUsers: false) and a sandbox RuntimeClass, because kaniko is +// no isolation boundary: the Dockerfile's RUN steps execute in its container; +// - an egress gate ahead of everything else, so nothing runs before the +// namespace's NetworkPolicy is enforced for this pod (cmd/felis egress-gate); +// - activeDeadlineSeconds + backoffLimit=0 + per-container CPU, memory and +// ephemeral-storage limits so a runaway or poisoned build cannot exhaust the +// node (spec §16); // - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a // CRITICAL CVE fails the Pod and therefore the Job — the only retained // automatic admission gate (spec §16). @@ -149,6 +182,18 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { if err != nil { return nil, err } + diskCap := p.DiskLimit + if diskCap == "" { + diskCap = defaultDiskLimit + } + kanikoDisk, err := resource.ParseQuantity(diskCap) + if err != nil { + return nil, fmt.Errorf("build: invalid disk limit %q: %w", diskCap, err) + } + kanikoRq := kanikoDiskRq.DeepCopy() + if kanikoDisk.Cmp(kanikoRq) < 0 { + kanikoRq = kanikoDisk.DeepCopy() + } if p.FelisImage == "" { return nil, fmt.Errorf("build: FelisImage is required: the push container runs it") } @@ -187,7 +232,35 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { // credentials) is first fetched into a shared emptyDir; a ref Kaniko can read // in place (s3://, or a path an installer pre-mounted) passes through untouched. contextPath := p.ContextRef - initContainers := []corev1.Container{} + + // The gate runs before anything else. A new pod's NetworkPolicy is programmed + // asynchronously: live on k3s (kube-router), a build-labelled pod reached the + // internet and the Kubernetes API for the first ~0.7 s of its life. The gate + // holds the pod until a destination the policy denies stops answering, so the + // Dockerfile never runs inside that window. + gateSec := sec.DeepCopy() + gateSec.ReadOnlyRootFilesystem = boolPtr(true) + gateSec.RunAsNonRoot = boolPtr(true) + gateSec.RunAsUser = int64Ptr(nonRootUID) + gate := corev1.Container{ + Name: ContainerGate, + Image: p.FelisImage, + Args: []string{"egress-gate"}, + Resources: corev1.ResourceRequirements{ + Limits: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("100m"), + corev1.ResourceMemory: resource.MustParse("64Mi"), + corev1.ResourceEphemeralStorage: gateDisk.limit, + }, + Requests: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("10m"), + corev1.ResourceMemory: resource.MustParse("16Mi"), + corev1.ResourceEphemeralStorage: gateDisk.request, + }, + }, + SecurityContext: gateSec, + } + initContainers := []corev1.Container{gate} imageMount := corev1.VolumeMount{Name: imageVolume, MountPath: imageMountPath} kanikoMounts := []corev1.VolumeMount{imageMount} podVolumes := []corev1.Volume{{ @@ -232,7 +305,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { }}, }}, VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}}, - Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)}, + Resources: withDisk(limits, fetchDisk), SecurityContext: fetchSec, } initContainers = append(initContainers, fetch) @@ -269,7 +342,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { "--skip-tls-verify-pull", }, VolumeMounts: kanikoMounts, - Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)}, + Resources: withDisk(limits, diskBounds{request: kanikoRq, limit: kanikoDisk}), SecurityContext: kanikoSec, } initContainers = append(initContainers, kaniko) @@ -298,7 +371,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { Image: p.TrivyImage, Args: trivyArgs, VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}}, - Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)}, + Resources: withDisk(limits, trivyDisk), SecurityContext: sec, } initContainers = append(initContainers, trivy) @@ -328,7 +401,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { secretEnv("FELIS_REGISTRY_PASSWORD", naming.RegistryPushPasswordKey), }, VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}}, - Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)}, + Resources: withDisk(limits, pushDisk), SecurityContext: pushSec, } @@ -350,16 +423,44 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { RestartPolicy: corev1.RestartPolicyNever, ServiceAccountName: p.ServiceAccount, AutomountServiceAccountToken: boolPtr(false), - InitContainers: initContainers, - Containers: []corev1.Container{push}, - Volumes: podVolumes, + // RUN steps execute in kaniko's container with root and three + // capabilities; RuntimeDefault takes away the syscalls a container + // never needs, among them most kernel-escape primitives (unshare, + // mount, keyctl, bpf). Every step was checked to run under it live. + SecurityContext: &corev1.PodSecurityContext{ + SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault}, + }, + InitContainers: initContainers, + Containers: []corev1.Container{push}, + Volumes: podVolumes, }, }, }, } + if p.UserNamespaces { + job.Spec.Template.Spec.HostUsers = boolPtr(false) + } + if p.RuntimeClass != "" { + rc := p.RuntimeClass + job.Spec.Template.Spec.RuntimeClassName = &rc + } return job, nil } +// nonRootUID is the distroless nonroot user the platform image ships as. +const nonRootUID = 65532 + +// withDisk is the resources block for one build container: the CPU and memory +// caps with their schedulable floor (buildRequests), plus its ephemeral-storage +// request and limit. +func withDisk(limits corev1.ResourceList, d diskBounds) corev1.ResourceRequirements { + lim := limits.DeepCopy() + lim[corev1.ResourceEphemeralStorage] = d.limit + req := buildRequests(limits) + req[corev1.ResourceEphemeralStorage] = d.request + return corev1.ResourceRequirements{Limits: lim, Requests: req} +} + // isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit // lane derives when the API is the blob transport — i.e. a context only the // fetch initContainer can turn into a local path for Kaniko. @@ -516,6 +617,36 @@ func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount { } } +// BuildLimitRange bounds any container in the build namespace that arrives +// without its own limits. Build Jobs set every limit themselves (BuildJob); this +// is the backstop for anything else that lands in the namespace, which shares +// the node's disk with the game worlds. +func BuildLimitRange(namespace string) *corev1.LimitRange { + return &corev1.LimitRange{ + ObjectMeta: metav1.ObjectMeta{ + Name: "felis-build-limits", + Namespace: namespace, + Labels: map[string]string{ + LabelManagedBy: managedByValue, + LabelComponent: componentValue, + }, + }, + Spec: corev1.LimitRangeSpec{Limits: []corev1.LimitRangeItem{{ + Type: corev1.LimitTypeContainer, + Default: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("1"), + corev1.ResourceMemory: resource.MustParse("1Gi"), + corev1.ResourceEphemeralStorage: resource.MustParse("1Gi"), + }, + DefaultRequest: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("100m"), + corev1.ResourceMemory: resource.MustParse("128Mi"), + corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"), + }, + }}}, + } +} + // resourceLimits parses the CPU/memory limits into a ResourceList. func resourceLimits(cpu, mem string) (corev1.ResourceList, error) { if cpu == "" { diff --git a/internal/build/jobspec_test.go b/internal/build/jobspec_test.go index 4e31dd6..df2c3be 100644 --- a/internal/build/jobspec_test.go +++ b/internal/build/jobspec_test.go @@ -171,10 +171,10 @@ func TestBuildJobScansBeforePush(t *testing.T) { t.Fatalf("BuildJob: %v", err) } inits := job.Spec.Template.Spec.InitContainers - if len(inits) != 2 || inits[0].Name != ContainerKaniko || inits[1].Name != ContainerTrivy { - t.Fatalf("initContainers = %v, want [kaniko trivy]", initNames(inits)) + if len(inits) != 3 || inits[0].Name != ContainerGate || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy { + t.Fatalf("initContainers = %v, want [egress-gate kaniko trivy]", initNames(inits)) } - kaniko, trivy := inits[0], inits[1] + kaniko, trivy := inits[1], inits[2] for _, want := range []string{"--destination=" + p.ImageRef, "--no-push", "--tar-path=" + imageTarPath} { if !hasArg(kaniko.Args, want) { t.Errorf("kaniko args = %v, want %s", kaniko.Args, want) @@ -339,9 +339,9 @@ func TestBuildJobTrivyDBRepositoryOverride(t *testing.T) { if err != nil { t.Fatalf("BuildJob: %v", err) } - trivy := job.Spec.Template.Spec.InitContainers[1] + trivy := job.Spec.Template.Spec.InitContainers[2] if trivy.Name != ContainerTrivy { - t.Fatalf("initContainers = %v, want trivy second", initNames(job.Spec.Template.Spec.InitContainers)) + t.Fatalf("initContainers = %v, want trivy third", initNames(job.Spec.Template.Spec.InitContainers)) } if !argPairPresent(trivy.Args, "--db-repository", p.TrivyDBRepository) { t.Errorf("trivy args = %v, want --db-repository %s", trivy.Args, p.TrivyDBRepository) @@ -423,10 +423,11 @@ func TestBuildJobFetchesHTTPContext(t *testing.T) { t.Fatalf("BuildJob: %v", err) } inits := job.Spec.Template.Spec.InitContainers - if len(inits) != 3 || inits[0].Name != ContainerFetch || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy { - t.Fatalf("initContainers = %v, want [%s %s %s]", initNames(inits), ContainerFetch, ContainerKaniko, ContainerTrivy) + if len(inits) != 4 || inits[0].Name != ContainerGate || inits[1].Name != ContainerFetch || + inits[2].Name != ContainerKaniko || inits[3].Name != ContainerTrivy { + t.Fatalf("initContainers = %v, want [%s %s %s %s]", initNames(inits), ContainerGate, ContainerFetch, ContainerKaniko, ContainerTrivy) } - fetch, kaniko := inits[0], inits[1] + fetch, kaniko := inits[1], inits[2] if fetch.Image != p.FelisImage { t.Errorf("fetch image = %q, want the platform image %q", fetch.Image, p.FelisImage) } @@ -494,8 +495,8 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) { if err != nil { t.Fatalf("BuildJob: %v", err) } - if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 2 || inits[0].Name != ContainerKaniko { - t.Errorf("a native ref must render just kaniko + trivy, got %v", initNames(inits)) + if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 3 || inits[1].Name != ContainerKaniko { + t.Errorf("a native ref must render just the gate, kaniko and trivy, got %v", initNames(inits)) } for _, v := range job.Spec.Template.Spec.Volumes { if v.Name == contextVolume { @@ -504,6 +505,111 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) { } } +// Nothing runs before the egress gate, and the gate itself holds nothing: no +// credential, no root, no writable filesystem. +func TestBuildJobGatesEgressFirst(t *testing.T) { + p := sampleJobParams() + p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx" + job, err := BuildJob(p) + if err != nil { + t.Fatalf("BuildJob: %v", err) + } + gate := job.Spec.Template.Spec.InitContainers[0] + if gate.Name != ContainerGate || gate.Image != p.FelisImage || len(gate.Args) != 1 || gate.Args[0] != "egress-gate" { + t.Fatalf("first initContainer = %s %s %v, want the platform image's egress-gate", gate.Name, gate.Image, gate.Args) + } + if len(gate.Env) != 0 || len(gate.VolumeMounts) != 0 { + t.Errorf("the gate must hold nothing, got env %v mounts %v", gate.Env, gate.VolumeMounts) + } + sc := gate.SecurityContext + if sc == nil || sc.RunAsNonRoot == nil || !*sc.RunAsNonRoot || sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem { + t.Errorf("the gate must run non-root on a read-only root, got %#v", sc) + } +} + +// The pod runs under RuntimeDefault seccomp always, and in a user namespace or +// a sandbox runtime when the install asks for them. +func TestBuildJobSandboxing(t *testing.T) { + job, err := BuildJob(sampleJobParams()) + if err != nil { + t.Fatalf("BuildJob: %v", err) + } + spec := job.Spec.Template.Spec + if spec.SecurityContext == nil || spec.SecurityContext.SeccompProfile == nil || + spec.SecurityContext.SeccompProfile.Type != corev1.SeccompProfileTypeRuntimeDefault { + t.Fatalf("pod securityContext = %#v, want seccompProfile RuntimeDefault", spec.SecurityContext) + } + if spec.HostUsers != nil || spec.RuntimeClassName != nil { + t.Errorf("defaults must leave hostUsers and runtimeClassName unset, got %v / %v", spec.HostUsers, spec.RuntimeClassName) + } + p := sampleJobParams() + p.UserNamespaces, p.RuntimeClass = true, "gvisor" + if job, err = BuildJob(p); err != nil { + t.Fatalf("BuildJob: %v", err) + } + spec = job.Spec.Template.Spec + if spec.HostUsers == nil || *spec.HostUsers { + t.Errorf("UserNamespaces must render hostUsers: false, got %v", spec.HostUsers) + } + if spec.RuntimeClassName == nil || *spec.RuntimeClassName != "gvisor" { + t.Errorf("runtimeClassName = %v, want gvisor", spec.RuntimeClassName) + } +} + +// Every container carries an ephemeral-storage request and limit, and kaniko's +// limit, the largest, is the configured disk cap. +func TestBuildJobBoundsEphemeralStorage(t *testing.T) { + p := sampleJobParams() + p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx" + job, err := BuildJob(p) + if err != nil { + t.Fatalf("BuildJob: %v", err) + } + all := append(append([]corev1.Container{}, job.Spec.Template.Spec.InitContainers...), job.Spec.Template.Spec.Containers...) + for _, c := range all { + lim, lok := c.Resources.Limits[corev1.ResourceEphemeralStorage] + req, rok := c.Resources.Requests[corev1.ResourceEphemeralStorage] + if !lok || !rok || lim.IsZero() || req.Cmp(lim) > 0 { + t.Errorf("container %s ephemeral-storage request %v limit %v", c.Name, req.String(), lim.String()) + } + if c.Name == ContainerKaniko && lim.String() != defaultDiskLimit { + t.Errorf("kaniko ephemeral-storage limit = %s, want the default %s", lim.String(), defaultDiskLimit) + } + } + p.DiskLimit = "512Mi" + if job, err = BuildJob(p); err != nil { + t.Fatalf("BuildJob: %v", err) + } + for _, c := range job.Spec.Template.Spec.InitContainers { + if c.Name != ContainerKaniko { + continue + } + lim := c.Resources.Limits[corev1.ResourceEphemeralStorage] + req := c.Resources.Requests[corev1.ResourceEphemeralStorage] + if lim.String() != "512Mi" || req.Cmp(lim) > 0 { + t.Errorf("a 512Mi disk cap rendered limit %s request %s", lim.String(), req.String()) + } + } + p.DiskLimit = "lots" + if _, err := BuildJob(p); err == nil { + t.Error("an unparsable disk limit was accepted") + } +} + +func TestBuildLimitRangeCoversEphemeralStorage(t *testing.T) { + lr := BuildLimitRange("felis-build") + if lr.Namespace != "felis-build" || len(lr.Spec.Limits) != 1 { + t.Fatalf("limit range = %#v", lr) + } + item := lr.Spec.Limits[0] + for _, res := range []corev1.ResourceName{corev1.ResourceCPU, corev1.ResourceMemory, corev1.ResourceEphemeralStorage} { + d, dr := item.Default[res], item.DefaultRequest[res] + if d.IsZero() || dr.IsZero() || dr.Cmp(d) > 0 { + t.Errorf("%s default %s request %s", res, d.String(), dr.String()) + } + } +} + func initNames(cs []corev1.Container) []string { names := make([]string, 0, len(cs)) for _, c := range cs { diff --git a/internal/build/userns.go b/internal/build/userns.go new file mode 100644 index 0000000..bc41914 --- /dev/null +++ b/internal/build/userns.go @@ -0,0 +1,128 @@ +package build + +import ( + "context" + "fmt" + "strconv" + "time" + + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +// Vars so tests can shrink them. The probe pod's image is the api's own, so it is +// already on the node; the timeout covers a slow pod start, and a runtime that +// cannot do user namespaces fails the pod well within it. +var ( + usernsProbeTimeout = 3 * time.Minute + usernsProbePoll = 2 * time.Second +) + +// UsernsProbeJob renders the one-shot Job that asks the cluster whether a build +// pod can run with hostUsers: false. The pod has the parts of a build pod that +// need idmapped mounts and namespaced capabilities: root with kaniko's three +// capabilities, the RuntimeDefault seccomp profile and an emptyDir. It runs +// `felis version`, which touches nothing. +func UsernsProbeJob(namespace, serviceAccount, image, name string) *batchv1.Job { + labels := map[string]string{LabelManagedBy: managedByValue, LabelComponent: "userns-probe"} + small := corev1.ResourceRequirements{ + Limits: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("100m"), + corev1.ResourceMemory: resource.MustParse("64Mi"), + corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"), + }, + Requests: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("10m"), + corev1.ResourceMemory: resource.MustParse("16Mi"), + corev1.ResourceEphemeralStorage: resource.MustParse("16Mi"), + }, + } + return &batchv1.Job{ + ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: namespace, Labels: labels}, + Spec: batchv1.JobSpec{ + BackoffLimit: int32Ptr(0), + ActiveDeadlineSeconds: int64Ptr(int64(usernsProbeTimeout / time.Second)), + TTLSecondsAfterFinished: int32Ptr(300), + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: labels}, + Spec: corev1.PodSpec{ + RestartPolicy: corev1.RestartPolicyNever, + ServiceAccountName: serviceAccount, + AutomountServiceAccountToken: boolPtr(false), + HostUsers: boolPtr(false), + SecurityContext: &corev1.PodSecurityContext{ + SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault}, + }, + Containers: []corev1.Container{{ + Name: "probe", + Image: image, + Args: []string{"version"}, + Resources: small, + SecurityContext: &corev1.SecurityContext{ + Privileged: boolPtr(false), + AllowPrivilegeEscalation: boolPtr(false), + RunAsUser: int64Ptr(0), + Capabilities: &corev1.Capabilities{ + Drop: []corev1.Capability{"ALL"}, + Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE", "FOWNER"}, + }, + }, + VolumeMounts: []corev1.VolumeMount{{Name: "scratch", MountPath: "/scratch"}}, + }}, + Volumes: []corev1.Volume{{ + Name: "scratch", + VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{SizeLimit: quantityPtr(resource.MustParse("16Mi"))}}, + }}, + }, + }, + }, + } +} + +// ProbeUserNamespaces runs UsernsProbeJob and reports whether its pod succeeded. +// A pod that fails, or never starts before the timeout, answers false; err is set +// only when the Job could not be created or read. The Job is deleted afterwards. +func (k *K8sJobs) ProbeUserNamespaces(ctx context.Context, image string) (bool, error) { + name := "userns-probe-" + strconv.FormatInt(time.Now().UnixNano(), 36) + job := UsernsProbeJob(k.cfg.Namespace, k.cfg.ServiceAccount, image, name) + if err := k.c.Create(ctx, job); err != nil { + return false, fmt.Errorf("create the probe job: %w", err) + } + defer func() { + bg := metav1.DeletePropagationBackground + del := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: k.cfg.Namespace, Name: name}} + _ = k.c.Delete(context.WithoutCancel(ctx), del, &client.DeleteOptions{PropagationPolicy: &bg}) + }() + deadline := time.Now().Add(usernsProbeTimeout) + for { + var got batchv1.Job + err := k.c.Get(ctx, types.NamespacedName{Namespace: k.cfg.Namespace, Name: name}, &got) + if err != nil && !apierrors.IsNotFound(err) { + return false, fmt.Errorf("read the probe job: %w", err) + } + for _, cond := range got.Status.Conditions { + if cond.Status != corev1.ConditionTrue { + continue + } + switch cond.Type { + case batchv1.JobComplete: + return true, nil + case batchv1.JobFailed: + return false, nil + } + } + if time.Now().After(deadline) { + return false, nil + } + select { + case <-ctx.Done(): + return false, ctx.Err() + case <-time.After(usernsProbePoll): + } + } +} diff --git a/internal/build/userns_test.go b/internal/build/userns_test.go new file mode 100644 index 0000000..d5566ea --- /dev/null +++ b/internal/build/userns_test.go @@ -0,0 +1,144 @@ +package build + +import ( + "context" + "sync/atomic" + "testing" + "time" + + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/runtime" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" +) + +// "auto" follows the probe through every copy of the Config; "on" and "off" +// ignore it. +func TestUserNamespacesMode(t *testing.T) { + probe := new(atomic.Bool) + for _, tc := range []struct { + mode string + probe bool + want bool + }{ + {"", false, false}, {"", true, true}, {UserNamespacesAuto, true, true}, + {UserNamespacesOn, false, true}, {UserNamespacesOff, true, false}, + } { + b, _, jb := newBuilder() + b.Config.UserNamespaces = tc.mode + b.Config.UserNamespacesProbe = probe + probe.Store(tc.probe) + if _, err := b.Submit(context.Background(), goodRequest()); err != nil { + t.Fatalf("Submit: %v", err) + } + if got := jb.created[0].UserNamespaces; got != tc.want { + t.Errorf("mode %q, probe %v: UserNamespaces = %v, want %v", tc.mode, tc.probe, got, tc.want) + } + } + b, _, jb := newBuilder() + if _, err := b.Submit(context.Background(), goodRequest()); err != nil || jb.created[0].UserNamespaces { + t.Errorf("no probe wired: err %v, UserNamespaces %v", err, jb.created[0].UserNamespaces) + } +} + +// The probe pod has the shape that needs user-namespace support, and runs +// nothing that matters. +func TestUsernsProbeJobShape(t *testing.T) { + job := UsernsProbeJob("felis-build", "felis-build", "felis:test", "userns-probe-x") + spec := job.Spec.Template.Spec + if spec.HostUsers == nil || *spec.HostUsers { + t.Fatal("the probe must ask for hostUsers: false") + } + if spec.AutomountServiceAccountToken == nil || *spec.AutomountServiceAccountToken || spec.ServiceAccountName != "felis-build" { + t.Error("the probe must run as the bare build SA with no token") + } + c := spec.Containers[0] + if c.Image != "felis:test" || len(c.Args) != 1 || c.Args[0] != "version" { + t.Errorf("probe container runs %s %v", c.Image, c.Args) + } + if c.SecurityContext.RunAsUser == nil || *c.SecurityContext.RunAsUser != 0 || len(c.SecurityContext.Capabilities.Add) != 3 { + t.Error("the probe must run as root with kaniko's capabilities, the case user namespaces must carry") + } + if len(spec.Volumes) != 1 || spec.Volumes[0].EmptyDir == nil { + t.Error("the probe must mount an emptyDir, as build pods do") + } + if job.Labels[LabelManagedBy] != managedByValue { + t.Error("the probe must sit under the build egress policy") + } +} + +func TestProbeUserNamespaces(t *testing.T) { + timeout, poll := usernsProbeTimeout, usernsProbePoll + t.Cleanup(func() { usernsProbeTimeout, usernsProbePoll = timeout, poll }) + usernsProbePoll = 5 * time.Millisecond + + scheme := runtime.NewScheme() + _ = clientgoscheme.AddToScheme(scheme) + for _, tc := range []struct { + name string + cond batchv1.JobConditionType + want bool + }{ + {"complete", batchv1.JobComplete, true}, + {"failed", batchv1.JobFailed, false}, + {"never finishes", "", false}, + } { + t.Run(tc.name, func(t *testing.T) { + // A finished probe must answer long before the timeout would. + usernsProbeTimeout = 5 * time.Second + if tc.cond == "" { + usernsProbeTimeout = 100 * time.Millisecond + } + start := time.Now() + cl := fake.NewClientBuilder().WithScheme(scheme).WithStatusSubresource(&batchv1.Job{}).Build() + jobs := NewK8sJobs(cl, Config{}) + type result struct { + ok bool + err error + } + done := make(chan result, 1) + go func() { + ok, err := jobs.ProbeUserNamespaces(context.Background(), "felis:test") + done <- result{ok, err} + }() + if tc.cond != "" { + finishProbe(t, cl, tc.cond) + } + r := <-done + if r.err != nil || r.ok != tc.want { + t.Fatalf("ProbeUserNamespaces = (%v, %v), want (%v, nil)", r.ok, r.err, tc.want) + } + if tc.cond != "" && time.Since(start) > 2*time.Second { + t.Fatalf("the probe answered after %s: it waited out the timeout instead of reading the verdict", time.Since(start)) + } + var left batchv1.JobList + if err := cl.List(context.Background(), &left, client.InNamespace(defaultNamespace)); err != nil || len(left.Items) != 0 { + t.Fatalf("probe jobs left behind: %d (%v)", len(left.Items), err) + } + }) + } +} + +// finishProbe waits for the probe Job to appear and marks it finished. +func finishProbe(t *testing.T, cl client.Client, cond batchv1.JobConditionType) { + t.Helper() + deadline := time.Now().Add(time.Second) + for time.Now().Before(deadline) { + var jobs batchv1.JobList + if err := cl.List(context.Background(), &jobs, client.InNamespace(defaultNamespace)); err != nil { + t.Fatal(err) + } + if len(jobs.Items) == 1 { + job := jobs.Items[0] + job.Status.Conditions = []batchv1.JobCondition{{Type: cond, Status: corev1.ConditionTrue}} + if err := cl.Status().Update(context.Background(), &job); err != nil { + t.Fatal(err) + } + return + } + time.Sleep(2 * time.Millisecond) + } + t.Fatal("the probe job never appeared") +} diff --git a/internal/config/config.go b/internal/config/config.go index 764113c..838e7c4 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -161,6 +161,16 @@ type RegistryConfig struct { TrivyImage string `toml:"trivy_image"` BuildCPULimit string `toml:"build_cpu_limit"` BuildMemLimit string `toml:"build_mem_limit"` + // BuildDiskLimit caps a build pod's ephemeral storage (context, unpacked base + // image and image tarball together). Empty keeps 12Gi. + BuildDiskLimit string `toml:"build_disk_limit"` + // BuildUserNamespaces runs build pods in a user namespace (hostUsers: false): + // "auto" (the default) turns it on when felis-api's startup probe pod ran + // with it, "on" always, "off" never. + BuildUserNamespaces string `toml:"build_user_namespaces"` + // BuildRuntimeClass runs build pods under a sandbox RuntimeClass such as + // gVisor or Kata. Empty runs them under the node's default runtime. + BuildRuntimeClass string `toml:"build_runtime_class"` // TrivyDBRepository points Trivy at an OCI repository holding the // vulnerability DB (--db-repository). Trivy's default fetches from // mirror.gcr.io/ghcr.io, which the build egress lock denies — so on a @@ -448,6 +458,11 @@ func (c *Config) Validate() error { if c.Registry.URL != "" && strings.Contains(c.Registry.URL, "://") { return fmt.Errorf("config: [registry] url %q must be a bare host[:port] with no scheme (e.g. registry.felis.svc:5000); a scheme breaks the user-modpack build lane's derived push target", c.Registry.URL) } + switch c.Registry.BuildUserNamespaces { + case "", "auto", "on", "off": + default: + return fmt.Errorf("config: [registry] build_user_namespaces %q must be auto, on or off", c.Registry.BuildUserNamespaces) + } // [smtp] is optional as a whole, but once a host is named the block must be // deliverable: a From address (relays reject MAIL FROM:<>) and a sane port. // Fail at load, not at the first OTP a player is waiting on. diff --git a/internal/config/config_test.go b/internal/config/config_test.go index 01a6157..82db3dd 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -51,6 +51,9 @@ kaniko_image = "registry.felis.svc:5000/mirror/kaniko:v1.23.2" trivy_image = "registry.felis.svc:5000/mirror/trivy:0.58.1" build_cpu_limit = "1" build_mem_limit = "2Gi" +build_disk_limit = "8Gi" +build_user_namespaces = "off" +build_runtime_class = "gvisor" [archive] store = "tarLocal" @@ -95,6 +98,23 @@ func TestLoadValid(t *testing.T) { if cfg.Registry.BuildCPULimit != "1" || cfg.Registry.BuildMemLimit != "2Gi" { t.Errorf("build resource overrides = %q / %q", cfg.Registry.BuildCPULimit, cfg.Registry.BuildMemLimit) } + if r := cfg.Registry; r.BuildDiskLimit != "8Gi" || r.BuildUserNamespaces != "off" || r.BuildRuntimeClass != "gvisor" { + t.Errorf("build isolation overrides = %q / %q / %q", r.BuildDiskLimit, r.BuildUserNamespaces, r.BuildRuntimeClass) + } +} + +func TestLoadRejectsUnknownBuildUserNamespaces(t *testing.T) { + _, err := config.Load(writeTOML(t, ` +[server] +root_domain = "mc.example.net" +[database] +url = "postgres://felis@db/felis" +[registry] +build_user_namespaces = "yes" +`)) + if err == nil || !strings.Contains(err.Error(), "build_user_namespaces") { + t.Fatalf("err = %v, want build_user_namespaces rejected", err) + } } func TestLoadAppliesDefaults(t *testing.T) { diff --git a/internal/platform/bundle.go b/internal/platform/bundle.go index d7cdd17..f567dbc 100644 --- a/internal/platform/bundle.go +++ b/internal/platform/bundle.go @@ -27,8 +27,8 @@ type Object interface { // namespaced Roles + RoleBindings — felis-api and felis-operator always, plus the // destructive felis-reaper identity only when retention is enabled, gated with its // CronJob), the two weak Job SAs (build/restore, which have NO Role anywhere), the -// build-namespace egress NetworkPolicy, and the minecraft-namespace ingress -// NetworkPolicies. +// build-namespace egress NetworkPolicy and LimitRange, and the minecraft-namespace +// ingress NetworkPolicies. // // Scope: this is the authorization + network fence (spec §21, §22) plus the // running control-plane workloads it fences — the felis-api / felis-operator @@ -93,6 +93,9 @@ func Objects(p Params) []Object { }) buildNP.TypeMeta = metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"} objs = append(objs, buildNP) + buildLR := build.BuildLimitRange(p.BuildNamespace) + buildLR.TypeMeta = metav1.TypeMeta{APIVersion: "v1", Kind: "LimitRange"} + objs = append(objs, buildLR) // Minecraft-namespace ingress fence (default-deny + RCON + game), the server // egress fence, and the registry's ingress fence.