fix(build): 构建 pod 等出口策略生效再运行,加 seccomp、可选 user namespace 与磁盘上限,上下文解包限总字节与条目数

This commit is contained in:
Lemon-miaow committed 2026-09-24 20:42:14 +08:00
1 parent 9bda3a52fa
commit 13b65e19ec
17 files changed
+948 -29

No files matched your search

+32 -1
View File
@@ -9,6 +9,7 @@ import (
"os"
"regexp"
"strings"
"sync/atomic"
"time"
"felis.lolicon.best/internal/api"
@@ -152,11 +153,13 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// The fetch initContainer runs THIS image's fetch-context entrypoint, so the
// build config carries the api's own image (the platform sets FELIS_IMAGE).
buildCfg.FelisImage = os.Getenv("FELIS_IMAGE")
buildJobs := build.NewK8sJobs(cl, buildCfg)
builder := &build.Builder{
Store: build.NewPGStore(drv.DB()),
Jobs: build.NewK8sJobs(cl, buildCfg),
Jobs: buildJobs,
Config: buildCfg,
}
go probeBuildUserNamespaces(ctx, buildJobs, buildCfg, stderr)
// User-modpack approval lane (user-directed extension over §16; see
// internal/submit). An ordinary user may only SUBMIT a
@@ -457,6 +460,11 @@ func buildConfig(cfg *config.Config) build.Config {
TrivyImage: cfg.Registry.TrivyImage,
CPULimit: cfg.Registry.BuildCPULimit,
MemLimit: cfg.Registry.BuildMemLimit,
DiskLimit: cfg.Registry.BuildDiskLimit,
// "auto" follows the startup probe (see probeBuildUserNamespaces).
UserNamespaces: cfg.Registry.BuildUserNamespaces,
UserNamespacesProbe: new(atomic.Bool),
RuntimeClass: cfg.Registry.BuildRuntimeClass,
// Empty keeps Trivy's own default; an install with builds points this at
// the internal DB mirror (see config.RegistryConfig.TrivyDBRepository).
TrivyDBRepository: cfg.Registry.TrivyDBRepository,
@@ -464,6 +472,29 @@ func buildConfig(cfg *config.Config) build.Config {
}
}
// probeBuildUserNamespaces settles build_user_namespaces = "auto": one probe
// pod with hostUsers: false tells whether this node's kernel and runtime can run
// build pods in a user namespace. Builds submitted before it answers run without.
func probeBuildUserNamespaces(ctx context.Context, jobs *build.K8sJobs, cfg build.Config, stderr io.Writer) {
if mode := cfg.UserNamespaces; mode != "" && mode != build.UserNamespacesAuto {
return
}
if cfg.FelisImage == "" {
fmt.Fprintln(stderr, "felis api: FELIS_IMAGE unset — build pods run without a user namespace")
return
}
ok, err := jobs.ProbeUserNamespaces(ctx, cfg.FelisImage)
cfg.UserNamespacesProbe.Store(ok)
switch {
case ok:
fmt.Fprintln(stderr, "felis api: build pods run in a user namespace (hostUsers: false)")
case err != nil:
fmt.Fprintf(stderr, "felis api: build pods run without a user namespace: the probe failed: %v\n", err)
default:
fmt.Fprintln(stderr, "felis api: build pods run without a user namespace: this node cannot start a pod with hostUsers: false")
}
}
// internalAPIBaseURL resolves the platform's internal-face base URL: the address
// the platform rendered into this pod (felis API base URL env), or — for a
// hand-rolled deployment that set none — the platform default control namespace,
+66
View File
@@ -0,0 +1,66 @@
package main
import (
"flag"
"fmt"
"io"
"net"
"os"
"time"
)
// Vars so tests can shrink them. A dial that neither connects nor is refused
// within egressDialTimeout counts as blocked: a policy that drops packets looks
// exactly like that.
var (
egressDialTimeout = 500 * time.Millisecond
egressPollInterval = 200 * time.Millisecond
)
// cmdEgressGate is the first initContainer of every build pod. The pod's
// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s:
// a build-labelled pod reached the internet and the Kubernetes API for its first
// ~0.7 s), so the gate dials a destination the policy denies until it stops
// answering, and only then lets the pod's next container, eventually the
// untrusted Dockerfile, start.
//
// The default probe is the Kubernetes API Service, which the kubelet names in
// every pod's environment and the build policy never admits. A probe that still
// answers after --wait means the policy is not enforced at all (a CNI without
// NetworkPolicy support, or k3s run with --disable-network-policy), and the
// build fails closed.
func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError)
fs.SetOutput(stderr)
probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused")
if err := fs.Parse(args); err != nil {
return 2
}
if *probe == "" {
host, port := os.Getenv("KUBERNETES_SERVICE_HOST"), os.Getenv("KUBERNETES_SERVICE_PORT")
if host == "" || port == "" {
fmt.Fprintln(stderr, "felis egress-gate: no --probe and no KUBERNETES_SERVICE_HOST/PORT to default to")
return 2
}
*probe = net.JoinHostPort(host, port)
}
start := time.Now()
for {
conn, err := net.DialTimeout("tcp", *probe, egressDialTimeout)
if err != nil {
fmt.Fprintf(stdout, "felis egress-gate: %s is unreachable after %s (%v); the egress lock is in effect\n",
*probe, time.Since(start).Round(time.Millisecond), err)
return 0
}
_ = conn.Close()
if time.Since(start) >= *wait {
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+
"(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n",
*probe, *wait)
return 1
}
time.Sleep(egressPollInterval)
}
}
+98
View File
@@ -0,0 +1,98 @@
package main
import (
"bytes"
"net"
"strings"
"testing"
"time"
)
func shrinkEgressGate(t *testing.T) {
t.Helper()
dial, poll := egressDialTimeout, egressPollInterval
egressDialTimeout, egressPollInterval = 200*time.Millisecond, 10*time.Millisecond
t.Cleanup(func() { egressDialTimeout, egressPollInterval = dial, poll })
}
// The gate holds while the probe answers and lets the pod go on once the policy
// lands, which the test plays by closing the listener.
func TestEgressGateWaitsForTheLock(t *testing.T) {
shrinkEgressGate(t)
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
accepted := make(chan struct{}, 100)
go func() {
for {
c, err := ln.Accept()
if err != nil {
return
}
_ = c.Close()
accepted <- struct{}{}
}
}()
go func() {
for i := 0; i < 3; i++ {
<-accepted
}
_ = ln.Close()
}()
var out, errb bytes.Buffer
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "10s"}, &out, &errb); code != 0 {
t.Fatalf("exit %d: %s", code, errb.String())
}
if !strings.Contains(out.String(), "egress lock is in effect") {
t.Errorf("stdout = %q", out.String())
}
}
// A probe that keeps answering means no policy is enforced: the build must not run.
func TestEgressGateRefusesAnOpenNetwork(t *testing.T) {
shrinkEgressGate(t)
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
defer ln.Close()
go func() {
for {
c, err := ln.Accept()
if err != nil {
return
}
_ = c.Close()
}
}()
var out, errb bytes.Buffer
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "100ms"}, &out, &errb); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
if !strings.Contains(errb.String(), "not enforced") {
t.Errorf("stderr = %q", errb.String())
}
}
func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) {
shrinkEgressGate(t)
t.Setenv("KUBERNETES_SERVICE_HOST", "")
t.Setenv("KUBERNETES_SERVICE_PORT", "")
var out, errb bytes.Buffer
if code := cmdEgressGate(nil, &out, &errb); code != 2 {
t.Fatalf("exit %d without a probe, want 2", code)
}
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
host, port, _ := net.SplitHostPort(ln.Addr().String())
_ = ln.Close() // closed: the lock reads as in effect at once
t.Setenv("KUBERNETES_SERVICE_HOST", host)
t.Setenv("KUBERNETES_SERVICE_PORT", port)
out.Reset()
if code := cmdEgressGate(nil, &out, &errb); code != 0 || !strings.Contains(out.String(), ln.Addr().String()) {
t.Fatalf("exit %d, stdout %q", code, out.String())
}
}
+27 -2
View File
@@ -136,13 +136,27 @@ func fetchContextWithRetry(ctx context.Context, client *http.Client, url, token
}
}
// maxContextBytes / maxContextEntries bound what one context may expand to. The
// compressed upload is capped at 1 GiB, but gzip turns that into hundreds of GiB
// or millions of empty files, and the emptyDir's 4 GiB sizeLimit is only
// enforced by the kubelet's periodic sweep, after the disk has filled. The byte
// cap matches that sizeLimit; the entry cap is far above any real modpack (a
// large one is a few thousand files) and far below an inode exhaustion.
//
// Vars, not consts, so tests can shrink them.
var (
maxContextBytes int64 = 4 << 30
maxContextEntries = 200_000
)
// extractTarGz streams a gzip'd tarball into root, creating directories as
// needed. Every entry is vetted BEFORE anything is written: a path that is
// absolute or escapes root (via ".."), a link (symlink or hardlink), or any
// special file kind aborts the whole extraction. Refusing rather than skipping is
// deliberate — a context that needs one of those constructs is not a context this
// transport carries, and silently dropping entries would build from a corpus the
// submitter did not upload.
// submitter did not upload. The whole extraction is also bounded by
// maxContextBytes and maxContextEntries.
func extractTarGz(r io.Reader, root string) error {
if err := os.MkdirAll(root, 0o755); err != nil {
return fmt.Errorf("create context dir: %w", err)
@@ -153,6 +167,8 @@ func extractTarGz(r io.Reader, root string) error {
}
defer zr.Close()
tr := tar.NewReader(zr)
var written int64
entries := 0
for {
hdr, err := tr.Next()
if errors.Is(err, io.EOF) {
@@ -161,6 +177,9 @@ func extractTarGz(r io.Reader, root string) error {
if err != nil {
return fmt.Errorf("read context tarball: %w", err)
}
if entries++; entries > maxContextEntries {
return fmt.Errorf("the build context has more than %d entries", maxContextEntries)
}
name := filepath.Clean(hdr.Name)
if name == "." {
continue
@@ -188,10 +207,16 @@ func extractTarGz(r io.Reader, root string) error {
if err != nil {
return fmt.Errorf("create %q: %w", name, err)
}
if _, err := io.Copy(f, tr); err != nil {
n, err := io.Copy(f, io.LimitReader(tr, maxContextBytes-written+1))
written += n
if err != nil {
_ = f.Close()
return fmt.Errorf("write %q: %w", name, err)
}
if written > maxContextBytes {
_ = f.Close()
return fmt.Errorf("the build context expands past %d bytes", maxContextBytes)
}
if err := f.Close(); err != nil {
return fmt.Errorf("close %q: %w", name, err)
}
+24
View File
@@ -121,6 +121,30 @@ func TestExtractTarGzRefusesEscapes(t *testing.T) {
}
}
// A context that expands past the byte or entry cap is refused, however small
// it was compressed: gzip bombs and inode floods stop at the cap.
func TestExtractTarGzCapsExpansion(t *testing.T) {
bytesCap, entriesCap := maxContextBytes, maxContextEntries
t.Cleanup(func() { maxContextBytes, maxContextEntries = bytesCap, entriesCap })
maxContextBytes, maxContextEntries = 1000, 5
fits := tgzBody(t, tarEntry{name: "a", body: strings.Repeat("x", 600)}, tarEntry{name: "b", body: strings.Repeat("y", 400)})
if err := extractTarGz(bytes.NewReader(fits), t.TempDir()); err != nil {
t.Fatalf("a context exactly at the byte cap: %v", err)
}
big := tgzBody(t, tarEntry{name: "a", body: strings.Repeat("x", 600)}, tarEntry{name: "b", body: strings.Repeat("y", 401)})
if err := extractTarGz(bytes.NewReader(big), t.TempDir()); err == nil || !strings.Contains(err.Error(), "expands past") {
t.Fatalf("one byte over the cap: err = %v", err)
}
var many []tarEntry
for i := 0; i < 6; i++ {
many = append(many, tarEntry{name: "d" + string(rune('0'+i)) + "/", typ: tar.TypeDir})
}
if err := extractTarGz(bytes.NewReader(tgzBody(t, many...)), t.TempDir()); err == nil || !strings.Contains(err.Error(), "entries") {
t.Fatalf("six entries over a cap of five: err = %v", err)
}
}
// The command end to end: it dials the URL with the bearer token from the
// environment, and refuses to run without it (the internal face would 401
// anyway; failing at parse time is the honest earlier error).
+2
View File
@@ -21,6 +21,7 @@ Commands:
restore Extract a world archive into a world volume (internal Job entrypoint)
backup Archive a world into the backup store and record it (internal Job entrypoint)
files List/read/write one file in a stopped server's world (internal Job entrypoint)
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
push-image Push a scanned image tarball to the registry (internal Job entrypoint)
registry-gate Authorize registry writes in front of registry:2 (internal sidecar entrypoint)
@@ -56,6 +57,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
"restore": cmdRestore,
"backup": cmdBackup,
"files": cmdFiles,
"egress-gate": cmdEgressGate,
"fetch-context": cmdFetchContext,
"push-image": cmdPushImage,
"registry-gate": cmdRegistryGate,
+1 -1
View File
@@ -2352,7 +2352,7 @@ persisted_registry_block() {
out="$(awk '
/^[[:space:]]*\[/ { sect = $0; next }
sect ~ /^[[:space:]]*\[registry\][[:space:]]*$/ &&
/^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|user_uploads_context)[[:space:]]*=/ { print }
/^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|user_uploads_context)[[:space:]]*=/ { print }
sect ~ /^[[:space:]]*\[registry\.s3\][[:space:]]*$/ && /^[[:space:]]*[A-Za-z_]+[[:space:]]*=/ {
if (!s3hdr) { printf "[registry.s3]\n"; s3hdr = 1 }
print
+6
View File
@@ -1037,6 +1037,9 @@ build_namespace = "stale-ns"
kaniko_image = "registry.felis.svc:5000/mirror/kaniko-executor:v1.24.0"
trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"
trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"
build_disk_limit = "20Gi"
build_user_namespaces = "off"
build_runtime_class = "gvisor"
[registry.s3]
endpoint = "https://s3.example"
@@ -1074,6 +1077,9 @@ expect "a re-run carries the trivy vulnerability-DB mirror" \
'trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"' "$out"
expect "a re-run carries the trivy java-DB mirror" \
'trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"' "$out"
expect "a re-run carries the build disk cap" 'build_disk_limit = "20Gi"' "$out"
expect "a re-run carries the build user-namespace mode" 'build_user_namespaces = "off"' "$out"
expect "a re-run carries the build runtime class" 'build_runtime_class = "gvisor"' "$out"
expect "a re-run carries the [registry.s3] uploads subtable" "[registry.s3]" "$out"
expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "$out"
expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out"
+81 -3
View File
@@ -454,9 +454,9 @@ internal registry.
`writeBuildError` (JobPhase→Failed). [GO-TESTED for the mapping.] The underlying
cause — a kaniko build error, the **Trivy CRITICAL-CVE gate** failing the build
(spec §16), or the final push — is in the Job's pod logs and is
[INTEGRATION-ONLY]. The pod runs `kaniko` (builds a tarball, never pushes) and
`trivy` (scans that tarball) as init containers, then `push` — so a CVE-rejected
image never reaches the registry. Inspect every step:
[INTEGRATION-ONLY]. The pod runs `egress-gate` (§8f), `context-fetch`, `kaniko`
(builds a tarball, never pushes) and `trivy` (scans that tarball) as init
containers, then `push` — so a CVE-rejected image never reaches the registry. Inspect every step:
```
kubectl logs -n felis-build job/<build-job> --all-containers --prefix
@@ -490,6 +490,9 @@ trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"
trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"
build_cpu_limit = "2"
build_mem_limit = "4Gi"
build_disk_limit = "12Gi" # §8f
build_user_namespaces = "auto" # §8f: auto | on | off
build_runtime_class = "" # §8f: e.g. "gvisor"
```
Mirror the executor images into the registry once. On the node itself, push
@@ -560,6 +563,81 @@ artifacts — every real modpack — and that download fails closed too. Mirror
above); the Java DB refreshes far less often than the vulnerability DB, so a
one-off mirror is usually fine.
### 8f. Build isolation model and residual risk
A Dockerfile's `RUN` steps execute inside the kaniko container, as root. Kaniko
is a daemonless image builder; it is **not** a sandbox. What stands between an
approved-but-hostile Dockerfile and the node is the pod around it:
| Layer | What it does | Where |
|---|---|---|
| Admin approval | Nothing builds until an administrator approves the submission | submit lane |
| Weak identity | `felis-build` SA, no Role anywhere, no token mounted | §8b |
| Egress lock | `felis-build-egress`: cluster DNS, the registry, the api internal face, nothing else | §8c |
| Egress gate | first init container; holds the pod until the lock is enforced for it | `felis egress-gate` |
| Capabilities | every container drops ALL; kaniko gets back only CHOWN, DAC_OVERRIDE, FOWNER to unpack base images | jobspec |
| seccomp | the whole pod runs under the runtime's default profile (no `unshare`, `mount`, `keyctl`, `bpf`, …) | jobspec |
| User namespace | with `build_user_namespaces` on, root in the pod is an unprivileged uid on the node | below |
| Sandbox runtime | optional `build_runtime_class` (gVisor, Kata) | below |
| Credentials | the registry credential lives only in the `push` container; the service token only in `context-fetch` | jobspec |
| Resources | CPU, memory and ephemeral-storage limits per container; `activeDeadlineSeconds`; the context extraction stops at 4 GiB or 200 000 entries | jobspec, `felis fetch-context` |
| Namespace backstop | `felis-build-limits` LimitRange gives any container without limits 1 CPU / 1 GiB / 1 GiB disk | bundle |
**Egress gate.** The CNI programs a new pod's NetworkPolicy a moment after the
pod starts. On k3s (kube-router), a pod in `felis-build` could reach the internet
and the Kubernetes API for its first ~0.7 s. `egress-gate` dials the Kubernetes
API Service, which the build policy never admits, and exits once it stops
answering. The build pod log shows the wait:
```
felis egress-gate: 10.43.0.1:443 is unreachable after 612ms (...); the egress lock is in effect
```
If the probe still answers after two minutes the gate exits 1 and the build
fails: `the build namespace's NetworkPolicy is not enforced`. The cluster is
running without NetworkPolicy enforcement (a CNI without it, or k3s started with
`--disable-network-policy`); fix the cluster, not the gate.
**User namespaces (`build_user_namespaces`).** With `hostUsers: false`, uid 0
in the build pod maps to an unprivileged uid range on the node, so a container
escape lands as nobody. It needs Kubernetes ≥ 1.33, containerd 2.x, and a kernel
with idmapped mounts on the node filesystem (5.19+ upstream; the RHEL/CentOS
Stream 9 kernels carry the backport). `auto`, the default, lets felis-api decide
at startup: it runs one `userns-probe-*` Job in `felis-build` and turns the
feature on only when that pod ran. The api log says which way it went:
```
felis api: build pods run in a user namespace (hostUsers: false)
felis api: build pods run without a user namespace: this node cannot start a pod with hostUsers: false
```
`on` forces it (builds then fail to start on a node that cannot do it), `off`
never uses it.
**Sandbox runtime (`build_runtime_class`).** Naming a RuntimeClass runs build
pods under it, for example gVisor (`runsc`) or Kata. The class must exist
(`kubectl get runtimeclass`), and kaniko must work under it: gVisor needs its
default `overlay2` rootfs, and Kata needs nested virtualization on a VM node.
Leave it empty unless you have installed and tested one.
**Disk (`build_disk_limit`, default `12Gi`).** This caps kaniko's writable layer
(the unpacked base image) and, as the largest limit in the pod, the pod's total
disk: extracted context, unpacked base image and image tarball together. The
kubelet enforces it by eviction on its housekeeping sweep, so a build that runs
past it is killed within seconds. The build then fails with `Evicted` in
`kubectl -n felis-build describe pod`. Raise it for very large modpacks, and
keep the node's free disk above it.
**Residual risk.** Without a user namespace or a sandbox runtime, the build runs
as root in a container on the same kernel as the game servers and the control
plane. Seccomp and the dropped capabilities remove the common escape primitives.
A kernel vulnerability reachable through the remaining syscalls still reaches
the node, and on a single-node install the node is the whole platform. Admin
approval is the control that remains: read the Dockerfile and download the
context before approving. On a node
where the probe comes back negative, a kernel upgrade that brings idmapped
mounts is the cheapest hardening available.
---
## 9. Registry push/pull failures (spec §15)
+42
View File
@@ -37,6 +37,7 @@ import (
"errors"
"fmt"
"strings"
"sync/atomic"
"time"
"felis.lolicon.best/internal/metrics"
@@ -243,6 +244,37 @@ type Config struct {
// CPULimit / MemLimit cap each build container (spec §16: resource limits).
CPULimit string
MemLimit string
// DiskLimit caps the build pod's ephemeral storage: the extracted context,
// the base image kaniko unpacks and the image tarball together.
DiskLimit string
// UserNamespaces selects hostUsers: false for build pods: UserNamespacesOn,
// UserNamespacesOff, or UserNamespacesAuto (the default), which follows
// UserNamespacesProbe.
UserNamespaces string
// UserNamespacesProbe carries ProbeUserNamespaces' verdict to every copy of
// this Config; nil or false keeps "auto" off.
UserNamespacesProbe *atomic.Bool
// RuntimeClass runs build pods under a sandbox RuntimeClass (gVisor, Kata)
// when set. The class must exist on the cluster.
RuntimeClass string
}
// Values of Config.UserNamespaces.
const (
UserNamespacesAuto = "auto"
UserNamespacesOn = "on"
UserNamespacesOff = "off"
)
// userNamespaces resolves Config.UserNamespaces for one build.
func (c Config) userNamespaces() bool {
switch c.UserNamespaces {
case UserNamespacesOn:
return true
case UserNamespacesOff:
return false
}
return c.UserNamespacesProbe != nil && c.UserNamespacesProbe.Load()
}
// Defaults applied when a Config field is left zero.
@@ -255,6 +287,7 @@ const (
defaultMaxDockerfile = 256 * 1024 // 256 KiB
defaultCPULimit = "2"
defaultMemLimit = "4Gi"
defaultDiskLimit = "12Gi"
)
// withDefaults returns a copy of c with zero fields filled, so a partially
@@ -284,6 +317,12 @@ func (c Config) withDefaults() Config {
if c.MemLimit == "" {
c.MemLimit = defaultMemLimit
}
if c.DiskLimit == "" {
c.DiskLimit = defaultDiskLimit
}
if c.UserNamespaces == "" {
c.UserNamespaces = UserNamespacesAuto
}
return c
}
@@ -380,6 +419,9 @@ func (b *Builder) jobParams(bld *Build, cfg Config) JobParams {
Deadline: cfg.Deadline,
CPULimit: cfg.CPULimit,
MemLimit: cfg.MemLimit,
DiskLimit: cfg.DiskLimit,
UserNamespaces: cfg.userNamespaces(),
RuntimeClass: cfg.RuntimeClass,
}
}
+141 -10
View File
@@ -33,6 +33,7 @@ const (
// build-log streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8)
// follows the same container this Job defines — one source of truth for the name.
const (
ContainerGate = "egress-gate"
ContainerKaniko = "kaniko"
ContainerTrivy = "trivy"
ContainerPush = "push"
@@ -64,6 +65,23 @@ var imageSizeLimit = resource.MustParse("10Gi")
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
var contextSizeLimit = resource.MustParse("4Gi")
// Per-container ephemeral-storage bounds (writable layer + logs; emptyDirs count
// toward the pod as a whole). Kaniko's limit is the operator's disk cap because
// kaniko unpacks the base image into its own root filesystem, which no emptyDir
// bound covers; it is also the largest limit in the pod, so it becomes the
// pod-level cap the kubelet holds context + unpacked rootfs + image tarball to.
// Trivy keeps its vulnerability and Java DBs (about 1.4 GiB live) in its layer.
// The others write nothing but logs.
var (
gateDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("64Mi")}
fetchDisk = diskBounds{request: resource.MustParse("64Mi"), limit: resource.MustParse("256Mi")}
kanikoDiskRq = resource.MustParse("1Gi")
trivyDisk = diskBounds{request: resource.MustParse("256Mi"), limit: resource.MustParse("4Gi")}
pushDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("256Mi")}
)
type diskBounds struct{ request, limit resource.Quantity }
// buildJobTTL is how long a finished build Job survives before the Job
// controller deletes it — and with it the Pod whose kaniko log is the admin
// failure-triage surface (GET /api/v1/images/build/{id}/logs).
@@ -105,6 +123,15 @@ type JobParams struct {
Deadline time.Duration
CPULimit string
MemLimit string
// DiskLimit caps kaniko's ephemeral storage, and with it the pod's (see
// kanikoDiskRq). Empty applies defaultDiskLimit.
DiskLimit string
// UserNamespaces runs the pod with hostUsers: false, so root in the build
// containers is an unprivileged uid on the node. It needs a kernel and runtime
// with idmapped mounts; Config.UserNamespaces decides.
UserNamespaces bool
// RuntimeClass, when set, runs the pod under that RuntimeClass (gVisor, Kata).
RuntimeClass string
}
// BuildJobName is the deterministic Job name for a build id.
@@ -127,8 +154,14 @@ func buildLabels(p JobParams) map[string]string {
// reach the K8s API (spec §16, §21);
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
// so docker-in-docker / privileged is never needed (spec §16, §22);
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
// - the RuntimeDefault seccomp profile on the whole pod, and optionally a user
// namespace (hostUsers: false) and a sandbox RuntimeClass, because kaniko is
// no isolation boundary: the Dockerfile's RUN steps execute in its container;
// - an egress gate ahead of everything else, so nothing runs before the
// namespace's NetworkPolicy is enforced for this pod (cmd/felis egress-gate);
// - activeDeadlineSeconds + backoffLimit=0 + per-container CPU, memory and
// ephemeral-storage limits so a runaway or poisoned build cannot exhaust the
// node (spec §16);
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
// automatic admission gate (spec §16).
@@ -149,6 +182,18 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
if err != nil {
return nil, err
}
diskCap := p.DiskLimit
if diskCap == "" {
diskCap = defaultDiskLimit
}
kanikoDisk, err := resource.ParseQuantity(diskCap)
if err != nil {
return nil, fmt.Errorf("build: invalid disk limit %q: %w", diskCap, err)
}
kanikoRq := kanikoDiskRq.DeepCopy()
if kanikoDisk.Cmp(kanikoRq) < 0 {
kanikoRq = kanikoDisk.DeepCopy()
}
if p.FelisImage == "" {
return nil, fmt.Errorf("build: FelisImage is required: the push container runs it")
}
@@ -187,7 +232,35 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
contextPath := p.ContextRef
initContainers := []corev1.Container{}
// The gate runs before anything else. A new pod's NetworkPolicy is programmed
// asynchronously: live on k3s (kube-router), a build-labelled pod reached the
// internet and the Kubernetes API for the first ~0.7 s of its life. The gate
// holds the pod until a destination the policy denies stops answering, so the
// Dockerfile never runs inside that window.
gateSec := sec.DeepCopy()
gateSec.ReadOnlyRootFilesystem = boolPtr(true)
gateSec.RunAsNonRoot = boolPtr(true)
gateSec.RunAsUser = int64Ptr(nonRootUID)
gate := corev1.Container{
Name: ContainerGate,
Image: p.FelisImage,
Args: []string{"egress-gate"},
Resources: corev1.ResourceRequirements{
Limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("100m"),
corev1.ResourceMemory: resource.MustParse("64Mi"),
corev1.ResourceEphemeralStorage: gateDisk.limit,
},
Requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("10m"),
corev1.ResourceMemory: resource.MustParse("16Mi"),
corev1.ResourceEphemeralStorage: gateDisk.request,
},
},
SecurityContext: gateSec,
}
initContainers := []corev1.Container{gate}
imageMount := corev1.VolumeMount{Name: imageVolume, MountPath: imageMountPath}
kanikoMounts := []corev1.VolumeMount{imageMount}
podVolumes := []corev1.Volume{{
@@ -232,7 +305,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
}},
}},
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, fetchDisk),
SecurityContext: fetchSec,
}
initContainers = append(initContainers, fetch)
@@ -269,7 +342,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
"--skip-tls-verify-pull",
},
VolumeMounts: kanikoMounts,
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, diskBounds{request: kanikoRq, limit: kanikoDisk}),
SecurityContext: kanikoSec,
}
initContainers = append(initContainers, kaniko)
@@ -298,7 +371,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
Image: p.TrivyImage,
Args: trivyArgs,
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, trivyDisk),
SecurityContext: sec,
}
initContainers = append(initContainers, trivy)
@@ -328,7 +401,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
secretEnv("FELIS_REGISTRY_PASSWORD", naming.RegistryPushPasswordKey),
},
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, pushDisk),
SecurityContext: pushSec,
}
@@ -350,16 +423,44 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
InitContainers: initContainers,
Containers: []corev1.Container{push},
Volumes: podVolumes,
// RUN steps execute in kaniko's container with root and three
// capabilities; RuntimeDefault takes away the syscalls a container
// never needs, among them most kernel-escape primitives (unshare,
// mount, keyctl, bpf). Every step was checked to run under it live.
SecurityContext: &corev1.PodSecurityContext{
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
},
InitContainers: initContainers,
Containers: []corev1.Container{push},
Volumes: podVolumes,
},
},
},
}
if p.UserNamespaces {
job.Spec.Template.Spec.HostUsers = boolPtr(false)
}
if p.RuntimeClass != "" {
rc := p.RuntimeClass
job.Spec.Template.Spec.RuntimeClassName = &rc
}
return job, nil
}
// nonRootUID is the distroless nonroot user the platform image ships as.
const nonRootUID = 65532
// withDisk is the resources block for one build container: the CPU and memory
// caps with their schedulable floor (buildRequests), plus its ephemeral-storage
// request and limit.
func withDisk(limits corev1.ResourceList, d diskBounds) corev1.ResourceRequirements {
lim := limits.DeepCopy()
lim[corev1.ResourceEphemeralStorage] = d.limit
req := buildRequests(limits)
req[corev1.ResourceEphemeralStorage] = d.request
return corev1.ResourceRequirements{Limits: lim, Requests: req}
}
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
// lane derives when the API is the blob transport — i.e. a context only the
// fetch initContainer can turn into a local path for Kaniko.
@@ -516,6 +617,36 @@ func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
}
}
// BuildLimitRange bounds any container in the build namespace that arrives
// without its own limits. Build Jobs set every limit themselves (BuildJob); this
// is the backstop for anything else that lands in the namespace, which shares
// the node's disk with the game worlds.
func BuildLimitRange(namespace string) *corev1.LimitRange {
return &corev1.LimitRange{
ObjectMeta: metav1.ObjectMeta{
Name: "felis-build-limits",
Namespace: namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
Spec: corev1.LimitRangeSpec{Limits: []corev1.LimitRangeItem{{
Type: corev1.LimitTypeContainer,
Default: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("1"),
corev1.ResourceMemory: resource.MustParse("1Gi"),
corev1.ResourceEphemeralStorage: resource.MustParse("1Gi"),
},
DefaultRequest: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("100m"),
corev1.ResourceMemory: resource.MustParse("128Mi"),
corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"),
},
}}},
}
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
+116 -10
View File
@@ -171,10 +171,10 @@ func TestBuildJobScansBeforePush(t *testing.T) {
t.Fatalf("BuildJob: %v", err)
}
inits := job.Spec.Template.Spec.InitContainers
if len(inits) != 2 || inits[0].Name != ContainerKaniko || inits[1].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [kaniko trivy]", initNames(inits))
if len(inits) != 3 || inits[0].Name != ContainerGate || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [egress-gate kaniko trivy]", initNames(inits))
}
kaniko, trivy := inits[0], inits[1]
kaniko, trivy := inits[1], inits[2]
for _, want := range []string{"--destination=" + p.ImageRef, "--no-push", "--tar-path=" + imageTarPath} {
if !hasArg(kaniko.Args, want) {
t.Errorf("kaniko args = %v, want %s", kaniko.Args, want)
@@ -339,9 +339,9 @@ func TestBuildJobTrivyDBRepositoryOverride(t *testing.T) {
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
trivy := job.Spec.Template.Spec.InitContainers[1]
trivy := job.Spec.Template.Spec.InitContainers[2]
if trivy.Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want trivy second", initNames(job.Spec.Template.Spec.InitContainers))
t.Fatalf("initContainers = %v, want trivy third", initNames(job.Spec.Template.Spec.InitContainers))
}
if !argPairPresent(trivy.Args, "--db-repository", p.TrivyDBRepository) {
t.Errorf("trivy args = %v, want --db-repository %s", trivy.Args, p.TrivyDBRepository)
@@ -423,10 +423,11 @@ func TestBuildJobFetchesHTTPContext(t *testing.T) {
t.Fatalf("BuildJob: %v", err)
}
inits := job.Spec.Template.Spec.InitContainers
if len(inits) != 3 || inits[0].Name != ContainerFetch || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [%s %s %s]", initNames(inits), ContainerFetch, ContainerKaniko, ContainerTrivy)
if len(inits) != 4 || inits[0].Name != ContainerGate || inits[1].Name != ContainerFetch ||
inits[2].Name != ContainerKaniko || inits[3].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [%s %s %s %s]", initNames(inits), ContainerGate, ContainerFetch, ContainerKaniko, ContainerTrivy)
}
fetch, kaniko := inits[0], inits[1]
fetch, kaniko := inits[1], inits[2]
if fetch.Image != p.FelisImage {
t.Errorf("fetch image = %q, want the platform image %q", fetch.Image, p.FelisImage)
}
@@ -494,8 +495,8 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) {
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 2 || inits[0].Name != ContainerKaniko {
t.Errorf("a native ref must render just kaniko + trivy, got %v", initNames(inits))
if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 3 || inits[1].Name != ContainerKaniko {
t.Errorf("a native ref must render just the gate, kaniko and trivy, got %v", initNames(inits))
}
for _, v := range job.Spec.Template.Spec.Volumes {
if v.Name == contextVolume {
@@ -504,6 +505,111 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) {
}
}
// Nothing runs before the egress gate, and the gate itself holds nothing: no
// credential, no root, no writable filesystem.
func TestBuildJobGatesEgressFirst(t *testing.T) {
p := sampleJobParams()
p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx"
job, err := BuildJob(p)
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
gate := job.Spec.Template.Spec.InitContainers[0]
if gate.Name != ContainerGate || gate.Image != p.FelisImage || len(gate.Args) != 1 || gate.Args[0] != "egress-gate" {
t.Fatalf("first initContainer = %s %s %v, want the platform image's egress-gate", gate.Name, gate.Image, gate.Args)
}
if len(gate.Env) != 0 || len(gate.VolumeMounts) != 0 {
t.Errorf("the gate must hold nothing, got env %v mounts %v", gate.Env, gate.VolumeMounts)
}
sc := gate.SecurityContext
if sc == nil || sc.RunAsNonRoot == nil || !*sc.RunAsNonRoot || sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
t.Errorf("the gate must run non-root on a read-only root, got %#v", sc)
}
}
// The pod runs under RuntimeDefault seccomp always, and in a user namespace or
// a sandbox runtime when the install asks for them.
func TestBuildJobSandboxing(t *testing.T) {
job, err := BuildJob(sampleJobParams())
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
spec := job.Spec.Template.Spec
if spec.SecurityContext == nil || spec.SecurityContext.SeccompProfile == nil ||
spec.SecurityContext.SeccompProfile.Type != corev1.SeccompProfileTypeRuntimeDefault {
t.Fatalf("pod securityContext = %#v, want seccompProfile RuntimeDefault", spec.SecurityContext)
}
if spec.HostUsers != nil || spec.RuntimeClassName != nil {
t.Errorf("defaults must leave hostUsers and runtimeClassName unset, got %v / %v", spec.HostUsers, spec.RuntimeClassName)
}
p := sampleJobParams()
p.UserNamespaces, p.RuntimeClass = true, "gvisor"
if job, err = BuildJob(p); err != nil {
t.Fatalf("BuildJob: %v", err)
}
spec = job.Spec.Template.Spec
if spec.HostUsers == nil || *spec.HostUsers {
t.Errorf("UserNamespaces must render hostUsers: false, got %v", spec.HostUsers)
}
if spec.RuntimeClassName == nil || *spec.RuntimeClassName != "gvisor" {
t.Errorf("runtimeClassName = %v, want gvisor", spec.RuntimeClassName)
}
}
// Every container carries an ephemeral-storage request and limit, and kaniko's
// limit, the largest, is the configured disk cap.
func TestBuildJobBoundsEphemeralStorage(t *testing.T) {
p := sampleJobParams()
p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx"
job, err := BuildJob(p)
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
all := append(append([]corev1.Container{}, job.Spec.Template.Spec.InitContainers...), job.Spec.Template.Spec.Containers...)
for _, c := range all {
lim, lok := c.Resources.Limits[corev1.ResourceEphemeralStorage]
req, rok := c.Resources.Requests[corev1.ResourceEphemeralStorage]
if !lok || !rok || lim.IsZero() || req.Cmp(lim) > 0 {
t.Errorf("container %s ephemeral-storage request %v limit %v", c.Name, req.String(), lim.String())
}
if c.Name == ContainerKaniko && lim.String() != defaultDiskLimit {
t.Errorf("kaniko ephemeral-storage limit = %s, want the default %s", lim.String(), defaultDiskLimit)
}
}
p.DiskLimit = "512Mi"
if job, err = BuildJob(p); err != nil {
t.Fatalf("BuildJob: %v", err)
}
for _, c := range job.Spec.Template.Spec.InitContainers {
if c.Name != ContainerKaniko {
continue
}
lim := c.Resources.Limits[corev1.ResourceEphemeralStorage]
req := c.Resources.Requests[corev1.ResourceEphemeralStorage]
if lim.String() != "512Mi" || req.Cmp(lim) > 0 {
t.Errorf("a 512Mi disk cap rendered limit %s request %s", lim.String(), req.String())
}
}
p.DiskLimit = "lots"
if _, err := BuildJob(p); err == nil {
t.Error("an unparsable disk limit was accepted")
}
}
func TestBuildLimitRangeCoversEphemeralStorage(t *testing.T) {
lr := BuildLimitRange("felis-build")
if lr.Namespace != "felis-build" || len(lr.Spec.Limits) != 1 {
t.Fatalf("limit range = %#v", lr)
}
item := lr.Spec.Limits[0]
for _, res := range []corev1.ResourceName{corev1.ResourceCPU, corev1.ResourceMemory, corev1.ResourceEphemeralStorage} {
d, dr := item.Default[res], item.DefaultRequest[res]
if d.IsZero() || dr.IsZero() || dr.Cmp(d) > 0 {
t.Errorf("%s default %s request %s", res, d.String(), dr.String())
}
}
}
func initNames(cs []corev1.Container) []string {
names := make([]string, 0, len(cs))
for _, c := range cs {
+128
View File
@@ -0,0 +1,128 @@
package build
import (
"context"
"fmt"
"strconv"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// Vars so tests can shrink them. The probe pod's image is the api's own, so it is
// already on the node; the timeout covers a slow pod start, and a runtime that
// cannot do user namespaces fails the pod well within it.
var (
usernsProbeTimeout = 3 * time.Minute
usernsProbePoll = 2 * time.Second
)
// UsernsProbeJob renders the one-shot Job that asks the cluster whether a build
// pod can run with hostUsers: false. The pod has the parts of a build pod that
// need idmapped mounts and namespaced capabilities: root with kaniko's three
// capabilities, the RuntimeDefault seccomp profile and an emptyDir. It runs
// `felis version`, which touches nothing.
func UsernsProbeJob(namespace, serviceAccount, image, name string) *batchv1.Job {
labels := map[string]string{LabelManagedBy: managedByValue, LabelComponent: "userns-probe"}
small := corev1.ResourceRequirements{
Limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("100m"),
corev1.ResourceMemory: resource.MustParse("64Mi"),
corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"),
},
Requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("10m"),
corev1.ResourceMemory: resource.MustParse("16Mi"),
corev1.ResourceEphemeralStorage: resource.MustParse("16Mi"),
},
}
return &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: namespace, Labels: labels},
Spec: batchv1.JobSpec{
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(int64(usernsProbeTimeout / time.Second)),
TTLSecondsAfterFinished: int32Ptr(300),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: labels},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: serviceAccount,
AutomountServiceAccountToken: boolPtr(false),
HostUsers: boolPtr(false),
SecurityContext: &corev1.PodSecurityContext{
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
},
Containers: []corev1.Container{{
Name: "probe",
Image: image,
Args: []string{"version"},
Resources: small,
SecurityContext: &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
RunAsUser: int64Ptr(0),
Capabilities: &corev1.Capabilities{
Drop: []corev1.Capability{"ALL"},
Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE", "FOWNER"},
},
},
VolumeMounts: []corev1.VolumeMount{{Name: "scratch", MountPath: "/scratch"}},
}},
Volumes: []corev1.Volume{{
Name: "scratch",
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{SizeLimit: quantityPtr(resource.MustParse("16Mi"))}},
}},
},
},
},
}
}
// ProbeUserNamespaces runs UsernsProbeJob and reports whether its pod succeeded.
// A pod that fails, or never starts before the timeout, answers false; err is set
// only when the Job could not be created or read. The Job is deleted afterwards.
func (k *K8sJobs) ProbeUserNamespaces(ctx context.Context, image string) (bool, error) {
name := "userns-probe-" + strconv.FormatInt(time.Now().UnixNano(), 36)
job := UsernsProbeJob(k.cfg.Namespace, k.cfg.ServiceAccount, image, name)
if err := k.c.Create(ctx, job); err != nil {
return false, fmt.Errorf("create the probe job: %w", err)
}
defer func() {
bg := metav1.DeletePropagationBackground
del := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: k.cfg.Namespace, Name: name}}
_ = k.c.Delete(context.WithoutCancel(ctx), del, &client.DeleteOptions{PropagationPolicy: &bg})
}()
deadline := time.Now().Add(usernsProbeTimeout)
for {
var got batchv1.Job
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.cfg.Namespace, Name: name}, &got)
if err != nil && !apierrors.IsNotFound(err) {
return false, fmt.Errorf("read the probe job: %w", err)
}
for _, cond := range got.Status.Conditions {
if cond.Status != corev1.ConditionTrue {
continue
}
switch cond.Type {
case batchv1.JobComplete:
return true, nil
case batchv1.JobFailed:
return false, nil
}
}
if time.Now().After(deadline) {
return false, nil
}
select {
case <-ctx.Done():
return false, ctx.Err()
case <-time.After(usernsProbePoll):
}
}
}
+144
View File
@@ -0,0 +1,144 @@
package build
import (
"context"
"sync/atomic"
"testing"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
// "auto" follows the probe through every copy of the Config; "on" and "off"
// ignore it.
func TestUserNamespacesMode(t *testing.T) {
probe := new(atomic.Bool)
for _, tc := range []struct {
mode string
probe bool
want bool
}{
{"", false, false}, {"", true, true}, {UserNamespacesAuto, true, true},
{UserNamespacesOn, false, true}, {UserNamespacesOff, true, false},
} {
b, _, jb := newBuilder()
b.Config.UserNamespaces = tc.mode
b.Config.UserNamespacesProbe = probe
probe.Store(tc.probe)
if _, err := b.Submit(context.Background(), goodRequest()); err != nil {
t.Fatalf("Submit: %v", err)
}
if got := jb.created[0].UserNamespaces; got != tc.want {
t.Errorf("mode %q, probe %v: UserNamespaces = %v, want %v", tc.mode, tc.probe, got, tc.want)
}
}
b, _, jb := newBuilder()
if _, err := b.Submit(context.Background(), goodRequest()); err != nil || jb.created[0].UserNamespaces {
t.Errorf("no probe wired: err %v, UserNamespaces %v", err, jb.created[0].UserNamespaces)
}
}
// The probe pod has the shape that needs user-namespace support, and runs
// nothing that matters.
func TestUsernsProbeJobShape(t *testing.T) {
job := UsernsProbeJob("felis-build", "felis-build", "felis:test", "userns-probe-x")
spec := job.Spec.Template.Spec
if spec.HostUsers == nil || *spec.HostUsers {
t.Fatal("the probe must ask for hostUsers: false")
}
if spec.AutomountServiceAccountToken == nil || *spec.AutomountServiceAccountToken || spec.ServiceAccountName != "felis-build" {
t.Error("the probe must run as the bare build SA with no token")
}
c := spec.Containers[0]
if c.Image != "felis:test" || len(c.Args) != 1 || c.Args[0] != "version" {
t.Errorf("probe container runs %s %v", c.Image, c.Args)
}
if c.SecurityContext.RunAsUser == nil || *c.SecurityContext.RunAsUser != 0 || len(c.SecurityContext.Capabilities.Add) != 3 {
t.Error("the probe must run as root with kaniko's capabilities, the case user namespaces must carry")
}
if len(spec.Volumes) != 1 || spec.Volumes[0].EmptyDir == nil {
t.Error("the probe must mount an emptyDir, as build pods do")
}
if job.Labels[LabelManagedBy] != managedByValue {
t.Error("the probe must sit under the build egress policy")
}
}
func TestProbeUserNamespaces(t *testing.T) {
timeout, poll := usernsProbeTimeout, usernsProbePoll
t.Cleanup(func() { usernsProbeTimeout, usernsProbePoll = timeout, poll })
usernsProbePoll = 5 * time.Millisecond
scheme := runtime.NewScheme()
_ = clientgoscheme.AddToScheme(scheme)
for _, tc := range []struct {
name string
cond batchv1.JobConditionType
want bool
}{
{"complete", batchv1.JobComplete, true},
{"failed", batchv1.JobFailed, false},
{"never finishes", "", false},
} {
t.Run(tc.name, func(t *testing.T) {
// A finished probe must answer long before the timeout would.
usernsProbeTimeout = 5 * time.Second
if tc.cond == "" {
usernsProbeTimeout = 100 * time.Millisecond
}
start := time.Now()
cl := fake.NewClientBuilder().WithScheme(scheme).WithStatusSubresource(&batchv1.Job{}).Build()
jobs := NewK8sJobs(cl, Config{})
type result struct {
ok bool
err error
}
done := make(chan result, 1)
go func() {
ok, err := jobs.ProbeUserNamespaces(context.Background(), "felis:test")
done <- result{ok, err}
}()
if tc.cond != "" {
finishProbe(t, cl, tc.cond)
}
r := <-done
if r.err != nil || r.ok != tc.want {
t.Fatalf("ProbeUserNamespaces = (%v, %v), want (%v, nil)", r.ok, r.err, tc.want)
}
if tc.cond != "" && time.Since(start) > 2*time.Second {
t.Fatalf("the probe answered after %s: it waited out the timeout instead of reading the verdict", time.Since(start))
}
var left batchv1.JobList
if err := cl.List(context.Background(), &left, client.InNamespace(defaultNamespace)); err != nil || len(left.Items) != 0 {
t.Fatalf("probe jobs left behind: %d (%v)", len(left.Items), err)
}
})
}
}
// finishProbe waits for the probe Job to appear and marks it finished.
func finishProbe(t *testing.T, cl client.Client, cond batchv1.JobConditionType) {
t.Helper()
deadline := time.Now().Add(time.Second)
for time.Now().Before(deadline) {
var jobs batchv1.JobList
if err := cl.List(context.Background(), &jobs, client.InNamespace(defaultNamespace)); err != nil {
t.Fatal(err)
}
if len(jobs.Items) == 1 {
job := jobs.Items[0]
job.Status.Conditions = []batchv1.JobCondition{{Type: cond, Status: corev1.ConditionTrue}}
if err := cl.Status().Update(context.Background(), &job); err != nil {
t.Fatal(err)
}
return
}
time.Sleep(2 * time.Millisecond)
}
t.Fatal("the probe job never appeared")
}
+15
View File
@@ -161,6 +161,16 @@ type RegistryConfig struct {
TrivyImage string `toml:"trivy_image"`
BuildCPULimit string `toml:"build_cpu_limit"`
BuildMemLimit string `toml:"build_mem_limit"`
// BuildDiskLimit caps a build pod's ephemeral storage (context, unpacked base
// image and image tarball together). Empty keeps 12Gi.
BuildDiskLimit string `toml:"build_disk_limit"`
// BuildUserNamespaces runs build pods in a user namespace (hostUsers: false):
// "auto" (the default) turns it on when felis-api's startup probe pod ran
// with it, "on" always, "off" never.
BuildUserNamespaces string `toml:"build_user_namespaces"`
// BuildRuntimeClass runs build pods under a sandbox RuntimeClass such as
// gVisor or Kata. Empty runs them under the node's default runtime.
BuildRuntimeClass string `toml:"build_runtime_class"`
// TrivyDBRepository points Trivy at an OCI repository holding the
// vulnerability DB (--db-repository). Trivy's default fetches from
// mirror.gcr.io/ghcr.io, which the build egress lock denies — so on a
@@ -448,6 +458,11 @@ func (c *Config) Validate() error {
if c.Registry.URL != "" && strings.Contains(c.Registry.URL, "://") {
return fmt.Errorf("config: [registry] url %q must be a bare host[:port] with no scheme (e.g. registry.felis.svc:5000); a scheme breaks the user-modpack build lane's derived push target", c.Registry.URL)
}
switch c.Registry.BuildUserNamespaces {
case "", "auto", "on", "off":
default:
return fmt.Errorf("config: [registry] build_user_namespaces %q must be auto, on or off", c.Registry.BuildUserNamespaces)
}
// [smtp] is optional as a whole, but once a host is named the block must be
// deliverable: a From address (relays reject MAIL FROM:<>) and a sane port.
// Fail at load, not at the first OTP a player is waiting on.
+20
View File
@@ -51,6 +51,9 @@ kaniko_image = "registry.felis.svc:5000/mirror/kaniko:v1.23.2"
trivy_image = "registry.felis.svc:5000/mirror/trivy:0.58.1"
build_cpu_limit = "1"
build_mem_limit = "2Gi"
build_disk_limit = "8Gi"
build_user_namespaces = "off"
build_runtime_class = "gvisor"
[archive]
store = "tarLocal"
@@ -95,6 +98,23 @@ func TestLoadValid(t *testing.T) {
if cfg.Registry.BuildCPULimit != "1" || cfg.Registry.BuildMemLimit != "2Gi" {
t.Errorf("build resource overrides = %q / %q", cfg.Registry.BuildCPULimit, cfg.Registry.BuildMemLimit)
}
if r := cfg.Registry; r.BuildDiskLimit != "8Gi" || r.BuildUserNamespaces != "off" || r.BuildRuntimeClass != "gvisor" {
t.Errorf("build isolation overrides = %q / %q / %q", r.BuildDiskLimit, r.BuildUserNamespaces, r.BuildRuntimeClass)
}
}
func TestLoadRejectsUnknownBuildUserNamespaces(t *testing.T) {
_, err := config.Load(writeTOML(t, `
[server]
root_domain = "mc.example.net"
[database]
url = "postgres://felis@db/felis"
[registry]
build_user_namespaces = "yes"
`))
if err == nil || !strings.Contains(err.Error(), "build_user_namespaces") {
t.Fatalf("err = %v, want build_user_namespaces rejected", err)
}
}
func TestLoadAppliesDefaults(t *testing.T) {
+5 -2
View File
@@ -27,8 +27,8 @@ type Object interface {
// namespaced Roles + RoleBindings — felis-api and felis-operator always, plus the
// destructive felis-reaper identity only when retention is enabled, gated with its
// CronJob), the two weak Job SAs (build/restore, which have NO Role anywhere), the
// build-namespace egress NetworkPolicy, and the minecraft-namespace ingress
// NetworkPolicies.
// build-namespace egress NetworkPolicy and LimitRange, and the minecraft-namespace
// ingress NetworkPolicies.
//
// Scope: this is the authorization + network fence (spec §21, §22) plus the
// running control-plane workloads it fences — the felis-api / felis-operator
@@ -93,6 +93,9 @@ func Objects(p Params) []Object {
})
buildNP.TypeMeta = metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"}
objs = append(objs, buildNP)
buildLR := build.BuildLimitRange(p.BuildNamespace)
buildLR.TypeMeta = metav1.TypeMeta{APIVersion: "v1", Kind: "LimitRange"}
objs = append(objs, buildLR)
// Minecraft-namespace ingress fence (default-deny + RCON + game), the server
// egress fence, and the registry's ingress fence.