fix(build): 构建 pod 等出口策略生效再运行,加 seccomp、可选 user namespace 与磁盘上限,上下文解包限总字节与条目数

This commit is contained in:
Lemon-miaow committed 2026-09-24 20:42:14 +08:00
1 parent 9bda3a52fa
commit 13b65e19ec
17 files changed
+948 -29

No files matched your search

+42
View File
@@ -37,6 +37,7 @@ import (
"errors"
"fmt"
"strings"
"sync/atomic"
"time"
"felis.lolicon.best/internal/metrics"
@@ -243,6 +244,37 @@ type Config struct {
// CPULimit / MemLimit cap each build container (spec §16: resource limits).
CPULimit string
MemLimit string
// DiskLimit caps the build pod's ephemeral storage: the extracted context,
// the base image kaniko unpacks and the image tarball together.
DiskLimit string
// UserNamespaces selects hostUsers: false for build pods: UserNamespacesOn,
// UserNamespacesOff, or UserNamespacesAuto (the default), which follows
// UserNamespacesProbe.
UserNamespaces string
// UserNamespacesProbe carries ProbeUserNamespaces' verdict to every copy of
// this Config; nil or false keeps "auto" off.
UserNamespacesProbe *atomic.Bool
// RuntimeClass runs build pods under a sandbox RuntimeClass (gVisor, Kata)
// when set. The class must exist on the cluster.
RuntimeClass string
}
// Values of Config.UserNamespaces.
const (
UserNamespacesAuto = "auto"
UserNamespacesOn = "on"
UserNamespacesOff = "off"
)
// userNamespaces resolves Config.UserNamespaces for one build.
func (c Config) userNamespaces() bool {
switch c.UserNamespaces {
case UserNamespacesOn:
return true
case UserNamespacesOff:
return false
}
return c.UserNamespacesProbe != nil && c.UserNamespacesProbe.Load()
}
// Defaults applied when a Config field is left zero.
@@ -255,6 +287,7 @@ const (
defaultMaxDockerfile = 256 * 1024 // 256 KiB
defaultCPULimit = "2"
defaultMemLimit = "4Gi"
defaultDiskLimit = "12Gi"
)
// withDefaults returns a copy of c with zero fields filled, so a partially
@@ -284,6 +317,12 @@ func (c Config) withDefaults() Config {
if c.MemLimit == "" {
c.MemLimit = defaultMemLimit
}
if c.DiskLimit == "" {
c.DiskLimit = defaultDiskLimit
}
if c.UserNamespaces == "" {
c.UserNamespaces = UserNamespacesAuto
}
return c
}
@@ -380,6 +419,9 @@ func (b *Builder) jobParams(bld *Build, cfg Config) JobParams {
Deadline: cfg.Deadline,
CPULimit: cfg.CPULimit,
MemLimit: cfg.MemLimit,
DiskLimit: cfg.DiskLimit,
UserNamespaces: cfg.userNamespaces(),
RuntimeClass: cfg.RuntimeClass,
}
}
+141 -10
View File
@@ -33,6 +33,7 @@ const (
// build-log streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8)
// follows the same container this Job defines — one source of truth for the name.
const (
ContainerGate = "egress-gate"
ContainerKaniko = "kaniko"
ContainerTrivy = "trivy"
ContainerPush = "push"
@@ -64,6 +65,23 @@ var imageSizeLimit = resource.MustParse("10Gi")
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
var contextSizeLimit = resource.MustParse("4Gi")
// Per-container ephemeral-storage bounds (writable layer + logs; emptyDirs count
// toward the pod as a whole). Kaniko's limit is the operator's disk cap because
// kaniko unpacks the base image into its own root filesystem, which no emptyDir
// bound covers; it is also the largest limit in the pod, so it becomes the
// pod-level cap the kubelet holds context + unpacked rootfs + image tarball to.
// Trivy keeps its vulnerability and Java DBs (about 1.4 GiB live) in its layer.
// The others write nothing but logs.
var (
gateDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("64Mi")}
fetchDisk = diskBounds{request: resource.MustParse("64Mi"), limit: resource.MustParse("256Mi")}
kanikoDiskRq = resource.MustParse("1Gi")
trivyDisk = diskBounds{request: resource.MustParse("256Mi"), limit: resource.MustParse("4Gi")}
pushDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("256Mi")}
)
type diskBounds struct{ request, limit resource.Quantity }
// buildJobTTL is how long a finished build Job survives before the Job
// controller deletes it — and with it the Pod whose kaniko log is the admin
// failure-triage surface (GET /api/v1/images/build/{id}/logs).
@@ -105,6 +123,15 @@ type JobParams struct {
Deadline time.Duration
CPULimit string
MemLimit string
// DiskLimit caps kaniko's ephemeral storage, and with it the pod's (see
// kanikoDiskRq). Empty applies defaultDiskLimit.
DiskLimit string
// UserNamespaces runs the pod with hostUsers: false, so root in the build
// containers is an unprivileged uid on the node. It needs a kernel and runtime
// with idmapped mounts; Config.UserNamespaces decides.
UserNamespaces bool
// RuntimeClass, when set, runs the pod under that RuntimeClass (gVisor, Kata).
RuntimeClass string
}
// BuildJobName is the deterministic Job name for a build id.
@@ -127,8 +154,14 @@ func buildLabels(p JobParams) map[string]string {
// reach the K8s API (spec §16, §21);
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
// so docker-in-docker / privileged is never needed (spec §16, §22);
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
// - the RuntimeDefault seccomp profile on the whole pod, and optionally a user
// namespace (hostUsers: false) and a sandbox RuntimeClass, because kaniko is
// no isolation boundary: the Dockerfile's RUN steps execute in its container;
// - an egress gate ahead of everything else, so nothing runs before the
// namespace's NetworkPolicy is enforced for this pod (cmd/felis egress-gate);
// - activeDeadlineSeconds + backoffLimit=0 + per-container CPU, memory and
// ephemeral-storage limits so a runaway or poisoned build cannot exhaust the
// node (spec §16);
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
// automatic admission gate (spec §16).
@@ -149,6 +182,18 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
if err != nil {
return nil, err
}
diskCap := p.DiskLimit
if diskCap == "" {
diskCap = defaultDiskLimit
}
kanikoDisk, err := resource.ParseQuantity(diskCap)
if err != nil {
return nil, fmt.Errorf("build: invalid disk limit %q: %w", diskCap, err)
}
kanikoRq := kanikoDiskRq.DeepCopy()
if kanikoDisk.Cmp(kanikoRq) < 0 {
kanikoRq = kanikoDisk.DeepCopy()
}
if p.FelisImage == "" {
return nil, fmt.Errorf("build: FelisImage is required: the push container runs it")
}
@@ -187,7 +232,35 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
contextPath := p.ContextRef
initContainers := []corev1.Container{}
// The gate runs before anything else. A new pod's NetworkPolicy is programmed
// asynchronously: live on k3s (kube-router), a build-labelled pod reached the
// internet and the Kubernetes API for the first ~0.7 s of its life. The gate
// holds the pod until a destination the policy denies stops answering, so the
// Dockerfile never runs inside that window.
gateSec := sec.DeepCopy()
gateSec.ReadOnlyRootFilesystem = boolPtr(true)
gateSec.RunAsNonRoot = boolPtr(true)
gateSec.RunAsUser = int64Ptr(nonRootUID)
gate := corev1.Container{
Name: ContainerGate,
Image: p.FelisImage,
Args: []string{"egress-gate"},
Resources: corev1.ResourceRequirements{
Limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("100m"),
corev1.ResourceMemory: resource.MustParse("64Mi"),
corev1.ResourceEphemeralStorage: gateDisk.limit,
},
Requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("10m"),
corev1.ResourceMemory: resource.MustParse("16Mi"),
corev1.ResourceEphemeralStorage: gateDisk.request,
},
},
SecurityContext: gateSec,
}
initContainers := []corev1.Container{gate}
imageMount := corev1.VolumeMount{Name: imageVolume, MountPath: imageMountPath}
kanikoMounts := []corev1.VolumeMount{imageMount}
podVolumes := []corev1.Volume{{
@@ -232,7 +305,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
}},
}},
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, fetchDisk),
SecurityContext: fetchSec,
}
initContainers = append(initContainers, fetch)
@@ -269,7 +342,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
"--skip-tls-verify-pull",
},
VolumeMounts: kanikoMounts,
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, diskBounds{request: kanikoRq, limit: kanikoDisk}),
SecurityContext: kanikoSec,
}
initContainers = append(initContainers, kaniko)
@@ -298,7 +371,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
Image: p.TrivyImage,
Args: trivyArgs,
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, trivyDisk),
SecurityContext: sec,
}
initContainers = append(initContainers, trivy)
@@ -328,7 +401,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
secretEnv("FELIS_REGISTRY_PASSWORD", naming.RegistryPushPasswordKey),
},
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
Resources: withDisk(limits, pushDisk),
SecurityContext: pushSec,
}
@@ -350,16 +423,44 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
InitContainers: initContainers,
Containers: []corev1.Container{push},
Volumes: podVolumes,
// RUN steps execute in kaniko's container with root and three
// capabilities; RuntimeDefault takes away the syscalls a container
// never needs, among them most kernel-escape primitives (unshare,
// mount, keyctl, bpf). Every step was checked to run under it live.
SecurityContext: &corev1.PodSecurityContext{
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
},
InitContainers: initContainers,
Containers: []corev1.Container{push},
Volumes: podVolumes,
},
},
},
}
if p.UserNamespaces {
job.Spec.Template.Spec.HostUsers = boolPtr(false)
}
if p.RuntimeClass != "" {
rc := p.RuntimeClass
job.Spec.Template.Spec.RuntimeClassName = &rc
}
return job, nil
}
// nonRootUID is the distroless nonroot user the platform image ships as.
const nonRootUID = 65532
// withDisk is the resources block for one build container: the CPU and memory
// caps with their schedulable floor (buildRequests), plus its ephemeral-storage
// request and limit.
func withDisk(limits corev1.ResourceList, d diskBounds) corev1.ResourceRequirements {
lim := limits.DeepCopy()
lim[corev1.ResourceEphemeralStorage] = d.limit
req := buildRequests(limits)
req[corev1.ResourceEphemeralStorage] = d.request
return corev1.ResourceRequirements{Limits: lim, Requests: req}
}
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
// lane derives when the API is the blob transport — i.e. a context only the
// fetch initContainer can turn into a local path for Kaniko.
@@ -516,6 +617,36 @@ func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
}
}
// BuildLimitRange bounds any container in the build namespace that arrives
// without its own limits. Build Jobs set every limit themselves (BuildJob); this
// is the backstop for anything else that lands in the namespace, which shares
// the node's disk with the game worlds.
func BuildLimitRange(namespace string) *corev1.LimitRange {
return &corev1.LimitRange{
ObjectMeta: metav1.ObjectMeta{
Name: "felis-build-limits",
Namespace: namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
Spec: corev1.LimitRangeSpec{Limits: []corev1.LimitRangeItem{{
Type: corev1.LimitTypeContainer,
Default: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("1"),
corev1.ResourceMemory: resource.MustParse("1Gi"),
corev1.ResourceEphemeralStorage: resource.MustParse("1Gi"),
},
DefaultRequest: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("100m"),
corev1.ResourceMemory: resource.MustParse("128Mi"),
corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"),
},
}}},
}
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
+116 -10
View File
@@ -171,10 +171,10 @@ func TestBuildJobScansBeforePush(t *testing.T) {
t.Fatalf("BuildJob: %v", err)
}
inits := job.Spec.Template.Spec.InitContainers
if len(inits) != 2 || inits[0].Name != ContainerKaniko || inits[1].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [kaniko trivy]", initNames(inits))
if len(inits) != 3 || inits[0].Name != ContainerGate || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [egress-gate kaniko trivy]", initNames(inits))
}
kaniko, trivy := inits[0], inits[1]
kaniko, trivy := inits[1], inits[2]
for _, want := range []string{"--destination=" + p.ImageRef, "--no-push", "--tar-path=" + imageTarPath} {
if !hasArg(kaniko.Args, want) {
t.Errorf("kaniko args = %v, want %s", kaniko.Args, want)
@@ -339,9 +339,9 @@ func TestBuildJobTrivyDBRepositoryOverride(t *testing.T) {
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
trivy := job.Spec.Template.Spec.InitContainers[1]
trivy := job.Spec.Template.Spec.InitContainers[2]
if trivy.Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want trivy second", initNames(job.Spec.Template.Spec.InitContainers))
t.Fatalf("initContainers = %v, want trivy third", initNames(job.Spec.Template.Spec.InitContainers))
}
if !argPairPresent(trivy.Args, "--db-repository", p.TrivyDBRepository) {
t.Errorf("trivy args = %v, want --db-repository %s", trivy.Args, p.TrivyDBRepository)
@@ -423,10 +423,11 @@ func TestBuildJobFetchesHTTPContext(t *testing.T) {
t.Fatalf("BuildJob: %v", err)
}
inits := job.Spec.Template.Spec.InitContainers
if len(inits) != 3 || inits[0].Name != ContainerFetch || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [%s %s %s]", initNames(inits), ContainerFetch, ContainerKaniko, ContainerTrivy)
if len(inits) != 4 || inits[0].Name != ContainerGate || inits[1].Name != ContainerFetch ||
inits[2].Name != ContainerKaniko || inits[3].Name != ContainerTrivy {
t.Fatalf("initContainers = %v, want [%s %s %s %s]", initNames(inits), ContainerGate, ContainerFetch, ContainerKaniko, ContainerTrivy)
}
fetch, kaniko := inits[0], inits[1]
fetch, kaniko := inits[1], inits[2]
if fetch.Image != p.FelisImage {
t.Errorf("fetch image = %q, want the platform image %q", fetch.Image, p.FelisImage)
}
@@ -494,8 +495,8 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) {
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 2 || inits[0].Name != ContainerKaniko {
t.Errorf("a native ref must render just kaniko + trivy, got %v", initNames(inits))
if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 3 || inits[1].Name != ContainerKaniko {
t.Errorf("a native ref must render just the gate, kaniko and trivy, got %v", initNames(inits))
}
for _, v := range job.Spec.Template.Spec.Volumes {
if v.Name == contextVolume {
@@ -504,6 +505,111 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) {
}
}
// Nothing runs before the egress gate, and the gate itself holds nothing: no
// credential, no root, no writable filesystem.
func TestBuildJobGatesEgressFirst(t *testing.T) {
p := sampleJobParams()
p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx"
job, err := BuildJob(p)
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
gate := job.Spec.Template.Spec.InitContainers[0]
if gate.Name != ContainerGate || gate.Image != p.FelisImage || len(gate.Args) != 1 || gate.Args[0] != "egress-gate" {
t.Fatalf("first initContainer = %s %s %v, want the platform image's egress-gate", gate.Name, gate.Image, gate.Args)
}
if len(gate.Env) != 0 || len(gate.VolumeMounts) != 0 {
t.Errorf("the gate must hold nothing, got env %v mounts %v", gate.Env, gate.VolumeMounts)
}
sc := gate.SecurityContext
if sc == nil || sc.RunAsNonRoot == nil || !*sc.RunAsNonRoot || sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
t.Errorf("the gate must run non-root on a read-only root, got %#v", sc)
}
}
// The pod runs under RuntimeDefault seccomp always, and in a user namespace or
// a sandbox runtime when the install asks for them.
func TestBuildJobSandboxing(t *testing.T) {
job, err := BuildJob(sampleJobParams())
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
spec := job.Spec.Template.Spec
if spec.SecurityContext == nil || spec.SecurityContext.SeccompProfile == nil ||
spec.SecurityContext.SeccompProfile.Type != corev1.SeccompProfileTypeRuntimeDefault {
t.Fatalf("pod securityContext = %#v, want seccompProfile RuntimeDefault", spec.SecurityContext)
}
if spec.HostUsers != nil || spec.RuntimeClassName != nil {
t.Errorf("defaults must leave hostUsers and runtimeClassName unset, got %v / %v", spec.HostUsers, spec.RuntimeClassName)
}
p := sampleJobParams()
p.UserNamespaces, p.RuntimeClass = true, "gvisor"
if job, err = BuildJob(p); err != nil {
t.Fatalf("BuildJob: %v", err)
}
spec = job.Spec.Template.Spec
if spec.HostUsers == nil || *spec.HostUsers {
t.Errorf("UserNamespaces must render hostUsers: false, got %v", spec.HostUsers)
}
if spec.RuntimeClassName == nil || *spec.RuntimeClassName != "gvisor" {
t.Errorf("runtimeClassName = %v, want gvisor", spec.RuntimeClassName)
}
}
// Every container carries an ephemeral-storage request and limit, and kaniko's
// limit, the largest, is the configured disk cap.
func TestBuildJobBoundsEphemeralStorage(t *testing.T) {
p := sampleJobParams()
p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx"
job, err := BuildJob(p)
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
all := append(append([]corev1.Container{}, job.Spec.Template.Spec.InitContainers...), job.Spec.Template.Spec.Containers...)
for _, c := range all {
lim, lok := c.Resources.Limits[corev1.ResourceEphemeralStorage]
req, rok := c.Resources.Requests[corev1.ResourceEphemeralStorage]
if !lok || !rok || lim.IsZero() || req.Cmp(lim) > 0 {
t.Errorf("container %s ephemeral-storage request %v limit %v", c.Name, req.String(), lim.String())
}
if c.Name == ContainerKaniko && lim.String() != defaultDiskLimit {
t.Errorf("kaniko ephemeral-storage limit = %s, want the default %s", lim.String(), defaultDiskLimit)
}
}
p.DiskLimit = "512Mi"
if job, err = BuildJob(p); err != nil {
t.Fatalf("BuildJob: %v", err)
}
for _, c := range job.Spec.Template.Spec.InitContainers {
if c.Name != ContainerKaniko {
continue
}
lim := c.Resources.Limits[corev1.ResourceEphemeralStorage]
req := c.Resources.Requests[corev1.ResourceEphemeralStorage]
if lim.String() != "512Mi" || req.Cmp(lim) > 0 {
t.Errorf("a 512Mi disk cap rendered limit %s request %s", lim.String(), req.String())
}
}
p.DiskLimit = "lots"
if _, err := BuildJob(p); err == nil {
t.Error("an unparsable disk limit was accepted")
}
}
func TestBuildLimitRangeCoversEphemeralStorage(t *testing.T) {
lr := BuildLimitRange("felis-build")
if lr.Namespace != "felis-build" || len(lr.Spec.Limits) != 1 {
t.Fatalf("limit range = %#v", lr)
}
item := lr.Spec.Limits[0]
for _, res := range []corev1.ResourceName{corev1.ResourceCPU, corev1.ResourceMemory, corev1.ResourceEphemeralStorage} {
d, dr := item.Default[res], item.DefaultRequest[res]
if d.IsZero() || dr.IsZero() || dr.Cmp(d) > 0 {
t.Errorf("%s default %s request %s", res, d.String(), dr.String())
}
}
}
func initNames(cs []corev1.Container) []string {
names := make([]string, 0, len(cs))
for _, c := range cs {
+128
View File
@@ -0,0 +1,128 @@
package build
import (
"context"
"fmt"
"strconv"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// Vars so tests can shrink them. The probe pod's image is the api's own, so it is
// already on the node; the timeout covers a slow pod start, and a runtime that
// cannot do user namespaces fails the pod well within it.
var (
usernsProbeTimeout = 3 * time.Minute
usernsProbePoll = 2 * time.Second
)
// UsernsProbeJob renders the one-shot Job that asks the cluster whether a build
// pod can run with hostUsers: false. The pod has the parts of a build pod that
// need idmapped mounts and namespaced capabilities: root with kaniko's three
// capabilities, the RuntimeDefault seccomp profile and an emptyDir. It runs
// `felis version`, which touches nothing.
func UsernsProbeJob(namespace, serviceAccount, image, name string) *batchv1.Job {
labels := map[string]string{LabelManagedBy: managedByValue, LabelComponent: "userns-probe"}
small := corev1.ResourceRequirements{
Limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("100m"),
corev1.ResourceMemory: resource.MustParse("64Mi"),
corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"),
},
Requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("10m"),
corev1.ResourceMemory: resource.MustParse("16Mi"),
corev1.ResourceEphemeralStorage: resource.MustParse("16Mi"),
},
}
return &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: namespace, Labels: labels},
Spec: batchv1.JobSpec{
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(int64(usernsProbeTimeout / time.Second)),
TTLSecondsAfterFinished: int32Ptr(300),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: labels},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: serviceAccount,
AutomountServiceAccountToken: boolPtr(false),
HostUsers: boolPtr(false),
SecurityContext: &corev1.PodSecurityContext{
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
},
Containers: []corev1.Container{{
Name: "probe",
Image: image,
Args: []string{"version"},
Resources: small,
SecurityContext: &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
RunAsUser: int64Ptr(0),
Capabilities: &corev1.Capabilities{
Drop: []corev1.Capability{"ALL"},
Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE", "FOWNER"},
},
},
VolumeMounts: []corev1.VolumeMount{{Name: "scratch", MountPath: "/scratch"}},
}},
Volumes: []corev1.Volume{{
Name: "scratch",
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{SizeLimit: quantityPtr(resource.MustParse("16Mi"))}},
}},
},
},
},
}
}
// ProbeUserNamespaces runs UsernsProbeJob and reports whether its pod succeeded.
// A pod that fails, or never starts before the timeout, answers false; err is set
// only when the Job could not be created or read. The Job is deleted afterwards.
func (k *K8sJobs) ProbeUserNamespaces(ctx context.Context, image string) (bool, error) {
name := "userns-probe-" + strconv.FormatInt(time.Now().UnixNano(), 36)
job := UsernsProbeJob(k.cfg.Namespace, k.cfg.ServiceAccount, image, name)
if err := k.c.Create(ctx, job); err != nil {
return false, fmt.Errorf("create the probe job: %w", err)
}
defer func() {
bg := metav1.DeletePropagationBackground
del := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: k.cfg.Namespace, Name: name}}
_ = k.c.Delete(context.WithoutCancel(ctx), del, &client.DeleteOptions{PropagationPolicy: &bg})
}()
deadline := time.Now().Add(usernsProbeTimeout)
for {
var got batchv1.Job
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.cfg.Namespace, Name: name}, &got)
if err != nil && !apierrors.IsNotFound(err) {
return false, fmt.Errorf("read the probe job: %w", err)
}
for _, cond := range got.Status.Conditions {
if cond.Status != corev1.ConditionTrue {
continue
}
switch cond.Type {
case batchv1.JobComplete:
return true, nil
case batchv1.JobFailed:
return false, nil
}
}
if time.Now().After(deadline) {
return false, nil
}
select {
case <-ctx.Done():
return false, ctx.Err()
case <-time.After(usernsProbePoll):
}
}
}
+144
View File
@@ -0,0 +1,144 @@
package build
import (
"context"
"sync/atomic"
"testing"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
// "auto" follows the probe through every copy of the Config; "on" and "off"
// ignore it.
func TestUserNamespacesMode(t *testing.T) {
probe := new(atomic.Bool)
for _, tc := range []struct {
mode string
probe bool
want bool
}{
{"", false, false}, {"", true, true}, {UserNamespacesAuto, true, true},
{UserNamespacesOn, false, true}, {UserNamespacesOff, true, false},
} {
b, _, jb := newBuilder()
b.Config.UserNamespaces = tc.mode
b.Config.UserNamespacesProbe = probe
probe.Store(tc.probe)
if _, err := b.Submit(context.Background(), goodRequest()); err != nil {
t.Fatalf("Submit: %v", err)
}
if got := jb.created[0].UserNamespaces; got != tc.want {
t.Errorf("mode %q, probe %v: UserNamespaces = %v, want %v", tc.mode, tc.probe, got, tc.want)
}
}
b, _, jb := newBuilder()
if _, err := b.Submit(context.Background(), goodRequest()); err != nil || jb.created[0].UserNamespaces {
t.Errorf("no probe wired: err %v, UserNamespaces %v", err, jb.created[0].UserNamespaces)
}
}
// The probe pod has the shape that needs user-namespace support, and runs
// nothing that matters.
func TestUsernsProbeJobShape(t *testing.T) {
job := UsernsProbeJob("felis-build", "felis-build", "felis:test", "userns-probe-x")
spec := job.Spec.Template.Spec
if spec.HostUsers == nil || *spec.HostUsers {
t.Fatal("the probe must ask for hostUsers: false")
}
if spec.AutomountServiceAccountToken == nil || *spec.AutomountServiceAccountToken || spec.ServiceAccountName != "felis-build" {
t.Error("the probe must run as the bare build SA with no token")
}
c := spec.Containers[0]
if c.Image != "felis:test" || len(c.Args) != 1 || c.Args[0] != "version" {
t.Errorf("probe container runs %s %v", c.Image, c.Args)
}
if c.SecurityContext.RunAsUser == nil || *c.SecurityContext.RunAsUser != 0 || len(c.SecurityContext.Capabilities.Add) != 3 {
t.Error("the probe must run as root with kaniko's capabilities, the case user namespaces must carry")
}
if len(spec.Volumes) != 1 || spec.Volumes[0].EmptyDir == nil {
t.Error("the probe must mount an emptyDir, as build pods do")
}
if job.Labels[LabelManagedBy] != managedByValue {
t.Error("the probe must sit under the build egress policy")
}
}
func TestProbeUserNamespaces(t *testing.T) {
timeout, poll := usernsProbeTimeout, usernsProbePoll
t.Cleanup(func() { usernsProbeTimeout, usernsProbePoll = timeout, poll })
usernsProbePoll = 5 * time.Millisecond
scheme := runtime.NewScheme()
_ = clientgoscheme.AddToScheme(scheme)
for _, tc := range []struct {
name string
cond batchv1.JobConditionType
want bool
}{
{"complete", batchv1.JobComplete, true},
{"failed", batchv1.JobFailed, false},
{"never finishes", "", false},
} {
t.Run(tc.name, func(t *testing.T) {
// A finished probe must answer long before the timeout would.
usernsProbeTimeout = 5 * time.Second
if tc.cond == "" {
usernsProbeTimeout = 100 * time.Millisecond
}
start := time.Now()
cl := fake.NewClientBuilder().WithScheme(scheme).WithStatusSubresource(&batchv1.Job{}).Build()
jobs := NewK8sJobs(cl, Config{})
type result struct {
ok bool
err error
}
done := make(chan result, 1)
go func() {
ok, err := jobs.ProbeUserNamespaces(context.Background(), "felis:test")
done <- result{ok, err}
}()
if tc.cond != "" {
finishProbe(t, cl, tc.cond)
}
r := <-done
if r.err != nil || r.ok != tc.want {
t.Fatalf("ProbeUserNamespaces = (%v, %v), want (%v, nil)", r.ok, r.err, tc.want)
}
if tc.cond != "" && time.Since(start) > 2*time.Second {
t.Fatalf("the probe answered after %s: it waited out the timeout instead of reading the verdict", time.Since(start))
}
var left batchv1.JobList
if err := cl.List(context.Background(), &left, client.InNamespace(defaultNamespace)); err != nil || len(left.Items) != 0 {
t.Fatalf("probe jobs left behind: %d (%v)", len(left.Items), err)
}
})
}
}
// finishProbe waits for the probe Job to appear and marks it finished.
func finishProbe(t *testing.T, cl client.Client, cond batchv1.JobConditionType) {
t.Helper()
deadline := time.Now().Add(time.Second)
for time.Now().Before(deadline) {
var jobs batchv1.JobList
if err := cl.List(context.Background(), &jobs, client.InNamespace(defaultNamespace)); err != nil {
t.Fatal(err)
}
if len(jobs.Items) == 1 {
job := jobs.Items[0]
job.Status.Conditions = []batchv1.JobCondition{{Type: cond, Status: corev1.ConditionTrue}}
if err := cl.Status().Update(context.Background(), &job); err != nil {
t.Fatal(err)
}
return
}
time.Sleep(2 * time.Millisecond)
}
t.Fatal("the probe job never appeared")
}
+15
View File
@@ -161,6 +161,16 @@ type RegistryConfig struct {
TrivyImage string `toml:"trivy_image"`
BuildCPULimit string `toml:"build_cpu_limit"`
BuildMemLimit string `toml:"build_mem_limit"`
// BuildDiskLimit caps a build pod's ephemeral storage (context, unpacked base
// image and image tarball together). Empty keeps 12Gi.
BuildDiskLimit string `toml:"build_disk_limit"`
// BuildUserNamespaces runs build pods in a user namespace (hostUsers: false):
// "auto" (the default) turns it on when felis-api's startup probe pod ran
// with it, "on" always, "off" never.
BuildUserNamespaces string `toml:"build_user_namespaces"`
// BuildRuntimeClass runs build pods under a sandbox RuntimeClass such as
// gVisor or Kata. Empty runs them under the node's default runtime.
BuildRuntimeClass string `toml:"build_runtime_class"`
// TrivyDBRepository points Trivy at an OCI repository holding the
// vulnerability DB (--db-repository). Trivy's default fetches from
// mirror.gcr.io/ghcr.io, which the build egress lock denies — so on a
@@ -448,6 +458,11 @@ func (c *Config) Validate() error {
if c.Registry.URL != "" && strings.Contains(c.Registry.URL, "://") {
return fmt.Errorf("config: [registry] url %q must be a bare host[:port] with no scheme (e.g. registry.felis.svc:5000); a scheme breaks the user-modpack build lane's derived push target", c.Registry.URL)
}
switch c.Registry.BuildUserNamespaces {
case "", "auto", "on", "off":
default:
return fmt.Errorf("config: [registry] build_user_namespaces %q must be auto, on or off", c.Registry.BuildUserNamespaces)
}
// [smtp] is optional as a whole, but once a host is named the block must be
// deliverable: a From address (relays reject MAIL FROM:<>) and a sane port.
// Fail at load, not at the first OTP a player is waiting on.
+20
View File
@@ -51,6 +51,9 @@ kaniko_image = "registry.felis.svc:5000/mirror/kaniko:v1.23.2"
trivy_image = "registry.felis.svc:5000/mirror/trivy:0.58.1"
build_cpu_limit = "1"
build_mem_limit = "2Gi"
build_disk_limit = "8Gi"
build_user_namespaces = "off"
build_runtime_class = "gvisor"
[archive]
store = "tarLocal"
@@ -95,6 +98,23 @@ func TestLoadValid(t *testing.T) {
if cfg.Registry.BuildCPULimit != "1" || cfg.Registry.BuildMemLimit != "2Gi" {
t.Errorf("build resource overrides = %q / %q", cfg.Registry.BuildCPULimit, cfg.Registry.BuildMemLimit)
}
if r := cfg.Registry; r.BuildDiskLimit != "8Gi" || r.BuildUserNamespaces != "off" || r.BuildRuntimeClass != "gvisor" {
t.Errorf("build isolation overrides = %q / %q / %q", r.BuildDiskLimit, r.BuildUserNamespaces, r.BuildRuntimeClass)
}
}
func TestLoadRejectsUnknownBuildUserNamespaces(t *testing.T) {
_, err := config.Load(writeTOML(t, `
[server]
root_domain = "mc.example.net"
[database]
url = "postgres://felis@db/felis"
[registry]
build_user_namespaces = "yes"
`))
if err == nil || !strings.Contains(err.Error(), "build_user_namespaces") {
t.Fatalf("err = %v, want build_user_namespaces rejected", err)
}
}
func TestLoadAppliesDefaults(t *testing.T) {
+5 -2
View File
@@ -27,8 +27,8 @@ type Object interface {
// namespaced Roles + RoleBindings — felis-api and felis-operator always, plus the
// destructive felis-reaper identity only when retention is enabled, gated with its
// CronJob), the two weak Job SAs (build/restore, which have NO Role anywhere), the
// build-namespace egress NetworkPolicy, and the minecraft-namespace ingress
// NetworkPolicies.
// build-namespace egress NetworkPolicy and LimitRange, and the minecraft-namespace
// ingress NetworkPolicies.
//
// Scope: this is the authorization + network fence (spec §21, §22) plus the
// running control-plane workloads it fences — the felis-api / felis-operator
@@ -93,6 +93,9 @@ func Objects(p Params) []Object {
})
buildNP.TypeMeta = metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"}
objs = append(objs, buildNP)
buildLR := build.BuildLimitRange(p.BuildNamespace)
buildLR.TypeMeta = metav1.TypeMeta{APIVersion: "v1", Kind: "LimitRange"}
objs = append(objs, buildLR)
// Minecraft-namespace ingress fence (default-deny + RCON + game), the server
// egress fence, and the registry's ingress fence.