fix(build): 构建 pod 等出口策略生效再运行,加 seccomp、可选 user namespace 与磁盘上限,上下文解包限总字节与条目数
This commit is contained in:
17 files changed
+948
-29
No files matched your search
@@ -37,6 +37,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/metrics"
|
||||
@@ -243,6 +244,37 @@ type Config struct {
|
||||
// CPULimit / MemLimit cap each build container (spec §16: resource limits).
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
// DiskLimit caps the build pod's ephemeral storage: the extracted context,
|
||||
// the base image kaniko unpacks and the image tarball together.
|
||||
DiskLimit string
|
||||
// UserNamespaces selects hostUsers: false for build pods: UserNamespacesOn,
|
||||
// UserNamespacesOff, or UserNamespacesAuto (the default), which follows
|
||||
// UserNamespacesProbe.
|
||||
UserNamespaces string
|
||||
// UserNamespacesProbe carries ProbeUserNamespaces' verdict to every copy of
|
||||
// this Config; nil or false keeps "auto" off.
|
||||
UserNamespacesProbe *atomic.Bool
|
||||
// RuntimeClass runs build pods under a sandbox RuntimeClass (gVisor, Kata)
|
||||
// when set. The class must exist on the cluster.
|
||||
RuntimeClass string
|
||||
}
|
||||
|
||||
// Values of Config.UserNamespaces.
|
||||
const (
|
||||
UserNamespacesAuto = "auto"
|
||||
UserNamespacesOn = "on"
|
||||
UserNamespacesOff = "off"
|
||||
)
|
||||
|
||||
// userNamespaces resolves Config.UserNamespaces for one build.
|
||||
func (c Config) userNamespaces() bool {
|
||||
switch c.UserNamespaces {
|
||||
case UserNamespacesOn:
|
||||
return true
|
||||
case UserNamespacesOff:
|
||||
return false
|
||||
}
|
||||
return c.UserNamespacesProbe != nil && c.UserNamespacesProbe.Load()
|
||||
}
|
||||
|
||||
// Defaults applied when a Config field is left zero.
|
||||
@@ -255,6 +287,7 @@ const (
|
||||
defaultMaxDockerfile = 256 * 1024 // 256 KiB
|
||||
defaultCPULimit = "2"
|
||||
defaultMemLimit = "4Gi"
|
||||
defaultDiskLimit = "12Gi"
|
||||
)
|
||||
|
||||
// withDefaults returns a copy of c with zero fields filled, so a partially
|
||||
@@ -284,6 +317,12 @@ func (c Config) withDefaults() Config {
|
||||
if c.MemLimit == "" {
|
||||
c.MemLimit = defaultMemLimit
|
||||
}
|
||||
if c.DiskLimit == "" {
|
||||
c.DiskLimit = defaultDiskLimit
|
||||
}
|
||||
if c.UserNamespaces == "" {
|
||||
c.UserNamespaces = UserNamespacesAuto
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
@@ -380,6 +419,9 @@ func (b *Builder) jobParams(bld *Build, cfg Config) JobParams {
|
||||
Deadline: cfg.Deadline,
|
||||
CPULimit: cfg.CPULimit,
|
||||
MemLimit: cfg.MemLimit,
|
||||
DiskLimit: cfg.DiskLimit,
|
||||
UserNamespaces: cfg.userNamespaces(),
|
||||
RuntimeClass: cfg.RuntimeClass,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+141
-10
@@ -33,6 +33,7 @@ const (
|
||||
// build-log streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8)
|
||||
// follows the same container this Job defines — one source of truth for the name.
|
||||
const (
|
||||
ContainerGate = "egress-gate"
|
||||
ContainerKaniko = "kaniko"
|
||||
ContainerTrivy = "trivy"
|
||||
ContainerPush = "push"
|
||||
@@ -64,6 +65,23 @@ var imageSizeLimit = resource.MustParse("10Gi")
|
||||
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
|
||||
var contextSizeLimit = resource.MustParse("4Gi")
|
||||
|
||||
// Per-container ephemeral-storage bounds (writable layer + logs; emptyDirs count
|
||||
// toward the pod as a whole). Kaniko's limit is the operator's disk cap because
|
||||
// kaniko unpacks the base image into its own root filesystem, which no emptyDir
|
||||
// bound covers; it is also the largest limit in the pod, so it becomes the
|
||||
// pod-level cap the kubelet holds context + unpacked rootfs + image tarball to.
|
||||
// Trivy keeps its vulnerability and Java DBs (about 1.4 GiB live) in its layer.
|
||||
// The others write nothing but logs.
|
||||
var (
|
||||
gateDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("64Mi")}
|
||||
fetchDisk = diskBounds{request: resource.MustParse("64Mi"), limit: resource.MustParse("256Mi")}
|
||||
kanikoDiskRq = resource.MustParse("1Gi")
|
||||
trivyDisk = diskBounds{request: resource.MustParse("256Mi"), limit: resource.MustParse("4Gi")}
|
||||
pushDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("256Mi")}
|
||||
)
|
||||
|
||||
type diskBounds struct{ request, limit resource.Quantity }
|
||||
|
||||
// buildJobTTL is how long a finished build Job survives before the Job
|
||||
// controller deletes it — and with it the Pod whose kaniko log is the admin
|
||||
// failure-triage surface (GET /api/v1/images/build/{id}/logs).
|
||||
@@ -105,6 +123,15 @@ type JobParams struct {
|
||||
Deadline time.Duration
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
// DiskLimit caps kaniko's ephemeral storage, and with it the pod's (see
|
||||
// kanikoDiskRq). Empty applies defaultDiskLimit.
|
||||
DiskLimit string
|
||||
// UserNamespaces runs the pod with hostUsers: false, so root in the build
|
||||
// containers is an unprivileged uid on the node. It needs a kernel and runtime
|
||||
// with idmapped mounts; Config.UserNamespaces decides.
|
||||
UserNamespaces bool
|
||||
// RuntimeClass, when set, runs the pod under that RuntimeClass (gVisor, Kata).
|
||||
RuntimeClass string
|
||||
}
|
||||
|
||||
// BuildJobName is the deterministic Job name for a build id.
|
||||
@@ -127,8 +154,14 @@ func buildLabels(p JobParams) map[string]string {
|
||||
// reach the K8s API (spec §16, §21);
|
||||
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
|
||||
// so docker-in-docker / privileged is never needed (spec §16, §22);
|
||||
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
|
||||
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
|
||||
// - the RuntimeDefault seccomp profile on the whole pod, and optionally a user
|
||||
// namespace (hostUsers: false) and a sandbox RuntimeClass, because kaniko is
|
||||
// no isolation boundary: the Dockerfile's RUN steps execute in its container;
|
||||
// - an egress gate ahead of everything else, so nothing runs before the
|
||||
// namespace's NetworkPolicy is enforced for this pod (cmd/felis egress-gate);
|
||||
// - activeDeadlineSeconds + backoffLimit=0 + per-container CPU, memory and
|
||||
// ephemeral-storage limits so a runaway or poisoned build cannot exhaust the
|
||||
// node (spec §16);
|
||||
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
|
||||
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
|
||||
// automatic admission gate (spec §16).
|
||||
@@ -149,6 +182,18 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
diskCap := p.DiskLimit
|
||||
if diskCap == "" {
|
||||
diskCap = defaultDiskLimit
|
||||
}
|
||||
kanikoDisk, err := resource.ParseQuantity(diskCap)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("build: invalid disk limit %q: %w", diskCap, err)
|
||||
}
|
||||
kanikoRq := kanikoDiskRq.DeepCopy()
|
||||
if kanikoDisk.Cmp(kanikoRq) < 0 {
|
||||
kanikoRq = kanikoDisk.DeepCopy()
|
||||
}
|
||||
if p.FelisImage == "" {
|
||||
return nil, fmt.Errorf("build: FelisImage is required: the push container runs it")
|
||||
}
|
||||
@@ -187,7 +232,35 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
|
||||
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
|
||||
contextPath := p.ContextRef
|
||||
initContainers := []corev1.Container{}
|
||||
|
||||
// The gate runs before anything else. A new pod's NetworkPolicy is programmed
|
||||
// asynchronously: live on k3s (kube-router), a build-labelled pod reached the
|
||||
// internet and the Kubernetes API for the first ~0.7 s of its life. The gate
|
||||
// holds the pod until a destination the policy denies stops answering, so the
|
||||
// Dockerfile never runs inside that window.
|
||||
gateSec := sec.DeepCopy()
|
||||
gateSec.ReadOnlyRootFilesystem = boolPtr(true)
|
||||
gateSec.RunAsNonRoot = boolPtr(true)
|
||||
gateSec.RunAsUser = int64Ptr(nonRootUID)
|
||||
gate := corev1.Container{
|
||||
Name: ContainerGate,
|
||||
Image: p.FelisImage,
|
||||
Args: []string{"egress-gate"},
|
||||
Resources: corev1.ResourceRequirements{
|
||||
Limits: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("100m"),
|
||||
corev1.ResourceMemory: resource.MustParse("64Mi"),
|
||||
corev1.ResourceEphemeralStorage: gateDisk.limit,
|
||||
},
|
||||
Requests: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("10m"),
|
||||
corev1.ResourceMemory: resource.MustParse("16Mi"),
|
||||
corev1.ResourceEphemeralStorage: gateDisk.request,
|
||||
},
|
||||
},
|
||||
SecurityContext: gateSec,
|
||||
}
|
||||
initContainers := []corev1.Container{gate}
|
||||
imageMount := corev1.VolumeMount{Name: imageVolume, MountPath: imageMountPath}
|
||||
kanikoMounts := []corev1.VolumeMount{imageMount}
|
||||
podVolumes := []corev1.Volume{{
|
||||
@@ -232,7 +305,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
}},
|
||||
}},
|
||||
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
Resources: withDisk(limits, fetchDisk),
|
||||
SecurityContext: fetchSec,
|
||||
}
|
||||
initContainers = append(initContainers, fetch)
|
||||
@@ -269,7 +342,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
"--skip-tls-verify-pull",
|
||||
},
|
||||
VolumeMounts: kanikoMounts,
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
Resources: withDisk(limits, diskBounds{request: kanikoRq, limit: kanikoDisk}),
|
||||
SecurityContext: kanikoSec,
|
||||
}
|
||||
initContainers = append(initContainers, kaniko)
|
||||
@@ -298,7 +371,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
Image: p.TrivyImage,
|
||||
Args: trivyArgs,
|
||||
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
Resources: withDisk(limits, trivyDisk),
|
||||
SecurityContext: sec,
|
||||
}
|
||||
initContainers = append(initContainers, trivy)
|
||||
@@ -328,7 +401,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
secretEnv("FELIS_REGISTRY_PASSWORD", naming.RegistryPushPasswordKey),
|
||||
},
|
||||
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
Resources: withDisk(limits, pushDisk),
|
||||
SecurityContext: pushSec,
|
||||
}
|
||||
|
||||
@@ -350,16 +423,44 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: p.ServiceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
InitContainers: initContainers,
|
||||
Containers: []corev1.Container{push},
|
||||
Volumes: podVolumes,
|
||||
// RUN steps execute in kaniko's container with root and three
|
||||
// capabilities; RuntimeDefault takes away the syscalls a container
|
||||
// never needs, among them most kernel-escape primitives (unshare,
|
||||
// mount, keyctl, bpf). Every step was checked to run under it live.
|
||||
SecurityContext: &corev1.PodSecurityContext{
|
||||
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
|
||||
},
|
||||
InitContainers: initContainers,
|
||||
Containers: []corev1.Container{push},
|
||||
Volumes: podVolumes,
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
if p.UserNamespaces {
|
||||
job.Spec.Template.Spec.HostUsers = boolPtr(false)
|
||||
}
|
||||
if p.RuntimeClass != "" {
|
||||
rc := p.RuntimeClass
|
||||
job.Spec.Template.Spec.RuntimeClassName = &rc
|
||||
}
|
||||
return job, nil
|
||||
}
|
||||
|
||||
// nonRootUID is the distroless nonroot user the platform image ships as.
|
||||
const nonRootUID = 65532
|
||||
|
||||
// withDisk is the resources block for one build container: the CPU and memory
|
||||
// caps with their schedulable floor (buildRequests), plus its ephemeral-storage
|
||||
// request and limit.
|
||||
func withDisk(limits corev1.ResourceList, d diskBounds) corev1.ResourceRequirements {
|
||||
lim := limits.DeepCopy()
|
||||
lim[corev1.ResourceEphemeralStorage] = d.limit
|
||||
req := buildRequests(limits)
|
||||
req[corev1.ResourceEphemeralStorage] = d.request
|
||||
return corev1.ResourceRequirements{Limits: lim, Requests: req}
|
||||
}
|
||||
|
||||
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
|
||||
// lane derives when the API is the blob transport — i.e. a context only the
|
||||
// fetch initContainer can turn into a local path for Kaniko.
|
||||
@@ -516,6 +617,36 @@ func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
|
||||
}
|
||||
}
|
||||
|
||||
// BuildLimitRange bounds any container in the build namespace that arrives
|
||||
// without its own limits. Build Jobs set every limit themselves (BuildJob); this
|
||||
// is the backstop for anything else that lands in the namespace, which shares
|
||||
// the node's disk with the game worlds.
|
||||
func BuildLimitRange(namespace string) *corev1.LimitRange {
|
||||
return &corev1.LimitRange{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "felis-build-limits",
|
||||
Namespace: namespace,
|
||||
Labels: map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
},
|
||||
},
|
||||
Spec: corev1.LimitRangeSpec{Limits: []corev1.LimitRangeItem{{
|
||||
Type: corev1.LimitTypeContainer,
|
||||
Default: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("1"),
|
||||
corev1.ResourceMemory: resource.MustParse("1Gi"),
|
||||
corev1.ResourceEphemeralStorage: resource.MustParse("1Gi"),
|
||||
},
|
||||
DefaultRequest: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("100m"),
|
||||
corev1.ResourceMemory: resource.MustParse("128Mi"),
|
||||
corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"),
|
||||
},
|
||||
}}},
|
||||
}
|
||||
}
|
||||
|
||||
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
||||
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||
if cpu == "" {
|
||||
|
||||
+116
-10
@@ -171,10 +171,10 @@ func TestBuildJobScansBeforePush(t *testing.T) {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
inits := job.Spec.Template.Spec.InitContainers
|
||||
if len(inits) != 2 || inits[0].Name != ContainerKaniko || inits[1].Name != ContainerTrivy {
|
||||
t.Fatalf("initContainers = %v, want [kaniko trivy]", initNames(inits))
|
||||
if len(inits) != 3 || inits[0].Name != ContainerGate || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy {
|
||||
t.Fatalf("initContainers = %v, want [egress-gate kaniko trivy]", initNames(inits))
|
||||
}
|
||||
kaniko, trivy := inits[0], inits[1]
|
||||
kaniko, trivy := inits[1], inits[2]
|
||||
for _, want := range []string{"--destination=" + p.ImageRef, "--no-push", "--tar-path=" + imageTarPath} {
|
||||
if !hasArg(kaniko.Args, want) {
|
||||
t.Errorf("kaniko args = %v, want %s", kaniko.Args, want)
|
||||
@@ -339,9 +339,9 @@ func TestBuildJobTrivyDBRepositoryOverride(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
trivy := job.Spec.Template.Spec.InitContainers[1]
|
||||
trivy := job.Spec.Template.Spec.InitContainers[2]
|
||||
if trivy.Name != ContainerTrivy {
|
||||
t.Fatalf("initContainers = %v, want trivy second", initNames(job.Spec.Template.Spec.InitContainers))
|
||||
t.Fatalf("initContainers = %v, want trivy third", initNames(job.Spec.Template.Spec.InitContainers))
|
||||
}
|
||||
if !argPairPresent(trivy.Args, "--db-repository", p.TrivyDBRepository) {
|
||||
t.Errorf("trivy args = %v, want --db-repository %s", trivy.Args, p.TrivyDBRepository)
|
||||
@@ -423,10 +423,11 @@ func TestBuildJobFetchesHTTPContext(t *testing.T) {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
inits := job.Spec.Template.Spec.InitContainers
|
||||
if len(inits) != 3 || inits[0].Name != ContainerFetch || inits[1].Name != ContainerKaniko || inits[2].Name != ContainerTrivy {
|
||||
t.Fatalf("initContainers = %v, want [%s %s %s]", initNames(inits), ContainerFetch, ContainerKaniko, ContainerTrivy)
|
||||
if len(inits) != 4 || inits[0].Name != ContainerGate || inits[1].Name != ContainerFetch ||
|
||||
inits[2].Name != ContainerKaniko || inits[3].Name != ContainerTrivy {
|
||||
t.Fatalf("initContainers = %v, want [%s %s %s %s]", initNames(inits), ContainerGate, ContainerFetch, ContainerKaniko, ContainerTrivy)
|
||||
}
|
||||
fetch, kaniko := inits[0], inits[1]
|
||||
fetch, kaniko := inits[1], inits[2]
|
||||
if fetch.Image != p.FelisImage {
|
||||
t.Errorf("fetch image = %q, want the platform image %q", fetch.Image, p.FelisImage)
|
||||
}
|
||||
@@ -494,8 +495,8 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 2 || inits[0].Name != ContainerKaniko {
|
||||
t.Errorf("a native ref must render just kaniko + trivy, got %v", initNames(inits))
|
||||
if inits := job.Spec.Template.Spec.InitContainers; len(inits) != 3 || inits[1].Name != ContainerKaniko {
|
||||
t.Errorf("a native ref must render just the gate, kaniko and trivy, got %v", initNames(inits))
|
||||
}
|
||||
for _, v := range job.Spec.Template.Spec.Volumes {
|
||||
if v.Name == contextVolume {
|
||||
@@ -504,6 +505,111 @@ func TestBuildJobNativeContextNeedsNoFetch(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing runs before the egress gate, and the gate itself holds nothing: no
|
||||
// credential, no root, no writable filesystem.
|
||||
func TestBuildJobGatesEgressFirst(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx"
|
||||
job, err := BuildJob(p)
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
gate := job.Spec.Template.Spec.InitContainers[0]
|
||||
if gate.Name != ContainerGate || gate.Image != p.FelisImage || len(gate.Args) != 1 || gate.Args[0] != "egress-gate" {
|
||||
t.Fatalf("first initContainer = %s %s %v, want the platform image's egress-gate", gate.Name, gate.Image, gate.Args)
|
||||
}
|
||||
if len(gate.Env) != 0 || len(gate.VolumeMounts) != 0 {
|
||||
t.Errorf("the gate must hold nothing, got env %v mounts %v", gate.Env, gate.VolumeMounts)
|
||||
}
|
||||
sc := gate.SecurityContext
|
||||
if sc == nil || sc.RunAsNonRoot == nil || !*sc.RunAsNonRoot || sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem {
|
||||
t.Errorf("the gate must run non-root on a read-only root, got %#v", sc)
|
||||
}
|
||||
}
|
||||
|
||||
// The pod runs under RuntimeDefault seccomp always, and in a user namespace or
|
||||
// a sandbox runtime when the install asks for them.
|
||||
func TestBuildJobSandboxing(t *testing.T) {
|
||||
job, err := BuildJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
spec := job.Spec.Template.Spec
|
||||
if spec.SecurityContext == nil || spec.SecurityContext.SeccompProfile == nil ||
|
||||
spec.SecurityContext.SeccompProfile.Type != corev1.SeccompProfileTypeRuntimeDefault {
|
||||
t.Fatalf("pod securityContext = %#v, want seccompProfile RuntimeDefault", spec.SecurityContext)
|
||||
}
|
||||
if spec.HostUsers != nil || spec.RuntimeClassName != nil {
|
||||
t.Errorf("defaults must leave hostUsers and runtimeClassName unset, got %v / %v", spec.HostUsers, spec.RuntimeClassName)
|
||||
}
|
||||
p := sampleJobParams()
|
||||
p.UserNamespaces, p.RuntimeClass = true, "gvisor"
|
||||
if job, err = BuildJob(p); err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
spec = job.Spec.Template.Spec
|
||||
if spec.HostUsers == nil || *spec.HostUsers {
|
||||
t.Errorf("UserNamespaces must render hostUsers: false, got %v", spec.HostUsers)
|
||||
}
|
||||
if spec.RuntimeClassName == nil || *spec.RuntimeClassName != "gvisor" {
|
||||
t.Errorf("runtimeClassName = %v, want gvisor", spec.RuntimeClassName)
|
||||
}
|
||||
}
|
||||
|
||||
// Every container carries an ephemeral-storage request and limit, and kaniko's
|
||||
// limit, the largest, is the configured disk cap.
|
||||
func TestBuildJobBoundsEphemeralStorage(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
p.ContextRef = "http://felis-api-internal.felis.svc.cluster.local:8081/ctx"
|
||||
job, err := BuildJob(p)
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
all := append(append([]corev1.Container{}, job.Spec.Template.Spec.InitContainers...), job.Spec.Template.Spec.Containers...)
|
||||
for _, c := range all {
|
||||
lim, lok := c.Resources.Limits[corev1.ResourceEphemeralStorage]
|
||||
req, rok := c.Resources.Requests[corev1.ResourceEphemeralStorage]
|
||||
if !lok || !rok || lim.IsZero() || req.Cmp(lim) > 0 {
|
||||
t.Errorf("container %s ephemeral-storage request %v limit %v", c.Name, req.String(), lim.String())
|
||||
}
|
||||
if c.Name == ContainerKaniko && lim.String() != defaultDiskLimit {
|
||||
t.Errorf("kaniko ephemeral-storage limit = %s, want the default %s", lim.String(), defaultDiskLimit)
|
||||
}
|
||||
}
|
||||
p.DiskLimit = "512Mi"
|
||||
if job, err = BuildJob(p); err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
for _, c := range job.Spec.Template.Spec.InitContainers {
|
||||
if c.Name != ContainerKaniko {
|
||||
continue
|
||||
}
|
||||
lim := c.Resources.Limits[corev1.ResourceEphemeralStorage]
|
||||
req := c.Resources.Requests[corev1.ResourceEphemeralStorage]
|
||||
if lim.String() != "512Mi" || req.Cmp(lim) > 0 {
|
||||
t.Errorf("a 512Mi disk cap rendered limit %s request %s", lim.String(), req.String())
|
||||
}
|
||||
}
|
||||
p.DiskLimit = "lots"
|
||||
if _, err := BuildJob(p); err == nil {
|
||||
t.Error("an unparsable disk limit was accepted")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildLimitRangeCoversEphemeralStorage(t *testing.T) {
|
||||
lr := BuildLimitRange("felis-build")
|
||||
if lr.Namespace != "felis-build" || len(lr.Spec.Limits) != 1 {
|
||||
t.Fatalf("limit range = %#v", lr)
|
||||
}
|
||||
item := lr.Spec.Limits[0]
|
||||
for _, res := range []corev1.ResourceName{corev1.ResourceCPU, corev1.ResourceMemory, corev1.ResourceEphemeralStorage} {
|
||||
d, dr := item.Default[res], item.DefaultRequest[res]
|
||||
if d.IsZero() || dr.IsZero() || dr.Cmp(d) > 0 {
|
||||
t.Errorf("%s default %s request %s", res, d.String(), dr.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func initNames(cs []corev1.Container) []string {
|
||||
names := make([]string, 0, len(cs))
|
||||
for _, c := range cs {
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// Vars so tests can shrink them. The probe pod's image is the api's own, so it is
|
||||
// already on the node; the timeout covers a slow pod start, and a runtime that
|
||||
// cannot do user namespaces fails the pod well within it.
|
||||
var (
|
||||
usernsProbeTimeout = 3 * time.Minute
|
||||
usernsProbePoll = 2 * time.Second
|
||||
)
|
||||
|
||||
// UsernsProbeJob renders the one-shot Job that asks the cluster whether a build
|
||||
// pod can run with hostUsers: false. The pod has the parts of a build pod that
|
||||
// need idmapped mounts and namespaced capabilities: root with kaniko's three
|
||||
// capabilities, the RuntimeDefault seccomp profile and an emptyDir. It runs
|
||||
// `felis version`, which touches nothing.
|
||||
func UsernsProbeJob(namespace, serviceAccount, image, name string) *batchv1.Job {
|
||||
labels := map[string]string{LabelManagedBy: managedByValue, LabelComponent: "userns-probe"}
|
||||
small := corev1.ResourceRequirements{
|
||||
Limits: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("100m"),
|
||||
corev1.ResourceMemory: resource.MustParse("64Mi"),
|
||||
corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"),
|
||||
},
|
||||
Requests: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("10m"),
|
||||
corev1.ResourceMemory: resource.MustParse("16Mi"),
|
||||
corev1.ResourceEphemeralStorage: resource.MustParse("16Mi"),
|
||||
},
|
||||
}
|
||||
return &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: namespace, Labels: labels},
|
||||
Spec: batchv1.JobSpec{
|
||||
BackoffLimit: int32Ptr(0),
|
||||
ActiveDeadlineSeconds: int64Ptr(int64(usernsProbeTimeout / time.Second)),
|
||||
TTLSecondsAfterFinished: int32Ptr(300),
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: labels},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: serviceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
HostUsers: boolPtr(false),
|
||||
SecurityContext: &corev1.PodSecurityContext{
|
||||
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
|
||||
},
|
||||
Containers: []corev1.Container{{
|
||||
Name: "probe",
|
||||
Image: image,
|
||||
Args: []string{"version"},
|
||||
Resources: small,
|
||||
SecurityContext: &corev1.SecurityContext{
|
||||
Privileged: boolPtr(false),
|
||||
AllowPrivilegeEscalation: boolPtr(false),
|
||||
RunAsUser: int64Ptr(0),
|
||||
Capabilities: &corev1.Capabilities{
|
||||
Drop: []corev1.Capability{"ALL"},
|
||||
Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE", "FOWNER"},
|
||||
},
|
||||
},
|
||||
VolumeMounts: []corev1.VolumeMount{{Name: "scratch", MountPath: "/scratch"}},
|
||||
}},
|
||||
Volumes: []corev1.Volume{{
|
||||
Name: "scratch",
|
||||
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{SizeLimit: quantityPtr(resource.MustParse("16Mi"))}},
|
||||
}},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// ProbeUserNamespaces runs UsernsProbeJob and reports whether its pod succeeded.
|
||||
// A pod that fails, or never starts before the timeout, answers false; err is set
|
||||
// only when the Job could not be created or read. The Job is deleted afterwards.
|
||||
func (k *K8sJobs) ProbeUserNamespaces(ctx context.Context, image string) (bool, error) {
|
||||
name := "userns-probe-" + strconv.FormatInt(time.Now().UnixNano(), 36)
|
||||
job := UsernsProbeJob(k.cfg.Namespace, k.cfg.ServiceAccount, image, name)
|
||||
if err := k.c.Create(ctx, job); err != nil {
|
||||
return false, fmt.Errorf("create the probe job: %w", err)
|
||||
}
|
||||
defer func() {
|
||||
bg := metav1.DeletePropagationBackground
|
||||
del := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: k.cfg.Namespace, Name: name}}
|
||||
_ = k.c.Delete(context.WithoutCancel(ctx), del, &client.DeleteOptions{PropagationPolicy: &bg})
|
||||
}()
|
||||
deadline := time.Now().Add(usernsProbeTimeout)
|
||||
for {
|
||||
var got batchv1.Job
|
||||
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.cfg.Namespace, Name: name}, &got)
|
||||
if err != nil && !apierrors.IsNotFound(err) {
|
||||
return false, fmt.Errorf("read the probe job: %w", err)
|
||||
}
|
||||
for _, cond := range got.Status.Conditions {
|
||||
if cond.Status != corev1.ConditionTrue {
|
||||
continue
|
||||
}
|
||||
switch cond.Type {
|
||||
case batchv1.JobComplete:
|
||||
return true, nil
|
||||
case batchv1.JobFailed:
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
return false, nil
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return false, ctx.Err()
|
||||
case <-time.After(usernsProbePoll):
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,144 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
// "auto" follows the probe through every copy of the Config; "on" and "off"
|
||||
// ignore it.
|
||||
func TestUserNamespacesMode(t *testing.T) {
|
||||
probe := new(atomic.Bool)
|
||||
for _, tc := range []struct {
|
||||
mode string
|
||||
probe bool
|
||||
want bool
|
||||
}{
|
||||
{"", false, false}, {"", true, true}, {UserNamespacesAuto, true, true},
|
||||
{UserNamespacesOn, false, true}, {UserNamespacesOff, true, false},
|
||||
} {
|
||||
b, _, jb := newBuilder()
|
||||
b.Config.UserNamespaces = tc.mode
|
||||
b.Config.UserNamespacesProbe = probe
|
||||
probe.Store(tc.probe)
|
||||
if _, err := b.Submit(context.Background(), goodRequest()); err != nil {
|
||||
t.Fatalf("Submit: %v", err)
|
||||
}
|
||||
if got := jb.created[0].UserNamespaces; got != tc.want {
|
||||
t.Errorf("mode %q, probe %v: UserNamespaces = %v, want %v", tc.mode, tc.probe, got, tc.want)
|
||||
}
|
||||
}
|
||||
b, _, jb := newBuilder()
|
||||
if _, err := b.Submit(context.Background(), goodRequest()); err != nil || jb.created[0].UserNamespaces {
|
||||
t.Errorf("no probe wired: err %v, UserNamespaces %v", err, jb.created[0].UserNamespaces)
|
||||
}
|
||||
}
|
||||
|
||||
// The probe pod has the shape that needs user-namespace support, and runs
|
||||
// nothing that matters.
|
||||
func TestUsernsProbeJobShape(t *testing.T) {
|
||||
job := UsernsProbeJob("felis-build", "felis-build", "felis:test", "userns-probe-x")
|
||||
spec := job.Spec.Template.Spec
|
||||
if spec.HostUsers == nil || *spec.HostUsers {
|
||||
t.Fatal("the probe must ask for hostUsers: false")
|
||||
}
|
||||
if spec.AutomountServiceAccountToken == nil || *spec.AutomountServiceAccountToken || spec.ServiceAccountName != "felis-build" {
|
||||
t.Error("the probe must run as the bare build SA with no token")
|
||||
}
|
||||
c := spec.Containers[0]
|
||||
if c.Image != "felis:test" || len(c.Args) != 1 || c.Args[0] != "version" {
|
||||
t.Errorf("probe container runs %s %v", c.Image, c.Args)
|
||||
}
|
||||
if c.SecurityContext.RunAsUser == nil || *c.SecurityContext.RunAsUser != 0 || len(c.SecurityContext.Capabilities.Add) != 3 {
|
||||
t.Error("the probe must run as root with kaniko's capabilities, the case user namespaces must carry")
|
||||
}
|
||||
if len(spec.Volumes) != 1 || spec.Volumes[0].EmptyDir == nil {
|
||||
t.Error("the probe must mount an emptyDir, as build pods do")
|
||||
}
|
||||
if job.Labels[LabelManagedBy] != managedByValue {
|
||||
t.Error("the probe must sit under the build egress policy")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbeUserNamespaces(t *testing.T) {
|
||||
timeout, poll := usernsProbeTimeout, usernsProbePoll
|
||||
t.Cleanup(func() { usernsProbeTimeout, usernsProbePoll = timeout, poll })
|
||||
usernsProbePoll = 5 * time.Millisecond
|
||||
|
||||
scheme := runtime.NewScheme()
|
||||
_ = clientgoscheme.AddToScheme(scheme)
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
cond batchv1.JobConditionType
|
||||
want bool
|
||||
}{
|
||||
{"complete", batchv1.JobComplete, true},
|
||||
{"failed", batchv1.JobFailed, false},
|
||||
{"never finishes", "", false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
// A finished probe must answer long before the timeout would.
|
||||
usernsProbeTimeout = 5 * time.Second
|
||||
if tc.cond == "" {
|
||||
usernsProbeTimeout = 100 * time.Millisecond
|
||||
}
|
||||
start := time.Now()
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithStatusSubresource(&batchv1.Job{}).Build()
|
||||
jobs := NewK8sJobs(cl, Config{})
|
||||
type result struct {
|
||||
ok bool
|
||||
err error
|
||||
}
|
||||
done := make(chan result, 1)
|
||||
go func() {
|
||||
ok, err := jobs.ProbeUserNamespaces(context.Background(), "felis:test")
|
||||
done <- result{ok, err}
|
||||
}()
|
||||
if tc.cond != "" {
|
||||
finishProbe(t, cl, tc.cond)
|
||||
}
|
||||
r := <-done
|
||||
if r.err != nil || r.ok != tc.want {
|
||||
t.Fatalf("ProbeUserNamespaces = (%v, %v), want (%v, nil)", r.ok, r.err, tc.want)
|
||||
}
|
||||
if tc.cond != "" && time.Since(start) > 2*time.Second {
|
||||
t.Fatalf("the probe answered after %s: it waited out the timeout instead of reading the verdict", time.Since(start))
|
||||
}
|
||||
var left batchv1.JobList
|
||||
if err := cl.List(context.Background(), &left, client.InNamespace(defaultNamespace)); err != nil || len(left.Items) != 0 {
|
||||
t.Fatalf("probe jobs left behind: %d (%v)", len(left.Items), err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// finishProbe waits for the probe Job to appear and marks it finished.
|
||||
func finishProbe(t *testing.T, cl client.Client, cond batchv1.JobConditionType) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
var jobs batchv1.JobList
|
||||
if err := cl.List(context.Background(), &jobs, client.InNamespace(defaultNamespace)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(jobs.Items) == 1 {
|
||||
job := jobs.Items[0]
|
||||
job.Status.Conditions = []batchv1.JobCondition{{Type: cond, Status: corev1.ConditionTrue}}
|
||||
if err := cl.Status().Update(context.Background(), &job); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return
|
||||
}
|
||||
time.Sleep(2 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("the probe job never appeared")
|
||||
}
|
||||
Reference in new issue
Block a user