package build import ( "fmt" "strings" "time" "felis.lolicon.best/internal/naming" batchv1 "k8s.io/api/batch/v1" corev1 "k8s.io/api/core/v1" networkingv1 "k8s.io/api/networking/v1" "k8s.io/apimachinery/pkg/api/resource" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/util/intstr" ) // Label keys applied to build objects. ManagedBy doubles as the NetworkPolicy // pod selector, so every build Pod is captured by the egress lock. const ( LabelManagedBy = "app.kubernetes.io/managed-by" LabelComponent = "app.kubernetes.io/component" LabelBuildID = "felis.lolicon.best/build-id" managedByValue = "felis-build" componentValue = "image-build" ) // Container names within the build Pod. Kaniko is the initContainer that builds // the image into a tarball — its log IS the "build log" an admin watches (spec // §16); Trivy is the next initContainer, whose CRITICAL-CVE verdict gates both the // push and admission and is surfaced via the build status, not the log stream; // Push is the main container that publishes the scanned tarball. Exported so the // build-log streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) // follows the same container this Job defines — one source of truth for the name. const ( ContainerGate = "egress-gate" ContainerKaniko = "kaniko" ContainerTrivy = "trivy" ContainerPush = "push" // ContainerFetch is the initContainer that pulls a submission's build context // from the felis-api internal face and extracts it into the shared emptyDir. // It exists only for an http(s) ContextRef (see BuildJob); a ref Kaniko can // read natively (s3://) or a pre-mounted path renders no such container. ContainerFetch = "context-fetch" // contextVolume/contextMountPath carry a fetched build context: the fetch // initContainer writes the extracted tree there, Kaniko reads it read-only. contextVolume = "context" contextMountPath = "/context" // imageVolume/imageTarPath carry the built image from Kaniko (--tar-path) to // Trivy (--input) and then to the push container. imageVolume = "image" imageMountPath = "/image" imageTarPath = imageMountPath + "/image.tar" ) // imageSizeLimit bounds the built image tarball. A modpack image is typically a // JRE, a server jar and a few hundred MiB of mods; 10 GiB leaves ample room while // still stopping a runaway build from filling the node's disk. var imageSizeLimit = resource.MustParse("10Gi") // contextSizeLimit bounds the extracted (attacker-controlled) context tree so a // tarball bomb wedges the build pod instead of the node's disk. The compressed // upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room. var contextSizeLimit = resource.MustParse("4Gi") // Per-container ephemeral-storage bounds (writable layer + logs; emptyDirs count // toward the pod as a whole). Kaniko's limit is the operator's disk cap because // kaniko unpacks the base image into its own root filesystem, which no emptyDir // bound covers; it is also the largest limit in the pod, so it becomes the // pod-level cap the kubelet holds context + unpacked rootfs + image tarball to. // Trivy keeps its vulnerability and Java DBs (about 1.4 GiB live) in its layer. // The others write nothing but logs. var ( gateDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("64Mi")} fetchDisk = diskBounds{request: resource.MustParse("64Mi"), limit: resource.MustParse("256Mi")} kanikoDiskRq = resource.MustParse("1Gi") trivyDisk = diskBounds{request: resource.MustParse("256Mi"), limit: resource.MustParse("4Gi")} pushDisk = diskBounds{request: resource.MustParse("16Mi"), limit: resource.MustParse("256Mi")} ) type diskBounds struct{ request, limit resource.Quantity } // buildJobTTL is how long a finished build Job survives before the Job // controller deletes it — and with it the Pod whose kaniko log is the admin // failure-triage surface (GET /api/v1/images/build/{id}/logs). // // Every other Job family the platform renders carries a TTL (fileedit 2m, // backup/restore 10m); the build lane deliberately keeps a much longer one // because the logs are the point. Without ANY TTL the Job and its completed // Pod accumulate one pair per build forever: they count against the node's // pod budget (110 on stock k3s), grow etcd, and eventually block new builds. // Sync already tolerates a vanished Job (JobUnknown → failed; terminal builds // are returned unchanged), so a week-old log falling off costs a 404, not a // status flip. const buildJobTTL = 7 * 24 * time.Hour // JobParams are the rendered inputs to a build Job. They are derived from a // Build + Config by the Builder; jobspec is a pure function of them so the // security-critical Job shape is unit-tested without a cluster. type JobParams struct { BuildID string ImageRef string ContextRef string // ContextDigest, when set, is passed to the fetch container, which refuses a // context whose sha256 differs (see Request.ContextDigest). ContextDigest string Namespace string ServiceAccount string RegistryURL string // FelisImage runs the context-fetch initContainer (the felis binary's // fetch-context entrypoint). Required when ContextRef is an http(s) URL. FelisImage string // TrivyDBRepository overrides Trivy's vulnerability-DB source (the // --db-repository flag). Empty keeps Trivy's own default; see // build.Config.TrivyDBRepository for why an in-cluster install sets it. TrivyDBRepository string // TrivyJavaDBRepository overrides Trivy's Java-DB source (the // --java-db-repository flag), fetched lazily when the image contains Java // artifacts; empty keeps Trivy's own default, which the build egress lock // denies — a jar-bearing image then fails the scan. TrivyJavaDBRepository string KanikoImage string TrivyImage string Deadline time.Duration CPULimit string MemLimit string // DiskLimit caps kaniko's ephemeral storage, and with it the pod's (see // kanikoDiskRq). Empty applies defaultDiskLimit. DiskLimit string // UserNamespaces runs the pod with hostUsers: false, so root in the build // containers is an unprivileged uid on the node. It needs a kernel and runtime // with idmapped mounts; Config.UserNamespaces decides. UserNamespaces bool // RuntimeClass, when set, runs the pod under that RuntimeClass (gVisor, Kata). RuntimeClass string } // BuildJobName is the deterministic Job name for a build id. func BuildJobName(buildID string) string { return "build-" + buildID } func buildLabels(p JobParams) map[string]string { return map[string]string{ LabelManagedBy: managedByValue, LabelComponent: componentValue, LabelBuildID: p.BuildID, } } // BuildJob renders the Kaniko+Trivy build Job (spec §16). Every isolation // guarantee the spec demands is encoded here and asserted by jobspec_test.go, // because no cluster runs in this environment: // // - runs in the isolated felis-build namespace with the weak felis-build SA // (never the felis-api SA) and does NOT mount the SA token, so it cannot // reach the K8s API (spec §16, §21); // - no privileged container — Kaniko builds the Dockerfile without a daemon, // so docker-in-docker / privileged is never needed (spec §16, §22); // - the RuntimeDefault seccomp profile on the whole pod, and optionally a user // namespace (hostUsers: false) and a sandbox RuntimeClass, because kaniko is // no isolation boundary: the Dockerfile's RUN steps execute in its container; // - an egress gate ahead of everything else, so nothing runs before the // namespace's NetworkPolicy is enforced for this pod (cmd/felis egress-gate); // - activeDeadlineSeconds + backoffLimit=0 + per-container CPU, memory and // ephemeral-storage limits so a runaway or poisoned build cannot exhaust the // node (spec §16); // - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a // CRITICAL CVE fails the Pod and therefore the Job — the only retained // automatic admission gate (spec §16). // // Sequencing: kaniko builds with --no-push into a tarball, trivy scans that // tarball, and only then does the push container publish it. So: // // - an image that fails the scan is never published — it used to be pushed to // the final tag first and scanned after, overwriting whatever that tag held; // - the registry credential lives in the push container alone. Kaniko executes // the untrusted Dockerfile and holds no credential at all, and the registry // refuses anonymous writes (internal/registrygate). // // The Pod succeeds only if kaniko built, trivy found no CRITICAL CVE, and the push // landed. func BuildJob(p JobParams) (*batchv1.Job, error) { limits, err := resourceLimits(p.CPULimit, p.MemLimit) if err != nil { return nil, err } diskCap := p.DiskLimit if diskCap == "" { diskCap = defaultDiskLimit } kanikoDisk, err := resource.ParseQuantity(diskCap) if err != nil { return nil, fmt.Errorf("build: invalid disk limit %q: %w", diskCap, err) } kanikoRq := kanikoDiskRq.DeepCopy() if kanikoDisk.Cmp(kanikoRq) < 0 { kanikoRq = kanikoDisk.DeepCopy() } if p.FelisImage == "" { return nil, fmt.Errorf("build: FelisImage is required: the push container runs it") } deadline := int64(p.Deadline / time.Second) if deadline <= 0 { deadline = int64(defaultDeadline / time.Second) } // Hardened container security context baseline: no privilege, no privilege // escalation, drop all capabilities. Kaniko needs a writable root filesystem // to unpack layers, so we do not force read-only root here; it also needs a // minimal capability subset added back (kanikoSec below), while fetch and // trivy run with exactly this baseline. sec := &corev1.SecurityContext{ Privileged: boolPtr(false), AllowPrivilegeEscalation: boolPtr(false), Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}}, } // Kaniko unpacks base-image layers as root, and the tar apply must chown/chmod // files to the owners the layer recorded — impossible under drop-ALL (live: // "failed to get filesystem from image: chown /etc/gshadow: operation not // permitted" for any FROM ; scratch builds masked this because // COPY only ever creates files kaniko itself owns). Add back exactly the caps // the unpack needs and nothing else: CHOWN/FOWNER for the ownership and mode // restore, DAC_OVERRIDE to write entries whose bits would otherwise exclude // even root once the capability-based exemption is gone. kanikoSec := sec.DeepCopy() kanikoSec.Capabilities = &corev1.Capabilities{ Drop: []corev1.Capability{"ALL"}, Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE", "FOWNER"}, } // The context Kaniko reads. An http(s) ref (the submit lane's derived ref: the // API streams the blob on its internal face, because the build Pod can neither // mount the control-plane uploads PVC across namespaces nor hold object-store // credentials) is first fetched into a shared emptyDir; a ref Kaniko can read // in place (s3://, or a path an installer pre-mounted) passes through untouched. contextPath := p.ContextRef // The gate runs before anything else. A new pod's NetworkPolicy is programmed // asynchronously: live on k3s (kube-router), a build-labelled pod reached the // internet and the Kubernetes API for the first ~0.7 s of its life. The gate // holds the pod until a destination the policy denies stops answering, so the // Dockerfile never runs inside that window. gateSec := sec.DeepCopy() gateSec.ReadOnlyRootFilesystem = boolPtr(true) gateSec.RunAsNonRoot = boolPtr(true) gateSec.RunAsUser = int64Ptr(nonRootUID) gate := corev1.Container{ Name: ContainerGate, Image: p.FelisImage, Args: []string{"egress-gate"}, Resources: corev1.ResourceRequirements{ Limits: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("100m"), corev1.ResourceMemory: resource.MustParse("64Mi"), corev1.ResourceEphemeralStorage: gateDisk.limit, }, Requests: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("10m"), corev1.ResourceMemory: resource.MustParse("16Mi"), corev1.ResourceEphemeralStorage: gateDisk.request, }, }, SecurityContext: gateSec, } initContainers := []corev1.Container{gate} imageMount := corev1.VolumeMount{Name: imageVolume, MountPath: imageMountPath} kanikoMounts := []corev1.VolumeMount{imageMount} podVolumes := []corev1.Volume{{ Name: imageVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{ SizeLimit: quantityPtr(imageSizeLimit), }}, }} if IsHTTPContextRef(p.ContextRef) { contextPath = contextMountPath // The fetch container runs as root while Kaniko keeps the image default // (also root): Kaniko re-copies the Dockerfile out of the context and // chowns/chmods it to the SOURCE file's owner, which fails for any other // owner without CAP_CHOWN/CAP_FOWNER — capabilities this pod deliberately // drops (the live drill hit exactly this: "copying dockerfile: chown // /kaniko/Dockerfile: operation not permitted" with the distroless uid // 65532). Extracting as root, the uid Kaniko itself runs as, keeps the // context owned by the only user that can satisfy that copy. The pod is // root by necessity regardless: Kaniko unpacks base-image layers into its // own filesystem. fetchSec := sec.DeepCopy() fetchSec.RunAsUser = int64Ptr(0) fetchSec.RunAsGroup = int64Ptr(0) fetch := corev1.Container{ Name: ContainerFetch, Image: p.FelisImage, Args: fetchArgs(p), // The internal face is service-token gated, and the token is read from a // Secret the installer materializes in THIS namespace (secretKeyRef is // namespace-local). It is mounted into this initContainer only: the Kaniko // container executes the untrusted Dockerfile and must never hold it, and // pod containers share neither environment nor PID namespace. Env: []corev1.EnvVar{{ Name: "FELIS_SERVICE_TOKEN", ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: naming.ServiceTokenSecretName}, Key: naming.ServiceTokenSecretKey, }}, }}, VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}}, Resources: withDisk(limits, fetchDisk), SecurityContext: fetchSec, } initContainers = append(initContainers, fetch) kanikoMounts = append(kanikoMounts, corev1.VolumeMount{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true}) podVolumes = append(podVolumes, corev1.Volume{ Name: contextVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{ // The extracted tree is attacker-controlled; bound it so a tarball // bomb wedges THIS pod (admitted failure) instead of filling the // node's disk. The compressed upload is capped at 1 GiB by the // submit lane, and 4 GiB leaves room for a typical expansion. SizeLimit: quantityPtr(contextSizeLimit), }}, }) } kaniko := corev1.Container{ Name: ContainerKaniko, Image: p.KanikoImage, Args: []string{ "--dockerfile=Dockerfile", "--context=" + contextPath, // --destination only names the image inside the tarball; --no-push // keeps Kaniko off the registry's write path entirely. "--destination=" + p.ImageRef, "--no-push", "--tar-path=" + imageTarPath, // The internal registry is in-cluster only and serves plain HTTP; it // is never a public ingress (spec §17). A Dockerfile's `FROM // registry.felis.svc:5000/...` fails with "server gave HTTP response // to HTTPS client" without these, breaking every build based on a // platform image (the canonical modpack shape). "--insecure-pull", "--skip-tls-verify-pull", }, VolumeMounts: kanikoMounts, Resources: withDisk(limits, diskBounds{request: kanikoRq, limit: kanikoDisk}), SecurityContext: kanikoSec, } initContainers = append(initContainers, kaniko) trivyArgs := []string{ "image", "--input", imageTarPath, "--exit-code", "1", "--severity", "CRITICAL", "--no-progress", "--insecure", } // The DB source is configurable because the default (mirror.gcr.io/ghcr.io) // is exactly what the build egress lock denies: an install that never mirrors // the DB cannot complete a scan, and the gate fails closed on purpose. The // supported shape is the internal registry (`--insecure` above already covers // its plain HTTP). if p.TrivyDBRepository != "" { trivyArgs = append(trivyArgs, "--db-repository", p.TrivyDBRepository) } if p.TrivyJavaDBRepository != "" { trivyArgs = append(trivyArgs, "--java-db-repository", p.TrivyJavaDBRepository) } trivy := corev1.Container{ Name: ContainerTrivy, Image: p.TrivyImage, Args: trivyArgs, VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}}, Resources: withDisk(limits, trivyDisk), SecurityContext: sec, } initContainers = append(initContainers, trivy) // The publish step: the only container that holds the registry credential, // read from a Secret the installer materializes in this namespace. It runs // the felis binary (internal/imagepush), which only reads the tarball. pushSec := sec.DeepCopy() pushSec.ReadOnlyRootFilesystem = boolPtr(true) secretEnv := func(name, key string) corev1.EnvVar { return corev1.EnvVar{Name: name, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: naming.RegistryPushSecretName}, Key: key, }}} } push := corev1.Container{ Name: ContainerPush, Image: p.FelisImage, Args: []string{ "push-image", "--tar=" + imageTarPath, "--ref=" + p.ImageRef, "--scheme=http", }, Env: []corev1.EnvVar{ secretEnv("FELIS_REGISTRY_USERNAME", naming.RegistryPushUsernameKey), secretEnv("FELIS_REGISTRY_PASSWORD", naming.RegistryPushPasswordKey), }, VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}}, Resources: withDisk(limits, pushDisk), SecurityContext: pushSec, } job := &batchv1.Job{ ObjectMeta: metav1.ObjectMeta{ Name: BuildJobName(p.BuildID), Namespace: p.Namespace, Labels: buildLabels(p), }, Spec: batchv1.JobSpec{ // A poisoned build must not loop — one shot, then a terminal verdict. BackoffLimit: int32Ptr(0), ActiveDeadlineSeconds: int64Ptr(deadline), // ...and a finished one must not linger forever (see buildJobTTL). TTLSecondsAfterFinished: int32Ptr(int32(buildJobTTL / time.Second)), Template: corev1.PodTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)}, Spec: corev1.PodSpec{ RestartPolicy: corev1.RestartPolicyNever, ServiceAccountName: p.ServiceAccount, AutomountServiceAccountToken: boolPtr(false), // RUN steps execute in kaniko's container with root and three // capabilities; RuntimeDefault takes away the syscalls a container // never needs, among them most kernel-escape primitives (unshare, // mount, keyctl, bpf). Every step was checked to run under it live. SecurityContext: &corev1.PodSecurityContext{ SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault}, }, InitContainers: initContainers, Containers: []corev1.Container{push}, Volumes: podVolumes, }, }, }, } if p.UserNamespaces { job.Spec.Template.Spec.HostUsers = boolPtr(false) } if p.RuntimeClass != "" { rc := p.RuntimeClass job.Spec.Template.Spec.RuntimeClassName = &rc } return job, nil } // nonRootUID is the distroless nonroot user the platform image ships as. const nonRootUID = 65532 // withDisk is the resources block for one build container: the CPU and memory // caps with their schedulable floor (buildRequests), plus its ephemeral-storage // request and limit. func withDisk(limits corev1.ResourceList, d diskBounds) corev1.ResourceRequirements { lim := limits.DeepCopy() lim[corev1.ResourceEphemeralStorage] = d.limit req := buildRequests(limits) req[corev1.ResourceEphemeralStorage] = d.request return corev1.ResourceRequirements{Limits: lim, Requests: req} } // fetchArgs is the context-fetch container's argv. The digest flag rides along // only when the build pins one; admin builds from a URL they supplied have none. func fetchArgs(p JobParams) []string { args := []string{"fetch-context", "--url=" + p.ContextRef, "--out=" + contextMountPath} if p.ContextDigest != "" { args = append(args, "--sha256="+p.ContextDigest) } return args } // IsHTTPContextRef reports whether ref is an http(s) URL — the shape the submit // lane derives when the API is the blob transport — i.e. a context only the // fetch initContainer can turn into a local path for Kaniko. func IsHTTPContextRef(ref string) bool { return strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://") } // ClusterDNSPeer selects the cluster DNS pods (CoreDNS in kube-system, labelled // k8s-app=kube-dns on k3s and upstream alike) — the only resolver a sandboxed pod // needs. func ClusterDNSPeer() networkingv1.NetworkPolicyPeer { return networkingv1.NetworkPolicyPeer{ NamespaceSelector: &metav1.LabelSelector{ MatchLabels: map[string]string{"kubernetes.io/metadata.name": "kube-system"}, }, PodSelector: &metav1.LabelSelector{ MatchLabels: map[string]string{"k8s-app": "kube-dns"}, }, } } // NetPolParams parameterises the build-namespace egress lock. type NetPolParams struct { Namespace string RegistryNamespace string RegistryPort int32 // ControlNamespace and APIPort are where the felis-api internal face lives: // the fetch initContainer's only egress besides DNS and the registry. Both // defaults (felis, 8081) match platform.DefaultControlNamespace and the // internal listener, so an unset Params is still the safe shape. ControlNamespace string APIPort int32 // PackageSourceCIDRs is an optional, explicit allowlist of external package // mirrors (spec §16: egress 仅 registry + 包源). Empty means the most // locked-down default — no internet egress at all (默认拒外网). PackageSourceCIDRs []string } // BuildNetworkPolicy renders the default-deny egress policy for build Pods // (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects // build Pods by the managed-by label, denies all ingress, and allows egress // only to the cluster DNS pods, the internal registry, felis-api's internal // face, and any explicitly configured package mirrors. There is deliberately no allow-all egress rule. func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy { port := p.RegistryPort if port == 0 { port = 5000 } controlNS := p.ControlNamespace if controlNS == "" { controlNS = "felis" } apiPort := p.APIPort if apiPort == 0 { apiPort = 8081 } dnsUDP := corev1.ProtocolUDP dnsTCP := corev1.ProtocolTCP dns53 := intstr.FromInt32(53) regPort := intstr.FromInt32(port) ctxPort := intstr.FromInt32(apiPort) egress := []networkingv1.NetworkPolicyEgressRule{ // DNS resolution, to the cluster resolver only. Port 53 to ANY address // would be an exfiltration channel out of an otherwise sealed sandbox // (a Dockerfile RUN can speak DNS, or anything else, to a resolver it // controls); the cluster DNS Service is DNATed to these pods before the // policy is evaluated, so selecting them is what "resolve names" means. { To: []networkingv1.NetworkPolicyPeer{ClusterDNSPeer()}, Ports: []networkingv1.NetworkPolicyPort{ {Protocol: &dnsUDP, Port: &dns53}, {Protocol: &dnsTCP, Port: &dns53}, }, }, // The internal registry, selected by the namespace's immutable // kubernetes.io/metadata.name label, on the registry port only. { To: []networkingv1.NetworkPolicyPeer{{ NamespaceSelector: &metav1.LabelSelector{ MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.RegistryNamespace}, }, }}, Ports: []networkingv1.NetworkPolicyPort{ {Protocol: &dnsTCP, Port: ®Port}, }, }, // felis-api's internal face, where the fetch initContainer streams the // submission's build context from. Without this rule the build Pod could // not read the context and every user build would fail in its first init // step — the default-deny here is exactly why the transport had to be // planned, not assumed. { To: []networkingv1.NetworkPolicyPeer{{ NamespaceSelector: &metav1.LabelSelector{ MatchLabels: map[string]string{"kubernetes.io/metadata.name": controlNS}, }, }}, Ports: []networkingv1.NetworkPolicyPort{ {Protocol: &dnsTCP, Port: &ctxPort}, }, }, } // Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet. for _, cidr := range p.PackageSourceCIDRs { egress = append(egress, networkingv1.NetworkPolicyEgressRule{ To: []networkingv1.NetworkPolicyPeer{{ IPBlock: &networkingv1.IPBlock{CIDR: cidr}, }}, }) } return &networkingv1.NetworkPolicy{ ObjectMeta: metav1.ObjectMeta{ Name: "felis-build-egress", Namespace: p.Namespace, Labels: map[string]string{ LabelManagedBy: managedByValue, LabelComponent: componentValue, }, }, Spec: networkingv1.NetworkPolicySpec{ PodSelector: metav1.LabelSelector{ MatchLabels: map[string]string{LabelManagedBy: managedByValue}, }, PolicyTypes: []networkingv1.PolicyType{ networkingv1.PolicyTypeIngress, networkingv1.PolicyTypeEgress, }, // Empty Ingress slice = deny all ingress: nothing connects to a // build Pod. Ingress: []networkingv1.NetworkPolicyIngressRule{}, Egress: egress, }, } } // BuildServiceAccount renders the weak build SA (spec §16, §21). It is the most // dangerous identity in the platform if mis-scoped, so it is created bare: no // secrets, token auto-mounting disabled, and — by virtue of having no Role or // RoleBinding anywhere — zero K8s API permissions. Its only capability is // network reachability to push to the registry, which RBAC does not grant. func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount { return &corev1.ServiceAccount{ ObjectMeta: metav1.ObjectMeta{ Name: name, Namespace: namespace, Labels: map[string]string{ LabelManagedBy: managedByValue, LabelComponent: componentValue, }, }, AutomountServiceAccountToken: boolPtr(false), } } // BuildLimitRange bounds any container in the build namespace that arrives // without its own limits. Build Jobs set every limit themselves (BuildJob); this // is the backstop for anything else that lands in the namespace, which shares // the node's disk with the game worlds. func BuildLimitRange(namespace string) *corev1.LimitRange { return &corev1.LimitRange{ ObjectMeta: metav1.ObjectMeta{ Name: "felis-build-limits", Namespace: namespace, Labels: map[string]string{ LabelManagedBy: managedByValue, LabelComponent: componentValue, }, }, Spec: corev1.LimitRangeSpec{Limits: []corev1.LimitRangeItem{{ Type: corev1.LimitTypeContainer, Default: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("1"), corev1.ResourceMemory: resource.MustParse("1Gi"), corev1.ResourceEphemeralStorage: resource.MustParse("1Gi"), }, DefaultRequest: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("100m"), corev1.ResourceMemory: resource.MustParse("128Mi"), corev1.ResourceEphemeralStorage: resource.MustParse("64Mi"), }, }}}, } } // resourceLimits parses the CPU/memory limits into a ResourceList. func resourceLimits(cpu, mem string) (corev1.ResourceList, error) { if cpu == "" { cpu = defaultCPULimit } if mem == "" { mem = defaultMemLimit } cpuQty, err := resource.ParseQuantity(cpu) if err != nil { return nil, fmt.Errorf("build: invalid cpu limit %q: %w", cpu, err) } memQty, err := resource.ParseQuantity(mem) if err != nil { return nil, fmt.Errorf("build: invalid memory limit %q: %w", mem, err) } return corev1.ResourceList{ corev1.ResourceCPU: cpuQty, corev1.ResourceMemory: memQty, }, nil } // buildRequests is the scheduler floor a build container asks for while its // configured limit stays the safety cap. Reserving the full cap as a request is // what once made a default install on the platform's starter node (4 vCPU / // 5.5 GiB) unable to schedule ANY build — caught by the live end-to-end drill, not // by any unit test. A build is best-effort batch work: it may be throttled or // evicted under contention, which fails the Job loudly, and the caps still stop a // runaway build from exhausting the node. func buildRequests(limits corev1.ResourceList) corev1.ResourceList { req := corev1.ResourceList{} for res, floor := range map[corev1.ResourceName]resource.Quantity{ corev1.ResourceCPU: resource.MustParse("250m"), corev1.ResourceMemory: resource.MustParse("512Mi"), } { limit, ok := limits[res] if ok && limit.Cmp(floor) < 0 { floor = limit // never ask for more than the cap } req[res] = floor } return req } func boolPtr(b bool) *bool { return &b } func int32Ptr(i int32) *int32 { return &i } // quantityPtr returns a pointer to a copy of q for a VolumeSource (the API object // only ever gets serialized, but a shared pointer across rendered Jobs invites // accidental aliasing). func quantityPtr(q resource.Quantity) *resource.Quantity { return &q } func int64Ptr(i int64) *int64 { return &i }