Files
Felis/internal/build/jobspec.go
T
Lemon-miaow 02fd2de502 fix(build): three drill-driven fixes so the lane actually completes on a starter node
The first live build (Kaniko v1.24, 4 vCPU / 5.5 GiB node) walked the new
transport end to end and hit three real defects, each invisible to unit tests:

- The Job requested its FULL limits (2 CPU / 4Gi per container), so the build
  Pod never scheduled on the platform's own starter node: FailedScheduling /
  Insufficient memory, Pending forever. Requests are now a small floor
  (250m / 512Mi, never above a configured cap) while the limits stay the
  safety caps.
- Kaniko re-copies the Dockerfile out of the context and chowns/chmods it to
  the source owner; a 65532-owned context (the distroless felis image uid)
  fails that under the pod's dropped capabilities ('copying dockerfile:
  chown /kaniko/Dockerfile: operation not permitted'). The fetch container
  now extracts as root — the uid Kaniko already runs as — so the copy
  succeeds; the pod was root by necessity regardless.
- Trivy's DB fetch is exactly what the build egress lock denies: the scan
  step failed closed on mirror.gcr.io. New [registry] trivy_db_repository
  renders --db-repository, and docs/troubleshooting.md §8e now carries the
  verified mirror recipe (docker pull/tag/push of aquasec/trivy-db:2 into the
  internal registry; --insecure already covers its plain HTTP).

Verified live after this batch: fetch initContainer streamed the blob through
the API + netpol + token, Kaniko built and pushed registry.felis.svc:5000/
user-uploads/sub-<id>:latest, and Trivy scanned against the mirrored DB.
2026-09-22 23:01:25 +08:00

453 lines
17 KiB
Go

package build
import (
"fmt"
"strings"
"time"
"felis.lolicon.best/internal/naming"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
networkingv1 "k8s.io/api/networking/v1"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/util/intstr"
)
// Label keys applied to build objects. ManagedBy doubles as the NetworkPolicy
// pod selector, so every build Pod is captured by the egress lock.
const (
LabelManagedBy = "app.kubernetes.io/managed-by"
LabelComponent = "app.kubernetes.io/component"
LabelBuildID = "felis.lolicon.best/build-id"
managedByValue = "felis-build"
componentValue = "image-build"
)
// Container names within the build Pod. Kaniko is the initContainer that builds
// and pushes the image — its log IS the "build log" an admin watches (spec §16);
// Trivy is the main container whose CRITICAL-CVE verdict gates admission and is
// surfaced via the build status, not the log stream. Exported so the build-log
// streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) follows the
// same container this Job defines — one source of truth for the name.
const (
ContainerKaniko = "kaniko"
ContainerTrivy = "trivy"
// ContainerFetch is the initContainer that pulls a submission's build context
// from the felis-api internal face and extracts it into the shared emptyDir.
// It exists only for an http(s) ContextRef (see BuildJob); a ref Kaniko can
// read natively (s3://) or a pre-mounted path renders no such container.
ContainerFetch = "context-fetch"
// contextVolume/contextMountPath carry a fetched build context: the fetch
// initContainer writes the extracted tree there, Kaniko reads it read-only.
contextVolume = "context"
contextMountPath = "/context"
)
// contextSizeLimit bounds the extracted (attacker-controlled) context tree so a
// tarball bomb wedges the build pod instead of the node's disk. The compressed
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
var contextSizeLimit = resource.MustParse("4Gi")
// JobParams are the rendered inputs to a build Job. They are derived from a
// Build + Config by the Builder; jobspec is a pure function of them so the
// security-critical Job shape is unit-tested without a cluster.
type JobParams struct {
BuildID string
ImageRef string
ContextRef string
Namespace string
ServiceAccount string
RegistryURL string
// FelisImage runs the context-fetch initContainer (the felis binary's
// fetch-context entrypoint). Required when ContextRef is an http(s) URL.
FelisImage string
// TrivyDBRepository overrides Trivy's vulnerability-DB source (the
// --db-repository flag). Empty keeps Trivy's own default; see
// build.Config.TrivyDBRepository for why an in-cluster install sets it.
TrivyDBRepository string
KanikoImage string
TrivyImage string
Deadline time.Duration
CPULimit string
MemLimit string
}
// BuildJobName is the deterministic Job name for a build id.
func BuildJobName(buildID string) string { return "build-" + buildID }
func buildLabels(p JobParams) map[string]string {
return map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
LabelBuildID: p.BuildID,
}
}
// BuildJob renders the Kaniko+Trivy build Job (spec §16). Every isolation
// guarantee the spec demands is encoded here and asserted by jobspec_test.go,
// because no cluster runs in this environment:
//
// - runs in the isolated felis-build namespace with the weak felis-build SA
// (never the felis-api SA) and does NOT mount the SA token, so it cannot
// reach the K8s API (spec §16, §21);
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
// so docker-in-docker / privileged is never needed (spec §16, §22);
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
// automatic admission gate (spec §16).
//
// Sequencing: kaniko runs as an initContainer (build + push to the internal
// registry) and trivy as the main container (scan the pushed ref). The Pod
// succeeds only if kaniko pushed AND trivy found no CRITICAL CVE.
func BuildJob(p JobParams) (*batchv1.Job, error) {
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
if err != nil {
return nil, err
}
deadline := int64(p.Deadline / time.Second)
if deadline <= 0 {
deadline = int64(defaultDeadline / time.Second)
}
// Hardened container security context shared by both build containers: no
// privilege, no privilege escalation, drop all capabilities. Kaniko needs a
// writable root filesystem to unpack layers, so we do not force read-only
// root here, but it gains no privilege.
sec := &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
}
// The context Kaniko reads. An http(s) ref (the submit lane's derived ref: the
// API streams the blob on its internal face, because the build Pod can neither
// mount the control-plane uploads PVC across namespaces nor hold object-store
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
contextPath := p.ContextRef
initContainers := []corev1.Container{}
var kanikoMounts []corev1.VolumeMount
var podVolumes []corev1.Volume
if isHTTPContextRef(p.ContextRef) {
if p.FelisImage == "" {
return nil, fmt.Errorf("build: context ref %q needs FelisImage for the fetch initContainer", p.ContextRef)
}
contextPath = contextMountPath
// The fetch container runs as root while Kaniko keeps the image default
// (also root): Kaniko re-copies the Dockerfile out of the context and
// chowns/chmods it to the SOURCE file's owner, which fails for any other
// owner without CAP_CHOWN/CAP_FOWNER — capabilities this pod deliberately
// drops (the live drill hit exactly this: "copying dockerfile: chown
// /kaniko/Dockerfile: operation not permitted" with the distroless uid
// 65532). Extracting as root, the uid Kaniko itself runs as, keeps the
// context owned by the only user that can satisfy that copy. The pod is
// root by necessity regardless: Kaniko unpacks base-image layers into its
// own filesystem.
fetchSec := sec.DeepCopy()
fetchSec.RunAsUser = int64Ptr(0)
fetchSec.RunAsGroup = int64Ptr(0)
fetch := corev1.Container{
Name: ContainerFetch,
Image: p.FelisImage,
Args: []string{
"fetch-context",
"--url=" + p.ContextRef,
"--out=" + contextMountPath,
},
// The internal face is service-token gated, and the token is read from a
// Secret the installer materializes in THIS namespace (secretKeyRef is
// namespace-local). It is mounted into this initContainer only: the Kaniko
// container executes the untrusted Dockerfile and must never hold it, and
// pod containers share neither environment nor PID namespace.
Env: []corev1.EnvVar{{
Name: "FELIS_SERVICE_TOKEN",
ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{Name: naming.ServiceTokenSecretName},
Key: naming.ServiceTokenSecretKey,
}},
}},
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
SecurityContext: fetchSec,
}
initContainers = append(initContainers, fetch)
kanikoMounts = []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true}}
podVolumes = []corev1.Volume{{
Name: contextVolume,
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
// The extracted tree is attacker-controlled; bound it so a tarball
// bomb wedges THIS pod (admitted failure) instead of filling the
// node's disk. The compressed upload is capped at 1 GiB by the
// submit lane, and 4 GiB leaves room for a typical expansion.
SizeLimit: sizeLimitPtr(),
}},
}}
}
kaniko := corev1.Container{
Name: ContainerKaniko,
Image: p.KanikoImage,
Args: []string{
"--dockerfile=Dockerfile",
"--context=" + contextPath,
"--destination=" + p.ImageRef,
// The internal registry is in-cluster only and may serve plain HTTP;
// it is never a public ingress (spec §17).
"--insecure",
"--skip-tls-verify",
},
VolumeMounts: kanikoMounts,
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
SecurityContext: sec,
}
initContainers = append(initContainers, kaniko)
trivyArgs := []string{
"image",
"--exit-code", "1",
"--severity", "CRITICAL",
"--no-progress",
"--insecure",
}
// The DB source is configurable because the default (mirror.gcr.io/ghcr.io)
// is exactly what the build egress lock denies: an install that never mirrors
// the DB cannot complete a scan, and the gate fails closed on purpose. The
// supported shape is the internal registry (`--insecure` above already covers
// its plain HTTP).
if p.TrivyDBRepository != "" {
trivyArgs = append(trivyArgs, "--db-repository", p.TrivyDBRepository)
}
trivyArgs = append(trivyArgs, p.ImageRef)
trivy := corev1.Container{
Name: ContainerTrivy,
Image: p.TrivyImage,
Args: trivyArgs,
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
SecurityContext: sec,
}
job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{
Name: BuildJobName(p.BuildID),
Namespace: p.Namespace,
Labels: buildLabels(p),
},
Spec: batchv1.JobSpec{
// A poisoned build must not loop — one shot, then a terminal verdict.
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(deadline),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
InitContainers: initContainers,
Containers: []corev1.Container{trivy},
Volumes: podVolumes,
},
},
},
}
return job, nil
}
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
// lane derives when the API is the blob transport — i.e. a context only the
// fetch initContainer can turn into a local path for Kaniko.
func isHTTPContextRef(ref string) bool {
return strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://")
}
// NetPolParams parameterises the build-namespace egress lock.
type NetPolParams struct {
Namespace string
RegistryNamespace string
RegistryPort int32
// ControlNamespace and APIPort are where the felis-api internal face lives:
// the fetch initContainer's only egress besides DNS and the registry. Both
// defaults (felis, 8081) match platform.DefaultControlNamespace and the
// internal listener, so an unset Params is still the safe shape.
ControlNamespace string
APIPort int32
// PackageSourceCIDRs is an optional, explicit allowlist of external package
// mirrors (spec §16: egress 仅 registry + 包源). Empty means the most
// locked-down default — no internet egress at all (默认拒外网).
PackageSourceCIDRs []string
}
// BuildNetworkPolicy renders the default-deny egress policy for build Pods
// (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects
// build Pods by the managed-by label, denies all ingress, and allows egress
// only to DNS, the internal registry, and any explicitly configured package
// mirrors. There is deliberately no allow-all egress rule.
func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
port := p.RegistryPort
if port == 0 {
port = 5000
}
controlNS := p.ControlNamespace
if controlNS == "" {
controlNS = "felis"
}
apiPort := p.APIPort
if apiPort == 0 {
apiPort = 8081
}
dnsUDP := corev1.ProtocolUDP
dnsTCP := corev1.ProtocolTCP
dns53 := intstr.FromInt32(53)
regPort := intstr.FromInt32(port)
ctxPort := intstr.FromInt32(apiPort)
egress := []networkingv1.NetworkPolicyEgressRule{
// DNS resolution: port-restricted to 53, so this is not an open-internet
// hole — name resolution only.
{
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsUDP, Port: &dns53},
{Protocol: &dnsTCP, Port: &dns53},
},
},
// The internal registry, selected by the namespace's immutable
// kubernetes.io/metadata.name label, on the registry port only.
{
To: []networkingv1.NetworkPolicyPeer{{
NamespaceSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.RegistryNamespace},
},
}},
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsTCP, Port: &regPort},
},
},
// felis-api's internal face, where the fetch initContainer streams the
// submission's build context from. Without this rule the build Pod could
// not read the context and every user build would fail in its first init
// step — the default-deny here is exactly why the transport had to be
// planned, not assumed.
{
To: []networkingv1.NetworkPolicyPeer{{
NamespaceSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"kubernetes.io/metadata.name": controlNS},
},
}},
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsTCP, Port: &ctxPort},
},
},
}
// Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet.
for _, cidr := range p.PackageSourceCIDRs {
egress = append(egress, networkingv1.NetworkPolicyEgressRule{
To: []networkingv1.NetworkPolicyPeer{{
IPBlock: &networkingv1.IPBlock{CIDR: cidr},
}},
})
}
return &networkingv1.NetworkPolicy{
ObjectMeta: metav1.ObjectMeta{
Name: "felis-build-egress",
Namespace: p.Namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
Spec: networkingv1.NetworkPolicySpec{
PodSelector: metav1.LabelSelector{
MatchLabels: map[string]string{LabelManagedBy: managedByValue},
},
PolicyTypes: []networkingv1.PolicyType{
networkingv1.PolicyTypeIngress,
networkingv1.PolicyTypeEgress,
},
// Empty Ingress slice = deny all ingress: nothing connects to a
// build Pod.
Ingress: []networkingv1.NetworkPolicyIngressRule{},
Egress: egress,
},
}
}
// BuildServiceAccount renders the weak build SA (spec §16, §21). It is the most
// dangerous identity in the platform if mis-scoped, so it is created bare: no
// secrets, token auto-mounting disabled, and — by virtue of having no Role or
// RoleBinding anywhere — zero K8s API permissions. Its only capability is
// network reachability to push to the registry, which RBAC does not grant.
func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
return &corev1.ServiceAccount{
ObjectMeta: metav1.ObjectMeta{
Name: name,
Namespace: namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
AutomountServiceAccountToken: boolPtr(false),
}
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
cpu = defaultCPULimit
}
if mem == "" {
mem = defaultMemLimit
}
cpuQty, err := resource.ParseQuantity(cpu)
if err != nil {
return nil, fmt.Errorf("build: invalid cpu limit %q: %w", cpu, err)
}
memQty, err := resource.ParseQuantity(mem)
if err != nil {
return nil, fmt.Errorf("build: invalid memory limit %q: %w", mem, err)
}
return corev1.ResourceList{
corev1.ResourceCPU: cpuQty,
corev1.ResourceMemory: memQty,
}, nil
}
// buildRequests is the scheduler floor a build container asks for while its
// configured limit stays the safety cap. Reserving the full cap as a request is
// what once made a default install on the platform's starter node (4 vCPU /
// 5.5 GiB) unable to schedule ANY build — caught by the live end-to-end drill, not
// by any unit test. A build is best-effort batch work: it may be throttled or
// evicted under contention, which fails the Job loudly, and the caps still stop a
// runaway build from exhausting the node.
func buildRequests(limits corev1.ResourceList) corev1.ResourceList {
req := corev1.ResourceList{}
for res, floor := range map[corev1.ResourceName]resource.Quantity{
corev1.ResourceCPU: resource.MustParse("250m"),
corev1.ResourceMemory: resource.MustParse("512Mi"),
} {
limit, ok := limits[res]
if ok && limit.Cmp(floor) < 0 {
floor = limit // never ask for more than the cap
}
req[res] = floor
}
return req
}
func boolPtr(b bool) *bool { return &b }
func int32Ptr(i int32) *int32 { return &i }
// sizeLimitPtr returns a copy of contextSizeLimit for a VolumeSource (the API
// object only ever gets serialized, but a shared pointer across rendered Jobs
// invites accidental aliasing).
func sizeLimitPtr() *resource.Quantity {
q := contextSizeLimit
return &q
}
func int64Ptr(i int64) *int64 { return &i }