Files
Felis/internal/build/jobspec.go
T
Lemon-miaow f79e5ebb5e feat(build): make the user-modpack build lane read its context (closes the last functional gap)
A submitted modpack was durable but unreadable: the uploads PVC cannot cross
namespaces (felis-api mounts it; Kaniko runs in felis-build) and the s3 lane
handed the sandboxed build Pod no credentials, so NO user build could ever
consume its context. The transport is now the API itself:

- submit: derived context refs become the internal-face URL
  /api/v1/internal/submissions/{id}/context (service-token gated), and Blobs
  gains Open (local + s3) with an ErrBlobNotFound sentinel for the route's 404.
- api: serves that route on the internal face only (openapi.yaml updated; the
  route-coverage test enforces it).
- build: an http(s) context renders a context-fetch initContainer (the felis
  image's new fetch-context entrypoint) that streams the blob with the
  namespace-local service-token Secret — never mounted into Kaniko — and
  extracts it under a zip-slip guard into a size-limited emptyDir that Kaniko
  reads read-only as --context=/context.
- platform/install: the api Deployment carries its own internal base URL; the
  build namespace gets the token Secret through the existing replica mechanism
  (bootstrap.sh + felis setup); the build egress lock opens exactly the control
  namespace on the internal port.
- cmd/felis: fetch-context entrypoint (registered, documented, unit-tested for
  escapes/symlinks/non-gzip).

Tests cover rendering, hardening, the s3/local Open paths, and the route's
404/503 mapping. Verified next on the real single-node cluster with Kaniko.
2026-09-22 22:45:09 +08:00

405 lines
15 KiB
Go

package build
import (
"fmt"
"strings"
"time"
"felis.lolicon.best/internal/naming"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
networkingv1 "k8s.io/api/networking/v1"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/util/intstr"
)
// Label keys applied to build objects. ManagedBy doubles as the NetworkPolicy
// pod selector, so every build Pod is captured by the egress lock.
const (
LabelManagedBy = "app.kubernetes.io/managed-by"
LabelComponent = "app.kubernetes.io/component"
LabelBuildID = "felis.lolicon.best/build-id"
managedByValue = "felis-build"
componentValue = "image-build"
)
// Container names within the build Pod. Kaniko is the initContainer that builds
// and pushes the image — its log IS the "build log" an admin watches (spec §16);
// Trivy is the main container whose CRITICAL-CVE verdict gates admission and is
// surfaced via the build status, not the log stream. Exported so the build-log
// streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) follows the
// same container this Job defines — one source of truth for the name.
const (
ContainerKaniko = "kaniko"
ContainerTrivy = "trivy"
// ContainerFetch is the initContainer that pulls a submission's build context
// from the felis-api internal face and extracts it into the shared emptyDir.
// It exists only for an http(s) ContextRef (see BuildJob); a ref Kaniko can
// read natively (s3://) or a pre-mounted path renders no such container.
ContainerFetch = "context-fetch"
// contextVolume/contextMountPath carry a fetched build context: the fetch
// initContainer writes the extracted tree there, Kaniko reads it read-only.
contextVolume = "context"
contextMountPath = "/context"
)
// contextSizeLimit bounds the extracted (attacker-controlled) context tree so a
// tarball bomb wedges the build pod instead of the node's disk. The compressed
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
var contextSizeLimit = resource.MustParse("4Gi")
// JobParams are the rendered inputs to a build Job. They are derived from a
// Build + Config by the Builder; jobspec is a pure function of them so the
// security-critical Job shape is unit-tested without a cluster.
type JobParams struct {
BuildID string
ImageRef string
ContextRef string
Namespace string
ServiceAccount string
RegistryURL string
// FelisImage runs the context-fetch initContainer (the felis binary's
// fetch-context entrypoint). Required when ContextRef is an http(s) URL.
FelisImage string
KanikoImage string
TrivyImage string
Deadline time.Duration
CPULimit string
MemLimit string
}
// BuildJobName is the deterministic Job name for a build id.
func BuildJobName(buildID string) string { return "build-" + buildID }
func buildLabels(p JobParams) map[string]string {
return map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
LabelBuildID: p.BuildID,
}
}
// BuildJob renders the Kaniko+Trivy build Job (spec §16). Every isolation
// guarantee the spec demands is encoded here and asserted by jobspec_test.go,
// because no cluster runs in this environment:
//
// - runs in the isolated felis-build namespace with the weak felis-build SA
// (never the felis-api SA) and does NOT mount the SA token, so it cannot
// reach the K8s API (spec §16, §21);
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
// so docker-in-docker / privileged is never needed (spec §16, §22);
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
// automatic admission gate (spec §16).
//
// Sequencing: kaniko runs as an initContainer (build + push to the internal
// registry) and trivy as the main container (scan the pushed ref). The Pod
// succeeds only if kaniko pushed AND trivy found no CRITICAL CVE.
func BuildJob(p JobParams) (*batchv1.Job, error) {
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
if err != nil {
return nil, err
}
deadline := int64(p.Deadline / time.Second)
if deadline <= 0 {
deadline = int64(defaultDeadline / time.Second)
}
// Hardened container security context shared by both build containers: no
// privilege, no privilege escalation, drop all capabilities. Kaniko needs a
// writable root filesystem to unpack layers, so we do not force read-only
// root here, but it gains no privilege.
sec := &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
}
// The context Kaniko reads. An http(s) ref (the submit lane's derived ref: the
// API streams the blob on its internal face, because the build Pod can neither
// mount the control-plane uploads PVC across namespaces nor hold object-store
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
contextPath := p.ContextRef
initContainers := []corev1.Container{}
var kanikoMounts []corev1.VolumeMount
var podVolumes []corev1.Volume
if isHTTPContextRef(p.ContextRef) {
if p.FelisImage == "" {
return nil, fmt.Errorf("build: context ref %q needs FelisImage for the fetch initContainer", p.ContextRef)
}
contextPath = contextMountPath
fetch := corev1.Container{
Name: ContainerFetch,
Image: p.FelisImage,
Args: []string{
"fetch-context",
"--url=" + p.ContextRef,
"--out=" + contextMountPath,
},
// The internal face is service-token gated, and the token is read from a
// Secret the installer materializes in THIS namespace (secretKeyRef is
// namespace-local). It is mounted into this initContainer only: the Kaniko
// container executes the untrusted Dockerfile and must never hold it, and
// pod containers share neither environment nor PID namespace.
Env: []corev1.EnvVar{{
Name: "FELIS_SERVICE_TOKEN",
ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{Name: naming.ServiceTokenSecretName},
Key: naming.ServiceTokenSecretKey,
}},
}},
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: sec,
}
initContainers = append(initContainers, fetch)
kanikoMounts = []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true}}
podVolumes = []corev1.Volume{{
Name: contextVolume,
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
// The extracted tree is attacker-controlled; bound it so a tarball
// bomb wedges THIS pod (admitted failure) instead of filling the
// node's disk. The compressed upload is capped at 1 GiB by the
// submit lane, and 4 GiB leaves room for a typical expansion.
SizeLimit: sizeLimitPtr(),
}},
}}
}
kaniko := corev1.Container{
Name: ContainerKaniko,
Image: p.KanikoImage,
Args: []string{
"--dockerfile=Dockerfile",
"--context=" + contextPath,
"--destination=" + p.ImageRef,
// The internal registry is in-cluster only and may serve plain HTTP;
// it is never a public ingress (spec §17).
"--insecure",
"--skip-tls-verify",
},
VolumeMounts: kanikoMounts,
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: sec,
}
initContainers = append(initContainers, kaniko)
trivy := corev1.Container{
Name: ContainerTrivy,
Image: p.TrivyImage,
Args: []string{
"image",
"--exit-code", "1",
"--severity", "CRITICAL",
"--no-progress",
"--insecure",
p.ImageRef,
},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: sec,
}
job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{
Name: BuildJobName(p.BuildID),
Namespace: p.Namespace,
Labels: buildLabels(p),
},
Spec: batchv1.JobSpec{
// A poisoned build must not loop — one shot, then a terminal verdict.
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(deadline),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
InitContainers: initContainers,
Containers: []corev1.Container{trivy},
Volumes: podVolumes,
},
},
},
}
return job, nil
}
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
// lane derives when the API is the blob transport — i.e. a context only the
// fetch initContainer can turn into a local path for Kaniko.
func isHTTPContextRef(ref string) bool {
return strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://")
}
// NetPolParams parameterises the build-namespace egress lock.
type NetPolParams struct {
Namespace string
RegistryNamespace string
RegistryPort int32
// ControlNamespace and APIPort are where the felis-api internal face lives:
// the fetch initContainer's only egress besides DNS and the registry. Both
// defaults (felis, 8081) match platform.DefaultControlNamespace and the
// internal listener, so an unset Params is still the safe shape.
ControlNamespace string
APIPort int32
// PackageSourceCIDRs is an optional, explicit allowlist of external package
// mirrors (spec §16: egress 仅 registry + 包源). Empty means the most
// locked-down default — no internet egress at all (默认拒外网).
PackageSourceCIDRs []string
}
// BuildNetworkPolicy renders the default-deny egress policy for build Pods
// (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects
// build Pods by the managed-by label, denies all ingress, and allows egress
// only to DNS, the internal registry, and any explicitly configured package
// mirrors. There is deliberately no allow-all egress rule.
func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
port := p.RegistryPort
if port == 0 {
port = 5000
}
controlNS := p.ControlNamespace
if controlNS == "" {
controlNS = "felis"
}
apiPort := p.APIPort
if apiPort == 0 {
apiPort = 8081
}
dnsUDP := corev1.ProtocolUDP
dnsTCP := corev1.ProtocolTCP
dns53 := intstr.FromInt32(53)
regPort := intstr.FromInt32(port)
ctxPort := intstr.FromInt32(apiPort)
egress := []networkingv1.NetworkPolicyEgressRule{
// DNS resolution: port-restricted to 53, so this is not an open-internet
// hole — name resolution only.
{
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsUDP, Port: &dns53},
{Protocol: &dnsTCP, Port: &dns53},
},
},
// The internal registry, selected by the namespace's immutable
// kubernetes.io/metadata.name label, on the registry port only.
{
To: []networkingv1.NetworkPolicyPeer{{
NamespaceSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.RegistryNamespace},
},
}},
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsTCP, Port: &regPort},
},
},
// felis-api's internal face, where the fetch initContainer streams the
// submission's build context from. Without this rule the build Pod could
// not read the context and every user build would fail in its first init
// step — the default-deny here is exactly why the transport had to be
// planned, not assumed.
{
To: []networkingv1.NetworkPolicyPeer{{
NamespaceSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"kubernetes.io/metadata.name": controlNS},
},
}},
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsTCP, Port: &ctxPort},
},
},
}
// Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet.
for _, cidr := range p.PackageSourceCIDRs {
egress = append(egress, networkingv1.NetworkPolicyEgressRule{
To: []networkingv1.NetworkPolicyPeer{{
IPBlock: &networkingv1.IPBlock{CIDR: cidr},
}},
})
}
return &networkingv1.NetworkPolicy{
ObjectMeta: metav1.ObjectMeta{
Name: "felis-build-egress",
Namespace: p.Namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
Spec: networkingv1.NetworkPolicySpec{
PodSelector: metav1.LabelSelector{
MatchLabels: map[string]string{LabelManagedBy: managedByValue},
},
PolicyTypes: []networkingv1.PolicyType{
networkingv1.PolicyTypeIngress,
networkingv1.PolicyTypeEgress,
},
// Empty Ingress slice = deny all ingress: nothing connects to a
// build Pod.
Ingress: []networkingv1.NetworkPolicyIngressRule{},
Egress: egress,
},
}
}
// BuildServiceAccount renders the weak build SA (spec §16, §21). It is the most
// dangerous identity in the platform if mis-scoped, so it is created bare: no
// secrets, token auto-mounting disabled, and — by virtue of having no Role or
// RoleBinding anywhere — zero K8s API permissions. Its only capability is
// network reachability to push to the registry, which RBAC does not grant.
func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
return &corev1.ServiceAccount{
ObjectMeta: metav1.ObjectMeta{
Name: name,
Namespace: namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
AutomountServiceAccountToken: boolPtr(false),
}
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
cpu = defaultCPULimit
}
if mem == "" {
mem = defaultMemLimit
}
cpuQty, err := resource.ParseQuantity(cpu)
if err != nil {
return nil, fmt.Errorf("build: invalid cpu limit %q: %w", cpu, err)
}
memQty, err := resource.ParseQuantity(mem)
if err != nil {
return nil, fmt.Errorf("build: invalid memory limit %q: %w", mem, err)
}
return corev1.ResourceList{
corev1.ResourceCPU: cpuQty,
corev1.ResourceMemory: memQty,
}, nil
}
func boolPtr(b bool) *bool { return &b }
func int32Ptr(i int32) *int32 { return &i }
// sizeLimitPtr returns a copy of contextSizeLimit for a VolumeSource (the API
// object only ever gets serialized, but a shared pointer across rendered Jobs
// invites accidental aliasing).
func sizeLimitPtr() *resource.Quantity {
q := contextSizeLimit
return &q
}
func int64Ptr(i int64) *int64 { return &i }