A submitted modpack was durable but unreadable: the uploads PVC cannot cross
namespaces (felis-api mounts it; Kaniko runs in felis-build) and the s3 lane
handed the sandboxed build Pod no credentials, so NO user build could ever
consume its context. The transport is now the API itself:
- submit: derived context refs become the internal-face URL
/api/v1/internal/submissions/{id}/context (service-token gated), and Blobs
gains Open (local + s3) with an ErrBlobNotFound sentinel for the route's 404.
- api: serves that route on the internal face only (openapi.yaml updated; the
route-coverage test enforces it).
- build: an http(s) context renders a context-fetch initContainer (the felis
image's new fetch-context entrypoint) that streams the blob with the
namespace-local service-token Secret — never mounted into Kaniko — and
extracts it under a zip-slip guard into a size-limited emptyDir that Kaniko
reads read-only as --context=/context.
- platform/install: the api Deployment carries its own internal base URL; the
build namespace gets the token Secret through the existing replica mechanism
(bootstrap.sh + felis setup); the build egress lock opens exactly the control
namespace on the internal port.
- cmd/felis: fetch-context entrypoint (registered, documented, unit-tested for
escapes/symlinks/non-gzip).
Tests cover rendering, hardening, the s3/local Open paths, and the route's
404/503 mapping. Verified next on the real single-node cluster with Kaniko.
405 lines
15 KiB
Go
405 lines
15 KiB
Go
package build
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"felis.lolicon.best/internal/naming"
|
|
batchv1 "k8s.io/api/batch/v1"
|
|
corev1 "k8s.io/api/core/v1"
|
|
networkingv1 "k8s.io/api/networking/v1"
|
|
"k8s.io/apimachinery/pkg/api/resource"
|
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
|
"k8s.io/apimachinery/pkg/util/intstr"
|
|
)
|
|
|
|
// Label keys applied to build objects. ManagedBy doubles as the NetworkPolicy
|
|
// pod selector, so every build Pod is captured by the egress lock.
|
|
const (
|
|
LabelManagedBy = "app.kubernetes.io/managed-by"
|
|
LabelComponent = "app.kubernetes.io/component"
|
|
LabelBuildID = "felis.lolicon.best/build-id"
|
|
|
|
managedByValue = "felis-build"
|
|
componentValue = "image-build"
|
|
)
|
|
|
|
// Container names within the build Pod. Kaniko is the initContainer that builds
|
|
// and pushes the image — its log IS the "build log" an admin watches (spec §16);
|
|
// Trivy is the main container whose CRITICAL-CVE verdict gates admission and is
|
|
// surfaced via the build status, not the log stream. Exported so the build-log
|
|
// streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) follows the
|
|
// same container this Job defines — one source of truth for the name.
|
|
const (
|
|
ContainerKaniko = "kaniko"
|
|
ContainerTrivy = "trivy"
|
|
// ContainerFetch is the initContainer that pulls a submission's build context
|
|
// from the felis-api internal face and extracts it into the shared emptyDir.
|
|
// It exists only for an http(s) ContextRef (see BuildJob); a ref Kaniko can
|
|
// read natively (s3://) or a pre-mounted path renders no such container.
|
|
ContainerFetch = "context-fetch"
|
|
|
|
// contextVolume/contextMountPath carry a fetched build context: the fetch
|
|
// initContainer writes the extracted tree there, Kaniko reads it read-only.
|
|
contextVolume = "context"
|
|
contextMountPath = "/context"
|
|
)
|
|
|
|
// contextSizeLimit bounds the extracted (attacker-controlled) context tree so a
|
|
// tarball bomb wedges the build pod instead of the node's disk. The compressed
|
|
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
|
|
var contextSizeLimit = resource.MustParse("4Gi")
|
|
|
|
// JobParams are the rendered inputs to a build Job. They are derived from a
|
|
// Build + Config by the Builder; jobspec is a pure function of them so the
|
|
// security-critical Job shape is unit-tested without a cluster.
|
|
type JobParams struct {
|
|
BuildID string
|
|
ImageRef string
|
|
ContextRef string
|
|
Namespace string
|
|
ServiceAccount string
|
|
RegistryURL string
|
|
// FelisImage runs the context-fetch initContainer (the felis binary's
|
|
// fetch-context entrypoint). Required when ContextRef is an http(s) URL.
|
|
FelisImage string
|
|
KanikoImage string
|
|
TrivyImage string
|
|
Deadline time.Duration
|
|
CPULimit string
|
|
MemLimit string
|
|
}
|
|
|
|
// BuildJobName is the deterministic Job name for a build id.
|
|
func BuildJobName(buildID string) string { return "build-" + buildID }
|
|
|
|
func buildLabels(p JobParams) map[string]string {
|
|
return map[string]string{
|
|
LabelManagedBy: managedByValue,
|
|
LabelComponent: componentValue,
|
|
LabelBuildID: p.BuildID,
|
|
}
|
|
}
|
|
|
|
// BuildJob renders the Kaniko+Trivy build Job (spec §16). Every isolation
|
|
// guarantee the spec demands is encoded here and asserted by jobspec_test.go,
|
|
// because no cluster runs in this environment:
|
|
//
|
|
// - runs in the isolated felis-build namespace with the weak felis-build SA
|
|
// (never the felis-api SA) and does NOT mount the SA token, so it cannot
|
|
// reach the K8s API (spec §16, §21);
|
|
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
|
|
// so docker-in-docker / privileged is never needed (spec §16, §22);
|
|
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
|
|
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
|
|
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
|
|
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
|
|
// automatic admission gate (spec §16).
|
|
//
|
|
// Sequencing: kaniko runs as an initContainer (build + push to the internal
|
|
// registry) and trivy as the main container (scan the pushed ref). The Pod
|
|
// succeeds only if kaniko pushed AND trivy found no CRITICAL CVE.
|
|
func BuildJob(p JobParams) (*batchv1.Job, error) {
|
|
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
deadline := int64(p.Deadline / time.Second)
|
|
if deadline <= 0 {
|
|
deadline = int64(defaultDeadline / time.Second)
|
|
}
|
|
|
|
// Hardened container security context shared by both build containers: no
|
|
// privilege, no privilege escalation, drop all capabilities. Kaniko needs a
|
|
// writable root filesystem to unpack layers, so we do not force read-only
|
|
// root here, but it gains no privilege.
|
|
sec := &corev1.SecurityContext{
|
|
Privileged: boolPtr(false),
|
|
AllowPrivilegeEscalation: boolPtr(false),
|
|
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
|
}
|
|
|
|
// The context Kaniko reads. An http(s) ref (the submit lane's derived ref: the
|
|
// API streams the blob on its internal face, because the build Pod can neither
|
|
// mount the control-plane uploads PVC across namespaces nor hold object-store
|
|
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
|
|
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
|
|
contextPath := p.ContextRef
|
|
initContainers := []corev1.Container{}
|
|
var kanikoMounts []corev1.VolumeMount
|
|
var podVolumes []corev1.Volume
|
|
if isHTTPContextRef(p.ContextRef) {
|
|
if p.FelisImage == "" {
|
|
return nil, fmt.Errorf("build: context ref %q needs FelisImage for the fetch initContainer", p.ContextRef)
|
|
}
|
|
contextPath = contextMountPath
|
|
fetch := corev1.Container{
|
|
Name: ContainerFetch,
|
|
Image: p.FelisImage,
|
|
Args: []string{
|
|
"fetch-context",
|
|
"--url=" + p.ContextRef,
|
|
"--out=" + contextMountPath,
|
|
},
|
|
// The internal face is service-token gated, and the token is read from a
|
|
// Secret the installer materializes in THIS namespace (secretKeyRef is
|
|
// namespace-local). It is mounted into this initContainer only: the Kaniko
|
|
// container executes the untrusted Dockerfile and must never hold it, and
|
|
// pod containers share neither environment nor PID namespace.
|
|
Env: []corev1.EnvVar{{
|
|
Name: "FELIS_SERVICE_TOKEN",
|
|
ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
|
|
LocalObjectReference: corev1.LocalObjectReference{Name: naming.ServiceTokenSecretName},
|
|
Key: naming.ServiceTokenSecretKey,
|
|
}},
|
|
}},
|
|
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
|
|
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
|
SecurityContext: sec,
|
|
}
|
|
initContainers = append(initContainers, fetch)
|
|
kanikoMounts = []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true}}
|
|
podVolumes = []corev1.Volume{{
|
|
Name: contextVolume,
|
|
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
|
|
// The extracted tree is attacker-controlled; bound it so a tarball
|
|
// bomb wedges THIS pod (admitted failure) instead of filling the
|
|
// node's disk. The compressed upload is capped at 1 GiB by the
|
|
// submit lane, and 4 GiB leaves room for a typical expansion.
|
|
SizeLimit: sizeLimitPtr(),
|
|
}},
|
|
}}
|
|
}
|
|
|
|
kaniko := corev1.Container{
|
|
Name: ContainerKaniko,
|
|
Image: p.KanikoImage,
|
|
Args: []string{
|
|
"--dockerfile=Dockerfile",
|
|
"--context=" + contextPath,
|
|
"--destination=" + p.ImageRef,
|
|
// The internal registry is in-cluster only and may serve plain HTTP;
|
|
// it is never a public ingress (spec §17).
|
|
"--insecure",
|
|
"--skip-tls-verify",
|
|
},
|
|
VolumeMounts: kanikoMounts,
|
|
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
|
SecurityContext: sec,
|
|
}
|
|
initContainers = append(initContainers, kaniko)
|
|
|
|
trivy := corev1.Container{
|
|
Name: ContainerTrivy,
|
|
Image: p.TrivyImage,
|
|
Args: []string{
|
|
"image",
|
|
"--exit-code", "1",
|
|
"--severity", "CRITICAL",
|
|
"--no-progress",
|
|
"--insecure",
|
|
p.ImageRef,
|
|
},
|
|
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
|
SecurityContext: sec,
|
|
}
|
|
|
|
job := &batchv1.Job{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: BuildJobName(p.BuildID),
|
|
Namespace: p.Namespace,
|
|
Labels: buildLabels(p),
|
|
},
|
|
Spec: batchv1.JobSpec{
|
|
// A poisoned build must not loop — one shot, then a terminal verdict.
|
|
BackoffLimit: int32Ptr(0),
|
|
ActiveDeadlineSeconds: int64Ptr(deadline),
|
|
Template: corev1.PodTemplateSpec{
|
|
ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)},
|
|
Spec: corev1.PodSpec{
|
|
RestartPolicy: corev1.RestartPolicyNever,
|
|
ServiceAccountName: p.ServiceAccount,
|
|
AutomountServiceAccountToken: boolPtr(false),
|
|
InitContainers: initContainers,
|
|
Containers: []corev1.Container{trivy},
|
|
Volumes: podVolumes,
|
|
},
|
|
},
|
|
},
|
|
}
|
|
return job, nil
|
|
}
|
|
|
|
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
|
|
// lane derives when the API is the blob transport — i.e. a context only the
|
|
// fetch initContainer can turn into a local path for Kaniko.
|
|
func isHTTPContextRef(ref string) bool {
|
|
return strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://")
|
|
}
|
|
|
|
// NetPolParams parameterises the build-namespace egress lock.
|
|
type NetPolParams struct {
|
|
Namespace string
|
|
RegistryNamespace string
|
|
RegistryPort int32
|
|
// ControlNamespace and APIPort are where the felis-api internal face lives:
|
|
// the fetch initContainer's only egress besides DNS and the registry. Both
|
|
// defaults (felis, 8081) match platform.DefaultControlNamespace and the
|
|
// internal listener, so an unset Params is still the safe shape.
|
|
ControlNamespace string
|
|
APIPort int32
|
|
// PackageSourceCIDRs is an optional, explicit allowlist of external package
|
|
// mirrors (spec §16: egress 仅 registry + 包源). Empty means the most
|
|
// locked-down default — no internet egress at all (默认拒外网).
|
|
PackageSourceCIDRs []string
|
|
}
|
|
|
|
// BuildNetworkPolicy renders the default-deny egress policy for build Pods
|
|
// (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects
|
|
// build Pods by the managed-by label, denies all ingress, and allows egress
|
|
// only to DNS, the internal registry, and any explicitly configured package
|
|
// mirrors. There is deliberately no allow-all egress rule.
|
|
func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
|
|
port := p.RegistryPort
|
|
if port == 0 {
|
|
port = 5000
|
|
}
|
|
controlNS := p.ControlNamespace
|
|
if controlNS == "" {
|
|
controlNS = "felis"
|
|
}
|
|
apiPort := p.APIPort
|
|
if apiPort == 0 {
|
|
apiPort = 8081
|
|
}
|
|
dnsUDP := corev1.ProtocolUDP
|
|
dnsTCP := corev1.ProtocolTCP
|
|
dns53 := intstr.FromInt32(53)
|
|
regPort := intstr.FromInt32(port)
|
|
ctxPort := intstr.FromInt32(apiPort)
|
|
|
|
egress := []networkingv1.NetworkPolicyEgressRule{
|
|
// DNS resolution: port-restricted to 53, so this is not an open-internet
|
|
// hole — name resolution only.
|
|
{
|
|
Ports: []networkingv1.NetworkPolicyPort{
|
|
{Protocol: &dnsUDP, Port: &dns53},
|
|
{Protocol: &dnsTCP, Port: &dns53},
|
|
},
|
|
},
|
|
// The internal registry, selected by the namespace's immutable
|
|
// kubernetes.io/metadata.name label, on the registry port only.
|
|
{
|
|
To: []networkingv1.NetworkPolicyPeer{{
|
|
NamespaceSelector: &metav1.LabelSelector{
|
|
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.RegistryNamespace},
|
|
},
|
|
}},
|
|
Ports: []networkingv1.NetworkPolicyPort{
|
|
{Protocol: &dnsTCP, Port: ®Port},
|
|
},
|
|
},
|
|
// felis-api's internal face, where the fetch initContainer streams the
|
|
// submission's build context from. Without this rule the build Pod could
|
|
// not read the context and every user build would fail in its first init
|
|
// step — the default-deny here is exactly why the transport had to be
|
|
// planned, not assumed.
|
|
{
|
|
To: []networkingv1.NetworkPolicyPeer{{
|
|
NamespaceSelector: &metav1.LabelSelector{
|
|
MatchLabels: map[string]string{"kubernetes.io/metadata.name": controlNS},
|
|
},
|
|
}},
|
|
Ports: []networkingv1.NetworkPolicyPort{
|
|
{Protocol: &dnsTCP, Port: &ctxPort},
|
|
},
|
|
},
|
|
}
|
|
// Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet.
|
|
for _, cidr := range p.PackageSourceCIDRs {
|
|
egress = append(egress, networkingv1.NetworkPolicyEgressRule{
|
|
To: []networkingv1.NetworkPolicyPeer{{
|
|
IPBlock: &networkingv1.IPBlock{CIDR: cidr},
|
|
}},
|
|
})
|
|
}
|
|
|
|
return &networkingv1.NetworkPolicy{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: "felis-build-egress",
|
|
Namespace: p.Namespace,
|
|
Labels: map[string]string{
|
|
LabelManagedBy: managedByValue,
|
|
LabelComponent: componentValue,
|
|
},
|
|
},
|
|
Spec: networkingv1.NetworkPolicySpec{
|
|
PodSelector: metav1.LabelSelector{
|
|
MatchLabels: map[string]string{LabelManagedBy: managedByValue},
|
|
},
|
|
PolicyTypes: []networkingv1.PolicyType{
|
|
networkingv1.PolicyTypeIngress,
|
|
networkingv1.PolicyTypeEgress,
|
|
},
|
|
// Empty Ingress slice = deny all ingress: nothing connects to a
|
|
// build Pod.
|
|
Ingress: []networkingv1.NetworkPolicyIngressRule{},
|
|
Egress: egress,
|
|
},
|
|
}
|
|
}
|
|
|
|
// BuildServiceAccount renders the weak build SA (spec §16, §21). It is the most
|
|
// dangerous identity in the platform if mis-scoped, so it is created bare: no
|
|
// secrets, token auto-mounting disabled, and — by virtue of having no Role or
|
|
// RoleBinding anywhere — zero K8s API permissions. Its only capability is
|
|
// network reachability to push to the registry, which RBAC does not grant.
|
|
func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
|
|
return &corev1.ServiceAccount{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: name,
|
|
Namespace: namespace,
|
|
Labels: map[string]string{
|
|
LabelManagedBy: managedByValue,
|
|
LabelComponent: componentValue,
|
|
},
|
|
},
|
|
AutomountServiceAccountToken: boolPtr(false),
|
|
}
|
|
}
|
|
|
|
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
|
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
|
if cpu == "" {
|
|
cpu = defaultCPULimit
|
|
}
|
|
if mem == "" {
|
|
mem = defaultMemLimit
|
|
}
|
|
cpuQty, err := resource.ParseQuantity(cpu)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("build: invalid cpu limit %q: %w", cpu, err)
|
|
}
|
|
memQty, err := resource.ParseQuantity(mem)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("build: invalid memory limit %q: %w", mem, err)
|
|
}
|
|
return corev1.ResourceList{
|
|
corev1.ResourceCPU: cpuQty,
|
|
corev1.ResourceMemory: memQty,
|
|
}, nil
|
|
}
|
|
|
|
func boolPtr(b bool) *bool { return &b }
|
|
func int32Ptr(i int32) *int32 { return &i }
|
|
|
|
// sizeLimitPtr returns a copy of contextSizeLimit for a VolumeSource (the API
|
|
// object only ever gets serialized, but a shared pointer across rendered Jobs
|
|
// invites accidental aliasing).
|
|
func sizeLimitPtr() *resource.Quantity {
|
|
q := contextSizeLimit
|
|
return &q
|
|
}
|
|
func int64Ptr(i int64) *int64 { return &i }
|