feat(registry): 写入改走鉴权网关,构建先扫描再推送

This commit is contained in:
Lemon-miaow committed 2026-09-24 14:25:17 +08:00
1 parent 8f684cedc0
commit 3424852a39
16 files changed
+1814 -145

No files matched your search

+111 -36
View File
@@ -26,14 +26,16 @@ const (
)
// Container names within the build Pod. Kaniko is the initContainer that builds
// and pushes the image — its log IS the "build log" an admin watches (spec §16);
// Trivy is the main container whose CRITICAL-CVE verdict gates admission and is
// surfaced via the build status, not the log stream. Exported so the build-log
// streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) follows the
// same container this Job defines — one source of truth for the name.
// the image into a tarball — its log IS the "build log" an admin watches (spec
// §16); Trivy is the next initContainer, whose CRITICAL-CVE verdict gates both the
// push and admission and is surfaced via the build status, not the log stream;
// Push is the main container that publishes the scanned tarball. Exported so the
// build-log streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8)
// follows the same container this Job defines — one source of truth for the name.
const (
ContainerKaniko = "kaniko"
ContainerTrivy = "trivy"
ContainerPush = "push"
// ContainerFetch is the initContainer that pulls a submission's build context
// from the felis-api internal face and extracts it into the shared emptyDir.
// It exists only for an http(s) ContextRef (see BuildJob); a ref Kaniko can
@@ -44,8 +46,19 @@ const (
// initContainer writes the extracted tree there, Kaniko reads it read-only.
contextVolume = "context"
contextMountPath = "/context"
// imageVolume/imageTarPath carry the built image from Kaniko (--tar-path) to
// Trivy (--input) and then to the push container.
imageVolume = "image"
imageMountPath = "/image"
imageTarPath = imageMountPath + "/image.tar"
)
// imageSizeLimit bounds the built image tarball. A modpack image is typically a
// JRE, a server jar and a few hundred MiB of mods; 10 GiB leaves ample room while
// still stopping a runaway build from filling the node's disk.
var imageSizeLimit = resource.MustParse("10Gi")
// contextSizeLimit bounds the extracted (attacker-controlled) context tree so a
// tarball bomb wedges the build pod instead of the node's disk. The compressed
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
@@ -120,14 +133,25 @@ func buildLabels(p JobParams) map[string]string {
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
// automatic admission gate (spec §16).
//
// Sequencing: kaniko runs as an initContainer (build + push to the internal
// registry) and trivy as the main container (scan the pushed ref). The Pod
// succeeds only if kaniko pushed AND trivy found no CRITICAL CVE.
// Sequencing: kaniko builds with --no-push into a tarball, trivy scans that
// tarball, and only then does the push container publish it. So:
//
// - an image that fails the scan is never published — it used to be pushed to
// the final tag first and scanned after, overwriting whatever that tag held;
// - the registry credential lives in the push container alone. Kaniko executes
// the untrusted Dockerfile and holds no credential at all, and the registry
// refuses anonymous writes (internal/registrygate).
//
// The Pod succeeds only if kaniko built, trivy found no CRITICAL CVE, and the push
// landed.
func BuildJob(p JobParams) (*batchv1.Job, error) {
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
if err != nil {
return nil, err
}
if p.FelisImage == "" {
return nil, fmt.Errorf("build: FelisImage is required: the push container runs it")
}
deadline := int64(p.Deadline / time.Second)
if deadline <= 0 {
deadline = int64(defaultDeadline / time.Second)
@@ -164,12 +188,15 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
contextPath := p.ContextRef
initContainers := []corev1.Container{}
var kanikoMounts []corev1.VolumeMount
var podVolumes []corev1.Volume
imageMount := corev1.VolumeMount{Name: imageVolume, MountPath: imageMountPath}
kanikoMounts := []corev1.VolumeMount{imageMount}
podVolumes := []corev1.Volume{{
Name: imageVolume,
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
SizeLimit: quantityPtr(imageSizeLimit),
}},
}}
if isHTTPContextRef(p.ContextRef) {
if p.FelisImage == "" {
return nil, fmt.Errorf("build: context ref %q needs FelisImage for the fetch initContainer", p.ContextRef)
}
contextPath = contextMountPath
// The fetch container runs as root while Kaniko keeps the image default
// (also root): Kaniko re-copies the Dockerfile out of the context and
@@ -209,17 +236,17 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
SecurityContext: fetchSec,
}
initContainers = append(initContainers, fetch)
kanikoMounts = []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true}}
podVolumes = []corev1.Volume{{
kanikoMounts = append(kanikoMounts, corev1.VolumeMount{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true})
podVolumes = append(podVolumes, corev1.Volume{
Name: contextVolume,
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
// The extracted tree is attacker-controlled; bound it so a tarball
// bomb wedges THIS pod (admitted failure) instead of filling the
// node's disk. The compressed upload is capped at 1 GiB by the
// submit lane, and 4 GiB leaves room for a typical expansion.
SizeLimit: sizeLimitPtr(),
SizeLimit: quantityPtr(contextSizeLimit),
}},
}}
})
}
kaniko := corev1.Container{
@@ -228,16 +255,16 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
Args: []string{
"--dockerfile=Dockerfile",
"--context=" + contextPath,
// --destination only names the image inside the tarball; --no-push
// keeps Kaniko off the registry's write path entirely.
"--destination=" + p.ImageRef,
// The internal registry is in-cluster only and may serve plain HTTP;
// it is never a public ingress (spec §17). Both directions need the
// insecure flags: --insecure/--skip-tls-verify cover the PUSH, while
// the pull side needs its own pair — a Dockerfile's `FROM
// registry.felis.svc:5000/...` otherwise fails with "server gave
// HTTP response to HTTPS client", breaking every build based on a
"--no-push",
"--tar-path=" + imageTarPath,
// The internal registry is in-cluster only and serves plain HTTP; it
// is never a public ingress (spec §17). A Dockerfile's `FROM
// registry.felis.svc:5000/...` fails with "server gave HTTP response
// to HTTPS client" without these, breaking every build based on a
// platform image (the canonical modpack shape).
"--insecure",
"--skip-tls-verify",
"--insecure-pull",
"--skip-tls-verify-pull",
},
@@ -249,6 +276,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
trivyArgs := []string{
"image",
"--input", imageTarPath,
"--exit-code", "1",
"--severity", "CRITICAL",
"--no-progress",
@@ -265,14 +293,44 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
if p.TrivyJavaDBRepository != "" {
trivyArgs = append(trivyArgs, "--java-db-repository", p.TrivyJavaDBRepository)
}
trivyArgs = append(trivyArgs, p.ImageRef)
trivy := corev1.Container{
Name: ContainerTrivy,
Image: p.TrivyImage,
Args: trivyArgs,
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
SecurityContext: sec,
}
initContainers = append(initContainers, trivy)
// The publish step: the only container that holds the registry credential,
// read from a Secret the installer materializes in this namespace. It runs
// the felis binary (internal/imagepush), which only reads the tarball.
pushSec := sec.DeepCopy()
pushSec.ReadOnlyRootFilesystem = boolPtr(true)
secretEnv := func(name, key string) corev1.EnvVar {
return corev1.EnvVar{Name: name, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{Name: naming.RegistryPushSecretName},
Key: key,
}}}
}
push := corev1.Container{
Name: ContainerPush,
Image: p.FelisImage,
Args: []string{
"push-image",
"--tar=" + imageTarPath,
"--ref=" + p.ImageRef,
"--scheme=http",
},
Env: []corev1.EnvVar{
secretEnv("FELIS_REGISTRY_USERNAME", naming.RegistryPushUsernameKey),
secretEnv("FELIS_REGISTRY_PASSWORD", naming.RegistryPushPasswordKey),
},
VolumeMounts: []corev1.VolumeMount{{Name: imageVolume, MountPath: imageMountPath, ReadOnly: true}},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
SecurityContext: pushSec,
}
job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{
@@ -293,7 +351,7 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
InitContainers: initContainers,
Containers: []corev1.Container{trivy},
Containers: []corev1.Container{push},
Volumes: podVolumes,
},
},
@@ -309,6 +367,20 @@ func isHTTPContextRef(ref string) bool {
return strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://")
}
// ClusterDNSPeer selects the cluster DNS pods (CoreDNS in kube-system, labelled
// k8s-app=kube-dns on k3s and upstream alike) — the only resolver a sandboxed pod
// needs.
func ClusterDNSPeer() networkingv1.NetworkPolicyPeer {
return networkingv1.NetworkPolicyPeer{
NamespaceSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"kubernetes.io/metadata.name": "kube-system"},
},
PodSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"k8s-app": "kube-dns"},
},
}
}
// NetPolParams parameterises the build-namespace egress lock.
type NetPolParams struct {
Namespace string
@@ -329,8 +401,8 @@ type NetPolParams struct {
// BuildNetworkPolicy renders the default-deny egress policy for build Pods
// (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects
// build Pods by the managed-by label, denies all ingress, and allows egress
// only to DNS, the internal registry, and any explicitly configured package
// mirrors. There is deliberately no allow-all egress rule.
// only to the cluster DNS pods, the internal registry, felis-api's internal
// face, and any explicitly configured package mirrors. There is deliberately no allow-all egress rule.
func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
port := p.RegistryPort
if port == 0 {
@@ -351,9 +423,13 @@ func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
ctxPort := intstr.FromInt32(apiPort)
egress := []networkingv1.NetworkPolicyEgressRule{
// DNS resolution: port-restricted to 53, so this is not an open-internet
// hole — name resolution only.
// DNS resolution, to the cluster resolver only. Port 53 to ANY address
// would be an exfiltration channel out of an otherwise sealed sandbox
// (a Dockerfile RUN can speak DNS, or anything else, to a resolver it
// controls); the cluster DNS Service is DNATed to these pods before the
// policy is evaluated, so selecting them is what "resolve names" means.
{
To: []networkingv1.NetworkPolicyPeer{ClusterDNSPeer()},
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsUDP, Port: &dns53},
{Protocol: &dnsTCP, Port: &dns53},
@@ -487,11 +563,10 @@ func buildRequests(limits corev1.ResourceList) corev1.ResourceList {
func boolPtr(b bool) *bool { return &b }
func int32Ptr(i int32) *int32 { return &i }
// sizeLimitPtr returns a copy of contextSizeLimit for a VolumeSource (the API
// object only ever gets serialized, but a shared pointer across rendered Jobs
// invites accidental aliasing).
func sizeLimitPtr() *resource.Quantity {
q := contextSizeLimit
// quantityPtr returns a pointer to a copy of q for a VolumeSource (the API object
// only ever gets serialized, but a shared pointer across rendered Jobs invites
// accidental aliasing).
func quantityPtr(q resource.Quantity) *resource.Quantity {
return &q
}
func int64Ptr(i int64) *int64 { return &i }