fix(platform): registry OOM (audit #46) + loopback hostPort — the node-side pull path

Two changes to the registry Deployment, both prerequisite to GC-durable images:

- Dedicated resource template: the control plane's 256Mi memory limit was a
  live-bite bug (#46) — pushing a 475MB layer OOM-killed the registry
  mid-upload (dmesg oom-kill, oom_score_adj 989) and the push failed; the
  same push completes in 2s with 2Gi. Registry limits are now 1 CPU / 2Gi.

- The container port carries hostPort 127.0.0.1:5000. Node containerd cannot
  reach the Service VIP (live stack: "Empty reply"), so the node-side pull
  path is a registries.yaml mirror rewriting registry.<ns>.svc:5000 onto
  http://127.0.0.1:5000, which lands on this hostPort. Loopback-only keeps
  the plain-HTTP registry off every other interface.

Tests pin both: exactly one port with hostIP 127.0.0.1, and a memory limit
>= 2Gi (exceeding the control-plane template) with the #46 evidence cited.
This commit is contained in:
Lemon-miaow committed 2026-09-23 18:55:23 +08:00
1 parent 92c06ac8dc
commit a9b275abbb
2 files changed
+57 -2

No files matched your search

+42 -2
View File
@@ -77,6 +77,13 @@ const (
registryName = "registry" registryName = "registry"
registryDataPath = "/var/lib/registry" registryDataPath = "/var/lib/registry"
registryStorageSize = "10Gi" registryStorageSize = "10Gi"
// registryLoopbackHost is the interface the registry's hostPort binds. Node-level
// containerd is the only client that needs it: it cannot reach the registry Service
// VIP (the live stack answered "Empty reply" to it), so the node's registries.yaml
// mirror rewrites registry.<ns>.svc:<port> onto http://127.0.0.1:<port> and the
// pull lands here. Loopback-only is deliberate — the registry serves plain HTTP
// and must never be reachable off the node.
registryLoopbackHost = "127.0.0.1"
configVolume = "config" configVolume = "config"
tmpVolume = "tmp" tmpVolume = "tmp"
@@ -671,6 +678,13 @@ func controlPlaneDeployment(p Params, sa string, container corev1.Container, vol
// target real. The registry never calls the K8s API, so its token auto-mount is // target real. The registry never calls the K8s API, so its token auto-mount is
// disabled (matching the weak build/restore SA hygiene), and REGISTRY_HTTP_ADDR // disabled (matching the weak build/restore SA hygiene), and REGISTRY_HTTP_ADDR
// pins its listen port to the Service port instead of trusting the image default. // pins its listen port to the Service port instead of trusting the image default.
//
// The container port also carries a loopback hostPort (registryLoopbackHost): it is
// the node-side pull path. The node's containerd cannot dial the Service VIP, so
// deploy/bootstrap.sh writes a registries.yaml mirror rewriting
// registry.<registry-ns>.svc:<port> onto http://127.0.0.1:<port>, and that request
// arrives at this hostPort — which is what lets kubelet re-pull a garbage-collected
// platform image without an operator re-import.
func registryDeployment(p Params) *appsv1.Deployment { func registryDeployment(p Params) *appsv1.Deployment {
p = p.withDefaults() p = p.withDefaults()
labels := registryLabels() labels := registryLabels()
@@ -682,7 +696,12 @@ func registryDeployment(p Params) *appsv1.Deployment {
{Name: "REGISTRY_HTTP_ADDR", Value: fmt.Sprintf(":%d", p.RegistryPort)}, {Name: "REGISTRY_HTTP_ADDR", Value: fmt.Sprintf(":%d", p.RegistryPort)},
}, },
Ports: []corev1.ContainerPort{ Ports: []corev1.ContainerPort{
{Name: registryName, ContainerPort: p.RegistryPort, Protocol: corev1.ProtocolTCP}, {
Name: registryName, ContainerPort: p.RegistryPort, Protocol: corev1.ProtocolTCP,
// The node-side pull path: containerd's mirror rewrites the Service
// name onto 127.0.0.1:<port> (see the doc comment above).
HostPort: p.RegistryPort, HostIP: registryLoopbackHost,
},
}, },
VolumeMounts: []corev1.VolumeMount{ VolumeMounts: []corev1.VolumeMount{
{Name: registryVolume, MountPath: registryDataPath}, {Name: registryVolume, MountPath: registryDataPath},
@@ -710,7 +729,7 @@ func registryDeployment(p Params) *appsv1.Deployment {
TimeoutSeconds: 3, TimeoutSeconds: 3,
FailureThreshold: 3, FailureThreshold: 3,
}, },
Resources: controlPlaneResources(), Resources: registryResources(),
SecurityContext: hardenedContainerSecurityContext(), SecurityContext: hardenedContainerSecurityContext(),
} }
@@ -854,6 +873,27 @@ func controlPlaneResources() corev1.ResourceRequirements {
} }
} }
// registryResources sizes the registry for what it actually does: it is the pull
// source for every image the node runs and the push target of every build, so its
// limits are not the control plane's. The memory limit is LOAD-BEARING, not
// tuning: a live drill pushing a 475MB layer into a 256Mi registry (the old
// control-plane template) OOM-killed the registry mid-upload — dmesg oom-kill,
// oom_score_adj 989, the push failed — and the identical push completed in
// 2 seconds once the limit was 2Gi. Large layers (Kaniko-built modpacks, the game
// images) are exactly that case, so 2Gi is the floor this renderer stands on.
func registryResources() corev1.ResourceRequirements {
return corev1.ResourceRequirements{
Requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("50m"),
corev1.ResourceMemory: resource.MustParse("64Mi"),
},
Limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("1"),
corev1.ResourceMemory: resource.MustParse("2Gi"),
},
}
}
// hardenedPodSecurityContext is the pod-level hardening shared by the // hardenedPodSecurityContext is the pod-level hardening shared by the
// control-plane workloads (api, operator, registry — the world-touching reaper // control-plane workloads (api, operator, registry — the world-touching reaper
// uses reaperPodSecurityContext instead): run as a fixed non-root uid/gid with a // uses reaperPodSecurityContext instead): run as a fixed non-root uid/gid with a
+15
View File
@@ -425,6 +425,21 @@ func TestRegistry_DeploymentServicePVC(t *testing.T) {
if v := envValue(c.Env, "REGISTRY_HTTP_ADDR"); v != ":5000" { if v := envValue(c.Env, "REGISTRY_HTTP_ADDR"); v != ":5000" {
t.Errorf("REGISTRY_HTTP_ADDR = %q, want :5000", v) t.Errorf("REGISTRY_HTTP_ADDR = %q, want :5000", v)
} }
// The node-side pull path: exactly one container port, mirrored by a LOOPBACK
// hostPort. Node containerd cannot dial the Service VIP, so its registries.yaml
// mirror rewrites the Service name onto 127.0.0.1:<port>; nothing else may be
// exposed (the registry serves plain HTTP).
if len(c.Ports) != 1 {
t.Fatalf("registry container ports = %+v, want exactly 1", c.Ports)
}
if p0 := c.Ports[0]; p0.ContainerPort != p.RegistryPort || p0.HostPort != p.RegistryPort || p0.HostIP != registryLoopbackHost {
t.Errorf("registry port = %+v, want container/host port %d bound to %s", p0, p.RegistryPort, registryLoopbackHost)
}
// The registry's limits are deliberately NOT the control-plane template's: audit
// #46 caught the registry OOM-killed mid-upload at 256Mi on a real 475MB-layer push.
if mem := c.Resources.Limits[corev1.ResourceMemory]; mem.Value() < 2*1024*1024*1024 {
t.Errorf("registry memory limit = %s, want >= 2Gi (audit #46: 256Mi OOM-killed on a 475MB-layer push)", mem.String())
}
// Registry never calls the K8s API ⇒ no auto-mounted token. // Registry never calls the K8s API ⇒ no auto-mounted token.
if ps.AutomountServiceAccountToken == nil || *ps.AutomountServiceAccountToken { if ps.AutomountServiceAccountToken == nil || *ps.AutomountServiceAccountToken {
t.Error("registry pod must set automountServiceAccountToken=false") t.Error("registry pod must set automountServiceAccountToken=false")