From 2755e41ff3154af2e0dbb6e7693631284113cfdd Mon Sep 17 00:00:00 2001 From: Lemon-miaow Date: Thu, 24 Sep 2026 02:03:20 +0800 Subject: [PATCH] =?UTF-8?q?fix(build):=20reap=20finished=20build=20Jobs=20?= =?UTF-8?q?=E2=80=94=20they=20accumulated=20forever?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every other Job family carries a TTLSecondsAfterFinished (fileedit 2m, backup/restore 10m) but the build lane never set one: each build left a completed Job + Pod in felis-build indefinitely (5 already on the drill cluster, oldest 26h), growing etcd and — since completed pods count against the node's pod budget (110 on stock k3s) — eventually blocking new builds. The code even anticipated GC it never got ("JobUnknown means the Job was not found (e.g. GC'd)"). Set a deliberately long TTL (7 days): the kaniko log is the admin failure-triage surface (GET /images/build/{id}/logs), so the window keeps a week of logs while bounding steady-state pods; terminal builds are idempotent under Sync, so a late log-404 is the only cost. --- internal/build/jobspec.go | 16 ++++++++++++++++ internal/build/jobspec_test.go | 17 +++++++++++++++++ 2 files changed, 33 insertions(+) diff --git a/internal/build/jobspec.go b/internal/build/jobspec.go index f8f363e..fe7044f 100644 --- a/internal/build/jobspec.go +++ b/internal/build/jobspec.go @@ -51,6 +51,20 @@ const ( // upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room. var contextSizeLimit = resource.MustParse("4Gi") +// buildJobTTL is how long a finished build Job survives before the Job +// controller deletes it — and with it the Pod whose kaniko log is the admin +// failure-triage surface (GET /api/v1/images/build/{id}/logs). +// +// Every other Job family the platform renders carries a TTL (fileedit 2m, +// backup/restore 10m); the build lane deliberately keeps a much longer one +// because the logs are the point. Without ANY TTL the Job and its completed +// Pod accumulate one pair per build forever: they count against the node's +// pod budget (110 on stock k3s), grow etcd, and eventually block new builds. +// Sync already tolerates a vanished Job (JobUnknown → failed; terminal builds +// are returned unchanged), so a week-old log falling off costs a 404, not a +// status flip. +const buildJobTTL = 7 * 24 * time.Hour + // JobParams are the rendered inputs to a build Job. They are derived from a // Build + Config by the Builder; jobspec is a pure function of them so the // security-critical Job shape is unit-tested without a cluster. @@ -241,6 +255,8 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { // A poisoned build must not loop — one shot, then a terminal verdict. BackoffLimit: int32Ptr(0), ActiveDeadlineSeconds: int64Ptr(deadline), + // ...and a finished one must not linger forever (see buildJobTTL). + TTLSecondsAfterFinished: int32Ptr(int32(buildJobTTL / time.Second)), Template: corev1.PodTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)}, Spec: corev1.PodSpec{ diff --git a/internal/build/jobspec_test.go b/internal/build/jobspec_test.go index 05a4ae3..5ad6d8e 100644 --- a/internal/build/jobspec_test.go +++ b/internal/build/jobspec_test.go @@ -71,6 +71,23 @@ func TestBuildJobIsBoundedAndOneShot(t *testing.T) { } } +// A finished build must not squat in the namespace forever: without a TTL the +// Job and its completed Pod accumulate one pair per build and eventually eat +// the node's pod budget. The window is deliberately long (see buildJobTTL) so +// the admin log stream stays useful for triage. +func TestBuildJobIsReapedAfterCompletion(t *testing.T) { + job, err := BuildJob(sampleJobParams()) + if err != nil { + t.Fatalf("BuildJob: %v", err) + } + if job.Spec.TTLSecondsAfterFinished == nil { + t.Fatal("TTLSecondsAfterFinished must be set so the finished Job (and its log Pod) is GC'd") + } + if got, want := *job.Spec.TTLSecondsAfterFinished, int32((7*24*time.Hour)/time.Second); got != want { + t.Errorf("TTLSecondsAfterFinished = %d, want %d (buildJobTTL)", got, want) + } +} + // No build container may be privileged or able to escalate, and all containers // must carry resource limits. func TestBuildJobContainersAreHardened(t *testing.T) {