diff --git a/internal/build/jobspec.go b/internal/build/jobspec.go index f8f363e..fe7044f 100644 --- a/internal/build/jobspec.go +++ b/internal/build/jobspec.go @@ -51,6 +51,20 @@ const ( // upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room. var contextSizeLimit = resource.MustParse("4Gi") +// buildJobTTL is how long a finished build Job survives before the Job +// controller deletes it — and with it the Pod whose kaniko log is the admin +// failure-triage surface (GET /api/v1/images/build/{id}/logs). +// +// Every other Job family the platform renders carries a TTL (fileedit 2m, +// backup/restore 10m); the build lane deliberately keeps a much longer one +// because the logs are the point. Without ANY TTL the Job and its completed +// Pod accumulate one pair per build forever: they count against the node's +// pod budget (110 on stock k3s), grow etcd, and eventually block new builds. +// Sync already tolerates a vanished Job (JobUnknown → failed; terminal builds +// are returned unchanged), so a week-old log falling off costs a 404, not a +// status flip. +const buildJobTTL = 7 * 24 * time.Hour + // JobParams are the rendered inputs to a build Job. They are derived from a // Build + Config by the Builder; jobspec is a pure function of them so the // security-critical Job shape is unit-tested without a cluster. @@ -241,6 +255,8 @@ func BuildJob(p JobParams) (*batchv1.Job, error) { // A poisoned build must not loop — one shot, then a terminal verdict. BackoffLimit: int32Ptr(0), ActiveDeadlineSeconds: int64Ptr(deadline), + // ...and a finished one must not linger forever (see buildJobTTL). + TTLSecondsAfterFinished: int32Ptr(int32(buildJobTTL / time.Second)), Template: corev1.PodTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)}, Spec: corev1.PodSpec{ diff --git a/internal/build/jobspec_test.go b/internal/build/jobspec_test.go index 05a4ae3..5ad6d8e 100644 --- a/internal/build/jobspec_test.go +++ b/internal/build/jobspec_test.go @@ -71,6 +71,23 @@ func TestBuildJobIsBoundedAndOneShot(t *testing.T) { } } +// A finished build must not squat in the namespace forever: without a TTL the +// Job and its completed Pod accumulate one pair per build and eventually eat +// the node's pod budget. The window is deliberately long (see buildJobTTL) so +// the admin log stream stays useful for triage. +func TestBuildJobIsReapedAfterCompletion(t *testing.T) { + job, err := BuildJob(sampleJobParams()) + if err != nil { + t.Fatalf("BuildJob: %v", err) + } + if job.Spec.TTLSecondsAfterFinished == nil { + t.Fatal("TTLSecondsAfterFinished must be set so the finished Job (and its log Pod) is GC'd") + } + if got, want := *job.Spec.TTLSecondsAfterFinished, int32((7*24*time.Hour)/time.Second); got != want { + t.Errorf("TTLSecondsAfterFinished = %d, want %d (buildJobTTL)", got, want) + } +} + // No build container may be privileged or able to escalate, and all containers // must carry resource limits. func TestBuildJobContainersAreHardened(t *testing.T) {