feat(core): add naming, RCON, store, config, and image-build libraries
Foundational libraries: deterministic resource naming, the RCON client, the Postgres store with embedded SQL migrations, configuration loading, and container image-build helpers.
This commit is contained in:
18 files changed
+3475
No files matched your search
@@ -0,0 +1,505 @@
|
||||
// Package build implements the image build subsystem (spec §16) — "the
|
||||
// platform's biggest security surface". A SysAdmin uploads a Dockerfile and a
|
||||
// context tarball; felis-api starts an in-cluster Kaniko Job that builds and
|
||||
// pushes to the internal registry, after which a Trivy scan gates admission to
|
||||
// the image whitelist.
|
||||
//
|
||||
// Trust model (spec §16, §22): we trust the SysAdmin at the *ingress* (only an
|
||||
// admin through Zero Trust may submit a build) but never trust the *Dockerfile
|
||||
// at runtime* — an arbitrary Dockerfile is build-time RCE whose victim is the
|
||||
// cluster, not the uploader. So the build Pod runs with a deliberately weak
|
||||
// service account in an isolated namespace that can only push to the registry
|
||||
// and cannot touch the minecraft namespace, the felis database, or the K8s API
|
||||
// (spec §21). Those isolation guarantees live in the Job/NetworkPolicy specs
|
||||
// (jobspec.go) and are asserted by unit tests, since no cluster runs here.
|
||||
//
|
||||
// The Trivy gate is enforced as the build Pod's *exit code*: a kaniko
|
||||
// initContainer builds and pushes, then a trivy container scans the pushed ref
|
||||
// with `--exit-code 1 --severity CRITICAL`. Therefore "Job Succeeded" is
|
||||
// equivalent to "pushed AND no CRITICAL CVE". felis-api observes the Job phase
|
||||
// and performs the database writes — the build Pod itself never has database
|
||||
// credentials (the weak-SA red line). On success the image is admitted to
|
||||
// image_whitelist with enabled=true (recording added_by); on failure the build
|
||||
// is marked failed and nothing is admitted (spec §16: the only retained
|
||||
// automatic gate).
|
||||
//
|
||||
// The Builder depends on the Store and Jobs interfaces, so submission, the
|
||||
// scan-gate translation, cancellation, and image admission are all unit-tested
|
||||
// against in-memory fakes. The Postgres (pgStore) and controller-runtime
|
||||
// (k8sJobs) implementations compile here but are exercised only by integration
|
||||
// tests against a live database / cluster.
|
||||
package build
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Status mirrors the build_status enum (spec §6).
|
||||
type Status string
|
||||
|
||||
const (
|
||||
StatusPending Status = "pending"
|
||||
StatusBuilding Status = "building"
|
||||
StatusSucceeded Status = "succeeded"
|
||||
StatusFailed Status = "failed"
|
||||
StatusCancelled Status = "cancelled"
|
||||
)
|
||||
|
||||
// terminal reports whether a status is final and no longer reconciled.
|
||||
func (s Status) terminal() bool {
|
||||
switch s {
|
||||
case StatusSucceeded, StatusFailed, StatusCancelled:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// JobPhase is the build Pod's lifecycle as observed from the K8s Job, decoupled
|
||||
// from any K8s type so the scan-gate translation stays unit-testable.
|
||||
type JobPhase int
|
||||
|
||||
const (
|
||||
// JobUnknown means the Job was not found (e.g. GC'd); treated as failed.
|
||||
JobUnknown JobPhase = iota
|
||||
JobPending
|
||||
JobRunning
|
||||
// JobSucceeded means kaniko pushed AND trivy found no CRITICAL CVE — the
|
||||
// scan gate passed (spec §16).
|
||||
JobSucceeded
|
||||
// JobFailed means kaniko failed OR trivy found a CRITICAL CVE — the build
|
||||
// is rejected and nothing is admitted.
|
||||
JobFailed
|
||||
)
|
||||
|
||||
// image admission sources (spec §6 image_whitelist.source).
|
||||
const (
|
||||
SourceBuilt = "built"
|
||||
SourceExternal = "external"
|
||||
)
|
||||
|
||||
// ErrNotFound is returned when a build id / image ref does not exist.
|
||||
var ErrNotFound = errors.New("build: not found")
|
||||
|
||||
// ErrAlreadyTerminal is returned by Cancel when the build has already finished.
|
||||
var ErrAlreadyTerminal = errors.New("build: already in a terminal state")
|
||||
|
||||
// ErrInvalid wraps every request-validation failure (bad image ref, missing /
|
||||
// oversize Dockerfile, missing context). Callers map it to a 400; it is kept
|
||||
// distinct from store/cluster failures so those surface as 500.
|
||||
var ErrInvalid = errors.New("build: invalid request")
|
||||
|
||||
// Request is the validated POST /images/build input (spec §16). The dockerfile
|
||||
// and context are archived for audit; the target ref must address the internal
|
||||
// registry (enforced in Validate).
|
||||
type Request struct {
|
||||
// ImageRef is the push target, e.g. registry.felis.svc:5000/foo:1.0. It must
|
||||
// be under the configured internal registry — a build can never push
|
||||
// elsewhere.
|
||||
ImageRef string
|
||||
// Dockerfile is the uploaded build recipe (size-capped).
|
||||
Dockerfile string
|
||||
// ContextRef locates the uploaded tar.gz context in object storage / a PVC
|
||||
// (spec §17: Kaniko pulls it; Git context is intentionally not supported).
|
||||
ContextRef string
|
||||
// BaseImage is the resolved FROM, recorded for audit only — it is NOT a hard
|
||||
// gate (spec §16: base FROM is not hard-gated; the scan + egress lock cover
|
||||
// poisoned bases).
|
||||
BaseImage string
|
||||
// RequestedBy is the admin Access email, used as the audit actor and the
|
||||
// added_by of any admitted image.
|
||||
RequestedBy string
|
||||
}
|
||||
|
||||
// Build mirrors an image_builds row (spec §6).
|
||||
type Build struct {
|
||||
ID string `json:"id"`
|
||||
ImageRef string `json:"image_ref"`
|
||||
Status Status `json:"status"`
|
||||
Dockerfile string `json:"dockerfile,omitempty"`
|
||||
ContextRef string `json:"context_ref,omitempty"`
|
||||
BaseImage string `json:"base_image,omitempty"`
|
||||
RequestedBy string `json:"requested_by"`
|
||||
JobName string `json:"job_name,omitempty"`
|
||||
LogRef string `json:"log_ref,omitempty"`
|
||||
Error string `json:"error,omitempty"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
FinishedAt *time.Time `json:"finished_at,omitempty"`
|
||||
}
|
||||
|
||||
// Image mirrors an image_whitelist row (spec §6): the dynamic, auditable image
|
||||
// admission list that the create-server form reads from.
|
||||
type Image struct {
|
||||
ImageRef string `json:"image_ref"`
|
||||
Source string `json:"source"`
|
||||
BuildID string `json:"build_id,omitempty"`
|
||||
AddedBy string `json:"added_by"`
|
||||
Enabled bool `json:"enabled"`
|
||||
AddedAt time.Time `json:"added_at"`
|
||||
}
|
||||
|
||||
// Store is the business-layer persistence the Builder depends on (image_builds
|
||||
// + image_whitelist). It is an interface so the Builder is tested against an
|
||||
// in-memory fake; the Postgres implementation (pgStore) is integration-tested
|
||||
// only.
|
||||
type Store interface {
|
||||
// CreateBuild inserts a new image_builds row (status pending).
|
||||
CreateBuild(ctx context.Context, b *Build) error
|
||||
// GetBuild loads one build, or ErrNotFound.
|
||||
GetBuild(ctx context.Context, id string) (*Build, error)
|
||||
// SetBuildJob records the Job name and advances status to building.
|
||||
SetBuildJob(ctx context.Context, id, jobName string) error
|
||||
// FinishBuild sets a terminal status, an optional error, and finished_at.
|
||||
FinishBuild(ctx context.Context, id string, status Status, errMsg string, at time.Time) error
|
||||
// ListUnfinishedBuilds returns builds still being reconciled (status pending
|
||||
// or building), oldest first — the work list for SyncAll.
|
||||
ListUnfinishedBuilds(ctx context.Context) ([]Build, error)
|
||||
// AdmitBuiltImage upserts an image_whitelist row with enabled=true and
|
||||
// source=built (the scan-gate success path, spec §16). It records added_by.
|
||||
AdmitBuiltImage(ctx context.Context, img Image) error
|
||||
// ListImages returns the image whitelist.
|
||||
ListImages(ctx context.Context) ([]Image, error)
|
||||
// AddExternalImage upserts an externally-pushed image (spec §15 external
|
||||
// admission; source=external, no build_id).
|
||||
AddExternalImage(ctx context.Context, img Image) error
|
||||
// RemoveImage deletes an image_whitelist row, or ErrNotFound.
|
||||
RemoveImage(ctx context.Context, imageRef string) error
|
||||
}
|
||||
|
||||
// Jobs is the cluster-side build lifecycle the Builder depends on. It is an
|
||||
// interface so the scan-gate translation is tested against a fake; the
|
||||
// controller-runtime implementation (k8sJobs) is integration-tested only — it
|
||||
// requires a live cluster.
|
||||
type Jobs interface {
|
||||
// CreateBuildJob starts the Kaniko+Trivy Job for p in the felis-build
|
||||
// namespace and returns the Job name.
|
||||
CreateBuildJob(ctx context.Context, p JobParams) (jobName string, err error)
|
||||
// JobPhase reports the current phase of a previously-created Job.
|
||||
JobPhase(ctx context.Context, jobName string) (JobPhase, error)
|
||||
// CancelBuildJob deletes the Job (and its pods), tolerating not-found.
|
||||
CancelBuildJob(ctx context.Context, jobName string) error
|
||||
}
|
||||
|
||||
// Config parameterises the build subsystem from felis.toml (spec §24 [registry]
|
||||
// + safety limits). It is validated by withDefaults before use.
|
||||
type Config struct {
|
||||
// Namespace is the isolated build namespace (spec §16: felis-build).
|
||||
Namespace string
|
||||
// ServiceAccount is the weak SA the build Pod runs as. It MUST NOT be the
|
||||
// felis-api SA (spec §16 red line).
|
||||
ServiceAccount string
|
||||
// RegistryURL is the internal registry the build pushes to and Trivy scans
|
||||
// (spec §17). Image refs are validated to be under it.
|
||||
RegistryURL string
|
||||
// KanikoImage / TrivyImage are the executor images.
|
||||
KanikoImage string
|
||||
TrivyImage string
|
||||
// Deadline caps a build's wall-clock (spec §16: activeDeadlineSeconds).
|
||||
Deadline time.Duration
|
||||
// MaxDockerfileBytes caps the uploaded Dockerfile (spec §16: context size
|
||||
// limits). Zero applies the default.
|
||||
MaxDockerfileBytes int
|
||||
// CPULimit / MemLimit cap each build container (spec §16: resource limits).
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
}
|
||||
|
||||
// Defaults applied when a Config field is left zero.
|
||||
const (
|
||||
defaultNamespace = "felis-build"
|
||||
defaultServiceAccount = "felis-build"
|
||||
defaultKanikoImage = "gcr.io/kaniko-project/executor:latest"
|
||||
defaultTrivyImage = "aquasec/trivy:latest"
|
||||
defaultDeadline = 30 * time.Minute
|
||||
defaultMaxDockerfile = 256 * 1024 // 256 KiB
|
||||
defaultCPULimit = "2"
|
||||
defaultMemLimit = "4Gi"
|
||||
)
|
||||
|
||||
// withDefaults returns a copy of c with zero fields filled, so a partially
|
||||
// configured Config (or the zero value, in tests) is always usable.
|
||||
func (c Config) withDefaults() Config {
|
||||
if c.Namespace == "" {
|
||||
c.Namespace = defaultNamespace
|
||||
}
|
||||
if c.ServiceAccount == "" {
|
||||
c.ServiceAccount = defaultServiceAccount
|
||||
}
|
||||
if c.KanikoImage == "" {
|
||||
c.KanikoImage = defaultKanikoImage
|
||||
}
|
||||
if c.TrivyImage == "" {
|
||||
c.TrivyImage = defaultTrivyImage
|
||||
}
|
||||
if c.Deadline <= 0 {
|
||||
c.Deadline = defaultDeadline
|
||||
}
|
||||
if c.MaxDockerfileBytes <= 0 {
|
||||
c.MaxDockerfileBytes = defaultMaxDockerfile
|
||||
}
|
||||
if c.CPULimit == "" {
|
||||
c.CPULimit = defaultCPULimit
|
||||
}
|
||||
if c.MemLimit == "" {
|
||||
c.MemLimit = defaultMemLimit
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// Builder orchestrates the build subsystem. It holds no mutable state; the
|
||||
// clock and id generator are injectable for hermetic tests.
|
||||
type Builder struct {
|
||||
Store Store
|
||||
Jobs Jobs
|
||||
Config Config
|
||||
|
||||
// Now is the clock, injectable for tests. Defaults to time.Now.
|
||||
Now func() time.Time
|
||||
// IDGen mints build ids. Defaults to a time-based generator.
|
||||
IDGen func() string
|
||||
}
|
||||
|
||||
func (b *Builder) now() time.Time {
|
||||
if b.Now != nil {
|
||||
return b.Now()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
func (b *Builder) newID() string {
|
||||
if b.IDGen != nil {
|
||||
return b.IDGen()
|
||||
}
|
||||
return fmt.Sprintf("bld-%d", time.Now().UnixNano())
|
||||
}
|
||||
|
||||
// Submit validates req, records a pending build, and starts the Kaniko+Trivy
|
||||
// Job (spec §16). The build is returned in the building state once the Job is
|
||||
// created; if Job creation fails the build is marked failed so it never lingers
|
||||
// pending. The caller (felis-api) drives the build to a terminal state by
|
||||
// polling Sync / SyncAll.
|
||||
func (b *Builder) Submit(ctx context.Context, req Request) (*Build, error) {
|
||||
cfg := b.Config.withDefaults()
|
||||
if err := Validate(req, cfg); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
now := b.now()
|
||||
bld := &Build{
|
||||
ID: b.newID(),
|
||||
ImageRef: req.ImageRef,
|
||||
Status: StatusPending,
|
||||
Dockerfile: req.Dockerfile,
|
||||
ContextRef: req.ContextRef,
|
||||
BaseImage: req.BaseImage,
|
||||
RequestedBy: req.RequestedBy,
|
||||
CreatedAt: now,
|
||||
}
|
||||
if err := b.Store.CreateBuild(ctx, bld); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
jobName, err := b.Jobs.CreateBuildJob(ctx, b.jobParams(bld, cfg))
|
||||
if err != nil {
|
||||
// The pending row exists; mark it failed so it is not reconciled forever.
|
||||
_ = b.Store.FinishBuild(ctx, bld.ID, StatusFailed, "job creation failed: "+err.Error(), b.now())
|
||||
bld.Status = StatusFailed
|
||||
bld.Error = "job creation failed: " + err.Error()
|
||||
return bld, fmt.Errorf("build: create job: %w", err)
|
||||
}
|
||||
|
||||
if err := b.Store.SetBuildJob(ctx, bld.ID, jobName); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
bld.JobName = jobName
|
||||
bld.Status = StatusBuilding
|
||||
return bld, nil
|
||||
}
|
||||
|
||||
// jobParams projects a build + config onto the inputs jobspec.go renders.
|
||||
func (b *Builder) jobParams(bld *Build, cfg Config) JobParams {
|
||||
return JobParams{
|
||||
BuildID: bld.ID,
|
||||
ImageRef: bld.ImageRef,
|
||||
ContextRef: bld.ContextRef,
|
||||
Namespace: cfg.Namespace,
|
||||
ServiceAccount: cfg.ServiceAccount,
|
||||
RegistryURL: cfg.RegistryURL,
|
||||
KanikoImage: cfg.KanikoImage,
|
||||
TrivyImage: cfg.TrivyImage,
|
||||
Deadline: cfg.Deadline,
|
||||
CPULimit: cfg.CPULimit,
|
||||
MemLimit: cfg.MemLimit,
|
||||
}
|
||||
}
|
||||
|
||||
// Get returns a build by id, or ErrNotFound.
|
||||
func (b *Builder) Get(ctx context.Context, id string) (*Build, error) {
|
||||
return b.Store.GetBuild(ctx, id)
|
||||
}
|
||||
|
||||
// Sync reconciles one non-terminal build against its Job phase — the scan-gate
|
||||
// translation (spec §16). A terminal build is returned unchanged (idempotent).
|
||||
//
|
||||
// - JobSucceeded → status=succeeded AND the image is admitted to the whitelist
|
||||
// with enabled=true (kaniko pushed and trivy found no CRITICAL CVE).
|
||||
// - JobFailed / JobUnknown → status=failed, nothing admitted (a CRITICAL CVE
|
||||
// surfaces here as a failed Job, since trivy runs with --exit-code 1).
|
||||
// - JobPending / JobRunning → no change.
|
||||
//
|
||||
// The image admission is performed by felis-api (this code path), never by the
|
||||
// build Pod, which holds no database credentials.
|
||||
func (b *Builder) Sync(ctx context.Context, id string) (*Build, error) {
|
||||
bld, err := b.Store.GetBuild(ctx, id)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if bld.Status.terminal() {
|
||||
return bld, nil
|
||||
}
|
||||
if bld.JobName == "" {
|
||||
// Created but the Job name was never recorded; treat as failed rather
|
||||
// than reconcile forever against a phantom Job.
|
||||
return b.finish(ctx, bld, StatusFailed, "no build job recorded")
|
||||
}
|
||||
|
||||
phase, err := b.Jobs.JobPhase(ctx, bld.JobName)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
switch phase {
|
||||
case JobSucceeded:
|
||||
now := b.now()
|
||||
// Admit the image first; only then mark the build succeeded, so a
|
||||
// succeeded build always has its whitelist row (no admitted-but-not-
|
||||
// recorded window if the second write fails).
|
||||
if err := b.Store.AdmitBuiltImage(ctx, Image{
|
||||
ImageRef: bld.ImageRef,
|
||||
Source: SourceBuilt,
|
||||
BuildID: bld.ID,
|
||||
AddedBy: bld.RequestedBy,
|
||||
Enabled: true,
|
||||
AddedAt: now,
|
||||
}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return b.finishAt(ctx, bld, StatusSucceeded, "", now)
|
||||
case JobFailed, JobUnknown:
|
||||
return b.finish(ctx, bld, StatusFailed, "build job failed or scan found a CRITICAL CVE")
|
||||
default: // JobPending / JobRunning
|
||||
return bld, nil
|
||||
}
|
||||
}
|
||||
|
||||
// SyncAll reconciles every unfinished build and returns the count advanced to a
|
||||
// terminal state. felis-api calls this periodically (spec §16: the scan gate is
|
||||
// observed, not pushed by the build Pod).
|
||||
func (b *Builder) SyncAll(ctx context.Context) (int, error) {
|
||||
builds, err := b.Store.ListUnfinishedBuilds(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
advanced := 0
|
||||
for i := range builds {
|
||||
bld, err := b.Sync(ctx, builds[i].ID)
|
||||
if err != nil {
|
||||
return advanced, err
|
||||
}
|
||||
if bld.Status.terminal() {
|
||||
advanced++
|
||||
}
|
||||
}
|
||||
return advanced, nil
|
||||
}
|
||||
|
||||
// Cancel stops an in-flight build: delete its Job and mark it cancelled. A
|
||||
// build that has already finished returns ErrAlreadyTerminal.
|
||||
func (b *Builder) Cancel(ctx context.Context, id string) (*Build, error) {
|
||||
bld, err := b.Store.GetBuild(ctx, id)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if bld.Status.terminal() {
|
||||
return nil, ErrAlreadyTerminal
|
||||
}
|
||||
if bld.JobName != "" {
|
||||
if err := b.Jobs.CancelBuildJob(ctx, bld.JobName); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return b.finish(ctx, bld, StatusCancelled, "cancelled by administrator")
|
||||
}
|
||||
|
||||
// finish marks a build terminal at the current clock and returns the updated
|
||||
// view without a second round-trip.
|
||||
func (b *Builder) finish(ctx context.Context, bld *Build, status Status, msg string) (*Build, error) {
|
||||
return b.finishAt(ctx, bld, status, msg, b.now())
|
||||
}
|
||||
|
||||
func (b *Builder) finishAt(ctx context.Context, bld *Build, status Status, msg string, at time.Time) (*Build, error) {
|
||||
if err := b.Store.FinishBuild(ctx, bld.ID, status, msg, at); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
bld.Status = status
|
||||
bld.Error = msg
|
||||
finished := at
|
||||
bld.FinishedAt = &finished
|
||||
return bld, nil
|
||||
}
|
||||
|
||||
// ListImages returns the image whitelist (spec §15 create-server form source).
|
||||
func (b *Builder) ListImages(ctx context.Context) ([]Image, error) {
|
||||
return b.Store.ListImages(ctx)
|
||||
}
|
||||
|
||||
// AddExternalImage admits an externally-pushed image (spec §15). It is enabled
|
||||
// immediately; external images bypass the build pipeline but are still recorded
|
||||
// with added_by for audit.
|
||||
func (b *Builder) AddExternalImage(ctx context.Context, imageRef, addedBy string) (*Image, error) {
|
||||
if err := ValidateImageRef(imageRef); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
img := Image{
|
||||
ImageRef: imageRef,
|
||||
Source: SourceExternal,
|
||||
AddedBy: addedBy,
|
||||
Enabled: true,
|
||||
AddedAt: b.now(),
|
||||
}
|
||||
if err := b.Store.AddExternalImage(ctx, img); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &img, nil
|
||||
}
|
||||
|
||||
// RemoveImage withdraws an image from the whitelist (spec §22: dynamic,
|
||||
// auditable). It does not delete the underlying registry blob.
|
||||
func (b *Builder) RemoveImage(ctx context.Context, imageRef string) error {
|
||||
return b.Store.RemoveImage(ctx, imageRef)
|
||||
}
|
||||
|
||||
// ImageAdmitted reports whether a concrete image reference is on the whitelist
|
||||
// and enabled (spec §15: the create-server form may only choose an admitted
|
||||
// image). A disabled row never admits. A wildcard whitelist entry
|
||||
// ("registry/foo:*") admits any concrete tag on that repo (imageMatches); the
|
||||
// caller always passes a concrete ref, never a wildcard. An empty ref is never
|
||||
// admitted.
|
||||
func (b *Builder) ImageAdmitted(ctx context.Context, imageRef string) (bool, error) {
|
||||
if strings.TrimSpace(imageRef) == "" {
|
||||
return false, nil
|
||||
}
|
||||
images, err := b.Store.ListImages(ctx)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
for _, img := range images {
|
||||
if img.Enabled && imageMatches(imageRef, img.ImageRef) {
|
||||
return true, nil
|
||||
}
|
||||
}
|
||||
return false, nil
|
||||
}
|
||||
@@ -0,0 +1,478 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
var testNow = time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
// ---- in-memory fakes ----
|
||||
|
||||
type fakeStore struct {
|
||||
builds map[string]*Build
|
||||
images map[string]Image
|
||||
// call recorders
|
||||
admitted []Image
|
||||
finished []string // "id:status"
|
||||
removeErr error
|
||||
createErr error
|
||||
}
|
||||
|
||||
func newFakeStore() *fakeStore {
|
||||
return &fakeStore{builds: map[string]*Build{}, images: map[string]Image{}}
|
||||
}
|
||||
|
||||
func (f *fakeStore) CreateBuild(_ context.Context, b *Build) error {
|
||||
if f.createErr != nil {
|
||||
return f.createErr
|
||||
}
|
||||
cp := *b
|
||||
f.builds[b.ID] = &cp
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) GetBuild(_ context.Context, id string) (*Build, error) {
|
||||
b, ok := f.builds[id]
|
||||
if !ok {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
cp := *b
|
||||
return &cp, nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) SetBuildJob(_ context.Context, id, jobName string) error {
|
||||
b, ok := f.builds[id]
|
||||
if !ok {
|
||||
return ErrNotFound
|
||||
}
|
||||
b.JobName = jobName
|
||||
b.Status = StatusBuilding
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) FinishBuild(_ context.Context, id string, status Status, errMsg string, at time.Time) error {
|
||||
b, ok := f.builds[id]
|
||||
if !ok {
|
||||
return ErrNotFound
|
||||
}
|
||||
b.Status = status
|
||||
b.Error = errMsg
|
||||
finished := at
|
||||
b.FinishedAt = &finished
|
||||
f.finished = append(f.finished, id+":"+string(status))
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) ListUnfinishedBuilds(_ context.Context) ([]Build, error) {
|
||||
var out []Build
|
||||
for _, b := range f.builds {
|
||||
if !b.Status.terminal() {
|
||||
out = append(out, *b)
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) AdmitBuiltImage(_ context.Context, img Image) error {
|
||||
f.images[img.ImageRef] = img
|
||||
f.admitted = append(f.admitted, img)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) ListImages(_ context.Context) ([]Image, error) {
|
||||
var out []Image
|
||||
for _, img := range f.images {
|
||||
out = append(out, img)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) AddExternalImage(_ context.Context, img Image) error {
|
||||
f.images[img.ImageRef] = img
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeStore) RemoveImage(_ context.Context, ref string) error {
|
||||
if f.removeErr != nil {
|
||||
return f.removeErr
|
||||
}
|
||||
if _, ok := f.images[ref]; !ok {
|
||||
return ErrNotFound
|
||||
}
|
||||
delete(f.images, ref)
|
||||
return nil
|
||||
}
|
||||
|
||||
type fakeJobs struct {
|
||||
phase JobPhase
|
||||
phaseErr error
|
||||
createErr error
|
||||
|
||||
created []JobParams
|
||||
cancelled []string
|
||||
}
|
||||
|
||||
func (f *fakeJobs) CreateBuildJob(_ context.Context, p JobParams) (string, error) {
|
||||
if f.createErr != nil {
|
||||
return "", f.createErr
|
||||
}
|
||||
f.created = append(f.created, p)
|
||||
return BuildJobName(p.BuildID), nil
|
||||
}
|
||||
|
||||
func (f *fakeJobs) JobPhase(_ context.Context, _ string) (JobPhase, error) {
|
||||
return f.phase, f.phaseErr
|
||||
}
|
||||
|
||||
func (f *fakeJobs) CancelBuildJob(_ context.Context, jobName string) error {
|
||||
f.cancelled = append(f.cancelled, jobName)
|
||||
return nil
|
||||
}
|
||||
|
||||
// newBuilder wires a Builder over fresh fakes with a frozen clock and a
|
||||
// deterministic id generator.
|
||||
func newBuilder() (*Builder, *fakeStore, *fakeJobs) {
|
||||
st := newFakeStore()
|
||||
jb := &fakeJobs{phase: JobPending}
|
||||
n := 0
|
||||
b := &Builder{
|
||||
Store: st,
|
||||
Jobs: jb,
|
||||
Config: Config{
|
||||
RegistryURL: "registry.felis.svc:5000",
|
||||
},
|
||||
Now: func() time.Time { return testNow },
|
||||
IDGen: func() string {
|
||||
n++
|
||||
return "bld-" + string(rune('0'+n))
|
||||
},
|
||||
}
|
||||
return b, st, jb
|
||||
}
|
||||
|
||||
func goodRequest() Request {
|
||||
return Request{
|
||||
ImageRef: "registry.felis.svc:5000/mc-paper:1.0",
|
||||
Dockerfile: "FROM eclipse-temurin:21\nCOPY . /data\n",
|
||||
ContextRef: "tar://contexts/abc.tar.gz",
|
||||
RequestedBy: "[email protected]",
|
||||
}
|
||||
}
|
||||
|
||||
// ---- submission ----
|
||||
|
||||
func TestSubmitCreatesPendingBuildAndStartsJob(t *testing.T) {
|
||||
b, st, jb := newBuilder()
|
||||
bld, err := b.Submit(context.Background(), goodRequest())
|
||||
if err != nil {
|
||||
t.Fatalf("Submit: %v", err)
|
||||
}
|
||||
if bld.Status != StatusBuilding {
|
||||
t.Errorf("status = %q, want building", bld.Status)
|
||||
}
|
||||
if bld.JobName == "" {
|
||||
t.Error("job name not recorded")
|
||||
}
|
||||
if got := st.builds[bld.ID]; got == nil || got.Dockerfile == "" {
|
||||
t.Error("build row not persisted with dockerfile")
|
||||
}
|
||||
if len(jb.created) != 1 {
|
||||
t.Fatalf("CreateBuildJob calls = %d, want 1", len(jb.created))
|
||||
}
|
||||
// The Job must be parameterised with the weak build SA and the namespace,
|
||||
// never the api identity — this is the §16 red line, asserted at the seam.
|
||||
p := jb.created[0]
|
||||
if p.ServiceAccount != defaultServiceAccount {
|
||||
t.Errorf("job SA = %q, want %q", p.ServiceAccount, defaultServiceAccount)
|
||||
}
|
||||
if p.Namespace != defaultNamespace {
|
||||
t.Errorf("job namespace = %q, want %q", p.Namespace, defaultNamespace)
|
||||
}
|
||||
if p.RegistryURL != "registry.felis.svc:5000" {
|
||||
t.Errorf("job registry = %q", p.RegistryURL)
|
||||
}
|
||||
if p.Deadline <= 0 {
|
||||
t.Error("job deadline not set")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubmitRejectsExternalRegistryTarget(t *testing.T) {
|
||||
b, st, jb := newBuilder()
|
||||
req := goodRequest()
|
||||
req.ImageRef = "docker.io/library/evil:latest"
|
||||
if _, err := b.Submit(context.Background(), req); err == nil {
|
||||
t.Fatal("expected rejection of non-internal registry target")
|
||||
}
|
||||
if len(st.builds) != 0 {
|
||||
t.Error("a rejected build must not be persisted")
|
||||
}
|
||||
if len(jb.created) != 0 {
|
||||
t.Error("a rejected build must not start a job")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubmitRejectsEmptyAndOversizeDockerfile(t *testing.T) {
|
||||
b, _, _ := newBuilder()
|
||||
req := goodRequest()
|
||||
req.Dockerfile = " "
|
||||
if _, err := b.Submit(context.Background(), req); err == nil {
|
||||
t.Error("expected rejection of empty dockerfile")
|
||||
}
|
||||
|
||||
b2, _, _ := newBuilder()
|
||||
b2.Config.MaxDockerfileBytes = 16
|
||||
req2 := goodRequest()
|
||||
req2.Dockerfile = "FROM scratch\n# padding padding padding padding\n"
|
||||
if _, err := b2.Submit(context.Background(), req2); err == nil {
|
||||
t.Error("expected rejection of oversize dockerfile")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubmitRequiresContext(t *testing.T) {
|
||||
b, _, _ := newBuilder()
|
||||
req := goodRequest()
|
||||
req.ContextRef = ""
|
||||
if _, err := b.Submit(context.Background(), req); err == nil {
|
||||
t.Error("expected rejection when context reference is missing")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubmitMarksBuildFailedWhenJobCreationFails(t *testing.T) {
|
||||
b, st, jb := newBuilder()
|
||||
jb.createErr = errors.New("apiserver down")
|
||||
bld, err := b.Submit(context.Background(), goodRequest())
|
||||
if err == nil {
|
||||
t.Fatal("expected error when job creation fails")
|
||||
}
|
||||
if bld == nil || bld.Status != StatusFailed {
|
||||
t.Fatalf("build should be marked failed, got %+v", bld)
|
||||
}
|
||||
// The persisted row must not be left pending forever.
|
||||
if got := st.builds[bld.ID]; got == nil || got.Status != StatusFailed {
|
||||
t.Errorf("persisted status = %v, want failed", got)
|
||||
}
|
||||
}
|
||||
|
||||
// ---- the scan gate (Job phase -> DB) ----
|
||||
|
||||
func TestSyncSucceededAdmitsImageWithAddedBy(t *testing.T) {
|
||||
b, st, jb := newBuilder()
|
||||
bld, _ := b.Submit(context.Background(), goodRequest())
|
||||
|
||||
jb.phase = JobSucceeded
|
||||
got, err := b.Sync(context.Background(), bld.ID)
|
||||
if err != nil {
|
||||
t.Fatalf("Sync: %v", err)
|
||||
}
|
||||
if got.Status != StatusSucceeded {
|
||||
t.Errorf("status = %q, want succeeded", got.Status)
|
||||
}
|
||||
img, ok := st.images[bld.ImageRef]
|
||||
if !ok {
|
||||
t.Fatal("succeeded build did not admit its image to the whitelist")
|
||||
}
|
||||
if !img.Enabled {
|
||||
t.Error("admitted image must be enabled=true")
|
||||
}
|
||||
if img.Source != SourceBuilt {
|
||||
t.Errorf("source = %q, want built", img.Source)
|
||||
}
|
||||
if img.AddedBy != "[email protected]" {
|
||||
t.Errorf("added_by = %q, want the requester", img.AddedBy)
|
||||
}
|
||||
if img.BuildID != bld.ID {
|
||||
t.Errorf("build_id = %q, want %q", img.BuildID, bld.ID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSyncFailedDoesNotAdmitImage(t *testing.T) {
|
||||
// A failed Job is exactly the CRITICAL-CVE case: trivy --exit-code 1 fails
|
||||
// the Pod, so the gate is enforced as the Job verdict (spec §16).
|
||||
b, st, jb := newBuilder()
|
||||
bld, _ := b.Submit(context.Background(), goodRequest())
|
||||
|
||||
jb.phase = JobFailed
|
||||
got, err := b.Sync(context.Background(), bld.ID)
|
||||
if err != nil {
|
||||
t.Fatalf("Sync: %v", err)
|
||||
}
|
||||
if got.Status != StatusFailed {
|
||||
t.Errorf("status = %q, want failed", got.Status)
|
||||
}
|
||||
if len(st.images) != 0 {
|
||||
t.Error("a failed/CRITICAL build must NOT admit any image")
|
||||
}
|
||||
if len(st.admitted) != 0 {
|
||||
t.Error("AdmitBuiltImage must not be called on failure")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSyncRunningIsNoOp(t *testing.T) {
|
||||
b, _, jb := newBuilder()
|
||||
bld, _ := b.Submit(context.Background(), goodRequest())
|
||||
jb.phase = JobRunning
|
||||
got, err := b.Sync(context.Background(), bld.ID)
|
||||
if err != nil {
|
||||
t.Fatalf("Sync: %v", err)
|
||||
}
|
||||
if got.Status != StatusBuilding {
|
||||
t.Errorf("status = %q, want building (unchanged)", got.Status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSyncIsIdempotentOnTerminalBuild(t *testing.T) {
|
||||
b, st, jb := newBuilder()
|
||||
bld, _ := b.Submit(context.Background(), goodRequest())
|
||||
jb.phase = JobSucceeded
|
||||
if _, err := b.Sync(context.Background(), bld.ID); err != nil {
|
||||
t.Fatalf("first Sync: %v", err)
|
||||
}
|
||||
// Flip the phase to a value that would admit again; a terminal build must
|
||||
// not be re-reconciled.
|
||||
before := len(st.admitted)
|
||||
if _, err := b.Sync(context.Background(), bld.ID); err != nil {
|
||||
t.Fatalf("second Sync: %v", err)
|
||||
}
|
||||
if len(st.admitted) != before {
|
||||
t.Error("terminal build was re-admitted on a second Sync")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSyncAllAdvancesUnfinishedBuilds(t *testing.T) {
|
||||
b, _, jb := newBuilder()
|
||||
if _, err := b.Submit(context.Background(), goodRequest()); err != nil {
|
||||
t.Fatalf("submit: %v", err)
|
||||
}
|
||||
jb.phase = JobSucceeded
|
||||
n, err := b.SyncAll(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("SyncAll: %v", err)
|
||||
}
|
||||
if n != 1 {
|
||||
t.Errorf("advanced = %d, want 1", n)
|
||||
}
|
||||
}
|
||||
|
||||
// ---- cancellation ----
|
||||
|
||||
func TestCancelDeletesJobAndMarksCancelled(t *testing.T) {
|
||||
b, _, jb := newBuilder()
|
||||
bld, _ := b.Submit(context.Background(), goodRequest())
|
||||
got, err := b.Cancel(context.Background(), bld.ID)
|
||||
if err != nil {
|
||||
t.Fatalf("Cancel: %v", err)
|
||||
}
|
||||
if got.Status != StatusCancelled {
|
||||
t.Errorf("status = %q, want cancelled", got.Status)
|
||||
}
|
||||
if len(jb.cancelled) != 1 {
|
||||
t.Errorf("CancelBuildJob calls = %d, want 1", len(jb.cancelled))
|
||||
}
|
||||
}
|
||||
|
||||
func TestCancelTerminalBuildFails(t *testing.T) {
|
||||
b, _, jb := newBuilder()
|
||||
bld, _ := b.Submit(context.Background(), goodRequest())
|
||||
jb.phase = JobSucceeded
|
||||
if _, err := b.Sync(context.Background(), bld.ID); err != nil {
|
||||
t.Fatalf("Sync: %v", err)
|
||||
}
|
||||
if _, err := b.Cancel(context.Background(), bld.ID); !errors.Is(err, ErrAlreadyTerminal) {
|
||||
t.Errorf("Cancel on terminal build err = %v, want ErrAlreadyTerminal", err)
|
||||
}
|
||||
}
|
||||
|
||||
// ---- external admission ----
|
||||
|
||||
func TestAddExternalImageIsEnabledAndRecorded(t *testing.T) {
|
||||
b, st, _ := newBuilder()
|
||||
img, err := b.AddExternalImage(context.Background(), "registry.felis.svc:5000/ext:1", "[email protected]")
|
||||
if err != nil {
|
||||
t.Fatalf("AddExternalImage: %v", err)
|
||||
}
|
||||
if img.Source != SourceExternal || !img.Enabled {
|
||||
t.Errorf("external image = %+v, want enabled external", img)
|
||||
}
|
||||
if _, ok := st.images["registry.felis.svc:5000/ext:1"]; !ok {
|
||||
t.Error("external image not persisted")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAddExternalImageRejectsMalformedRef(t *testing.T) {
|
||||
b, _, _ := newBuilder()
|
||||
if _, err := b.AddExternalImage(context.Background(), "not a ref!!", "[email protected]"); err == nil {
|
||||
t.Error("expected rejection of malformed image ref")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemoveImage(t *testing.T) {
|
||||
b, st, _ := newBuilder()
|
||||
st.images["registry.felis.svc:5000/x:1"] = Image{ImageRef: "registry.felis.svc:5000/x:1"}
|
||||
if err := b.RemoveImage(context.Background(), "registry.felis.svc:5000/x:1"); err != nil {
|
||||
t.Fatalf("RemoveImage: %v", err)
|
||||
}
|
||||
if _, ok := st.images["registry.felis.svc:5000/x:1"]; ok {
|
||||
t.Error("image not removed")
|
||||
}
|
||||
}
|
||||
|
||||
// ---- whitelist admission for the create-server form (spec §15) ----
|
||||
|
||||
func TestImageMatches(t *testing.T) {
|
||||
cases := []struct {
|
||||
ref, pattern string
|
||||
want bool
|
||||
}{
|
||||
// exact match
|
||||
{"registry.felis.svc:5000/mc:1.0", "registry.felis.svc:5000/mc:1.0", true},
|
||||
// exact mismatch on tag
|
||||
{"registry.felis.svc:5000/mc:1.0", "registry.felis.svc:5000/mc:2.0", false},
|
||||
// tag wildcard admits any concrete tag on the repo
|
||||
{"registry.felis.svc:5000/mc:1.0", "registry.felis.svc:5000/mc:*", true},
|
||||
{"registry.felis.svc:5000/mc:anything", "registry.felis.svc:5000/mc:*", true},
|
||||
// wildcard does not cross repos
|
||||
{"registry.felis.svc:5000/other:1", "registry.felis.svc:5000/mc:*", false},
|
||||
// a registry-port colon is not a tag separator: an untagged ref under a
|
||||
// ported host has no tag, so a wildcard (which needs a non-empty tag) misses
|
||||
{"registry.felis.svc:5000/mc", "registry.felis.svc:5000/mc:*", false},
|
||||
// a non-wildcard pattern never matches via the prefix branch
|
||||
{"registry.felis.svc:5000/mc:1", "registry.felis.svc:5000/mc", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := imageMatches(c.ref, c.pattern); got != c.want {
|
||||
t.Errorf("imageMatches(%q, %q) = %v, want %v", c.ref, c.pattern, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageAdmitted(t *testing.T) {
|
||||
b, st, _ := newBuilder()
|
||||
st.images["registry.felis.svc:5000/exact:1"] = Image{ImageRef: "registry.felis.svc:5000/exact:1", Enabled: true}
|
||||
st.images["registry.felis.svc:5000/wild:*"] = Image{ImageRef: "registry.felis.svc:5000/wild:*", Enabled: true}
|
||||
st.images["registry.felis.svc:5000/off:1"] = Image{ImageRef: "registry.felis.svc:5000/off:1", Enabled: false}
|
||||
|
||||
cases := []struct {
|
||||
ref string
|
||||
want bool
|
||||
}{
|
||||
{"registry.felis.svc:5000/exact:1", true}, // exact, enabled
|
||||
{"registry.felis.svc:5000/exact:2", false}, // wrong tag
|
||||
{"registry.felis.svc:5000/wild:99", true}, // wildcard, enabled, port-colon safe
|
||||
{"registry.felis.svc:5000/off:1", false}, // present but disabled
|
||||
{"registry.felis.svc:5000/unknown:1", false}, // not on the list
|
||||
{"", false}, // empty ref
|
||||
{" ", false}, // blank ref
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := b.ImageAdmitted(context.Background(), c.ref)
|
||||
if err != nil {
|
||||
t.Fatalf("ImageAdmitted(%q): %v", c.ref, err)
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("ImageAdmitted(%q) = %v, want %v", c.ref, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,284 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/util/intstr"
|
||||
)
|
||||
|
||||
// Label keys applied to build objects. ManagedBy doubles as the NetworkPolicy
|
||||
// pod selector, so every build Pod is captured by the egress lock.
|
||||
const (
|
||||
LabelManagedBy = "app.kubernetes.io/managed-by"
|
||||
LabelComponent = "app.kubernetes.io/component"
|
||||
LabelBuildID = "felis.lolicon.best/build-id"
|
||||
|
||||
managedByValue = "felis-build"
|
||||
componentValue = "image-build"
|
||||
)
|
||||
|
||||
// Container names within the build Pod. Kaniko is the initContainer that builds
|
||||
// and pushes the image — its log IS the "build log" an admin watches (spec §16);
|
||||
// Trivy is the main container whose CRITICAL-CVE verdict gates admission and is
|
||||
// surfaced via the build status, not the log stream. Exported so the build-log
|
||||
// streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) follows the
|
||||
// same container this Job defines — one source of truth for the name.
|
||||
const (
|
||||
ContainerKaniko = "kaniko"
|
||||
ContainerTrivy = "trivy"
|
||||
)
|
||||
|
||||
// JobParams are the rendered inputs to a build Job. They are derived from a
|
||||
// Build + Config by the Builder; jobspec is a pure function of them so the
|
||||
// security-critical Job shape is unit-tested without a cluster.
|
||||
type JobParams struct {
|
||||
BuildID string
|
||||
ImageRef string
|
||||
ContextRef string
|
||||
Namespace string
|
||||
ServiceAccount string
|
||||
RegistryURL string
|
||||
KanikoImage string
|
||||
TrivyImage string
|
||||
Deadline time.Duration
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
}
|
||||
|
||||
// BuildJobName is the deterministic Job name for a build id.
|
||||
func BuildJobName(buildID string) string { return "build-" + buildID }
|
||||
|
||||
func buildLabels(p JobParams) map[string]string {
|
||||
return map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
LabelBuildID: p.BuildID,
|
||||
}
|
||||
}
|
||||
|
||||
// BuildJob renders the Kaniko+Trivy build Job (spec §16). Every isolation
|
||||
// guarantee the spec demands is encoded here and asserted by jobspec_test.go,
|
||||
// because no cluster runs in this environment:
|
||||
//
|
||||
// - runs in the isolated felis-build namespace with the weak felis-build SA
|
||||
// (never the felis-api SA) and does NOT mount the SA token, so it cannot
|
||||
// reach the K8s API (spec §16, §21);
|
||||
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
|
||||
// so docker-in-docker / privileged is never needed (spec §16, §22);
|
||||
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
|
||||
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
|
||||
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
|
||||
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
|
||||
// automatic admission gate (spec §16).
|
||||
//
|
||||
// Sequencing: kaniko runs as an initContainer (build + push to the internal
|
||||
// registry) and trivy as the main container (scan the pushed ref). The Pod
|
||||
// succeeds only if kaniko pushed AND trivy found no CRITICAL CVE.
|
||||
func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
deadline := int64(p.Deadline / time.Second)
|
||||
if deadline <= 0 {
|
||||
deadline = int64(defaultDeadline / time.Second)
|
||||
}
|
||||
|
||||
// Hardened container security context shared by both build containers: no
|
||||
// privilege, no privilege escalation, drop all capabilities. Kaniko needs a
|
||||
// writable root filesystem to unpack layers, so we do not force read-only
|
||||
// root here, but it gains no privilege.
|
||||
sec := &corev1.SecurityContext{
|
||||
Privileged: boolPtr(false),
|
||||
AllowPrivilegeEscalation: boolPtr(false),
|
||||
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
||||
}
|
||||
|
||||
kaniko := corev1.Container{
|
||||
Name: ContainerKaniko,
|
||||
Image: p.KanikoImage,
|
||||
Args: []string{
|
||||
"--dockerfile=Dockerfile",
|
||||
"--context=" + p.ContextRef,
|
||||
"--destination=" + p.ImageRef,
|
||||
// The internal registry is in-cluster only and may serve plain HTTP;
|
||||
// it is never a public ingress (spec §17).
|
||||
"--insecure",
|
||||
"--skip-tls-verify",
|
||||
},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||
SecurityContext: sec,
|
||||
}
|
||||
|
||||
trivy := corev1.Container{
|
||||
Name: ContainerTrivy,
|
||||
Image: p.TrivyImage,
|
||||
Args: []string{
|
||||
"image",
|
||||
"--exit-code", "1",
|
||||
"--severity", "CRITICAL",
|
||||
"--no-progress",
|
||||
"--insecure",
|
||||
p.ImageRef,
|
||||
},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||
SecurityContext: sec,
|
||||
}
|
||||
|
||||
job := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: BuildJobName(p.BuildID),
|
||||
Namespace: p.Namespace,
|
||||
Labels: buildLabels(p),
|
||||
},
|
||||
Spec: batchv1.JobSpec{
|
||||
// A poisoned build must not loop — one shot, then a terminal verdict.
|
||||
BackoffLimit: int32Ptr(0),
|
||||
ActiveDeadlineSeconds: int64Ptr(deadline),
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: p.ServiceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
InitContainers: []corev1.Container{kaniko},
|
||||
Containers: []corev1.Container{trivy},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
return job, nil
|
||||
}
|
||||
|
||||
// NetPolParams parameterises the build-namespace egress lock.
|
||||
type NetPolParams struct {
|
||||
Namespace string
|
||||
RegistryNamespace string
|
||||
RegistryPort int32
|
||||
// PackageSourceCIDRs is an optional, explicit allowlist of external package
|
||||
// mirrors (spec §16: egress 仅 registry + 包源). Empty means the most
|
||||
// locked-down default — no internet egress at all (默认拒外网).
|
||||
PackageSourceCIDRs []string
|
||||
}
|
||||
|
||||
// BuildNetworkPolicy renders the default-deny egress policy for build Pods
|
||||
// (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects
|
||||
// build Pods by the managed-by label, denies all ingress, and allows egress
|
||||
// only to DNS, the internal registry, and any explicitly configured package
|
||||
// mirrors. There is deliberately no allow-all egress rule.
|
||||
func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
|
||||
port := p.RegistryPort
|
||||
if port == 0 {
|
||||
port = 5000
|
||||
}
|
||||
dnsUDP := corev1.ProtocolUDP
|
||||
dnsTCP := corev1.ProtocolTCP
|
||||
dns53 := intstr.FromInt32(53)
|
||||
regPort := intstr.FromInt32(port)
|
||||
|
||||
egress := []networkingv1.NetworkPolicyEgressRule{
|
||||
// DNS resolution: port-restricted to 53, so this is not an open-internet
|
||||
// hole — name resolution only.
|
||||
{
|
||||
Ports: []networkingv1.NetworkPolicyPort{
|
||||
{Protocol: &dnsUDP, Port: &dns53},
|
||||
{Protocol: &dnsTCP, Port: &dns53},
|
||||
},
|
||||
},
|
||||
// The internal registry, selected by the namespace's immutable
|
||||
// kubernetes.io/metadata.name label, on the registry port only.
|
||||
{
|
||||
To: []networkingv1.NetworkPolicyPeer{{
|
||||
NamespaceSelector: &metav1.LabelSelector{
|
||||
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.RegistryNamespace},
|
||||
},
|
||||
}},
|
||||
Ports: []networkingv1.NetworkPolicyPort{
|
||||
{Protocol: &dnsTCP, Port: ®Port},
|
||||
},
|
||||
},
|
||||
}
|
||||
// Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet.
|
||||
for _, cidr := range p.PackageSourceCIDRs {
|
||||
egress = append(egress, networkingv1.NetworkPolicyEgressRule{
|
||||
To: []networkingv1.NetworkPolicyPeer{{
|
||||
IPBlock: &networkingv1.IPBlock{CIDR: cidr},
|
||||
}},
|
||||
})
|
||||
}
|
||||
|
||||
return &networkingv1.NetworkPolicy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "felis-build-egress",
|
||||
Namespace: p.Namespace,
|
||||
Labels: map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
},
|
||||
},
|
||||
Spec: networkingv1.NetworkPolicySpec{
|
||||
PodSelector: metav1.LabelSelector{
|
||||
MatchLabels: map[string]string{LabelManagedBy: managedByValue},
|
||||
},
|
||||
PolicyTypes: []networkingv1.PolicyType{
|
||||
networkingv1.PolicyTypeIngress,
|
||||
networkingv1.PolicyTypeEgress,
|
||||
},
|
||||
// Empty Ingress slice = deny all ingress: nothing connects to a
|
||||
// build Pod.
|
||||
Ingress: []networkingv1.NetworkPolicyIngressRule{},
|
||||
Egress: egress,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// BuildServiceAccount renders the weak build SA (spec §16, §21). It is the most
|
||||
// dangerous identity in the platform if mis-scoped, so it is created bare: no
|
||||
// secrets, token auto-mounting disabled, and — by virtue of having no Role or
|
||||
// RoleBinding anywhere — zero K8s API permissions. Its only capability is
|
||||
// network reachability to push to the registry, which RBAC does not grant.
|
||||
func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
|
||||
return &corev1.ServiceAccount{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: namespace,
|
||||
Labels: map[string]string{
|
||||
LabelManagedBy: managedByValue,
|
||||
LabelComponent: componentValue,
|
||||
},
|
||||
},
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
}
|
||||
}
|
||||
|
||||
// resourceLimits parses the CPU/memory limits into a ResourceList.
|
||||
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||
if cpu == "" {
|
||||
cpu = defaultCPULimit
|
||||
}
|
||||
if mem == "" {
|
||||
mem = defaultMemLimit
|
||||
}
|
||||
cpuQty, err := resource.ParseQuantity(cpu)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("build: invalid cpu limit %q: %w", cpu, err)
|
||||
}
|
||||
memQty, err := resource.ParseQuantity(mem)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("build: invalid memory limit %q: %w", mem, err)
|
||||
}
|
||||
return corev1.ResourceList{
|
||||
corev1.ResourceCPU: cpuQty,
|
||||
corev1.ResourceMemory: memQty,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
func int32Ptr(i int32) *int32 { return &i }
|
||||
func int64Ptr(i int64) *int64 { return &i }
|
||||
@@ -0,0 +1,306 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
)
|
||||
|
||||
func sampleJobParams() JobParams {
|
||||
return JobParams{
|
||||
BuildID: "bld-1",
|
||||
ImageRef: "registry.felis.svc:5000/mc-paper:1.0",
|
||||
ContextRef: "tar://contexts/abc.tar.gz",
|
||||
Namespace: defaultNamespace,
|
||||
ServiceAccount: defaultServiceAccount,
|
||||
RegistryURL: "registry.felis.svc:5000",
|
||||
KanikoImage: defaultKanikoImage,
|
||||
TrivyImage: defaultTrivyImage,
|
||||
Deadline: 30 * time.Minute,
|
||||
CPULimit: "2",
|
||||
MemLimit: "4Gi",
|
||||
}
|
||||
}
|
||||
|
||||
// The Job must run in the isolated build namespace under the weak build SA —
|
||||
// never the api namespace/identity, and never the minecraft namespace. This is
|
||||
// the central §16/§21 red line, asserted on the rendered spec because no cluster
|
||||
// runs here.
|
||||
func TestBuildJobRunsIsolatedUnderWeakSA(t *testing.T) {
|
||||
job, err := BuildJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
if job.Namespace != "felis-build" {
|
||||
t.Errorf("namespace = %q, want felis-build (never minecraft/felis-system)", job.Namespace)
|
||||
}
|
||||
sa := job.Spec.Template.Spec.ServiceAccountName
|
||||
if sa != "felis-build" {
|
||||
t.Errorf("service account = %q, want felis-build (never felis-api)", sa)
|
||||
}
|
||||
if sa == "felis-api" {
|
||||
t.Fatal("build Pod must NOT run as the felis-api SA")
|
||||
}
|
||||
// The SA token must not be mounted: with no token, the Pod cannot reach the
|
||||
// K8s API even if a Role were mis-bound.
|
||||
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
|
||||
t.Error("AutomountServiceAccountToken must be explicitly false")
|
||||
}
|
||||
}
|
||||
|
||||
// A poisoned build must terminate and not loop or run unbounded.
|
||||
func TestBuildJobIsBoundedAndOneShot(t *testing.T) {
|
||||
job, err := BuildJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
|
||||
t.Error("BackoffLimit must be 0 — a poisoned build must not retry")
|
||||
}
|
||||
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds <= 0 {
|
||||
t.Error("ActiveDeadlineSeconds must be a positive wall-clock cap")
|
||||
}
|
||||
if got := *job.Spec.ActiveDeadlineSeconds; got != int64((30 * time.Minute).Seconds()) {
|
||||
t.Errorf("ActiveDeadlineSeconds = %d, want 1800", got)
|
||||
}
|
||||
if job.Spec.Template.Spec.RestartPolicy != corev1.RestartPolicyNever {
|
||||
t.Error("RestartPolicy must be Never")
|
||||
}
|
||||
}
|
||||
|
||||
// No build container may be privileged or able to escalate, and all containers
|
||||
// must carry resource limits.
|
||||
func TestBuildJobContainersAreHardened(t *testing.T) {
|
||||
job, err := BuildJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
all := append([]corev1.Container{}, job.Spec.Template.Spec.InitContainers...)
|
||||
all = append(all, job.Spec.Template.Spec.Containers...)
|
||||
if len(all) < 2 {
|
||||
t.Fatalf("expected kaniko initContainer + trivy container, got %d containers", len(all))
|
||||
}
|
||||
for _, c := range all {
|
||||
sc := c.SecurityContext
|
||||
if sc == nil {
|
||||
t.Fatalf("container %q has no security context", c.Name)
|
||||
}
|
||||
if sc.Privileged == nil || *sc.Privileged {
|
||||
t.Errorf("container %q must not be privileged (Kaniko needs no daemon)", c.Name)
|
||||
}
|
||||
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
|
||||
t.Errorf("container %q must set allowPrivilegeEscalation=false", c.Name)
|
||||
}
|
||||
if sc.Capabilities == nil || len(sc.Capabilities.Drop) == 0 {
|
||||
t.Errorf("container %q must drop capabilities", c.Name)
|
||||
} else if string(sc.Capabilities.Drop[0]) != "ALL" {
|
||||
t.Errorf("container %q must drop ALL capabilities, got %v", c.Name, sc.Capabilities.Drop)
|
||||
}
|
||||
if c.Resources.Limits.Cpu().IsZero() || c.Resources.Limits.Memory().IsZero() {
|
||||
t.Errorf("container %q must carry CPU+memory limits", c.Name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// kaniko builds and pushes to the request's exact target; trivy gates admission
|
||||
// with --exit-code 1 --severity CRITICAL on that same ref.
|
||||
func TestBuildJobKanikoPushesAndTrivyGates(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
job, err := BuildJob(p)
|
||||
if err != nil {
|
||||
t.Fatalf("BuildJob: %v", err)
|
||||
}
|
||||
if len(job.Spec.Template.Spec.InitContainers) != 1 {
|
||||
t.Fatalf("expected exactly one (kaniko) initContainer")
|
||||
}
|
||||
kaniko := job.Spec.Template.Spec.InitContainers[0]
|
||||
if kaniko.Name != "kaniko" {
|
||||
t.Errorf("init container = %q, want kaniko", kaniko.Name)
|
||||
}
|
||||
if !hasArg(kaniko.Args, "--destination="+p.ImageRef) {
|
||||
t.Errorf("kaniko must push to %q, args=%v", p.ImageRef, kaniko.Args)
|
||||
}
|
||||
|
||||
if len(job.Spec.Template.Spec.Containers) != 1 {
|
||||
t.Fatalf("expected exactly one (trivy) main container")
|
||||
}
|
||||
trivy := job.Spec.Template.Spec.Containers[0]
|
||||
if trivy.Name != "trivy" {
|
||||
t.Errorf("main container = %q, want trivy", trivy.Name)
|
||||
}
|
||||
// The scan gate: a CRITICAL CVE must fail the Pod (and thus the Job).
|
||||
if !argPairPresent(trivy.Args, "--exit-code", "1") {
|
||||
t.Errorf("trivy must run with --exit-code 1, args=%v", trivy.Args)
|
||||
}
|
||||
if !argPairPresent(trivy.Args, "--severity", "CRITICAL") {
|
||||
t.Errorf("trivy must gate on --severity CRITICAL, args=%v", trivy.Args)
|
||||
}
|
||||
if !hasArg(trivy.Args, p.ImageRef) {
|
||||
t.Errorf("trivy must scan the pushed ref %q, args=%v", p.ImageRef, trivy.Args)
|
||||
}
|
||||
}
|
||||
|
||||
// The build namespace egress lock must be default-deny: deny all ingress, and
|
||||
// allow egress only to DNS + the internal registry — never an allow-all rule.
|
||||
func TestBuildNetworkPolicyIsDefaultDeny(t *testing.T) {
|
||||
np := BuildNetworkPolicy(NetPolParams{
|
||||
Namespace: "felis-build",
|
||||
RegistryNamespace: "felis-system",
|
||||
RegistryPort: 5000,
|
||||
})
|
||||
|
||||
if !hasPolicyType(np, networkingv1.PolicyTypeEgress) || !hasPolicyType(np, networkingv1.PolicyTypeIngress) {
|
||||
t.Fatal("policy must govern both Ingress and Egress")
|
||||
}
|
||||
// Ingress: empty rule slice = deny all.
|
||||
if len(np.Spec.Ingress) != 0 {
|
||||
t.Errorf("ingress must be empty (deny all), got %d rules", len(np.Spec.Ingress))
|
||||
}
|
||||
// The selector must capture every build Pod by the managed-by label.
|
||||
if np.Spec.PodSelector.MatchLabels[LabelManagedBy] != managedByValue {
|
||||
t.Errorf("pod selector must match managed-by=%s", managedByValue)
|
||||
}
|
||||
// No egress rule may be an allow-all (a rule with neither To peers nor Ports
|
||||
// would permit unrestricted egress).
|
||||
for i, rule := range np.Spec.Egress {
|
||||
if len(rule.To) == 0 && len(rule.Ports) == 0 {
|
||||
t.Errorf("egress rule %d is allow-all (no To, no Ports) — internet would be open", i)
|
||||
}
|
||||
}
|
||||
// The registry must be reachable (by namespace selector), and DNS allowed.
|
||||
if !egressAllowsNamespace(np, "felis-system") {
|
||||
t.Error("egress must allow the registry namespace")
|
||||
}
|
||||
if !egressAllowsPort(np, 53) {
|
||||
t.Error("egress must allow DNS (port 53)")
|
||||
}
|
||||
}
|
||||
|
||||
// With no package-source CIDRs configured, there must be zero IPBlock egress —
|
||||
// the most locked-down default (no open internet).
|
||||
func TestBuildNetworkPolicyDefaultsToNoInternet(t *testing.T) {
|
||||
np := BuildNetworkPolicy(NetPolParams{
|
||||
Namespace: "felis-build",
|
||||
RegistryNamespace: "felis-system",
|
||||
})
|
||||
for i, rule := range np.Spec.Egress {
|
||||
for _, peer := range rule.To {
|
||||
if peer.IPBlock != nil {
|
||||
t.Errorf("egress rule %d has an IPBlock but no package sources were configured", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildNetworkPolicyAllowsConfiguredPackageMirrors(t *testing.T) {
|
||||
cidr := "192.0.2.0/24"
|
||||
np := BuildNetworkPolicy(NetPolParams{
|
||||
Namespace: "felis-build",
|
||||
RegistryNamespace: "felis-system",
|
||||
PackageSourceCIDRs: []string{cidr},
|
||||
})
|
||||
found := false
|
||||
for _, rule := range np.Spec.Egress {
|
||||
for _, peer := range rule.To {
|
||||
if peer.IPBlock != nil && peer.IPBlock.CIDR == cidr {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("configured package-mirror CIDR %q not present in egress", cidr)
|
||||
}
|
||||
}
|
||||
|
||||
// The build SA is the most dangerous identity if mis-scoped: it must be bare —
|
||||
// no secrets, token automount disabled, and (by having no Role anywhere) no API
|
||||
// rights. We assert the spec-level properties the renderer controls.
|
||||
func TestBuildServiceAccountIsBare(t *testing.T) {
|
||||
sa := BuildServiceAccount("felis-build", "felis-build")
|
||||
if sa.Namespace != "felis-build" {
|
||||
t.Errorf("SA namespace = %q, want felis-build", sa.Namespace)
|
||||
}
|
||||
if sa.AutomountServiceAccountToken == nil || *sa.AutomountServiceAccountToken {
|
||||
t.Error("SA must disable token automounting")
|
||||
}
|
||||
if len(sa.Secrets) != 0 {
|
||||
t.Errorf("SA must carry no secrets, got %d", len(sa.Secrets))
|
||||
}
|
||||
if len(sa.ImagePullSecrets) != 0 {
|
||||
t.Errorf("SA must carry no image-pull secrets, got %d", len(sa.ImagePullSecrets))
|
||||
}
|
||||
}
|
||||
|
||||
// An invalid resource limit must surface as an error rather than render a Job
|
||||
// with no limits.
|
||||
func TestBuildJobRejectsBadResourceLimit(t *testing.T) {
|
||||
p := sampleJobParams()
|
||||
p.CPULimit = "not-a-quantity"
|
||||
if _, err := BuildJob(p); err == nil {
|
||||
t.Error("expected error for an unparseable CPU limit")
|
||||
}
|
||||
}
|
||||
|
||||
// ---- helpers ----
|
||||
|
||||
func hasArg(args []string, want string) bool {
|
||||
for _, a := range args {
|
||||
if a == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// argPairPresent reports whether flag is immediately followed by val (the
|
||||
// `--exit-code 1` two-token form).
|
||||
func argPairPresent(args []string, flag, val string) bool {
|
||||
for i := 0; i < len(args)-1; i++ {
|
||||
if args[i] == flag && args[i+1] == val {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func hasPolicyType(np *networkingv1.NetworkPolicy, t networkingv1.PolicyType) bool {
|
||||
for _, pt := range np.Spec.PolicyTypes {
|
||||
if pt == t {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func egressAllowsNamespace(np *networkingv1.NetworkPolicy, ns string) bool {
|
||||
for _, rule := range np.Spec.Egress {
|
||||
for _, peer := range rule.To {
|
||||
if peer.NamespaceSelector != nil &&
|
||||
peer.NamespaceSelector.MatchLabels["kubernetes.io/metadata.name"] == ns {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func egressAllowsPort(np *networkingv1.NetworkPolicy, port int32) bool {
|
||||
for _, rule := range np.Spec.Egress {
|
||||
for _, p := range rule.Ports {
|
||||
if p.Port != nil && p.Port.IntVal == port {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// guard against accidental shorthand: the test image ref must be host-qualified.
|
||||
func TestSampleRefIsHostQualified(t *testing.T) {
|
||||
if !strings.Contains(sampleJobParams().ImageRef, "/") {
|
||||
t.Fatal("sample image ref must be host-qualified")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// K8sJobs is the production Jobs backed by a controller-runtime client (spec
|
||||
// §16). It creates the Kaniko+Trivy build Job, reads its phase for the scan-gate
|
||||
// translation, and deletes it on cancel — nothing more. The cluster-bootstrap
|
||||
// objects (the felis-build namespace, the weak SA, and the egress
|
||||
// NetworkPolicy) are installed once by the deployment manifests (spec §21), not
|
||||
// per build, so this binding never needs to create them. It is integration-
|
||||
// tested against a live cluster, not the hermetic build_test.go suite.
|
||||
type K8sJobs struct {
|
||||
c client.Client
|
||||
cfg Config
|
||||
}
|
||||
|
||||
// NewK8sJobs builds a Jobs over c using cfg for the namespace and image refs.
|
||||
func NewK8sJobs(c client.Client, cfg Config) *K8sJobs {
|
||||
return &K8sJobs{c: c, cfg: cfg.withDefaults()}
|
||||
}
|
||||
|
||||
// CreateBuildJob renders and applies the build Job, returning its name. The Job
|
||||
// is the security-critical object; its shape is fixed by BuildJob (jobspec.go)
|
||||
// and asserted by jobspec_test.go.
|
||||
func (k *K8sJobs) CreateBuildJob(ctx context.Context, p JobParams) (string, error) {
|
||||
job, err := BuildJob(p)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := k.c.Create(ctx, job); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return job.Name, nil
|
||||
}
|
||||
|
||||
// JobPhase reads the Job and maps its status to a JobPhase. A missing Job
|
||||
// (GC'd, never created) is JobUnknown, which Sync treats as failed. The mapping
|
||||
// is deliberately conservative: a Job is Succeeded only when the Complete
|
||||
// condition is true, so a half-finished Job is never admitted.
|
||||
func (k *K8sJobs) JobPhase(ctx context.Context, jobName string) (JobPhase, error) {
|
||||
var job batchv1.Job
|
||||
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.cfg.Namespace, Name: jobName}, &job); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return JobUnknown, nil
|
||||
}
|
||||
return JobUnknown, err
|
||||
}
|
||||
for _, cond := range job.Status.Conditions {
|
||||
if cond.Status != "True" {
|
||||
continue
|
||||
}
|
||||
switch cond.Type {
|
||||
case batchv1.JobComplete:
|
||||
return JobSucceeded, nil
|
||||
case batchv1.JobFailed:
|
||||
// Covers a CRITICAL CVE (trivy --exit-code 1), a kaniko failure, and
|
||||
// DeadlineExceeded — all are a rejected build.
|
||||
return JobFailed, nil
|
||||
}
|
||||
}
|
||||
if job.Status.Active > 0 {
|
||||
return JobRunning, nil
|
||||
}
|
||||
return JobPending, nil
|
||||
}
|
||||
|
||||
// CancelBuildJob deletes the Job and, via background propagation, its pods. A
|
||||
// missing Job is not an error: cancellation is idempotent.
|
||||
func (k *K8sJobs) CancelBuildJob(ctx context.Context, jobName string) error {
|
||||
bg := metav1.DeletePropagationBackground
|
||||
obj := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: k.cfg.Namespace, Name: jobName},
|
||||
}
|
||||
if err := k.c.Delete(ctx, obj, &client.DeleteOptions{PropagationPolicy: &bg}); err != nil &&
|
||||
!apierrors.IsNotFound(err) {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"time"
|
||||
)
|
||||
|
||||
// PGStore is the production Store backed by Postgres (spec §6, §16). It writes
|
||||
// the two tables of the build subsystem — image_builds and image_whitelist —
|
||||
// and is the *only* component that holds database credentials: the build Pod
|
||||
// never does (the weak-SA red line). The SQL here is exercised by integration
|
||||
// tests against a live database, not the hermetic build_test.go suite. Every
|
||||
// statement is a narrow operation; there is no generic UPDATE escape hatch.
|
||||
type PGStore struct {
|
||||
db *sql.DB
|
||||
}
|
||||
|
||||
// NewPGStore wraps an existing pool (from store.PostgresDriver.DB()).
|
||||
func NewPGStore(db *sql.DB) *PGStore { return &PGStore{db: db} }
|
||||
|
||||
func (s *PGStore) CreateBuild(ctx context.Context, b *Build) error {
|
||||
const q = `INSERT INTO image_builds
|
||||
(id, image_ref, status, dockerfile, context_ref, base_image, requested_by, created_at)
|
||||
VALUES ($1, $2, $3, $4, NULLIF($5, ''), NULLIF($6, ''), $7, $8)`
|
||||
_, err := s.db.ExecContext(ctx, q,
|
||||
b.ID, b.ImageRef, string(b.Status), b.Dockerfile, b.ContextRef, b.BaseImage,
|
||||
b.RequestedBy, b.CreatedAt)
|
||||
return err
|
||||
}
|
||||
|
||||
func (s *PGStore) GetBuild(ctx context.Context, id string) (*Build, error) {
|
||||
const q = `SELECT id, image_ref, status, dockerfile, context_ref, base_image,
|
||||
requested_by, job_name, log_ref, error, created_at, finished_at
|
||||
FROM image_builds WHERE id = $1`
|
||||
return s.scanBuild(s.db.QueryRowContext(ctx, q, id))
|
||||
}
|
||||
|
||||
func (s *PGStore) scanBuild(row *sql.Row) (*Build, error) {
|
||||
var (
|
||||
b Build
|
||||
status string
|
||||
ctxRef, base, jobName, logRef, eMsg sql.NullString
|
||||
finished sql.NullTime
|
||||
)
|
||||
switch err := row.Scan(&b.ID, &b.ImageRef, &status, &b.Dockerfile, &ctxRef, &base,
|
||||
&b.RequestedBy, &jobName, &logRef, &eMsg, &b.CreatedAt, &finished); {
|
||||
case err == sql.ErrNoRows:
|
||||
return nil, ErrNotFound
|
||||
case err != nil:
|
||||
return nil, err
|
||||
}
|
||||
b.Status = Status(status)
|
||||
b.ContextRef = ctxRef.String
|
||||
b.BaseImage = base.String
|
||||
b.JobName = jobName.String
|
||||
b.LogRef = logRef.String
|
||||
b.Error = eMsg.String
|
||||
if finished.Valid {
|
||||
t := finished.Time
|
||||
b.FinishedAt = &t
|
||||
}
|
||||
return &b, nil
|
||||
}
|
||||
|
||||
func (s *PGStore) SetBuildJob(ctx context.Context, id, jobName string) error {
|
||||
const q = `UPDATE image_builds SET job_name = $2, status = 'building'
|
||||
WHERE id = $1 AND status = 'pending'`
|
||||
res, err := s.db.ExecContext(ctx, q, id, jobName)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return ErrNotFound
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *PGStore) FinishBuild(ctx context.Context, id string, status Status, errMsg string, at time.Time) error {
|
||||
const q = `UPDATE image_builds SET status = $2, error = NULLIF($3, ''), finished_at = $4
|
||||
WHERE id = $1`
|
||||
res, err := s.db.ExecContext(ctx, q, id, string(status), errMsg, at)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return ErrNotFound
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *PGStore) ListUnfinishedBuilds(ctx context.Context) ([]Build, error) {
|
||||
const q = `SELECT id, image_ref, status, dockerfile, context_ref, base_image,
|
||||
requested_by, job_name, log_ref, error, created_at, finished_at
|
||||
FROM image_builds WHERE status IN ('pending', 'building') ORDER BY created_at ASC`
|
||||
rows, err := s.db.QueryContext(ctx, q)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Build
|
||||
for rows.Next() {
|
||||
var (
|
||||
b Build
|
||||
status string
|
||||
ctxRef, base, jobName, logRef, eMsg sql.NullString
|
||||
finished sql.NullTime
|
||||
)
|
||||
if err := rows.Scan(&b.ID, &b.ImageRef, &status, &b.Dockerfile, &ctxRef, &base,
|
||||
&b.RequestedBy, &jobName, &logRef, &eMsg, &b.CreatedAt, &finished); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
b.Status = Status(status)
|
||||
b.ContextRef = ctxRef.String
|
||||
b.BaseImage = base.String
|
||||
b.JobName = jobName.String
|
||||
b.LogRef = logRef.String
|
||||
b.Error = eMsg.String
|
||||
if finished.Valid {
|
||||
t := finished.Time
|
||||
b.FinishedAt = &t
|
||||
}
|
||||
out = append(out, b)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// AdmitBuiltImage upserts the whitelist row on scan-gate success. ON CONFLICT
|
||||
// re-enables and re-stamps a previously-removed or superseded ref, so a rebuild
|
||||
// of the same tag re-admits it (spec §16).
|
||||
func (s *PGStore) AdmitBuiltImage(ctx context.Context, img Image) error {
|
||||
const q = `INSERT INTO image_whitelist
|
||||
(image_ref, source, build_id, added_by, enabled, added_at)
|
||||
VALUES ($1, 'built', NULLIF($2, ''), $3, true, $4)
|
||||
ON CONFLICT (image_ref) DO UPDATE
|
||||
SET source = 'built', build_id = EXCLUDED.build_id, added_by = EXCLUDED.added_by,
|
||||
enabled = true, added_at = EXCLUDED.added_at`
|
||||
_, err := s.db.ExecContext(ctx, q, img.ImageRef, img.BuildID, img.AddedBy, img.AddedAt)
|
||||
return err
|
||||
}
|
||||
|
||||
func (s *PGStore) ListImages(ctx context.Context) ([]Image, error) {
|
||||
const q = `SELECT image_ref, source, build_id, added_by, enabled, added_at
|
||||
FROM image_whitelist ORDER BY added_at DESC`
|
||||
rows, err := s.db.QueryContext(ctx, q)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Image
|
||||
for rows.Next() {
|
||||
var (
|
||||
img Image
|
||||
buildID sql.NullString
|
||||
)
|
||||
if err := rows.Scan(&img.ImageRef, &img.Source, &buildID, &img.AddedBy,
|
||||
&img.Enabled, &img.AddedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
img.BuildID = buildID.String
|
||||
out = append(out, img)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
func (s *PGStore) AddExternalImage(ctx context.Context, img Image) error {
|
||||
const q = `INSERT INTO image_whitelist
|
||||
(image_ref, source, added_by, enabled, added_at)
|
||||
VALUES ($1, 'external', $2, true, $3)
|
||||
ON CONFLICT (image_ref) DO UPDATE
|
||||
SET source = 'external', build_id = NULL, added_by = EXCLUDED.added_by,
|
||||
enabled = true, added_at = EXCLUDED.added_at`
|
||||
_, err := s.db.ExecContext(ctx, q, img.ImageRef, img.AddedBy, img.AddedAt)
|
||||
return err
|
||||
}
|
||||
|
||||
func (s *PGStore) RemoveImage(ctx context.Context, imageRef string) error {
|
||||
res, err := s.db.ExecContext(ctx, `DELETE FROM image_whitelist WHERE image_ref = $1`, imageRef)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return ErrNotFound
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
package build
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// invalidf builds a validation error wrapping ErrInvalid so the API layer maps
|
||||
// every malformed-request case to a single 400 path.
|
||||
func invalidf(format string, a ...any) error {
|
||||
return fmt.Errorf("%w: "+format, append([]any{ErrInvalid}, a...)...)
|
||||
}
|
||||
|
||||
// imageNameRE matches the path+tag of an image reference under the registry
|
||||
// host, e.g. "foo/bar:1.0" or "mc-paper:latest". It is intentionally strict:
|
||||
// lowercase path segments, an optional tag of the same alphabet, no digests, no
|
||||
// shell metacharacters that could escape into the Kaniko/Trivy argv.
|
||||
var imageNameRE = regexp.MustCompile(`^[a-z0-9]([a-z0-9._/-]*[a-z0-9])?(:[a-zA-Z0-9._-]+)?$`)
|
||||
|
||||
// Validate enforces the §16 admission rules on a build request: the target must
|
||||
// address the internal registry, the Dockerfile must be present and within the
|
||||
// size cap, and a context reference is required (Kaniko pulls it, spec §17).
|
||||
func Validate(req Request, cfg Config) error {
|
||||
cfg = cfg.withDefaults()
|
||||
if err := validateRegistryTarget(req.ImageRef, cfg.RegistryURL); err != nil {
|
||||
return err
|
||||
}
|
||||
if strings.TrimSpace(req.Dockerfile) == "" {
|
||||
return invalidf("dockerfile is required")
|
||||
}
|
||||
if len(req.Dockerfile) > cfg.MaxDockerfileBytes {
|
||||
return invalidf("dockerfile exceeds %d bytes", cfg.MaxDockerfileBytes)
|
||||
}
|
||||
if strings.TrimSpace(req.ContextRef) == "" {
|
||||
return invalidf("context reference is required")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ValidateImageRef checks a bare image reference (used by external admission,
|
||||
// where there is no registry-target constraint beyond well-formedness). A
|
||||
// whitelist entry may be a tag wildcard ("registry/foo:*", a legitimate
|
||||
// image_whitelist value per spec §15): the trailing ":*" is stripped before the
|
||||
// path is validated. This is safe because a wildcard is only ever compared
|
||||
// against concrete refs by imageMatches — it never reaches a Kaniko/Trivy argv,
|
||||
// unlike a build push target (validateRegistryTarget stays strictly concrete).
|
||||
func ValidateImageRef(ref string) error {
|
||||
if ref == "" {
|
||||
return invalidf("image reference is required")
|
||||
}
|
||||
host, rest, ok := splitRegistryHost(ref)
|
||||
if !ok {
|
||||
return invalidf("image reference %q must be host-qualified (host/path:tag)", ref)
|
||||
}
|
||||
if repo, isWildcard := strings.CutSuffix(rest, ":*"); isWildcard {
|
||||
rest = repo
|
||||
}
|
||||
if !imageNameRE.MatchString(rest) {
|
||||
return invalidf("invalid image path/tag %q", rest)
|
||||
}
|
||||
_ = host
|
||||
return nil
|
||||
}
|
||||
|
||||
// splitTag splits an image reference's path from its tag. It is registry-port
|
||||
// safe: only a colon *after* the final path separator is a tag separator, so
|
||||
// "registry:5000/foo" splits to ("registry:5000/foo", ""), never a bogus tag.
|
||||
func splitTag(ref string) (repo, tag string) {
|
||||
slash := strings.LastIndexByte(ref, '/')
|
||||
colon := strings.LastIndexByte(ref, ':')
|
||||
if colon > slash {
|
||||
return ref[:colon], ref[colon+1:]
|
||||
}
|
||||
return ref, ""
|
||||
}
|
||||
|
||||
// imageMatches reports whether a concrete image ref is admitted by a whitelist
|
||||
// pattern. A pattern is either exact ("registry/foo:1.0") or a tag wildcard
|
||||
// ("registry/foo:*", spec §15) that matches any non-empty tag on the same repo.
|
||||
// A wildcard never matches an untagged ref — admission is always to a concrete
|
||||
// tag.
|
||||
func imageMatches(ref, pattern string) bool {
|
||||
if ref == pattern {
|
||||
return true
|
||||
}
|
||||
prefix, ok := strings.CutSuffix(pattern, ":*")
|
||||
if !ok {
|
||||
return false
|
||||
}
|
||||
repo, tag := splitTag(ref)
|
||||
return repo == prefix && tag != ""
|
||||
}
|
||||
|
||||
// validateRegistryTarget enforces that a build pushes only to the configured
|
||||
// internal registry — never an arbitrary external host (spec §16: the build can
|
||||
// never push elsewhere; the registry is not a public ingress). When RegistryURL
|
||||
// is unset (tests / not configured) the host constraint is skipped but the
|
||||
// path/tag are still validated.
|
||||
func validateRegistryTarget(ref, registryURL string) error {
|
||||
if ref == "" {
|
||||
return invalidf("image reference is required")
|
||||
}
|
||||
host, rest, ok := splitRegistryHost(ref)
|
||||
if !ok {
|
||||
return invalidf("image reference %q must target the internal registry (host/path:tag)", ref)
|
||||
}
|
||||
if !imageNameRE.MatchString(rest) {
|
||||
return invalidf("invalid image path/tag %q", rest)
|
||||
}
|
||||
if registryURL != "" && host != registryHost(registryURL) {
|
||||
return invalidf("image reference %q must target the internal registry %q, not %q",
|
||||
ref, registryHost(registryURL), host)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// splitRegistryHost separates the registry host from the remaining path+tag. A
|
||||
// reference is host-qualified only if the first segment looks like a registry
|
||||
// host — it contains a '.' or ':' (port), matching containerd's heuristic.
|
||||
// "foo/bar:1" (Docker Hub shorthand) is rejected: builds must be explicit about
|
||||
// the internal registry.
|
||||
func splitRegistryHost(ref string) (host, rest string, ok bool) {
|
||||
slash := strings.IndexByte(ref, '/')
|
||||
if slash < 0 {
|
||||
return "", "", false
|
||||
}
|
||||
host = ref[:slash]
|
||||
rest = ref[slash+1:]
|
||||
if !strings.ContainsAny(host, ".:") {
|
||||
return "", "", false
|
||||
}
|
||||
if rest == "" {
|
||||
return "", "", false
|
||||
}
|
||||
return host, rest, true
|
||||
}
|
||||
|
||||
// registryHost strips any scheme and path from a configured registry URL,
|
||||
// leaving the host[:port] that an image reference must match.
|
||||
func registryHost(registryURL string) string {
|
||||
h := registryURL
|
||||
if i := strings.Index(h, "://"); i >= 0 {
|
||||
h = h[i+3:]
|
||||
}
|
||||
if i := strings.IndexByte(h, '/'); i >= 0 {
|
||||
h = h[:i]
|
||||
}
|
||||
return h
|
||||
}
|
||||
Reference in new issue
Block a user