feat(core): add naming, RCON, store, config, and image-build libraries

Foundational libraries: deterministic resource naming, the RCON client, the Postgres store with embedded SQL migrations, configuration loading, and container image-build helpers.
This commit is contained in:
flyemoji committed 2026-06-26 23:31:58 +09:00
1 parent 7fbebfe843
commit 708cdfc5b8
18 files changed
+3475

No files matched your search

+505
View File
@@ -0,0 +1,505 @@
// Package build implements the image build subsystem (spec §16) — "the
// platform's biggest security surface". A SysAdmin uploads a Dockerfile and a
// context tarball; felis-api starts an in-cluster Kaniko Job that builds and
// pushes to the internal registry, after which a Trivy scan gates admission to
// the image whitelist.
//
// Trust model (spec §16, §22): we trust the SysAdmin at the *ingress* (only an
// admin through Zero Trust may submit a build) but never trust the *Dockerfile
// at runtime* — an arbitrary Dockerfile is build-time RCE whose victim is the
// cluster, not the uploader. So the build Pod runs with a deliberately weak
// service account in an isolated namespace that can only push to the registry
// and cannot touch the minecraft namespace, the felis database, or the K8s API
// (spec §21). Those isolation guarantees live in the Job/NetworkPolicy specs
// (jobspec.go) and are asserted by unit tests, since no cluster runs here.
//
// The Trivy gate is enforced as the build Pod's *exit code*: a kaniko
// initContainer builds and pushes, then a trivy container scans the pushed ref
// with `--exit-code 1 --severity CRITICAL`. Therefore "Job Succeeded" is
// equivalent to "pushed AND no CRITICAL CVE". felis-api observes the Job phase
// and performs the database writes — the build Pod itself never has database
// credentials (the weak-SA red line). On success the image is admitted to
// image_whitelist with enabled=true (recording added_by); on failure the build
// is marked failed and nothing is admitted (spec §16: the only retained
// automatic gate).
//
// The Builder depends on the Store and Jobs interfaces, so submission, the
// scan-gate translation, cancellation, and image admission are all unit-tested
// against in-memory fakes. The Postgres (pgStore) and controller-runtime
// (k8sJobs) implementations compile here but are exercised only by integration
// tests against a live database / cluster.
package build
import (
"context"
"errors"
"fmt"
"strings"
"time"
)
// Status mirrors the build_status enum (spec §6).
type Status string
const (
StatusPending Status = "pending"
StatusBuilding Status = "building"
StatusSucceeded Status = "succeeded"
StatusFailed Status = "failed"
StatusCancelled Status = "cancelled"
)
// terminal reports whether a status is final and no longer reconciled.
func (s Status) terminal() bool {
switch s {
case StatusSucceeded, StatusFailed, StatusCancelled:
return true
default:
return false
}
}
// JobPhase is the build Pod's lifecycle as observed from the K8s Job, decoupled
// from any K8s type so the scan-gate translation stays unit-testable.
type JobPhase int
const (
// JobUnknown means the Job was not found (e.g. GC'd); treated as failed.
JobUnknown JobPhase = iota
JobPending
JobRunning
// JobSucceeded means kaniko pushed AND trivy found no CRITICAL CVE — the
// scan gate passed (spec §16).
JobSucceeded
// JobFailed means kaniko failed OR trivy found a CRITICAL CVE — the build
// is rejected and nothing is admitted.
JobFailed
)
// image admission sources (spec §6 image_whitelist.source).
const (
SourceBuilt = "built"
SourceExternal = "external"
)
// ErrNotFound is returned when a build id / image ref does not exist.
var ErrNotFound = errors.New("build: not found")
// ErrAlreadyTerminal is returned by Cancel when the build has already finished.
var ErrAlreadyTerminal = errors.New("build: already in a terminal state")
// ErrInvalid wraps every request-validation failure (bad image ref, missing /
// oversize Dockerfile, missing context). Callers map it to a 400; it is kept
// distinct from store/cluster failures so those surface as 500.
var ErrInvalid = errors.New("build: invalid request")
// Request is the validated POST /images/build input (spec §16). The dockerfile
// and context are archived for audit; the target ref must address the internal
// registry (enforced in Validate).
type Request struct {
// ImageRef is the push target, e.g. registry.felis.svc:5000/foo:1.0. It must
// be under the configured internal registry — a build can never push
// elsewhere.
ImageRef string
// Dockerfile is the uploaded build recipe (size-capped).
Dockerfile string
// ContextRef locates the uploaded tar.gz context in object storage / a PVC
// (spec §17: Kaniko pulls it; Git context is intentionally not supported).
ContextRef string
// BaseImage is the resolved FROM, recorded for audit only — it is NOT a hard
// gate (spec §16: base FROM is not hard-gated; the scan + egress lock cover
// poisoned bases).
BaseImage string
// RequestedBy is the admin Access email, used as the audit actor and the
// added_by of any admitted image.
RequestedBy string
}
// Build mirrors an image_builds row (spec §6).
type Build struct {
ID string `json:"id"`
ImageRef string `json:"image_ref"`
Status Status `json:"status"`
Dockerfile string `json:"dockerfile,omitempty"`
ContextRef string `json:"context_ref,omitempty"`
BaseImage string `json:"base_image,omitempty"`
RequestedBy string `json:"requested_by"`
JobName string `json:"job_name,omitempty"`
LogRef string `json:"log_ref,omitempty"`
Error string `json:"error,omitempty"`
CreatedAt time.Time `json:"created_at"`
FinishedAt *time.Time `json:"finished_at,omitempty"`
}
// Image mirrors an image_whitelist row (spec §6): the dynamic, auditable image
// admission list that the create-server form reads from.
type Image struct {
ImageRef string `json:"image_ref"`
Source string `json:"source"`
BuildID string `json:"build_id,omitempty"`
AddedBy string `json:"added_by"`
Enabled bool `json:"enabled"`
AddedAt time.Time `json:"added_at"`
}
// Store is the business-layer persistence the Builder depends on (image_builds
// + image_whitelist). It is an interface so the Builder is tested against an
// in-memory fake; the Postgres implementation (pgStore) is integration-tested
// only.
type Store interface {
// CreateBuild inserts a new image_builds row (status pending).
CreateBuild(ctx context.Context, b *Build) error
// GetBuild loads one build, or ErrNotFound.
GetBuild(ctx context.Context, id string) (*Build, error)
// SetBuildJob records the Job name and advances status to building.
SetBuildJob(ctx context.Context, id, jobName string) error
// FinishBuild sets a terminal status, an optional error, and finished_at.
FinishBuild(ctx context.Context, id string, status Status, errMsg string, at time.Time) error
// ListUnfinishedBuilds returns builds still being reconciled (status pending
// or building), oldest first — the work list for SyncAll.
ListUnfinishedBuilds(ctx context.Context) ([]Build, error)
// AdmitBuiltImage upserts an image_whitelist row with enabled=true and
// source=built (the scan-gate success path, spec §16). It records added_by.
AdmitBuiltImage(ctx context.Context, img Image) error
// ListImages returns the image whitelist.
ListImages(ctx context.Context) ([]Image, error)
// AddExternalImage upserts an externally-pushed image (spec §15 external
// admission; source=external, no build_id).
AddExternalImage(ctx context.Context, img Image) error
// RemoveImage deletes an image_whitelist row, or ErrNotFound.
RemoveImage(ctx context.Context, imageRef string) error
}
// Jobs is the cluster-side build lifecycle the Builder depends on. It is an
// interface so the scan-gate translation is tested against a fake; the
// controller-runtime implementation (k8sJobs) is integration-tested only — it
// requires a live cluster.
type Jobs interface {
// CreateBuildJob starts the Kaniko+Trivy Job for p in the felis-build
// namespace and returns the Job name.
CreateBuildJob(ctx context.Context, p JobParams) (jobName string, err error)
// JobPhase reports the current phase of a previously-created Job.
JobPhase(ctx context.Context, jobName string) (JobPhase, error)
// CancelBuildJob deletes the Job (and its pods), tolerating not-found.
CancelBuildJob(ctx context.Context, jobName string) error
}
// Config parameterises the build subsystem from felis.toml (spec §24 [registry]
// + safety limits). It is validated by withDefaults before use.
type Config struct {
// Namespace is the isolated build namespace (spec §16: felis-build).
Namespace string
// ServiceAccount is the weak SA the build Pod runs as. It MUST NOT be the
// felis-api SA (spec §16 red line).
ServiceAccount string
// RegistryURL is the internal registry the build pushes to and Trivy scans
// (spec §17). Image refs are validated to be under it.
RegistryURL string
// KanikoImage / TrivyImage are the executor images.
KanikoImage string
TrivyImage string
// Deadline caps a build's wall-clock (spec §16: activeDeadlineSeconds).
Deadline time.Duration
// MaxDockerfileBytes caps the uploaded Dockerfile (spec §16: context size
// limits). Zero applies the default.
MaxDockerfileBytes int
// CPULimit / MemLimit cap each build container (spec §16: resource limits).
CPULimit string
MemLimit string
}
// Defaults applied when a Config field is left zero.
const (
defaultNamespace = "felis-build"
defaultServiceAccount = "felis-build"
defaultKanikoImage = "gcr.io/kaniko-project/executor:latest"
defaultTrivyImage = "aquasec/trivy:latest"
defaultDeadline = 30 * time.Minute
defaultMaxDockerfile = 256 * 1024 // 256 KiB
defaultCPULimit = "2"
defaultMemLimit = "4Gi"
)
// withDefaults returns a copy of c with zero fields filled, so a partially
// configured Config (or the zero value, in tests) is always usable.
func (c Config) withDefaults() Config {
if c.Namespace == "" {
c.Namespace = defaultNamespace
}
if c.ServiceAccount == "" {
c.ServiceAccount = defaultServiceAccount
}
if c.KanikoImage == "" {
c.KanikoImage = defaultKanikoImage
}
if c.TrivyImage == "" {
c.TrivyImage = defaultTrivyImage
}
if c.Deadline <= 0 {
c.Deadline = defaultDeadline
}
if c.MaxDockerfileBytes <= 0 {
c.MaxDockerfileBytes = defaultMaxDockerfile
}
if c.CPULimit == "" {
c.CPULimit = defaultCPULimit
}
if c.MemLimit == "" {
c.MemLimit = defaultMemLimit
}
return c
}
// Builder orchestrates the build subsystem. It holds no mutable state; the
// clock and id generator are injectable for hermetic tests.
type Builder struct {
Store Store
Jobs Jobs
Config Config
// Now is the clock, injectable for tests. Defaults to time.Now.
Now func() time.Time
// IDGen mints build ids. Defaults to a time-based generator.
IDGen func() string
}
func (b *Builder) now() time.Time {
if b.Now != nil {
return b.Now()
}
return time.Now()
}
func (b *Builder) newID() string {
if b.IDGen != nil {
return b.IDGen()
}
return fmt.Sprintf("bld-%d", time.Now().UnixNano())
}
// Submit validates req, records a pending build, and starts the Kaniko+Trivy
// Job (spec §16). The build is returned in the building state once the Job is
// created; if Job creation fails the build is marked failed so it never lingers
// pending. The caller (felis-api) drives the build to a terminal state by
// polling Sync / SyncAll.
func (b *Builder) Submit(ctx context.Context, req Request) (*Build, error) {
cfg := b.Config.withDefaults()
if err := Validate(req, cfg); err != nil {
return nil, err
}
now := b.now()
bld := &Build{
ID: b.newID(),
ImageRef: req.ImageRef,
Status: StatusPending,
Dockerfile: req.Dockerfile,
ContextRef: req.ContextRef,
BaseImage: req.BaseImage,
RequestedBy: req.RequestedBy,
CreatedAt: now,
}
if err := b.Store.CreateBuild(ctx, bld); err != nil {
return nil, err
}
jobName, err := b.Jobs.CreateBuildJob(ctx, b.jobParams(bld, cfg))
if err != nil {
// The pending row exists; mark it failed so it is not reconciled forever.
_ = b.Store.FinishBuild(ctx, bld.ID, StatusFailed, "job creation failed: "+err.Error(), b.now())
bld.Status = StatusFailed
bld.Error = "job creation failed: " + err.Error()
return bld, fmt.Errorf("build: create job: %w", err)
}
if err := b.Store.SetBuildJob(ctx, bld.ID, jobName); err != nil {
return nil, err
}
bld.JobName = jobName
bld.Status = StatusBuilding
return bld, nil
}
// jobParams projects a build + config onto the inputs jobspec.go renders.
func (b *Builder) jobParams(bld *Build, cfg Config) JobParams {
return JobParams{
BuildID: bld.ID,
ImageRef: bld.ImageRef,
ContextRef: bld.ContextRef,
Namespace: cfg.Namespace,
ServiceAccount: cfg.ServiceAccount,
RegistryURL: cfg.RegistryURL,
KanikoImage: cfg.KanikoImage,
TrivyImage: cfg.TrivyImage,
Deadline: cfg.Deadline,
CPULimit: cfg.CPULimit,
MemLimit: cfg.MemLimit,
}
}
// Get returns a build by id, or ErrNotFound.
func (b *Builder) Get(ctx context.Context, id string) (*Build, error) {
return b.Store.GetBuild(ctx, id)
}
// Sync reconciles one non-terminal build against its Job phase — the scan-gate
// translation (spec §16). A terminal build is returned unchanged (idempotent).
//
// - JobSucceeded → status=succeeded AND the image is admitted to the whitelist
// with enabled=true (kaniko pushed and trivy found no CRITICAL CVE).
// - JobFailed / JobUnknown → status=failed, nothing admitted (a CRITICAL CVE
// surfaces here as a failed Job, since trivy runs with --exit-code 1).
// - JobPending / JobRunning → no change.
//
// The image admission is performed by felis-api (this code path), never by the
// build Pod, which holds no database credentials.
func (b *Builder) Sync(ctx context.Context, id string) (*Build, error) {
bld, err := b.Store.GetBuild(ctx, id)
if err != nil {
return nil, err
}
if bld.Status.terminal() {
return bld, nil
}
if bld.JobName == "" {
// Created but the Job name was never recorded; treat as failed rather
// than reconcile forever against a phantom Job.
return b.finish(ctx, bld, StatusFailed, "no build job recorded")
}
phase, err := b.Jobs.JobPhase(ctx, bld.JobName)
if err != nil {
return nil, err
}
switch phase {
case JobSucceeded:
now := b.now()
// Admit the image first; only then mark the build succeeded, so a
// succeeded build always has its whitelist row (no admitted-but-not-
// recorded window if the second write fails).
if err := b.Store.AdmitBuiltImage(ctx, Image{
ImageRef: bld.ImageRef,
Source: SourceBuilt,
BuildID: bld.ID,
AddedBy: bld.RequestedBy,
Enabled: true,
AddedAt: now,
}); err != nil {
return nil, err
}
return b.finishAt(ctx, bld, StatusSucceeded, "", now)
case JobFailed, JobUnknown:
return b.finish(ctx, bld, StatusFailed, "build job failed or scan found a CRITICAL CVE")
default: // JobPending / JobRunning
return bld, nil
}
}
// SyncAll reconciles every unfinished build and returns the count advanced to a
// terminal state. felis-api calls this periodically (spec §16: the scan gate is
// observed, not pushed by the build Pod).
func (b *Builder) SyncAll(ctx context.Context) (int, error) {
builds, err := b.Store.ListUnfinishedBuilds(ctx)
if err != nil {
return 0, err
}
advanced := 0
for i := range builds {
bld, err := b.Sync(ctx, builds[i].ID)
if err != nil {
return advanced, err
}
if bld.Status.terminal() {
advanced++
}
}
return advanced, nil
}
// Cancel stops an in-flight build: delete its Job and mark it cancelled. A
// build that has already finished returns ErrAlreadyTerminal.
func (b *Builder) Cancel(ctx context.Context, id string) (*Build, error) {
bld, err := b.Store.GetBuild(ctx, id)
if err != nil {
return nil, err
}
if bld.Status.terminal() {
return nil, ErrAlreadyTerminal
}
if bld.JobName != "" {
if err := b.Jobs.CancelBuildJob(ctx, bld.JobName); err != nil {
return nil, err
}
}
return b.finish(ctx, bld, StatusCancelled, "cancelled by administrator")
}
// finish marks a build terminal at the current clock and returns the updated
// view without a second round-trip.
func (b *Builder) finish(ctx context.Context, bld *Build, status Status, msg string) (*Build, error) {
return b.finishAt(ctx, bld, status, msg, b.now())
}
func (b *Builder) finishAt(ctx context.Context, bld *Build, status Status, msg string, at time.Time) (*Build, error) {
if err := b.Store.FinishBuild(ctx, bld.ID, status, msg, at); err != nil {
return nil, err
}
bld.Status = status
bld.Error = msg
finished := at
bld.FinishedAt = &finished
return bld, nil
}
// ListImages returns the image whitelist (spec §15 create-server form source).
func (b *Builder) ListImages(ctx context.Context) ([]Image, error) {
return b.Store.ListImages(ctx)
}
// AddExternalImage admits an externally-pushed image (spec §15). It is enabled
// immediately; external images bypass the build pipeline but are still recorded
// with added_by for audit.
func (b *Builder) AddExternalImage(ctx context.Context, imageRef, addedBy string) (*Image, error) {
if err := ValidateImageRef(imageRef); err != nil {
return nil, err
}
img := Image{
ImageRef: imageRef,
Source: SourceExternal,
AddedBy: addedBy,
Enabled: true,
AddedAt: b.now(),
}
if err := b.Store.AddExternalImage(ctx, img); err != nil {
return nil, err
}
return &img, nil
}
// RemoveImage withdraws an image from the whitelist (spec §22: dynamic,
// auditable). It does not delete the underlying registry blob.
func (b *Builder) RemoveImage(ctx context.Context, imageRef string) error {
return b.Store.RemoveImage(ctx, imageRef)
}
// ImageAdmitted reports whether a concrete image reference is on the whitelist
// and enabled (spec §15: the create-server form may only choose an admitted
// image). A disabled row never admits. A wildcard whitelist entry
// ("registry/foo:*") admits any concrete tag on that repo (imageMatches); the
// caller always passes a concrete ref, never a wildcard. An empty ref is never
// admitted.
func (b *Builder) ImageAdmitted(ctx context.Context, imageRef string) (bool, error) {
if strings.TrimSpace(imageRef) == "" {
return false, nil
}
images, err := b.Store.ListImages(ctx)
if err != nil {
return false, err
}
for _, img := range images {
if img.Enabled && imageMatches(imageRef, img.ImageRef) {
return true, nil
}
}
return false, nil
}
+478
View File
@@ -0,0 +1,478 @@
package build
import (
"context"
"errors"
"testing"
"time"
)
var testNow = time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
// ---- in-memory fakes ----
type fakeStore struct {
builds map[string]*Build
images map[string]Image
// call recorders
admitted []Image
finished []string // "id:status"
removeErr error
createErr error
}
func newFakeStore() *fakeStore {
return &fakeStore{builds: map[string]*Build{}, images: map[string]Image{}}
}
func (f *fakeStore) CreateBuild(_ context.Context, b *Build) error {
if f.createErr != nil {
return f.createErr
}
cp := *b
f.builds[b.ID] = &cp
return nil
}
func (f *fakeStore) GetBuild(_ context.Context, id string) (*Build, error) {
b, ok := f.builds[id]
if !ok {
return nil, ErrNotFound
}
cp := *b
return &cp, nil
}
func (f *fakeStore) SetBuildJob(_ context.Context, id, jobName string) error {
b, ok := f.builds[id]
if !ok {
return ErrNotFound
}
b.JobName = jobName
b.Status = StatusBuilding
return nil
}
func (f *fakeStore) FinishBuild(_ context.Context, id string, status Status, errMsg string, at time.Time) error {
b, ok := f.builds[id]
if !ok {
return ErrNotFound
}
b.Status = status
b.Error = errMsg
finished := at
b.FinishedAt = &finished
f.finished = append(f.finished, id+":"+string(status))
return nil
}
func (f *fakeStore) ListUnfinishedBuilds(_ context.Context) ([]Build, error) {
var out []Build
for _, b := range f.builds {
if !b.Status.terminal() {
out = append(out, *b)
}
}
return out, nil
}
func (f *fakeStore) AdmitBuiltImage(_ context.Context, img Image) error {
f.images[img.ImageRef] = img
f.admitted = append(f.admitted, img)
return nil
}
func (f *fakeStore) ListImages(_ context.Context) ([]Image, error) {
var out []Image
for _, img := range f.images {
out = append(out, img)
}
return out, nil
}
func (f *fakeStore) AddExternalImage(_ context.Context, img Image) error {
f.images[img.ImageRef] = img
return nil
}
func (f *fakeStore) RemoveImage(_ context.Context, ref string) error {
if f.removeErr != nil {
return f.removeErr
}
if _, ok := f.images[ref]; !ok {
return ErrNotFound
}
delete(f.images, ref)
return nil
}
type fakeJobs struct {
phase JobPhase
phaseErr error
createErr error
created []JobParams
cancelled []string
}
func (f *fakeJobs) CreateBuildJob(_ context.Context, p JobParams) (string, error) {
if f.createErr != nil {
return "", f.createErr
}
f.created = append(f.created, p)
return BuildJobName(p.BuildID), nil
}
func (f *fakeJobs) JobPhase(_ context.Context, _ string) (JobPhase, error) {
return f.phase, f.phaseErr
}
func (f *fakeJobs) CancelBuildJob(_ context.Context, jobName string) error {
f.cancelled = append(f.cancelled, jobName)
return nil
}
// newBuilder wires a Builder over fresh fakes with a frozen clock and a
// deterministic id generator.
func newBuilder() (*Builder, *fakeStore, *fakeJobs) {
st := newFakeStore()
jb := &fakeJobs{phase: JobPending}
n := 0
b := &Builder{
Store: st,
Jobs: jb,
Config: Config{
RegistryURL: "registry.felis.svc:5000",
},
Now: func() time.Time { return testNow },
IDGen: func() string {
n++
return "bld-" + string(rune('0'+n))
},
}
return b, st, jb
}
func goodRequest() Request {
return Request{
ImageRef: "registry.felis.svc:5000/mc-paper:1.0",
Dockerfile: "FROM eclipse-temurin:21\nCOPY . /data\n",
ContextRef: "tar://contexts/abc.tar.gz",
RequestedBy: "[email protected]",
}
}
// ---- submission ----
func TestSubmitCreatesPendingBuildAndStartsJob(t *testing.T) {
b, st, jb := newBuilder()
bld, err := b.Submit(context.Background(), goodRequest())
if err != nil {
t.Fatalf("Submit: %v", err)
}
if bld.Status != StatusBuilding {
t.Errorf("status = %q, want building", bld.Status)
}
if bld.JobName == "" {
t.Error("job name not recorded")
}
if got := st.builds[bld.ID]; got == nil || got.Dockerfile == "" {
t.Error("build row not persisted with dockerfile")
}
if len(jb.created) != 1 {
t.Fatalf("CreateBuildJob calls = %d, want 1", len(jb.created))
}
// The Job must be parameterised with the weak build SA and the namespace,
// never the api identity — this is the §16 red line, asserted at the seam.
p := jb.created[0]
if p.ServiceAccount != defaultServiceAccount {
t.Errorf("job SA = %q, want %q", p.ServiceAccount, defaultServiceAccount)
}
if p.Namespace != defaultNamespace {
t.Errorf("job namespace = %q, want %q", p.Namespace, defaultNamespace)
}
if p.RegistryURL != "registry.felis.svc:5000" {
t.Errorf("job registry = %q", p.RegistryURL)
}
if p.Deadline <= 0 {
t.Error("job deadline not set")
}
}
func TestSubmitRejectsExternalRegistryTarget(t *testing.T) {
b, st, jb := newBuilder()
req := goodRequest()
req.ImageRef = "docker.io/library/evil:latest"
if _, err := b.Submit(context.Background(), req); err == nil {
t.Fatal("expected rejection of non-internal registry target")
}
if len(st.builds) != 0 {
t.Error("a rejected build must not be persisted")
}
if len(jb.created) != 0 {
t.Error("a rejected build must not start a job")
}
}
func TestSubmitRejectsEmptyAndOversizeDockerfile(t *testing.T) {
b, _, _ := newBuilder()
req := goodRequest()
req.Dockerfile = " "
if _, err := b.Submit(context.Background(), req); err == nil {
t.Error("expected rejection of empty dockerfile")
}
b2, _, _ := newBuilder()
b2.Config.MaxDockerfileBytes = 16
req2 := goodRequest()
req2.Dockerfile = "FROM scratch\n# padding padding padding padding\n"
if _, err := b2.Submit(context.Background(), req2); err == nil {
t.Error("expected rejection of oversize dockerfile")
}
}
func TestSubmitRequiresContext(t *testing.T) {
b, _, _ := newBuilder()
req := goodRequest()
req.ContextRef = ""
if _, err := b.Submit(context.Background(), req); err == nil {
t.Error("expected rejection when context reference is missing")
}
}
func TestSubmitMarksBuildFailedWhenJobCreationFails(t *testing.T) {
b, st, jb := newBuilder()
jb.createErr = errors.New("apiserver down")
bld, err := b.Submit(context.Background(), goodRequest())
if err == nil {
t.Fatal("expected error when job creation fails")
}
if bld == nil || bld.Status != StatusFailed {
t.Fatalf("build should be marked failed, got %+v", bld)
}
// The persisted row must not be left pending forever.
if got := st.builds[bld.ID]; got == nil || got.Status != StatusFailed {
t.Errorf("persisted status = %v, want failed", got)
}
}
// ---- the scan gate (Job phase -> DB) ----
func TestSyncSucceededAdmitsImageWithAddedBy(t *testing.T) {
b, st, jb := newBuilder()
bld, _ := b.Submit(context.Background(), goodRequest())
jb.phase = JobSucceeded
got, err := b.Sync(context.Background(), bld.ID)
if err != nil {
t.Fatalf("Sync: %v", err)
}
if got.Status != StatusSucceeded {
t.Errorf("status = %q, want succeeded", got.Status)
}
img, ok := st.images[bld.ImageRef]
if !ok {
t.Fatal("succeeded build did not admit its image to the whitelist")
}
if !img.Enabled {
t.Error("admitted image must be enabled=true")
}
if img.Source != SourceBuilt {
t.Errorf("source = %q, want built", img.Source)
}
if img.AddedBy != "[email protected]" {
t.Errorf("added_by = %q, want the requester", img.AddedBy)
}
if img.BuildID != bld.ID {
t.Errorf("build_id = %q, want %q", img.BuildID, bld.ID)
}
}
func TestSyncFailedDoesNotAdmitImage(t *testing.T) {
// A failed Job is exactly the CRITICAL-CVE case: trivy --exit-code 1 fails
// the Pod, so the gate is enforced as the Job verdict (spec §16).
b, st, jb := newBuilder()
bld, _ := b.Submit(context.Background(), goodRequest())
jb.phase = JobFailed
got, err := b.Sync(context.Background(), bld.ID)
if err != nil {
t.Fatalf("Sync: %v", err)
}
if got.Status != StatusFailed {
t.Errorf("status = %q, want failed", got.Status)
}
if len(st.images) != 0 {
t.Error("a failed/CRITICAL build must NOT admit any image")
}
if len(st.admitted) != 0 {
t.Error("AdmitBuiltImage must not be called on failure")
}
}
func TestSyncRunningIsNoOp(t *testing.T) {
b, _, jb := newBuilder()
bld, _ := b.Submit(context.Background(), goodRequest())
jb.phase = JobRunning
got, err := b.Sync(context.Background(), bld.ID)
if err != nil {
t.Fatalf("Sync: %v", err)
}
if got.Status != StatusBuilding {
t.Errorf("status = %q, want building (unchanged)", got.Status)
}
}
func TestSyncIsIdempotentOnTerminalBuild(t *testing.T) {
b, st, jb := newBuilder()
bld, _ := b.Submit(context.Background(), goodRequest())
jb.phase = JobSucceeded
if _, err := b.Sync(context.Background(), bld.ID); err != nil {
t.Fatalf("first Sync: %v", err)
}
// Flip the phase to a value that would admit again; a terminal build must
// not be re-reconciled.
before := len(st.admitted)
if _, err := b.Sync(context.Background(), bld.ID); err != nil {
t.Fatalf("second Sync: %v", err)
}
if len(st.admitted) != before {
t.Error("terminal build was re-admitted on a second Sync")
}
}
func TestSyncAllAdvancesUnfinishedBuilds(t *testing.T) {
b, _, jb := newBuilder()
if _, err := b.Submit(context.Background(), goodRequest()); err != nil {
t.Fatalf("submit: %v", err)
}
jb.phase = JobSucceeded
n, err := b.SyncAll(context.Background())
if err != nil {
t.Fatalf("SyncAll: %v", err)
}
if n != 1 {
t.Errorf("advanced = %d, want 1", n)
}
}
// ---- cancellation ----
func TestCancelDeletesJobAndMarksCancelled(t *testing.T) {
b, _, jb := newBuilder()
bld, _ := b.Submit(context.Background(), goodRequest())
got, err := b.Cancel(context.Background(), bld.ID)
if err != nil {
t.Fatalf("Cancel: %v", err)
}
if got.Status != StatusCancelled {
t.Errorf("status = %q, want cancelled", got.Status)
}
if len(jb.cancelled) != 1 {
t.Errorf("CancelBuildJob calls = %d, want 1", len(jb.cancelled))
}
}
func TestCancelTerminalBuildFails(t *testing.T) {
b, _, jb := newBuilder()
bld, _ := b.Submit(context.Background(), goodRequest())
jb.phase = JobSucceeded
if _, err := b.Sync(context.Background(), bld.ID); err != nil {
t.Fatalf("Sync: %v", err)
}
if _, err := b.Cancel(context.Background(), bld.ID); !errors.Is(err, ErrAlreadyTerminal) {
t.Errorf("Cancel on terminal build err = %v, want ErrAlreadyTerminal", err)
}
}
// ---- external admission ----
func TestAddExternalImageIsEnabledAndRecorded(t *testing.T) {
b, st, _ := newBuilder()
img, err := b.AddExternalImage(context.Background(), "registry.felis.svc:5000/ext:1", "[email protected]")
if err != nil {
t.Fatalf("AddExternalImage: %v", err)
}
if img.Source != SourceExternal || !img.Enabled {
t.Errorf("external image = %+v, want enabled external", img)
}
if _, ok := st.images["registry.felis.svc:5000/ext:1"]; !ok {
t.Error("external image not persisted")
}
}
func TestAddExternalImageRejectsMalformedRef(t *testing.T) {
b, _, _ := newBuilder()
if _, err := b.AddExternalImage(context.Background(), "not a ref!!", "[email protected]"); err == nil {
t.Error("expected rejection of malformed image ref")
}
}
func TestRemoveImage(t *testing.T) {
b, st, _ := newBuilder()
st.images["registry.felis.svc:5000/x:1"] = Image{ImageRef: "registry.felis.svc:5000/x:1"}
if err := b.RemoveImage(context.Background(), "registry.felis.svc:5000/x:1"); err != nil {
t.Fatalf("RemoveImage: %v", err)
}
if _, ok := st.images["registry.felis.svc:5000/x:1"]; ok {
t.Error("image not removed")
}
}
// ---- whitelist admission for the create-server form (spec §15) ----
func TestImageMatches(t *testing.T) {
cases := []struct {
ref, pattern string
want bool
}{
// exact match
{"registry.felis.svc:5000/mc:1.0", "registry.felis.svc:5000/mc:1.0", true},
// exact mismatch on tag
{"registry.felis.svc:5000/mc:1.0", "registry.felis.svc:5000/mc:2.0", false},
// tag wildcard admits any concrete tag on the repo
{"registry.felis.svc:5000/mc:1.0", "registry.felis.svc:5000/mc:*", true},
{"registry.felis.svc:5000/mc:anything", "registry.felis.svc:5000/mc:*", true},
// wildcard does not cross repos
{"registry.felis.svc:5000/other:1", "registry.felis.svc:5000/mc:*", false},
// a registry-port colon is not a tag separator: an untagged ref under a
// ported host has no tag, so a wildcard (which needs a non-empty tag) misses
{"registry.felis.svc:5000/mc", "registry.felis.svc:5000/mc:*", false},
// a non-wildcard pattern never matches via the prefix branch
{"registry.felis.svc:5000/mc:1", "registry.felis.svc:5000/mc", false},
}
for _, c := range cases {
if got := imageMatches(c.ref, c.pattern); got != c.want {
t.Errorf("imageMatches(%q, %q) = %v, want %v", c.ref, c.pattern, got, c.want)
}
}
}
func TestImageAdmitted(t *testing.T) {
b, st, _ := newBuilder()
st.images["registry.felis.svc:5000/exact:1"] = Image{ImageRef: "registry.felis.svc:5000/exact:1", Enabled: true}
st.images["registry.felis.svc:5000/wild:*"] = Image{ImageRef: "registry.felis.svc:5000/wild:*", Enabled: true}
st.images["registry.felis.svc:5000/off:1"] = Image{ImageRef: "registry.felis.svc:5000/off:1", Enabled: false}
cases := []struct {
ref string
want bool
}{
{"registry.felis.svc:5000/exact:1", true}, // exact, enabled
{"registry.felis.svc:5000/exact:2", false}, // wrong tag
{"registry.felis.svc:5000/wild:99", true}, // wildcard, enabled, port-colon safe
{"registry.felis.svc:5000/off:1", false}, // present but disabled
{"registry.felis.svc:5000/unknown:1", false}, // not on the list
{"", false}, // empty ref
{" ", false}, // blank ref
}
for _, c := range cases {
got, err := b.ImageAdmitted(context.Background(), c.ref)
if err != nil {
t.Fatalf("ImageAdmitted(%q): %v", c.ref, err)
}
if got != c.want {
t.Errorf("ImageAdmitted(%q) = %v, want %v", c.ref, got, c.want)
}
}
}
+284
View File
@@ -0,0 +1,284 @@
package build
import (
"fmt"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
networkingv1 "k8s.io/api/networking/v1"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/util/intstr"
)
// Label keys applied to build objects. ManagedBy doubles as the NetworkPolicy
// pod selector, so every build Pod is captured by the egress lock.
const (
LabelManagedBy = "app.kubernetes.io/managed-by"
LabelComponent = "app.kubernetes.io/component"
LabelBuildID = "felis.lolicon.best/build-id"
managedByValue = "felis-build"
componentValue = "image-build"
)
// Container names within the build Pod. Kaniko is the initContainer that builds
// and pushes the image — its log IS the "build log" an admin watches (spec §16);
// Trivy is the main container whose CRITICAL-CVE verdict gates admission and is
// surfaced via the build status, not the log stream. Exported so the build-log
// streamer (internal/api.K8sBuildLogStreamer, spec §416 日志流复用 §8) follows the
// same container this Job defines — one source of truth for the name.
const (
ContainerKaniko = "kaniko"
ContainerTrivy = "trivy"
)
// JobParams are the rendered inputs to a build Job. They are derived from a
// Build + Config by the Builder; jobspec is a pure function of them so the
// security-critical Job shape is unit-tested without a cluster.
type JobParams struct {
BuildID string
ImageRef string
ContextRef string
Namespace string
ServiceAccount string
RegistryURL string
KanikoImage string
TrivyImage string
Deadline time.Duration
CPULimit string
MemLimit string
}
// BuildJobName is the deterministic Job name for a build id.
func BuildJobName(buildID string) string { return "build-" + buildID }
func buildLabels(p JobParams) map[string]string {
return map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
LabelBuildID: p.BuildID,
}
}
// BuildJob renders the Kaniko+Trivy build Job (spec §16). Every isolation
// guarantee the spec demands is encoded here and asserted by jobspec_test.go,
// because no cluster runs in this environment:
//
// - runs in the isolated felis-build namespace with the weak felis-build SA
// (never the felis-api SA) and does NOT mount the SA token, so it cannot
// reach the K8s API (spec §16, §21);
// - no privileged container — Kaniko builds the Dockerfile without a daemon,
// so docker-in-docker / privileged is never needed (spec §16, §22);
// - activeDeadlineSeconds + backoffLimit=0 + per-container resource limits so
// a runaway or poisoned build cannot exhaust the cluster (spec §16);
// - the Trivy step runs with `--exit-code 1 --severity CRITICAL`, so a
// CRITICAL CVE fails the Pod and therefore the Job — the only retained
// automatic admission gate (spec §16).
//
// Sequencing: kaniko runs as an initContainer (build + push to the internal
// registry) and trivy as the main container (scan the pushed ref). The Pod
// succeeds only if kaniko pushed AND trivy found no CRITICAL CVE.
func BuildJob(p JobParams) (*batchv1.Job, error) {
limits, err := resourceLimits(p.CPULimit, p.MemLimit)
if err != nil {
return nil, err
}
deadline := int64(p.Deadline / time.Second)
if deadline <= 0 {
deadline = int64(defaultDeadline / time.Second)
}
// Hardened container security context shared by both build containers: no
// privilege, no privilege escalation, drop all capabilities. Kaniko needs a
// writable root filesystem to unpack layers, so we do not force read-only
// root here, but it gains no privilege.
sec := &corev1.SecurityContext{
Privileged: boolPtr(false),
AllowPrivilegeEscalation: boolPtr(false),
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
}
kaniko := corev1.Container{
Name: ContainerKaniko,
Image: p.KanikoImage,
Args: []string{
"--dockerfile=Dockerfile",
"--context=" + p.ContextRef,
"--destination=" + p.ImageRef,
// The internal registry is in-cluster only and may serve plain HTTP;
// it is never a public ingress (spec §17).
"--insecure",
"--skip-tls-verify",
},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: sec,
}
trivy := corev1.Container{
Name: ContainerTrivy,
Image: p.TrivyImage,
Args: []string{
"image",
"--exit-code", "1",
"--severity", "CRITICAL",
"--no-progress",
"--insecure",
p.ImageRef,
},
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
SecurityContext: sec,
}
job := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{
Name: BuildJobName(p.BuildID),
Namespace: p.Namespace,
Labels: buildLabels(p),
},
Spec: batchv1.JobSpec{
// A poisoned build must not loop — one shot, then a terminal verdict.
BackoffLimit: int32Ptr(0),
ActiveDeadlineSeconds: int64Ptr(deadline),
Template: corev1.PodTemplateSpec{
ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
ServiceAccountName: p.ServiceAccount,
AutomountServiceAccountToken: boolPtr(false),
InitContainers: []corev1.Container{kaniko},
Containers: []corev1.Container{trivy},
},
},
},
}
return job, nil
}
// NetPolParams parameterises the build-namespace egress lock.
type NetPolParams struct {
Namespace string
RegistryNamespace string
RegistryPort int32
// PackageSourceCIDRs is an optional, explicit allowlist of external package
// mirrors (spec §16: egress 仅 registry + 包源). Empty means the most
// locked-down default — no internet egress at all (默认拒外网).
PackageSourceCIDRs []string
}
// BuildNetworkPolicy renders the default-deny egress policy for build Pods
// (spec §16, §21: build ns egress 仅放 registry + 包源,默认拒外网). It selects
// build Pods by the managed-by label, denies all ingress, and allows egress
// only to DNS, the internal registry, and any explicitly configured package
// mirrors. There is deliberately no allow-all egress rule.
func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
port := p.RegistryPort
if port == 0 {
port = 5000
}
dnsUDP := corev1.ProtocolUDP
dnsTCP := corev1.ProtocolTCP
dns53 := intstr.FromInt32(53)
regPort := intstr.FromInt32(port)
egress := []networkingv1.NetworkPolicyEgressRule{
// DNS resolution: port-restricted to 53, so this is not an open-internet
// hole — name resolution only.
{
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsUDP, Port: &dns53},
{Protocol: &dnsTCP, Port: &dns53},
},
},
// The internal registry, selected by the namespace's immutable
// kubernetes.io/metadata.name label, on the registry port only.
{
To: []networkingv1.NetworkPolicyPeer{{
NamespaceSelector: &metav1.LabelSelector{
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.RegistryNamespace},
},
}},
Ports: []networkingv1.NetworkPolicyPort{
{Protocol: &dnsTCP, Port: &regPort},
},
},
}
// Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet.
for _, cidr := range p.PackageSourceCIDRs {
egress = append(egress, networkingv1.NetworkPolicyEgressRule{
To: []networkingv1.NetworkPolicyPeer{{
IPBlock: &networkingv1.IPBlock{CIDR: cidr},
}},
})
}
return &networkingv1.NetworkPolicy{
ObjectMeta: metav1.ObjectMeta{
Name: "felis-build-egress",
Namespace: p.Namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
Spec: networkingv1.NetworkPolicySpec{
PodSelector: metav1.LabelSelector{
MatchLabels: map[string]string{LabelManagedBy: managedByValue},
},
PolicyTypes: []networkingv1.PolicyType{
networkingv1.PolicyTypeIngress,
networkingv1.PolicyTypeEgress,
},
// Empty Ingress slice = deny all ingress: nothing connects to a
// build Pod.
Ingress: []networkingv1.NetworkPolicyIngressRule{},
Egress: egress,
},
}
}
// BuildServiceAccount renders the weak build SA (spec §16, §21). It is the most
// dangerous identity in the platform if mis-scoped, so it is created bare: no
// secrets, token auto-mounting disabled, and — by virtue of having no Role or
// RoleBinding anywhere — zero K8s API permissions. Its only capability is
// network reachability to push to the registry, which RBAC does not grant.
func BuildServiceAccount(namespace, name string) *corev1.ServiceAccount {
return &corev1.ServiceAccount{
ObjectMeta: metav1.ObjectMeta{
Name: name,
Namespace: namespace,
Labels: map[string]string{
LabelManagedBy: managedByValue,
LabelComponent: componentValue,
},
},
AutomountServiceAccountToken: boolPtr(false),
}
}
// resourceLimits parses the CPU/memory limits into a ResourceList.
func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
if cpu == "" {
cpu = defaultCPULimit
}
if mem == "" {
mem = defaultMemLimit
}
cpuQty, err := resource.ParseQuantity(cpu)
if err != nil {
return nil, fmt.Errorf("build: invalid cpu limit %q: %w", cpu, err)
}
memQty, err := resource.ParseQuantity(mem)
if err != nil {
return nil, fmt.Errorf("build: invalid memory limit %q: %w", mem, err)
}
return corev1.ResourceList{
corev1.ResourceCPU: cpuQty,
corev1.ResourceMemory: memQty,
}, nil
}
func boolPtr(b bool) *bool { return &b }
func int32Ptr(i int32) *int32 { return &i }
func int64Ptr(i int64) *int64 { return &i }
+306
View File
@@ -0,0 +1,306 @@
package build
import (
"strings"
"testing"
"time"
corev1 "k8s.io/api/core/v1"
networkingv1 "k8s.io/api/networking/v1"
)
func sampleJobParams() JobParams {
return JobParams{
BuildID: "bld-1",
ImageRef: "registry.felis.svc:5000/mc-paper:1.0",
ContextRef: "tar://contexts/abc.tar.gz",
Namespace: defaultNamespace,
ServiceAccount: defaultServiceAccount,
RegistryURL: "registry.felis.svc:5000",
KanikoImage: defaultKanikoImage,
TrivyImage: defaultTrivyImage,
Deadline: 30 * time.Minute,
CPULimit: "2",
MemLimit: "4Gi",
}
}
// The Job must run in the isolated build namespace under the weak build SA —
// never the api namespace/identity, and never the minecraft namespace. This is
// the central §16/§21 red line, asserted on the rendered spec because no cluster
// runs here.
func TestBuildJobRunsIsolatedUnderWeakSA(t *testing.T) {
job, err := BuildJob(sampleJobParams())
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
if job.Namespace != "felis-build" {
t.Errorf("namespace = %q, want felis-build (never minecraft/felis-system)", job.Namespace)
}
sa := job.Spec.Template.Spec.ServiceAccountName
if sa != "felis-build" {
t.Errorf("service account = %q, want felis-build (never felis-api)", sa)
}
if sa == "felis-api" {
t.Fatal("build Pod must NOT run as the felis-api SA")
}
// The SA token must not be mounted: with no token, the Pod cannot reach the
// K8s API even if a Role were mis-bound.
if amt := job.Spec.Template.Spec.AutomountServiceAccountToken; amt == nil || *amt {
t.Error("AutomountServiceAccountToken must be explicitly false")
}
}
// A poisoned build must terminate and not loop or run unbounded.
func TestBuildJobIsBoundedAndOneShot(t *testing.T) {
job, err := BuildJob(sampleJobParams())
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
if job.Spec.BackoffLimit == nil || *job.Spec.BackoffLimit != 0 {
t.Error("BackoffLimit must be 0 — a poisoned build must not retry")
}
if job.Spec.ActiveDeadlineSeconds == nil || *job.Spec.ActiveDeadlineSeconds <= 0 {
t.Error("ActiveDeadlineSeconds must be a positive wall-clock cap")
}
if got := *job.Spec.ActiveDeadlineSeconds; got != int64((30 * time.Minute).Seconds()) {
t.Errorf("ActiveDeadlineSeconds = %d, want 1800", got)
}
if job.Spec.Template.Spec.RestartPolicy != corev1.RestartPolicyNever {
t.Error("RestartPolicy must be Never")
}
}
// No build container may be privileged or able to escalate, and all containers
// must carry resource limits.
func TestBuildJobContainersAreHardened(t *testing.T) {
job, err := BuildJob(sampleJobParams())
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
all := append([]corev1.Container{}, job.Spec.Template.Spec.InitContainers...)
all = append(all, job.Spec.Template.Spec.Containers...)
if len(all) < 2 {
t.Fatalf("expected kaniko initContainer + trivy container, got %d containers", len(all))
}
for _, c := range all {
sc := c.SecurityContext
if sc == nil {
t.Fatalf("container %q has no security context", c.Name)
}
if sc.Privileged == nil || *sc.Privileged {
t.Errorf("container %q must not be privileged (Kaniko needs no daemon)", c.Name)
}
if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
t.Errorf("container %q must set allowPrivilegeEscalation=false", c.Name)
}
if sc.Capabilities == nil || len(sc.Capabilities.Drop) == 0 {
t.Errorf("container %q must drop capabilities", c.Name)
} else if string(sc.Capabilities.Drop[0]) != "ALL" {
t.Errorf("container %q must drop ALL capabilities, got %v", c.Name, sc.Capabilities.Drop)
}
if c.Resources.Limits.Cpu().IsZero() || c.Resources.Limits.Memory().IsZero() {
t.Errorf("container %q must carry CPU+memory limits", c.Name)
}
}
}
// kaniko builds and pushes to the request's exact target; trivy gates admission
// with --exit-code 1 --severity CRITICAL on that same ref.
func TestBuildJobKanikoPushesAndTrivyGates(t *testing.T) {
p := sampleJobParams()
job, err := BuildJob(p)
if err != nil {
t.Fatalf("BuildJob: %v", err)
}
if len(job.Spec.Template.Spec.InitContainers) != 1 {
t.Fatalf("expected exactly one (kaniko) initContainer")
}
kaniko := job.Spec.Template.Spec.InitContainers[0]
if kaniko.Name != "kaniko" {
t.Errorf("init container = %q, want kaniko", kaniko.Name)
}
if !hasArg(kaniko.Args, "--destination="+p.ImageRef) {
t.Errorf("kaniko must push to %q, args=%v", p.ImageRef, kaniko.Args)
}
if len(job.Spec.Template.Spec.Containers) != 1 {
t.Fatalf("expected exactly one (trivy) main container")
}
trivy := job.Spec.Template.Spec.Containers[0]
if trivy.Name != "trivy" {
t.Errorf("main container = %q, want trivy", trivy.Name)
}
// The scan gate: a CRITICAL CVE must fail the Pod (and thus the Job).
if !argPairPresent(trivy.Args, "--exit-code", "1") {
t.Errorf("trivy must run with --exit-code 1, args=%v", trivy.Args)
}
if !argPairPresent(trivy.Args, "--severity", "CRITICAL") {
t.Errorf("trivy must gate on --severity CRITICAL, args=%v", trivy.Args)
}
if !hasArg(trivy.Args, p.ImageRef) {
t.Errorf("trivy must scan the pushed ref %q, args=%v", p.ImageRef, trivy.Args)
}
}
// The build namespace egress lock must be default-deny: deny all ingress, and
// allow egress only to DNS + the internal registry — never an allow-all rule.
func TestBuildNetworkPolicyIsDefaultDeny(t *testing.T) {
np := BuildNetworkPolicy(NetPolParams{
Namespace: "felis-build",
RegistryNamespace: "felis-system",
RegistryPort: 5000,
})
if !hasPolicyType(np, networkingv1.PolicyTypeEgress) || !hasPolicyType(np, networkingv1.PolicyTypeIngress) {
t.Fatal("policy must govern both Ingress and Egress")
}
// Ingress: empty rule slice = deny all.
if len(np.Spec.Ingress) != 0 {
t.Errorf("ingress must be empty (deny all), got %d rules", len(np.Spec.Ingress))
}
// The selector must capture every build Pod by the managed-by label.
if np.Spec.PodSelector.MatchLabels[LabelManagedBy] != managedByValue {
t.Errorf("pod selector must match managed-by=%s", managedByValue)
}
// No egress rule may be an allow-all (a rule with neither To peers nor Ports
// would permit unrestricted egress).
for i, rule := range np.Spec.Egress {
if len(rule.To) == 0 && len(rule.Ports) == 0 {
t.Errorf("egress rule %d is allow-all (no To, no Ports) — internet would be open", i)
}
}
// The registry must be reachable (by namespace selector), and DNS allowed.
if !egressAllowsNamespace(np, "felis-system") {
t.Error("egress must allow the registry namespace")
}
if !egressAllowsPort(np, 53) {
t.Error("egress must allow DNS (port 53)")
}
}
// With no package-source CIDRs configured, there must be zero IPBlock egress —
// the most locked-down default (no open internet).
func TestBuildNetworkPolicyDefaultsToNoInternet(t *testing.T) {
np := BuildNetworkPolicy(NetPolParams{
Namespace: "felis-build",
RegistryNamespace: "felis-system",
})
for i, rule := range np.Spec.Egress {
for _, peer := range rule.To {
if peer.IPBlock != nil {
t.Errorf("egress rule %d has an IPBlock but no package sources were configured", i)
}
}
}
}
func TestBuildNetworkPolicyAllowsConfiguredPackageMirrors(t *testing.T) {
cidr := "192.0.2.0/24"
np := BuildNetworkPolicy(NetPolParams{
Namespace: "felis-build",
RegistryNamespace: "felis-system",
PackageSourceCIDRs: []string{cidr},
})
found := false
for _, rule := range np.Spec.Egress {
for _, peer := range rule.To {
if peer.IPBlock != nil && peer.IPBlock.CIDR == cidr {
found = true
}
}
}
if !found {
t.Errorf("configured package-mirror CIDR %q not present in egress", cidr)
}
}
// The build SA is the most dangerous identity if mis-scoped: it must be bare —
// no secrets, token automount disabled, and (by having no Role anywhere) no API
// rights. We assert the spec-level properties the renderer controls.
func TestBuildServiceAccountIsBare(t *testing.T) {
sa := BuildServiceAccount("felis-build", "felis-build")
if sa.Namespace != "felis-build" {
t.Errorf("SA namespace = %q, want felis-build", sa.Namespace)
}
if sa.AutomountServiceAccountToken == nil || *sa.AutomountServiceAccountToken {
t.Error("SA must disable token automounting")
}
if len(sa.Secrets) != 0 {
t.Errorf("SA must carry no secrets, got %d", len(sa.Secrets))
}
if len(sa.ImagePullSecrets) != 0 {
t.Errorf("SA must carry no image-pull secrets, got %d", len(sa.ImagePullSecrets))
}
}
// An invalid resource limit must surface as an error rather than render a Job
// with no limits.
func TestBuildJobRejectsBadResourceLimit(t *testing.T) {
p := sampleJobParams()
p.CPULimit = "not-a-quantity"
if _, err := BuildJob(p); err == nil {
t.Error("expected error for an unparseable CPU limit")
}
}
// ---- helpers ----
func hasArg(args []string, want string) bool {
for _, a := range args {
if a == want {
return true
}
}
return false
}
// argPairPresent reports whether flag is immediately followed by val (the
// `--exit-code 1` two-token form).
func argPairPresent(args []string, flag, val string) bool {
for i := 0; i < len(args)-1; i++ {
if args[i] == flag && args[i+1] == val {
return true
}
}
return false
}
func hasPolicyType(np *networkingv1.NetworkPolicy, t networkingv1.PolicyType) bool {
for _, pt := range np.Spec.PolicyTypes {
if pt == t {
return true
}
}
return false
}
func egressAllowsNamespace(np *networkingv1.NetworkPolicy, ns string) bool {
for _, rule := range np.Spec.Egress {
for _, peer := range rule.To {
if peer.NamespaceSelector != nil &&
peer.NamespaceSelector.MatchLabels["kubernetes.io/metadata.name"] == ns {
return true
}
}
}
return false
}
func egressAllowsPort(np *networkingv1.NetworkPolicy, port int32) bool {
for _, rule := range np.Spec.Egress {
for _, p := range rule.Ports {
if p.Port != nil && p.Port.IntVal == port {
return true
}
}
}
return false
}
// guard against accidental shorthand: the test image ref must be host-qualified.
func TestSampleRefIsHostQualified(t *testing.T) {
if !strings.Contains(sampleJobParams().ImageRef, "/") {
t.Fatal("sample image ref must be host-qualified")
}
}
+87
View File
@@ -0,0 +1,87 @@
package build
import (
"context"
batchv1 "k8s.io/api/batch/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// K8sJobs is the production Jobs backed by a controller-runtime client (spec
// §16). It creates the Kaniko+Trivy build Job, reads its phase for the scan-gate
// translation, and deletes it on cancel — nothing more. The cluster-bootstrap
// objects (the felis-build namespace, the weak SA, and the egress
// NetworkPolicy) are installed once by the deployment manifests (spec §21), not
// per build, so this binding never needs to create them. It is integration-
// tested against a live cluster, not the hermetic build_test.go suite.
type K8sJobs struct {
c client.Client
cfg Config
}
// NewK8sJobs builds a Jobs over c using cfg for the namespace and image refs.
func NewK8sJobs(c client.Client, cfg Config) *K8sJobs {
return &K8sJobs{c: c, cfg: cfg.withDefaults()}
}
// CreateBuildJob renders and applies the build Job, returning its name. The Job
// is the security-critical object; its shape is fixed by BuildJob (jobspec.go)
// and asserted by jobspec_test.go.
func (k *K8sJobs) CreateBuildJob(ctx context.Context, p JobParams) (string, error) {
job, err := BuildJob(p)
if err != nil {
return "", err
}
if err := k.c.Create(ctx, job); err != nil {
return "", err
}
return job.Name, nil
}
// JobPhase reads the Job and maps its status to a JobPhase. A missing Job
// (GC'd, never created) is JobUnknown, which Sync treats as failed. The mapping
// is deliberately conservative: a Job is Succeeded only when the Complete
// condition is true, so a half-finished Job is never admitted.
func (k *K8sJobs) JobPhase(ctx context.Context, jobName string) (JobPhase, error) {
var job batchv1.Job
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.cfg.Namespace, Name: jobName}, &job); err != nil {
if apierrors.IsNotFound(err) {
return JobUnknown, nil
}
return JobUnknown, err
}
for _, cond := range job.Status.Conditions {
if cond.Status != "True" {
continue
}
switch cond.Type {
case batchv1.JobComplete:
return JobSucceeded, nil
case batchv1.JobFailed:
// Covers a CRITICAL CVE (trivy --exit-code 1), a kaniko failure, and
// DeadlineExceeded — all are a rejected build.
return JobFailed, nil
}
}
if job.Status.Active > 0 {
return JobRunning, nil
}
return JobPending, nil
}
// CancelBuildJob deletes the Job and, via background propagation, its pods. A
// missing Job is not an error: cancellation is idempotent.
func (k *K8sJobs) CancelBuildJob(ctx context.Context, jobName string) error {
bg := metav1.DeletePropagationBackground
obj := &batchv1.Job{
ObjectMeta: metav1.ObjectMeta{Namespace: k.cfg.Namespace, Name: jobName},
}
if err := k.c.Delete(ctx, obj, &client.DeleteOptions{PropagationPolicy: &bg}); err != nil &&
!apierrors.IsNotFound(err) {
return err
}
return nil
}
+186
View File
@@ -0,0 +1,186 @@
package build
import (
"context"
"database/sql"
"time"
)
// PGStore is the production Store backed by Postgres (spec §6, §16). It writes
// the two tables of the build subsystem — image_builds and image_whitelist —
// and is the *only* component that holds database credentials: the build Pod
// never does (the weak-SA red line). The SQL here is exercised by integration
// tests against a live database, not the hermetic build_test.go suite. Every
// statement is a narrow operation; there is no generic UPDATE escape hatch.
type PGStore struct {
db *sql.DB
}
// NewPGStore wraps an existing pool (from store.PostgresDriver.DB()).
func NewPGStore(db *sql.DB) *PGStore { return &PGStore{db: db} }
func (s *PGStore) CreateBuild(ctx context.Context, b *Build) error {
const q = `INSERT INTO image_builds
(id, image_ref, status, dockerfile, context_ref, base_image, requested_by, created_at)
VALUES ($1, $2, $3, $4, NULLIF($5, ''), NULLIF($6, ''), $7, $8)`
_, err := s.db.ExecContext(ctx, q,
b.ID, b.ImageRef, string(b.Status), b.Dockerfile, b.ContextRef, b.BaseImage,
b.RequestedBy, b.CreatedAt)
return err
}
func (s *PGStore) GetBuild(ctx context.Context, id string) (*Build, error) {
const q = `SELECT id, image_ref, status, dockerfile, context_ref, base_image,
requested_by, job_name, log_ref, error, created_at, finished_at
FROM image_builds WHERE id = $1`
return s.scanBuild(s.db.QueryRowContext(ctx, q, id))
}
func (s *PGStore) scanBuild(row *sql.Row) (*Build, error) {
var (
b Build
status string
ctxRef, base, jobName, logRef, eMsg sql.NullString
finished sql.NullTime
)
switch err := row.Scan(&b.ID, &b.ImageRef, &status, &b.Dockerfile, &ctxRef, &base,
&b.RequestedBy, &jobName, &logRef, &eMsg, &b.CreatedAt, &finished); {
case err == sql.ErrNoRows:
return nil, ErrNotFound
case err != nil:
return nil, err
}
b.Status = Status(status)
b.ContextRef = ctxRef.String
b.BaseImage = base.String
b.JobName = jobName.String
b.LogRef = logRef.String
b.Error = eMsg.String
if finished.Valid {
t := finished.Time
b.FinishedAt = &t
}
return &b, nil
}
func (s *PGStore) SetBuildJob(ctx context.Context, id, jobName string) error {
const q = `UPDATE image_builds SET job_name = $2, status = 'building'
WHERE id = $1 AND status = 'pending'`
res, err := s.db.ExecContext(ctx, q, id, jobName)
if err != nil {
return err
}
if n, _ := res.RowsAffected(); n == 0 {
return ErrNotFound
}
return nil
}
func (s *PGStore) FinishBuild(ctx context.Context, id string, status Status, errMsg string, at time.Time) error {
const q = `UPDATE image_builds SET status = $2, error = NULLIF($3, ''), finished_at = $4
WHERE id = $1`
res, err := s.db.ExecContext(ctx, q, id, string(status), errMsg, at)
if err != nil {
return err
}
if n, _ := res.RowsAffected(); n == 0 {
return ErrNotFound
}
return nil
}
func (s *PGStore) ListUnfinishedBuilds(ctx context.Context) ([]Build, error) {
const q = `SELECT id, image_ref, status, dockerfile, context_ref, base_image,
requested_by, job_name, log_ref, error, created_at, finished_at
FROM image_builds WHERE status IN ('pending', 'building') ORDER BY created_at ASC`
rows, err := s.db.QueryContext(ctx, q)
if err != nil {
return nil, err
}
defer rows.Close()
var out []Build
for rows.Next() {
var (
b Build
status string
ctxRef, base, jobName, logRef, eMsg sql.NullString
finished sql.NullTime
)
if err := rows.Scan(&b.ID, &b.ImageRef, &status, &b.Dockerfile, &ctxRef, &base,
&b.RequestedBy, &jobName, &logRef, &eMsg, &b.CreatedAt, &finished); err != nil {
return nil, err
}
b.Status = Status(status)
b.ContextRef = ctxRef.String
b.BaseImage = base.String
b.JobName = jobName.String
b.LogRef = logRef.String
b.Error = eMsg.String
if finished.Valid {
t := finished.Time
b.FinishedAt = &t
}
out = append(out, b)
}
return out, rows.Err()
}
// AdmitBuiltImage upserts the whitelist row on scan-gate success. ON CONFLICT
// re-enables and re-stamps a previously-removed or superseded ref, so a rebuild
// of the same tag re-admits it (spec §16).
func (s *PGStore) AdmitBuiltImage(ctx context.Context, img Image) error {
const q = `INSERT INTO image_whitelist
(image_ref, source, build_id, added_by, enabled, added_at)
VALUES ($1, 'built', NULLIF($2, ''), $3, true, $4)
ON CONFLICT (image_ref) DO UPDATE
SET source = 'built', build_id = EXCLUDED.build_id, added_by = EXCLUDED.added_by,
enabled = true, added_at = EXCLUDED.added_at`
_, err := s.db.ExecContext(ctx, q, img.ImageRef, img.BuildID, img.AddedBy, img.AddedAt)
return err
}
func (s *PGStore) ListImages(ctx context.Context) ([]Image, error) {
const q = `SELECT image_ref, source, build_id, added_by, enabled, added_at
FROM image_whitelist ORDER BY added_at DESC`
rows, err := s.db.QueryContext(ctx, q)
if err != nil {
return nil, err
}
defer rows.Close()
var out []Image
for rows.Next() {
var (
img Image
buildID sql.NullString
)
if err := rows.Scan(&img.ImageRef, &img.Source, &buildID, &img.AddedBy,
&img.Enabled, &img.AddedAt); err != nil {
return nil, err
}
img.BuildID = buildID.String
out = append(out, img)
}
return out, rows.Err()
}
func (s *PGStore) AddExternalImage(ctx context.Context, img Image) error {
const q = `INSERT INTO image_whitelist
(image_ref, source, added_by, enabled, added_at)
VALUES ($1, 'external', $2, true, $3)
ON CONFLICT (image_ref) DO UPDATE
SET source = 'external', build_id = NULL, added_by = EXCLUDED.added_by,
enabled = true, added_at = EXCLUDED.added_at`
_, err := s.db.ExecContext(ctx, q, img.ImageRef, img.AddedBy, img.AddedAt)
return err
}
func (s *PGStore) RemoveImage(ctx context.Context, imageRef string) error {
res, err := s.db.ExecContext(ctx, `DELETE FROM image_whitelist WHERE image_ref = $1`, imageRef)
if err != nil {
return err
}
if n, _ := res.RowsAffected(); n == 0 {
return ErrNotFound
}
return nil
}
+150
View File
@@ -0,0 +1,150 @@
package build
import (
"fmt"
"regexp"
"strings"
)
// invalidf builds a validation error wrapping ErrInvalid so the API layer maps
// every malformed-request case to a single 400 path.
func invalidf(format string, a ...any) error {
return fmt.Errorf("%w: "+format, append([]any{ErrInvalid}, a...)...)
}
// imageNameRE matches the path+tag of an image reference under the registry
// host, e.g. "foo/bar:1.0" or "mc-paper:latest". It is intentionally strict:
// lowercase path segments, an optional tag of the same alphabet, no digests, no
// shell metacharacters that could escape into the Kaniko/Trivy argv.
var imageNameRE = regexp.MustCompile(`^[a-z0-9]([a-z0-9._/-]*[a-z0-9])?(:[a-zA-Z0-9._-]+)?$`)
// Validate enforces the §16 admission rules on a build request: the target must
// address the internal registry, the Dockerfile must be present and within the
// size cap, and a context reference is required (Kaniko pulls it, spec §17).
func Validate(req Request, cfg Config) error {
cfg = cfg.withDefaults()
if err := validateRegistryTarget(req.ImageRef, cfg.RegistryURL); err != nil {
return err
}
if strings.TrimSpace(req.Dockerfile) == "" {
return invalidf("dockerfile is required")
}
if len(req.Dockerfile) > cfg.MaxDockerfileBytes {
return invalidf("dockerfile exceeds %d bytes", cfg.MaxDockerfileBytes)
}
if strings.TrimSpace(req.ContextRef) == "" {
return invalidf("context reference is required")
}
return nil
}
// ValidateImageRef checks a bare image reference (used by external admission,
// where there is no registry-target constraint beyond well-formedness). A
// whitelist entry may be a tag wildcard ("registry/foo:*", a legitimate
// image_whitelist value per spec §15): the trailing ":*" is stripped before the
// path is validated. This is safe because a wildcard is only ever compared
// against concrete refs by imageMatches — it never reaches a Kaniko/Trivy argv,
// unlike a build push target (validateRegistryTarget stays strictly concrete).
func ValidateImageRef(ref string) error {
if ref == "" {
return invalidf("image reference is required")
}
host, rest, ok := splitRegistryHost(ref)
if !ok {
return invalidf("image reference %q must be host-qualified (host/path:tag)", ref)
}
if repo, isWildcard := strings.CutSuffix(rest, ":*"); isWildcard {
rest = repo
}
if !imageNameRE.MatchString(rest) {
return invalidf("invalid image path/tag %q", rest)
}
_ = host
return nil
}
// splitTag splits an image reference's path from its tag. It is registry-port
// safe: only a colon *after* the final path separator is a tag separator, so
// "registry:5000/foo" splits to ("registry:5000/foo", ""), never a bogus tag.
func splitTag(ref string) (repo, tag string) {
slash := strings.LastIndexByte(ref, '/')
colon := strings.LastIndexByte(ref, ':')
if colon > slash {
return ref[:colon], ref[colon+1:]
}
return ref, ""
}
// imageMatches reports whether a concrete image ref is admitted by a whitelist
// pattern. A pattern is either exact ("registry/foo:1.0") or a tag wildcard
// ("registry/foo:*", spec §15) that matches any non-empty tag on the same repo.
// A wildcard never matches an untagged ref — admission is always to a concrete
// tag.
func imageMatches(ref, pattern string) bool {
if ref == pattern {
return true
}
prefix, ok := strings.CutSuffix(pattern, ":*")
if !ok {
return false
}
repo, tag := splitTag(ref)
return repo == prefix && tag != ""
}
// validateRegistryTarget enforces that a build pushes only to the configured
// internal registry — never an arbitrary external host (spec §16: the build can
// never push elsewhere; the registry is not a public ingress). When RegistryURL
// is unset (tests / not configured) the host constraint is skipped but the
// path/tag are still validated.
func validateRegistryTarget(ref, registryURL string) error {
if ref == "" {
return invalidf("image reference is required")
}
host, rest, ok := splitRegistryHost(ref)
if !ok {
return invalidf("image reference %q must target the internal registry (host/path:tag)", ref)
}
if !imageNameRE.MatchString(rest) {
return invalidf("invalid image path/tag %q", rest)
}
if registryURL != "" && host != registryHost(registryURL) {
return invalidf("image reference %q must target the internal registry %q, not %q",
ref, registryHost(registryURL), host)
}
return nil
}
// splitRegistryHost separates the registry host from the remaining path+tag. A
// reference is host-qualified only if the first segment looks like a registry
// host — it contains a '.' or ':' (port), matching containerd's heuristic.
// "foo/bar:1" (Docker Hub shorthand) is rejected: builds must be explicit about
// the internal registry.
func splitRegistryHost(ref string) (host, rest string, ok bool) {
slash := strings.IndexByte(ref, '/')
if slash < 0 {
return "", "", false
}
host = ref[:slash]
rest = ref[slash+1:]
if !strings.ContainsAny(host, ".:") {
return "", "", false
}
if rest == "" {
return "", "", false
}
return host, rest, true
}
// registryHost strips any scheme and path from a configured registry URL,
// leaving the host[:port] that an image reference must match.
func registryHost(registryURL string) string {
h := registryURL
if i := strings.Index(h, "://"); i >= 0 {
h = h[i+3:]
}
if i := strings.IndexByte(h, '/'); i >= 0 {
h = h[:i]
}
return h
}