Unverified Commit e836a73a authored by Lemon-miaow's avatar Lemon-miaow
Browse files

fix(build): 构建并发上限与排队,构建命名空间加 ResourceQuota,SyncAll 逐个容错

parent e521cf59
Loading
Loading
Loading
Loading
+1 −0
Changes for cmd/felis/api.go: 1 added line, 0 removed lines.
Original line number Diff line number Diff line
@@ -465,6 +465,7 @@ func buildConfig(cfg *config.Config) build.Config {
		UserNamespaces:      cfg.Registry.BuildUserNamespaces,
		UserNamespacesProbe: new(atomic.Bool),
		RuntimeClass:        cfg.Registry.BuildRuntimeClass,
		MaxConcurrent:       cfg.Registry.MaxConcurrentBuilds,
		// Empty keeps Trivy's own default; an install with builds points this at
		// the internal DB mirror (see config.RegistryConfig.TrivyDBRepository).
		TrivyDBRepository:     cfg.Registry.TrivyDBRepository,
+1 −1
Changes for deploy/bootstrap.sh: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -2352,7 +2352,7 @@ persisted_registry_block() {
    out="$(awk '
      /^[[:space:]]*\[/ { sect = $0; next }
      sect ~ /^[[:space:]]*\[registry\][[:space:]]*$/ &&
        /^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|user_uploads_context)[[:space:]]*=/ { print }
        /^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|max_concurrent_builds|user_uploads_context)[[:space:]]*=/ { print }
      sect ~ /^[[:space:]]*\[registry\.s3\][[:space:]]*$/ && /^[[:space:]]*[A-Za-z_]+[[:space:]]*=/ {
        if (!s3hdr) { printf "[registry.s3]\n"; s3hdr = 1 }
        print
+3 −1
Changes for docs/troubleshooting.md: 3 added lines, 1 removed line.
Original line number Diff line number Diff line
@@ -493,6 +493,7 @@ build_mem_limit = "4Gi"
build_disk_limit = "12Gi"          # §8f
build_user_namespaces = "auto"     # §8f: auto | on | off
build_runtime_class = ""           # §8f: e.g. "gvisor"
max_concurrent_builds = 2          # §8f: 1-6; later builds queue
```

Mirror the executor images into the registry once. On the node itself, push
@@ -582,7 +583,8 @@ approved-but-hostile Dockerfile and the node is the pod around it:
| Sandbox runtime | optional `build_runtime_class` (gVisor, Kata) | below |
| Credentials | the registry credential lives only in the `push` container; the service token only in `context-fetch` | jobspec |
| Resources | CPU, memory and ephemeral-storage limits per container; `activeDeadlineSeconds`; the context extraction stops at 4 GiB or 200 000 entries | jobspec, `felis fetch-context` |
| Namespace backstop | `felis-build-limits` LimitRange gives any container without limits 1 CPU / 1 GiB / 1 GiB disk | bundle |
| Namespace backstop | `felis-build-limits` LimitRange gives any container without limits 1 CPU / 1 GiB / 1 GiB disk; `felis-build-quota` allows 8 running pods and no PVCs | bundle |
| Concurrency | at most `[registry] max_concurrent_builds` (default 2, at most 6) builds run; later ones wait as `pending` (Queued) and start oldest first | `build.Builder` |

**Reviewed bytes.** Every upload records the sha256 of the archive, and the
review page shows it. The context download carries the same value in the
+117 −12
Changes for internal/build/build.go: 117 added lines, 12 removed lines.
Original line number Diff line number Diff line
@@ -37,6 +37,7 @@ import (
	"errors"
	"fmt"
	"strings"
	"sync"
	"sync/atomic"
	"time"

@@ -267,6 +268,11 @@ type Config struct {
	// when set. The class must exist on the cluster.
	RuntimeClass string

	// MaxConcurrent caps how many build Jobs run at once. A build submitted past
	// the cap stays pending, and SyncAll starts queued builds oldest first as
	// running ones finish. Zero applies the default; MaxConcurrentLimit bounds it.
	MaxConcurrent int

	// ContextOrigin is the scheme://host[:port] of the platform's internal API
	// face, the only host an http(s) ContextRef may name: the fetch step presents
	// the service token to it. Empty refuses every http(s) context.
@@ -302,8 +308,14 @@ const (
	defaultCPULimit       = "2"
	defaultMemLimit       = "4Gi"
	defaultDiskLimit      = "12Gi"
	defaultMaxConcurrent  = 2
)

// MaxConcurrentLimit is the highest MaxConcurrent the platform accepts. The
// build namespace's pod quota (BuildResourceQuota) leaves room for exactly this
// many build pods plus the user-namespace probe.
const MaxConcurrentLimit = 6

// withDefaults returns a copy of c with zero fields filled, so a partially
// configured Config (or the zero value, in tests) is always usable.
func (c Config) withDefaults() Config {
@@ -337,16 +349,25 @@ func (c Config) withDefaults() Config {
	if c.UserNamespaces == "" {
		c.UserNamespaces = UserNamespacesAuto
	}
	if c.MaxConcurrent <= 0 {
		c.MaxConcurrent = defaultMaxConcurrent
	}
	if c.MaxConcurrent > MaxConcurrentLimit {
		c.MaxConcurrent = MaxConcurrentLimit
	}
	return c
}

// Builder orchestrates the build subsystem. It holds no mutable state; the
// clock and id generator are injectable for hermetic tests.
// Builder orchestrates the build subsystem. The clock and id generator are
// injectable for hermetic tests. Its only state is the lock that serializes
// starting Jobs, so two submissions cannot both take the last free slot.
type Builder struct {
	Store  Store
	Jobs   Jobs
	Config Config

	startMu sync.Mutex

	// Now is the clock, injectable for tests. Defaults to time.Now.
	Now func() time.Time
	// IDGen mints build ids. Defaults to a time-based generator.
@@ -368,10 +389,11 @@ func (b *Builder) newID() string {
}

// Submit validates req, records a pending build, and starts the Kaniko+Trivy
// Job (spec §16). The build is returned in the building state once the Job is
// created; if Job creation fails the build is marked failed so it never lingers
// pending. The caller (felis-api) drives the build to a terminal state by
// polling Sync / SyncAll.
// Job (spec §16) when fewer than MaxConcurrent builds are running. Past the cap
// the build is returned pending and waits in the queue SyncAll drains. A started
// build is returned building; if Job creation fails it is marked failed so it
// never lingers pending. The caller (felis-api) drives the build to a terminal
// state by polling Sync / SyncAll.
func (b *Builder) Submit(ctx context.Context, req Request) (*Build, error) {
	cfg := b.Config.withDefaults()
	if err := Validate(req, cfg); err != nil {
@@ -394,6 +416,34 @@ func (b *Builder) Submit(ctx context.Context, req Request) (*Build, error) {
		return nil, err
	}

	b.startMu.Lock()
	defer b.startMu.Unlock()
	running, err := b.running(ctx)
	if err != nil || running >= cfg.MaxConcurrent {
		// Queued. When the count could not be read, SyncAll retries the start.
		return bld, nil
	}
	return b.start(ctx, bld, cfg)
}

// running counts builds whose Job has been started and not yet reconciled to a
// terminal state.
func (b *Builder) running(ctx context.Context) (int, error) {
	builds, err := b.Store.ListUnfinishedBuilds(ctx)
	if err != nil {
		return 0, err
	}
	n := 0
	for i := range builds {
		if builds[i].Status == StatusBuilding {
			n++
		}
	}
	return n, nil
}

// start creates bld's Job and records it. The caller holds startMu.
func (b *Builder) start(ctx context.Context, bld *Build, cfg Config) (*Build, error) {
	jobName, err := b.Jobs.CreateBuildJob(ctx, b.jobParams(bld, cfg))
	if err != nil {
		// The pending row exists; mark it failed so it is not reconciled forever.
@@ -466,8 +516,12 @@ func (b *Builder) Sync(ctx context.Context, id string) (*Build, error) {
		return bld, nil
	}
	if bld.JobName == "" {
		// Created but the Job name was never recorded; treat as failed rather
		// than reconcile forever against a phantom Job.
		if bld.Status == StatusPending {
			// Queued behind MaxConcurrent; SyncAll starts it.
			return bld, nil
		}
		// Building with no Job name recorded; treat as failed rather than
		// reconcile forever against a phantom Job.
		return b.finish(ctx, bld, StatusFailed, "no build job recorded")
	}

@@ -499,25 +553,76 @@ func (b *Builder) Sync(ctx context.Context, id string) (*Build, error) {
	}
}

// SyncAll reconciles every unfinished build and returns the count advanced to a
// SyncAll reconciles every running build, then starts queued builds oldest
// first while fewer than MaxConcurrent run. It returns the count advanced to a
// terminal state. felis-api calls this periodically (spec §16: the scan gate is
// observed, not pushed by the build Pod).
// observed, not pushed by the build Pod). One build's error does not stop the
// rest; every error comes back joined.
func (b *Builder) SyncAll(ctx context.Context) (int, error) {
	builds, err := b.Store.ListUnfinishedBuilds(ctx)
	if err != nil {
		return 0, err
	}
	advanced := 0
	var errs []error
	for i := range builds {
		if builds[i].Status == StatusPending && builds[i].JobName == "" {
			continue // queued; startQueued below
		}
		bld, err := b.Sync(ctx, builds[i].ID)
		if err != nil {
			return advanced, err
			errs = append(errs, fmt.Errorf("build %s: %w", builds[i].ID, err))
			continue
		}
		if bld.Status.terminal() {
			advanced++
		}
	}
	return advanced, nil
	started, err := b.startQueued(ctx)
	if err != nil {
		errs = append(errs, err)
	}
	advanced += started
	return advanced, errors.Join(errs...)
}

// startQueued starts pending builds, oldest first, while fewer than
// MaxConcurrent run. It returns how many it failed outright (a Job that could
// not be created), which count as advanced to a terminal state.
func (b *Builder) startQueued(ctx context.Context) (int, error) {
	cfg := b.Config.withDefaults()
	b.startMu.Lock()
	defer b.startMu.Unlock()
	builds, err := b.Store.ListUnfinishedBuilds(ctx)
	if err != nil {
		return 0, err
	}
	running := 0
	for i := range builds {
		if builds[i].Status == StatusBuilding {
			running++
		}
	}
	failed := 0
	var errs []error
	for i := range builds {
		if running >= cfg.MaxConcurrent {
			break
		}
		bld := &builds[i]
		if bld.Status != StatusPending || bld.JobName != "" {
			continue
		}
		if _, err := b.start(ctx, bld, cfg); err != nil {
			errs = append(errs, fmt.Errorf("build %s: %w", bld.ID, err))
			if bld.Status == StatusFailed {
				failed++
			}
			continue
		}
		running++
	}
	return failed, errors.Join(errs...)
}

// Cancel stops an in-flight build: delete its Job and mark it cancelled. A
+88 −1
Changes for internal/build/build_test.go: 88 added lines, 1 removed line.
Original line number Diff line number Diff line
@@ -3,6 +3,7 @@ package build
import (
	"context"
	"errors"
	"sort"
	"strings"
	"testing"
	"time"
@@ -77,6 +78,13 @@ func (f *fakeStore) ListUnfinishedBuilds(_ context.Context) ([]Build, error) {
			out = append(out, *b)
		}
	}
	// Oldest first, like the SQL; the frozen test clock ties, so the id breaks it.
	sort.Slice(out, func(i, j int) bool {
		if !out[i].CreatedAt.Equal(out[j].CreatedAt) {
			return out[i].CreatedAt.Before(out[j].CreatedAt)
		}
		return out[i].ID < out[j].ID
	})
	return out, nil
}

@@ -114,6 +122,9 @@ type fakeJobs struct {
	phase     JobPhase
	phaseErr  error
	createErr error
	// Per-job overrides of phase / phaseErr, keyed by Job name.
	phases    map[string]JobPhase
	phaseErrs map[string]error

	created   []JobParams
	cancelled []string
@@ -127,7 +138,13 @@ func (f *fakeJobs) CreateBuildJob(_ context.Context, p JobParams) (string, error
	return BuildJobName(p.BuildID), nil
}

func (f *fakeJobs) JobPhase(_ context.Context, _ string) (JobPhase, error) {
func (f *fakeJobs) JobPhase(_ context.Context, name string) (JobPhase, error) {
	if err, ok := f.phaseErrs[name]; ok {
		return JobUnknown, err
	}
	if p, ok := f.phases[name]; ok {
		return p, nil
	}
	return f.phase, f.phaseErr
}

@@ -445,6 +462,76 @@ func TestSyncAllAdvancesUnfinishedBuilds(t *testing.T) {
	}
}

// Builds past MaxConcurrent wait pending and start oldest first as running ones
// finish (build-supply-chain-12).
func TestSubmitQueuesPastMaxConcurrent(t *testing.T) {
	b, st, jb := newBuilder()
	b.Config.MaxConcurrent = 1
	ctx := context.Background()
	first, err := b.Submit(ctx, goodRequest())
	if err != nil || first.Status != StatusBuilding {
		t.Fatalf("first = %+v, %v; want building", first, err)
	}
	var queued []*Build
	for range 2 {
		q, err := b.Submit(ctx, goodRequest())
		if err != nil || q.Status != StatusPending || q.JobName != "" {
			t.Fatalf("queued = %+v, %v; want pending with no Job", q, err)
		}
		queued = append(queued, q)
	}
	if len(jb.created) != 1 {
		t.Fatalf("jobs created = %d, want 1", len(jb.created))
	}

	// A queued build is not a phantom: Sync leaves it alone.
	if got, err := b.Sync(ctx, queued[0].ID); err != nil || got.Status != StatusPending {
		t.Fatalf("Sync(queued) = %+v, %v; want still pending", got, err)
	}
	// Nothing frees up while the first runs.
	if _, err := b.SyncAll(ctx); err != nil || len(jb.created) != 1 {
		t.Fatalf("SyncAll while full: err %v, jobs %d", err, len(jb.created))
	}

	jb.phases = map[string]JobPhase{first.JobName: JobSucceeded}
	n, err := b.SyncAll(ctx)
	if err != nil {
		t.Fatalf("SyncAll: %v", err)
	}
	if n != 1 || len(jb.created) != 2 || jb.created[1].BuildID != queued[0].ID {
		t.Fatalf("advanced %d, jobs %v; want the first finished and the oldest queued started", n, jb.created)
	}
	if st.builds[queued[0].ID].Status != StatusBuilding || st.builds[queued[1].ID].Status != StatusPending {
		t.Fatalf("statuses = %s, %s; want building, pending",
			st.builds[queued[0].ID].Status, st.builds[queued[1].ID].Status)
	}
}

// One build whose Job cannot be read does not stall the others.
func TestSyncAllContinuesPastOneBuildsError(t *testing.T) {
	b, st, jb := newBuilder()
	b.Config.MaxConcurrent = 3
	ctx := context.Background()
	var ids []string
	for range 3 {
		bld, err := b.Submit(ctx, goodRequest())
		if err != nil {
			t.Fatal(err)
		}
		ids = append(ids, bld.ID)
	}
	jb.phase = JobSucceeded
	jb.phaseErrs = map[string]error{BuildJobName(ids[0]): errors.New("apiserver hiccup")}
	n, err := b.SyncAll(ctx)
	if err == nil || !strings.Contains(err.Error(), ids[0]) {
		t.Fatalf("err = %v, want it to name %s", err, ids[0])
	}
	if n != 2 || st.builds[ids[1]].Status != StatusSucceeded || st.builds[ids[2]].Status != StatusSucceeded {
		t.Fatalf("advanced %d; statuses %s %s; want the other two succeeded", n,
			st.builds[ids[1]].Status, st.builds[ids[2]].Status)
	}
}

// TestImageBuildFailuresMetricCountsFailedBuilds asserts felis_image_build_failures_total
// (spec §23) advances exactly once per failed build from BOTH terminal-failure
// producers, and never on a successful build. There are two distinct Inc sites —
Loading