Unverified Commit bb68fefe authored by Lemon-miaow's avatar Lemon-miaow
Browse files

fix(quota): make the claim gate atomic, and stop zeroing the storage cache

Two defects in the §9.3 quota path, both invisible to the hermetic suite:

- Audit #4's TOCTOU was real and documented: QuotaCheck and ClaimServer were
  separate statements, so two concurrent claims by one user for two different
  ownerless servers both read count < max_servers and both won. The gate now
  lives inside ClaimServer, in the SAME transaction as the ownership write,
  under pg_advisory_xact_lock(hashtext(user_id)) — the aggregate read, the
  four-dimension re-check (shared with QuotaCheck via one helper so the two
  cannot drift), and the UPDATE are one serialized decision. The loser gets
  ErrQuotaExceeded, which both claim handlers map to the same 403 the
  sequential path gives; the server row is additionally taken FOR UPDATE so
  same-server races still resolve to exactly one winner.

- The server PATCH path called UpdateServerResources(..., 0) for storage even
  though a resources patch cannot change storage. The cached columns are the
  ONLY input to the quota aggregate, so every resource patch silently dropped
  that server's storage contribution from its owner's cap. The handler now
  reads the current spec and passes storage through.

Red-then-green: the new pgint test drives two real concurrent claims against
max_servers=1 (before: both win; now: exactly one win + one gated 403, and the
DB shows one owned row); the hermetic suite pins the 403 mapping and the
storage-preserving cache write.
parent d829267f
Loading
Loading
Loading
Loading
+34 −4
Changes for internal/api/api_test.go: 34 added lines, 4 removed lines.
Original line number Diff line number Diff line
@@ -38,6 +38,14 @@ type fakeRepo struct {
	owners    map[string]string
	ownersErr error
	claimOK   map[string]bool // name -> claim succeeds; absent name -> ErrNotFound
	// claimQuotaRefuse simulates ClaimServer's atomic quota gate (audit #4)
	// refusing a name whose advisory pre-check already passed.
	claimQuotaRefuse map[string]bool
	// serverResources / resourceUpdates mirror the cached resource columns:
	// ServerResources is what the resize path reads (to preserve storage), and
	// UpdateServerResources records the write for assertions.
	serverResources map[string]ResourceSpec
	resourceUpdates map[string]ResourceSpec
	audits          []AuditEntry
	joins           []string
	// create-server seeding (spec §15)
@@ -206,7 +214,8 @@ func newFakeRepo() *fakeRepo {
		allowlist: map[string]map[string]bool{}, allowUUID: map[string]map[string]bool{},
		mine:    map[string][]MyServerView{},
		owners:  map[string]string{},
		claimOK: map[string]bool{},
		claimOK: map[string]bool{}, claimQuotaRefuse: map[string]bool{},
		serverResources: map[string]ResourceSpec{}, resourceUpdates: map[string]ResourceSpec{},
		seeded: map[string]bool{}, aliases: map[string]string{},
		linkCodes: map[string]fakeLinkCode{}, links: map[string]string{},
		linkAuthSource:    map[string]string{},
@@ -249,10 +258,13 @@ func (f *fakeRepo) QuotaCheck(_ context.Context, userID string, _ string, _ Reso
	return f.QuotaAvailable(context.TODO(), userID)
}

func (f *fakeRepo) UpdateServerResources(_ context.Context, _ string, _, _, _ int) error { return nil }
func (f *fakeRepo) UpdateServerResources(_ context.Context, name string, cpu, mem, stor int) error {
	f.resourceUpdates[name] = ResourceSpec{CPUMilli: cpu, MemoryMB: mem, StorageMB: stor}
	return nil
}

func (f *fakeRepo) ServerResources(_ context.Context, _ string) (ResourceSpec, error) {
	return ResourceSpec{}, nil
func (f *fakeRepo) ServerResources(_ context.Context, name string) (ResourceSpec, error) {
	return f.serverResources[name], nil
}
func (f *fakeRepo) CreateLinkCode(_ context.Context, code, mcUUID, authSource string, expiresAt time.Time) error {
	f.linkCodes[code] = fakeLinkCode{mcUUID: mcUUID, authSource: authSource, expiresAt: expiresAt}
@@ -649,6 +661,9 @@ func (f *fakeRepo) ClaimServer(_ context.Context, n, u string) (bool, error) {
	if !present {
		return false, ErrNotFound
	}
	if ok && f.claimQuotaRefuse[n] {
		return false, ErrQuotaExceeded // mirrors the atomic gate losing the race
	}
	return ok, nil
}
func (f *fakeRepo) RecordJoin(_ context.Context, n, uuid string) error {
@@ -1783,6 +1798,21 @@ func TestClaimStateMachine(t *testing.T) {
			t.Fatalf("code = %d body %s", w.Code, w.Body.String())
		}
	})
	t.Run("atomic gate refusal -> 403 quota_exceeded", func(t *testing.T) {
		// The advisory pre-check passed, but ClaimServer's serialized re-check
		// (audit #4) refuses: the caller must see the same 403, not a 500.
		repo := newFakeRepo()
		repo.linked["u1"] = true
		repo.quota["u1"] = true
		repo.claimOK["survival"] = true
		repo.claimQuotaRefuse["survival"] = true
		api := newTestAPI(repo, newFakeCluster())
		api.External = staticExternal{p: user}
		w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/claim", "", nil)
		if w.Code != http.StatusForbidden || decodeErr(t, w) != "quota_exceeded" {
			t.Fatalf("code = %d body %s, want 403 quota_exceeded", w.Code, w.Body.String())
		}
	})
	t.Run("already claimed -> 409", func(t *testing.T) {
		repo := newFakeRepo()
		repo.linked["u1"] = true
+7 −0
Changes for internal/api/errors.go: 7 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -15,6 +15,13 @@ var (
	ErrNotFound = errors.New("not found")
	// ErrConflict means an atomic precondition failed (e.g. claim lost the race).
	ErrConflict = errors.New("conflict")
	// ErrQuotaExceeded means an ownership write would push the user over a quota
	// cap (spec §9.3). ClaimServer — the atomic gate — returns it when a claim
	// passes the handler's advisory pre-check but loses the serialized re-check
	// (two concurrent claims by one user); handlers map it to a 403
	// quota_exceeded, the same answer the pre-check gives, so the CONCURRENT case
	// and the SEQUENTIAL case are indistinguishable to the caller.
	ErrQuotaExceeded = errors.New("server quota exhausted")
	// ErrLinkCodeInvalid means an account-link code is unknown or expired (spec
	// §10). It is a client error (the verify endpoint exists; the code is bad), so
	// handlers map it to 400, not 404.
+6 −0
Changes for internal/api/handlers_internal.go: 6 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -240,6 +240,12 @@ func (a *API) handleInternalClaim(w http.ResponseWriter, r *http.Request) {
	// 412 above; a lost race (0 rows) is 409.
	claimed, err := a.Repo.ClaimServer(r.Context(), name, userID)
	if err != nil {
		// Same atomic quota gate as the external face (audit #4): the concurrent
		// loser gets the sequential 403, never an over-provisioned tenant.
		if errors.Is(err, ErrQuotaExceeded) {
			writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted"))
			return
		}
		a.writeLookupError(w, r, err)
		return
	}
+20 −0
Changes for internal/api/handlers_patch_test.go: 20 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -127,6 +127,26 @@ func TestPatchServerMemoryOverride(t *testing.T) {
	}
}

// TestPatchServerPreservesStorageCache pins the storage dimension of the quota
// aggregate: a resources patch cannot change storage, so the cached storage
// must survive it — passing 0 would silently zero the owner's aggregate (the
// cached columns are QuotaCheck's only input) from that patch onward.
func TestPatchServerPreservesStorageCache(t *testing.T) {
	api, repo, _, _ := newPatchAPI()
	repo.byName["survival"].OwnerID = "u1"
	repo.quota["u1"] = true
	repo.serverResources["survival"] = ResourceSpec{StorageMB: 10240}

	w := patchSurvival(api, `{"memory":"2Gi","resources":{"memory":"4Gi","cpu":"2"}}`)
	if w.Code != http.StatusOK {
		t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
	}
	got := repo.resourceUpdates["survival"]
	if got.CPUMilli != 2000 || got.MemoryMB != 4096 || got.StorageMB != 10240 {
		t.Fatalf("resource cache = %+v, want cpu 2000 / mem 4096 / storage preserved 10240", got)
	}
}

// TestPatchServerRejections is the validation matrix: each malformed request is
// rejected with the right status and stable error code, and (critically) NOTHING
// reaches the cluster on a rejection — the analog of create's "no CRD written".
+20 −1
Changes for internal/api/handlers_user.go: 20 added lines, 1 removed line.
Original line number Diff line number Diff line
@@ -142,6 +142,13 @@ func (a *API) handleClaim(w http.ResponseWriter, r *http.Request) {
	// ③ atomic claim
	claimed, err := a.Repo.ClaimServer(r.Context(), name, p.UserID)
	if err != nil {
		// The atomic gate re-checks quota under the per-user lock (audit #4): a
		// concurrent claim that spent the last slot surfaces here, with the same
		// 403 the pre-check gives sequentially.
		if errors.Is(err, ErrQuotaExceeded) {
			writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted"))
			return
		}
		a.writeLookupError(w, r, err)
		return
	}
@@ -759,7 +766,19 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
			a.writeLookupError(w, r, err)
			return
		}
		_ = a.Repo.UpdateServerResources(r.Context(), name, newCPU, newMemMB, 0)
		// A resource patch cannot change storage, so its cached contribution must
		// be preserved: passing 0 would silently zero the storage dimension of the
		// owner's four-cap aggregate (the cached columns are its only input).
		storMB := 0
		if rec != nil {
			cur, err := a.Repo.ServerResources(r.Context(), name)
			if err != nil {
				writeError(w, r, err)
				return
			}
			storMB = cur.StorageMB
		}
		_ = a.Repo.UpdateServerResources(r.Context(), name, newCPU, newMemMB, storMB)
	} else {
		if err := a.Cluster.PatchServerSpec(r.Context(), name, patch); err != nil {
			a.writeLookupError(w, r, err)
Loading