fix(api): make OTP-start throttle atomic to close concurrent-burst bypass

The email-OTP resend cooldown checked the window with a peek (allowed)
and only recorded it after delivery. For OTP that throttle is the sole
defense and each admitted send is a real, non-idempotent email, so a
burst of truly concurrent starts all passed the peek before any recorded
and every one mailed: N concurrent starts bombed a mailbox with N codes.

Add an atomic reserve/release pair to cooldownLimiter: reserve checks and
records the window in one critical section under the mutex, so a
concurrent burst yields exactly one winner; release rolls a reservation
back only if it is still the current one, so a slow failing caller never
clobbers a newer holder. handleEmailOTPStart now reserves both the
principal and the recipient key up front and defers a rollback that frees
both windows on any mint, create, or delivery error — preserving the old
"a failed send does not consume the cooldown" property, now race-free.

The wake path keeps allowed→record: its real gate is the running cap and
its side effect (SetDesiredState) is idempotent, so the peek gap is
harmless there.

Tests: a frozen-clock gate-mailer fires 8 concurrent starts for one
victim from one principal and asserts exactly one mail and one 202; a
flaky-mailer test proves a failed delivery releases the window so an
immediate retry in the same instant is admitted.
This commit is contained in:
flyemoji committed 2026-06-30 23:04:07 +09:00
1 parent 98739044a5
commit 879b1777f5
3 files changed
+182 -8

No files matched your search

+116
View File
@@ -5,6 +5,8 @@ import (
"errors"
"net/http"
"net/http/httptest"
"sync"
"sync/atomic"
"testing"
"time"
)
@@ -413,3 +415,117 @@ func TestEmailOTPMailerError(t *testing.T) {
}
}
}
// flakyMailer fails its first SendOTP, then succeeds, so a test can observe whether
// a failed delivery left the cooldown consumed (a retry would be wrongly throttled)
// or released (the retry is admitted, as it must be).
type flakyMailer struct {
calls int
}
func (m *flakyMailer) SendOTP(_ context.Context, _, _ string) error {
m.calls++
if m.calls == 1 {
return errors.New("smtp down")
}
return nil
}
// TestEmailOTPStartFailedDeliveryReleasesCooldown covers the new reserve→rollback
// path: a send that reserves the cooldown but then fails to deliver must release it,
// so the very next attempt at the same instant is admitted rather than 429'd. Without
// the deferred release a transient SMTP blip would lock a player out for the whole
// window — strictly worse than the throttle is meant to be.
func TestEmailOTPStartFailedDeliveryReleasesCooldown(t *testing.T) {
user := &Principal{UserID: "u1", Email: "[email protected]", Role: "user"}
mailer := &flakyMailer{}
api := newTestAPI(newFakeRepo(), newFakeCluster())
api.External = staticExternal{p: user}
api.Mailer = mailer
api.Now = func() time.Time { return time.Unix(1_700_000_000, 0) } // frozen: same window
eh := api.ExternalHandler()
if w := do(eh, "POST", "/api/v1/account/email/start", `{"email":"[email protected]"}`, nil); w.Code < 500 {
t.Fatalf("first send (mailer fails): code = %d, want 5xx (%s)", w.Code, w.Body.String())
}
// Same instant, same caller and recipient: had the failed send burned the window
// this would be a 429. The rollback frees it, so the retry delivers.
if w := do(eh, "POST", "/api/v1/account/email/start", `{"email":"[email protected]"}`, nil); w.Code != http.StatusAccepted {
t.Fatalf("retry after failed delivery: code = %d, want 202 (the failed send must release the cooldown) (%s)", w.Code, w.Body.String())
}
if mailer.calls != 2 {
t.Errorf("mailer calls = %d, want 2 (one failed, one delivered)", mailer.calls)
}
}
// gateMailer blocks every SendOTP until all concurrent callers have arrived, making
// any check-then-act window in the throttle deterministically observable instead of
// scheduler-dependent. It is the committed counterpart of the adversarial burst probe.
type gateMailer struct {
calls int64
entered chan struct{}
release chan struct{}
}
func (m *gateMailer) SendOTP(_ context.Context, _, _ string) error {
atomic.AddInt64(&m.calls, 1)
m.entered <- struct{}{}
<-m.release
return nil
}
// TestEmailOTPStartConcurrentBurstBounded is the regression guard for the email-bomb
// closure under concurrency. net/http serves each request on its own goroutine, so a
// throttle that peeks then records in two steps lets a burst of starts for one victim
// from one principal all slip through together. The atomic reserve admits exactly one;
// the rest get 429. (Runs without -race — the gate makes the race deterministic.)
func TestEmailOTPStartConcurrentBurstBounded(t *testing.T) {
const n = 8
mailer := &gateMailer{entered: make(chan struct{}, n), release: make(chan struct{})}
api := newTestAPI(newFakeRepo(), newFakeCluster())
api.External = staticExternal{p: &Principal{UserID: "u1", Email: "[email protected]", Role: "user"}}
api.Mailer = mailer
api.Now = func() time.Time { return time.Unix(1_700_000_000, 0) } // frozen: one shared window
eh := api.ExternalHandler()
var wg sync.WaitGroup
codes := make([]int, n)
for i := 0; i < n; i++ {
wg.Add(1)
go func(i int) {
defer wg.Done()
w := do(eh, "POST", "/api/v1/account/email/start", `{"email":"[email protected]"}`, nil)
codes[i] = w.Code
}(i)
}
// Drain whoever reached the mailer, then release them. A correct throttle admits
// exactly one; a buggy one lets several arrive before any records, and they pile
// up here within the deadline.
deadline := time.After(2 * time.Second)
reached := 0
loop:
for reached < n {
select {
case <-mailer.entered:
reached++
case <-deadline:
break loop
}
}
close(mailer.release)
wg.Wait()
accepted := 0
for _, c := range codes {
if c == http.StatusAccepted {
accepted++
}
}
if got := atomic.LoadInt64(&mailer.calls); got != 1 {
t.Errorf("burst delivered %d mails for one victim from one account in one window; want exactly 1 (accepted=%d, reached=%d)", got, accepted, reached)
}
if accepted != 1 {
t.Errorf("burst accepted %d starts; want exactly 1 (the rest must be 429)", accepted)
}
}