fix(submit): 分块上传的预算检查复用一分钟内读过的 blob 大小,存储读不到时回 503 让面板自动重试
This commit is contained in:
13 files changed
+335
-25
No files matched your search
+8
-2
@@ -5884,7 +5884,11 @@ paths:
|
|||||||
the same context cap (400) and storage budget (403) as a single upload.
|
the same context cap (400) and storage budget (403) as a single upload.
|
||||||
A part that breaks off is cut back off, so the staged bytes are always a
|
A part that breaks off is cut back off, so the staged bytes are always a
|
||||||
prefix of the file. One request per upload at a time (409 upload_busy).
|
prefix of the file. One request per upload at a time (409 upload_busy).
|
||||||
Staged bytes untouched for 24 hours are deleted.
|
Staged bytes untouched for 24 hours are deleted. The budget check reads
|
||||||
|
blob sizes remembered for up to a minute; when a size has to be read and
|
||||||
|
the uploads store does not answer, the answer is 503
|
||||||
|
uploads_store_unavailable with Retry-After, and the same part can be
|
||||||
|
sent again.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: app
|
x-felis-tier: app
|
||||||
security: [{ sessionCookie: [] }]
|
security: [{ sessionCookie: [] }]
|
||||||
@@ -5938,7 +5942,9 @@ paths:
|
|||||||
way, and deletes the staged copy. Holds the same per-user upload
|
way, and deletes the staged copy. Holds the same per-user upload
|
||||||
cooldown (429) and writes the same submission.upload audit event.
|
cooldown (429) and writes the same submission.upload audit event.
|
||||||
Nothing staged is 400. After a failure the staged bytes stay, for a
|
Nothing staged is 400. After a failure the staged bytes stay, for a
|
||||||
retry.
|
retry. The budget is checked against every blob's size read from the
|
||||||
|
uploads store; a store that does not answer is 503
|
||||||
|
uploads_store_unavailable with Retry-After.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: app
|
x-felis-tier: app
|
||||||
security: [{ sessionCookie: [] }]
|
security: [{ sessionCookie: [] }]
|
||||||
|
|||||||
@@ -824,6 +824,7 @@ func (f *fakeRepo) SetAllowlistWake(_ context.Context, n, uuid string, canWake b
|
|||||||
}
|
}
|
||||||
return ErrNotFound
|
return ErrNotFound
|
||||||
}
|
}
|
||||||
|
|
||||||
// RequestRetire and CancelRetire mirror PGRepo's: the first request time is
|
// RequestRetire and CancelRetire mirror PGRepo's: the first request time is
|
||||||
// kept, a deletion stays a deletion, and only an admin cancels a deletion.
|
// kept, a deletion stays a deletion, and only an admin cancels a deletion.
|
||||||
func (f *fakeRepo) RequestRetire(_ context.Context, n string, del bool) (RetireState, error) {
|
func (f *fakeRepo) RequestRetire(_ context.Context, n string, del bool) (RetireState, error) {
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ import (
|
|||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
|
"time"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/submit"
|
"felis.lolicon.best/internal/submit"
|
||||||
)
|
)
|
||||||
@@ -502,6 +503,9 @@ func (a *API) handleDeleteSubmission(w http.ResponseWriter, r *http.Request) {
|
|||||||
// errSubmissionsUnavailable is returned when the approval lane is not configured
|
// errSubmissionsUnavailable is returned when the approval lane is not configured
|
||||||
// on this api instance (a nil Submissions service), so the admin/app boundary is
|
// on this api instance (a nil Submissions service), so the admin/app boundary is
|
||||||
// still exercised before the subsystem is wired in.
|
// still exercised before the subsystem is wired in.
|
||||||
|
// uploadsStoreRetry is the Retry-After on uploads_store_unavailable.
|
||||||
|
const uploadsStoreRetry = 5 * time.Second
|
||||||
|
|
||||||
var errSubmissionsUnavailable = newError(http.StatusServiceUnavailable, "submissions_unavailable",
|
var errSubmissionsUnavailable = newError(http.StatusServiceUnavailable, "submissions_unavailable",
|
||||||
"modpack submission subsystem is not configured")
|
"modpack submission subsystem is not configured")
|
||||||
|
|
||||||
@@ -511,7 +515,10 @@ var errSubmissionsUnavailable = newError(http.StatusServiceUnavailable, "submiss
|
|||||||
// allowance is 403 (the same status the server-resource quota answers with), a
|
// allowance is 403 (the same status the server-resource quota answers with), a
|
||||||
// full uploads store (every user's uploads together at their cap, or the volume
|
// full uploads store (every user's uploads together at their cap, or the volume
|
||||||
// short of free space) is 507, and an unconfigured upload transport is 503 (the store this deployment set has no
|
// short of free space) is 507, and an unconfigured upload transport is 503 (the store this deployment set has no
|
||||||
// implemented transport — an honest "not available here", not a client error). Everything
|
// implemented transport — an honest "not available here", not a client error). A
|
||||||
|
// blob store that did not answer the budget check is 503 uploads_store_unavailable
|
||||||
|
// with Retry-After: nothing was written and the same request can be sent again.
|
||||||
|
// Everything
|
||||||
// else — including a build.ErrInvalid raised by the pre-CAS build.Validate (a
|
// else — including a build.ErrInvalid raised by the pre-CAS build.Validate (a
|
||||||
// platform registry/context MISCONFIGURATION, never client input, since every
|
// platform registry/context MISCONFIGURATION, never client input, since every
|
||||||
// build input is platform-derived) and a post-CAS Submit hand-off failure — is a
|
// build input is platform-derived) and a post-CAS Submit hand-off failure — is a
|
||||||
@@ -541,6 +548,12 @@ func writeSubmitError(w http.ResponseWriter, r *http.Request, err error) {
|
|||||||
case errors.Is(err, submit.ErrUploadsUnavailable):
|
case errors.Is(err, submit.ErrUploadsUnavailable):
|
||||||
writeError(w, r, newError(http.StatusServiceUnavailable, "uploads_unavailable",
|
writeError(w, r, newError(http.StatusServiceUnavailable, "uploads_unavailable",
|
||||||
"modpack upload transport is not configured"))
|
"modpack upload transport is not configured"))
|
||||||
|
case errors.Is(err, submit.ErrStoreUnavailable):
|
||||||
|
// The budget could not be checked; nothing was written. The panel's upload
|
||||||
|
// loop sends the same request again.
|
||||||
|
log.Printf("api: %s %s: %v", r.Method, r.URL.Path, err)
|
||||||
|
writeError(w, r, newError(http.StatusServiceUnavailable, "uploads_store_unavailable",
|
||||||
|
"the uploads store did not answer; send the request again").retryAfter(uploadsStoreRetry))
|
||||||
case errors.Is(err, submit.ErrUploadBusy):
|
case errors.Is(err, submit.ErrUploadBusy):
|
||||||
writeError(w, r, newError(http.StatusConflict, "upload_busy",
|
writeError(w, r, newError(http.StatusConflict, "upload_busy",
|
||||||
"another request is still writing this upload; read where it stands and continue from there"))
|
"another request is still writing this upload; read where it stands and continue from there"))
|
||||||
|
|||||||
@@ -60,12 +60,16 @@ func TestContextUploadErrors(t *testing.T) {
|
|||||||
code int
|
code int
|
||||||
want string
|
want string
|
||||||
msg string
|
msg string
|
||||||
|
// retry is the Retry-After the answer must carry, if any.
|
||||||
|
retry string
|
||||||
}{
|
}{
|
||||||
{"offset mismatch", &submit.OffsetMismatchError{Received: 12}, 409, "upload_offset_mismatch", "the upload holds 12 bytes"},
|
{"offset mismatch", &submit.OffsetMismatchError{Received: 12}, 409, "upload_offset_mismatch", "the upload holds 12 bytes", ""},
|
||||||
{"busy", submit.ErrUploadBusy, 409, "upload_busy", ""},
|
{"busy", submit.ErrUploadBusy, 409, "upload_busy", "", ""},
|
||||||
{"part too large", fmt.Errorf("submit: write upload part: %w", submit.ErrPartTooLarge), 413, "part_too_large", ""},
|
{"part too large", fmt.Errorf("submit: write upload part: %w", submit.ErrPartTooLarge), 413, "part_too_large", "", ""},
|
||||||
{"not owned", submit.ErrNotFound, 404, "not_found", ""},
|
{"not owned", submit.ErrNotFound, 404, "not_found", "", ""},
|
||||||
{"no part store", submit.ErrUploadsUnavailable, 503, "uploads_unavailable", ""},
|
{"no part store", submit.ErrUploadsUnavailable, 503, "uploads_unavailable", "", ""},
|
||||||
|
{"store did not answer", fmt.Errorf("%w: size of the context of sub-3: dial tcp: i/o timeout", submit.ErrStoreUnavailable),
|
||||||
|
503, "uploads_store_unavailable", "send the request again", "5"},
|
||||||
} {
|
} {
|
||||||
fs := &fakeSubmissions{chunkErr: tc.err}
|
fs := &fakeSubmissions{chunkErr: tc.err}
|
||||||
w := do(appSubAPI(fs).ExternalHandler(), "PUT", "/api/v1/me/submissions/sub-9/context/upload?offset=12", "abcd", nil)
|
w := do(appSubAPI(fs).ExternalHandler(), "PUT", "/api/v1/me/submissions/sub-9/context/upload?offset=12", "abcd", nil)
|
||||||
@@ -75,6 +79,9 @@ func TestContextUploadErrors(t *testing.T) {
|
|||||||
if tc.msg != "" && !strings.Contains(w.Body.String(), tc.msg) {
|
if tc.msg != "" && !strings.Contains(w.Body.String(), tc.msg) {
|
||||||
t.Errorf("%s: body %s does not say %q", tc.name, w.Body.String(), tc.msg)
|
t.Errorf("%s: body %s does not say %q", tc.name, w.Body.String(), tc.msg)
|
||||||
}
|
}
|
||||||
|
if got := w.Header().Get("Retry-After"); got != tc.retry {
|
||||||
|
t.Errorf("%s: Retry-After = %q, want %q", tc.name, got, tc.retry)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,175 @@
|
|||||||
|
package submit
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// countingBlobs counts the Size reads per id.
|
||||||
|
type countingBlobs struct {
|
||||||
|
*fakeBlobs
|
||||||
|
sizes map[string]int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *countingBlobs) Size(ctx context.Context, id string) (int64, bool, error) {
|
||||||
|
c.sizes[id]++
|
||||||
|
return c.fakeBlobs.Size(ctx, id)
|
||||||
|
}
|
||||||
|
|
||||||
|
// putThenFail stores the bytes and still fails, like an object-store upload that
|
||||||
|
// completed but whose answer was lost.
|
||||||
|
type putThenFail struct{ *fakeBlobs }
|
||||||
|
|
||||||
|
func (p putThenFail) Put(ctx context.Context, id string, r io.Reader) (int64, error) {
|
||||||
|
if _, err := p.fakeBlobs.Put(ctx, id, r); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return 0, errors.New("put: connection reset")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The budget check ahead of each part reads every other blob's size once per
|
||||||
|
// blobSizeTTL; the one on completion reads them all from the store.
|
||||||
|
func TestPartBudgetRemembersBlobSizes(t *testing.T) {
|
||||||
|
m, _, fb, a := newChunkedManager(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
clock := testNow
|
||||||
|
m.Now = func() time.Time { return clock }
|
||||||
|
b, err := m.Create(ctx, CreateRequest{DisplayName: "B", SubmittedBy: "user-2"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
fb.stored[b.ID] = []byte("xyz")
|
||||||
|
cb := &countingBlobs{fakeBlobs: fb, sizes: map[string]int{}}
|
||||||
|
m.Blobs = cb
|
||||||
|
|
||||||
|
sendPart(t, m, a, 0, "\x1f\x8b\x08\x00")
|
||||||
|
sendPart(t, m, a, 4, "abcd")
|
||||||
|
if n := cb.sizes[b.ID]; n != 1 {
|
||||||
|
t.Fatalf("two parts read the other blob's size %d times, want once", n)
|
||||||
|
}
|
||||||
|
clock = clock.Add(blobSizeTTL)
|
||||||
|
sendPart(t, m, a, 8, "ef")
|
||||||
|
if n := cb.sizes[b.ID]; n != 2 {
|
||||||
|
t.Fatalf("a part once the TTL ran out: %d reads in all, want 2", n)
|
||||||
|
}
|
||||||
|
if _, err := m.CompleteUpload(ctx, a, "user-1"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if n := cb.sizes[b.ID]; n != 3 {
|
||||||
|
t.Fatalf("completion within the TTL: %d reads in all, want 3 (it reads the store)", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// What this Manager stores or deletes counts in the next part's budget at once,
|
||||||
|
// without waiting out the TTL.
|
||||||
|
func TestPartBudgetSeesThisManagersOwnWritesAtOnce(t *testing.T) {
|
||||||
|
m, _, _, a := newChunkedManager(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
m.MaxStoredBytesPerUser = 10
|
||||||
|
m.PartMaxBytes = 16
|
||||||
|
b, err := m.Create(ctx, CreateRequest{DisplayName: "B", SubmittedBy: "user-1"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
// The first part remembers B as holding nothing.
|
||||||
|
sendPart(t, m, a, 0, "\x1f\x8b")
|
||||||
|
if _, err := m.UploadContext(ctx, b.ID, "user-1", strings.NewReader("\x1f\x8b\x08\x00ab")); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, err := m.UploadPart(ctx, a, "user-1", 2, strings.NewReader("cde")); !errors.Is(err, ErrQuotaExceeded) {
|
||||||
|
t.Fatalf("5 bytes staged beside B's 6 under a 10-byte budget = %v, want ErrQuotaExceeded", err)
|
||||||
|
}
|
||||||
|
if _, err := m.Withdraw(ctx, b.ID, "user-1"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
// Its row is gone, so nothing counts it; nor is its size kept, or the
|
||||||
|
// remembered sizes would grow with every submission ever deleted.
|
||||||
|
if _, kept := m.sizes[b.ID]; kept {
|
||||||
|
t.Fatal("a withdrawn submission's blob size is still remembered")
|
||||||
|
}
|
||||||
|
if got := sendPart(t, m, a, 2, "cdefgh"); got != 8 {
|
||||||
|
t.Fatalf("staged %d after B was withdrawn, want 8", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPartBudgetSeesAReapedBlobAtOnce(t *testing.T) {
|
||||||
|
m, st, _, a := newChunkedManager(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
m.MaxStoredBytesPerUser = 10
|
||||||
|
m.PartMaxBytes = 16
|
||||||
|
b, err := m.Create(ctx, CreateRequest{DisplayName: "B", SubmittedBy: "user-1"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, err := m.UploadContext(ctx, b.ID, "user-1", strings.NewReader("\x1f\x8b\x08\x00ab")); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
reviewed := testNow.Add(-48 * time.Hour)
|
||||||
|
st.subs[b.ID].Status = StatusRejected
|
||||||
|
st.subs[b.ID].ReviewedAt = &reviewed
|
||||||
|
if n, err := m.ReapRejected(ctx, 24*time.Hour); n != 1 || err != nil {
|
||||||
|
t.Fatalf("ReapRejected = %d, %v; want 1, nil", n, err)
|
||||||
|
}
|
||||||
|
if got := sendPart(t, m, a, 0, "\x1f\x8b\x08\x00abcd"); got != 8 {
|
||||||
|
t.Fatalf("staged %d after B's blob was reaped, want 8", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A Put that failed may still have replaced the blob, so its size is read again.
|
||||||
|
func TestPartBudgetRereadsABlobAfterAFailedPut(t *testing.T) {
|
||||||
|
m, _, fb, a := newChunkedManager(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
m.MaxStoredBytesPerUser = 10
|
||||||
|
m.PartMaxBytes = 16
|
||||||
|
b, err := m.Create(ctx, CreateRequest{DisplayName: "B", SubmittedBy: "user-1"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if _, err := m.UploadContext(ctx, b.ID, "user-1", strings.NewReader("\x1f\x8b\x08\x00ab")); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
m.Blobs = putThenFail{fb}
|
||||||
|
if _, err := m.UploadContext(ctx, b.ID, "user-1", strings.NewReader("\x1f\x8b")); err == nil {
|
||||||
|
t.Fatal("setup: the failing Put succeeded")
|
||||||
|
}
|
||||||
|
m.Blobs = fb
|
||||||
|
if got := sendPart(t, m, a, 0, "\x1f\x8b\x08\x00abcd"); got != 8 {
|
||||||
|
t.Fatalf("staged %d beside B's 2 stored bytes, want 8", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A size that cannot be read fails the check as ErrStoreUnavailable, which the
|
||||||
|
// API answers 503 for the client to send again, never as a bare error (500).
|
||||||
|
func TestBudgetReadFailuresAreRetryable(t *testing.T) {
|
||||||
|
m, _, fb, a := newChunkedManager(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
if _, err := m.Create(ctx, CreateRequest{DisplayName: "B", SubmittedBy: "user-2"}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
fb.sizeErr = errors.New("dial tcp: i/o timeout")
|
||||||
|
if _, err := m.UploadPart(ctx, a, "user-1", 0, strings.NewReader("\x1f\x8b\x08\x00")); !errors.Is(err, ErrStoreUnavailable) {
|
||||||
|
t.Fatalf("a part with the blob store down = %v, want ErrStoreUnavailable", err)
|
||||||
|
}
|
||||||
|
if _, err := m.UploadContext(ctx, a, "user-1", strings.NewReader("\x1f\x8b\x08\x00")); !errors.Is(err, ErrStoreUnavailable) {
|
||||||
|
t.Fatalf("an upload with the blob store down = %v, want ErrStoreUnavailable", err)
|
||||||
|
}
|
||||||
|
if _, ok := fb.stored[a]; ok {
|
||||||
|
t.Fatal("an upload whose budget could not be checked was stored")
|
||||||
|
}
|
||||||
|
|
||||||
|
fb.sizeErr = nil
|
||||||
|
notDir := filepath.Join(t.TempDir(), "parts")
|
||||||
|
if err := os.WriteFile(notDir, nil, 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
m.Parts = &PartStore{Dir: notDir}
|
||||||
|
if _, err := m.UploadContext(ctx, a, "user-1", strings.NewReader("\x1f\x8b\x08\x00")); !errors.Is(err, ErrStoreUnavailable) {
|
||||||
|
t.Fatalf("an upload with the staging directory unreadable = %v, want ErrStoreUnavailable", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
+89
-13
@@ -67,6 +67,7 @@ import (
|
|||||||
"log"
|
"log"
|
||||||
"regexp"
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
|
"sync"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/build"
|
"felis.lolicon.best/internal/build"
|
||||||
@@ -109,6 +110,11 @@ var (
|
|||||||
// was never uploaded). The internal context-fetch route maps it to 404, the
|
// was never uploaded). The internal context-fetch route maps it to 404, the
|
||||||
// same distinction Exists draws for Approve.
|
// same distinction Exists draws for Approve.
|
||||||
ErrBlobNotFound = errors.New("submit: context blob not found")
|
ErrBlobNotFound = errors.New("submit: context blob not found")
|
||||||
|
// ErrStoreUnavailable means the sizes the storage budgets are checked against
|
||||||
|
// could not be read: the blob store (an object store, most likely) or the
|
||||||
|
// staging directory failed to answer. Nothing was written and the request can
|
||||||
|
// simply be sent again, so the API answers 503 and the panel retries.
|
||||||
|
ErrStoreUnavailable = errors.New("submit: the uploads store did not answer")
|
||||||
)
|
)
|
||||||
|
|
||||||
// invalidf wraps ErrInvalid so every malformed-request case maps to one 400.
|
// invalidf wraps ErrInvalid so every malformed-request case maps to one 400.
|
||||||
@@ -335,8 +341,9 @@ type Blobs interface {
|
|||||||
Open(ctx context.Context, id string) (io.ReadCloser, error)
|
Open(ctx context.Context, id string) (io.ReadCloser, error)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Manager orchestrates the approval lane. It holds no mutable state; the clock
|
// Manager orchestrates the approval lane. Its only mutable state is the blob
|
||||||
// and id generator are injectable for hermetic tests.
|
// sizes it remembers for the budget checks (blobSize); the clock and id
|
||||||
|
// generator are injectable for hermetic tests. Use it by pointer.
|
||||||
type Manager struct {
|
type Manager struct {
|
||||||
Store Store
|
Store Store
|
||||||
Builds Builds
|
Builds Builds
|
||||||
@@ -390,8 +397,27 @@ type Manager struct {
|
|||||||
|
|
||||||
Now func() time.Time
|
Now func() time.Time
|
||||||
IDGen func() string
|
IDGen func() string
|
||||||
|
|
||||||
|
// sizes remembers each blob's size as last read or written here, so the
|
||||||
|
// budget check ahead of every chunked-upload part does not stat every blob
|
||||||
|
// in the store again (blobSize).
|
||||||
|
sizesMu sync.Mutex
|
||||||
|
sizes map[string]knownSize
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// knownSize is a blob's size and when this Manager last read or wrote it.
|
||||||
|
type knownSize struct {
|
||||||
|
n int64
|
||||||
|
at time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
// blobSizeTTL is how long a blob size read from the store stands in for it in
|
||||||
|
// the budget check ahead of a chunked-upload part. The blobs this Manager writes
|
||||||
|
// or deletes update it at once; only another api replica's changes wait out the
|
||||||
|
// TTL, and the check that decides what is stored (UploadContext, which
|
||||||
|
// CompleteUpload goes through) always reads the store.
|
||||||
|
const blobSizeTTL = time.Minute
|
||||||
|
|
||||||
// ContextLimit is the effective cap on one uploaded context.
|
// ContextLimit is the effective cap on one uploaded context.
|
||||||
func (m *Manager) ContextLimit() int64 { return m.maxContextBytes() }
|
func (m *Manager) ContextLimit() int64 { return m.maxContextBytes() }
|
||||||
|
|
||||||
@@ -602,7 +628,7 @@ func (m *Manager) UploadContext(ctx context.Context, id, submittedBy string, r i
|
|||||||
return nil, invalidf("build context must be a gzip-compressed tarball (.tar.gz)")
|
return nil, invalidf("build context must be a gzip-compressed tarball (.tar.gz)")
|
||||||
}
|
}
|
||||||
|
|
||||||
limit, over, err := m.uploadLimit(ctx, submittedBy, id)
|
limit, over, err := m.uploadLimit(ctx, submittedBy, id, true)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -614,9 +640,14 @@ func (m *Manager) UploadContext(ctx context.Context, id, submittedBy string, r i
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
h := sha256.New()
|
h := sha256.New()
|
||||||
if _, err := m.Blobs.Put(ctx, id, io.TeeReader(&cappedReader{r: br, left: limit, over: over}, h)); err != nil {
|
n, err := m.Blobs.Put(ctx, id, io.TeeReader(&cappedReader{r: br, left: limit, over: over}, h))
|
||||||
|
if err != nil {
|
||||||
|
// A failed Put may or may not have replaced the blob; the next read
|
||||||
|
// finds out.
|
||||||
|
m.forgetSize(id)
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
m.noteSize(id, n)
|
||||||
digest := hex.EncodeToString(h.Sum(nil))
|
digest := hex.EncodeToString(h.Sum(nil))
|
||||||
won, err := m.Store.SetContextDigest(ctx, id, digest)
|
won, err := m.Store.SetContextDigest(ctx, id, digest)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -658,9 +689,10 @@ func (m *Manager) ownPending(ctx context.Context, id, submittedBy string) (*Subm
|
|||||||
// package: a burst that reaches two api replicas (or any direct caller of the
|
// package: a burst that reaches two api replicas (or any direct caller of the
|
||||||
// Manager) can overshoot by up to one blob per interleaved upload — each write
|
// Manager) can overshoot by up to one blob per interleaved upload — each write
|
||||||
// still bounded by the single-blob cap — while the API's per-user upload
|
// still bounded by the single-blob cap — while the API's per-user upload
|
||||||
// reservation collapses the single-replica case.
|
// reservation collapses the single-replica case. fresh reads every blob's size
|
||||||
func (m *Manager) uploadLimit(ctx context.Context, submittedBy, id string) (int64, error, error) {
|
// from the store; otherwise a size read within blobSizeTTL stands in for it.
|
||||||
used, total, err := m.storedBytes(ctx, submittedBy, id)
|
func (m *Manager) uploadLimit(ctx context.Context, submittedBy, id string, fresh bool) (int64, error, error) {
|
||||||
|
used, total, err := m.storedBytes(ctx, submittedBy, id, fresh)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, nil, err
|
return 0, nil, err
|
||||||
}
|
}
|
||||||
@@ -735,7 +767,9 @@ func (m *Manager) UploadPart(ctx context.Context, id, submittedBy string, offset
|
|||||||
return UploadProgress{}, invalidf("build context must be a gzip-compressed tarball (.tar.gz)")
|
return UploadProgress{}, invalidf("build context must be a gzip-compressed tarball (.tar.gz)")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
limit, over, err := m.uploadLimit(ctx, submittedBy, id)
|
// Remembered blob sizes: a part is one of many, and what is finally stored is
|
||||||
|
// checked against the store itself on completion.
|
||||||
|
limit, over, err := m.uploadLimit(ctx, submittedBy, id, false)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return UploadProgress{}, err
|
return UploadProgress{}, err
|
||||||
}
|
}
|
||||||
@@ -798,8 +832,10 @@ func (m *Manager) ReapStaleParts(olderThan time.Duration) (int, error) {
|
|||||||
// the sums cannot drift from what is actually occupying the volume (including
|
// the sums cannot drift from what is actually occupying the volume (including
|
||||||
// blobs uploaded before any budget existed). A staged chunked upload counts as
|
// blobs uploaded before any budget existed). A staged chunked upload counts as
|
||||||
// well, so parts spread over several pending submissions cannot hold more than
|
// well, so parts spread over several pending submissions cannot hold more than
|
||||||
// the budget allows.
|
// the budget allows. With fresh unset, a blob size read or written within
|
||||||
func (m *Manager) storedBytes(ctx context.Context, submittedBy, excludeID string) (user, total int64, err error) {
|
// blobSizeTTL is used as it stands (blobSize). A size that cannot be read is
|
||||||
|
// ErrStoreUnavailable, for the caller to try again.
|
||||||
|
func (m *Manager) storedBytes(ctx context.Context, submittedBy, excludeID string, fresh bool) (user, total int64, err error) {
|
||||||
subs, err := m.Store.ListSubmissions(ctx)
|
subs, err := m.Store.ListSubmissions(ctx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, 0, err
|
return 0, 0, err
|
||||||
@@ -808,14 +844,14 @@ func (m *Manager) storedBytes(ctx context.Context, submittedBy, excludeID string
|
|||||||
if s.ID == excludeID {
|
if s.ID == excludeID {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
n, _, err := m.Blobs.Size(ctx, s.ID)
|
n, err := m.blobSize(ctx, s.ID, fresh)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, 0, err
|
return 0, 0, err
|
||||||
}
|
}
|
||||||
if m.Parts != nil {
|
if m.Parts != nil {
|
||||||
staged, _, err := m.Parts.Size(s.ID)
|
staged, _, err := m.Parts.Size(s.ID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, 0, err
|
return 0, 0, fmt.Errorf("%w: size of the staged upload of %s: %v", ErrStoreUnavailable, s.ID, err)
|
||||||
}
|
}
|
||||||
n += staged
|
n += staged
|
||||||
}
|
}
|
||||||
@@ -827,6 +863,43 @@ func (m *Manager) storedBytes(ctx context.Context, submittedBy, excludeID string
|
|||||||
return user, total, nil
|
return user, total, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// blobSize is id's stored blob size: the one remembered from within blobSizeTTL
|
||||||
|
// unless fresh, else read from the store and remembered.
|
||||||
|
func (m *Manager) blobSize(ctx context.Context, id string, fresh bool) (int64, error) {
|
||||||
|
now := m.now()
|
||||||
|
if !fresh {
|
||||||
|
m.sizesMu.Lock()
|
||||||
|
k, ok := m.sizes[id]
|
||||||
|
m.sizesMu.Unlock()
|
||||||
|
if ok && now.Sub(k.at) < blobSizeTTL {
|
||||||
|
return k.n, nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
n, _, err := m.Blobs.Size(ctx, id)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("%w: size of the context of %s: %v", ErrStoreUnavailable, id, err)
|
||||||
|
}
|
||||||
|
m.noteSizeAt(id, n, now)
|
||||||
|
return n, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (m *Manager) noteSize(id string, n int64) { m.noteSizeAt(id, n, m.now()) }
|
||||||
|
|
||||||
|
func (m *Manager) noteSizeAt(id string, n int64, at time.Time) {
|
||||||
|
m.sizesMu.Lock()
|
||||||
|
defer m.sizesMu.Unlock()
|
||||||
|
if m.sizes == nil {
|
||||||
|
m.sizes = map[string]knownSize{}
|
||||||
|
}
|
||||||
|
m.sizes[id] = knownSize{n: n, at: at}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (m *Manager) forgetSize(id string) {
|
||||||
|
m.sizesMu.Lock()
|
||||||
|
defer m.sizesMu.Unlock()
|
||||||
|
delete(m.sizes, id)
|
||||||
|
}
|
||||||
|
|
||||||
// ReapRejected deletes the uploaded context of every submission rejected more
|
// ReapRejected deletes the uploaded context of every submission rejected more
|
||||||
// than olderThan ago, keeping the row. It returns how many blobs it deleted and
|
// than olderThan ago, keeping the row. It returns how many blobs it deleted and
|
||||||
// carries on past a blob it cannot delete, reporting the first such error.
|
// carries on past a blob it cannot delete, reporting the first such error.
|
||||||
@@ -848,6 +921,7 @@ func (m *Manager) ReapRejected(ctx context.Context, olderThan time.Duration) (in
|
|||||||
_, ok, err := m.Blobs.Size(ctx, s.ID)
|
_, ok, err := m.Blobs.Size(ctx, s.ID)
|
||||||
if err == nil && ok {
|
if err == nil && ok {
|
||||||
if err = m.Blobs.Delete(ctx, s.ID); err == nil {
|
if err = m.Blobs.Delete(ctx, s.ID); err == nil {
|
||||||
|
m.noteSize(s.ID, 0)
|
||||||
reaped++
|
reaped++
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1133,7 +1207,9 @@ func (m *Manager) deleteBlob(ctx context.Context, id string) error {
|
|||||||
if m.Blobs == nil {
|
if m.Blobs == nil {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
if err := m.Blobs.Delete(ctx, id); err != nil {
|
err := m.Blobs.Delete(ctx, id)
|
||||||
|
m.forgetSize(id)
|
||||||
|
if err != nil {
|
||||||
return fmt.Errorf("submit: submission removed, but its uploaded context could not be deleted (it may remain on the uploads store): %w", err)
|
return fmt.Errorf("submit: submission removed, but its uploaded context could not be deleted (it may remain on the uploads store): %w", err)
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
@@ -73,6 +73,7 @@
|
|||||||
"uploads_unavailable": "Uploads aren't available right now.",
|
"uploads_unavailable": "Uploads aren't available right now.",
|
||||||
"upload_busy": "Another upload of this submission is still running — wait a moment and try again.",
|
"upload_busy": "Another upload of this submission is still running — wait a moment and try again.",
|
||||||
"upload_offset_mismatch": "The upload fell out of step with the server — try again to pick up where it stopped.",
|
"upload_offset_mismatch": "The upload fell out of step with the server — try again to pick up where it stopped.",
|
||||||
|
"uploads_store_unavailable": "The uploads store did not answer — try again in a moment; what was already sent is kept.",
|
||||||
"part_too_large": "A piece of the upload was larger than the server accepts.",
|
"part_too_large": "A piece of the upload was larger than the server accepts.",
|
||||||
"backup_unavailable": "Backups aren't available right now.",
|
"backup_unavailable": "Backups aren't available right now.",
|
||||||
"account_retired": "This account has been retired — sign in with the account it was migrated to.",
|
"account_retired": "This account has been retired — sign in with the account it was migrated to.",
|
||||||
|
|||||||
@@ -73,6 +73,7 @@
|
|||||||
"uploads_unavailable": "上传功能当前不可用。",
|
"uploads_unavailable": "上传功能当前不可用。",
|
||||||
"upload_busy": "这个投稿的另一次上传仍在进行——请稍等片刻再试。",
|
"upload_busy": "这个投稿的另一次上传仍在进行——请稍等片刻再试。",
|
||||||
"upload_offset_mismatch": "上传进度与服务器对不上——重试即可从中断处接着传。",
|
"upload_offset_mismatch": "上传进度与服务器对不上——重试即可从中断处接着传。",
|
||||||
|
"uploads_store_unavailable": "上传存储暂时没有响应——稍后重试即可,已传的部分会保留。",
|
||||||
"part_too_large": "上传的某一片超过了服务器接受的大小。",
|
"part_too_large": "上传的某一片超过了服务器接受的大小。",
|
||||||
"backup_unavailable": "备份功能当前不可用。",
|
"backup_unavailable": "备份功能当前不可用。",
|
||||||
"account_retired": "该账户已退役——请使用迁移后的账户登录。",
|
"account_retired": "该账户已退役——请使用迁移后的账户登录。",
|
||||||
|
|||||||
@@ -660,6 +660,13 @@ describe("image whitelist and builds wire shapes", () => {
|
|||||||
expect(humanizeError({ code: "submission_cooldown" })).toMatch(/try again/i);
|
expect(humanizeError({ code: "submission_cooldown" })).toMatch(/try again/i);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("says an uploads store that did not answer is worth another try", async () => {
|
||||||
|
const { humanizeError } = await import("./api");
|
||||||
|
expect(humanizeError({ status: 503, code: "uploads_store_unavailable" })).toBe(
|
||||||
|
"The uploads store did not answer — try again in a moment; what was already sent is kept.",
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
it("withdrawSubmission DELETEs /me/submissions/{id}", async () => {
|
it("withdrawSubmission DELETEs /me/submissions/{id}", async () => {
|
||||||
const sub = { id: "sub-4", status: "pending_review" };
|
const sub = { id: "sub-4", status: "pending_review" };
|
||||||
const fetchSpy = fakeFetch(sub);
|
const fetchSpy = fakeFetch(sub);
|
||||||
|
|||||||
@@ -1170,6 +1170,8 @@ export function humanizeError(e: unknown): string {
|
|||||||
return t("upload_busy");
|
return t("upload_busy");
|
||||||
case "upload_offset_mismatch":
|
case "upload_offset_mismatch":
|
||||||
return t("upload_offset_mismatch");
|
return t("upload_offset_mismatch");
|
||||||
|
case "uploads_store_unavailable":
|
||||||
|
return t("uploads_store_unavailable");
|
||||||
case "part_too_large":
|
case "part_too_large":
|
||||||
return t("part_too_large");
|
return t("part_too_large");
|
||||||
case "backup_unavailable":
|
case "backup_unavailable":
|
||||||
|
|||||||
@@ -134,6 +134,16 @@ describe("uploadContext", () => {
|
|||||||
expect((await sentParts()).map(([o]) => o)).toEqual([0, 4, 8]);
|
expect((await sentParts()).map(([o]) => o)).toEqual([0, 4, 8]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("sends a part again when the uploads store did not answer the budget check", async () => {
|
||||||
|
calls.getContextUpload.mockResolvedValueOnce(at(0)).mockResolvedValueOnce(at(4));
|
||||||
|
acceptParts();
|
||||||
|
calls.putContextPart.mockImplementationOnce(async (_id, offset: number, part: Blob) => at(offset + part.size));
|
||||||
|
calls.putContextPart.mockRejectedValueOnce({ status: 503, code: "uploads_store_unavailable", message: "" });
|
||||||
|
await uploadContext("sub-1", FILE, { sleep });
|
||||||
|
expect((await sentParts()).map(([o]) => o)).toEqual([0, 4, 4, 8]);
|
||||||
|
expect(calls.completeContextUpload).toHaveBeenCalledTimes(1);
|
||||||
|
});
|
||||||
|
|
||||||
it("gives up at once on a refusal the next attempt cannot outlast", async () => {
|
it("gives up at once on a refusal the next attempt cannot outlast", async () => {
|
||||||
calls.getContextUpload.mockResolvedValue(at(0));
|
calls.getContextUpload.mockResolvedValue(at(0));
|
||||||
const refused = { status: 400, code: "bad_request", message: "context must be a gzip-compressed tarball" };
|
const refused = { status: 400, code: "bad_request", message: "context must be a gzip-compressed tarball" };
|
||||||
@@ -216,6 +226,15 @@ describe("uploadContext completion", () => {
|
|||||||
expect(sleep.mock.calls.map((c) => c[0])).toEqual([1000, 2000]);
|
expect(sleep.mock.calls.map((c) => c[0])).toEqual([1000, 2000]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("completes again when the uploads store did not answer, the staged bytes still all there", async () => {
|
||||||
|
calls.getContextUpload.mockResolvedValueOnce(at(0)).mockResolvedValue(at(10));
|
||||||
|
calls.completeContextUpload
|
||||||
|
.mockRejectedValueOnce({ status: 503, code: "uploads_store_unavailable", message: "" })
|
||||||
|
.mockResolvedValueOnce({ id: "sub-1" });
|
||||||
|
await uploadContext("sub-1", FILE, { sleep });
|
||||||
|
expect(calls.completeContextUpload).toHaveBeenCalledTimes(2);
|
||||||
|
});
|
||||||
|
|
||||||
it("surfaces the real refusal when completing again answers with one", async () => {
|
it("surfaces the real refusal when completing again answers with one", async () => {
|
||||||
calls.getContextUpload.mockResolvedValueOnce(at(0)).mockResolvedValue(at(10));
|
calls.getContextUpload.mockResolvedValueOnce(at(0)).mockResolvedValue(at(10));
|
||||||
const quota = { status: 403, code: "submission_quota_exceeded", message: "budget" };
|
const quota = { status: 403, code: "submission_quota_exceeded", message: "budget" };
|
||||||
|
|||||||
@@ -28,14 +28,16 @@ export const MAX_ATTEMPTS = 8;
|
|||||||
export const MAX_COMPLETE_WAITS = 40;
|
export const MAX_COMPLETE_WAITS = 40;
|
||||||
|
|
||||||
// A transient answer is one the next attempt can outlast: no response at all, a
|
// A transient answer is one the next attempt can outlast: no response at all, a
|
||||||
// tunnel or ingress page in place of the API's (upstream_unavailable), or a 409
|
// tunnel or ingress page in place of the API's (upstream_unavailable), an uploads
|
||||||
// that means "ask where the upload stands and send again".
|
// store that did not answer the budget check (uploads_store_unavailable), or a
|
||||||
|
// 409 that means "ask where the upload stands and send again".
|
||||||
export function isTransient(e: unknown): boolean {
|
export function isTransient(e: unknown): boolean {
|
||||||
const err = e as Partial<ApiError> | null;
|
const err = e as Partial<ApiError> | null;
|
||||||
if (!err || typeof err.code !== "string") return false;
|
if (!err || typeof err.code !== "string") return false;
|
||||||
return (
|
return (
|
||||||
err.code === "network_error" ||
|
err.code === "network_error" ||
|
||||||
err.code === "upstream_unavailable" ||
|
err.code === "upstream_unavailable" ||
|
||||||
|
err.code === "uploads_store_unavailable" ||
|
||||||
err.code === "upload_busy" ||
|
err.code === "upload_busy" ||
|
||||||
err.code === "upload_offset_mismatch"
|
err.code === "upload_offset_mismatch"
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -1971,7 +1971,7 @@ export interface paths {
|
|||||||
get: operations["getContextUpload"];
|
get: operations["getContextUpload"];
|
||||||
/**
|
/**
|
||||||
* Append one part of your chunked context upload.
|
* Append one part of your chunked context upload.
|
||||||
* @description The body is the part's raw bytes, at most part_max_bytes (32 MiB). offset is where they start: 0 starts the upload over, and anything else must equal the staged length, or the answer is 409 upload_offset_mismatch and the client reads GET for where to resume. The first part must open with the gzip magic (400). The staged total meets the same context cap (400) and storage budget (403) as a single upload. A part that breaks off is cut back off, so the staged bytes are always a prefix of the file. One request per upload at a time (409 upload_busy). Staged bytes untouched for 24 hours are deleted.
|
* @description The body is the part's raw bytes, at most part_max_bytes (32 MiB). offset is where they start: 0 starts the upload over, and anything else must equal the staged length, or the answer is 409 upload_offset_mismatch and the client reads GET for where to resume. The first part must open with the gzip magic (400). The staged total meets the same context cap (400) and storage budget (403) as a single upload. A part that breaks off is cut back off, so the staged bytes are always a prefix of the file. One request per upload at a time (409 upload_busy). Staged bytes untouched for 24 hours are deleted. The budget check reads blob sizes remembered for up to a minute; when a size has to be read and the uploads store does not answer, the answer is 503 uploads_store_unavailable with Retry-After, and the same part can be sent again.
|
||||||
*/
|
*/
|
||||||
put: operations["putContextUploadPart"];
|
put: operations["putContextUploadPart"];
|
||||||
post?: never;
|
post?: never;
|
||||||
@@ -1992,7 +1992,7 @@ export interface paths {
|
|||||||
put?: never;
|
put?: never;
|
||||||
/**
|
/**
|
||||||
* Store your staged chunked upload as the submission's build context.
|
* Store your staged chunked upload as the submission's build context.
|
||||||
* @description Runs every check of POST /api/v1/me/submissions/{id}/context on the staged bytes (format, cap, budget, room), records the digest the same way, and deletes the staged copy. Holds the same per-user upload cooldown (429) and writes the same submission.upload audit event. Nothing staged is 400. After a failure the staged bytes stay, for a retry.
|
* @description Runs every check of POST /api/v1/me/submissions/{id}/context on the staged bytes (format, cap, budget, room), records the digest the same way, and deletes the staged copy. Holds the same per-user upload cooldown (429) and writes the same submission.upload audit event. Nothing staged is 400. After a failure the staged bytes stay, for a retry. The budget is checked against every blob's size read from the uploads store; a store that does not answer is 503 uploads_store_unavailable with Retry-After.
|
||||||
*/
|
*/
|
||||||
post: operations["completeContextUpload"];
|
post: operations["completeContextUpload"];
|
||||||
delete?: never;
|
delete?: never;
|
||||||
|
|||||||
Reference in new issue
Block a user