fix(files): 模组包上传也逐段核对 SHA-256、长文件名能写、卡住的导出按时停掉、世界被占用时文件页说明原因并等它结束

This commit is contained in:
Lemon-miaow committed 2026-09-29 01:26:24 +08:00
1 parent 1d1549cea2
commit cb3065da1d
59 files changed
+2538 -338

No files matched your search

+24
View File
@@ -494,6 +494,9 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
if fileStage != nil {
go expireFileSessions(ctx, fileStage, fileSessionSweep, stderr)
}
if exporter != nil {
go expireExports(ctx, a, exportSweep)
}
go retention.Loop(ctx, drv.DB(), retention.Policy{Audit: auditRetention}, retentionInterval, slog.Default())
servers := []*http.Server{internalSrv, externalSrv}
@@ -824,6 +827,27 @@ func expireFileSessions(ctx context.Context, s *fileedit.Stage, every time.Durat
}
}
// exportSweep is how often expireExports runs: an export whose Job never
// connected is stopped within a minute of going stale.
const exportSweep = time.Minute
// expireExports runs the export sweep (api.API.ExpireExports) on a ticker. The
// export routes sweep as they are called, and an owner who closed the tab calls
// none; a Job whose Pod never got going would then keep the server from
// starting until the Job's deadline.
func expireExports(ctx context.Context, a interface{ ExpireExports() }, every time.Duration) {
t := time.NewTicker(every)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
a.ExpireExports()
}
}
}
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
// submissions rejected more than submit.RejectedContextRetention ago, and the
// chunked uploads left untouched for submit.StalePartRetention. Without it a
+25
View File
@@ -286,3 +286,28 @@ func TestExpireFileSessions(t *testing.T) {
t.Fatalf("said %q", got)
}
}
type sweepCount struct{ n atomic.Int32 }
func (s *sweepCount) ExpireExports() { s.n.Add(1) }
// TestExpireExports: the loop sweeps on each tick, and returns once felis-api
// shuts down.
func TestExpireExports(t *testing.T) {
var s sweepCount
ctx, cancel := context.WithCancel(context.Background())
done := make(chan struct{})
go func() { expireExports(ctx, &s, time.Millisecond); close(done) }()
for deadline := time.Now().Add(5 * time.Second); s.n.Load() < 3; time.Sleep(time.Millisecond) {
if time.Now().After(deadline) {
cancel()
t.Fatalf("swept %d times in 5s at a 1ms tick", s.n.Load())
}
}
cancel()
select {
case <-done:
case <-time.After(5 * time.Second):
t.Fatal("the loop outlived its context")
}
}
+30 -8
View File
@@ -3,6 +3,7 @@ package main
import (
"context"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"encoding/json"
"errors"
@@ -14,6 +15,7 @@ import (
"net/http"
"os"
"os/signal"
"strconv"
"strings"
"syscall"
"time"
@@ -67,6 +69,7 @@ func cmdExport(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis export: --target-url and %s are required\n", worldexport.TokenEnv)
return 2
}
limitHeapToCgroup()
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
@@ -222,19 +225,30 @@ func (d *digestReader) Read(p []byte) (int, error) {
return n, err
}
// streamExport runs write straight into the body of the PUT. An error from
// write aborts the chunked body, and felis-api then cuts the browser's download
// off rather than end it; that error is the one reported, since the PUT's own
// error only wraps it. When the PUT ends first, write is stopped.
// streamExport runs write straight into the body of the PUT, hashing it as it
// goes. Once write has finished, the SHA-256 of all it wrote rides the
// request's trailer (worldexport.DigestTrailer), and felis-api holds back the
// last bytes from the browser until what it received hashes the same. An error
// from write aborts the chunked body before the trailer, and felis-api then
// cuts the browser's download off rather than end it; that error is the one
// reported, since the PUT's own error only wraps it. When the PUT ends first,
// write is stopped.
func streamExport(ctx context.Context, target, token, contentType string, size int64, write func(io.Writer) error) error {
pr, pw := io.Pipe()
trailer := http.Header{worldexport.DigestTrailer: nil}
werr := make(chan error, 1)
go func() {
err := write(pw)
sum := sha256.New()
err := write(io.MultiWriter(pw, sum))
if err == nil {
// Set before the body ends: the transport reads the trailer once it
// has read the body to its end.
trailer.Set(worldexport.DigestTrailer, "sha-256=:"+base64.StdEncoding.EncodeToString(sum.Sum(nil))+":")
}
pw.CloseWithError(err)
werr <- err
}()
err := putExport(ctx, target, token, contentType, pr, size)
err := putExport(ctx, target, token, contentType, pr, size, trailer)
pr.CloseWithError(io.ErrClosedPipe)
if w := <-werr; w != nil && !errors.Is(w, io.ErrClosedPipe) {
return w
@@ -248,12 +262,20 @@ func streamExport(ctx context.Context, target, token, contentType string, size i
// redirects. felis-api answers only after the whole download, which the Job's
// activeDeadlineSeconds bounds, so the header timeout is a backstop for a
// wedged endpoint and not the real limit.
func putExport(ctx context.Context, target, token, contentType string, body io.Reader, size int64) error {
//
// The body always goes chunked, which is what lets it end with a trailer; a
// size the Job knows (-1 when it does not) goes as worldexport.LengthHeader in
// place of Content-Length.
func putExport(ctx context.Context, target, token, contentType string, body io.Reader, size int64, trailer http.Header) error {
req, err := http.NewRequestWithContext(ctx, http.MethodPut, target, body)
if err != nil {
return err
}
req.ContentLength = size
req.ContentLength = -1
req.Trailer = trailer
if size >= 0 {
req.Header.Set(worldexport.LengthHeader, strconv.FormatInt(size, 10))
}
req.Header.Set("Authorization", "Bearer "+token)
req.Header.Set("Content-Type", contentType)
client := &http.Client{
+25 -2
View File
@@ -8,6 +8,7 @@ import (
"context"
"crypto/rand"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"errors"
"io"
@@ -54,6 +55,25 @@ func receiveExport(t *testing.T, reply func(w http.ResponseWriter)) *exportRecei
func noContent(w http.ResponseWriter) { w.WriteHeader(http.StatusNoContent) }
// sentWhole fails unless the upload rcv got ended with the Content-Digest
// trailer of its own bytes, and declared length as its size (-1: none).
func sentWhole(t *testing.T, rcv *exportReceiver, length int64) {
t.Helper()
sum := sha256.Sum256(rcv.body)
want := "sha-256=:" + base64.StdEncoding.EncodeToString(sum[:]) + ":"
wantLength := ""
if length >= 0 {
wantLength = strconv.FormatInt(length, 10)
}
r := rcv.req
if rcv.readErr != nil || r.Trailer.Get(worldexport.DigestTrailer) != want || r.Header.Get(worldexport.LengthHeader) != wantLength ||
r.ContentLength != -1 || strings.Join(r.TransferEncoding, ",") != "chunked" {
t.Fatalf("upload read %v, trailer %v, %s %q, length %d, encoding %v; want trailer %q and %s %q, chunked",
rcv.readErr, r.Trailer, worldexport.LengthHeader, r.Header.Get(worldexport.LengthHeader), r.ContentLength, r.TransferEncoding,
want, worldexport.LengthHeader, wantLength)
}
}
func tarEntries(t *testing.T, archive []byte) map[string]string {
t.Helper()
gz, err := gzip.NewReader(bytes.NewReader(archive))
@@ -141,6 +161,7 @@ func TestCmdExportWorld(t *testing.T) {
if got := tarEntries(t, rcv.body); !reflect.DeepEqual(got, want) {
t.Fatalf("archive holds %v\nwant %v", got, want)
}
sentWhole(t, rcv, -1)
want2 := "felis export: left out 1 entries a tar cannot hold (symbolic links, devices, sockets)\n" +
"felis export: left out 2 files that hold platform secrets\n" +
"felis export: server=survival mode=world downloaded\n"
@@ -297,6 +318,7 @@ func TestCmdExportBackup(t *testing.T) {
if got := tarEntries(t, rcv.body); !reflect.DeepEqual(got, want) {
t.Fatalf("archive holds %d entries, want exactly the redacted properties, level.dat and the region file", len(got))
}
sentWhole(t, rcv, -1)
if want := "felis export: left out 1 files that hold platform secrets\nfelis export: server=survival mode=backup downloaded\n"; stdout.String() != want {
t.Errorf("stdout = %q, want %q", stdout.String(), want)
}
@@ -417,9 +439,10 @@ func TestCmdExportFiles(t *testing.T) {
if code != 0 || stdout != "felis export: server=survival mode=files downloaded\n" {
t.Fatalf("exit %d, stdout %q, stderr %q", code, stdout, stderr)
}
if string(rcv.body) != want || rcv.req.ContentLength != int64(len(want)) || rcv.req.Header.Get("Content-Type") != "application/octet-stream" {
t.Fatalf("body %q, length %d, type %q; want %q", rcv.body, rcv.req.ContentLength, rcv.req.Header.Get("Content-Type"), want)
if string(rcv.body) != want || rcv.req.Header.Get("Content-Type") != "application/octet-stream" {
t.Fatalf("body %q, type %q; want %q", rcv.body, rcv.req.Header.Get("Content-Type"), want)
}
sentWhole(t, rcv, int64(len(want)))
})
}
+35 -1
View File
@@ -57,6 +57,7 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
fmt.Fprintln(stderr, "felis files: --op is required")
return 2
}
limitHeapToCgroup()
req := fileedit.Request{
Op: *op, Path: *path, To: *to, Expect: *expect, CreateOnly: *createOnly, Overwrite: *overwrite,
}
@@ -86,6 +87,13 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
req.Upload = &fileedit.Upload{
Size: *size, SHA256: *sum,
Open: func() (io.ReadCloser, error) { return fetchUpload(ctx, *sourceURL, token) },
Landed: func() {
if err := reportLanded(ctx, *sourceURL, token); err != nil {
// The file is in place; felis-api drops its copy when it
// has sat idle long enough, and the panel cancels it too.
fmt.Fprintf(stderr, "felis files: tell felis-api the upload landed: %v\n", err)
}
},
}
}
// An upload or an unzip (the only ops that report progress) can run long
@@ -112,7 +120,7 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
// fetchUpload opens the staged upload on felis-api's internal face. There is no
// retry: the token opens the upload once (fileedit.Stage), so a second attempt
// could only be refused, and the caller retries the failed Job whole (a file
// sent in parts stays staged until it has been served whole once, so that retry
// sent in parts stays staged until its Job reports it landed, so that retry
// does not send it again). Redirects are refused because the request carries the
// token and the internal face never redirects; the header timeout catches a
// wedged endpoint, and the Job's activeDeadlineSeconds bounds the body.
@@ -136,3 +144,29 @@ func fetchUpload(ctx context.Context, url, token string) (io.ReadCloser, error)
}
return resp.Body, nil
}
// reportLanded tells felis-api the upload's file is in place (DELETE on the URL
// it was fetched from, with the same token), so it deletes the copy it staged.
// One try: the file has landed whatever the answer, and a copy nobody deletes
// is dropped once it has sat idle for fileedit.SessionIdle.
func reportLanded(ctx context.Context, url, token string) error {
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
req, err := http.NewRequestWithContext(ctx, http.MethodDelete, url, nil)
if err != nil {
return err
}
req.Header.Set("Authorization", "Bearer "+token)
client := &http.Client{
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
}
resp, err := client.Do(req)
if err != nil {
return err
}
resp.Body.Close()
if resp.StatusCode != http.StatusNoContent {
return fmt.Errorf("DELETE returned %s", resp.Status)
}
return nil
}
+86 -8
View File
@@ -3,6 +3,7 @@ package main
import (
"archive/zip"
"bytes"
"cmp"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
@@ -35,19 +36,43 @@ func filesResult(t *testing.T, stdout string) fileedit.Result {
return res
}
// stagedUpload serves body to a request carrying Bearer token, and 404 to any
// other, the way felis-api's internal face does.
func stagedUpload(t *testing.T, token string, body []byte) *httptest.Server {
// stagedSource is felis-api's internal face for one staged upload. It serves
// body to a GET carrying Bearer token and 404 to any other, and answers the
// DELETE that reports the file landed with landedCode (204 when unset),
// redirecting to landedTo when that is a redirect. reports counts those
// DELETEs, each with the token and at the path the bytes came from.
type stagedSource struct {
*httptest.Server
reports, strays atomic.Int32
landedCode int
landedTo string
}
func stagedUpload(t *testing.T, token string, body []byte) *stagedSource {
t.Helper()
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.Header.Get("Authorization") != "Bearer "+token {
s := &stagedSource{}
s.Server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.Header.Get("Authorization") != "Bearer "+token || r.URL.Path != "/u" {
s.strays.Add(1)
http.Error(w, "no such upload", http.StatusNotFound)
return
}
w.Write(body)
switch r.Method {
case http.MethodGet:
w.Write(body)
case http.MethodDelete:
s.reports.Add(1)
if s.landedTo != "" {
w.Header().Set("Location", s.landedTo)
}
w.WriteHeader(cmp.Or(s.landedCode, http.StatusNoContent))
default:
s.strays.Add(1)
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
}
}))
t.Cleanup(srv.Close)
return srv
t.Cleanup(s.Close)
return s
}
func uploadArgs(root, sourceURL string, body []byte) []string {
@@ -89,8 +114,58 @@ func TestCmdFilesUpload(t *testing.T) {
if err != nil || !bytes.Equal(got, body) {
t.Fatalf("landed %q, %v", got, err)
}
if n, strays := srv.reports.Load(), srv.strays.Load(); n != 1 || strays != 0 || stderr.Len() != 0 {
t.Fatalf("reported landed %d times, %d stray requests, stderr %q; want once", n, strays, stderr.String())
}
})
t.Run("a file already there is a result, and nothing is reported landed", func(t *testing.T) {
root := uploadRoot(t)
if err := os.WriteFile(filepath.Join(root, "plugins", "a.jar"), []byte("old!"), 0o644); err != nil {
t.Fatal(err)
}
srv := stagedUpload(t, "tok", body)
t.Setenv(fileedit.UploadTokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeExists || srv.reports.Load() != 0 {
t.Fatalf("result = %+v, reported landed %d times", res, srv.reports.Load())
}
})
// The file is in place whatever felis-api answers, so the Job still succeeds
// and says why the staged copy may linger. A redirect is not followed, since
// the request carries the token.
for name, tc := range map[string]struct {
code int
stderr string
}{
"refused": {http.StatusNotFound, "felis files: tell felis-api the upload landed: DELETE returned 404 Not Found\n"},
"redirected": {http.StatusFound, "felis files: tell felis-api the upload landed: DELETE returned 302 Found\n"},
} {
t.Run("a landed report "+name+" still lands the file", func(t *testing.T) {
var elsewhere atomic.Int32
away := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) { elsewhere.Add(1) }))
defer away.Close()
root := uploadRoot(t)
srv := stagedUpload(t, "tok", body)
srv.landedCode, srv.landedTo = tc.code, away.URL+"/u"
t.Setenv(fileedit.UploadTokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != "" || stderr.String() != tc.stderr || elsewhere.Load() != 0 {
t.Fatalf("result = %+v, stderr %q, redirect followed %d times", res, stderr.String(), elsewhere.Load())
}
if got, err := os.ReadFile(filepath.Join(root, "plugins", "a.jar")); err != nil || !bytes.Equal(got, body) {
t.Fatalf("landed %q, %v", got, err)
}
})
}
// A refused fetch is the Job failing, never a Result: the API answers it with a
// 500 the caller retries whole.
t.Run("a refused fetch exits 1 and lands nothing", func(t *testing.T) {
@@ -104,6 +179,9 @@ func TestCmdFilesUpload(t *testing.T) {
if !strings.Contains(stderr.String(), "404") {
t.Fatalf("stderr %q does not name the status", stderr.String())
}
if srv.reports.Load() != 0 {
t.Fatal("a refused fetch was reported landed")
}
if _, err := os.Lstat(filepath.Join(root, "plugins", "a.jar")); !os.IsNotExist(err) {
t.Fatalf("a refused fetch left a file: %v", err)
}
+48
View File
@@ -0,0 +1,48 @@
package main
import (
"os"
"runtime/debug"
"strconv"
"strings"
)
// cgroupMemoryFiles are where a container reads the memory it is allowed:
// cgroup v2 first, then v1.
var cgroupMemoryFiles = []string{"/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory/memory.limit_in_bytes"}
// limitHeapToCgroup sets the Go heap's soft limit from the container's memory
// limit, so the collector works harder as a Job nears it and the kernel does not
// kill the Job first. An extraction or a folder zipped for download keeps a few
// hundred bytes per entry for as long as it runs; without the limit the heap
// grows to twice that before a collection, and a 256 MiB Job was killed at
// 400,000 entries whose live heap was 115 MB. GOMEMLIMIT set by hand wins.
func limitHeapToCgroup() {
if os.Getenv("GOMEMLIMIT") != "" {
return
}
for _, f := range cgroupMemoryFiles {
b, err := os.ReadFile(f)
if err != nil {
continue
}
if n, ok := softMemoryLimit(string(b)); ok {
debug.SetMemoryLimit(n)
}
return
}
}
// softMemoryLimit answers three fifths of the limit a cgroup memory file holds,
// or false for "max" (no limit) and anything unreadable. The rest is left for
// what the kernel charges the container beyond the Go heap: the page cache of
// the files it reads and writes, and the inodes it creates. Under a 256 MiB
// limit, 400,000 extracted entries peaked at 184 MB resident with the heap held
// to 150 MiB.
func softMemoryLimit(content string) (int64, bool) {
n, err := strconv.ParseInt(strings.TrimSpace(content), 10, 64)
if err != nil || n <= 0 {
return 0, false
}
return n / 5 * 3, true
}
+58
View File
@@ -0,0 +1,58 @@
package main
import (
"math"
"os"
"path/filepath"
"runtime/debug"
"testing"
)
func TestSoftMemoryLimit(t *testing.T) {
for _, c := range []struct {
in string
want int64
ok bool
}{
{"268435456\n", 161061273, true}, // 256 MiB, as memory.max holds it
{"max\n", 0, false},
{"0\n", 0, false},
{"-1", 0, false},
{"", 0, false},
} {
got, ok := softMemoryLimit(c.in)
if got != c.want || ok != c.ok {
t.Errorf("softMemoryLimit(%q) = %d %v, want %d %v", c.in, got, ok, c.want, c.ok)
}
}
}
// TestLimitHeapToCgroup checks the limit comes from the first cgroup file there
// is, and that GOMEMLIMIT set by hand leaves the heap alone.
func TestLimitHeapToCgroup(t *testing.T) {
prevFiles, prevLimit := cgroupMemoryFiles, debug.SetMemoryLimit(-1)
t.Cleanup(func() { cgroupMemoryFiles = prevFiles; debug.SetMemoryLimit(prevLimit) })
dir := t.TempDir()
v1, v1b := filepath.Join(dir, "v1"), filepath.Join(dir, "v1b")
if err := os.WriteFile(v1, []byte("268435456\n"), 0o600); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(v1b, []byte("536870912\n"), 0o600); err != nil {
t.Fatal(err)
}
cgroupMemoryFiles = []string{filepath.Join(dir, "missing"), v1, v1b}
t.Setenv("GOMEMLIMIT", "")
debug.SetMemoryLimit(math.MaxInt64)
limitHeapToCgroup()
if got := debug.SetMemoryLimit(-1); got != 161061273 {
t.Fatalf("limit = %d, want three fifths of 256 MiB", got)
}
t.Setenv("GOMEMLIMIT", "1GiB")
debug.SetMemoryLimit(math.MaxInt64)
limitHeapToCgroup()
if got := debug.SetMemoryLimit(-1); got != math.MaxInt64 {
t.Fatalf("limit = %d with GOMEMLIMIT set, want it left alone", got)
}
}