Loading cmd/felis/api.go +32 −7 Changes for cmd/felis/api.go: 32 added lines, 7 removed lines. Original line number Diff line number Diff line Loading @@ -10,6 +10,7 @@ import ( "os" "regexp" "strings" "sync" "sync/atomic" "time" Loading Loading @@ -415,15 +416,17 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int { } go reapRejectedContexts(ctx, submissions, stderr) select { case <-ctx.Done(): shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = internalSrv.Shutdown(shutdownCtx) _ = externalSrv.Shutdown(shutdownCtx) servers := []*http.Server{internalSrv, externalSrv} if httpsSrv != nil { _ = httpsSrv.Shutdown(shutdownCtx) servers = append(servers, httpsSrv) } for _, srv := range servers { srv.RegisterOnShutdown(a.CloseStreams) } select { case <-ctx.Done(): shutdownServers(servers, apiShutdownGrace, stderr) return 0 case err := <-errc: if err != nil && err != http.ErrServerClosed { Loading @@ -443,8 +446,30 @@ const ( // apiIdleTimeout caps how long a kept-alive connection may sit idle between // requests before the server closes it, bounding idle-connection exhaustion. apiIdleTimeout = 120 * time.Second // apiShutdownGrace is how long the listeners drain after SIGTERM. The pod gets // the Kubernetes default of 30s before SIGKILL; this leaves the rest for the // process to exit. apiShutdownGrace = 20 * time.Second ) // shutdownServers drains every listener at once under one deadline: in turn, a // slow first listener would spend the time the others needed. Log streams end // through RegisterOnShutdown (API.CloseStreams); what is still running when the // deadline passes is cut off with the process. func shutdownServers(servers []*http.Server, grace time.Duration, stderr io.Writer) { ctx, cancel := context.WithTimeout(context.Background(), grace) defer cancel() var wg sync.WaitGroup for _, srv := range servers { wg.Go(func() { if err := srv.Shutdown(ctx); err != nil { fmt.Fprintf(stderr, "felis api: shutdown %s: %v\n", srv.Addr, err) } }) } wg.Wait() } // newAPIServer builds an http.Server with hardened header/idle timeouts (gosec // G112) shared by all three felis-api listeners (internal, external, https). // WriteTimeout and ReadTimeout are deliberately LEFT UNSET: the external and https Loading cmd/felis/api_test.go +48 −0 Changes for cmd/felis/api_test.go: 48 added lines, 0 removed lines. Original line number Diff line number Diff line Loading @@ -4,8 +4,12 @@ import ( "context" "errors" "fmt" "io" "net" "net/http" "sync/atomic" "testing" "time" "felis.lolicon.best/internal/api" "felis.lolicon.best/internal/build" Loading Loading @@ -97,6 +101,50 @@ func TestNewAPIServerSetsHardenedTimeouts(t *testing.T) { } } // Every listener drains at once, and the shutdown hook (API.CloseStreams in // cmdAPI) runs on each: two listeners each holding a request that ends when // the hook fires take one hook's worth of time, well inside the deadline. func TestShutdownServersDrainsListenersTogether(t *testing.T) { release := make(chan struct{}) var hooks int32 started := make(chan struct{}, 2) var servers []*http.Server for range 2 { srv := newAPIServer("", http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { started <- struct{}{} <-release })) srv.RegisterOnShutdown(func() { if atomic.AddInt32(&hooks, 1) == 1 { time.AfterFunc(100*time.Millisecond, func() { close(release) }) } }) l, err := net.Listen("tcp", "127.0.0.1:0") if err != nil { t.Fatal(err) } srv.Addr = l.Addr().String() go func() { _ = srv.Serve(l) }() go func() { if resp, err := http.Get("http://" + srv.Addr); err == nil { resp.Body.Close() } }() servers = append(servers, srv) } <-started <-started begun := time.Now() shutdownServers(servers, 5*time.Second, io.Discard) if took := time.Since(begun); took > 2*time.Second { t.Fatalf("shutdown took %v", took) } if got := atomic.LoadInt32(&hooks); got != 2 { t.Fatalf("shutdown hook ran %d times, want once per listener", got) } } type fakeRefStore struct { images []build.Image builds []build.Build Loading docs/openapi.yaml +44 −4 Changes for docs/openapi.yaml: 44 added lines, 4 removed lines. Original line number Diff line number Diff line Loading @@ -31,6 +31,28 @@ info: and the panel). Admin-tier external operations additionally require the admin Access path. See `x-felis-face` / `x-felis-tier` on each operation. Behaviour every operation shares, and so not repeated under each: * Every response carries `X-Request-Id` (a well-formed inbound one is kept), `X-Content-Type-Options: nosniff`, `X-Frame-Options: DENY`, `Referrer-Policy: no-referrer` and `Content-Security-Policy: default-src 'none'`. `Strict-Transport-Security` is added when the request came through the TLS edge (`X-Forwarded-Proto: https`). * A path no operation serves is `404 not_found`; a path served under other methods is `405 method_not_allowed` with an `Allow` header. * A POST/PUT/PATCH/DELETE a browser sends from another site (`Sec-Fetch-Site` `same-site` or `cross-site`, or an `Origin` whose host is not the request's) is `403 cross_site`, before authentication. Callers that send neither header (the plugins, scripts) are unaffected. * A JSON body over 1 MiB is `413 too_large`. A request body must keep arriving: after 30 s it has to average 16 KiB/s or the connection is closed. * Event streams (the console and build logs) tag each line with `id:` (unix seconds); an EventSource that reconnects with `Last-Event-ID` within the hour resumes from that second instead of the tailed backlog. The server re-checks the caller every minute and ends the stream with `event: revoked` once the session or the access is gone; a stream also closes after 30 minutes and on server shutdown, and the client simply reconnects. servers: - url: https://api-internal.{root_domain} description: >- Loading Loading @@ -1443,9 +1465,16 @@ paths: security: [{ accessJWT: [] }] parameters: - { name: name, in: path, required: true, schema: { type: string } } - name: Last-Event-ID in: header required: false description: The id of the last line received; resumes the stream from that second (within the hour). schema: { type: string } responses: '200': description: An event stream of log lines. description: >- An event stream of log lines (`id:` + `data:` per line, `:` comments as keep-alives), ended by `event: revoked` when access is withdrawn. content: text/event-stream: schema: { type: string } Loading Loading @@ -1871,7 +1900,11 @@ paths: get: tags: [servers] operationId: status summary: Status of your own server. summary: Status of a server; the full record for its owner and staff. description: >- Anyone signed in may ask. The owner and staff get the whole projection; anyone else gets what the game's own server list shows: name, subdomain, displayName, phase, ready, playersOnline and playersMax. x-felis-face: [external] x-felis-tier: app security: [{ accessJWT: [] }] Loading @@ -1879,7 +1912,7 @@ paths: - { name: name, in: path, required: true, schema: { type: string } } responses: '200': description: The server's status projection. description: The server's status projection (trimmed for non-owners). content: application/json: schema: { $ref: '#/components/schemas/ServerInfo' } Loading Loading @@ -4707,9 +4740,16 @@ paths: security: [{ accessJWT: [] }] parameters: - { name: id, in: path, required: true, schema: { type: string } } - name: Last-Event-ID in: header required: false description: The id of the last line received; resumes the stream from that second (within the hour). schema: { type: string } responses: '200': description: An event stream of build log lines. description: >- An event stream of build log lines (`id:` + `data:` per line), ended by `event: revoked` when the caller is no longer staff. content: text/event-stream: schema: { type: string } Loading docs/troubleshooting.md +22 −1 Changes for docs/troubleshooting.md: 22 added lines, 1 removed line. Original line number Diff line number Diff line Loading @@ -1238,6 +1238,26 @@ All four mandated metrics have real producers; scrape them when triaging: - `felis_reaper_worlds_deleted_total` — increments only after a world PVC is actually deleted post-backup (§10); a spike here means worlds crossed the 15d idle line — cross-check that join events are flowing (§10 risk vectors). - `felis_http_requests_total{face,method,route,code}` and `felis_http_request_duration_seconds{face,route}` — every API request, by the route pattern it matched (`/api/v1/servers/{name}/status`, never the raw path; `unmatched` for a path no route serves). `face` is `internal` or `external`. Log streams count as requests but stay out of the latency histogram. A rise in `code=~"5.."` on one route narrows a failure to one handler; `route="unmatched"` rising is someone scanning. - The Go runtime and process series (`go_*`, `process_*`) of `felis-api`: goroutines, heap, open file descriptors. Goroutines that climb without falling back usually mean streams or uploads that never end. The API also writes one access-log line per request to its log, in logfmt: `face`, `method`, `route`, `path`, `status`, `duration_ms`, `bytes`, `request_id` (the id in every error envelope) and `principal` (the user id, once signed in). Successful probes and scrapes are left out. ```bash kubectl -n felis logs deploy/felis-api | grep 'msg=request' | grep 'status=5' kubectl -n felis logs deploy/felis-api | grep 'request_id=<id from the error>' ``` ### Scraping Loading @@ -1250,7 +1270,8 @@ annotated Service endpoints picks them up as is. `felis_build_info{component="operator"}`, and controller-runtime's `controller_runtime_reconcile_*` / `workqueue_*` series. - `felis-api` internal face `:8081/metrics` (Service `felis-api-internal`) — `felis_build_info{component="api"}`, `felis_build_info{component="api"}`, the `felis_http_*` request series, the `go_*`/`process_*` runtime series, `felis_image_build_failures_total`, and the sign-in series of §17 (`felis_mail_total`, `felis_rate_limited_total`, `felis_auth_otp_lockouts_total`, `felis_auth_failures_total`, Loading internal/api/api.go +77 −11 Changes for internal/api/api.go: 77 added lines, 11 removed lines. Original line number Diff line number Diff line Loading @@ -14,7 +14,9 @@ package api import ( "context" "log/slog" "net/http" "strings" "sync" "time" Loading Loading @@ -181,6 +183,10 @@ type API struct { // X-Forwarded-For behind an operator proxy). Empty means the TCP peer. ClientIPHeader string // AccessLog receives one line per API request (observe.go). Nil logs logfmt // to stderr. AccessLog *slog.Logger // Now is the clock, injectable for tests. Defaults to time.Now. Now func() time.Time Loading @@ -200,6 +206,26 @@ type API struct { authDoorBuckets *bucketSet mailOnce sync.Once mailBuckets *bucketSet drainInit sync.Once drainClose sync.Once drain chan struct{} } // streamsClosing is closed once CloseStreams runs. func (a *API) streamsClosing() <-chan struct{} { a.drainInit.Do(func() { a.drain = make(chan struct{}) }) return a.drain } // CloseStreams ends every log stream this API is relaying, now and from now on. // http.Server.Shutdown waits for handlers to return and cancels nothing, so a // console left open would hold the process until the pod's grace period ran out; // register this with RegisterOnShutdown. The EventSource on the other end // reconnects, and resumes from its Last-Event-ID on the next instance. func (a *API) CloseStreams() { a.streamsClosing() a.drainClose.Do(func() { close(a.drain) }) } // panelURL returns the public player-console origin ("https://console.<root>"), Loading Loading @@ -625,14 +651,14 @@ func (a *API) externalAPIRoutes() []apiRoute { // InternalHandler builds the internal-face http.Handler: service-token auth, no // Zero Trust (spec §14 red line). /healthz and /readyz are unauthenticated. func (a *API) InternalHandler() http.Handler { return a.buildFace(a.internalAPIRoutes(), a.requireInternal) return a.buildFace("internal", a.internalAPIRoutes(), a.requireInternal) } // ExternalHandler builds the external-face http.Handler: Access-JWT auth on every // /api/v1 route, with admin-tier routes additionally gated by the admin Access // path inside their handlers. func (a *API) ExternalHandler() http.Handler { return a.buildFace(a.externalAPIRoutes(), a.requireExternal) return a.buildFace("external", a.externalAPIRoutes(), a.requireExternal) } // buildFace assembles one face from its route table. Public routes are mounted Loading @@ -641,7 +667,7 @@ func (a *API) ExternalHandler() http.Handler { // adminOnly, and Owner routes in ownerOnly. Because both faces are built from the // same table the OpenAPI parity test reads, the served surface and the documented // surface cannot drift apart without failing the build. func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler { func (a *API) buildFace(face string, routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler { mux := http.NewServeMux() auth := http.NewServeMux() for _, rt := range routes { Loading @@ -651,7 +677,7 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler if rt.AuthDoor { h = a.throttleAuthDoor(h) } mux.HandleFunc(pattern, h) mux.HandleFunc(pattern, tagRoute(rt.Pattern, h)) continue } h := rt.h Loading @@ -667,22 +693,61 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler if !rt.SetupAllowed { h = a.requireOnboarded(h) } auth.HandleFunc(pattern, h) auth.HandleFunc(pattern, tagRoute(rt.Pattern, h)) } guarded := guard(auth) mux.Handle("/api/v1/", http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if _, pattern := auth.Handler(r); pattern == "" { http.NotFound(w, r) _, pattern := auth.Handler(r) if pattern == "" { writeNoRoute(w, r, mux, auth) return } // Named before the guard runs, so a refused request is counted under // the route it asked for. noteRoute(r, pattern) guarded.ServeHTTP(w, r) })) return a.baseChain(mux) top := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if _, pattern := mux.Handler(r); pattern == "" { writeNoRoute(w, r, mux, auth) return } mux.ServeHTTP(w, r) }) return a.baseChain(face, top) } // baseChain wraps a handler in the cross-cutting middleware shared by both faces: // the request id, then the access log and metrics (which see the final status of // everything inside), the response security headers, the cross-site write fence, // the request-body read deadline, and panic recovery. func (a *API) baseChain(face string, h http.Handler) http.Handler { return withRequestID(a.observe(face, withSecurityHeaders(rejectCrossSiteWrites(withBodyDeadline(withRecover(h)))))) } // baseChain wraps a handler in the cross-cutting middleware shared by both faces. func (a *API) baseChain(h http.Handler) http.Handler { return withRequestID(withRecover(h)) // writeNoRoute answers a request no route took, in the API's error envelope: 405 // with an Allow header when the path exists under other methods, 404 otherwise. // ServeMux's own answers are plain text and turn a wrong method on a guarded // route into a 404, since the guarded routes sit behind one catch-all. func writeNoRoute(w http.ResponseWriter, r *http.Request, muxes ...*http.ServeMux) { var allow []string for _, m := range []string{http.MethodGet, http.MethodPost, http.MethodPut, http.MethodPatch, http.MethodDelete} { probe := r.Clone(r.Context()) probe.Method = m for _, mux := range muxes { if _, p := mux.Handler(probe); p != "" && p != "/api/v1/" { allow = append(allow, m) break } } } if len(allow) > 0 { w.Header().Set("Allow", strings.Join(allow, ", ")) writeError(w, r, newError(http.StatusMethodNotAllowed, "method_not_allowed", "%s is not allowed here; use %s", r.Method, strings.Join(allow, ", "))) return } writeError(w, r, newError(http.StatusNotFound, "not_found", "no such endpoint")) } // requireOnboarded fences an authenticated route behind the setup-lockdown: a Loading Loading @@ -734,6 +799,7 @@ type ctxKey int const ( ctxKeyRequestID ctxKey = iota ctxKeyPrincipal ctxKeyReqInfo ) func requestIDFromContext(ctx context.Context) string { Loading Loading
cmd/felis/api.go +32 −7 Changes for cmd/felis/api.go: 32 added lines, 7 removed lines. Original line number Diff line number Diff line Loading @@ -10,6 +10,7 @@ import ( "os" "regexp" "strings" "sync" "sync/atomic" "time" Loading Loading @@ -415,15 +416,17 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int { } go reapRejectedContexts(ctx, submissions, stderr) select { case <-ctx.Done(): shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() _ = internalSrv.Shutdown(shutdownCtx) _ = externalSrv.Shutdown(shutdownCtx) servers := []*http.Server{internalSrv, externalSrv} if httpsSrv != nil { _ = httpsSrv.Shutdown(shutdownCtx) servers = append(servers, httpsSrv) } for _, srv := range servers { srv.RegisterOnShutdown(a.CloseStreams) } select { case <-ctx.Done(): shutdownServers(servers, apiShutdownGrace, stderr) return 0 case err := <-errc: if err != nil && err != http.ErrServerClosed { Loading @@ -443,8 +446,30 @@ const ( // apiIdleTimeout caps how long a kept-alive connection may sit idle between // requests before the server closes it, bounding idle-connection exhaustion. apiIdleTimeout = 120 * time.Second // apiShutdownGrace is how long the listeners drain after SIGTERM. The pod gets // the Kubernetes default of 30s before SIGKILL; this leaves the rest for the // process to exit. apiShutdownGrace = 20 * time.Second ) // shutdownServers drains every listener at once under one deadline: in turn, a // slow first listener would spend the time the others needed. Log streams end // through RegisterOnShutdown (API.CloseStreams); what is still running when the // deadline passes is cut off with the process. func shutdownServers(servers []*http.Server, grace time.Duration, stderr io.Writer) { ctx, cancel := context.WithTimeout(context.Background(), grace) defer cancel() var wg sync.WaitGroup for _, srv := range servers { wg.Go(func() { if err := srv.Shutdown(ctx); err != nil { fmt.Fprintf(stderr, "felis api: shutdown %s: %v\n", srv.Addr, err) } }) } wg.Wait() } // newAPIServer builds an http.Server with hardened header/idle timeouts (gosec // G112) shared by all three felis-api listeners (internal, external, https). // WriteTimeout and ReadTimeout are deliberately LEFT UNSET: the external and https Loading
cmd/felis/api_test.go +48 −0 Changes for cmd/felis/api_test.go: 48 added lines, 0 removed lines. Original line number Diff line number Diff line Loading @@ -4,8 +4,12 @@ import ( "context" "errors" "fmt" "io" "net" "net/http" "sync/atomic" "testing" "time" "felis.lolicon.best/internal/api" "felis.lolicon.best/internal/build" Loading Loading @@ -97,6 +101,50 @@ func TestNewAPIServerSetsHardenedTimeouts(t *testing.T) { } } // Every listener drains at once, and the shutdown hook (API.CloseStreams in // cmdAPI) runs on each: two listeners each holding a request that ends when // the hook fires take one hook's worth of time, well inside the deadline. func TestShutdownServersDrainsListenersTogether(t *testing.T) { release := make(chan struct{}) var hooks int32 started := make(chan struct{}, 2) var servers []*http.Server for range 2 { srv := newAPIServer("", http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { started <- struct{}{} <-release })) srv.RegisterOnShutdown(func() { if atomic.AddInt32(&hooks, 1) == 1 { time.AfterFunc(100*time.Millisecond, func() { close(release) }) } }) l, err := net.Listen("tcp", "127.0.0.1:0") if err != nil { t.Fatal(err) } srv.Addr = l.Addr().String() go func() { _ = srv.Serve(l) }() go func() { if resp, err := http.Get("http://" + srv.Addr); err == nil { resp.Body.Close() } }() servers = append(servers, srv) } <-started <-started begun := time.Now() shutdownServers(servers, 5*time.Second, io.Discard) if took := time.Since(begun); took > 2*time.Second { t.Fatalf("shutdown took %v", took) } if got := atomic.LoadInt32(&hooks); got != 2 { t.Fatalf("shutdown hook ran %d times, want once per listener", got) } } type fakeRefStore struct { images []build.Image builds []build.Build Loading
docs/openapi.yaml +44 −4 Changes for docs/openapi.yaml: 44 added lines, 4 removed lines. Original line number Diff line number Diff line Loading @@ -31,6 +31,28 @@ info: and the panel). Admin-tier external operations additionally require the admin Access path. See `x-felis-face` / `x-felis-tier` on each operation. Behaviour every operation shares, and so not repeated under each: * Every response carries `X-Request-Id` (a well-formed inbound one is kept), `X-Content-Type-Options: nosniff`, `X-Frame-Options: DENY`, `Referrer-Policy: no-referrer` and `Content-Security-Policy: default-src 'none'`. `Strict-Transport-Security` is added when the request came through the TLS edge (`X-Forwarded-Proto: https`). * A path no operation serves is `404 not_found`; a path served under other methods is `405 method_not_allowed` with an `Allow` header. * A POST/PUT/PATCH/DELETE a browser sends from another site (`Sec-Fetch-Site` `same-site` or `cross-site`, or an `Origin` whose host is not the request's) is `403 cross_site`, before authentication. Callers that send neither header (the plugins, scripts) are unaffected. * A JSON body over 1 MiB is `413 too_large`. A request body must keep arriving: after 30 s it has to average 16 KiB/s or the connection is closed. * Event streams (the console and build logs) tag each line with `id:` (unix seconds); an EventSource that reconnects with `Last-Event-ID` within the hour resumes from that second instead of the tailed backlog. The server re-checks the caller every minute and ends the stream with `event: revoked` once the session or the access is gone; a stream also closes after 30 minutes and on server shutdown, and the client simply reconnects. servers: - url: https://api-internal.{root_domain} description: >- Loading Loading @@ -1443,9 +1465,16 @@ paths: security: [{ accessJWT: [] }] parameters: - { name: name, in: path, required: true, schema: { type: string } } - name: Last-Event-ID in: header required: false description: The id of the last line received; resumes the stream from that second (within the hour). schema: { type: string } responses: '200': description: An event stream of log lines. description: >- An event stream of log lines (`id:` + `data:` per line, `:` comments as keep-alives), ended by `event: revoked` when access is withdrawn. content: text/event-stream: schema: { type: string } Loading Loading @@ -1871,7 +1900,11 @@ paths: get: tags: [servers] operationId: status summary: Status of your own server. summary: Status of a server; the full record for its owner and staff. description: >- Anyone signed in may ask. The owner and staff get the whole projection; anyone else gets what the game's own server list shows: name, subdomain, displayName, phase, ready, playersOnline and playersMax. x-felis-face: [external] x-felis-tier: app security: [{ accessJWT: [] }] Loading @@ -1879,7 +1912,7 @@ paths: - { name: name, in: path, required: true, schema: { type: string } } responses: '200': description: The server's status projection. description: The server's status projection (trimmed for non-owners). content: application/json: schema: { $ref: '#/components/schemas/ServerInfo' } Loading Loading @@ -4707,9 +4740,16 @@ paths: security: [{ accessJWT: [] }] parameters: - { name: id, in: path, required: true, schema: { type: string } } - name: Last-Event-ID in: header required: false description: The id of the last line received; resumes the stream from that second (within the hour). schema: { type: string } responses: '200': description: An event stream of build log lines. description: >- An event stream of build log lines (`id:` + `data:` per line), ended by `event: revoked` when the caller is no longer staff. content: text/event-stream: schema: { type: string } Loading
docs/troubleshooting.md +22 −1 Changes for docs/troubleshooting.md: 22 added lines, 1 removed line. Original line number Diff line number Diff line Loading @@ -1238,6 +1238,26 @@ All four mandated metrics have real producers; scrape them when triaging: - `felis_reaper_worlds_deleted_total` — increments only after a world PVC is actually deleted post-backup (§10); a spike here means worlds crossed the 15d idle line — cross-check that join events are flowing (§10 risk vectors). - `felis_http_requests_total{face,method,route,code}` and `felis_http_request_duration_seconds{face,route}` — every API request, by the route pattern it matched (`/api/v1/servers/{name}/status`, never the raw path; `unmatched` for a path no route serves). `face` is `internal` or `external`. Log streams count as requests but stay out of the latency histogram. A rise in `code=~"5.."` on one route narrows a failure to one handler; `route="unmatched"` rising is someone scanning. - The Go runtime and process series (`go_*`, `process_*`) of `felis-api`: goroutines, heap, open file descriptors. Goroutines that climb without falling back usually mean streams or uploads that never end. The API also writes one access-log line per request to its log, in logfmt: `face`, `method`, `route`, `path`, `status`, `duration_ms`, `bytes`, `request_id` (the id in every error envelope) and `principal` (the user id, once signed in). Successful probes and scrapes are left out. ```bash kubectl -n felis logs deploy/felis-api | grep 'msg=request' | grep 'status=5' kubectl -n felis logs deploy/felis-api | grep 'request_id=<id from the error>' ``` ### Scraping Loading @@ -1250,7 +1270,8 @@ annotated Service endpoints picks them up as is. `felis_build_info{component="operator"}`, and controller-runtime's `controller_runtime_reconcile_*` / `workqueue_*` series. - `felis-api` internal face `:8081/metrics` (Service `felis-api-internal`) — `felis_build_info{component="api"}`, `felis_build_info{component="api"}`, the `felis_http_*` request series, the `go_*`/`process_*` runtime series, `felis_image_build_failures_total`, and the sign-in series of §17 (`felis_mail_total`, `felis_rate_limited_total`, `felis_auth_otp_lockouts_total`, `felis_auth_failures_total`, Loading
internal/api/api.go +77 −11 Changes for internal/api/api.go: 77 added lines, 11 removed lines. Original line number Diff line number Diff line Loading @@ -14,7 +14,9 @@ package api import ( "context" "log/slog" "net/http" "strings" "sync" "time" Loading Loading @@ -181,6 +183,10 @@ type API struct { // X-Forwarded-For behind an operator proxy). Empty means the TCP peer. ClientIPHeader string // AccessLog receives one line per API request (observe.go). Nil logs logfmt // to stderr. AccessLog *slog.Logger // Now is the clock, injectable for tests. Defaults to time.Now. Now func() time.Time Loading @@ -200,6 +206,26 @@ type API struct { authDoorBuckets *bucketSet mailOnce sync.Once mailBuckets *bucketSet drainInit sync.Once drainClose sync.Once drain chan struct{} } // streamsClosing is closed once CloseStreams runs. func (a *API) streamsClosing() <-chan struct{} { a.drainInit.Do(func() { a.drain = make(chan struct{}) }) return a.drain } // CloseStreams ends every log stream this API is relaying, now and from now on. // http.Server.Shutdown waits for handlers to return and cancels nothing, so a // console left open would hold the process until the pod's grace period ran out; // register this with RegisterOnShutdown. The EventSource on the other end // reconnects, and resumes from its Last-Event-ID on the next instance. func (a *API) CloseStreams() { a.streamsClosing() a.drainClose.Do(func() { close(a.drain) }) } // panelURL returns the public player-console origin ("https://console.<root>"), Loading Loading @@ -625,14 +651,14 @@ func (a *API) externalAPIRoutes() []apiRoute { // InternalHandler builds the internal-face http.Handler: service-token auth, no // Zero Trust (spec §14 red line). /healthz and /readyz are unauthenticated. func (a *API) InternalHandler() http.Handler { return a.buildFace(a.internalAPIRoutes(), a.requireInternal) return a.buildFace("internal", a.internalAPIRoutes(), a.requireInternal) } // ExternalHandler builds the external-face http.Handler: Access-JWT auth on every // /api/v1 route, with admin-tier routes additionally gated by the admin Access // path inside their handlers. func (a *API) ExternalHandler() http.Handler { return a.buildFace(a.externalAPIRoutes(), a.requireExternal) return a.buildFace("external", a.externalAPIRoutes(), a.requireExternal) } // buildFace assembles one face from its route table. Public routes are mounted Loading @@ -641,7 +667,7 @@ func (a *API) ExternalHandler() http.Handler { // adminOnly, and Owner routes in ownerOnly. Because both faces are built from the // same table the OpenAPI parity test reads, the served surface and the documented // surface cannot drift apart without failing the build. func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler { func (a *API) buildFace(face string, routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler { mux := http.NewServeMux() auth := http.NewServeMux() for _, rt := range routes { Loading @@ -651,7 +677,7 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler if rt.AuthDoor { h = a.throttleAuthDoor(h) } mux.HandleFunc(pattern, h) mux.HandleFunc(pattern, tagRoute(rt.Pattern, h)) continue } h := rt.h Loading @@ -667,22 +693,61 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler if !rt.SetupAllowed { h = a.requireOnboarded(h) } auth.HandleFunc(pattern, h) auth.HandleFunc(pattern, tagRoute(rt.Pattern, h)) } guarded := guard(auth) mux.Handle("/api/v1/", http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if _, pattern := auth.Handler(r); pattern == "" { http.NotFound(w, r) _, pattern := auth.Handler(r) if pattern == "" { writeNoRoute(w, r, mux, auth) return } // Named before the guard runs, so a refused request is counted under // the route it asked for. noteRoute(r, pattern) guarded.ServeHTTP(w, r) })) return a.baseChain(mux) top := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if _, pattern := mux.Handler(r); pattern == "" { writeNoRoute(w, r, mux, auth) return } mux.ServeHTTP(w, r) }) return a.baseChain(face, top) } // baseChain wraps a handler in the cross-cutting middleware shared by both faces: // the request id, then the access log and metrics (which see the final status of // everything inside), the response security headers, the cross-site write fence, // the request-body read deadline, and panic recovery. func (a *API) baseChain(face string, h http.Handler) http.Handler { return withRequestID(a.observe(face, withSecurityHeaders(rejectCrossSiteWrites(withBodyDeadline(withRecover(h)))))) } // baseChain wraps a handler in the cross-cutting middleware shared by both faces. func (a *API) baseChain(h http.Handler) http.Handler { return withRequestID(withRecover(h)) // writeNoRoute answers a request no route took, in the API's error envelope: 405 // with an Allow header when the path exists under other methods, 404 otherwise. // ServeMux's own answers are plain text and turn a wrong method on a guarded // route into a 404, since the guarded routes sit behind one catch-all. func writeNoRoute(w http.ResponseWriter, r *http.Request, muxes ...*http.ServeMux) { var allow []string for _, m := range []string{http.MethodGet, http.MethodPost, http.MethodPut, http.MethodPatch, http.MethodDelete} { probe := r.Clone(r.Context()) probe.Method = m for _, mux := range muxes { if _, p := mux.Handler(probe); p != "" && p != "/api/v1/" { allow = append(allow, m) break } } } if len(allow) > 0 { w.Header().Set("Allow", strings.Join(allow, ", ")) writeError(w, r, newError(http.StatusMethodNotAllowed, "method_not_allowed", "%s is not allowed here; use %s", r.Method, strings.Join(allow, ", "))) return } writeError(w, r, newError(http.StatusNotFound, "not_found", "no such endpoint")) } // requireOnboarded fences an authenticated route behind the setup-lockdown: a Loading Loading @@ -734,6 +799,7 @@ type ctxKey int const ( ctxKeyRequestID ctxKey = iota ctxKeyPrincipal ctxKeyReqInfo ) func requestIDFromContext(ctx context.Context) string { Loading