Files
Felis/internal/api/handlers_logstream.go
T

139 lines
5.6 KiB
Go

package api
import (
"context"
"errors"
"net/http"
"felis.lolicon.best/internal/naming"
)
// handleServerConsole streams the caller's server console as Server-Sent Events
// (spec §8, 读写分离: 读=pods/log follow; spec §262 GET /servers/{name}/console #
// SSE). It is the read counterpart to handleCommand (写=RCON): the write side
// dials RCON, this side relays the pod log so the panel sees join/聊天/异步打印
// and live boot progress. Like the write side it is app-tier — your OWN server —
// gated by isOwnerOrAdmin.
//
// The order mirrors handleCommand up to the point of streaming, then diverges:
//
// ① name validation (400)
// ② ownership: owner or admin, else 403 (404 if the server is unknown)
// ③ the nil-Logs guard (503), so a misconstructed API fails clean, not panics
// ④ open the follow stream; map its errors to a normal JSON envelope:
// ErrNotFound → 409 not_running (no running pod to read)
// ErrConsoleUnavailable → 503 (a pod exists but its log can't be opened)
// ⑤ relay as SSE (relayLogStream), past which no error envelope is possible
//
// It deliberately does NOT pre-check readiness the way handleCommand does. A
// write only makes sense on a Ready server, but a *read* most wants the logs
// precisely while the server is booting — and §141 readiness IS an RCON probe a
// booting pod fails *while emitting the very boot logs the operator wants to
// watch*. So the read path gates on "is there a running pod" (inside StreamLogs),
// not on RCON readiness.
//
// The RCON password plays no part here at all: the read side never touches it
// (spec §286). The audit row is written BEFORE the relay, because once SSE
// framing begins the handler blocks for the lifetime of the stream — auditing
// after relayLogStream returns would mis-timestamp the attach to the detach.
func (a *API) handleServerConsole(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
// Ownership: owner or admin, mirroring handleCommand. An unknown server is 404.
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
// Logs is wired in production (cmd/felis); the nil guard only defends against a
// misconstructed API, failing as 503 rather than panicking — same contract as
// the write side's Console guard.
if a.Logs == nil {
writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable",
"console subsystem is not configured"))
return
}
// Bound concurrent SSE streams per principal BEFORE opening the follow stream, so
// an over-cap caller never even ties up a kube-apiserver connection. A stalled
// reader keeps this relay (and its upstream follow) alive indefinitely — the write
// deadline that actually severs it is a separate slice — so this cap is what stops
// one principal from accumulating unbounded leaked control-plane connections. The
// slot is held for the whole relay and released on every return path.
release, ok := a.streamGate().acquire(streamKey(p))
if !ok {
writeError(w, r, newError(http.StatusTooManyRequests, "too_many_streams",
"too many open console streams; close one and retry"))
return
}
defer release()
// Open the follow stream. Every error must be resolved HERE, into a normal JSON
// envelope, because relayLogStream commits the 200 + SSE headers and no error
// body can follow it.
ctx := r.Context()
if since, ok := logSinceFromRequest(r, a.now()); ok {
ctx = withLogSince(ctx, since)
}
src, err := a.Logs.StreamLogs(ctx, name)
switch {
case errors.Is(err, ErrConsoleUnavailable):
writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable",
"server console is currently unreachable; wake the server and retry"))
return
case errors.Is(err, ErrNotFound):
// No running pod to read — the server is stopped or not yet scheduled. This
// is the read-side analogue of handleCommand's readiness 409.
writeError(w, r, newError(http.StatusConflict, "not_running",
"server is not running; wake it before attaching to the console"))
return
case err != nil:
writeError(w, r, newError(http.StatusInternalServerError, "internal",
"could not open server console"))
return
}
a.audit(r, "console.attach", name)
relayLogStream(w, r, src, a.streamRecheck(r, func(ctx context.Context, p *Principal) error {
rec, err := a.Repo.ServerByName(ctx, name)
switch {
case errors.Is(err, ErrNotFound):
return errForbidden // the server is gone, and the grant with it
case err != nil:
return nil
case !a.isOwnerOrAdmin(p, rec):
return errForbidden
}
return nil
}), a.streamsClosing())
}
// streamRecheck builds the streamGuard for an external-face stream: it re-runs the
// authentication the stream opened with (the session may have been revoked or
// expired, the account disabled), the op.console staff gate, and then allow for the
// route's own rule. An unreachable session store is not a verdict (see streamGuard).
func (a *API) streamRecheck(r *http.Request, allow func(ctx context.Context, p *Principal) error) streamGuard {
return func(ctx context.Context) error {
p, err := a.External.Authenticate(r)
switch {
case errors.Is(err, errAuthBackend):
return nil
case err != nil || p == nil:
return errUnauthorized
case hostIsAdminConsole(r, a.RootDomain, a.AdminHostname) && !p.IsAdmin():
return errForbidden
}
return allow(ctx, p)
}
}