From b508fccc6fb6bd1dd43c15aa2634dbc68a666e90 Mon Sep 17 00:00:00 2001 From: Minseong Choi Date: Fri, 26 Jun 2026 23:32:38 +0900 Subject: [PATCH] feat(api): add felis-api service with permissions, modpack lane, and fleet read The dual-faced felis-api: internal (service) and external (public/app/admin) routes behind a Zero-Trust guard. Includes the access domain (whitelist, ban, and LuckPerms permission/group control over the owner-gated RCON path), the modpack submission endpoints, and the admin-tier SysAdmin fleet read. Structured access fields are charset-validated before assembly so no field can splice a second RCON command. --- internal/api/api.go | 387 +++++++++ internal/api/api_test.go | 868 ++++++++++++++++++++ internal/api/auth.go | 157 ++++ internal/api/cluster.go | 92 +++ internal/api/console.go | 140 ++++ internal/api/errors.go | 78 ++ internal/api/handlers_access.go | 353 ++++++++ internal/api/handlers_access_test.go | 385 +++++++++ internal/api/handlers_account.go | 148 ++++ internal/api/handlers_account_test.go | 265 ++++++ internal/api/handlers_backups.go | 142 ++++ internal/api/handlers_backups_test.go | 293 +++++++ internal/api/handlers_console.go | 122 +++ internal/api/handlers_console_test.go | 196 +++++ internal/api/handlers_create_test.go | 317 +++++++ internal/api/handlers_internal.go | 359 ++++++++ internal/api/handlers_internal_menu_test.go | 219 +++++ internal/api/handlers_internal_wake_test.go | 176 ++++ internal/api/handlers_logstream.go | 89 ++ internal/api/handlers_logstream_test.go | 480 +++++++++++ internal/api/handlers_patch_test.go | 254 ++++++ internal/api/handlers_user.go | 721 ++++++++++++++++ internal/api/images.go | 238 ++++++ internal/api/images_logs_test.go | 100 +++ internal/api/images_test.go | 265 ++++++ internal/api/k8scluster.go | 157 ++++ internal/api/logstream.go | 311 +++++++ internal/api/middleware.go | 85 ++ internal/api/openapi_test.go | 198 +++++ internal/api/pgrepo.go | 359 ++++++++ internal/api/repo.go | 142 ++++ internal/api/restore_wire_test.go | 13 + internal/api/restorer.go | 26 + internal/api/submissions.go | 187 +++++ internal/api/submissions_test.go | 300 +++++++ internal/api/util.go | 23 + 36 files changed, 8645 insertions(+) create mode 100644 internal/api/api.go create mode 100644 internal/api/api_test.go create mode 100644 internal/api/auth.go create mode 100644 internal/api/cluster.go create mode 100644 internal/api/console.go create mode 100644 internal/api/errors.go create mode 100644 internal/api/handlers_access.go create mode 100644 internal/api/handlers_access_test.go create mode 100644 internal/api/handlers_account.go create mode 100644 internal/api/handlers_account_test.go create mode 100644 internal/api/handlers_backups.go create mode 100644 internal/api/handlers_backups_test.go create mode 100644 internal/api/handlers_console.go create mode 100644 internal/api/handlers_console_test.go create mode 100644 internal/api/handlers_create_test.go create mode 100644 internal/api/handlers_internal.go create mode 100644 internal/api/handlers_internal_menu_test.go create mode 100644 internal/api/handlers_internal_wake_test.go create mode 100644 internal/api/handlers_logstream.go create mode 100644 internal/api/handlers_logstream_test.go create mode 100644 internal/api/handlers_patch_test.go create mode 100644 internal/api/handlers_user.go create mode 100644 internal/api/images.go create mode 100644 internal/api/images_logs_test.go create mode 100644 internal/api/images_test.go create mode 100644 internal/api/k8scluster.go create mode 100644 internal/api/logstream.go create mode 100644 internal/api/middleware.go create mode 100644 internal/api/openapi_test.go create mode 100644 internal/api/pgrepo.go create mode 100644 internal/api/repo.go create mode 100644 internal/api/restore_wire_test.go create mode 100644 internal/api/restorer.go create mode 100644 internal/api/submissions.go create mode 100644 internal/api/submissions_test.go create mode 100644 internal/api/util.go diff --git a/internal/api/api.go b/internal/api/api.go new file mode 100644 index 0000000..8a4376e --- /dev/null +++ b/internal/api/api.go @@ -0,0 +1,387 @@ +// Package api implements felis-api: one binary serving two faces (spec §7). +// +// The internal face (velocity / backend callbacks) authenticates with a static +// service token and is never wrapped in Zero Trust. The external face (people / +// panel) authenticates with a Cloudflare Access JWT; admin-tier operations +// additionally require the admin Access path (spec §14, graded by operation). +// +// Handlers depend on the Repo and Cluster interfaces, so the request routing, +// dual-face auth, input validation and authorization are all unit-tested with +// in-memory fakes. The Postgres (pgRepo) and controller-runtime (k8sCluster) +// implementations compile here but are exercised only by integration tests +// against a live database / cluster. +package api + +import ( + "context" + "net/http" + "sync" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" +) + +// API holds the dependencies shared by every handler. +type API struct { + Repo Repo + Cluster Cluster + Internal InternalAuth + External ExternalAuth + + // Builder is the image build subsystem (spec §16). It is optional: when nil + // the /images routes report 503 rather than 404, so the admin boundary is + // still exercised even before the subsystem is wired in. + Builder ImageBuilder + + // Console is the synchronous RCON write channel (spec §8 写=RCON). It is + // wired in production (cmd/felis); a nil Console makes the command route report + // 503 rather than panic, so the ownership boundary is still exercised in tests. + Console Console + + // Logs is the read-side console channel (spec §8 读=pods/log follow). Like + // Console it is wired in production (cmd/felis); a nil Logs makes the console + // route report 503 rather than panic, so the ownership boundary is still + // exercised in tests. + Logs LogStreamer + + // BuildLogs is the read-side build-log channel (spec §16, §416 日志流复用 §8): + // it streams the build Job's Pod log (kaniko) in the build namespace, distinct + // from Logs which streams a server pod in the minecraft namespace. Like Logs it + // is wired in production (cmd/felis); a nil BuildLogs makes the admin-only + // build-logs route report 503 rather than panic, so the admin boundary is still + // exercised in tests. + BuildLogs LogStreamer + + // Restorer starts a world restore from a backup (spec §466). It is optional: + // when nil the restore-backup route reports 503, so the restore authorization + // boundary is exercised before the restore-Job executor is wired. The list and + // authorization paths do not need it (they read the Repo), only the kick-off. + Restorer Restorer + + // Submissions is the user-modpack approval lane (a user-directed extension over + // the §16 build subsystem; see internal/submit). It is optional: when + // nil the /me/submissions and /submissions routes report 503 rather than 404, so + // the app/admin boundary is still exercised even before the subsystem is wired + // in. An ordinary user may only SUBMIT here; an admin approves before any build + // runs, distinct from Builder which an admin drives directly. + Submissions SubmissionService + + // RootDomain is injected from config (spec §2). It is the only place the + // deployment zone enters the API; hostnames are validated against it and + // never hardcoded. + RootDomain string + + // WakeCooldown throttles repeated wakes per server (spec §9.1: cooldown hangs + // on the wake lever). Zero disables throttling. + WakeCooldown time.Duration + + // MaxRunningServers caps how many servers may be desired-Running cluster-wide + // (spec §9.1: the concurrency-上限 lever hanging on the same wake chokepoint as + // cooldown and autostartPolicy). Zero — the default — disables it: §9.2 wires + // only autostartPolicy + cooldown as active wake gates, so this lever ships + // inert, exactly like a zero WakeCooldown, and a deployment opts in by setting + // a positive value. Enforced via withinRunningCap on the wake path. + MaxRunningServers int + + // Now is the clock, injectable for tests. Defaults to time.Now. + Now func() time.Time + + cooldownOnce sync.Once + cooldown *cooldownLimiter +} + +// now returns the current time using the injected clock. +func (a *API) now() time.Time { + if a.Now != nil { + return a.Now() + } + return time.Now() +} + +// limiter lazily builds the wake cooldown limiter bound to a's clock. +func (a *API) limiter() *cooldownLimiter { + a.cooldownOnce.Do(func() { + a.cooldown = &cooldownLimiter{now: a.now, last: map[string]time.Time{}} + }) + return a.cooldown +} + +// apiRoute is one served HTTP route. Each face exposes its routes as a single +// table (internalAPIRoutes / externalAPIRoutes) so that one declaration drives +// BOTH handler construction here AND the OpenAPI parity test (openapi_test.go): +// the doc is checked against the routes actually served — in both directions — +// rather than against a re-derivation, which a http.ServeMux cannot be asked to +// confirm (it exposes no way to enumerate its registered patterns). +type apiRoute struct { + Method string + Pattern string + + // Public routes are unauthenticated and mounted on the outer mux — the health + // probes, since kubelet / Cloudflare hold no token. Everything else is mounted + // behind the face's auth middleware. + Public bool + + // Admin marks an external-face route that is additionally gated on the admin + // Zero-Trust path (the adminOnly wrapper + Principal.IsAdmin() inside the + // handler). Internal-face routes never set it. + Admin bool + + h http.HandlerFunc +} + +// internalAPIRoutes is the internal face's served route table (spec §7, §14): +// service-token auth, never Zero Trust. It carries both health probes. +func (a *API) internalAPIRoutes() []apiRoute { + return []apiRoute{ + {Method: "GET", Pattern: "/healthz", Public: true, h: a.handleHealthz}, + {Method: "GET", Pattern: "/readyz", Public: true, h: a.handleReadyz}, + + {Method: "GET", Pattern: "/api/v1/servers", h: a.handleListServers}, + {Method: "GET", Pattern: "/api/v1/servers/by-host/{host}", h: a.handleByHost}, + {Method: "POST", Pattern: "/api/v1/internal/servers/{name}/ready", h: a.handleReady}, + {Method: "POST", Pattern: "/api/v1/internal/servers/{name}/join-event", h: a.handleJoinEvent}, + // Domain-autostart (spec §9.1, §14): velocity drives the wake lever and polls + // status with its service token, identifying the joining player by online-mode + // UUID. These live on the internal face because velocity holds no web Principal; + // the external face keeps its own Principal-gated wake/status for the panel. + {Method: "POST", Pattern: "/api/v1/internal/servers/{name}/wake", h: a.handleInternalWake}, + {Method: "GET", Pattern: "/api/v1/internal/servers/{name}/status", h: a.handleStatus}, + // Lobby `/menu` (spec §12): the felis-paper lobby is a pure UI face holding no + // token, so velocity drives these on its behalf — claim by online-mode UUID + // (the lobby's `Claim & Start`, separate from the autostartPolicy-gated wake) + // and the menu projection that adds the ownership-derived `claimable` the §11 + // list/status views never carry. + {Method: "POST", Pattern: "/api/v1/internal/servers/{name}/claim", h: a.handleInternalClaim}, + {Method: "GET", Pattern: "/api/v1/internal/servers/{name}/menu", h: a.handleInternalMenuStatus}, + // Account linking (spec §10): the in-game /link side mints a one-time code for a + // verified UUID. Internal-only — the code is born from an online-mode UUID the + // web never holds (account_link_codes has no user_id column). + {Method: "POST", Pattern: "/api/v1/internal/account/link/code", h: a.handleCreateLinkCode}, + } +} + +// externalAPIRoutes is the external face's served route table (spec §7, §14): +// Cloudflare Access-JWT auth on every /api/v1 route; the Admin entries are +// additionally gated on the admin Zero-Trust path. It exposes liveness only — +// readiness is an internal concern. +func (a *API) externalAPIRoutes() []apiRoute { + return []apiRoute{ + {Method: "GET", Pattern: "/healthz", Public: true, h: a.handleHealthz}, + + // App-auth tier: operations on your own servers (spec §14). + {Method: "POST", Pattern: "/api/v1/servers/{name}/wake", h: a.handleWake}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/stop", h: a.handleStop}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/claim", h: a.handleClaim}, + // Console write (spec §8 写=RCON): owner/admin-gated inside the handler, so + // it sits in the app-tier block (操作自己服 → app 鉴权), not behind adminOnly. + {Method: "POST", Pattern: "/api/v1/servers/{name}/command", h: a.handleCommand}, + // Console read (spec §8 读=pods/log follow; §262 SSE). Owner/admin-gated inside + // the handler, app-tier like the write side. + {Method: "GET", Pattern: "/api/v1/servers/{name}/console", h: a.handleServerConsole}, + // Access / permissions (spec §7): structured whitelist / ban / LuckPerms + // operations translated to RCON commands over the same owner-gated console + // path as /command. An owner manages their OWN claimed node; an admin manages + // ANY node — both via isOwnerOrAdmin inside issueAccessCommand, so these stay + // app-tier (not behind adminOnly). Each command is assembled only from + // strict-charset-validated structured fields (handlers_access.go). + {Method: "POST", Pattern: "/api/v1/servers/{name}/access/whitelist", h: a.handleAccessWhitelist}, + {Method: "GET", Pattern: "/api/v1/servers/{name}/access/whitelist", h: a.handleAccessWhitelistList}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/access/ban", h: a.handleAccessBan}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/access/permission", h: a.handleAccessPermission}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/access/group", h: a.handleAccessGroup}, + {Method: "GET", Pattern: "/api/v1/servers/{name}/status", h: a.handleStatus}, + // Identity self-read (spec §14 tiering): the panel reads this once at boot to + // learn its own tier and decide which navigation surfaces to render. App-tier — + // every authenticated principal may read its OWN identity. is_admin is the + // server-computed Principal.IsAdmin() (Role + admin Access path), so the client + // never re-derives the graded-ZT rule; it remains UX truth, not enforcement. + {Method: "GET", Pattern: "/api/v1/me", h: a.handleMe}, + {Method: "GET", Pattern: "/api/v1/me/servers", h: a.handleMyServers}, + // World backups (spec §7, §466). Both are app-tier: GET /backups is scoped + // inside the handler (admin sees all; a user sees only worlds they formerly + // owned), and restore is gated by owner-or-admin PLUS a former-owner match, so + // neither sits behind adminOnly. + {Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup}, + // Account linking (spec §10), web side: /start reports link status (it is the + // pointer handleClaim's 412 emits), /verify consumes the in-game code and binds + // the account. App-tier, not admin — linking your own account is an ordinary + // authenticated operation. + {Method: "POST", Pattern: "/api/v1/account/link/start", h: a.handleLinkStart}, + {Method: "POST", Pattern: "/api/v1/account/link/verify", h: a.handleLinkVerify}, + // Modpack submission (user-directed lane over §16), user side: a user files an upload for review + // and lists their own. App-tier — the submitter and the "my uploads" scope are + // both taken from the principal, never the body, so an ordinary authenticated + // session is the correct gate (the admin verdict lives below, behind adminOnly). + {Method: "POST", Pattern: "/api/v1/me/submissions", h: a.handleCreateSubmission}, + {Method: "GET", Pattern: "/api/v1/me/submissions", h: a.handleMySubmissions}, + // Admin (Zero-Trust) tier: create / mutate spec / image admission. These gate + // on Principal.IsAdmin() inside the handler via the adminOnly wrapper, so the + // boundary is exercised even where the body is a later-phase stub. + {Method: "POST", Pattern: "/api/v1/servers", Admin: true, h: a.handleCreateServer}, + {Method: "PATCH", Pattern: "/api/v1/servers/{name}", Admin: true, h: a.handlePatchServer}, + // SysAdmin fleet read: the whole-fleet lifecycle list for the cockpit's + // FleetTable (panel/DESIGN-WEB-3SIDES.md — a frontend extension, not a spec §7 + // route). Admin-tier — it reads every owner's server, so it gates on the admin + // Zero-Trust path, unlike the app-tier /me/servers. A path distinct from the + // internal velocity GET /api/v1/servers on purpose: the parity test forbids one + // {method, path} from carrying both the service and admin tiers. + {Method: "GET", Pattern: "/api/v1/fleet", Admin: true, h: a.handleFleet}, + // Image build + whitelist (spec §16, §15). Every route is admin-tier: a build + // is build-time RCE against the cluster, so submission requires the admin + // Zero-Trust path, not merely an authenticated session. + {Method: "POST", Pattern: "/api/v1/images/build", Admin: true, h: a.handleBuildImage}, + {Method: "GET", Pattern: "/api/v1/images/build/{id}", Admin: true, h: a.handleGetBuild}, + {Method: "GET", Pattern: "/api/v1/images/build/{id}/logs", Admin: true, h: a.handleBuildLogs}, + {Method: "POST", Pattern: "/api/v1/images/build/{id}/cancel", Admin: true, h: a.handleCancelBuild}, + {Method: "GET", Pattern: "/api/v1/images", Admin: true, h: a.handleListImages}, + {Method: "POST", Pattern: "/api/v1/images", Admin: true, h: a.handleAddImage}, + {Method: "DELETE", Pattern: "/api/v1/images", Admin: true, h: a.handleRemoveImage}, + // Modpack submission review (user-directed lane over §16), admin side: the queue of every user's + // uploads and the approve/reject verdicts. Admin-tier — listing reads other + // users' uploads and approving starts a build-time Kaniko job, so both require + // the admin Zero-Trust path, exactly like a direct /images/build. + {Method: "GET", Pattern: "/api/v1/submissions", Admin: true, h: a.handleListSubmissions}, + {Method: "POST", Pattern: "/api/v1/submissions/{id}/approve", Admin: true, h: a.handleApproveSubmission}, + {Method: "POST", Pattern: "/api/v1/submissions/{id}/reject", Admin: true, h: a.handleRejectSubmission}, + } +} + +// InternalHandler builds the internal-face http.Handler: service-token auth, no +// Zero Trust (spec §14 red line). /healthz and /readyz are unauthenticated. +func (a *API) InternalHandler() http.Handler { + return a.buildFace(a.internalAPIRoutes(), a.requireInternal) +} + +// ExternalHandler builds the external-face http.Handler: Access-JWT auth on every +// /api/v1 route, with admin-tier routes additionally gated by the admin Access +// path inside their handlers. +func (a *API) ExternalHandler() http.Handler { + return a.buildFace(a.externalAPIRoutes(), a.requireExternal) +} + +// buildFace assembles one face from its route table. Public routes are mounted +// unauthenticated on the outer mux; the rest go on an inner mux behind guard +// (requireInternal / requireExternal), with Admin routes additionally wrapped in +// adminOnly. Because both faces are built from the same table the OpenAPI parity +// test reads, the served surface and the documented surface cannot drift apart +// without failing the build. +func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler { + mux := http.NewServeMux() + auth := http.NewServeMux() + for _, rt := range routes { + pattern := rt.Method + " " + rt.Pattern + if rt.Public { + mux.HandleFunc(pattern, rt.h) + continue + } + h := rt.h + if rt.Admin { + h = a.adminOnly(rt.h) + } + auth.HandleFunc(pattern, h) + } + mux.Handle("/api/v1/", guard(auth)) + return a.baseChain(mux) +} + +// baseChain wraps a handler in the cross-cutting middleware shared by both faces. +func (a *API) baseChain(h http.Handler) http.Handler { + return withRequestID(withRecover(h)) +} + +// ---- request context plumbing ---- + +type ctxKey int + +const ( + ctxKeyRequestID ctxKey = iota + ctxKeyPrincipal +) + +func requestIDFromContext(ctx context.Context) string { + if v, ok := ctx.Value(ctxKeyRequestID).(string); ok { + return v + } + return "" +} + +// principalFromContext returns the authenticated external-face caller, or nil. +func principalFromContext(ctx context.Context) *Principal { + if v, ok := ctx.Value(ctxKeyPrincipal).(*Principal); ok { + return v + } + return nil +} + +// ---- wake cooldown ---- + +// cooldownLimiter is an in-memory per-server rate limiter for the wake lever. +// It is process-local; with multiple api replicas the effective cooldown is +// per-replica, which is acceptable because the operator reconcile is idempotent. +type cooldownLimiter struct { + mu sync.Mutex + now func() time.Time + last map[string]time.Time + window time.Duration +} + +// allow reports whether name may wake now, recording the attempt when allowed. +func (c *cooldownLimiter) allow(name string, window time.Duration) bool { + if window <= 0 { + return true + } + c.mu.Lock() + defer c.mu.Unlock() + t := c.now() + if last, ok := c.last[name]; ok && t.Sub(last) < window { + return false + } + c.last[name] = t + return true +} + +// ---- running-server cap ---- + +// withinRunningCap reports whether waking info's server is allowed under the +// global running-server cap (spec §9.1: the concurrency-上限 lever hangs on the +// same wake chokepoint as cooldown and autostartPolicy). The cap counts servers +// whose spec.desiredState is already Running — CRD lifecycle truth read through +// ListServers, never the Postgres business layer (spec §1) — and admits the wake +// only while that count stays below the cap. +// +// Two short-circuits keep it both correct and cheap: +// +// - A non-positive cap disables the lever (the default; mirrors WakeCooldown's +// "zero disables"). §9.2 wires only autostartPolicy + cooldown as active wake +// gates, so the cap ships inert and a deployment opts in by setting a positive +// value. Disabled, it never lists the cluster — the default wake path pays +// nothing for a lever nobody turned on. +// - A target already desired-Running is idempotent: re-waking it adds no load, +// so it is always admitted and need not be counted (and a stop→wake flip of a +// server that was the Nth running one is never wedged by its own slot). +// +// Like the cooldown limiter this is a soft throttle, not a transactional +// invariant: the count-then-flip is not atomic, so a burst of concurrent wakes +// can momentarily exceed the cap. That is acceptable because the operator +// reconcile is idempotent and the §18 reaper / §9.3 quota bound steady-state +// load; the cap exists to refuse an obvious flood, not to hold a hard ceiling. +func (a *API) withinRunningCap(ctx context.Context, info *ServerInfo) (bool, error) { + if a.MaxRunningServers <= 0 { + return true, nil + } + if info.DesiredState == string(v1alpha1.DesiredRunning) { + return true, nil + } + servers, err := a.Cluster.ListServers(ctx) + if err != nil { + return false, err + } + running := 0 + for _, s := range servers { + if s.DesiredState == string(v1alpha1.DesiredRunning) { + running++ + } + } + return running < a.MaxRunningServers, nil +} diff --git a/internal/api/api_test.go b/internal/api/api_test.go new file mode 100644 index 0000000..5f063e1 --- /dev/null +++ b/internal/api/api_test.go @@ -0,0 +1,868 @@ +package api + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "github.com/golang-jwt/jwt/v5" +) + +const testRoot = "mc.example.net" // neutral; never a deployment domain + +// ---- fakes ---- + +type fakeRepo struct { + bySub map[string]*ServerRecord + byName map[string]*ServerRecord + linked map[string]bool + quota map[string]bool + allowlist map[string]map[string]bool // name -> user_id -> on list (web face) + // allowUUID mirrors the UUID-keyed server_allowlist table itself, the gate the + // internal/velocity wake uses (name -> mc_uuid -> on list). allowlist above is + // the account_links-bridged web view of the same data. + allowUUID map[string]map[string]bool + mine map[string][]MyServerView + claimOK map[string]bool // name -> claim succeeds; absent name -> ErrNotFound + audits []AuditEntry + joins []string + // create-server seeding (spec §15) + seeded map[string]bool // name -> servers row exists + aliases map[string]string // subdomain -> bound server name + seedErr error + // account linking (spec §10) + linkCodes map[string]fakeLinkCode // code -> pending binding + links map[string]string // mc_uuid -> user_id (mirrors UNIQUE(mc_uuid)) + // world backups (spec §7, §22). A nil slice lists empty. + backups []fakeBackup +} + +// fakeBackup mirrors a world_backups row: the client-facing view plus the +// server-side backup_ref the list queries never expose. +type fakeBackup struct { + view BackupView + ref string +} + +// fakeLinkCode mirrors an account_link_codes row. +type fakeLinkCode struct { + mcUUID string + expiresAt time.Time +} + +func newFakeRepo() *fakeRepo { + return &fakeRepo{ + bySub: map[string]*ServerRecord{}, byName: map[string]*ServerRecord{}, + linked: map[string]bool{}, quota: map[string]bool{}, + allowlist: map[string]map[string]bool{}, allowUUID: map[string]map[string]bool{}, + mine: map[string][]MyServerView{}, + claimOK: map[string]bool{}, + seeded: map[string]bool{}, aliases: map[string]string{}, + linkCodes: map[string]fakeLinkCode{}, links: map[string]string{}, + } +} + +func (f *fakeRepo) ServerBySubdomain(_ context.Context, s string) (*ServerRecord, error) { + if r, ok := f.bySub[s]; ok { + return r, nil + } + return nil, ErrNotFound +} +func (f *fakeRepo) ServerByName(_ context.Context, n string) (*ServerRecord, error) { + if r, ok := f.byName[n]; ok { + return r, nil + } + return nil, ErrNotFound +} +func (f *fakeRepo) IsLinked(_ context.Context, u string) (bool, error) { return f.linked[u], nil } +func (f *fakeRepo) QuotaAvailable(_ context.Context, u string) (bool, error) { return f.quota[u], nil } +func (f *fakeRepo) CreateLinkCode(_ context.Context, code, mcUUID string, expiresAt time.Time) error { + f.linkCodes[code] = fakeLinkCode{mcUUID: mcUUID, expiresAt: expiresAt} + return nil +} + +// VerifyLinkCode mirrors PGRepo.VerifyLinkCode exactly so the hermetic tests +// exercise the same contract the integration impl honors: strict expiry against +// the passed clock, a different-user UUID → ErrConflict WITHOUT consuming the +// code, same (user, uuid) idempotent, and the code consumed only on success. +func (f *fakeRepo) VerifyLinkCode(_ context.Context, userID, code string, now time.Time) (string, error) { + rec, ok := f.linkCodes[code] + if !ok || !rec.expiresAt.After(now) { + return "", ErrLinkCodeInvalid + } + if existing, ok := f.links[rec.mcUUID]; ok && existing != userID { + return "", ErrConflict // do not consume another user's pending code + } + f.links[rec.mcUUID] = userID + f.linked[userID] = true + delete(f.linkCodes, code) + return rec.mcUUID, nil +} +func (f *fakeRepo) UserInAllowlist(_ context.Context, n, u string) (bool, error) { + return f.allowlist[n][u], nil +} +func (f *fakeRepo) UUIDInAllowlist(_ context.Context, n, uuid string) (bool, error) { + return f.allowUUID[n][uuid], nil +} +func (f *fakeRepo) UserByMCUUID(_ context.Context, uuid string) (string, error) { + if u, ok := f.links[uuid]; ok { + return u, nil + } + return "", ErrNotFound +} +func (f *fakeRepo) ClaimServer(_ context.Context, n, u string) (bool, error) { + ok, present := f.claimOK[n] + if !present { + return false, ErrNotFound + } + return ok, nil +} +func (f *fakeRepo) RecordJoin(_ context.Context, n, uuid string) error { + if _, ok := f.byName[n]; !ok { + return ErrNotFound + } + f.joins = append(f.joins, n+":"+uuid) + return nil +} +func (f *fakeRepo) MyServers(_ context.Context, u string) ([]MyServerView, error) { + return f.mine[u], nil +} +func (f *fakeRepo) SeedServer(_ context.Context, name, subdomain string) error { + if f.seedErr != nil { + return f.seedErr + } + if bound, ok := f.aliases[subdomain]; ok && bound != name { + return ErrConflict + } + f.seeded[name] = true + f.aliases[subdomain] = name + return nil +} +func (f *fakeRepo) Audit(_ context.Context, e AuditEntry) error { + f.audits = append(f.audits, e) + return nil +} + +// AllBackups / BackupsForUser / LatestBackup mirror the PG queries' contract so +// the hermetic tests can't pass against a too-lenient fake: only status='present' +// rows are visible, the user scope is the former_owner column, and LatestBackup +// is the newest present row for a server (or ErrNotFound). +func (f *fakeRepo) AllBackups(_ context.Context) ([]BackupView, error) { + var out []BackupView + for _, b := range f.backups { + if b.view.Status == "present" { + out = append(out, b.view) + } + } + return out, nil +} +func (f *fakeRepo) BackupsForUser(_ context.Context, userID string) ([]BackupView, error) { + var out []BackupView + for _, b := range f.backups { + if b.view.Status == "present" && b.view.FormerOwner == userID { + out = append(out, b.view) + } + } + return out, nil +} +func (f *fakeRepo) LatestBackup(_ context.Context, serverName string) (*BackupRecord, error) { + var latest *fakeBackup + for i := range f.backups { + b := &f.backups[i] + if b.view.Status != "present" || b.view.ServerName != serverName { + continue + } + if latest == nil || b.view.CreatedAt.After(latest.view.CreatedAt) { + latest = b + } + } + if latest == nil { + return nil, ErrNotFound + } + return &BackupRecord{ + ID: latest.view.ID, ServerName: latest.view.ServerName, + FormerOwner: latest.view.FormerOwner, BackupRef: latest.ref, + SizeBytes: latest.view.SizeBytes, + }, nil +} + +// fakeRestorer records the restore it was asked to start and returns a canned +// error, mirroring the Restorer kick-off contract. The real restore Job is +// integration-only, so the handler is tested against this fake (spec §466). +type fakeRestorer struct { + err error + calls int + gotName string + gotRef string +} + +func (f *fakeRestorer) Restore(_ context.Context, name, ref string) error { + f.calls++ + f.gotName, f.gotRef = name, ref + return f.err +} + +type fakeCluster struct { + byName map[string]*ServerInfo + bySub map[string]*ServerInfo + list []ServerInfo + desired map[string]v1alpha1.DesiredState + created map[string]CreateServerInput // name -> the validated input it was created from + patched map[string]ServerSpecPatch // name -> the validated spec patch it received + createErr error +} + +func newFakeCluster() *fakeCluster { + return &fakeCluster{byName: map[string]*ServerInfo{}, bySub: map[string]*ServerInfo{}, + desired: map[string]v1alpha1.DesiredState{}, created: map[string]CreateServerInput{}, + patched: map[string]ServerSpecPatch{}} +} +func (c *fakeCluster) GetServer(_ context.Context, n string) (*ServerInfo, error) { + if s, ok := c.byName[n]; ok { + return s, nil + } + return nil, ErrNotFound +} +func (c *fakeCluster) GetBySubdomain(_ context.Context, s string) (*ServerInfo, error) { + if v, ok := c.bySub[s]; ok { + return v, nil + } + return nil, ErrNotFound +} +func (c *fakeCluster) ListServers(_ context.Context) ([]ServerInfo, error) { return c.list, nil } +func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.DesiredState) error { + c.desired[n] = s + return nil +} +func (c *fakeCluster) CreateServer(_ context.Context, in CreateServerInput) error { + if c.createErr != nil { + return c.createErr + } + if _, ok := c.byName[in.Name]; ok { + return ErrConflict + } + c.created[in.Name] = in + info := &ServerInfo{Name: in.Name, Subdomain: in.Subdomain, + AutostartPolicy: string(in.AutostartPolicy), + DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)} + c.byName[in.Name] = info + c.bySub[in.Subdomain] = info + return nil +} +func (c *fakeCluster) PatchServerSpec(_ context.Context, n string, p ServerSpecPatch) error { + info, ok := c.byName[n] + if !ok { + return ErrNotFound + } + c.patched[n] = p + // Apply only the fields the lifecycle view exposes, so a follow-up read sees + // the mutation (mirrors the real merge patch touching only non-nil fields). + if p.AutostartPolicy != nil { + info.AutostartPolicy = string(*p.AutostartPolicy) + } + return nil +} + +// fakeConsole records the command it was asked to run and returns a canned reply +// or error, mirroring the Console contract. The real K8sConsole's password +// resolution and RCON dial are integration-only, so the handler is tested +// against this fake (spec §8 写=RCON). +type fakeConsole struct { + reply string + err error + calls int + gotName string + gotCommand string +} + +func (f *fakeConsole) RunCommand(_ context.Context, name, command string) (string, error) { + f.calls++ + f.gotName, f.gotCommand = name, command + if f.err != nil { + return "", f.err + } + return f.reply, nil +} + +// staticExternal injects a fixed principal so handler logic is tested without +// real JWT crypto (which is exercised separately in TestAccessVerifier). +type staticExternal struct { + p *Principal + err error +} + +func (s staticExternal) Authenticate(*http.Request) (*Principal, error) { return s.p, s.err } + +type okInternal struct{} + +func (okInternal) Authenticate(*http.Request) error { return nil } + +// ---- helpers ---- + +func newTestAPI(repo Repo, cl Cluster) *API { + return &API{Repo: repo, Cluster: cl, Internal: okInternal{}, RootDomain: testRoot, + Now: func() time.Time { return time.Unix(1_700_000_000, 0) }} +} + +func do(h http.Handler, method, target, body string, headers map[string]string) *httptest.ResponseRecorder { + var r *http.Request + if body == "" { + r = httptest.NewRequest(method, target, nil) + } else { + r = httptest.NewRequest(method, target, strings.NewReader(body)) + } + for k, v := range headers { + r.Header.Set(k, v) + } + w := httptest.NewRecorder() + h.ServeHTTP(w, r) + return w +} + +func decodeErr(t *testing.T, w *httptest.ResponseRecorder) string { + t.Helper() + var raw map[string]map[string]string + if err := json.Unmarshal(w.Body.Bytes(), &raw); err != nil { + t.Fatalf("error body not JSON: %v (%s)", err, w.Body.String()) + } + return raw["error"]["code"] +} + +// ---- dual-face separation ---- + +func TestInternalFaceRequiresServiceToken(t *testing.T) { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.Internal = BearerTokenAuth{Token: "s3cr3t"} + h := api.InternalHandler() + + // no token -> 401 + if w := do(h, "GET", "/api/v1/servers", "", nil); w.Code != http.StatusUnauthorized { + t.Fatalf("no token: code = %d, want 401", w.Code) + } + // wrong token -> 401 + if w := do(h, "GET", "/api/v1/servers", "", map[string]string{"Authorization": "Bearer nope"}); w.Code != http.StatusUnauthorized { + t.Fatalf("wrong token: code = %d, want 401", w.Code) + } + // right token -> 200 + if w := do(h, "GET", "/api/v1/servers", "", map[string]string{"Authorization": "Bearer s3cr3t"}); w.Code != http.StatusOK { + t.Fatalf("right token: code = %d, want 200", w.Code) + } +} + +func TestHealthzIsUnauthenticated(t *testing.T) { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.Internal = BearerTokenAuth{Token: "s3cr3t"} + if w := do(api.InternalHandler(), "GET", "/healthz", "", nil); w.Code != http.StatusOK { + t.Fatalf("healthz code = %d, want 200", w.Code) + } +} + +func TestExternalFaceRequiresPrincipal(t *testing.T) { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.External = staticExternal{err: http.ErrNoCookie} // any auth error + if w := do(api.ExternalHandler(), "GET", "/api/v1/me/servers", "", nil); w.Code != http.StatusUnauthorized { + t.Fatalf("code = %d, want 401", w.Code) + } +} + +// TestMeIdentity proves GET /api/v1/me reports the server-computed identity the +// panel uses to gate its Admin / SysAdmin navigation. The load-bearing assertion +// is the third subtest: is_admin tracks Principal.IsAdmin(), so the admin ROLE is +// not sufficient — the request must ALSO have arrived via the admin Access path +// (ViaAdminAccess). An admin who reached the panel through the ordinary app path +// therefore reads is_admin=false and the panel hides the admin surfaces (which the +// backend would 403 regardless). This keeps the client from re-deriving graded ZT. +func TestMeIdentity(t *testing.T) { + get := func(p *Principal) map[string]any { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.External = staticExternal{p: p} + w := do(api.ExternalHandler(), "GET", "/api/v1/me", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (body %s)", w.Code, w.Body.String()) + } + var got map[string]any + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + return got + } + + t.Run("ordinary user: is_admin false", func(t *testing.T) { + got := get(&Principal{UserID: "u1", Email: "u1@example.net", Role: "user"}) + if got["user_id"] != "u1" || got["role"] != "user" || got["is_admin"] != false { + t.Fatalf("got %v, want user_id=u1 role=user is_admin=false", got) + } + }) + + t.Run("admin via admin Access path: is_admin true", func(t *testing.T) { + got := get(&Principal{UserID: "a1", Email: "a1@example.net", Role: "admin", ViaAdminAccess: true}) + if got["role"] != "admin" || got["is_admin"] != true { + t.Fatalf("got %v, want role=admin is_admin=true", got) + } + }) + + t.Run("admin via ordinary app path: is_admin false (graded ZT)", func(t *testing.T) { + got := get(&Principal{UserID: "a1", Email: "a1@example.net", Role: "admin", ViaAdminAccess: false}) + if got["role"] != "admin" { + t.Fatalf("role = %v, want admin", got["role"]) + } + if got["is_admin"] != false { + t.Fatalf("is_admin = %v, want false — the admin role alone must not grant admin tier "+ + "without the admin Access path", got["is_admin"]) + } + }) +} + +// ---- by-host ---- + +func TestByHost(t *testing.T) { + cl := newFakeCluster() + cl.bySub["survival"] = &ServerInfo{Name: "survival", Subdomain: "survival", Phase: "Running", Ready: true} + api := newTestAPI(newFakeRepo(), cl) + h := api.InternalHandler() + tok := map[string]string{"Authorization": "Bearer "} // okInternal ignores it + + t.Run("foreign domain rejected", func(t *testing.T) { + w := do(h, "GET", "/api/v1/servers/by-host/survival.evil.example.org", "", tok) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) + t.Run("multi-label rejected", func(t *testing.T) { + w := do(h, "GET", "/api/v1/servers/by-host/a.b."+testRoot, "", tok) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) + t.Run("unknown server 404", func(t *testing.T) { + w := do(h, "GET", "/api/v1/servers/by-host/creative."+testRoot, "", tok) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) + t.Run("found", func(t *testing.T) { + w := do(h, "GET", "/api/v1/servers/by-host/survival."+testRoot, "", tok) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + var info ServerInfo + if err := json.Unmarshal(w.Body.Bytes(), &info); err != nil || info.Name != "survival" { + t.Fatalf("unexpected body %s err %v", w.Body.String(), err) + } + }) +} + +// ---- fleet (SysAdmin cockpit read) ---- + +// TestFleetAdminRead proves the SysAdmin cockpit's fleet read is admin-tier AND +// fleet-wide. Two properties distinguish it from the app-tier /me/servers: a plain +// user is rejected by adminOnly before the handler runs, and an admin sees EVERY +// server the cluster reports (CRD truth via ListServers, §1) rather than a +// caller-scoped slice. +func TestFleetAdminRead(t *testing.T) { + cl := newFakeCluster() + cl.list = []ServerInfo{ + {Name: "survival", Phase: "Running", Ready: true}, + {Name: "creative", Phase: "Stopped"}, + {Name: "skyblock", Phase: "Running", Ready: true}, + } + + t.Run("plain user forbidden", func(t *testing.T) { + api := newTestAPI(newFakeRepo(), cl) + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user"}} + w := do(api.ExternalHandler(), "GET", "/api/v1/fleet", "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403 (a plain user must not read the fleet)", w.Code) + } + }) + + t.Run("admin via ordinary app path forbidden (graded ZT)", func(t *testing.T) { + // The admin ROLE alone is not enough: without the admin Access path adminOnly + // rejects, so the cockpit read cannot be reached by an admin who arrived via + // the ordinary app face — exactly as GET /api/v1/me reports is_admin=false there. + api := newTestAPI(newFakeRepo(), cl) + api.External = staticExternal{p: &Principal{UserID: "a1", Email: "a1@example.net", + Role: "admin", ViaAdminAccess: false}} + w := do(api.ExternalHandler(), "GET", "/api/v1/fleet", "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403 (admin role without the admin Access path)", w.Code) + } + }) + + t.Run("admin reads the whole fleet", func(t *testing.T) { + api := newTestAPI(newFakeRepo(), cl) + api.External = staticExternal{p: &Principal{UserID: "a1", Email: "a1@example.net", + Role: "admin", ViaAdminAccess: true}} + w := do(api.ExternalHandler(), "GET", "/api/v1/fleet", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + var got map[string][]ServerInfo + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v", err) + } + // Fleet-wide: all three servers, not a caller-scoped subset. + if len(got["servers"]) != 3 { + t.Fatalf("servers = %d, want 3 (the fleet read must not be caller-scoped)", len(got["servers"])) + } + }) +} + +// ---- join-event ---- + +func TestJoinEvent(t *testing.T) { + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival"} + api := newTestAPI(repo, newFakeCluster()) + h := api.InternalHandler() + + t.Run("missing uuid 400", func(t *testing.T) { + w := do(h, "POST", "/api/v1/internal/servers/survival/join-event", `{}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) + t.Run("records join", func(t *testing.T) { + w := do(h, "POST", "/api/v1/internal/servers/survival/join-event", + `{"mc_uuid":"11111111-1111-1111-1111-111111111111"}`, nil) + if w.Code != http.StatusNoContent { + t.Fatalf("code = %d, want 204 (%s)", w.Code, w.Body.String()) + } + if len(repo.joins) != 1 { + t.Fatalf("expected 1 recorded join, got %d", len(repo.joins)) + } + }) + t.Run("unknown server 404", func(t *testing.T) { + w := do(h, "POST", "/api/v1/internal/servers/missing/join-event", + `{"mc_uuid":"11111111-1111-1111-1111-111111111111"}`, nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) +} + +// ---- claim (§9.3) ---- + +func TestClaimStateMachine(t *testing.T) { + user := &Principal{UserID: "u1", Email: "u1@example.net", Role: "user", ViaAdminAccess: false} + + t.Run("not linked -> 412", func(t *testing.T) { + repo := newFakeRepo() + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/claim", "", nil) + if w.Code != http.StatusPreconditionFailed || decodeErr(t, w) != "not_linked" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("over quota -> 403", func(t *testing.T) { + repo := newFakeRepo() + repo.linked["u1"] = true + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/claim", "", nil) + if w.Code != http.StatusForbidden || decodeErr(t, w) != "quota_exceeded" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("already claimed -> 409", func(t *testing.T) { + repo := newFakeRepo() + repo.linked["u1"] = true + repo.quota["u1"] = true + repo.claimOK["survival"] = false // row exists but owner already set + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/claim", "", nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "already_claimed" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("success -> 200 + audit", func(t *testing.T) { + repo := newFakeRepo() + repo.linked["u1"] = true + repo.quota["u1"] = true + repo.claimOK["survival"] = true + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/claim", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "claim" || repo.audits[0].Actor != "u1@example.net" { + t.Fatalf("audit not written as expected: %+v", repo.audits) + } + }) +} + +// ---- wake (autostartPolicy gate + cooldown) ---- + +func TestWakeAutostartGate(t *testing.T) { + mk := func(policy string) (*API, *fakeCluster) { + repo := newFakeRepo() + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: policy} + api := newTestAPI(repo, cl) + return api, cl + } + stranger := &Principal{UserID: "stranger", Role: "user"} + + t.Run("public: any user wakes", func(t *testing.T) { + api, cl := mk("public") + api.External = staticExternal{p: stranger} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desiredState = %q, want Running", cl.desired["survival"]) + } + }) + t.Run("ownerOnly: stranger forbidden", func(t *testing.T) { + api, cl := mk("ownerOnly") + api.External = staticExternal{p: stranger} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if _, set := cl.desired["survival"]; set { + t.Fatal("desiredState must not change on a forbidden wake") + } + }) + t.Run("ownerOnly: owner wakes", func(t *testing.T) { + api, _ := mk("ownerOnly") + api.Repo.(*fakeRepo).byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + api.External = staticExternal{p: &Principal{UserID: "owner1", Role: "user"}} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("allowlist: only listed user wakes", func(t *testing.T) { + api, _ := mk("allowlist") + repo := api.Repo.(*fakeRepo) + repo.allowlist["survival"] = map[string]bool{"friend": true} + api.External = staticExternal{p: &Principal{UserID: "friend", Role: "user"}} + if w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusAccepted { + t.Fatalf("listed user: code = %d", w.Code) + } + api.External = staticExternal{p: stranger} + if w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusForbidden { + t.Fatalf("stranger: code = %d, want 403", w.Code) + } + }) +} + +func TestWakeCooldown(t *testing.T) { + repo := newFakeRepo() + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public"} + api := newTestAPI(repo, cl) + api.WakeCooldown = time.Minute + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user"}} + h := api.ExternalHandler() + + if w := do(h, "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusAccepted { + t.Fatalf("first wake code = %d", w.Code) + } + // clock is frozen, so the second wake is inside the cooldown window + if w := do(h, "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusTooManyRequests { + t.Fatalf("second wake code = %d, want 429", w.Code) + } +} + +// TestWakeRunningCap exercises the §9.1 cluster-wide concurrency lever on the +// external wake path. The cap counts CRD truth via ListServers (never Postgres, +// per §1) and is a default-off lever: MaxRunningServers <= 0 disables it exactly +// as a zero WakeCooldown disables the per-server throttle. A full cluster answers +// 503 at_capacity — distinct from the cooldown's 429 — and an already-Running +// target re-wakes idempotently regardless of the cap. +func TestWakeRunningCap(t *testing.T) { + // capAPI builds an external-face API whose cluster already holds `running` + // servers desired-Running plus a stopped "survival" target, with the cap set. + capAPI := func(cap, running int) (*API, *fakeCluster) { + cl := newFakeCluster() + target := &ServerInfo{Name: "survival", AutostartPolicy: "public", + DesiredState: string(v1alpha1.DesiredStopped)} + cl.byName["survival"] = target + cl.list = []ServerInfo{*target} + for i := 0; i < running; i++ { + cl.list = append(cl.list, ServerInfo{Name: fmt.Sprintf("running-%d", i), + DesiredState: string(v1alpha1.DesiredRunning)}) + } + api := newTestAPI(newFakeRepo(), cl) + api.MaxRunningServers = cap + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user"}} + return api, cl + } + + t.Run("at cap: wake rejected with 503 at_capacity", func(t *testing.T) { + api, cl := capAPI(2, 2) + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("code = %d, want 503", w.Code) + } + if code := decodeErr(t, w); code != "at_capacity" { + t.Fatalf("error code = %q, want at_capacity", code) + } + if _, set := cl.desired["survival"]; set { + t.Fatal("desiredState must not change when the cluster is at capacity") + } + }) + + t.Run("below cap: wake accepted", func(t *testing.T) { + api, cl := capAPI(3, 2) + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202", w.Code) + } + if cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desired = %q, want Running", cl.desired["survival"]) + } + }) + + t.Run("cap disabled (0): wake accepted even when many run", func(t *testing.T) { + api, _ := capAPI(0, 5) + if w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202", w.Code) + } + }) + + t.Run("already-Running target re-wakes despite a full cap (idempotent)", func(t *testing.T) { + api, cl := capAPI(2, 2) + cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning) + if w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202 (idempotent re-wake)", w.Code) + } + }) +} + +// ---- Zero-Trust admin boundary (§14) ---- + +func TestAdminBoundary(t *testing.T) { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + + t.Run("user role rejected before handler", func(t *testing.T) { + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user", ViaAdminAccess: false}} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", `{}`, nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + }) + t.Run("admin role without admin access rejected", func(t *testing.T) { + api.External = staticExternal{p: &Principal{UserID: "a", Role: "admin", ViaAdminAccess: false}} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", `{}`, nil) + if w.Code != http.StatusForbidden { + t.Fatalf("role=admin but panel path: code = %d, want 403", w.Code) + } + }) + t.Run("admin via admin-access reaches handler", func(t *testing.T) { + api.External = staticExternal{p: &Principal{UserID: "a", Role: "admin", ViaAdminAccess: true}} + // The body is empty so the real handler rejects it (validation 400 with no + // Builder → 503), but the point is that the admin Zero-Trust path is NOT + // stopped at the 403 boundary — it reaches the handler. + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", `{}`, nil) + if w.Code == http.StatusForbidden { + t.Fatalf("admin via admin-access must reach the handler, got 403") + } + }) +} + +// ---- error envelope ---- + +func TestErrorEnvelopeHasRequestID(t *testing.T) { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user"}} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/status", "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + if w.Header().Get("X-Request-Id") == "" { + t.Fatal("missing X-Request-Id response header") + } + var raw map[string]map[string]string + if err := json.Unmarshal(w.Body.Bytes(), &raw); err != nil { + t.Fatalf("body not JSON: %v", err) + } + if raw["error"]["request_id"] == "" { + t.Fatal("error envelope missing request_id") + } +} + +// ---- real AccessVerifier (JWT aud) ---- + +func TestAccessVerifier(t *testing.T) { + key := []byte("test-signing-key") + keyfunc := func(*jwt.Token) (any, error) { return key, nil } + v := AccessVerifier{Audience: "felis-app", AdminAudience: "felis-admin", Keyfunc: keyfunc} + + sign := func(claims accessClaims) string { + tok := jwt.NewWithClaims(jwt.SigningMethodHS256, claims) + s, err := tok.SignedString(key) + if err != nil { + t.Fatalf("sign: %v", err) + } + return s + } + exp := jwt.NewNumericDate(time.Now().Add(time.Hour)) + + t.Run("valid app token", func(t *testing.T) { + s := sign(accessClaims{Email: "u@example.net", RegisteredClaims: jwt.RegisteredClaims{ + Subject: "u1", Audience: jwt.ClaimStrings{"felis-app"}, ExpiresAt: exp}}) + r := httptest.NewRequest("GET", "/", nil) + r.Header.Set("Authorization", "Bearer "+s) + p, err := v.Authenticate(r) + if err != nil { + t.Fatalf("authenticate: %v", err) + } + if p.UserID != "u1" || p.Email != "u@example.net" || p.Role != "user" || p.ViaAdminAccess { + t.Fatalf("unexpected principal %+v", p) + } + }) + t.Run("admin audience sets ViaAdminAccess", func(t *testing.T) { + s := sign(accessClaims{Role: "admin", RegisteredClaims: jwt.RegisteredClaims{ + Subject: "a1", Audience: jwt.ClaimStrings{"felis-app", "felis-admin"}, ExpiresAt: exp}}) + r := httptest.NewRequest("GET", "/", nil) + r.Header.Set("Cf-Access-Jwt-Assertion", s) + p, err := v.Authenticate(r) + if err != nil { + t.Fatalf("authenticate: %v", err) + } + if !p.IsAdmin() { + t.Fatalf("expected admin principal, got %+v", p) + } + }) + t.Run("wrong audience rejected", func(t *testing.T) { + s := sign(accessClaims{RegisteredClaims: jwt.RegisteredClaims{ + Subject: "u1", Audience: jwt.ClaimStrings{"someone-else"}, ExpiresAt: exp}}) + r := httptest.NewRequest("GET", "/", nil) + r.Header.Set("Authorization", "Bearer "+s) + if _, err := v.Authenticate(r); err == nil { + t.Fatal("expected audience rejection") + } + }) + t.Run("wrong signing key rejected", func(t *testing.T) { + tok := jwt.NewWithClaims(jwt.SigningMethodHS256, accessClaims{RegisteredClaims: jwt.RegisteredClaims{ + Subject: "u1", Audience: jwt.ClaimStrings{"felis-app"}, ExpiresAt: exp}}) + s, _ := tok.SignedString([]byte("attacker-key")) + r := httptest.NewRequest("GET", "/", nil) + r.Header.Set("Authorization", "Bearer "+s) + if _, err := v.Authenticate(r); err == nil { + t.Fatal("expected signature rejection") + } + }) + t.Run("missing expiry rejected", func(t *testing.T) { + s := sign(accessClaims{RegisteredClaims: jwt.RegisteredClaims{ + Subject: "u1", Audience: jwt.ClaimStrings{"felis-app"}}}) + r := httptest.NewRequest("GET", "/", nil) + r.Header.Set("Authorization", "Bearer "+s) + if _, err := v.Authenticate(r); err == nil { + t.Fatal("expected missing-expiry rejection") + } + }) +} diff --git a/internal/api/auth.go b/internal/api/auth.go new file mode 100644 index 0000000..c606294 --- /dev/null +++ b/internal/api/auth.go @@ -0,0 +1,157 @@ +package api + +import ( + "crypto/subtle" + "fmt" + "net/http" + "strings" + + "github.com/golang-jwt/jwt/v5" +) + +// Principal is the authenticated external-face caller (spec §7, §14). The +// internal face (service token) never produces a Principal — it is a trusted +// machine caller, not a person. +type Principal struct { + // UserID is the stable web identity (SSO subject → users.id). + UserID string + // Email is the audited actor identity (spec §14: audit actor = Access email). + Email string + // Role is "admin" or "user" (mirrors users.role). + Role string + // ViaAdminAccess is true only when the request arrived through the admin.* + // Zero-Trust hostname (Cloudflare Access). Admin-tier operations require it + // in addition to Role=="admin" (spec §14: ZT is graded by operation). + ViaAdminAccess bool +} + +// IsAdmin reports whether the principal may perform admin-tier operations. +// Both the role claim and the admin Access path are required: a role=admin +// session arriving on panel.* must not bypass the Zero-Trust boundary. +func (p *Principal) IsAdmin() bool { + return p != nil && p.Role == "admin" && p.ViaAdminAccess +} + +// InternalAuth authenticates the internal face (velocity / backend callbacks): +// a static service token presented as a Bearer credential. The internal face +// is never wrapped in Zero Trust (spec §1.8, §14 red line). +type InternalAuth interface { + Authenticate(r *http.Request) error +} + +// ExternalAuth authenticates the external face (people / panel) and returns the +// resolved Principal. Production verifies a Cloudflare Access JWT and checks its +// audience; the verification key source (JWKS) is injected so the audience and +// expiry logic stay unit-testable. +type ExternalAuth interface { + Authenticate(r *http.Request) (*Principal, error) +} + +// BearerTokenAuth is the production InternalAuth: a constant-time comparison +// against the configured service token. A zero token fails closed so a +// misconfiguration can never silently disable internal-face auth. +type BearerTokenAuth struct { + Token string +} + +// Authenticate checks the Authorization: Bearer header against the token. +func (b BearerTokenAuth) Authenticate(r *http.Request) error { + if b.Token == "" { + return fmt.Errorf("internal auth not configured") + } + got := bearerToken(r) + if got == "" { + return fmt.Errorf("missing bearer token") + } + if subtle.ConstantTimeCompare([]byte(got), []byte(b.Token)) != 1 { + return fmt.Errorf("invalid service token") + } + return nil +} + +// AccessVerifier is the production ExternalAuth: it parses a Cloudflare Access +// JWT, verifies the signature with the injected key function, and enforces the +// configured audience (spec §7 "验 aud"). AdminAudience, when set, marks a token +// minted for the admin.* application so admin-tier routes can require it. +type AccessVerifier struct { + // Audience is the required `aud` claim for any external request. + Audience string + // AdminAudience, if non-empty and present in the token's aud set, flags the + // principal as having passed the admin Zero-Trust path. + AdminAudience string + // Keyfunc resolves the signing key (production: a JWKS-backed keyfunc). + Keyfunc jwt.Keyfunc +} + +// accessClaims are the subset of Access JWT claims we consume. +type accessClaims struct { + Email string `json:"email"` + Role string `json:"felis_role"` + jwt.RegisteredClaims +} + +// Authenticate verifies the Access JWT and maps it onto a Principal. +func (v AccessVerifier) Authenticate(r *http.Request) (*Principal, error) { + if v.Keyfunc == nil { + return nil, fmt.Errorf("external auth not configured") + } + raw := accessToken(r) + if raw == "" { + return nil, fmt.Errorf("missing access token") + } + + var claims accessClaims + parser := jwt.NewParser(jwt.WithExpirationRequired()) + if _, err := parser.ParseWithClaims(raw, &claims, v.Keyfunc); err != nil { + return nil, fmt.Errorf("invalid access token: %w", err) + } + + // Audience check: the configured app aud must be present. We do not delegate + // to jwt.WithAudience so we can additionally detect the admin audience. + if !audienceContains(claims.Audience, v.Audience) { + return nil, fmt.Errorf("token audience does not include %q", v.Audience) + } + if claims.Subject == "" { + return nil, fmt.Errorf("token missing subject") + } + + role := claims.Role + if role == "" { + role = "user" + } + return &Principal{ + UserID: claims.Subject, + Email: claims.Email, + Role: role, + ViaAdminAccess: v.AdminAudience != "" && audienceContains(claims.Audience, v.AdminAudience), + }, nil +} + +// bearerToken extracts a Bearer credential from the Authorization header. +func bearerToken(r *http.Request) string { + const prefix = "Bearer " + h := r.Header.Get("Authorization") + if len(h) > len(prefix) && strings.EqualFold(h[:len(prefix)], prefix) { + return strings.TrimSpace(h[len(prefix):]) + } + return "" +} + +// accessToken prefers the Cloudflare Access assertion header, falling back to a +// Bearer credential so the same verifier works behind a proxy or directly. +func accessToken(r *http.Request) string { + if h := r.Header.Get("Cf-Access-Jwt-Assertion"); h != "" { + return h + } + return bearerToken(r) +} + +// audienceContains reports whether want appears in the aud claim set. +func audienceContains(aud jwt.ClaimStrings, want string) bool { + for _, a := range aud { + if a == want { + return true + } + } + return false +} diff --git a/internal/api/cluster.go b/internal/api/cluster.go new file mode 100644 index 0000000..8e65deb --- /dev/null +++ b/internal/api/cluster.go @@ -0,0 +1,92 @@ +package api + +import ( + "context" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + corev1 "k8s.io/api/core/v1" +) + +// ServerInfo is the lifecycle view read from the MinecraftServer CRD + status +// (spec §4). The CRD is the source-of-truth; the API mutates its spec only via +// the app-tier desiredState lever (wake/stop, spec §9.1) and the admin-tier spec +// patch (spec §7 PATCH /servers/{name}) — never the business-layer fields, which +// live in Postgres (spec §22). +type ServerInfo struct { + Name string `json:"name"` + Subdomain string `json:"subdomain"` + Phase string `json:"phase"` + Ready bool `json:"ready"` + AutostartPolicy string `json:"autostartPolicy,omitempty"` + DesiredState string `json:"desiredState,omitempty"` + EndpointMode string `json:"endpointMode,omitempty"` + EndpointAddress string `json:"endpointAddress,omitempty"` + PlayersOnline int32 `json:"playersOnline"` + PlayersMax int32 `json:"playersMax"` +} + +// CreateServerInput is the validated, structured create-server form (spec §15). +// felis-api has already enforced naming, reservation, image whitelist, quota +// policy, and the §22 memory ceiling before this reaches the cluster: there is +// no free-form YAML path — every field is a typed, validated value. A created +// server starts DesiredState=Stopped and unowned (claimed later, spec §9.3). +type CreateServerInput struct { + Name string + Subdomain string + DisplayName string + Image string + JavaMemory string + StorageSize string + AutostartPolicy v1alpha1.AutostartPolicy + // Resources is the fully-resolved pod resource block. Its memory limit is the + // §22 ceiling — felis-api guarantees it is non-zero (the operator does not + // derive a cgroup limit from JavaMemory). + Resources corev1.ResourceRequirements +} + +// ServerSpecPatch is the resolved, validated set of admin-tier mutations applied +// to one MinecraftServer spec (spec §7 PATCH /servers/{name}). felis-api has +// already validated naming, admitted any new image against the whitelist, parsed +// the policy, and derived the §22 memory ceiling before this reaches the cluster: +// there is no free-form YAML path. Every field is a pointer — nil means "leave +// unchanged", so the patch touches only the fields the admin explicitly set. It +// deliberately omits the dual-write routing identity (subdomain, the object name) +// and the world PVC size, which felis-api rejects before constructing this so the +// CRD and Postgres never desync (spec §22). +type ServerSpecPatch struct { + DisplayName *string + AutostartPolicy *v1alpha1.AutostartPolicy + Image *string + // JavaMemory is the re-derived JVM heap string; Resources carries the matching + // pod block whose memory limit is the non-zero §22 ceiling. They move together + // (felis-api resolves both from the same form) or both stay nil. + JavaMemory *string + Resources *corev1.ResourceRequirements +} + +// Cluster is the lifecycle-layer access the API depends on: reads of the +// MinecraftServer CRD, the app-tier desiredState lever (spec §9.1), the admin-tier +// create (spec §15), and the admin-tier spec patch (spec §7). It is an interface +// so handlers are tested against a fake; the controller-runtime implementation +// (k8sCluster) is integration-tested only — it requires a live cluster. +type Cluster interface { + // GetServer reads one MinecraftServer's lifecycle view, or ErrNotFound. + GetServer(ctx context.Context, name string) (*ServerInfo, error) + // GetBySubdomain finds the MinecraftServer whose spec.subdomain matches, or + // ErrNotFound. + GetBySubdomain(ctx context.Context, subdomain string) (*ServerInfo, error) + // ListServers returns the lifecycle view of every MinecraftServer, for the + // velocity registration pull (spec §7 GET /servers). + ListServers(ctx context.Context) ([]ServerInfo, error) + // SetDesiredState flips spec.desiredState — the only write the API performs + // against the CRD (spec §9.1). It is idempotent. + SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error + // CreateServer creates a MinecraftServer CRD from the validated form (spec + // §15). It returns ErrConflict if a server of that name already exists. + CreateServer(ctx context.Context, in CreateServerInput) error + // PatchServerSpec applies an admin-tier spec mutation (spec §7 PATCH + // /servers/{name}): only the non-nil fields of the patch are written, via a + // merge patch so a concurrent operator status write is never clobbered. It + // returns ErrNotFound if no server of that name exists. + PatchServerSpec(ctx context.Context, name string, patch ServerSpecPatch) error +} diff --git a/internal/api/console.go b/internal/api/console.go new file mode 100644 index 0000000..6722732 --- /dev/null +++ b/internal/api/console.go @@ -0,0 +1,140 @@ +package api + +import ( + "context" + "fmt" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/rcon" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/types" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +// Console is the synchronous RCON write channel the API depends on (spec §8, +// 读写分离: 写=RCON). RunCommand sends one command to the named server's RCON +// endpoint and returns the server's reply. It returns ErrNotFound if no server +// of that name exists, and ErrConsoleUnavailable if the RCON channel cannot be +// reached (dial timeout / refused / auth rejected) — which, because readiness +// IS an RCON probe (spec §141), is the expected outcome when the server isn't +// truly Running. The RCON password is resolved internally and is NEVER part of +// any argument or return value (spec §286: RCON 密码绝不下发前端). +// +// It is an interface so handlers are tested against a fake (api_test.go); the +// controller-runtime implementation (K8sConsole) is integration-tested only. +type Console interface { + RunCommand(ctx context.Context, name, command string) (string, error) +} + +// K8sConsole is the production Console: it resolves the per-server RCON password +// from the Secret named by the CRD's spec.rcon.secretRef, dials the in-cluster +// RCON endpoint, and runs one command (spec §8 写=RCON). It mirrors the +// operator's readiness prober — the same address convention and the same +// secret-read path as internal/operator (rconAddress / reconciler.rconPassword) +// — so the single reviewed way to reach a server's RCON is the only way the API +// reaches it too. +// +// INTEGRATION-ONLY: like K8sCluster / pgRepo this needs a live cluster and a +// reachable RCON port; it compiles here but is exercised only by integration +// tests against a real cluster, never by the hermetic api_test.go suite. The +// Oracle verifies the handler layer (handleCommand) against a fake Console. +// +// Security: the resolved password authenticates the dial and is never logged +// nor placed in any return value — only Execute's reply body (the command +// output) flows back to the caller (spec §286). Port 25575 is reachable only +// from felis-api by NetworkPolicy (spec §8), so this dial is the single +// sanctioned write path. +type K8sConsole struct { + c client.Client + namespace string + timeout time.Duration +} + +// NewK8sConsole builds a Console over c, scoped to namespace. +func NewK8sConsole(c client.Client, namespace string) *K8sConsole { + return &K8sConsole{c: c, namespace: namespace, timeout: 5 * time.Second} +} + +// RunCommand reads the server CRD, resolves its RCON password, dials the +// in-cluster RCON endpoint, and runs command. Every unreachable/auth/secret +// failure collapses to ErrConsoleUnavailable (the handler maps it to 503) so no +// driver detail — and certainly no password — ever reaches the caller. +func (k *K8sConsole) RunCommand(ctx context.Context, name, command string) (string, error) { + var ms v1alpha1.MinecraftServer + if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil { + if apierrors.IsNotFound(err) { + return "", ErrNotFound + } + return "", err + } + if !ms.Spec.Rcon.Enabled { + // No RCON means no write channel at all (spec §8). + return "", ErrConsoleUnavailable + } + + password, err := k.rconPassword(ctx, &ms) + if err != nil { + // A missing/garbled secret is a server misconfiguration, but to the caller + // it still means the console cannot be reached — and the underlying error + // must not leak. Surface it as unavailable. + return "", ErrConsoleUnavailable + } + + timeout := k.timeout + if timeout <= 0 { + timeout = 5 * time.Second + } + if dl, ok := ctx.Deadline(); ok { + if remaining := time.Until(dl); remaining > 0 && remaining < timeout { + timeout = remaining + } + } + + conn, err := rcon.Dial(rconEndpoint(&ms), password, timeout) + if err != nil { + // Dial refused / timed out / auth rejected: the channel is not reachable. + // Per spec §141 readiness IS this probe, so this is the expected "not + // actually up" outcome — surfaced as 503, not 500. + return "", ErrConsoleUnavailable + } + defer conn.Close() + + out, err := conn.Execute(command) + if err != nil { + return "", ErrConsoleUnavailable + } + return out, nil +} + +// rconEndpoint is the in-cluster RCON address for a server, matching the +// operator's convention (internal/operator builders.rconAddress): the headless +// client Service is named after the server, in its own namespace. +func rconEndpoint(server *v1alpha1.MinecraftServer) string { + port := server.Spec.Rcon.Port + if port <= 0 { + port = rcon.DefaultPort + } + return fmt.Sprintf("%s.%s.svc.cluster.local:%d", server.Name, server.Namespace, port) +} + +// rconPassword resolves the RCON password from the Secret named by the CRD's +// spec.rcon.secretRef. It mirrors internal/operator reconciler.rconPassword +// exactly so the API reads the credential the same reviewed way the operator +// does. The returned value is used solely to authenticate the dial. +func (k *K8sConsole) rconPassword(ctx context.Context, server *v1alpha1.MinecraftServer) (string, error) { + ref := server.Spec.Rcon.SecretRef + if ref.Name == "" || ref.Key == "" { + return "", fmt.Errorf("rcon.secretRef.name and .key are required when rcon is enabled") + } + var secret corev1.Secret + if err := k.c.Get(ctx, types.NamespacedName{Namespace: server.Namespace, Name: ref.Name}, &secret); err != nil { + return "", err + } + b, ok := secret.Data[ref.Key] + if !ok { + return "", fmt.Errorf("secret %q has no key %q", ref.Name, ref.Key) + } + return string(b), nil +} diff --git a/internal/api/errors.go b/internal/api/errors.go new file mode 100644 index 0000000..5e64b11 --- /dev/null +++ b/internal/api/errors.go @@ -0,0 +1,78 @@ +package api + +import ( + "encoding/json" + "errors" + "fmt" + "net/http" +) + +// Sentinel errors the repository and cluster layers return so handlers can map +// domain outcomes onto HTTP status codes without leaking driver details. +var ( + // ErrNotFound means the requested server / record does not exist. + ErrNotFound = errors.New("not found") + // ErrConflict means an atomic precondition failed (e.g. claim lost the race). + ErrConflict = errors.New("conflict") + // ErrLinkCodeInvalid means an account-link code is unknown or expired (spec + // §10). It is a client error (the verify endpoint exists; the code is bad), so + // handlers map it to 400, not 404. + ErrLinkCodeInvalid = errors.New("link code invalid or expired") + // ErrConsoleUnavailable means the RCON write channel could not be reached — + // the dial timed out, was refused, or the password was rejected (spec §8). + // Because readiness IS an RCON probe (spec §141: phase=Running ⟺ RCON + // answers), a reachable failure here is a transient/racy "the server isn't + // actually up", not a server bug. Handlers map it to 503, not 500, so the + // caller is told to wake/retry rather than shown an opaque internal error. + ErrConsoleUnavailable = errors.New("server console is unavailable") +) + +// apiError is a handler-level error carrying an HTTP status and a stable, +// machine-readable code. The error envelope matches the platform convention: +// +// {"error": {"code": "...", "message": "...", "request_id": "..."}} +type apiError struct { + status int + code string + msg string +} + +func (e *apiError) Error() string { return e.msg } + +// newError builds an apiError with a formatted message. +func newError(status int, code, format string, a ...any) *apiError { + return &apiError{status: status, code: code, msg: fmt.Sprintf(format, a...)} +} + +// Common errors reused across handlers. +var ( + errUnauthorized = newError(http.StatusUnauthorized, "unauthorized", "authentication required") + errForbidden = newError(http.StatusForbidden, "forbidden", "not permitted") + errBadRequest = newError(http.StatusBadRequest, "bad_request", "invalid request") +) + +// writeJSON writes v as an indented JSON body with the given status. +func writeJSON(w http.ResponseWriter, status int, v any) { + w.Header().Set("Content-Type", "application/json; charset=utf-8") + w.WriteHeader(status) + enc := json.NewEncoder(w) + enc.SetEscapeHTML(false) + _ = enc.Encode(v) +} + +// writeError renders err as the standard error envelope. Non-apiError values +// collapse to a 500 so driver/internal details never reach the client. +func writeError(w http.ResponseWriter, r *http.Request, err error) { + var ae *apiError + if !errors.As(err, &ae) { + ae = newError(http.StatusInternalServerError, "internal", "internal error") + } + body := map[string]any{ + "error": map[string]string{ + "code": ae.code, + "message": ae.msg, + "request_id": requestIDFromContext(r.Context()), + }, + } + writeJSON(w, ae.status, body) +} diff --git a/internal/api/handlers_access.go b/internal/api/handlers_access.go new file mode 100644 index 0000000..257ad83 --- /dev/null +++ b/internal/api/handlers_access.go @@ -0,0 +1,353 @@ +package api + +import ( + "errors" + "fmt" + "net/http" + "regexp" + "strings" + + "felis.lolicon.best/internal/naming" +) + +// Access / permissions domain (spec §7). These endpoints let an owner manage +// who may join and what they may do on their OWN claimed node, and let an admin +// do the same on ANY node, by translating a small set of STRUCTURED fields into +// the live server's runtime authority — vanilla's whitelist/ban and the +// LuckPerms plugin — over the exact same owner-gated RCON path as POST +// /servers/{name}/command (handleCommand). +// +// Source-of-truth note (spec §1): the live MC server + LuckPerms is a THIRD +// authority, neither the CRD (lifecycle) nor Postgres (business). felis-api is +// only a command-issuer and read-projector here; it stores none of this state. +// +// The security difference from handleCommand is the whole point of this file. +// handleCommand validates ONE free-form line and forbids control characters so a +// newline cannot smuggle a second command. Here the command is ASSEMBLED from +// structured fields (player, node, world, group); a space or newline in any +// field would splice a second RCON command just the same. So every field is +// validated against a strict allow-list charset BEFORE it is ever concatenated +// into a command, and there is NO free-text field anywhere (a ban carries no +// reason string — that would be the one free-text injection vector). The +// anchored allow-list regexes are strictly stronger than handleCommand's +// control-character scan: Go's `$` is `\z` (absolute end, not `\Z`), so a +// trailing newline cannot sit before it and "player\n…" is rejected outright. +var ( + // mcNameRe matches a Java-edition username: 1–16 of [A-Za-z0-9_]. No space, + // separator, or control character can appear, so a validated name is safe to + // concatenate directly into an RCON command word. + mcNameRe = regexp.MustCompile(`^[A-Za-z0-9_]{1,16}$`) + // lpNodeRe matches a LuckPerms permission node: dotted segments with the + // wildcard, e.g. "essentials.fly" or "worldedit.*". Deliberately excludes + // space/`=`/`/` so a node can never carry a second token or a `world=` context. + lpNodeRe = regexp.MustCompile(`^[A-Za-z0-9_.*-]{1,64}$`) + // lpCtxRe matches a world name or a LuckPerms group: [A-Za-z0-9_-], 1–48. Used + // for both the optional `world=` context and a parent group. + lpCtxRe = regexp.MustCompile(`^[A-Za-z0-9_-]{1,48}$`) +) + +var ( + errInvalidPlayer = newError(http.StatusBadRequest, "bad_request", + "invalid player name (1–16 chars: letters, digits, underscore)") + errInvalidAction = newError(http.StatusBadRequest, "bad_request", "unknown action") + errInvalidNode = newError(http.StatusBadRequest, "bad_request", + "invalid permission node (allowed: letters, digits, . _ - *)") + errInvalidWorld = newError(http.StatusBadRequest, "bad_request", + "invalid world (allowed: letters, digits, _ -)") + errInvalidGroup = newError(http.StatusBadRequest, "bad_request", + "invalid group (allowed: letters, digits, _ -)") +) + +// issueAccessCommand is the shared spine of every §7 access mutation: resolve the +// named server, enforce owner-or-admin, require readiness, and run ONE +// already-validated RCON command, returning its reply. It centralises the +// gate / readiness / console-failure surface in exactly one reviewed place, +// mirroring handleCommand step-for-step, so each handler's only job is to +// validate its structured fields and assemble the command string. +// +// It writes the HTTP error and returns ok=false on any failure, so a caller just +// `return`s. It does NOT audit — the caller audits with a structured action +// label (e.g. "access.whitelist.add") so the trail records intent, not a raw +// "console.command". The RCON password is resolved inside the Console +// implementation and never appears in `command`, the reply, or any log (§286). +// +// command MUST be assembled only from charset-validated fields; a space or +// newline in it would splice a second RCON command. `name` (the one field from +// the path, not the body) is validated here. +func (a *API) issueAccessCommand(w http.ResponseWriter, r *http.Request, name, command string) (string, bool) { + p := principalFromContext(r.Context()) + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return "", false + } + + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return "", false + } + if !a.isOwnerOrAdmin(p, rec) { + writeError(w, r, errForbidden) + return "", false + } + + // Readiness pre-check: RCON cannot reach a stopped server (§141 Ready ⟺ RCON + // reachable), so changing access requires a Running node. This is a specific + // 409 rather than a blind dial; the RunCommand below still maps an unreachable + // channel to 503 because the invariant can drop between here and the dial. + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return "", false + } + if !info.Ready { + writeError(w, r, newError(http.StatusConflict, "not_running", + "server is not running; wake it before changing access")) + return "", false + } + + if a.Console == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable", + "console subsystem is not configured")) + return "", false + } + + out, err := a.Console.RunCommand(r.Context(), name, command) + switch { + case errors.Is(err, ErrConsoleUnavailable): + writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable", + "server console is currently unreachable; wake the server and retry")) + return "", false + case err != nil: + a.writeLookupError(w, r, err) + return "", false + } + return out, true +} + +// whitelistRequest is the body of POST .../access/whitelist (allow / disallow a +// player to join, spec §7). decodeJSON rejects unknown fields so no extra knob +// can smuggle in. +type whitelistRequest struct { + Action string `json:"action"` // add | remove + Player string `json:"player"` +} + +// handleAccessWhitelist adds or removes a player from the live whitelist via +// "whitelist add|remove ". App-tier, owner/admin-gated inside +// issueAccessCommand. +func (a *API) handleAccessWhitelist(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + var body whitelistRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + if !mcNameRe.MatchString(body.Player) { + writeError(w, r, errInvalidPlayer) + return + } + switch body.Action { + case "add", "remove": + default: + writeError(w, r, errInvalidAction) + return + } + + out, ok := a.issueAccessCommand(w, r, name, "whitelist "+body.Action+" "+body.Player) + if !ok { + return + } + a.audit(r, principalFromContext(r.Context()).Email, "access.whitelist."+body.Action, name) + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, "action": body.Action, "player": body.Player, "output": out, + }) +} + +// handleAccessWhitelistList is the read projector for the whitelist: it runs +// "whitelist list" and returns a best-effort parse PLUS the raw reply. The parse +// is vanilla-specific (INTEGRATION-ONLY against a real server); the raw output is +// always returned so the client has ground truth when the format differs. +func (a *API) handleAccessWhitelistList(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + out, ok := a.issueAccessCommand(w, r, name, "whitelist list") + if !ok { + return + } + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, "players": parseWhitelistOutput(out), "output": out, + }) +} + +// banRequest is the body of POST .../access/ban (deny / restore a player's +// ability to join, spec §7). It carries NO reason field on purpose: a free-text +// reason would be the one place a structured request could splice a second RCON +// command, and it buys nothing the audit log does not already record. +type banRequest struct { + Action string `json:"action"` // ban | pardon + Player string `json:"player"` +} + +// handleAccessBan bans or pardons a player via "ban|pardon ". +func (a *API) handleAccessBan(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + var body banRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + if !mcNameRe.MatchString(body.Player) { + writeError(w, r, errInvalidPlayer) + return + } + switch body.Action { + case "ban", "pardon": + default: + writeError(w, r, errInvalidAction) + return + } + + out, ok := a.issueAccessCommand(w, r, name, body.Action+" "+body.Player) + if !ok { + return + } + a.audit(r, principalFromContext(r.Context()).Email, "access.ban."+body.Action, name) + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, "action": body.Action, "player": body.Player, "output": out, + }) +} + +// permissionRequest is the body of POST .../access/permission: a fine-grained +// LuckPerms permission grant/deny on a single node, optionally scoped to a world +// (spec §7 细致的权限调整 + world 范围). +// +// Value is a *bool, NOT a bool, and this matters: a plain bool's zero value is +// false, so an OMITTED value would silently mean "permission set false", +// which is an explicit LuckPerms DENY — the exact opposite of the grant a caller +// who omits the field intends. nil therefore means "default to true (grant)"; +// an explicit false is a deliberate deny. +type permissionRequest struct { + Action string `json:"action"` // set | unset + Player string `json:"player"` + Node string `json:"node"` + Value *bool `json:"value,omitempty"` // set only; nil => true (grant) + World string `json:"world,omitempty"` // optional context; "" => global +} + +// handleAccessPermission sets or unsets a LuckPerms permission node for a player, +// optionally within a world context: +// +// set: lp user permission set [world=] +// unset: lp user permission unset [world=] +func (a *API) handleAccessPermission(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + var body permissionRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + if !mcNameRe.MatchString(body.Player) { + writeError(w, r, errInvalidPlayer) + return + } + if !lpNodeRe.MatchString(body.Node) { + writeError(w, r, errInvalidNode) + return + } + // World is optional: validate ONLY when present, else an omitted world would + // fail the charset check and the optional field would become mandatory. + if body.World != "" && !lpCtxRe.MatchString(body.World) { + writeError(w, r, errInvalidWorld) + return + } + + var cmd string + switch body.Action { + case "set": + value := true // nil => grant; see permissionRequest.Value. + if body.Value != nil { + value = *body.Value + } + cmd = fmt.Sprintf("lp user %s permission set %s %t", body.Player, body.Node, value) + case "unset": + cmd = fmt.Sprintf("lp user %s permission unset %s", body.Player, body.Node) + default: + writeError(w, r, errInvalidAction) + return + } + if body.World != "" { + cmd += " world=" + body.World + } + + out, ok := a.issueAccessCommand(w, r, name, cmd) + if !ok { + return + } + a.audit(r, principalFromContext(r.Context()).Email, "access.permission."+body.Action, name) + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, "action": body.Action, "player": body.Player, + "node": body.Node, "output": out, + }) +} + +// groupRequest is the body of POST .../access/group: add or remove a LuckPerms +// parent group for a player (spec §7 给其他玩家权限的管理 via group membership). +type groupRequest struct { + Action string `json:"action"` // add | remove + Player string `json:"player"` + Group string `json:"group"` +} + +// handleAccessGroup adds or removes a player's LuckPerms parent group via +// "lp user parent add|remove ". +func (a *API) handleAccessGroup(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + var body groupRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + if !mcNameRe.MatchString(body.Player) { + writeError(w, r, errInvalidPlayer) + return + } + if !lpCtxRe.MatchString(body.Group) { + writeError(w, r, errInvalidGroup) + return + } + switch body.Action { + case "add", "remove": + default: + writeError(w, r, errInvalidAction) + return + } + + out, ok := a.issueAccessCommand(w, r, name, "lp user "+body.Player+" parent "+body.Action+" "+body.Group) + if !ok { + return + } + a.audit(r, principalFromContext(r.Context()).Email, "access.group."+body.Action, name) + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, "action": body.Action, "player": body.Player, "group": body.Group, "output": out, + }) +} + +// parseWhitelistOutput extracts player names from vanilla's "whitelist list" +// reply, whose format is "There are N whitelisted player(s): a, b, c" (and "There +// are no whitelisted players" / a trailing colon for the empty case). The parse +// is best-effort and vanilla-specific — the raw reply is always returned +// alongside, so a different format (a plugin, a localised or future server) never +// loses information. Returns a non-nil empty slice so the JSON renders [] not null. +func parseWhitelistOutput(out string) []string { + players := []string{} + i := strings.LastIndex(out, ":") + if i < 0 { + return players + } + for _, part := range strings.Split(out[i+1:], ",") { + if p := strings.TrimSpace(part); p != "" { + players = append(players, p) + } + } + return players +} diff --git a/internal/api/handlers_access_test.go b/internal/api/handlers_access_test.go new file mode 100644 index 0000000..a61a239 --- /dev/null +++ b/internal/api/handlers_access_test.go @@ -0,0 +1,385 @@ +package api + +import ( + "encoding/json" + "net/http" + "testing" +) + +// mkAccess builds an API whose "survival" server is owned by owner1 and Ready, +// with a fresh fakeConsole wired — the shared fixture for the §7 access tests. +func mkAccess(t *testing.T) (*API, *fakeRepo, *fakeCluster, *fakeConsole) { + t.Helper() + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Running", Ready: true} + console := &fakeConsole{reply: "ok"} + api := newTestAPI(repo, cl) + api.Console = console + return api, repo, cl, console +} + +var accessOwner = &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"} + +// TestAccessTranslation pins the structured-field → RCON-command translation for +// every §7 action. The exact gotCommand is the contract LuckPerms / vanilla will +// receive, so each variant that hides a bug is checked: a permission grant with +// an OMITTED value must become "... true" (not the *bool zero "... false", which +// would be a silent DENY), an explicit false must stay false, and an optional +// world must append " world=" only when present. +func TestAccessTranslation(t *testing.T) { + cases := []struct { + name string + path string + body string + wantCmd string + wantLbl string // audit action label + }{ + {"whitelist add", "/api/v1/servers/survival/access/whitelist", + `{"action":"add","player":"Steve"}`, "whitelist add Steve", "access.whitelist.add"}, + {"whitelist remove", "/api/v1/servers/survival/access/whitelist", + `{"action":"remove","player":"Steve"}`, "whitelist remove Steve", "access.whitelist.remove"}, + {"ban", "/api/v1/servers/survival/access/ban", + `{"action":"ban","player":"Griefer_99"}`, "ban Griefer_99", "access.ban.ban"}, + {"pardon", "/api/v1/servers/survival/access/ban", + `{"action":"pardon","player":"Griefer_99"}`, "pardon Griefer_99", "access.ban.pardon"}, + // The *bool trap: omitted value defaults to true (grant), NOT false (deny). + {"permission set default grant", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials.fly"}`, + "lp user Steve permission set essentials.fly true", "access.permission.set"}, + {"permission set explicit false", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials.fly","value":false}`, + "lp user Steve permission set essentials.fly false", "access.permission.set"}, + {"permission set with world", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials.fly","world":"nether","value":true}`, + "lp user Steve permission set essentials.fly true world=nether", "access.permission.set"}, + {"permission unset", "/api/v1/servers/survival/access/permission", + `{"action":"unset","player":"Steve","node":"worldedit.*"}`, + "lp user Steve permission unset worldedit.*", "access.permission.unset"}, + {"permission unset with world", "/api/v1/servers/survival/access/permission", + `{"action":"unset","player":"Steve","node":"essentials.fly","world":"nether"}`, + "lp user Steve permission unset essentials.fly world=nether", "access.permission.unset"}, + {"group add", "/api/v1/servers/survival/access/group", + `{"action":"add","player":"Steve","group":"vip"}`, "lp user Steve parent add vip", "access.group.add"}, + {"group remove", "/api/v1/servers/survival/access/group", + `{"action":"remove","player":"Steve","group":"vip"}`, "lp user Steve parent remove vip", "access.group.remove"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + api, repo, _, console := mkAccess(t) + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", tc.path, tc.body, nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if console.calls != 1 { + t.Fatalf("console calls = %d, want 1", console.calls) + } + if console.gotName != "survival" || console.gotCommand != tc.wantCmd { + t.Fatalf("console saw (%q,%q), want (survival,%q)", console.gotName, console.gotCommand, tc.wantCmd) + } + if len(repo.audits) != 1 || repo.audits[0].Action != tc.wantLbl || repo.audits[0].Actor != "owner1@example.net" { + t.Fatalf("audit = %+v, want action %q", repo.audits, tc.wantLbl) + } + }) + } +} + +// TestAccessInjectionRejected is the security verification: a space, newline, or +// other off-charset character in ANY structured field must be rejected with 400 +// AND must never reach RCON. Without per-field validation each of these inputs +// would splice a second command into the assembled line — so this matrix, not the +// happy-path gotCommand assert, is what proves the injection defence. +func TestAccessInjectionRejected(t *testing.T) { + cases := []struct { + name, path, body string + }{ + // player field, across the routes that take one. + {"whitelist player space", "/api/v1/servers/survival/access/whitelist", `{"action":"add","player":"ev il"}`}, + {"whitelist player newline", "/api/v1/servers/survival/access/whitelist", `{"action":"add","player":"ev\nop x"}`}, + {"ban player semicolon", "/api/v1/servers/survival/access/ban", `{"action":"ban","player":"ev;il"}`}, + {"ban player space", "/api/v1/servers/survival/access/ban", `{"action":"ban","player":"ev il"}`}, + {"permission player space", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"ev il","node":"essentials.fly"}`}, + {"group player newline", "/api/v1/servers/survival/access/group", + `{"action":"add","player":"ev\nx","group":"vip"}`}, + // node field. + {"permission node space", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials fly"}`}, + {"permission node newline", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials.fly\nop x"}`}, + {"permission node equals", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"a=b"}`}, + // world field (optional — but when present, still validated). + {"permission world space", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials.fly","world":"ne ther"}`}, + {"permission world newline", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"essentials.fly","world":"a\nb"}`}, + // group field. + {"group group space", "/api/v1/servers/survival/access/group", + `{"action":"add","player":"Steve","group":"vi p"}`}, + {"group group newline", "/api/v1/servers/survival/access/group", + `{"action":"add","player":"Steve","group":"a\nb"}`}, + // Trailing / embedded control characters: the anchored regex uses Go's `$` + // (= \z, absolute end), so even a SINGLE trailing newline after an otherwise + // valid name is rejected — a Perl `$` (\Z) would have let "Steve\n" through. + {"whitelist player trailing newline", "/api/v1/servers/survival/access/whitelist", + `{"action":"add","player":"Steve\n"}`}, + {"whitelist player carriage return", "/api/v1/servers/survival/access/whitelist", + `{"action":"add","player":"Steve\r"}`}, + {"whitelist player tab", "/api/v1/servers/survival/access/whitelist", + `{"action":"add","player":"Ste\tve"}`}, + // Length bounds: a name past 16 chars / node past 64 chars must be rejected, + // not truncated, so an over-long field can never carry a smuggled tail. + {"whitelist player over 16 chars", "/api/v1/servers/survival/access/whitelist", + `{"action":"add","player":"AAAAAAAAAAAAAAAAA"}`}, + {"permission node over 64 chars", "/api/v1/servers/survival/access/permission", + `{"action":"set","player":"Steve","node":"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"}`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", tc.path, tc.body, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d body %s, want 400", w.Code, w.Body.String()) + } + if console.calls != 0 { + t.Fatalf("an off-charset field must not reach RCON (calls = %d)", console.calls) + } + }) + } +} + +// TestAccessUnknownAction rejects an action enum value no handler understands, +// before any RCON contact. +func TestAccessUnknownAction(t *testing.T) { + cases := []struct{ name, path, body string }{ + {"whitelist", "/api/v1/servers/survival/access/whitelist", `{"action":"frobnicate","player":"Steve"}`}, + {"ban", "/api/v1/servers/survival/access/ban", `{"action":"frobnicate","player":"Steve"}`}, + {"permission", "/api/v1/servers/survival/access/permission", + `{"action":"frobnicate","player":"Steve","node":"essentials.fly"}`}, + {"group", "/api/v1/servers/survival/access/group", `{"action":"frobnicate","player":"Steve","group":"vip"}`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", tc.path, tc.body, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + if console.calls != 0 { + t.Fatalf("an unknown action must not reach RCON (calls = %d)", console.calls) + } + }) + } +} + +// TestAccessUnknownField proves decodeJSON's strict mode locks the body shape for +// every mutating access route — no extra knob can ride in. +func TestAccessUnknownField(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/access/whitelist", + `{"action":"add","player":"Steve","extra":1}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + if console.calls != 0 { + t.Fatal("a body with an unknown field must not reach RCON") + } +} + +// TestAccessOwnerGate proves the shared issueAccessCommand gate: an owner and an +// admin pass (on their own / on any node), a non-owner non-admin is rejected with +// 403 and never reaches RCON. The gate is the same one handleCommand uses, so one +// route exercises it for all five. +func TestAccessOwnerGate(t *testing.T) { + body := `{"action":"add","player":"Steve"}` + path := "/api/v1/servers/survival/access/whitelist" + + t.Run("owner -> 200", func(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", path, body, nil) + if w.Code != http.StatusOK || console.calls != 1 { + t.Fatalf("code = %d calls = %d", w.Code, console.calls) + } + }) + + t.Run("admin on another's node -> 200", func(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "admin1@example.net", + Role: "admin", ViaAdminAccess: true}} + w := do(api.ExternalHandler(), "POST", path, body, nil) + if w.Code != http.StatusOK || console.calls != 1 { + t.Fatalf("code = %d calls = %d", w.Code, console.calls) + } + }) + + t.Run("non-owner -> 403, no RCON call", func(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}} + w := do(api.ExternalHandler(), "POST", path, body, nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if console.calls != 0 { + t.Fatal("a forbidden caller must not reach RCON") + } + }) +} + +// TestAccessReadinessAndFailures covers the non-happy lifecycle states the shared +// helper maps: unknown server → 404, stopped server → 409 (no RCON), unreachable +// console → 503 (no audit), nil console → 503. +func TestAccessReadinessAndFailures(t *testing.T) { + body := `{"action":"add","player":"Steve"}` + path := "/api/v1/servers/survival/access/whitelist" + + t.Run("unknown server -> 404", func(t *testing.T) { + api, _, _, _ := mkAccess(t) + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/missing/access/whitelist", body, nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) + + t.Run("not running -> 409 not_running, no RCON call", func(t *testing.T) { + api, _, cl, console := mkAccess(t) + cl.byName["survival"].Ready = false + cl.byName["survival"].Phase = "Stopped" + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", path, body, nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "not_running" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if console.calls != 0 { + t.Fatal("a non-ready server must not reach RCON") + } + }) + + t.Run("console unreachable -> 503, no audit", func(t *testing.T) { + api, repo, _, console := mkAccess(t) + console.err = ErrConsoleUnavailable + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", path, body, nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "console_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if len(repo.audits) != 0 { + t.Fatalf("a failed command must not audit: %+v", repo.audits) + } + }) + + t.Run("nil Console -> 503", func(t *testing.T) { + api, _, _, _ := mkAccess(t) + api.Console = nil + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "POST", path, body, nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "console_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) +} + +// TestAccessWhitelistList exercises the read projector: GET returns the parsed +// player list plus the raw reply, and is owner-gated like the writes. +func TestAccessWhitelistList(t *testing.T) { + t.Run("parses players and returns raw output", func(t *testing.T) { + api, repo, _, console := mkAccess(t) + console.reply = "There are 2 whitelisted player(s): Steve, Alex" + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/access/whitelist", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + var resp struct { + Name string `json:"name"` + Players []string `json:"players"` + Output string `json:"output"` + } + if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + if resp.Name != "survival" || resp.Output != console.reply { + t.Fatalf("unexpected response %+v", resp) + } + if len(resp.Players) != 2 || resp.Players[0] != "Steve" || resp.Players[1] != "Alex" { + t.Fatalf("players = %#v, want [Steve Alex]", resp.Players) + } + if console.gotCommand != "whitelist list" { + t.Fatalf("console got %q, want %q", console.gotCommand, "whitelist list") + } + // A read must never write the audit trail — only the mutating routes audit. + if len(repo.audits) != 0 { + t.Fatalf("GET whitelist must not audit: %+v", repo.audits) + } + }) + + t.Run("empty whitelist -> [] not null", func(t *testing.T) { + api, repo, _, console := mkAccess(t) + console.reply = "There are no whitelisted players" + api.External = staticExternal{p: accessOwner} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/access/whitelist", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + // players must render as [] (non-nil) so the client never sees null. + var raw map[string]json.RawMessage + if err := json.Unmarshal(w.Body.Bytes(), &raw); err != nil { + t.Fatalf("body not JSON: %v", err) + } + if string(raw["players"]) != "[]" { + t.Fatalf("players = %s, want []", raw["players"]) + } + if len(repo.audits) != 0 { + t.Fatalf("GET whitelist must not audit: %+v", repo.audits) + } + }) + + t.Run("non-owner -> 403, no RCON call", func(t *testing.T) { + api, _, _, console := mkAccess(t) + api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/access/whitelist", "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if console.calls != 0 { + t.Fatal("a forbidden caller must not reach RCON") + } + }) +} + +// TestParseWhitelistOutput unit-tests the vanilla parser directly, including the +// formats issueAccessCommand never produces but a real server might. +func TestParseWhitelistOutput(t *testing.T) { + cases := []struct { + in string + want []string + }{ + {"There are 2 whitelisted player(s): Steve, Alex", []string{"Steve", "Alex"}}, + {"There are 1 whitelisted player(s): Steve", []string{"Steve"}}, + {"There are no whitelisted players", []string{}}, + {"", []string{}}, + {"There are 0 whitelisted player(s):", []string{}}, + {"Names: a, b ,c", []string{"a", "b", "c"}}, + } + for _, tc := range cases { + got := parseWhitelistOutput(tc.in) + if got == nil { + t.Fatalf("parseWhitelistOutput(%q) = nil, want non-nil slice", tc.in) + } + if len(got) != len(tc.want) { + t.Fatalf("parseWhitelistOutput(%q) = %#v, want %#v", tc.in, got, tc.want) + } + for i := range got { + if got[i] != tc.want[i] { + t.Fatalf("parseWhitelistOutput(%q)[%d] = %q, want %q", tc.in, i, got[i], tc.want[i]) + } + } + } +} diff --git a/internal/api/handlers_account.go b/internal/api/handlers_account.go new file mode 100644 index 0000000..759e665 --- /dev/null +++ b/internal/api/handlers_account.go @@ -0,0 +1,148 @@ +package api + +import ( + "crypto/rand" + "errors" + "net/http" + "strings" + "time" +) + +// Account-linking endpoints (spec §10). The flow is forced by the +// account_link_codes schema, which carries mc_uuid but no user_id: +// +// 游戏内 /link → 生成一次性码 (internal: the in-game side has the verified UUID) +// → 玩家拿码 → 网页 verify 填码 (external: the web side has the logged-in user) +// → 写 account_links +// +// So code generation is internal-face and verification is external-face. A web +// endpoint cannot mint a code — it has no verified UUID to mint against — which +// is exactly what the schema (mc_uuid NOT NULL, no user_id) encodes. Verification +// is the load-bearing step: success there flips IsLinked true and unblocks every +// ownership operation (claim, §9.3), which otherwise dead-ends at a 412. + +const ( + // linkCodeTTL bounds how long a freshly minted code is accepted (spec §10: + // 短 TTL). Long enough to alt-tab from the game to the panel, short enough that + // a leaked code is useless minutes later. + linkCodeTTL = 10 * time.Minute + // linkCodeAlphabet is a 32-symbol set with the visually ambiguous characters + // I, O, 0 and 1 removed, so a player can read a code off chat and type it on the + // panel without confusion. 32 divides 256 evenly, so a uniform random byte + // reduced mod 32 is itself uniform — no modulo bias, no rejection sampling. + linkCodeAlphabet = "ABCDEFGHJKLMNPQRSTUVWXYZ23456789" + // linkCodeLen is the symbol count: a 32^8 ≈ 1.1e12 keyspace, far beyond brute + // force inside the TTL. + linkCodeLen = 8 +) + +// newLinkCode returns a cryptographically random, unambiguous link code. +func newLinkCode() (string, error) { + buf := make([]byte, linkCodeLen) + if _, err := rand.Read(buf); err != nil { + return "", err + } + for i, b := range buf { + buf[i] = linkCodeAlphabet[int(b)%len(linkCodeAlphabet)] + } + return string(buf), nil +} + +// createLinkCodeRequest is the in-game /link callback body (spec §10): the +// backend reports the verified UUID of the player who ran the command. +type createLinkCodeRequest struct { + MCUUID string `json:"mc_uuid"` +} + +// handleCreateLinkCode mints a one-time link code for a verified in-game UUID +// (spec §10, internal face). It is the server side of the in-game /link command: +// the backend has already established the UUID via online-mode auth, so the code +// is born bound to a trustworthy identity. The player carries the returned code +// to the panel and submits it to the external verify endpoint. +func (a *API) handleCreateLinkCode(w http.ResponseWriter, r *http.Request) { + var req createLinkCodeRequest + if err := decodeJSON(w, r, &req); err != nil { + writeError(w, r, err) + return + } + if req.MCUUID == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "mc_uuid is required")) + return + } + code, err := newLinkCode() + if err != nil { + writeError(w, r, err) + return + } + expiresAt := a.now().Add(linkCodeTTL) + if err := a.Repo.CreateLinkCode(r.Context(), code, req.MCUUID, expiresAt); err != nil { + writeError(w, r, err) + return + } + writeJSON(w, http.StatusCreated, map[string]any{ + "code": code, + "expires_at": expiresAt.UTC(), + }) +} + +// linkVerifyRequest is the panel verify-code body (spec §10): the logged-in user +// submits the code they were shown in-game. +type linkVerifyRequest struct { + Code string `json:"code"` +} + +// handleLinkVerify consumes a link code for the authenticated user and writes the +// account_links binding (spec §10, external app face). This is the load-bearing +// step of §10. The code is trimmed and uppercased so a player who typed it with +// stray spaces or in lowercase still matches the minted value. Outcomes: +// invalid/expired code → 400 invalid_code; the UUID already linked to a different +// user → 409 already_linked; otherwise the binding is written and IsLinked +// becomes true for this user. +func (a *API) handleLinkVerify(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + var req linkVerifyRequest + if err := decodeJSON(w, r, &req); err != nil { + writeError(w, r, err) + return + } + code := strings.ToUpper(strings.TrimSpace(req.Code)) + if code == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "code is required")) + return + } + mcUUID, err := a.Repo.VerifyLinkCode(r.Context(), p.UserID, code, a.now()) + switch { + case errors.Is(err, ErrLinkCodeInvalid): + writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "link code is invalid or expired")) + return + case errors.Is(err, ErrConflict): + writeError(w, r, newError(http.StatusConflict, "already_linked", + "that Minecraft account is already linked to another user")) + return + case err != nil: + writeError(w, r, err) + return + } + a.audit(r, p.Email, "account.link", "") + writeJSON(w, http.StatusOK, map[string]any{"linked": true, "mc_uuid": mcUUID}) +} + +// handleLinkStart reports the caller's link status and how to link (spec §10, +// external app face). It is the endpoint handleClaim's 412 points at. It +// deliberately does NOT mint a code: a code is born in-game (account_link_codes +// has no user_id column), so the web can only report status and relay the in-game +// instruction — minting here would contradict the schema. This keeps the pointer +// in handleClaim honest without pretending the web can originate a binding. +func (a *API) handleLinkStart(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + linked, err := a.Repo.IsLinked(r.Context(), p.UserID) + if err != nil { + writeError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{ + "linked": linked, + "instructions": "Run /link in-game to receive a one-time code, then submit it to " + + "POST /api/v1/account/link/verify.", + }) +} diff --git a/internal/api/handlers_account_test.go b/internal/api/handlers_account_test.go new file mode 100644 index 0000000..1f6afb5 --- /dev/null +++ b/internal/api/handlers_account_test.go @@ -0,0 +1,265 @@ +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" +) + +// acctBody decodes a success body into a generic object for field assertions. +func acctBody(t *testing.T, w *httptest.ResponseRecorder) map[string]any { + t.Helper() + var m map[string]any + if err := json.Unmarshal(w.Body.Bytes(), &m); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + return m +} + +// TestAccountLinkVertical walks the whole §10 flow across both faces and proves +// the point of the slice: linking reconnects the claim vertical that handleClaim +// (handlers_user.go:100) otherwise dead-ends with 412 not_linked. +func TestAccountLinkVertical(t *testing.T) { + const mcUUID = "11111111-1111-1111-1111-111111111111" + user := &Principal{UserID: "u1", Email: "u1@example.net", Role: "user"} + + repo := newFakeRepo() + // Pre-arm the downstream claim gates so a *successful* claim becomes possible + // the instant the link gate clears — that is what demonstrates the unblock. + repo.quota["u1"] = true + repo.claimOK["survival"] = true + + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + ih := api.InternalHandler() + eh := api.ExternalHandler() + + // Before linking, the claim dead-ends at the link gate. + if w := do(eh, "POST", "/api/v1/servers/survival/claim", "", nil); w.Code != http.StatusPreconditionFailed || decodeErr(t, w) != "not_linked" { + t.Fatalf("pre-link claim: code = %d body %s, want 412 not_linked", w.Code, w.Body.String()) + } + + // 1) in-game /link mints a one-time code (internal face). + w := do(ih, "POST", "/api/v1/internal/account/link/code", `{"mc_uuid":"`+mcUUID+`"}`, nil) + if w.Code != http.StatusCreated { + t.Fatalf("mint code: code = %d, want 201 (%s)", w.Code, w.Body.String()) + } + code, _ := acctBody(t, w)["code"].(string) + if len(code) != linkCodeLen { + t.Fatalf("minted code %q: len = %d, want %d", code, len(code), linkCodeLen) + } + + // 2) the player submits the code on the panel (external face). + w = do(eh, "POST", "/api/v1/account/link/verify", `{"code":"`+code+`"}`, nil) + if w.Code != http.StatusOK { + t.Fatalf("verify: code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if b := acctBody(t, w); b["linked"] != true || b["mc_uuid"] != mcUUID { + t.Fatalf("verify body = %v, want linked:true mc_uuid:%s", b, mcUUID) + } + // The link is audited as account.link by the principal's Access email. + if n := len(repo.audits); n != 1 || repo.audits[0].Action != "account.link" || repo.audits[0].Actor != "u1@example.net" { + t.Fatalf("link audit not written as expected: %+v", repo.audits) + } + + // 3) the identical claim now passes the link gate and succeeds — the vertical + // is reconnected, which is the whole reason this slice exists. + if w := do(eh, "POST", "/api/v1/servers/survival/claim", "", nil); w.Code != http.StatusOK { + t.Fatalf("post-link claim: code = %d body %s, want 200", w.Code, w.Body.String()) + } +} + +// TestCreateLinkCode covers the internal mint endpoint: input validation, strict +// decoding, and that a real code lands in the store with the API-clock TTL. +func TestCreateLinkCode(t *testing.T) { + const mcUUID = "22222222-2222-2222-2222-222222222222" + repo := newFakeRepo() + api := newTestAPI(repo, newFakeCluster()) + ih := api.InternalHandler() + + t.Run("missing uuid -> 400", func(t *testing.T) { + w := do(ih, "POST", "/api/v1/internal/account/link/code", `{}`, nil) + if w.Code != http.StatusBadRequest || decodeErr(t, w) != "bad_request" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("unknown field rejected (no user_id smuggling)", func(t *testing.T) { + // The schema deliberately has no user_id on a code; a caller must not be able + // to introduce one via an extra field. + w := do(ih, "POST", "/api/v1/internal/account/link/code", `{"mc_uuid":"`+mcUUID+`","user_id":"u1"}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("strict decode must reject unknown field, code = %d (%s)", w.Code, w.Body.String()) + } + }) + t.Run("mints into store with TTL", func(t *testing.T) { + w := do(ih, "POST", "/api/v1/internal/account/link/code", `{"mc_uuid":"`+mcUUID+`"}`, nil) + if w.Code != http.StatusCreated { + t.Fatalf("code = %d, want 201 (%s)", w.Code, w.Body.String()) + } + code, _ := acctBody(t, w)["code"].(string) + rec, ok := repo.linkCodes[code] + if !ok { + t.Fatalf("minted code %q was not persisted", code) + } + if rec.mcUUID != mcUUID { + t.Errorf("stored mc_uuid = %q, want %q", rec.mcUUID, mcUUID) + } + if want := api.now().Add(linkCodeTTL); !rec.expiresAt.Equal(want) { + t.Errorf("expiresAt = %v, want %v", rec.expiresAt, want) + } + for _, c := range code { + if !strings.ContainsRune(linkCodeAlphabet, c) { + t.Errorf("code %q contains out-of-alphabet rune %q", code, c) + } + } + }) +} + +// TestLinkVerifyRejections is the verify failure matrix. The expired case seeds a +// past-dated code directly: the test clock is frozen, so there is nothing to +// "advance" — planting an already-expired code is the only way to exercise the +// expiry branch. +func TestLinkVerifyRejections(t *testing.T) { + const mcUUID = "33333333-3333-3333-3333-333333333333" + user := &Principal{UserID: "u1", Email: "u1@example.net", Role: "user"} + mk := func(repo *fakeRepo) http.Handler { + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + return api.ExternalHandler() + } + + t.Run("empty code -> 400 bad_request", func(t *testing.T) { + w := do(mk(newFakeRepo()), "POST", "/api/v1/account/link/verify", `{"code":""}`, nil) + if w.Code != http.StatusBadRequest || decodeErr(t, w) != "bad_request" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("whitespace code -> 400 bad_request", func(t *testing.T) { + w := do(mk(newFakeRepo()), "POST", "/api/v1/account/link/verify", `{"code":" "}`, nil) + if w.Code != http.StatusBadRequest || decodeErr(t, w) != "bad_request" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("unknown field -> 400", func(t *testing.T) { + w := do(mk(newFakeRepo()), "POST", "/api/v1/account/link/verify", `{"token":"ABCDEFGH"}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("strict decode must reject unknown field, code = %d", w.Code) + } + }) + t.Run("unknown code -> 400 invalid_code", func(t *testing.T) { + w := do(mk(newFakeRepo()), "POST", "/api/v1/account/link/verify", `{"code":"NEVERMINT"}`, nil) + if w.Code != http.StatusBadRequest || decodeErr(t, w) != "invalid_code" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + t.Run("expired code -> 400 invalid_code", func(t *testing.T) { + repo := newFakeRepo() + // One second before the frozen test clock (time.Unix(1_700_000_000, 0)). + repo.linkCodes["EXPIREDXY"] = fakeLinkCode{mcUUID: mcUUID, expiresAt: time.Unix(1_699_999_999, 0)} + w := do(mk(repo), "POST", "/api/v1/account/link/verify", `{"code":"EXPIREDXY"}`, nil) + if w.Code != http.StatusBadRequest || decodeErr(t, w) != "invalid_code" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if _, ok := repo.linkCodes["EXPIREDXY"]; !ok { + t.Error("an expired code must not be consumed") + } + }) + t.Run("uuid linked to another user -> 409 already_linked", func(t *testing.T) { + repo := newFakeRepo() + repo.links[mcUUID] = "someone-else" + repo.linkCodes["FRESHCOD"] = fakeLinkCode{mcUUID: mcUUID, expiresAt: time.Unix(1_700_000_600, 0)} + w := do(mk(repo), "POST", "/api/v1/account/link/verify", `{"code":"FRESHCOD"}`, nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "already_linked" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + // A wrong-user submission must not burn the real owner's pending code. + if _, ok := repo.linkCodes["FRESHCOD"]; !ok { + t.Error("a conflicting verify must not consume the code") + } + }) +} + +// TestLinkVerifyIdempotent re-verifies the same (user, uuid) pair with a fresh +// code and expects a clean 200, not a self-conflict. +func TestLinkVerifyIdempotent(t *testing.T) { + const mcUUID = "44444444-4444-4444-4444-444444444444" + user := &Principal{UserID: "u1", Email: "u1@example.net", Role: "user"} + repo := newFakeRepo() + repo.links[mcUUID] = "u1" // already linked to THIS user + repo.linked["u1"] = true + repo.linkCodes["REVERIFYX"] = fakeLinkCode{mcUUID: mcUUID, expiresAt: time.Unix(1_700_000_600, 0)} + + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + w := do(api.ExternalHandler(), "POST", "/api/v1/account/link/verify", `{"code":"REVERIFYX"}`, nil) + if w.Code != http.StatusOK { + t.Fatalf("idempotent re-verify: code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if got := acctBody(t, w)["mc_uuid"]; got != mcUUID { + t.Errorf("mc_uuid = %v, want %s", got, mcUUID) + } + if repo.links[mcUUID] != "u1" { + t.Errorf("links[%s] = %q, want u1", mcUUID, repo.links[mcUUID]) + } +} + +// TestLinkStart pins the status endpoint handleClaim's 412 points at: it reports +// link state and instructions, and never mints (it has no UUID to mint against). +func TestLinkStart(t *testing.T) { + user := &Principal{UserID: "u1", Email: "u1@example.net", Role: "user"} + mk := func(repo *fakeRepo) http.Handler { + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: user} + return api.ExternalHandler() + } + + t.Run("not linked", func(t *testing.T) { + w := do(mk(newFakeRepo()), "POST", "/api/v1/account/link/start", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + b := acctBody(t, w) + if b["linked"] != false { + t.Errorf("linked = %v, want false", b["linked"]) + } + if s, _ := b["instructions"].(string); s == "" { + t.Error("start must return non-empty instructions") + } + }) + t.Run("already linked", func(t *testing.T) { + repo := newFakeRepo() + repo.linked["u1"] = true + w := do(mk(repo), "POST", "/api/v1/account/link/start", "", nil) + if b := acctBody(t, w); b["linked"] != true { + t.Errorf("linked = %v, want true", b["linked"]) + } + }) +} + +// TestAccountLinkFaceSeparation enforces the schema-forced face split: the code +// generator is internal-only, and verify/start are web-only. Crossing those +// faces must 404, not silently work. +func TestAccountLinkFaceSeparation(t *testing.T) { + user := &Principal{UserID: "u1", Email: "u1@example.net", Role: "user"} + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.External = staticExternal{p: user} + ih := api.InternalHandler() + eh := api.ExternalHandler() + + // The generator must not exist on the web face — a logged-in user could + // otherwise mint a code for a UUID they never proved they own. + if w := do(eh, "POST", "/api/v1/internal/account/link/code", `{"mc_uuid":"x"}`, nil); w.Code != http.StatusNotFound { + t.Errorf("code endpoint on external face: code = %d, want 404", w.Code) + } + // verify / start are web-only: they require a logged-in principal the internal + // face never carries. + if w := do(ih, "POST", "/api/v1/account/link/verify", `{"code":"ABCDEFGH"}`, nil); w.Code != http.StatusNotFound { + t.Errorf("verify endpoint on internal face: code = %d, want 404", w.Code) + } + if w := do(ih, "POST", "/api/v1/account/link/start", "", nil); w.Code != http.StatusNotFound { + t.Errorf("start endpoint on internal face: code = %d, want 404", w.Code) + } +} diff --git a/internal/api/handlers_backups.go b/internal/api/handlers_backups.go new file mode 100644 index 0000000..58b2e3d --- /dev/null +++ b/internal/api/handlers_backups.go @@ -0,0 +1,142 @@ +package api + +import ( + "errors" + "net/http" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/naming" +) + +// handleListBackups lists the world backups visible to the caller (spec §7 GET +// /api/v1/backups; world_backups in §22). It is app-tier: an admin sees every +// present backup; a regular user sees only the backups of worlds they formerly +// owned. The scope is decided here from the Principal and enforced by which Repo +// query runs (AllBackups vs BackupsForUser) — there is no client-supplied filter +// a user could widen, so "a user cannot see another's backups" is a property of +// the query, not of request parsing. +func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + + var ( + backups []BackupView + err error + ) + if p.IsAdmin() { + backups, err = a.Repo.AllBackups(r.Context()) + } else { + backups, err = a.Repo.BackupsForUser(r.Context(), p.UserID) + } + if err != nil { + writeError(w, r, err) + return + } + if backups == nil { + backups = []BackupView{} + } + writeJSON(w, http.StatusOK, map[string]any{"backups": backups}) +} + +// handleRestoreBackup starts restoring a server's world from its most recent +// backup (spec §7 POST /servers/{name}/restore-backup; spec §466: former_owner +// 3mo 内重新 claim → restore PVC). The authorization is deliberately stricter than +// ordinary owner-or-admin, in this order: +// +// ① name validation +// ② ServerByName — an unknown server is 404 +// ③ owner-or-admin, else 403. A released world's server row is unowned +// (owner_id NULL → OwnerID ""), so this also enforces "重新 claim": a former +// owner must re-claim the server before they can restore into it. +// ④ the latest present backup, else 404 no_backup +// ⑤ former-owner match: a non-admin may restore ONLY a world they formerly owned. +// The current-owner gate in ③ is not enough — user B who re-claims a released +// server could otherwise resurrect user A's world (the backup still carries +// former_owner=A), a data leak. Admin skips this check. +// ⑥ stopped gate: the world PVC must be free, so restore is refused unless the +// server is fully stopped. A running OR starting server still holds the RWO +// world volume, which a restore Job could not mount — a clean 409 beats a Job +// that fails to schedule. +// ⑦ hand off to the Restorer. Restore is asynchronous (a restore Job, like an +// image build Job), so success means "enqueued" and the handler answers 202. +// +// The opaque backup_ref is resolved server-side from the latest backup and handed +// to the Restorer directly; the client never names a backup by handle (spec §286 +// principle). +func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + // Ownership: owner or admin, mirroring handleStop. An unknown server is 404; an + // unowned (released) server fails the owner check for everyone but admin, which + // is exactly the "must re-claim first" rule. + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !a.isOwnerOrAdmin(p, rec) { + writeError(w, r, errForbidden) + return + } + + backup, err := a.Repo.LatestBackup(r.Context(), name) + if err != nil { + if errors.Is(err, ErrNotFound) { + writeError(w, r, newError(http.StatusNotFound, "no_backup", + "no restorable backup exists for this server")) + return + } + writeError(w, r, err) + return + } + + // A non-admin may restore only a world they formerly owned (spec §466). Without + // this a fresh claimant of a released server could resurrect the previous + // owner's world data. + if !p.IsAdmin() && backup.FormerOwner != p.UserID { + writeError(w, r, errForbidden) + return + } + + // Stopped gate: the world PVC must be free for the restore to write into it. + // Refuse unless the server is fully stopped — Ready means it is up, and any + // desiredState other than Stopped means it is up or coming up and still owns the + // RWO volume (spec §141 readiness is an RCON probe; DesiredStopped is the + // intent). This yields a specific 409 instead of a restore Job that cannot mount. + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if info.Ready || info.DesiredState != string(v1alpha1.DesiredStopped) { + writeError(w, r, newError(http.StatusConflict, "not_stopped", + "stop the server before restoring a backup")) + return + } + + // Restorer is optional: when unwired the endpoint reports 503 rather than + // panicking, so the authorization boundary above is exercised even before the + // restore-Job executor is wired (see Restorer). + if a.Restorer == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "restore_unavailable", + "restore subsystem is not configured")) + return + } + + if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil { + // ErrNotFound (server vanished from the execution backend) → 404; else 500. + a.writeLookupError(w, r, err) + return + } + + a.audit(r, p.Email, "backup.restore", name) + writeJSON(w, http.StatusAccepted, map[string]any{ + "name": name, + "status": "restoring", + "backup_id": backup.ID, + }) +} diff --git a/internal/api/handlers_backups_test.go b/internal/api/handlers_backups_test.go new file mode 100644 index 0000000..8521aa3 --- /dev/null +++ b/internal/api/handlers_backups_test.go @@ -0,0 +1,293 @@ +package api + +import ( + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" +) + +// TestListBackups exercises GET /api/v1/backups (spec §7): admin sees every +// present backup, a regular user sees only the worlds they formerly owned, and +// expired/deleted backups are never listed. It also asserts the opaque +// backup_ref never reaches the wire (spec §286 principle). +func TestListBackups(t *testing.T) { + mk := func() *API { + repo := newFakeRepo() + repo.backups = []fakeBackup{ + {view: BackupView{ID: "b1", ServerName: "alpha", FormerOwner: "owner1", + Status: "present", Reason: "inactive_15d", SizeBytes: 100, + CreatedAt: time.Unix(1_699_000_000, 0), ExpiresAt: time.Unix(1_706_000_000, 0)}, ref: "ref-b1"}, + {view: BackupView{ID: "b2", ServerName: "beta", FormerOwner: "owner2", + Status: "present", Reason: "manual", SizeBytes: 200, + CreatedAt: time.Unix(1_699_500_000, 0), ExpiresAt: time.Unix(1_706_000_000, 0)}, ref: "ref-b2"}, + {view: BackupView{ID: "b3", ServerName: "gamma", FormerOwner: "owner1", + Status: "deleted", Reason: "manual", SizeBytes: 50, + CreatedAt: time.Unix(1_698_000_000, 0), ExpiresAt: time.Unix(1_705_000_000, 0)}, ref: "ref-b3"}, + } + return newTestAPI(repo, newFakeCluster()) + } + + list := func(t *testing.T, w *httptest.ResponseRecorder) []BackupView { + t.Helper() + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + var resp struct { + Backups []BackupView `json:"backups"` + } + if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + return resp.Backups + } + ids := func(vs []BackupView) map[string]bool { + m := map[string]bool{} + for _, v := range vs { + m[v.ID] = true + } + return m + } + + t.Run("admin sees all present, not deleted", func(t *testing.T) { + api := mk() + api.External = staticExternal{p: &Principal{UserID: "admin1", Role: "admin", ViaAdminAccess: true}} + got := ids(list(t, do(api.ExternalHandler(), "GET", "/api/v1/backups", "", nil))) + if !got["b1"] || !got["b2"] || got["b3"] || len(got) != 2 { + t.Fatalf("admin scope = %v, want {b1,b2}", got) + } + }) + + t.Run("user sees only own former-owned present backups", func(t *testing.T) { + api := mk() + api.External = staticExternal{p: &Principal{UserID: "owner1", Role: "user"}} + got := ids(list(t, do(api.ExternalHandler(), "GET", "/api/v1/backups", "", nil))) + // b2 belongs to owner2; b3 is owner1's but deleted — neither is visible. + if !got["b1"] || got["b2"] || got["b3"] || len(got) != 1 { + t.Fatalf("owner1 scope = %v, want {b1}", got) + } + }) + + t.Run("user with no backups -> empty array, never null", func(t *testing.T) { + api := mk() + api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}} + w := do(api.ExternalHandler(), "GET", "/api/v1/backups", "", nil) + vs := list(t, w) + if vs == nil || len(vs) != 0 { + t.Fatalf("empty scope = %v, want a non-nil empty slice", vs) + } + }) + + t.Run("backup_ref never serialized", func(t *testing.T) { + api := mk() + api.External = staticExternal{p: &Principal{UserID: "admin1", Role: "admin", ViaAdminAccess: true}} + body := do(api.ExternalHandler(), "GET", "/api/v1/backups", "", nil).Body.String() + if strings.Contains(body, "backup_ref") || strings.Contains(body, "ref-b") { + t.Fatalf("response leaked the internal backup_ref: %s", body) + } + }) +} + +// TestRestoreBackup exercises POST /api/v1/servers/{name}/restore-backup (spec §7, +// §466). The default server is stopped, owned by owner1 who is also the backup's +// former_owner, with one present backup and a wired fakeRestorer. Subtests +// override only what they need. The focus is the layered authorization (owner + +// former-owner), the stopped gate, the 404/503 surface, and that restore is an +// async 202 kick-off. +func TestRestoreBackup(t *testing.T) { + owner := &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"} + + mk := func() (*API, *fakeRepo, *fakeCluster, *fakeRestorer) { + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + repo.backups = []fakeBackup{ + {view: BackupView{ID: "bk1", ServerName: "survival", FormerOwner: "owner1", + Status: "present", Reason: "inactive_15d", SizeBytes: 1024, + CreatedAt: time.Unix(1_699_000_000, 0), ExpiresAt: time.Unix(1_706_000_000, 0)}, ref: "world-archive-ref"}, + } + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped", + Ready: false, DesiredState: string(v1alpha1.DesiredStopped)} + restorer := &fakeRestorer{} + api := newTestAPI(repo, cl) + api.Restorer = restorer + return api, repo, cl, restorer + } + + const path = "/api/v1/servers/survival/restore-backup" + + t.Run("former owner restores -> 202 + restorer + audit", func(t *testing.T) { + api, repo, _, restorer := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String()) + } + var resp struct { + Name string `json:"name"` + Status string `json:"status"` + BackupID string `json:"backup_id"` + } + if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + if resp.Name != "survival" || resp.Status != "restoring" || resp.BackupID != "bk1" { + t.Fatalf("unexpected response %+v", resp) + } + if restorer.calls != 1 || restorer.gotName != "survival" || restorer.gotRef != "world-archive-ref" { + t.Fatalf("restorer saw (calls=%d,%q,%q), want (1,survival,world-archive-ref)", + restorer.calls, restorer.gotName, restorer.gotRef) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "backup.restore" || repo.audits[0].Actor != "owner1@example.net" { + t.Fatalf("audit not written as expected: %+v", repo.audits) + } + }) + + t.Run("admin restores another's world -> 202", func(t *testing.T) { + api, _, _, restorer := mk() + api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "admin1@example.net", + Role: "admin", ViaAdminAccess: true}} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String()) + } + if restorer.calls != 1 { + t.Fatalf("admin restore reached restorer %d times, want 1", restorer.calls) + } + }) + + t.Run("non-owner -> 403, no restore", func(t *testing.T) { + api, _, _, restorer := mk() + api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if restorer.calls != 0 { + t.Fatal("a forbidden caller must not start a restore") + } + }) + + t.Run("current owner who is not former_owner -> 403, no restore", func(t *testing.T) { + // The leak guard: user "newowner" re-claimed survival (so they pass the + // owner gate), but the backup belongs to "olduser". They must NOT be able to + // resurrect another person's world. + api, repo, _, restorer := mk() + repo.byName["survival"].OwnerID = "newowner" + repo.backups[0].view.FormerOwner = "olduser" + api.External = staticExternal{p: &Principal{UserID: "newowner", Role: "user"}} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403 (former-owner leak guard)", w.Code) + } + if restorer.calls != 0 { + t.Fatal("restoring another owner's world must not reach the restorer") + } + }) + + t.Run("unknown server -> 404", func(t *testing.T) { + api, _, _, _ := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/missing/restore-backup", "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) + + t.Run("no present backup -> 404 no_backup", func(t *testing.T) { + api, repo, _, restorer := mk() + repo.backups = nil + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusNotFound || decodeErr(t, w) != "no_backup" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if restorer.calls != 0 { + t.Fatal("no backup must not reach the restorer") + } + }) + + t.Run("only a deleted backup -> 404 no_backup", func(t *testing.T) { + api, repo, _, _ := mk() + repo.backups[0].view.Status = "deleted" + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusNotFound || decodeErr(t, w) != "no_backup" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + + t.Run("running server -> 409 not_stopped, no restore", func(t *testing.T) { + api, _, cl, restorer := mk() + cl.byName["survival"].Ready = true + cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning) + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if restorer.calls != 0 { + t.Fatal("a running server must not be restored into") + } + }) + + t.Run("starting server -> 409 not_stopped", func(t *testing.T) { + // desiredState=Running but not yet Ready: the world PVC is already mounted by + // the starting pod, so the stopped gate must still refuse. + api, _, cl, _ := mk() + cl.byName["survival"].Ready = false + cl.byName["survival"].DesiredState = string(v1alpha1.DesiredRunning) + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "not_stopped" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + + t.Run("nil Restorer -> 503 restore_unavailable", func(t *testing.T) { + api, _, _, _ := mk() + api.Restorer = nil + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "restore_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + + t.Run("restorer failure -> 500, not audited", func(t *testing.T) { + api, repo, _, restorer := mk() + restorer.err = errors.New("kick-off failed") + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusInternalServerError { + t.Fatalf("code = %d, want 500 (%s)", w.Code, w.Body.String()) + } + if len(repo.audits) != 0 { + t.Fatalf("a failed restore must not be audited: %+v", repo.audits) + } + }) + + t.Run("restorer reports unknown server -> 404", func(t *testing.T) { + api, _, _, restorer := mk() + restorer.err = ErrNotFound + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", path, "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) + + t.Run("invalid server name -> 400", func(t *testing.T) { + api, _, _, _ := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/X/restore-backup", "", nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) +} diff --git a/internal/api/handlers_console.go b/internal/api/handlers_console.go new file mode 100644 index 0000000..2c32e29 --- /dev/null +++ b/internal/api/handlers_console.go @@ -0,0 +1,122 @@ +package api + +import ( + "errors" + "net/http" + "strings" + + "felis.lolicon.best/internal/naming" +) + +// maxConsoleCommandLen caps the command body well under RCON's single-packet +// limit (internal/rcon maxPacketLen = 4096, minus framing). The cap is +// deliberately conservative — interactive console commands are short — so an +// over-long command fails as a clean 400 here rather than surfacing as an opaque +// 500 from the RCON writer. +const maxConsoleCommandLen = 1024 + +// commandRequest is the console-write body (spec §8 写=RCON). One field, one +// line: decodeJSON rejects unknown fields so no extra knob can smuggle in. +type commandRequest struct { + Command string `json:"command"` +} + +// handleCommand runs one RCON command against the caller's server and returns +// the server's reply (spec §8, 读写分离: 写=RCON; spec §7 POST +// /servers/{name}/command). It is app-tier — operating your OWN server — gated +// by isOwnerOrAdmin, mirroring handleStop. The order is authorization first, +// then a readiness pre-check, then the write: +// +// ① name + body validation (single line, bounded length) +// ② ownership: owner or admin, else 403 (404 if the server is unknown) +// ③ readiness: a write only makes sense on a Running server (§141 Ready ⟺ RCON +// reachable), so a non-Ready server is a specific 409, not a blind dial +// ④ the RCON write; an unreachable channel is ErrConsoleUnavailable → 503 (the +// §141 invariant can drop between the §③ check and the dial — the pre-check +// is for a better error, not a correctness guarantee) +// +// The RCON password is resolved entirely inside the Console implementation and +// never appears in the request or response (spec §286: RCON 密码绝不下发前端). +func (a *API) handleCommand(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + var body commandRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + + // A console command is exactly one line. Trim surrounding space, strip a + // single leading '/' (players type "/say hi"; RCON wants "say hi"), then + // reject control characters so one request can never smuggle a second command + // past a newline. + command := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(body.Command), "/")) + if command == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "command is required")) + return + } + if len(command) > maxConsoleCommandLen { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "command too long (max %d bytes)", maxConsoleCommandLen)) + return + } + if strings.IndexFunc(command, func(c rune) bool { return c < 0x20 }) >= 0 { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "command must be a single line (no control characters)")) + return + } + + // Ownership: owner or admin, mirroring handleStop. An unknown server is 404. + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !a.isOwnerOrAdmin(p, rec) { + writeError(w, r, errForbidden) + return + } + + // Readiness pre-check: a write only makes sense on a Running server. This + // yields a specific "not running" 409 instead of a blind dial that would time + // out. It is racy (the server can drop between here and the dial), so the + // write below still maps an unreachable channel to 503. + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !info.Ready { + writeError(w, r, newError(http.StatusConflict, "not_running", + "server is not running; wake it before sending console commands")) + return + } + + // Console is wired in production (cmd/felis); the nil guard only defends + // against a misconstructed API, failing as 503 rather than panicking. + if a.Console == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable", + "console subsystem is not configured")) + return + } + + output, err := a.Console.RunCommand(r.Context(), name, command) + switch { + case errors.Is(err, ErrConsoleUnavailable): + writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable", + "server console is currently unreachable; wake the server and retry")) + return + case err != nil: + // ErrNotFound (server vanished mid-request) → 404; anything else → 500. + a.writeLookupError(w, r, err) + return + } + + a.audit(r, p.Email, "console.command", name) + writeJSON(w, http.StatusOK, map[string]any{"name": name, "output": output}) +} diff --git a/internal/api/handlers_console_test.go b/internal/api/handlers_console_test.go new file mode 100644 index 0000000..aa95ab4 --- /dev/null +++ b/internal/api/handlers_console_test.go @@ -0,0 +1,196 @@ +package api + +import ( + "encoding/json" + "net/http" + "strings" + "testing" +) + +// TestConsoleCommand exercises the §8 RCON write handler end-to-end through the +// external app-tier router: ownership (owner/admin), the readiness gate, input +// validation (single line, bounded, no control chars), and the failure surface +// (unreachable console → 503). The real RCON dial is integration-only; this +// drives the handler against a fakeConsole. +func TestConsoleCommand(t *testing.T) { + owner := &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"} + + // mk builds an API whose "survival" server is owned by owner1 and Ready, with a + // fresh fakeConsole wired. Subtests override only what they need. + mk := func() (*API, *fakeRepo, *fakeCluster, *fakeConsole) { + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Running", Ready: true} + console := &fakeConsole{reply: "There are 3 of a max of 20 players online"} + api := newTestAPI(repo, cl) + api.Console = console + return api, repo, cl, console + } + + t.Run("owner runs command -> 200 + reply + audit", func(t *testing.T) { + api, repo, _, console := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"list"}`, nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + var resp struct { + Name string `json:"name"` + Output string `json:"output"` + } + if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + if resp.Name != "survival" || resp.Output != console.reply { + t.Fatalf("unexpected response %+v", resp) + } + if console.gotName != "survival" || console.gotCommand != "list" { + t.Fatalf("console saw (%q,%q), want (survival,list)", console.gotName, console.gotCommand) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "console.command" || repo.audits[0].Actor != "owner1@example.net" { + t.Fatalf("audit not written as expected: %+v", repo.audits) + } + }) + + t.Run("leading slash stripped before RCON", func(t *testing.T) { + api, _, _, console := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"/say hi"}`, nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if console.gotCommand != "say hi" { + t.Fatalf("console got %q, want %q", console.gotCommand, "say hi") + } + }) + + t.Run("admin runs command on another's server -> 200", func(t *testing.T) { + api, _, _, console := mk() + api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "admin1@example.net", + Role: "admin", ViaAdminAccess: true}} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"list"}`, nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if console.calls != 1 { + t.Fatalf("admin command reached RCON %d times, want 1", console.calls) + } + }) + + t.Run("non-owner -> 403, no RCON call", func(t *testing.T) { + api, _, _, console := mk() + api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"list"}`, nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if console.calls != 0 { + t.Fatal("a forbidden caller must not reach RCON") + } + }) + + t.Run("unknown server -> 404", func(t *testing.T) { + api, _, _, _ := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/missing/command", `{"command":"list"}`, nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + }) + + t.Run("not ready -> 409 not_running, no RCON call", func(t *testing.T) { + api, _, cl, console := mk() + cl.byName["survival"].Ready = false + cl.byName["survival"].Phase = "Stopped" + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"list"}`, nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "not_running" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if console.calls != 0 { + t.Fatal("a non-ready server must not reach RCON") + } + }) + + t.Run("empty command -> 400, no RCON call", func(t *testing.T) { + api, _, _, console := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":" "}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + if console.calls != 0 { + t.Fatal("an empty command must not reach RCON") + } + }) + + t.Run("slash-only command -> 400", func(t *testing.T) { + api, _, _, _ := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"/"}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) + + t.Run("control character (newline) -> 400, no RCON call", func(t *testing.T) { + api, _, _, console := mk() + api.External = staticExternal{p: owner} + // The \n is a JSON escape that decodes to a real newline; the handler must + // reject it so one request cannot smuggle a second command. + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", + `{"command":"say hi\nop attacker"}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if console.calls != 0 { + t.Fatal("a control-char command must not reach RCON") + } + }) + + t.Run("over-long command -> 400", func(t *testing.T) { + api, _, _, _ := mk() + api.External = staticExternal{p: owner} + long := strings.Repeat("a", maxConsoleCommandLen+1) + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", + `{"command":"`+long+`"}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) + + t.Run("unknown field -> 400", func(t *testing.T) { + api, _, _, _ := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", + `{"command":"list","extra":1}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) + + t.Run("console unreachable -> 503 console_unavailable", func(t *testing.T) { + api, repo, _, console := mk() + console.err = ErrConsoleUnavailable + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"list"}`, nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "console_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + // A failed write must not be audited as a successful command. + if len(repo.audits) != 0 { + t.Fatalf("unreachable console must not audit: %+v", repo.audits) + } + }) + + t.Run("nil Console -> 503", func(t *testing.T) { + api, _, _, _ := mk() + api.Console = nil + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/command", `{"command":"list"}`, nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "console_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) +} diff --git a/internal/api/handlers_create_test.go b/internal/api/handlers_create_test.go new file mode 100644 index 0000000..1a69dd6 --- /dev/null +++ b/internal/api/handlers_create_test.go @@ -0,0 +1,317 @@ +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" +) + +// decodeField returns one string field from a flat JSON success body. +func decodeField(t *testing.T, w *httptest.ResponseRecorder, field string) string { + t.Helper() + var raw map[string]any + if err := json.Unmarshal(w.Body.Bytes(), &raw); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + s, _ := raw[field].(string) + return s +} + +// newCreateAPI wires an admin-authenticated API with a Builder whose whitelist +// admits the canonical test image, plus handles to the fakes so a test can +// pre-seed state and assert what the §15 create flow wrote. +func newCreateAPI() (*API, *fakeRepo, *fakeCluster, *fakeBuilder) { + repo := newFakeRepo() + cl := newFakeCluster() + fb := &fakeBuilder{admitted: map[string]bool{admittedImage: true}} + api := newTestAPI(repo, cl) + api.Builder = fb + api.External = staticExternal{p: &Principal{UserID: "a", Email: "admin@example.net", + Role: "admin", ViaAdminAccess: true}} + return api, repo, cl, fb +} + +const admittedImage = "registry.felis.svc:5000/mc:1" + +// validCreateBody is a well-formed §15 form; tests mutate one field at a time by +// writing their own JSON. +const validCreateBody = `{"name":"survival","subdomain":"survival",` + + `"image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi"}` + +// TestCreateServerSuccess covers the happy path end-to-end: the form is +// validated, the business rows are seeded, the CRD is created cold and unowned, +// and the §22 memory ceiling is materialized on the created spec. +func TestCreateServerSuccess(t *testing.T) { + api, repo, cl, _ := newCreateAPI() + + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", validCreateBody, nil) + if w.Code != http.StatusCreated { + t.Fatalf("code = %d, want 201 (%s)", w.Code, w.Body.String()) + } + + in, ok := cl.created["survival"] + if !ok { + t.Fatal("CreateServer was not called for survival") + } + if in.Subdomain != "survival" || in.Image != admittedImage { + t.Errorf("unexpected created input %+v", in) + } + // JavaMemory is the DERIVED JVM heap, not the raw K8s quantity: 2Gi ceiling + // minus 512Mi off-heap headroom = 1536Mi, in JVM-valid "M" units (never "Gi"). + if in.JavaMemory != "1536M" { + t.Errorf("JavaMemory = %q, want 1536M (2Gi limit minus headroom)", in.JavaMemory) + } + if in.StorageSize != "10Gi" { + t.Errorf("StorageSize = %q, want 10Gi", in.StorageSize) + } + // Default autostart policy is the safe ownerOnly. + if in.AutostartPolicy != v1alpha1.AutostartOwnerOnly { + t.Errorf("autostartPolicy = %q, want ownerOnly", in.AutostartPolicy) + } + + // §22 ceiling: the memory limit must be present, non-zero, and match memory. + memLim, has := in.Resources.Limits[corev1.ResourceMemory] + if !has || memLim.IsZero() { + t.Fatalf("memory limit missing or zero: %+v", in.Resources.Limits) + } + if want := resource.MustParse("2Gi"); memLim.Cmp(want) != 0 { + t.Errorf("memory limit = %s, want 2Gi", memLim.String()) + } + if memReq := in.Resources.Requests[corev1.ResourceMemory]; memReq.Cmp(resource.MustParse("2Gi")) != 0 { + t.Errorf("memory request = %s, want 2Gi", memReq.String()) + } + + // Business rows seeded for the unowned server, alias bound. + if !repo.seeded["survival"] { + t.Error("servers row was not seeded") + } + if repo.aliases["survival"] != "survival" { + t.Errorf("subdomain alias = %q, want survival", repo.aliases["survival"]) + } + // Quota is NOT consulted on create (the server is unowned; quota is charged at + // claim). repo.quota is empty here yet the create still succeeded. + + // Audit written with the admin's identity. + if len(repo.audits) != 1 || repo.audits[0].Action != "server.create" || + repo.audits[0].Actor != "admin@example.net" { + t.Fatalf("audit not written as expected: %+v", repo.audits) + } + + // Response envelope. + if got := decodeField(t, w, "desiredState"); got != "Stopped" { + t.Errorf("desiredState = %q, want Stopped", got) + } +} + +// TestCreateServerResourceOverride confirms the optional resources block widens +// the CPU envelope and can override the memory limit/request independently. +func TestCreateServerResourceOverride(t *testing.T) { + api, _, cl, _ := newCreateAPI() + body := `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1",` + + `"memory":"2Gi","storage":"10Gi","resources":{"cpu":"2","cpuRequest":"500m",` + + `"memory":"4Gi","memoryRequest":"1Gi"}}` + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", body, nil) + if w.Code != http.StatusCreated { + t.Fatalf("code = %d, want 201 (%s)", w.Code, w.Body.String()) + } + in := cl.created["survival"] + if lim := in.Resources.Limits[corev1.ResourceMemory]; lim.Cmp(resource.MustParse("4Gi")) != 0 { + t.Errorf("memory limit = %s, want 4Gi (override)", lim.String()) + } + if req := in.Resources.Requests[corev1.ResourceMemory]; req.Cmp(resource.MustParse("1Gi")) != 0 { + t.Errorf("memory request = %s, want 1Gi (override)", req.String()) + } + if lim := in.Resources.Limits[corev1.ResourceCPU]; lim.Cmp(resource.MustParse("2")) != 0 { + t.Errorf("cpu limit = %s, want 2", lim.String()) + } + if req := in.Resources.Requests[corev1.ResourceCPU]; req.Cmp(resource.MustParse("500m")) != 0 { + t.Errorf("cpu request = %s, want 500m", req.String()) + } + // The JVM heap derives from the FINAL (overridden) 4Gi ceiling, not the + // top-level 2Gi: 4Gi minus 1Gi (25%) headroom = 3072Mi. + if in.JavaMemory != "3072M" { + t.Errorf("JavaMemory = %q, want 3072M (derived from overridden 4Gi limit)", in.JavaMemory) + } +} + +// TestDeriveJavaHeap pins the JVM heap derivation: always JVM-valid "M" units +// (never "Gi"), always strictly below the cgroup ceiling, with headroom that is +// the larger of 512Mi or 25%, capped at half the limit for small ceilings. +func TestDeriveJavaHeap(t *testing.T) { + cases := []struct{ limit, want string }{ + {"2Gi", "1536M"}, // 2048 - 512 (25% == floor) + {"4Gi", "3072M"}, // 4096 - 1024 (25%) + {"8Gi", "6144M"}, // 8192 - 2048 (25%) + {"1Gi", "512M"}, // 1024 - 512 (floor, capped at half) + {"512Mi", "256M"}, // 512 - 256 (floor capped at half) + } + for _, c := range cases { + got := deriveJavaHeap(resource.MustParse(c.limit)) + if got != c.want { + t.Errorf("deriveJavaHeap(%s) = %q, want %q", c.limit, got, c.want) + } + // Invariant: the derived heap must never reach the ceiling. + heap := resource.MustParse(got) + if heap.Cmp(resource.MustParse(c.limit)) >= 0 { + t.Errorf("deriveJavaHeap(%s) = %s is not below the ceiling", c.limit, got) + } + } +} + +// TestCreateServerRejections is the validation matrix: every malformed or +// conflicting request is rejected with the right status and stable error code, +// and (critically) nothing reaches the cluster on a rejection. +func TestCreateServerRejections(t *testing.T) { + cases := []struct { + name string + body string + setup func(*fakeRepo, *fakeCluster) + wantCode int + wantErr string + check func(*testing.T, *fakeRepo, *fakeCluster) + }{ + { + name: "bad name", + body: `{"name":"Bad_Name","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_name", + }, + { + name: "reserved name", + body: `{"name":"admin","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_name", + }, + { + name: "bad subdomain", + body: `{"name":"survival","subdomain":"Bad.Sub","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_subdomain", + }, + { + name: "reserved subdomain", + body: `{"name":"survival","subdomain":"lobby","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_subdomain", + }, + { + name: "bad autostart policy", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi","autostartPolicy":"sometimes"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "missing memory", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "non-positive memory", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"0","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "garbage memory quantity", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"lots","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "memory request exceeds limit", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi","resources":{"memoryRequest":"4Gi"}}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "missing storage", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"2Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "empty image", + body: `{"name":"survival","subdomain":"survival","image":"","memory":"2Gi","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "image not whitelisted", + body: `{"name":"survival","subdomain":"survival","image":"docker.io/evil:latest","memory":"2Gi","storage":"10Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "image_not_whitelisted", + }, + { + name: "unknown field (no free YAML)", + body: `{"name":"survival","subdomain":"survival","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi","apiVersion":"felis/v1alpha1"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "subdomain taken in cluster", + body: validCreateBody, + setup: func(_ *fakeRepo, cl *fakeCluster) { + cl.bySub["survival"] = &ServerInfo{Name: "other", Subdomain: "survival"} + }, + wantCode: http.StatusConflict, wantErr: "subdomain_taken", + }, + { + name: "subdomain bound to different server in PG", + body: validCreateBody, + setup: func(r *fakeRepo, _ *fakeCluster) { r.aliases["survival"] = "someoneelse" }, + wantCode: http.StatusConflict, wantErr: "subdomain_taken", + }, + { + // A dup-name create with a FRESH subdomain must be rejected BEFORE any + // PG write. Otherwise SeedServer binds the new alias onto the + // pre-existing server and commits it, leaving a stray alias after the + // create 409s — a failed request must not mutate state. + name: "duplicate server name", + body: `{"name":"survival","subdomain":"fresh","image":"registry.felis.svc:5000/mc:1","memory":"2Gi","storage":"10Gi"}`, + setup: func(_ *fakeRepo, cl *fakeCluster) { + cl.byName["survival"] = &ServerInfo{Name: "survival", Subdomain: "legacy"} + }, + wantCode: http.StatusConflict, wantErr: "already_exists", + check: func(t *testing.T, repo *fakeRepo, _ *fakeCluster) { + if repo.aliases["fresh"] != "" { + t.Errorf("stray alias bound on failed create: fresh -> %q", repo.aliases["fresh"]) + } + if repo.seeded["survival"] { + t.Error("a failed dup-name create must not seed a servers row") + } + }, + }, + } + + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + api, repo, cl, _ := newCreateAPI() + if c.setup != nil { + c.setup(repo, cl) + } + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", c.body, nil) + if w.Code != c.wantCode { + t.Fatalf("code = %d, want %d (%s)", w.Code, c.wantCode, w.Body.String()) + } + if got := decodeErr(t, w); got != c.wantErr { + t.Errorf("error code = %q, want %q", got, c.wantErr) + } + // Every rejection short-circuits before CreateServer, so no rejected + // request may have created a CRD. (The dup-name setup pre-seeds byName + // directly, never cl.created.) + if len(cl.created) != 0 { + t.Errorf("a rejected create must not create a CRD, got %+v", cl.created) + } + if c.check != nil { + c.check(t, repo, cl) + } + }) + } +} + +// TestCreateServerWithoutBuilderIs503 proves create requires the build subsystem +// (its whitelist is the image-admission source) and fails before any write. +func TestCreateServerWithoutBuilderIs503(t *testing.T) { + api, _, cl, _ := newCreateAPI() + api.Builder = nil + w := do(api.ExternalHandler(), "POST", "/api/v1/servers", validCreateBody, nil) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("code = %d, want 503 (%s)", w.Code, w.Body.String()) + } + if _, created := cl.created["survival"]; created { + t.Error("no CRD may be created without a Builder") + } +} diff --git a/internal/api/handlers_internal.go b/internal/api/handlers_internal.go new file mode 100644 index 0000000..6f6cf99 --- /dev/null +++ b/internal/api/handlers_internal.go @@ -0,0 +1,359 @@ +package api + +import ( + "context" + "errors" + "net/http" + "strings" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/naming" +) + +// handleHealthz is a liveness probe: the process is up. +func (a *API) handleHealthz(w http.ResponseWriter, r *http.Request) { + writeJSON(w, http.StatusOK, map[string]string{"status": "ok"}) +} + +// handleReadyz is a readiness probe. A full implementation also checks the DB, +// the K8s API and the CRD informer (spec §7); here it reports the configured +// dependencies are wired. Dependency pinging lands with the integration layer. +func (a *API) handleReadyz(w http.ResponseWriter, r *http.Request) { + if a.Repo == nil || a.Cluster == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "not_ready", "dependencies not wired")) + return + } + writeJSON(w, http.StatusOK, map[string]string{"status": "ready"}) +} + +// handleListServers serves the velocity registration pull (spec §7 GET /servers): +// the lifecycle view of every MinecraftServer, read from the CRD + status. +func (a *API) handleListServers(w http.ResponseWriter, r *http.Request) { + servers, err := a.Cluster.ListServers(r.Context()) + if err != nil { + writeError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"servers": servers}) +} + +// handleByHost resolves host=subdomain.{root_domain} to its server (spec §7 +// GET /servers/by-host/{host}). The host is validated against the configured +// root domain — the only place the deployment zone enters the lookup. +func (a *API) handleByHost(w http.ResponseWriter, r *http.Request) { + host := strings.ToLower(r.PathValue("host")) + if err := naming.ValidateHostname(host, a.RootDomain); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_host", "invalid host: %v", err)) + return + } + subdomain := strings.TrimSuffix(host, "."+a.RootDomain) + + info, err := a.Cluster.GetBySubdomain(r.Context(), subdomain) + if err != nil { + a.writeLookupError(w, r, err) + return + } + writeJSON(w, http.StatusOK, info) +} + +// handleReady accepts a backend's push that a server is up (spec §7 +// /internal/servers/{name}/ready). The RCON probe is the authoritative gate, so +// this is advisory: it audits the signal and returns 204. +func (a *API) handleReady(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + _ = a.Repo.Audit(r.Context(), AuditEntry{ + Actor: "backend", Source: "internal", Action: "ready", ServerName: name, + RequestID: requestIDFromContext(r.Context()), + }) + w.WriteHeader(http.StatusNoContent) +} + +// joinEventRequest is the velocity real-player-join report body. +type joinEventRequest struct { + MCUUID string `json:"mc_uuid"` +} + +// handleJoinEvent records a real player join (spec §7 /internal/.../join-event): +// it bumps last_active_at, clears reaper warnings, and auto-appends the UUID to +// the allowlist. This is what keeps an active server alive against the reaper. +func (a *API) handleJoinEvent(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + var req joinEventRequest + if err := decodeJSON(w, r, &req); err != nil { + writeError(w, r, err) + return + } + if req.MCUUID == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "mc_uuid is required")) + return + } + if err := a.Repo.RecordJoin(r.Context(), name, req.MCUUID); err != nil { + a.writeLookupError(w, r, err) + return + } + w.WriteHeader(http.StatusNoContent) +} + +// internalWakeRequest is the velocity domain-autostart wake body: the verified +// online-mode UUID of the player whose connection triggered the wake. +type internalWakeRequest struct { + MCUUID string `json:"mc_uuid"` +} + +// handleInternalWake is the internal-face wake (spec §9.1, §14): velocity drives +// domain-autostart with its service token, identifying the joining player by +// online-mode UUID rather than a web Principal. It pulls the same single lever as +// the external wake — autostartPolicy gate, then the shared per-server cooldown, +// then flip the CRD desiredState to Running — and reports the current phase so +// velocity knows whether to hold the player in its waiting queue or transfer +// immediately. The cooldown limiter is shared with the external face, so a wake +// already in flight (whatever its origin) returns 429; velocity treats that as +// "already waking, keep waiting", not a hard failure. +func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + var req internalWakeRequest + if err := decodeJSON(w, r, &req); err != nil { + writeError(w, r, err) + return + } + if req.MCUUID == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "mc_uuid is required")) + return + } + + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil && !errors.Is(err, ErrNotFound) { + writeError(w, r, err) + return + } + + if err := a.authorizeWakeByUUID(r.Context(), req.MCUUID, info, rec); err != nil { + writeError(w, r, err) + return + } + if !a.limiter().allow(name, a.WakeCooldown) { + writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly")) + return + } + // Global running-server cap (spec §9.1), shared with the external wake. velocity + // treats 503 at_capacity as "cluster full, hold the player", distinct from the + // 429 cooldown's "already waking, keep waiting". + ok, err := a.withinRunningCap(r.Context(), info) + if err != nil { + writeError(w, r, err) + return + } + if !ok { + writeError(w, r, newError(http.StatusServiceUnavailable, "at_capacity", + "the cluster is at its running-server cap (spec §9.1); retry once a server stops")) + return + } + + if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil { + writeError(w, r, err) + return + } + _ = a.Repo.Audit(r.Context(), AuditEntry{ + Actor: "velocity", Source: "internal", Action: "wake", ServerName: name, + RequestID: requestIDFromContext(r.Context()), + }) + writeJSON(w, http.StatusAccepted, map[string]any{ + "name": name, "desiredState": "Running", + "phase": info.Phase, "ready": info.Ready, + }) +} + +// internalClaimRequest is the velocity `Claim & Start` body: the verified +// online-mode UUID of the player claiming an ownerless server (spec §9.3, §12). +type internalClaimRequest struct { + MCUUID string `json:"mc_uuid"` +} + +// handleInternalClaim is the internal-face claim (spec §9.3, §12): the lobby's +// `Claim & Start` button drives it through velocity, identifying the claiming +// player by their verified online-mode UUID rather than a web Principal. It pulls +// the same atomic UPDATE...WHERE owner_id IS NULL lever as the external claim and +// the same quota gate, but resolves identity by UUID. The two operations §12 +// describes — claim then wake — stay separate on purpose: claim needs a link plus +// quota (here), wake needs the autostartPolicy gate (handleInternalWake); velocity +// follows a 200 here with a wake call. An unlinked UUID can own nothing, so it is +// the internal-face equivalent of the external claim's 412 not_linked, distinct +// from a 404 for a missing server. +func (a *API) handleInternalClaim(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + var req internalClaimRequest + if err := decodeJSON(w, r, &req); err != nil { + writeError(w, r, err) + return + } + if req.MCUUID == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "mc_uuid is required")) + return + } + + // ① resolve identity by UUID. An unlinked UUID (no account_links row) cannot + // establish ownership; a successful resolve already implies linked, so there is + // no separate IsLinked check (mirrors the external claim's order, link → quota + // → write). ErrNotFound here is "claimer not linked" (412), never "server + // missing" — that distinction is the claim call's, below. + userID, err := a.Repo.UserByMCUUID(r.Context(), req.MCUUID) + if err != nil { + if errors.Is(err, ErrNotFound) { + writeError(w, r, newError(http.StatusPreconditionFailed, "not_linked", + "link your Minecraft account before claiming (see /api/v1/account/link/start)")) + return + } + writeError(w, r, err) + return + } + + // ② quota gate, evaluated before the ownership write (mirrors handleClaim). + ok, err := a.Repo.QuotaAvailable(r.Context(), userID) + if err != nil { + writeError(w, r, err) + return + } + if !ok { + writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted")) + return + } + + // ③ atomic claim. A missing server is 404 (writeLookupError), distinct from the + // 412 above; a lost race (0 rows) is 409. + claimed, err := a.Repo.ClaimServer(r.Context(), name, userID) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !claimed { + writeError(w, r, newError(http.StatusConflict, "already_claimed", "server is already claimed")) + return + } + + _ = a.Repo.Audit(r.Context(), AuditEntry{ + Actor: "velocity", Source: "internal", Action: "claim", ServerName: name, + RequestID: requestIDFromContext(r.Context()), + }) + writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true}) +} + +// handleInternalMenuStatus is the lobby `/menu` projection (spec §12): everything +// the lobby GUI needs to render one server tile, composed from the lifecycle view +// (phase/ready/players from the CRD status) and the business ownership row +// (claimable = nobody owns it yet). It is the only internal response carrying +// claimable, so it has its own shape — the §11 list/by-host/status views never +// expose ownership, and folding owner data into ServerInfo would force the +// lifecycle layer to consult Postgres. +// +// claimable is ownership-only and UUID-independent: it reports whether the server +// is ownerless, not whether *this* player may claim it (the link + quota gates are +// the claim call's, not the menu's). The lobby uses it purely to choose between +// rendering `Claim & Start` (ownerless) and `Join`/`Wake` (owned). A server known +// to the cluster but missing its servers-row is treated as ownerless, so it still +// renders a sane tile rather than erroring. +func (a *API) handleInternalMenuStatus(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + claimable := true + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil && !errors.Is(err, ErrNotFound) { + writeError(w, r, err) + return + } + if rec != nil && rec.OwnerID != "" { + claimable = false + } + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, + "phase": info.Phase, + "ready": info.Ready, + "playersOnline": info.PlayersOnline, + "playersMax": info.PlayersMax, + "claimable": claimable, + }) +} + +// authorizeWakeByUUID is the internal-face counterpart of authorizeWake (spec +// §9.4): it applies the autostartPolicy gate for a wake driven by velocity, where +// the joining player is known only by their verified online-mode UUID rather than +// a web Principal. There is no admin tier on this path — a raw UUID carries no +// panel role — but the owner bypass still applies, mirroring the external gate: +// the owner waking their own server by domain passes under any policy. An unlinked +// UUID (no account_links row) cannot establish ownership and falls through to the +// policy gate, so ownerOnly/unset fails safe exactly as on the web face. +func (a *API) authorizeWakeByUUID(ctx context.Context, mcUUID string, info *ServerInfo, rec *ServerRecord) error { + // public needs no identity at all — skip the account_links resolution. + if info.AutostartPolicy == string(v1alpha1.AutostartPublic) { + return nil + } + // Owner bypass: resolve the UUID to its linked user and compare to the owner. + // A missing link is not an error here — it just means "not the owner". + if rec != nil && rec.OwnerID != "" { + switch userID, err := a.Repo.UserByMCUUID(ctx, mcUUID); { + case err == nil: + if userID == rec.OwnerID { + return nil + } + case errors.Is(err, ErrNotFound): + // unlinked UUID → fall through to the policy gate + default: + return err + } + } + switch info.AutostartPolicy { + case string(v1alpha1.AutostartAllowlist): + ok, err := a.Repo.UUIDInAllowlist(ctx, info.Name, mcUUID) + if err != nil { + return err + } + if ok { + return nil + } + return errForbidden + default: // ownerOnly or unset → only the owner (handled above) may wake + return errForbidden + } +} + +// writeLookupError maps a repo/cluster lookup error onto an HTTP status: a +// missing record is 404, an atomic precondition failure is 409, anything else is +// an opaque 500. +func (a *API) writeLookupError(w http.ResponseWriter, r *http.Request, err error) { + switch { + case errors.Is(err, ErrNotFound): + writeError(w, r, newError(http.StatusNotFound, "not_found", "not found")) + case errors.Is(err, ErrConflict): + writeError(w, r, newError(http.StatusConflict, "conflict", "conflict")) + default: + writeError(w, r, err) + } +} diff --git a/internal/api/handlers_internal_menu_test.go b/internal/api/handlers_internal_menu_test.go new file mode 100644 index 0000000..52268ad --- /dev/null +++ b/internal/api/handlers_internal_menu_test.go @@ -0,0 +1,219 @@ +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "testing" +) + +// The internal-face claim + menu pair (spec §9.3, §12) is what velocity drives for +// the felis-paper lobby, which holds no token of its own. The claim mirrors the +// external Principal-gated claim but keys identity off the verified online-mode +// UUID; the menu adds the ownership-derived `claimable` flag the §11 lifecycle +// views never carry. These assert the exact wire keys, not just status, because a +// key velocity's parser reads but the handler never emits fails silently — every +// tile would read claimable=false forever and a status-only test would still pass. + +const menuUUID = "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" + +func internalClaim(api *API, body string) *httptest.ResponseRecorder { + return do(api.InternalHandler(), "POST", "/api/v1/internal/servers/survival/claim", body, nil) +} + +func internalMenu(api *API) *httptest.ResponseRecorder { + return do(api.InternalHandler(), "GET", "/api/v1/internal/servers/survival/menu", "", nil) +} + +func TestInternalClaim(t *testing.T) { + body := `{"mc_uuid":"` + menuUUID + `"}` + + // linkAndQuota wires an API whose menuUUID is linked to userID and (optionally) + // has quota headroom, with the survival server present and claimable. + setup := func() (*API, *fakeRepo) { + repo := newFakeRepo() + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped"} + return newTestAPI(repo, cl), repo + } + + t.Run("happy path: linked + quota + ownerless → 200 claimed", func(t *testing.T) { + api, repo := setup() + repo.links[menuUUID] = "user1" // account_links: UUID → user1 (implies linked) + repo.quota["user1"] = true + repo.claimOK["survival"] = true // UPDATE ... WHERE owner_id IS NULL hits 1 row + + w := internalClaim(api, body) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + var got map[string]any + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + // Exact wire contract velocity's ClaimResponse parser reads. + if got["name"] != "survival" { + t.Fatalf("name = %v, want survival", got["name"]) + } + if got["claimed"] != true { + t.Fatalf("claimed = %v, want true", got["claimed"]) + } + // The claim is by the resolved user, and it is audited as a velocity action. + repo.assertClaimAudit(t, "survival") + }) + + t.Run("unlinked UUID → 412 not_linked, no claim", func(t *testing.T) { + api, repo := setup() + repo.claimOK["survival"] = true // would succeed if we got that far + // menuUUID is absent from repo.links → UserByMCUUID returns ErrNotFound. + + w := internalClaim(api, body) + if w.Code != http.StatusPreconditionFailed { + t.Fatalf("code = %d, want 412 (%s)", w.Code, w.Body.String()) + } + if code := decodeErr(t, w); code != "not_linked" { + t.Fatalf("error code = %q, want not_linked", code) + } + if len(repo.audits) != 0 { + t.Fatal("an unlinked claim must not be audited as a claim") + } + }) + + t.Run("over quota → 403 quota_exceeded", func(t *testing.T) { + api, repo := setup() + repo.links[menuUUID] = "user1" + repo.quota["user1"] = false // at the ceiling + repo.claimOK["survival"] = true + + w := internalClaim(api, body) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403 (%s)", w.Code, w.Body.String()) + } + if code := decodeErr(t, w); code != "quota_exceeded" { + t.Fatalf("error code = %q, want quota_exceeded", code) + } + }) + + t.Run("already claimed (lost race) → 409", func(t *testing.T) { + api, repo := setup() + repo.links[menuUUID] = "user1" + repo.quota["user1"] = true + repo.claimOK["survival"] = false // present but UPDATE hits 0 rows + + w := internalClaim(api, body) + if w.Code != http.StatusConflict { + t.Fatalf("code = %d, want 409 (%s)", w.Code, w.Body.String()) + } + if code := decodeErr(t, w); code != "already_claimed" { + t.Fatalf("error code = %q, want already_claimed", code) + } + }) + + t.Run("missing server → 404, distinct from the unlinked 412", func(t *testing.T) { + api, repo := setup() + repo.links[menuUUID] = "user1" + repo.quota["user1"] = true + // survival absent from repo.claimOK → ClaimServer returns ErrNotFound. + + w := internalClaim(api, body) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404 (%s)", w.Code, w.Body.String()) + } + }) + + t.Run("missing mc_uuid is rejected", func(t *testing.T) { + api, repo := setup() + repo.claimOK["survival"] = true + w := internalClaim(api, `{}`) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + }) +} + +func TestInternalMenuStatus(t *testing.T) { + setup := func() (*API, *fakeRepo, *fakeCluster) { + repo := newFakeRepo() + cl := newFakeCluster() + return newTestAPI(repo, cl), repo, cl + } + + t.Run("ownerless server → claimable, with exact keys", func(t *testing.T) { + api, _, cl := setup() + cl.byName["survival"] = &ServerInfo{ + Name: "survival", Phase: "Running", Ready: true, + PlayersOnline: 3, PlayersMax: 20, + } + // no servers-row at all → treated as ownerless → claimable. + + w := internalMenu(api) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + got := decodeMenu(t, w) + // Assert every key the lobby's MenuStatus parser reads, by exact name — + // the §11-trap guard: a renamed/absent key would read as a zero value. + assertEq(t, "name", got["name"], "survival") + assertEq(t, "phase", got["phase"], "Running") + assertEq(t, "ready", got["ready"], true) + assertEq(t, "playersOnline", got["playersOnline"], float64(3)) + assertEq(t, "playersMax", got["playersMax"], float64(20)) + assertEq(t, "claimable", got["claimable"], true) + }) + + t.Run("owned server → not claimable", func(t *testing.T) { + api, repo, cl := setup() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped"} + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + + got := decodeMenu(t, internalMenu(api)) + assertEq(t, "claimable", got["claimable"], false) + }) + + t.Run("seeded but ownerless (empty owner_id) → claimable", func(t *testing.T) { + api, repo, cl := setup() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped"} + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: ""} + + got := decodeMenu(t, internalMenu(api)) + assertEq(t, "claimable", got["claimable"], true) + }) + + t.Run("unknown server → 404", func(t *testing.T) { + api, _, _ := setup() + if w := internalMenu(api); w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404 (%s)", w.Code, w.Body.String()) + } + }) +} + +// ---- local helpers ---- + +func decodeMenu(t *testing.T, w *httptest.ResponseRecorder) map[string]any { + t.Helper() + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + var got map[string]any + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + return got +} + +func assertEq(t *testing.T, key string, got, want any) { + t.Helper() + if got != want { + t.Fatalf("%s = %v (%T), want %v (%T)", key, got, got, want, want) + } +} + +func (f *fakeRepo) assertClaimAudit(t *testing.T, name string) { + t.Helper() + for _, e := range f.audits { + if e.Action == "claim" && e.ServerName == name && e.Actor == "velocity" && e.Source == "internal" { + return + } + } + t.Fatalf("no velocity/internal claim audit for %q in %+v", name, f.audits) +} diff --git a/internal/api/handlers_internal_wake_test.go b/internal/api/handlers_internal_wake_test.go new file mode 100644 index 0000000..18d95ce --- /dev/null +++ b/internal/api/handlers_internal_wake_test.go @@ -0,0 +1,176 @@ +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" +) + +// The internal-face wake (spec §9.1, §14) is what velocity drives for +// domain-autostart: service-token auth, the joining player identified by their +// verified online-mode UUID rather than a web Principal. These exercise the same +// autostartPolicy gate as the external wake but keyed by UUID, plus the shared +// cooldown and the phase reported back so velocity can decide whether to wait. + +const wakeUUID = "11111111-2222-3333-4444-555555555555" + +func newInternalWakeAPI(policy string) (*API, *fakeCluster) { + repo := newFakeRepo() + cl := newFakeCluster() + cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped", AutostartPolicy: policy} + return newTestAPI(repo, cl), cl +} + +func internalWake(api *API, body string) *httptest.ResponseRecorder { + return do(api.InternalHandler(), "POST", "/api/v1/internal/servers/survival/wake", body, nil) +} + +func TestInternalWakeAutostartGate(t *testing.T) { + body := `{"mc_uuid":"` + wakeUUID + `"}` + + t.Run("public: any UUID wakes and gets phase back", func(t *testing.T) { + api, cl := newInternalWakeAPI("public") + w := internalWake(api, body) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desiredState = %q, want Running", cl.desired["survival"]) + } + // velocity reads phase/ready to decide whether to hold the player. + var got map[string]any + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v (%s)", err, w.Body.String()) + } + if got["phase"] != "Stopped" { + t.Fatalf("phase = %v, want Stopped", got["phase"]) + } + if got["ready"] != false { + t.Fatalf("ready = %v, want false", got["ready"]) + } + }) + + t.Run("missing mc_uuid is rejected", func(t *testing.T) { + api, cl := newInternalWakeAPI("public") + w := internalWake(api, `{}`) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } + if _, set := cl.desired["survival"]; set { + t.Fatal("desiredState must not change when the UUID is missing") + } + }) + + t.Run("ownerOnly: unlinked UUID forbidden", func(t *testing.T) { + api, cl := newInternalWakeAPI("ownerOnly") + api.Repo.(*fakeRepo).byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + w := internalWake(api, body) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if _, set := cl.desired["survival"]; set { + t.Fatal("desiredState must not change on a forbidden wake") + } + }) + + t.Run("ownerOnly: owner's linked UUID wakes", func(t *testing.T) { + api, cl := newInternalWakeAPI("ownerOnly") + repo := api.Repo.(*fakeRepo) + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + repo.links[wakeUUID] = "owner1" // account_links: this UUID belongs to owner1 + if w := internalWake(api, body); w.Code != http.StatusAccepted { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desiredState = %q, want Running", cl.desired["survival"]) + } + }) + + t.Run("ownerOnly: a different linked user is still forbidden", func(t *testing.T) { + api, _ := newInternalWakeAPI("ownerOnly") + repo := api.Repo.(*fakeRepo) + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + repo.links[wakeUUID] = "someone-else" + if w := internalWake(api, body); w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + }) + + t.Run("allowlist: only a listed UUID wakes", func(t *testing.T) { + api, _ := newInternalWakeAPI("allowlist") + repo := api.Repo.(*fakeRepo) + repo.allowUUID["survival"] = map[string]bool{wakeUUID: true} + if w := internalWake(api, body); w.Code != http.StatusAccepted { + t.Fatalf("listed UUID: code = %d body %s", w.Code, w.Body.String()) + } + + api2, _ := newInternalWakeAPI("allowlist") // a different, unlisted UUID + other := `{"mc_uuid":"99999999-0000-0000-0000-000000000000"}` + if w := internalWake(api2, other); w.Code != http.StatusForbidden { + t.Fatalf("unlisted UUID: code = %d, want 403", w.Code) + } + }) + + t.Run("allowlist: owner bypasses the list even when not on it", func(t *testing.T) { + api, cl := newInternalWakeAPI("allowlist") + repo := api.Repo.(*fakeRepo) + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + repo.links[wakeUUID] = "owner1" // owner, but allowUUID is empty + if w := internalWake(api, body); w.Code != http.StatusAccepted { + t.Fatalf("owner: code = %d body %s", w.Code, w.Body.String()) + } + if cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desiredState = %q, want Running", cl.desired["survival"]) + } + }) +} + +func TestInternalWakeCooldownIsShared(t *testing.T) { + api, _ := newInternalWakeAPI("public") + api.WakeCooldown = time.Minute + body := `{"mc_uuid":"` + wakeUUID + `"}` + + if w := internalWake(api, body); w.Code != http.StatusAccepted { + t.Fatalf("first wake code = %d", w.Code) + } + // clock is frozen, so the second wake is inside the cooldown window; velocity + // reads this 429 as "already waking, keep waiting", not a failure. + if w := internalWake(api, body); w.Code != http.StatusTooManyRequests { + t.Fatalf("second wake code = %d, want 429", w.Code) + } +} + +// TestInternalWakeRunningCap proves the §9.1 cap is shared by the velocity-driven +// internal wake, not only the web wake: with one slot and another server already +// desired-Running, a join-triggered wake of a stopped server is held with 503 +// at_capacity and leaves desiredState untouched. velocity reads this as "cluster +// full, hold the player", distinct from the 429 cooldown's "already waking". +func TestInternalWakeRunningCap(t *testing.T) { + api, cl := newInternalWakeAPI("public") + api.MaxRunningServers = 1 + cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}} + body := `{"mc_uuid":"` + wakeUUID + `"}` + + w := internalWake(api, body) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("code = %d, want 503 at_capacity (body %s)", w.Code, w.Body.String()) + } + if code := decodeErr(t, w); code != "at_capacity" { + t.Fatalf("error code = %q, want at_capacity", code) + } + if _, set := cl.desired["survival"]; set { + t.Fatal("desiredState must not change when the cluster is at capacity") + } +} + +func TestInternalStatusServedOnInternalFace(t *testing.T) { + api, _ := newInternalWakeAPI("public") + w := do(api.InternalHandler(), "GET", "/api/v1/internal/servers/survival/status", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("status code = %d body %s", w.Code, w.Body.String()) + } +} diff --git a/internal/api/handlers_logstream.go b/internal/api/handlers_logstream.go new file mode 100644 index 0000000..52d2d95 --- /dev/null +++ b/internal/api/handlers_logstream.go @@ -0,0 +1,89 @@ +package api + +import ( + "errors" + "net/http" + + "felis.lolicon.best/internal/naming" +) + +// handleServerConsole streams the caller's server console as Server-Sent Events +// (spec §8, 读写分离: 读=pods/log follow; spec §262 GET /servers/{name}/console # +// SSE). It is the read counterpart to handleCommand (写=RCON): the write side +// dials RCON, this side relays the pod log so the panel sees join/聊天/异步打印 +// and live boot progress. Like the write side it is app-tier — your OWN server — +// gated by isOwnerOrAdmin. +// +// The order mirrors handleCommand up to the point of streaming, then diverges: +// +// ① name validation (400) +// ② ownership: owner or admin, else 403 (404 if the server is unknown) +// ③ the nil-Logs guard (503), so a misconstructed API fails clean, not panics +// ④ open the follow stream; map its errors to a normal JSON envelope: +// ErrNotFound → 409 not_running (no running pod to read) +// ErrConsoleUnavailable → 503 (a pod exists but its log can't be opened) +// ⑤ relay as SSE (relayLogStream), past which no error envelope is possible +// +// It deliberately does NOT pre-check readiness the way handleCommand does. A +// write only makes sense on a Ready server, but a *read* most wants the logs +// precisely while the server is booting — and §141 readiness IS an RCON probe a +// booting pod fails *while emitting the very boot logs the operator wants to +// watch*. So the read path gates on "is there a running pod" (inside StreamLogs), +// not on RCON readiness. +// +// The RCON password plays no part here at all: the read side never touches it +// (spec §286). The audit row is written BEFORE the relay, because once SSE +// framing begins the handler blocks for the lifetime of the stream — auditing +// after relayLogStream returns would mis-timestamp the attach to the detach. +func (a *API) handleServerConsole(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + // Ownership: owner or admin, mirroring handleCommand. An unknown server is 404. + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !a.isOwnerOrAdmin(p, rec) { + writeError(w, r, errForbidden) + return + } + + // Logs is wired in production (cmd/felis); the nil guard only defends against a + // misconstructed API, failing as 503 rather than panicking — same contract as + // the write side's Console guard. + if a.Logs == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable", + "console subsystem is not configured")) + return + } + + // Open the follow stream. Every error must be resolved HERE, into a normal JSON + // envelope, because relayLogStream commits the 200 + SSE headers and no error + // body can follow it. + src, err := a.Logs.StreamLogs(r.Context(), name) + switch { + case errors.Is(err, ErrConsoleUnavailable): + writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable", + "server console is currently unreachable; wake the server and retry")) + return + case errors.Is(err, ErrNotFound): + // No running pod to read — the server is stopped or not yet scheduled. This + // is the read-side analogue of handleCommand's readiness 409. + writeError(w, r, newError(http.StatusConflict, "not_running", + "server is not running; wake it before attaching to the console")) + return + case err != nil: + writeError(w, r, newError(http.StatusInternalServerError, "internal", + "could not open server console")) + return + } + + a.audit(r, p.Email, "console.attach", name) + relayLogStream(w, r, src) +} diff --git a/internal/api/handlers_logstream_test.go b/internal/api/handlers_logstream_test.go new file mode 100644 index 0000000..269a5b4 --- /dev/null +++ b/internal/api/handlers_logstream_test.go @@ -0,0 +1,480 @@ +package api + +import ( + "context" + "io" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" + "time" +) + +// fakeLogStreamer drives the §8 read-side handler without a cluster: it returns a +// canned source (or error) and records what it was asked to stream. The real +// K8sLogStreamer's pod selection and pods/log follow are integration-only, so the +// handler is tested against this fake (spec §8 读=pods/log follow). +type fakeLogStreamer struct { + src io.ReadCloser + err error + calls int + gotName string + gotCtx context.Context + + // srcFromCtx, when set, builds the returned source from the context + // StreamLogs actually receives, instead of returning the pre-baked src. The + // teardown test uses it to prove the handler threads r.Context() through: + // the real K8sLogStreamer opens the pods/log follow under the passed ctx, so + // a source whose only cancellation path is that ctx faithfully models the + // leak surface. A handler that passed context.Background() would hand this a + // never-cancelled context and hang. + srcFromCtx func(ctx context.Context) io.ReadCloser +} + +func (f *fakeLogStreamer) StreamLogs(ctx context.Context, name string) (io.ReadCloser, error) { + f.calls++ + f.gotName = name + f.gotCtx = ctx + if f.err != nil { + return nil, f.err + } + if f.srcFromCtx != nil { + return f.srcFromCtx(ctx), nil + } + return f.src, nil +} + +// recordReadCloser is a finite log source that records whether Close ran, so the +// relay's teardown (deferred src.Close) can be asserted. +type recordReadCloser struct { + r *strings.Reader + closed bool +} + +func (s *recordReadCloser) Read(p []byte) (int, error) { return s.r.Read(p) } +func (s *recordReadCloser) Close() error { s.closed = true; return nil } + +// ctxBlockingReadCloser emits one line, then blocks until its context is +// cancelled — modeling a live `pods/log` follow stream that yields boot output +// and then waits. It is how the disconnect-teardown test proves a client +// disconnect (request-context cancel) unblocks the relay and releases the stream. +type ctxBlockingReadCloser struct { + ctx context.Context + first []byte + firstRead chan struct{} + sentFirst bool + closed chan struct{} +} + +func (b *ctxBlockingReadCloser) Read(p []byte) (int, error) { + if !b.sentFirst { + b.sentFirst = true + n := copy(p, b.first) + close(b.firstRead) // signal the relay has begun streaming + return n, nil + } + // No more data until the request context is cancelled (client disconnect). + <-b.ctx.Done() + return 0, b.ctx.Err() +} + +func (b *ctxBlockingReadCloser) Close() error { + close(b.closed) + return nil +} + +// TestServerConsoleStream exercises the §8 read-side SSE relay end-to-end through +// the external app-tier router: ownership (owner/admin), the nil-Logs and +// open-stream failure surface, and the SSE framing itself. The real pod-log +// follow is integration-only; this drives the handler against a fakeLogStreamer. +func TestServerConsoleStream(t *testing.T) { + owner := &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"} + + // mk builds an API whose "survival" server is owned by owner1, with a fresh + // fakeLogStreamer wired. Subtests override src/err as needed. + mk := func() (*API, *fakeRepo, *fakeLogStreamer) { + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + streamer := &fakeLogStreamer{} + api := newTestAPI(repo, newFakeCluster()) + api.Logs = streamer + return api, repo, streamer + } + + t.Run("owner attaches -> 200 SSE framing + flush + audit", func(t *testing.T) { + api, repo, streamer := mk() + streamer.src = &recordReadCloser{r: strings.NewReader("line one\nline two\n")} + api.External = staticExternal{p: owner} + + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) + + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if ct := w.Header().Get("Content-Type"); ct != "text/event-stream" { + t.Fatalf("Content-Type = %q, want text/event-stream", ct) + } + if cc := w.Header().Get("Cache-Control"); cc != "no-cache" { + t.Fatalf("Cache-Control = %q, want no-cache", cc) + } + // Each log line is framed as a single SSE data event. + for _, want := range []string{"data: line one\n\n", "data: line two\n\n"} { + if !strings.Contains(w.Body.String(), want) { + t.Fatalf("body missing %q; got %q", want, w.Body.String()) + } + } + if !w.Flushed { + t.Fatal("SSE relay must flush per event (recorder not flushed)") + } + if !streamer.src.(*recordReadCloser).closed { + t.Fatal("relay must Close the log source when the stream ends") + } + if streamer.gotName != "survival" { + t.Fatalf("streamer saw name %q, want survival", streamer.gotName) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "console.attach" || repo.audits[0].Actor != "owner1@example.net" { + t.Fatalf("audit not written as expected: %+v", repo.audits) + } + }) + + t.Run("admin attaches to another's server -> 200", func(t *testing.T) { + api, _, streamer := mk() + streamer.src = &recordReadCloser{r: strings.NewReader("boot\n")} + api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "admin1@example.net", + Role: "admin", ViaAdminAccess: true}} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if streamer.calls != 1 { + t.Fatalf("admin attach reached the streamer %d times, want 1", streamer.calls) + } + }) + + t.Run("non-owner -> 403, no stream opened", func(t *testing.T) { + api, _, streamer := mk() + api.External = staticExternal{p: &Principal{UserID: "stranger", Role: "user"}} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) + if w.Code != http.StatusForbidden { + t.Fatalf("code = %d, want 403", w.Code) + } + if streamer.calls != 0 { + t.Fatal("a forbidden caller must not open a log stream") + } + }) + + t.Run("unknown server -> 404", func(t *testing.T) { + api, _, streamer := mk() + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/missing/console", "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } + if streamer.calls != 0 { + t.Fatal("an unknown server must not open a log stream") + } + }) + + t.Run("nil Logs -> 503 console_unavailable", func(t *testing.T) { + api, _, _ := mk() + api.Logs = nil + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "console_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) + + t.Run("no running pod -> 409 not_running, no audit", func(t *testing.T) { + api, repo, streamer := mk() + streamer.err = ErrNotFound + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) + if w.Code != http.StatusConflict || decodeErr(t, w) != "not_running" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + if len(repo.audits) != 0 { + t.Fatalf("a failed attach must not audit: %+v", repo.audits) + } + }) + + t.Run("stream unavailable -> 503 console_unavailable", func(t *testing.T) { + api, _, streamer := mk() + streamer.err = ErrConsoleUnavailable + api.External = staticExternal{p: owner} + w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) + if w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "console_unavailable" { + t.Fatalf("code = %d body %s", w.Code, w.Body.String()) + } + }) +} + +// idleReadCloser is a perfectly quiet log follow: every Read blocks until the +// context is cancelled, yielding no line at all. It models a Minecraft server +// that has booted and gone silent (no chat, no log output), which is exactly the +// case the SSE heartbeat exists to keep alive. +type idleReadCloser struct { + ctx context.Context + closed chan struct{} +} + +func (b *idleReadCloser) Read(p []byte) (int, error) { + <-b.ctx.Done() + return 0, b.ctx.Err() +} + +func (b *idleReadCloser) Close() error { + if b.closed != nil { + close(b.closed) + } + return nil +} + +// signalWriter wraps the recorder so the test learns the moment the relay makes +// its first body write WITHOUT racing on the recorder's buffer: on an idle stream +// that first write can only be a heartbeat (no data line will ever arrive), so +// closing `fired` there lets the test cancel deterministically instead of sleeping +// a guessed interval. Flush is forwarded because relayLogStream type-asserts +// http.Flusher and flushes every event. +type signalWriter struct { + http.ResponseWriter + fired chan struct{} + once sync.Once +} + +func (s *signalWriter) Write(p []byte) (int, error) { + s.once.Do(func() { close(s.fired) }) + return s.ResponseWriter.Write(p) +} + +func (s *signalWriter) Flush() { + if f, ok := s.ResponseWriter.(http.Flusher); ok { + f.Flush() + } +} + +// TestServerConsoleHeartbeat proves the §8 relay keeps an idle stream warm: with +// no log line ever arriving, it must still emit SSE comment heartbeats (so an +// intermediary's idle timeout — Cloudflare's ~100s — never tears the console +// down) and must emit NO data event. heartbeatInterval is shrunk so a heartbeat +// lands within the test window; the first body write is necessarily that +// heartbeat, so the test waits for it rather than sleeping a fixed duration. +func TestServerConsoleHeartbeat(t *testing.T) { + old := heartbeatInterval + heartbeatInterval = 2 * time.Millisecond + defer func() { heartbeatInterval = old }() + + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"}} + + ctx, cancel := context.WithCancel(context.Background()) + streamer := &fakeLogStreamer{srcFromCtx: func(streamCtx context.Context) io.ReadCloser { + return &idleReadCloser{ctx: streamCtx} + }} + api.Logs = streamer + + rec := httptest.NewRecorder() + w := &signalWriter{ResponseWriter: rec, fired: make(chan struct{})} + r := httptest.NewRequest("GET", "/api/v1/servers/survival/console", nil).WithContext(ctx) + + done := make(chan struct{}) + go func() { + api.ExternalHandler().ServeHTTP(w, r) + close(done) + }() + + // Wait for the relay's first body write — on an idle stream, a heartbeat — then + // disconnect. No sleep-on-a-guessed-interval. + select { + case <-w.fired: + case <-time.After(2 * time.Second): + t.Fatal("idle relay never emitted a heartbeat") + } + cancel() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("handler did not return after client disconnect") + } + + // Read the body only after <-done, so the relay's writes happen-before this. + body := rec.Body.String() + if !strings.Contains(body, sseHeartbeat) { + t.Fatalf("idle relay emitted no SSE heartbeat comment; body = %q", body) + } + // An idle stream carries keep-alives only — never a data event. + if strings.Contains(body, "data:") { + t.Fatalf("idle relay must emit only heartbeats, got a data event: %q", body) + } +} + +// interleaveWriter signals once the relay has written BOTH a data event and a +// heartbeat, so a test can prove the two coexist on one stream without racing on +// the recorder. Each relay event is a single Write (fmt.Fprintf / io.WriteString), +// and the relay is the lone writer, so inspecting p per Write is race-free. +type interleaveWriter struct { + http.ResponseWriter + sawData bool + sawBeat bool + both chan struct{} + once sync.Once +} + +func (b *interleaveWriter) Write(p []byte) (int, error) { + s := string(p) + if strings.HasPrefix(s, "data:") { + b.sawData = true + } + if s == sseHeartbeat { + b.sawBeat = true + } + if b.sawData && b.sawBeat { + b.once.Do(func() { close(b.both) }) + } + return b.ResponseWriter.Write(p) +} + +func (b *interleaveWriter) Flush() { + if f, ok := b.ResponseWriter.(http.Flusher); ok { + f.Flush() + } +} + +// TestServerConsoleHeartbeatInterleavesWithData proves a heartbeat following a +// real log line does not corrupt it: the source emits one line then goes idle, so +// the relay writes a data event and then — past the (shrunk) interval — a +// keep-alive. The select serializes the two writes, so the data event must appear +// intact (no comment spliced into the middle of `data: …\n\n`). +func TestServerConsoleHeartbeatInterleavesWithData(t *testing.T) { + old := heartbeatInterval + heartbeatInterval = 2 * time.Millisecond + defer func() { heartbeatInterval = old }() + + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"}} + + ctx, cancel := context.WithCancel(context.Background()) + streamer := &fakeLogStreamer{srcFromCtx: func(streamCtx context.Context) io.ReadCloser { + return &ctxBlockingReadCloser{ + ctx: streamCtx, + first: []byte("boot line\n"), + firstRead: make(chan struct{}), + closed: make(chan struct{}), + } + }} + api.Logs = streamer + + rec := httptest.NewRecorder() + w := &interleaveWriter{ResponseWriter: rec, both: make(chan struct{})} + r := httptest.NewRequest("GET", "/api/v1/servers/survival/console", nil).WithContext(ctx) + + done := make(chan struct{}) + go func() { + api.ExternalHandler().ServeHTTP(w, r) + close(done) + }() + + // Wait until the relay has emitted the data event AND a heartbeat after it. + select { + case <-w.both: + case <-time.After(2 * time.Second): + t.Fatal("relay did not produce both a data event and a heartbeat") + } + cancel() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("handler did not return after disconnect") + } + + body := rec.Body.String() + // The data event must appear intact — a heartbeat must not splice into it. + if !strings.Contains(body, "data: boot line\n\n") { + t.Fatalf("data event not intact in interleaved stream; body = %q", body) + } + if !strings.Contains(body, sseHeartbeat) { + t.Fatalf("no heartbeat after the idle gap; body = %q", body) + } +} + +// TestServerConsoleDisconnectTeardown proves the linchpin of the read-side relay: +// a client disconnect (request-context cancel) unblocks the follow stream and the +// deferred Close releases it — no leaked apiserver connection. The source emits +// one line, then blocks on its context; cancelling the request must make the +// handler return AND Close the source. +func TestServerConsoleDisconnectTeardown(t *testing.T) { + repo := newFakeRepo() + repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + api := newTestAPI(repo, newFakeCluster()) + api.External = staticExternal{p: &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"}} + + ctx, cancel := context.WithCancel(context.Background()) + // The channels are owned by the test, but the source itself is built inside + // StreamLogs from the context the handler passes in — NOT from the test's ctx. + // This is what actually backs the teardown claim: the source's only + // cancellation path is the context the handler threaded through, mirroring the + // real K8sLogStreamer (GetLogs(...).Stream(ctx)). If handleServerConsole + // streamed under context.Background() instead of r.Context(), this source's + // second Read would block on a never-cancelled context, the handler would + // never return, and <-done below would time out — the test fails closed on the + // exact leak it exists to prevent. + firstRead := make(chan struct{}) + closed := make(chan struct{}) + streamer := &fakeLogStreamer{srcFromCtx: func(streamCtx context.Context) io.ReadCloser { + return &ctxBlockingReadCloser{ + ctx: streamCtx, + first: []byte("boot progress 50%\n"), + firstRead: firstRead, + closed: closed, + } + }} + api.Logs = streamer + + r := httptest.NewRequest("GET", "/api/v1/servers/survival/console", nil).WithContext(ctx) + w := httptest.NewRecorder() + + done := make(chan struct{}) + go func() { + api.ExternalHandler().ServeHTTP(w, r) + close(done) + }() + + // Wait until the relay has streamed the first line and is blocked on the next + // read, then simulate the client going away. + select { + case <-firstRead: + case <-time.After(2 * time.Second): + t.Fatal("relay never read the first log line") + } + cancel() + + // The handler must return promptly... + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("handler did not return after client disconnect (leaked stream)") + } + // ...and the deferred Close must have released the upstream stream. + select { + case <-closed: + case <-time.After(2 * time.Second): + t.Fatal("relay did not Close the log source on disconnect") + } + // Belt-and-suspenders: the context StreamLogs received must itself be Done. + // Read after <-done, so the write inside StreamLogs is happens-before this. + // Together with the source-from-ctx wiring above, this nails the one + // production-critical property — the relay streams under the request context. + select { + case <-streamer.gotCtx.Done(): + default: + t.Fatal("StreamLogs did not receive the cancellable request context (gotCtx not Done)") + } + if !strings.Contains(w.Body.String(), "data: boot progress 50%\n\n") { + t.Fatalf("expected the first event before disconnect, got %q", w.Body.String()) + } +} diff --git a/internal/api/handlers_patch_test.go b/internal/api/handlers_patch_test.go new file mode 100644 index 0000000..0a3fdba --- /dev/null +++ b/internal/api/handlers_patch_test.go @@ -0,0 +1,254 @@ +package api + +import ( + "net/http" + "net/http/httptest" + "testing" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" +) + +// newPatchAPI wires the same admin-authenticated API as create, then pre-seeds an +// existing "survival" server so a PATCH has a live target to mutate. +func newPatchAPI() (*API, *fakeRepo, *fakeCluster, *fakeBuilder) { + api, repo, cl, fb := newCreateAPI() + cl.byName["survival"] = &ServerInfo{Name: "survival", Subdomain: "survival", + AutostartPolicy: string(v1alpha1.AutostartOwnerOnly), + DesiredState: string(v1alpha1.DesiredStopped), Phase: string(v1alpha1.PhaseStopped)} + return api, repo, cl, fb +} + +func patchSurvival(api *API, body string) *httptest.ResponseRecorder { + return do(api.ExternalHandler(), "PATCH", "/api/v1/servers/survival", body, nil) +} + +// TestPatchServerDisplayName covers the simplest spec mutation end-to-end: a +// cosmetic field is patched, the merge reaches the cluster, and the admin is +// audited. It also pins that a displayName patch needs no image admission. +func TestPatchServerDisplayName(t *testing.T) { + api, repo, cl, _ := newPatchAPI() + + w := patchSurvival(api, `{"displayName":"Survival Realm"}`) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + p, ok := cl.patched["survival"] + if !ok { + t.Fatal("PatchServerSpec was not called for survival") + } + if p.DisplayName == nil || *p.DisplayName != "Survival Realm" { + t.Errorf("patched displayName = %v, want \"Survival Realm\"", p.DisplayName) + } + // Only the field set was carried; nothing else was touched. + if p.AutostartPolicy != nil || p.Image != nil || p.JavaMemory != nil || p.Resources != nil { + t.Errorf("unexpected extra fields in patch: %+v", p) + } + if len(repo.audits) != 1 || repo.audits[0].Action != "server.patch" || + repo.audits[0].Actor != "admin@example.net" { + t.Fatalf("audit not written as expected: %+v", repo.audits) + } +} + +// TestPatchServerAutostartPolicy confirms a valid policy is parsed onto the CRD +// and the lifecycle view reflects it (the fake applies the merge). +func TestPatchServerAutostartPolicy(t *testing.T) { + api, _, cl, _ := newPatchAPI() + + w := patchSurvival(api, `{"autostartPolicy":"public"}`) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + p := cl.patched["survival"] + if p.AutostartPolicy == nil || *p.AutostartPolicy != v1alpha1.AutostartPublic { + t.Fatalf("patched autostartPolicy = %v, want public", p.AutostartPolicy) + } + if got := cl.byName["survival"].AutostartPolicy; got != string(v1alpha1.AutostartPublic) { + t.Errorf("merged view autostartPolicy = %q, want public", got) + } +} + +// TestPatchServerImageReAdmitted proves a new image is re-checked against the +// whitelist before it reaches the CRD — admission is the only legal image source. +func TestPatchServerImageReAdmitted(t *testing.T) { + api, _, cl, _ := newPatchAPI() + + w := patchSurvival(api, `{"image":"`+admittedImage+`"}`) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if p := cl.patched["survival"]; p.Image == nil || *p.Image != admittedImage { + t.Fatalf("patched image = %v, want %q", p.Image, admittedImage) + } +} + +// TestPatchServerMemoryReDerives confirms a memory change re-derives the §22 +// ceiling and the JVM heap exactly as create does — from the FINAL limit. +func TestPatchServerMemoryReDerives(t *testing.T) { + api, _, cl, _ := newPatchAPI() + + w := patchSurvival(api, `{"memory":"2Gi"}`) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + p := cl.patched["survival"] + if p.JavaMemory == nil || *p.JavaMemory != "1536M" { + t.Errorf("patched JavaMemory = %v, want 1536M", p.JavaMemory) + } + if p.Resources == nil { + t.Fatal("memory patch must carry a resolved resources block (§22 ceiling)") + } + memLim, has := p.Resources.Limits[corev1.ResourceMemory] + if !has || memLim.IsZero() || memLim.Cmp(resource.MustParse("2Gi")) != 0 { + t.Errorf("memory ceiling = %v, want non-zero 2Gi", p.Resources.Limits) + } +} + +// TestPatchServerMemoryOverride confirms the resources block widens the envelope +// and the heap derives from the OVERRIDDEN ceiling, mirroring create. +func TestPatchServerMemoryOverride(t *testing.T) { + api, _, cl, _ := newPatchAPI() + + w := patchSurvival(api, `{"memory":"2Gi","resources":{"memory":"4Gi","cpu":"2"}}`) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + p := cl.patched["survival"] + if p.JavaMemory == nil || *p.JavaMemory != "3072M" { + t.Errorf("patched JavaMemory = %v, want 3072M (derived from 4Gi override)", p.JavaMemory) + } + if lim := p.Resources.Limits[corev1.ResourceMemory]; lim.Cmp(resource.MustParse("4Gi")) != 0 { + t.Errorf("memory limit = %s, want 4Gi (override)", lim.String()) + } + if lim := p.Resources.Limits[corev1.ResourceCPU]; lim.Cmp(resource.MustParse("2")) != 0 { + t.Errorf("cpu limit = %s, want 2", lim.String()) + } +} + +// TestPatchServerRejections is the validation matrix: each malformed request is +// rejected with the right status and stable error code, and (critically) NOTHING +// reaches the cluster on a rejection — the analog of create's "no CRD written". +func TestPatchServerRejections(t *testing.T) { + cases := []struct { + name string + target string // defaults to /api/v1/servers/survival + body string + wantCode int + wantErr string + }{ + { + name: "empty patch", + body: `{}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "storage is immutable", + body: `{"storage":"20Gi"}`, + wantCode: http.StatusBadRequest, wantErr: "storage_immutable", + }, + { + name: "bad name in path", + target: "/api/v1/servers/Bad_Name", + body: `{"displayName":"x"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_name", + }, + { + name: "empty autostart policy", + body: `{"autostartPolicy":""}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "bad autostart policy", + body: `{"autostartPolicy":"sometimes"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "empty image", + body: `{"image":""}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "image not whitelisted", + body: `{"image":"docker.io/evil:latest"}`, + wantCode: http.StatusBadRequest, wantErr: "image_not_whitelisted", + }, + { + name: "non-positive memory", + body: `{"memory":"0"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "garbage memory quantity", + body: `{"memory":"lots"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "memory request exceeds limit", + body: `{"memory":"2Gi","resources":{"memoryRequest":"4Gi"}}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "resources without memory", + body: `{"resources":{"cpu":"2"}}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + name: "unknown field (no free YAML)", + body: `{"subdomain":"renamed"}`, + wantCode: http.StatusBadRequest, wantErr: "bad_request", + }, + { + // A well-formed patch against a missing server reaches the cluster, which + // returns ErrNotFound -> 404. The fake records nothing on NotFound, so the + // "no spec patched" invariant still holds. + name: "server not found", + target: "/api/v1/servers/ghost", + body: `{"displayName":"x"}`, + wantCode: http.StatusNotFound, wantErr: "not_found", + }, + } + + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + api, _, cl, _ := newPatchAPI() + target := c.target + if target == "" { + target = "/api/v1/servers/survival" + } + w := do(api.ExternalHandler(), "PATCH", target, c.body, nil) + if w.Code != c.wantCode { + t.Fatalf("code = %d, want %d (%s)", w.Code, c.wantCode, w.Body.String()) + } + if got := decodeErr(t, w); got != c.wantErr { + t.Errorf("error code = %q, want %q", got, c.wantErr) + } + // Every rejection short-circuits before a spec is written. (404 reaches + // the cluster but the fake records nothing for a missing server.) + if len(cl.patched) != 0 { + t.Errorf("a rejected patch must not write a spec, got %+v", cl.patched) + } + }) + } +} + +// TestPatchServerImageWithoutBuilderIs503 proves the Builder requirement is +// scoped to IMAGE patches only: a non-image patch succeeds without a Builder, +// while an image patch fails 503 before any spec is written. +func TestPatchServerImageWithoutBuilderIs503(t *testing.T) { + api, _, cl, _ := newPatchAPI() + api.Builder = nil + + // A displayName patch needs no admission source — it still succeeds. + if w := patchSurvival(api, `{"displayName":"x"}`); w.Code != http.StatusOK { + t.Fatalf("displayName patch without Builder: code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + // An image patch has no whitelist to check against -> 503, nothing written. + w := patchSurvival(api, `{"image":"`+admittedImage+`"}`) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("image patch without Builder: code = %d, want 503 (%s)", w.Code, w.Body.String()) + } + if p := cl.patched["survival"]; p.Image != nil { + t.Error("no image may be patched without a Builder") + } +} diff --git a/internal/api/handlers_user.go b/internal/api/handlers_user.go new file mode 100644 index 0000000..06bd569 --- /dev/null +++ b/internal/api/handlers_user.go @@ -0,0 +1,721 @@ +package api + +import ( + "context" + "errors" + "fmt" + "net/http" + "strings" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/naming" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" +) + +// handleWake is the single lever (spec §9.1): it authorizes per autostartPolicy, +// applies the cooldown, and flips the CRD desiredState to Running. It does not +// transfer the player — the web flow shows status and a connect hint (spec §9.2). +func (a *API) handleWake(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil && !errors.Is(err, ErrNotFound) { + writeError(w, r, err) + return + } + + if err := a.authorizeWake(r.Context(), p, info, rec); err != nil { + writeError(w, r, err) + return + } + if !a.limiter().allow(name, a.WakeCooldown) { + writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly")) + return + } + // Global running-server cap (spec §9.1). Distinct from the per-server cooldown: + // 503 at_capacity means the cluster is full, not that this server is throttled. + ok, err := a.withinRunningCap(r.Context(), info) + if err != nil { + writeError(w, r, err) + return + } + if !ok { + writeError(w, r, newError(http.StatusServiceUnavailable, "at_capacity", + "the cluster is at its running-server cap (spec §9.1); retry once a server stops")) + return + } + + if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil { + writeError(w, r, err) + return + } + a.audit(r, p.Email, "wake", name) + writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"}) +} + +// handleStop flips desiredState to Stopped. Only the owner or an admin may stop a +// server (spec §14: operating someone else's server is admin-tier). +func (a *API) handleStop(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !a.isOwnerOrAdmin(p, rec) { + writeError(w, r, errForbidden) + return + } + + if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredStopped); err != nil { + writeError(w, r, err) + return + } + a.audit(r, p.Email, "stop", name) + writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Stopped"}) +} + +// handleClaim is the atomic claim transaction (spec §9.3): require a verified +// account link, enforce the quota gate, then UPDATE ... WHERE owner_id IS NULL. +// A lost race (0 rows) is 409. +func (a *API) handleClaim(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + // ① verified account link + linked, err := a.Repo.IsLinked(r.Context(), p.UserID) + if err != nil { + writeError(w, r, err) + return + } + if !linked { + writeError(w, r, newError(http.StatusPreconditionFailed, "not_linked", + "link your Minecraft account before claiming (see /api/v1/account/link/start)")) + return + } + + // ② quota gate, evaluated before the ownership write + ok, err := a.Repo.QuotaAvailable(r.Context(), p.UserID) + if err != nil { + writeError(w, r, err) + return + } + if !ok { + writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted")) + return + } + + // ③ atomic claim + claimed, err := a.Repo.ClaimServer(r.Context(), name, p.UserID) + if err != nil { + a.writeLookupError(w, r, err) + return + } + if !claimed { + writeError(w, r, newError(http.StatusConflict, "already_claimed", "server is already claimed")) + return + } + + // ④ audit. Allowlist population happens on first successful join (spec §9.4). + a.audit(r, p.Email, "claim", name) + writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true}) +} + +// handleStatus returns the CRD status view (spec §7 GET /servers/{name}/status). +func (a *API) handleStatus(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + info, err := a.Cluster.GetServer(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return + } + writeJSON(w, http.StatusOK, info) +} + +// handleMe returns the calling principal's own identity (spec §14 tiering). The +// panel reads it once at boot to decide which navigation surfaces to render: +// the User-Side for everyone, the Admin/SysAdmin sides only when is_admin. This +// is UX truth, NOT a security control — every admin route is independently gated +// by adminOnly + Principal.IsAdmin() server-side, so hiding a nav item never +// widens access. is_admin is computed here as IsAdmin() (Role=="admin" AND the +// admin Access path), so the client never re-derives the graded-ZT rule. App-tier: +// a principal reads only its OWN identity — the response is sourced entirely from +// the verified token, no lookup escapes it. +func (a *API) handleMe(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + writeJSON(w, http.StatusOK, map[string]any{ + "user_id": p.UserID, + "email": p.Email, + "role": p.Role, + "is_admin": p.IsAdmin(), + }) +} + +// handleMyServers lists what the caller owns or may claim (spec §7 GET /me/servers). +func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + servers, err := a.Repo.MyServers(r.Context(), p.UserID) + if err != nil { + writeError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"servers": servers}) +} + +// handleFleet is the SysAdmin cockpit's fleet-wide read: the lifecycle view of +// EVERY MinecraftServer, from CRD + status. It is the admin-tier counterpart of +// the app-tier handleMyServers — where /me/servers scopes to the caller, this +// returns the whole fleet, so it gates on the admin Zero-Trust path via adminOnly. +// +// This is a frontend-cockpit-driven extension (the SysAdmin FleetTable in +// panel/DESIGN-WEB-3SIDES.md), NOT a spec §7 route: §7 lists only the internal +// velocity pull (GET /servers, service-tier) and the app-tier GET /me/servers, +// neither of which is an external admin read. It reuses Cluster.ListServers (the +// same CRD-truth source as the velocity pull, §1) but is a DISTINCT handler so +// each route's provenance and tier stay honest, and so the two never share a +// {method, path} key — the OpenAPI parity test forbids one path carrying both the +// service and admin tiers across faces. CRD truth only: owner and the other +// Postgres business fields are deliberately not joined here (§1 — the CRD is the +// lifecycle authority, Postgres the business authority; this read stays on the +// lifecycle side). +func (a *API) handleFleet(w http.ResponseWriter, r *http.Request) { + servers, err := a.Cluster.ListServers(r.Context()) + if err != nil { + writeError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"servers": servers}) +} + +// createServerRequest is the structured §15 create-server form. This is the +// ONLY way to create a server from the Web: every field is a typed, validated +// value and decodeJSON rejects unknown fields, so a caller can never smuggle +// free-form YAML or raw CRD fields through this endpoint. +type createServerRequest struct { + Name string `json:"name"` + Subdomain string `json:"subdomain"` + DisplayName string `json:"displayName,omitempty"` + Image string `json:"image"` + Memory string `json:"memory"` + Storage string `json:"storage"` + AutostartPolicy string `json:"autostartPolicy,omitempty"` + Resources *resourceRequest `json:"resources,omitempty"` +} + +// resourceRequest is the optional override block. The required top-level memory +// already sets the pod memory limit+request (the §22 ceiling); these fields let +// an admin widen/narrow the cgroup envelope. Each is a Kubernetes quantity +// string ("500m", "2", "1Gi"). +type resourceRequest struct { + CPU string `json:"cpu,omitempty"` + CPURequest string `json:"cpuRequest,omitempty"` + Memory string `json:"memory,omitempty"` + MemoryRequest string `json:"memoryRequest,omitempty"` +} + +// handleCreateServer (spec §15) is the admin-tier structured create flow: +// validate the form, admit the image against the whitelist, seed the business +// rows, then create the MinecraftServer CRD cold (DesiredState=Stopped) and +// unowned. There is no free-YAML path — the request is a typed form. +func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) { + // The image whitelist lives in the build subsystem; with no Builder there is + // no admission source, so create cannot run safely. + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + p := principalFromContext(r.Context()) + + var body createServerRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + + // Server name and subdomain both obey the §22 portability rule and the + // reservation list. + if err := naming.ValidateServerName(body.Name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + if err := naming.ValidateServerName(body.Subdomain); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_subdomain", "invalid subdomain: %v", err)) + return + } + + policy, err := parseAutostartPolicy(body.AutostartPolicy) + if err != nil { + writeError(w, r, err) + return + } + + // Memory is required: it is both the JVM heap hint and the default pod memory + // limit+request. resolveResources fails closed if the §22 ceiling would be + // zero, so a CRD is never written without a concrete memory limit. + javaMemory, resources, err := resolveResources(body.Memory, body.Resources) + if err != nil { + writeError(w, r, err) + return + } + + storage, err := parseStorageSize(body.Storage) + if err != nil { + writeError(w, r, err) + return + } + + // Image admission is data-driven (the whitelist), never a free image string. + if strings.TrimSpace(body.Image) == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "image is required")) + return + } + admitted, err := a.Builder.ImageAdmitted(r.Context(), body.Image) + if err != nil { + writeError(w, r, err) + return + } + if !admitted { + writeError(w, r, newError(http.StatusBadRequest, "image_not_whitelisted", + "image %q is not on the whitelist", body.Image)) + return + } + + // Quota is intentionally NOT enforced here. §15 creates an UNOWNED server + // (owner_id NULL); the per-user quota is charged at claim time (spec §9.3 / + // §22). The create path is quota-free by design, not by oversight. + + // Reject a duplicate subdomain before any write. The CRD list is the + // lifecycle source; SeedServer re-checks the PG alias atomically below. + switch _, err := a.Cluster.GetBySubdomain(r.Context(), body.Subdomain); { + case err == nil: + writeError(w, r, newError(http.StatusConflict, "subdomain_taken", + "subdomain %q is already in use", body.Subdomain)) + return + case !errors.Is(err, ErrNotFound): + writeError(w, r, err) + return + } + + // Reject a duplicate server NAME before any write too. Without this, a + // dup-name create (fresh subdomain) would reach SeedServer, which would bind + // the new alias onto the PRE-EXISTING server and commit it — then CreateServer + // fails 409 but the stray alias persists. Checking the CRD here keeps the + // failed create side-effect-free. The narrow concurrent same-name race stays + // inside the documented non-transactional tradeoff, backstopped by + // CreateServer's AlreadyExists→409 below. + switch _, err := a.Cluster.GetServer(r.Context(), body.Name); { + case err == nil: + writeError(w, r, newError(http.StatusConflict, "already_exists", + "a server named %q already exists", body.Name)) + return + case !errors.Is(err, ErrNotFound): + writeError(w, r, err) + return + } + + // Seed the business rows FIRST (servers + alias). ClaimServer needs the row, + // so a CRD-only server would be unclaimable. PG-first means a later CRD + // failure leaves a claimable ghost row — acceptable, not transactional. + if err := a.Repo.SeedServer(r.Context(), body.Name, body.Subdomain); err != nil { + if errors.Is(err, ErrConflict) { + writeError(w, r, newError(http.StatusConflict, "subdomain_taken", + "subdomain %q is already in use", body.Subdomain)) + return + } + writeError(w, r, err) + return + } + + in := CreateServerInput{ + Name: body.Name, + Subdomain: body.Subdomain, + DisplayName: body.DisplayName, + Image: body.Image, + JavaMemory: javaMemory, + StorageSize: storage, + AutostartPolicy: policy, + Resources: resources, + } + if err := a.Cluster.CreateServer(r.Context(), in); err != nil { + if errors.Is(err, ErrConflict) { + writeError(w, r, newError(http.StatusConflict, "already_exists", + "a server named %q already exists", body.Name)) + return + } + writeError(w, r, err) + return + } + + a.audit(r, p.Email, "server.create", body.Name) + writeJSON(w, http.StatusCreated, map[string]any{ + "name": body.Name, + "subdomain": body.Subdomain, + "desiredState": string(v1alpha1.DesiredStopped), + }) +} + +// parseAutostartPolicy maps the form value to a CRD policy. An empty value +// defaults to the safest policy (ownerOnly); any other unknown value is a 400. +func parseAutostartPolicy(s string) (v1alpha1.AutostartPolicy, error) { + switch s { + case "": + return v1alpha1.AutostartOwnerOnly, nil + case string(v1alpha1.AutostartOwnerOnly): + return v1alpha1.AutostartOwnerOnly, nil + case string(v1alpha1.AutostartPublic): + return v1alpha1.AutostartPublic, nil + case string(v1alpha1.AutostartAllowlist): + return v1alpha1.AutostartAllowlist, nil + default: + return "", newError(http.StatusBadRequest, "bad_request", + "invalid autostartPolicy %q (want ownerOnly, public, or allowlist)", s) + } +} + +// resolveResources turns the required memory string and the optional override +// block into the pod resource requirements. The top-level memory seeds both the +// memory limit and request; the override block may widen CPU and memory. It +// fails closed: the returned limit's memory is guaranteed non-zero so the §22 +// ceiling is never absent from the CRD. The first return value is the derived +// JVM max-heap string (JavaMemory / -Xmx), computed from the FINAL memory limit +// — NOT the raw form value — so an overridden ceiling is honored and the heap +// stays below the cgroup limit (see deriveJavaHeap). +func resolveResources(memory string, rr *resourceRequest) (string, corev1.ResourceRequirements, error) { + memQ, err := parsePositiveQuantity(memory, "memory") + if err != nil { + return "", corev1.ResourceRequirements{}, err + } + limits := corev1.ResourceList{corev1.ResourceMemory: memQ} + requests := corev1.ResourceList{corev1.ResourceMemory: memQ} + + if rr != nil { + if rr.Memory != "" { + q, err := parsePositiveQuantity(rr.Memory, "resources.memory") + if err != nil { + return "", corev1.ResourceRequirements{}, err + } + limits[corev1.ResourceMemory] = q + } + if rr.MemoryRequest != "" { + q, err := parsePositiveQuantity(rr.MemoryRequest, "resources.memoryRequest") + if err != nil { + return "", corev1.ResourceRequirements{}, err + } + requests[corev1.ResourceMemory] = q + } + if rr.CPU != "" { + q, err := parsePositiveQuantity(rr.CPU, "resources.cpu") + if err != nil { + return "", corev1.ResourceRequirements{}, err + } + limits[corev1.ResourceCPU] = q + } + if rr.CPURequest != "" { + q, err := parsePositiveQuantity(rr.CPURequest, "resources.cpuRequest") + if err != nil { + return "", corev1.ResourceRequirements{}, err + } + requests[corev1.ResourceCPU] = q + } + } + + // A request that exceeds its limit is rejected by Kubernetes; fail fast here + // with a clear 400 instead of letting the CRD write bounce. + if memReq, memLim := requests[corev1.ResourceMemory], limits[corev1.ResourceMemory]; memReq.Cmp(memLim) > 0 { + return "", corev1.ResourceRequirements{}, newError(http.StatusBadRequest, "bad_request", + "memory request %s exceeds limit %s", memReq.String(), memLim.String()) + } + if cpuReq, hasReq := requests[corev1.ResourceCPU]; hasReq { + if cpuLim, hasLim := limits[corev1.ResourceCPU]; hasLim && cpuReq.Cmp(cpuLim) > 0 { + return "", corev1.ResourceRequirements{}, newError(http.StatusBadRequest, "bad_request", + "cpu request %s exceeds limit %s", cpuReq.String(), cpuLim.String()) + } + } + + // §22 fail-closed: never hand the operator a CRD without a concrete memory + // ceiling. This cannot trigger given the positive memQ above, but the assert + // guarantees the invariant survives future edits. + memLim, ok := limits[corev1.ResourceMemory] + if !ok || memLim.IsZero() { + return "", corev1.ResourceRequirements{}, newError(http.StatusInternalServerError, "internal", + "refusing to create a server without a memory ceiling (§22)") + } + + return deriveJavaHeap(memLim), corev1.ResourceRequirements{Limits: limits, Requests: requests}, nil +} + +// deriveJavaHeap converts the pod memory ceiling into a JVM max-heap string +// (JavaMemory → JAVA_MEMORY → -Xmx). Two reasons the raw K8s quantity cannot be +// forwarded as-is: +// +// - Format: the JVM's -Xmx accepts k/m/g (1024-based) suffixes, NOT the +// Kubernetes "Ki/Mi/Gi" forms. "-Xmx2Gi" fails to start the JVM, so we emit +// a plain "M" value, which both -Xmx and the container entrypoint accept. +// - Headroom: metaspace, thread stacks, Netty direct buffers and GC structures +// live OUTSIDE the heap. Setting -Xmx to the full cgroup limit guarantees an +// eventual OOMKill, so we reserve off-heap room (the larger of 512Mi or 25%, +// capped at half the limit) and size the heap to what remains. +// +// This is a sane default the operator/runtime may later refine; it is purely a +// derivation of the §22 ceiling and never exceeds it. +func deriveJavaHeap(limit resource.Quantity) string { + const mib = int64(1024 * 1024) + bytes := limit.Value() + + reserve := bytes / 4 + if floor := 512 * mib; reserve < floor { + reserve = floor + } + if half := bytes / 2; reserve > half { + reserve = half + } + + heapMiB := (bytes - reserve) / mib + if heapMiB < 1 { + heapMiB = 1 + } + return fmt.Sprintf("%dM", heapMiB) +} + +// parseStorageSize validates the required storage size and returns the +// canonical quantity string for the PVC (spec §15). +func parseStorageSize(s string) (string, error) { + q, err := parsePositiveQuantity(s, "storage") + if err != nil { + return "", err + } + return q.String(), nil +} + +// parsePositiveQuantity parses a Kubernetes quantity string and rejects any +// non-positive value with a 400 naming the offending field. +func parsePositiveQuantity(s, field string) (resource.Quantity, error) { + q, err := resource.ParseQuantity(s) + if err != nil { + return resource.Quantity{}, newError(http.StatusBadRequest, "bad_request", + "invalid %s quantity %q: %v", field, s, err) + } + if q.Sign() <= 0 { + return resource.Quantity{}, newError(http.StatusBadRequest, "bad_request", + "%s must be a positive quantity", field) + } + return q, nil +} + +// patchServerRequest is the structured §7 PATCH /servers/{name} form. Like the +// §15 create form it is a CLOSED set of typed fields (decodeJSON rejects unknown +// fields), so an admin can never smuggle raw CRD/YAML knobs through a patch. +// Every field is a pointer: a nil pointer means "absent — leave unchanged", +// which a plain zero value could not distinguish from "set to empty". It mutates +// only CRD-authoritative spec fields (spec §22); it deliberately has no field for +// the dual-write routing identity (name is the immutable object key; subdomain +// would desync the Postgres alias) nor for the world PVC size (see below). +type patchServerRequest struct { + DisplayName *string `json:"displayName,omitempty"` + AutostartPolicy *string `json:"autostartPolicy,omitempty"` + Image *string `json:"image,omitempty"` + Memory *string `json:"memory,omitempty"` + Resources *resourceRequest `json:"resources,omitempty"` + // Storage is recognized only so the endpoint can reject it with a precise + // reason rather than an opaque "unknown field": a StatefulSet's PVC capacity + // is immutable except for storage-class-gated expansion, which this build does + // not orchestrate. Accepting it would write a CRD change the operator cannot + // honor, so it is refused (storage_immutable) instead of silently dropped. + Storage *string `json:"storage,omitempty"` +} + +// handlePatchServer (spec §7 PATCH /servers/{name}) is the admin-tier spec +// mutation: it validates the structured form, re-admits any new image against the +// whitelist, re-derives the §22 memory ceiling, and applies a merge patch to the +// MinecraftServer CRD. Only CRD-authoritative fields move; the business layer +// (Postgres) is untouched, so the two never desync (spec §22). The admin gate is +// the adminOnly wrapper in routing — every caller here is already an admin. +func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) { + p := principalFromContext(r.Context()) + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return + } + + var body patchServerRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + + // An empty patch is a client mistake, not a no-op success. + if body.DisplayName == nil && body.AutostartPolicy == nil && body.Image == nil && + body.Memory == nil && body.Resources == nil && body.Storage == nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "patch must set at least one field")) + return + } + if body.Storage != nil { + writeError(w, r, newError(http.StatusBadRequest, "storage_immutable", + "storage size cannot be changed through this endpoint (PVC capacity is immutable)")) + return + } + + // Build the resolved patch field-by-field, validating each present field with + // the SAME helpers the create form uses. `changed` records what actually moves + // so the response and audit name the real mutation. + var patch ServerSpecPatch + var changed []string + + if body.DisplayName != nil { + patch.DisplayName = body.DisplayName + changed = append(changed, "displayName") + } + + if body.AutostartPolicy != nil { + // Unlike create, an explicit empty policy is rejected rather than defaulted: + // a patch states an intent, so "" is ambiguous, not "the safe default". + if *body.AutostartPolicy == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "autostartPolicy cannot be empty")) + return + } + policy, err := parseAutostartPolicy(*body.AutostartPolicy) + if err != nil { + writeError(w, r, err) + return + } + patch.AutostartPolicy = &policy + changed = append(changed, "autostartPolicy") + } + + if body.Image != nil { + // A new image must be re-admitted against the whitelist, exactly as create + // does — admission is the only source of a legal image. With no Builder + // there is no whitelist to check against, so the change cannot run safely. + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + if strings.TrimSpace(*body.Image) == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "image cannot be empty")) + return + } + admitted, err := a.Builder.ImageAdmitted(r.Context(), *body.Image) + if err != nil { + writeError(w, r, err) + return + } + if !admitted { + writeError(w, r, newError(http.StatusBadRequest, "image_not_whitelisted", + "image %q is not on the whitelist", *body.Image)) + return + } + patch.Image = body.Image + changed = append(changed, "image") + } + + // Memory and the resource overrides move together: resolveResources derives the + // JVM heap and the §22 non-zero ceiling from the FINAL memory limit, and the + // override block is meaningless without that base. A resources-only patch has no + // base ceiling to widen (this endpoint does not read the current spec back), so + // it is rejected rather than guessed. + if body.Memory != nil { + javaMemory, resources, err := resolveResources(*body.Memory, body.Resources) + if err != nil { + writeError(w, r, err) + return + } + patch.JavaMemory = &javaMemory + patch.Resources = &resources + changed = append(changed, "memory") + if body.Resources != nil { + changed = append(changed, "resources") + } + } else if body.Resources != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "resources overrides require memory to be set in the same patch")) + return + } + + if err := a.Cluster.PatchServerSpec(r.Context(), name, patch); err != nil { + a.writeLookupError(w, r, err) + return + } + + a.audit(r, p.Email, "server.patch", name) + writeJSON(w, http.StatusOK, map[string]any{ + "name": name, + "patched": changed, + }) +} + +// ---- authorization helpers ---- + +// authorizeWake applies the autostartPolicy gate (spec §9.4). The owner and any +// admin may always wake; otherwise the policy decides. An empty/unknown policy +// fails safe (owner-only). +func (a *API) authorizeWake(ctx context.Context, p *Principal, info *ServerInfo, rec *ServerRecord) error { + if p.IsAdmin() { + return nil + } + if rec != nil && rec.OwnerID != "" && rec.OwnerID == p.UserID { + return nil + } + switch info.AutostartPolicy { + case string(v1alpha1.AutostartPublic): + return nil + case string(v1alpha1.AutostartAllowlist): + ok, err := a.Repo.UserInAllowlist(ctx, info.Name, p.UserID) + if err != nil { + return err + } + if ok { + return nil + } + return errForbidden + default: // ownerOnly or unset → only owner/admin, already handled above + return errForbidden + } +} + +// isOwnerOrAdmin reports whether p owns rec or is an admin. +func (a *API) isOwnerOrAdmin(p *Principal, rec *ServerRecord) bool { + if p.IsAdmin() { + return true + } + return rec != nil && rec.OwnerID != "" && rec.OwnerID == p.UserID +} + +// audit writes a best-effort audit row; a logging failure must not fail the +// underlying operation, which already succeeded. +func (a *API) audit(r *http.Request, actor, action, server string) { + _ = a.Repo.Audit(r.Context(), AuditEntry{ + Actor: actor, + Source: "external", + Action: action, + ServerName: server, + RequestID: requestIDFromContext(r.Context()), + }) +} diff --git a/internal/api/images.go b/internal/api/images.go new file mode 100644 index 0000000..fa31571 --- /dev/null +++ b/internal/api/images.go @@ -0,0 +1,238 @@ +package api + +import ( + "context" + "errors" + "net/http" + + "felis.lolicon.best/internal/build" + "k8s.io/apimachinery/pkg/util/validation" +) + +// ImageBuilder is the build-subsystem surface the API depends on (spec §16). It +// is an interface so the image handlers are unit-tested against a fake; the +// production implementation is *build.Builder. All build operations are +// admin-tier (Zero Trust), enforced by adminOnly before these handlers run. +type ImageBuilder interface { + Submit(ctx context.Context, req build.Request) (*build.Build, error) + // Sync reconciles a build against its Job and returns the current view, so a + // GET doubles as the reconcile tick (idempotent on terminal builds). + Sync(ctx context.Context, id string) (*build.Build, error) + Cancel(ctx context.Context, id string) (*build.Build, error) + ListImages(ctx context.Context) ([]build.Image, error) + AddExternalImage(ctx context.Context, imageRef, addedBy string) (*build.Image, error) + RemoveImage(ctx context.Context, imageRef string) error + // ImageAdmitted reports whether a concrete image ref is whitelisted and + // enabled — the §15 create-server form gate (admission is data-driven, never + // a free image string from the body). + ImageAdmitted(ctx context.Context, imageRef string) (bool, error) +} + +// buildImageRequest is the POST /images/build body (spec §16). The push target, +// the uploaded Dockerfile, and the context reference are required; base_image is +// recorded for audit only. The requester identity comes from the Access +// principal, never the body. +type buildImageRequest struct { + ImageRef string `json:"image_ref"` + Dockerfile string `json:"dockerfile"` + ContextRef string `json:"context_ref"` + BaseImage string `json:"base_image,omitempty"` +} + +// addImageRequest is the POST /images body for external admission (spec §15). +type addImageRequest struct { + ImageRef string `json:"image_ref"` +} + +// handleBuildImage starts a build (admin-tier). It maps validation failures to +// 400 and everything else to the standard envelope. +func (a *API) handleBuildImage(w http.ResponseWriter, r *http.Request) { + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + p := principalFromContext(r.Context()) + var body buildImageRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + bld, err := a.Builder.Submit(r.Context(), build.Request{ + ImageRef: body.ImageRef, + Dockerfile: body.Dockerfile, + ContextRef: body.ContextRef, + BaseImage: body.BaseImage, + RequestedBy: p.Email, + }) + if err != nil { + writeBuildError(w, r, err) + return + } + a.audit(r, p.Email, "image.build", bld.ImageRef) + writeJSON(w, http.StatusAccepted, bld) +} + +// handleGetBuild returns a build, reconciling it against its Job first so +// polling drives the scan-gate translation without a separate background loop. +func (a *API) handleGetBuild(w http.ResponseWriter, r *http.Request) { + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + id := r.PathValue("id") + bld, err := a.Builder.Sync(r.Context(), id) + if err != nil { + writeBuildError(w, r, err) + return + } + writeJSON(w, http.StatusOK, bld) +} + +// handleBuildLogs (spec §16, §416 日志流复用 §8) streams the build Pod's log to the +// admin as Server-Sent Events. It funnels through relayLogStream — the very same +// §8 read-side relay the server console uses — after a.BuildLogs selects the build +// Job's Pod in the build namespace by build-id label and follows its kaniko +// container. It is admin-tier (adminOnly gates the route); the relay never dials +// RCON and never resolves a secret, it only reads pod logs. +// +// Every error path is resolved BEFORE the relay writes the first SSE byte — once +// the stream headers ship the status code is fixed and no error envelope can +// follow. The buildID becomes a label-selector value (build-id=), so a +// malformed id is rejected up front (400) rather than concatenated into a +// selector, mirroring how handleServerConsole validates the server name before +// touching the streamer. +func (a *API) handleBuildLogs(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + // Reject anything that is not a valid label value before it reaches the + // selector: it cannot match a real build Pod (whose build-id label is itself a + // valid label value) and, left unchecked, a crafted id (e.g. one containing a + // comma) could inject a second selector requirement. 400 is the honest answer. + if errs := validation.IsValidLabelValue(id); len(errs) > 0 { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "build id is not a valid build identifier")) + return + } + if a.BuildLogs == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "build_logs_unavailable", + "build log streaming is not configured")) + return + } + p := principalFromContext(r.Context()) + src, err := a.BuildLogs.StreamLogs(r.Context(), id) + switch { + case errors.Is(err, ErrNotFound): + writeError(w, r, newError(http.StatusNotFound, "not_found", + "no build pod found for this id (the build has not started yet, or its pod was cleaned up)")) + return + case errors.Is(err, ErrConsoleUnavailable): + writeError(w, r, newError(http.StatusServiceUnavailable, "build_logs_unavailable", + "build logs are not reachable yet (the build pod may still be pulling its image); retry shortly")) + return + case err != nil: + writeError(w, r, newError(http.StatusInternalServerError, "internal", + "could not open build logs")) + return + } + a.audit(r, p.Email, "image.build.logs", id) + relayLogStream(w, r, src) +} + +// handleCancelBuild cancels an in-flight build (admin-tier). A build that has +// already finished is 409. +func (a *API) handleCancelBuild(w http.ResponseWriter, r *http.Request) { + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + p := principalFromContext(r.Context()) + id := r.PathValue("id") + bld, err := a.Builder.Cancel(r.Context(), id) + if err != nil { + writeBuildError(w, r, err) + return + } + a.audit(r, p.Email, "image.build.cancel", bld.ImageRef) + writeJSON(w, http.StatusOK, bld) +} + +// handleListImages returns the image whitelist (spec §15: the create-server form +// source). +func (a *API) handleListImages(w http.ResponseWriter, r *http.Request) { + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + images, err := a.Builder.ListImages(r.Context()) + if err != nil { + writeBuildError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"images": images}) +} + +// handleAddImage admits an externally-built image (spec §15). It is admin-tier +// and enabled immediately, recorded with added_by for audit. +func (a *API) handleAddImage(w http.ResponseWriter, r *http.Request) { + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + p := principalFromContext(r.Context()) + var body addImageRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + img, err := a.Builder.AddExternalImage(r.Context(), body.ImageRef, p.Email) + if err != nil { + writeBuildError(w, r, err) + return + } + a.audit(r, p.Email, "image.admit", img.ImageRef) + writeJSON(w, http.StatusCreated, img) +} + +// handleRemoveImage withdraws an image from the whitelist (spec §22). The ref is +// taken from the ?ref= query parameter because image references contain '/' and +// ':' that do not round-trip cleanly through a path segment. +func (a *API) handleRemoveImage(w http.ResponseWriter, r *http.Request) { + if a.Builder == nil { + writeError(w, r, errBuildUnavailable) + return + } + p := principalFromContext(r.Context()) + ref := r.URL.Query().Get("ref") + if ref == "" { + writeError(w, r, newError(http.StatusBadRequest, "bad_request", + "the ?ref= query parameter is required")) + return + } + if err := a.Builder.RemoveImage(r.Context(), ref); err != nil { + writeBuildError(w, r, err) + return + } + a.audit(r, p.Email, "image.remove", ref) + w.WriteHeader(http.StatusNoContent) +} + +// errBuildUnavailable is returned when the build subsystem is not configured on +// this api instance. +var errBuildUnavailable = newError(http.StatusServiceUnavailable, "build_unavailable", + "image build subsystem is not configured") + +// writeBuildError maps build-package errors onto HTTP status codes: a validation +// failure is 400, a missing build/image is 404, an already-terminal build is +// 409, and anything else collapses to a 500 by writeError. +func writeBuildError(w http.ResponseWriter, r *http.Request, err error) { + switch { + case errors.Is(err, build.ErrInvalid): + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "%s", err.Error())) + case errors.Is(err, build.ErrNotFound): + writeError(w, r, newError(http.StatusNotFound, "not_found", "build or image not found")) + case errors.Is(err, build.ErrAlreadyTerminal): + writeError(w, r, newError(http.StatusConflict, "already_terminal", + "build has already finished")) + default: + writeError(w, r, err) + } +} diff --git a/internal/api/images_logs_test.go b/internal/api/images_logs_test.go new file mode 100644 index 0000000..7daad74 --- /dev/null +++ b/internal/api/images_logs_test.go @@ -0,0 +1,100 @@ +package api + +import ( + "io" + "net/http" + "strings" + "testing" +) + +// adminAPIWithBuildLogs is adminAPI plus a build-log streamer, so the §16 +// build-log handler is exercised against a fake (the real K8sBuildLogStreamer's +// pod selection + pods/log follow are integration-only). A bare fakeBuilder +// satisfies the admin gate's interface dependency. +func adminAPIWithBuildLogs(streamer LogStreamer) *API { + api := adminAPI(&fakeBuilder{}) + api.BuildLogs = streamer + return api +} + +// A valid build id streams the build Pod log as SSE and forwards the id to the +// build-namespace streamer (spec §16, §416 日志流复用 §8). A finite source (one +// line then EOF) lets the relay return without blocking on the heartbeat. +func TestBuildLogsStreamsForAdmin(t *testing.T) { + streamer := &fakeLogStreamer{src: io.NopCloser(strings.NewReader("building image...\n"))} + api := adminAPIWithBuildLogs(streamer) + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/bld-1/logs", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if ct := w.Header().Get("Content-Type"); ct != "text/event-stream" { + t.Errorf("Content-Type = %q, want text/event-stream", ct) + } + if streamer.calls != 1 { + t.Fatalf("streamer called %d times, want 1", streamer.calls) + } + if streamer.gotName != "bld-1" { + t.Errorf("streamer got id %q, want bld-1 (the build id must reach the streamer)", streamer.gotName) + } + if !strings.Contains(w.Body.String(), "data: building image...\n\n") { + t.Errorf("body missing the streamed log line as an SSE data event: %q", w.Body.String()) + } +} + +// A malformed build id is rejected up front (400) and never reaches the streamer: +// the id becomes a label-selector value, so a crafted id (here a comma + '=' that +// would inject a second selector requirement) must be refused before selection. +func TestBuildLogsRejectsMalformedID(t *testing.T) { + streamer := &fakeLogStreamer{src: io.NopCloser(strings.NewReader("x\n"))} + api := adminAPIWithBuildLogs(streamer) + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/bld-1,evil=x/logs", "", nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400 (%s)", w.Code, w.Body.String()) + } + if decodeErr(t, w) != "bad_request" { + t.Errorf("error code = %q, want bad_request", decodeErr(t, w)) + } + if streamer.calls != 0 { + t.Errorf("streamer was called %d times for a malformed id; must be 0 (no selector built)", streamer.calls) + } +} + +// With no BuildLogs streamer configured, the route reports 503 — but only after +// the admin gate and id validation, so the boundary is still enforced. +func TestBuildLogsUnconfiguredIs503(t *testing.T) { + api := adminAPI(&fakeBuilder{}) // BuildLogs deliberately left nil + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/bld-1/logs", "", nil) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("code = %d, want 503 (%s)", w.Code, w.Body.String()) + } + if decodeErr(t, w) != "build_logs_unavailable" { + t.Errorf("error code = %q, want build_logs_unavailable", decodeErr(t, w)) + } +} + +// No build Pod for the id (not scheduled yet, or GC'd after completion) is 404 — +// distinct from the 409 the §8 server console returns for a stopped server, since +// a build's pod absence is "not found", not "not running". +func TestBuildLogsNotFoundIs404(t *testing.T) { + api := adminAPIWithBuildLogs(&fakeLogStreamer{err: ErrNotFound}) + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/bld-1/logs", "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404 (%s)", w.Code, w.Body.String()) + } + if decodeErr(t, w) != "not_found" { + t.Errorf("error code = %q, want not_found", decodeErr(t, w)) + } +} + +// A pod that exists but whose log stream cannot be opened (e.g. kaniko's image is +// still pulling, container Waiting) is a transient 503 the client retries. +func TestBuildLogsUnavailableIs503(t *testing.T) { + api := adminAPIWithBuildLogs(&fakeLogStreamer{err: ErrConsoleUnavailable}) + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/bld-1/logs", "", nil) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("code = %d, want 503 (%s)", w.Code, w.Body.String()) + } + if decodeErr(t, w) != "build_logs_unavailable" { + t.Errorf("error code = %q, want build_logs_unavailable", decodeErr(t, w)) + } +} diff --git a/internal/api/images_test.go b/internal/api/images_test.go new file mode 100644 index 0000000..eec2b3e --- /dev/null +++ b/internal/api/images_test.go @@ -0,0 +1,265 @@ +package api + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "testing" + + "felis.lolicon.best/internal/build" +) + +// fakeBuilder is an in-memory ImageBuilder for the handler tests. Each field is +// the canned outcome of the matching call; the recorders let a test assert what +// the handler forwarded. +type fakeBuilder struct { + submitted *build.Request + submitErr error + syncErr error + cancelErr error + addedRef string + addedBy string + addErr error + removedRef string + removeErr error + images []build.Image + listErr error + lastBuildID string + admitted map[string]bool + admitErr error +} + +func (f *fakeBuilder) Submit(_ context.Context, req build.Request) (*build.Build, error) { + if f.submitErr != nil { + return nil, f.submitErr + } + cp := req + f.submitted = &cp + return &build.Build{ID: "bld-1", ImageRef: req.ImageRef, Status: build.StatusBuilding, + RequestedBy: req.RequestedBy}, nil +} + +func (f *fakeBuilder) Sync(_ context.Context, id string) (*build.Build, error) { + f.lastBuildID = id + if f.syncErr != nil { + return nil, f.syncErr + } + return &build.Build{ID: id, ImageRef: "registry.felis.svc:5000/x:1", Status: build.StatusSucceeded}, nil +} + +func (f *fakeBuilder) Cancel(_ context.Context, id string) (*build.Build, error) { + f.lastBuildID = id + if f.cancelErr != nil { + return nil, f.cancelErr + } + return &build.Build{ID: id, ImageRef: "registry.felis.svc:5000/x:1", Status: build.StatusCancelled}, nil +} + +func (f *fakeBuilder) ListImages(context.Context) ([]build.Image, error) { + return f.images, f.listErr +} + +func (f *fakeBuilder) AddExternalImage(_ context.Context, ref, addedBy string) (*build.Image, error) { + if f.addErr != nil { + return nil, f.addErr + } + f.addedRef, f.addedBy = ref, addedBy + return &build.Image{ImageRef: ref, Source: build.SourceExternal, AddedBy: addedBy, Enabled: true}, nil +} + +func (f *fakeBuilder) RemoveImage(_ context.Context, ref string) error { + if f.removeErr != nil { + return f.removeErr + } + f.removedRef = ref + return nil +} + +func (f *fakeBuilder) ImageAdmitted(_ context.Context, ref string) (bool, error) { + if f.admitErr != nil { + return false, f.admitErr + } + return f.admitted[ref], nil +} + +func adminAPI(b ImageBuilder) *API { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.Builder = b + api.External = staticExternal{p: &Principal{UserID: "a", Email: "admin@example.net", + Role: "admin", ViaAdminAccess: true}} + return api +} + +// Every image route is admin-tier: a non-admin is rejected before the handler. +func TestImageRoutesAreAdminOnly(t *testing.T) { + cases := []struct { + method, target, body string + }{ + {"POST", "/api/v1/images/build", `{"image_ref":"registry.felis.svc:5000/x:1","dockerfile":"FROM x","context_ref":"c"}`}, + {"GET", "/api/v1/images/build/bld-1", ""}, + {"GET", "/api/v1/images/build/bld-1/logs", ""}, + {"POST", "/api/v1/images/build/bld-1/cancel", ""}, + {"GET", "/api/v1/images", ""}, + {"POST", "/api/v1/images", `{"image_ref":"registry.felis.svc:5000/x:1"}`}, + {"DELETE", "/api/v1/images?ref=registry.felis.svc:5000/x:1", ""}, + } + for _, c := range cases { + api := adminAPI(&fakeBuilder{}) + // downgrade to a plain user + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user"}} + w := do(api.ExternalHandler(), c.method, c.target, c.body, nil) + if w.Code != http.StatusForbidden { + t.Errorf("%s %s: code = %d, want 403", c.method, c.target, w.Code) + } + } +} + +func TestBuildImageSubmits(t *testing.T) { + fb := &fakeBuilder{} + api := adminAPI(fb) + body := `{"image_ref":"registry.felis.svc:5000/mc:1","dockerfile":"FROM eclipse-temurin:21","context_ref":"tar://c.tgz"}` + w := do(api.ExternalHandler(), "POST", "/api/v1/images/build", body, nil) + if w.Code != http.StatusAccepted { + t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String()) + } + if fb.submitted == nil { + t.Fatal("Submit was not called") + } + // The requester identity must come from the Access principal, never the body. + if fb.submitted.RequestedBy != "admin@example.net" { + t.Errorf("requested_by = %q, want the admin email", fb.submitted.RequestedBy) + } + if fb.submitted.ImageRef != "registry.felis.svc:5000/mc:1" { + t.Errorf("image_ref = %q", fb.submitted.ImageRef) + } +} + +func TestBuildImageValidationIs400(t *testing.T) { + fb := &fakeBuilder{submitErr: fmt.Errorf("%w: bad ref", build.ErrInvalid)} + api := adminAPI(fb) + body := `{"image_ref":"docker.io/evil:1","dockerfile":"FROM x","context_ref":"c"}` + w := do(api.ExternalHandler(), "POST", "/api/v1/images/build", body, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400 (%s)", w.Code, w.Body.String()) + } + if decodeErr(t, w) != "bad_request" { + t.Errorf("error code = %q, want bad_request", decodeErr(t, w)) + } +} + +func TestGetBuildReconcilesOnRead(t *testing.T) { + fb := &fakeBuilder{} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/bld-9", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if fb.lastBuildID != "bld-9" { + t.Errorf("Sync called with %q, want bld-9", fb.lastBuildID) + } + var bld build.Build + if err := json.Unmarshal(w.Body.Bytes(), &bld); err != nil || bld.Status != build.StatusSucceeded { + t.Fatalf("unexpected body %s err %v", w.Body.String(), err) + } +} + +func TestGetBuildNotFoundIs404(t *testing.T) { + fb := &fakeBuilder{syncErr: build.ErrNotFound} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "GET", "/api/v1/images/build/missing", "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404", w.Code) + } +} + +func TestCancelBuild(t *testing.T) { + fb := &fakeBuilder{} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "POST", "/api/v1/images/build/bld-1/cancel", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if fb.lastBuildID != "bld-1" { + t.Errorf("Cancel called with %q", fb.lastBuildID) + } +} + +func TestCancelTerminalBuildIs409(t *testing.T) { + fb := &fakeBuilder{cancelErr: build.ErrAlreadyTerminal} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "POST", "/api/v1/images/build/bld-1/cancel", "", nil) + if w.Code != http.StatusConflict { + t.Fatalf("code = %d, want 409 (%s)", w.Code, w.Body.String()) + } +} + +func TestListImages(t *testing.T) { + fb := &fakeBuilder{images: []build.Image{ + {ImageRef: "registry.felis.svc:5000/a:1", Source: build.SourceBuilt, Enabled: true}, + }} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "GET", "/api/v1/images", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + var got map[string][]build.Image + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v", err) + } + if len(got["images"]) != 1 { + t.Fatalf("images = %d, want 1", len(got["images"])) + } +} + +func TestAddExternalImage(t *testing.T) { + fb := &fakeBuilder{} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "POST", "/api/v1/images", + `{"image_ref":"registry.felis.svc:5000/ext:1"}`, nil) + if w.Code != http.StatusCreated { + t.Fatalf("code = %d, want 201 (%s)", w.Code, w.Body.String()) + } + if fb.addedRef != "registry.felis.svc:5000/ext:1" { + t.Errorf("added ref = %q", fb.addedRef) + } + if fb.addedBy != "admin@example.net" { + t.Errorf("added_by = %q, want the admin email", fb.addedBy) + } +} + +func TestRemoveImage(t *testing.T) { + fb := &fakeBuilder{} + api := adminAPI(fb) + w := do(api.ExternalHandler(), "DELETE", "/api/v1/images?ref=registry.felis.svc:5000/x:1", "", nil) + if w.Code != http.StatusNoContent { + t.Fatalf("code = %d, want 204 (%s)", w.Code, w.Body.String()) + } + if fb.removedRef != "registry.felis.svc:5000/x:1" { + t.Errorf("removed ref = %q", fb.removedRef) + } +} + +func TestRemoveImageRequiresRef(t *testing.T) { + api := adminAPI(&fakeBuilder{}) + w := do(api.ExternalHandler(), "DELETE", "/api/v1/images", "", nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400", w.Code) + } +} + +// When no Builder is configured, image routes report 503 — but only after the +// admin gate, so the boundary is still enforced. +func TestImageRoutesWithoutBuilderAre503(t *testing.T) { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.Builder = nil + api.External = staticExternal{p: &Principal{UserID: "a", Email: "admin@example.net", + Role: "admin", ViaAdminAccess: true}} + w := do(api.ExternalHandler(), "GET", "/api/v1/images", "", nil) + if w.Code != http.StatusServiceUnavailable { + t.Fatalf("code = %d, want 503", w.Code) + } +} + +// Compile-time proof that the production Builder satisfies the API interface. +var _ ImageBuilder = (*build.Builder)(nil) diff --git a/internal/api/k8scluster.go b/internal/api/k8scluster.go new file mode 100644 index 0000000..c4fa3a3 --- /dev/null +++ b/internal/api/k8scluster.go @@ -0,0 +1,157 @@ +package api + +import ( + "context" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +// K8sCluster is the production Cluster backed by a controller-runtime client +// (spec §4). It reads the MinecraftServer CRD and performs the API's spec writes +// — the app-tier desiredState lever, the admin-tier create, and the admin-tier +// spec patch — each via a merge patch. It is integration-tested against a live +// cluster, not the hermetic api_test.go suite. +type K8sCluster struct { + c client.Client + namespace string +} + +// NewK8sCluster builds a Cluster over c, scoped to namespace. +func NewK8sCluster(c client.Client, namespace string) *K8sCluster { + return &K8sCluster{c: c, namespace: namespace} +} + +func (k *K8sCluster) GetServer(ctx context.Context, name string) (*ServerInfo, error) { + var ms v1alpha1.MinecraftServer + if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil { + if apierrors.IsNotFound(err) { + return nil, ErrNotFound + } + return nil, err + } + return serverInfo(&ms), nil +} + +func (k *K8sCluster) GetBySubdomain(ctx context.Context, subdomain string) (*ServerInfo, error) { + var list v1alpha1.MinecraftServerList + if err := k.c.List(ctx, &list, client.InNamespace(k.namespace)); err != nil { + return nil, err + } + for i := range list.Items { + if list.Items[i].Spec.Subdomain == subdomain { + return serverInfo(&list.Items[i]), nil + } + } + return nil, ErrNotFound +} + +func (k *K8sCluster) ListServers(ctx context.Context) ([]ServerInfo, error) { + var list v1alpha1.MinecraftServerList + if err := k.c.List(ctx, &list, client.InNamespace(k.namespace)); err != nil { + return nil, err + } + out := make([]ServerInfo, 0, len(list.Items)) + for i := range list.Items { + out = append(out, *serverInfo(&list.Items[i])) + } + return out, nil +} + +// CreateServer creates a MinecraftServer CRD from the validated §15 form. The +// server starts DesiredState=Stopped (created cold, woken later) and unowned — +// ownership is established by a later claim (spec §9.3). felis-api has already +// guaranteed the §22 memory ceiling lives in in.Resources, so the operator +// never has to derive a cgroup limit from JavaMemory. An existing name maps to +// ErrConflict so the handler returns 409. +func (k *K8sCluster) CreateServer(ctx context.Context, in CreateServerInput) error { + ms := &v1alpha1.MinecraftServer{ + ObjectMeta: metav1.ObjectMeta{ + Name: in.Name, + Namespace: k.namespace, + }, + Spec: v1alpha1.MinecraftServerSpec{ + Subdomain: in.Subdomain, + DisplayName: in.DisplayName, + Image: in.Image, + JavaMemory: in.JavaMemory, + DesiredState: v1alpha1.DesiredStopped, + AutostartPolicy: in.AutostartPolicy, + Storage: v1alpha1.StorageSpec{Size: in.StorageSize}, + Resources: in.Resources, + }, + } + if err := k.c.Create(ctx, ms); err != nil { + if apierrors.IsAlreadyExists(err) { + return ErrConflict + } + return err + } + return nil +} + +// SetDesiredState patches spec.desiredState with a merge patch so concurrent +// status writes by the operator are never clobbered (spec §9.1). +func (k *K8sCluster) SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error { + var ms v1alpha1.MinecraftServer + if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil { + if apierrors.IsNotFound(err) { + return ErrNotFound + } + return err + } + patch := client.MergeFrom(ms.DeepCopy()) + ms.Spec.DesiredState = state + return k.c.Patch(ctx, &ms, patch) +} + +// PatchServerSpec applies the admin-tier spec mutation (spec §7) with the same +// merge-patch discipline as SetDesiredState: read, copy, mutate only the fields +// the admin set, patch — so an operator status write racing in parallel survives. +// felis-api has already validated every field and resolved the §22 ceiling, so +// here we only translate the non-nil patch fields onto the live spec. +func (k *K8sCluster) PatchServerSpec(ctx context.Context, name string, p ServerSpecPatch) error { + var ms v1alpha1.MinecraftServer + if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &ms); err != nil { + if apierrors.IsNotFound(err) { + return ErrNotFound + } + return err + } + patch := client.MergeFrom(ms.DeepCopy()) + if p.DisplayName != nil { + ms.Spec.DisplayName = *p.DisplayName + } + if p.AutostartPolicy != nil { + ms.Spec.AutostartPolicy = *p.AutostartPolicy + } + if p.Image != nil { + ms.Spec.Image = *p.Image + } + if p.JavaMemory != nil { + ms.Spec.JavaMemory = *p.JavaMemory + } + if p.Resources != nil { + ms.Spec.Resources = *p.Resources + } + return k.c.Patch(ctx, &ms, patch) +} + +// serverInfo projects a MinecraftServer onto the API's lifecycle view. +func serverInfo(ms *v1alpha1.MinecraftServer) *ServerInfo { + return &ServerInfo{ + Name: ms.Name, + Subdomain: ms.Spec.Subdomain, + Phase: string(ms.Status.Phase), + Ready: ms.Status.Ready, + AutostartPolicy: string(ms.Spec.AutostartPolicy), + DesiredState: string(ms.Spec.DesiredState), + EndpointMode: string(ms.Status.Endpoint.Mode), + EndpointAddress: ms.Status.Endpoint.Address, + PlayersOnline: ms.Status.Players.Online, + PlayersMax: ms.Status.Players.Max, + } +} diff --git a/internal/api/logstream.go b/internal/api/logstream.go new file mode 100644 index 0000000..c4cb582 --- /dev/null +++ b/internal/api/logstream.go @@ -0,0 +1,311 @@ +package api + +import ( + "bufio" + "context" + "fmt" + "io" + "net/http" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/build" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/client-go/kubernetes" +) + +// LogStreamer is the read-side console channel (spec §8, 读写分离: 读=pods/log +// follow). StreamLogs opens a *follow* stream of the named server's container log +// and returns it as an io.ReadCloser, which relayLogStream copies to the client +// as Server-Sent Events (spec §262: GET /servers/{name}/console # SSE). It is the +// counterpart to Console (写=RCON): the read side never dials RCON and never +// resolves the RCON password — it only reads pod logs (看 join/聊天/异步打印 and +// boot progress). +// +// Contract: it returns ErrNotFound when the server has no running pod (stopped or +// not yet scheduled — the handler maps it to 409 not_running) and +// ErrConsoleUnavailable when a pod exists but its log stream cannot be opened (the +// handler maps it to 503). The returned stream MUST be closed by the caller; +// relayLogStream does so. +// +// It is an interface so the handler is tested against a fake +// (handlers_logstream_test.go); the client-go implementation (K8sLogStreamer) is +// integration-only. +type LogStreamer interface { + StreamLogs(ctx context.Context, name string) (io.ReadCloser, error) +} + +// maxLogLineBytes bounds a single log line the relay will buffer before it gives +// up on the stream. Real server log lines are well under 1 KiB; the generous cap +// (256 KiB) defends against a pathological unbounded line forcing felis-api to +// buffer without limit. A line longer than this ends the stream rather than +// growing memory without bound. +const maxLogLineBytes = 256 * 1024 + +// sseHeartbeat is the keep-alive the relay emits on an idle stream. A line whose +// first character is ':' is an SSE *comment*: EventSource ignores it entirely (it +// is never delivered as a message), but it is still bytes on the wire, which is +// all a proxy idle timer cares about. Without it a quiet Minecraft server (no +// chat / no log output) would have its console connection torn down by an +// intermediary, forcing a reconnect that replays the tail backlog. +const sseHeartbeat = ": keepalive\n\n" + +// heartbeatInterval is how often relayLogStream emits sseHeartbeat while no log +// line is flowing. It must sit comfortably under the shortest idle timeout in the +// path — Cloudflare's proxy drops an idle streamed response at ~100s — while +// staying quiet enough not to be chatty; 25s gives ~4 keep-alives per timeout +// window. It is a var, not a const, ONLY so a test can shrink it to observe a +// heartbeat without waiting; production never reassigns it. +var heartbeatInterval = 25 * time.Second + +// relayLogStream is the shared §8 read-side relay: it copies a line-oriented log +// source to the client as Server-Sent Events (spec §262 SSE, NOT WebSocket). It +// is the single reusable artifact the server console (handleServerConsole) and, +// in a later micro-slice, the build-log endpoint (handleBuildLogs, spec §16/§416 +// 日志流复用 §8) both funnel through, so the SSE framing and flush discipline are +// written and tested exactly once. +// +// It MUST be called only after the caller has committed to streaming — every +// error case resolved and already written as a normal JSON envelope. Once the SSE +// headers ship the status code is fixed and no error envelope can follow, so the +// handler maps nil-dep / not-running / unavailable BEFORE handing the source here. +// +// Teardown is governed by the caller's request context: the source is opened with +// r.Context(), so a client disconnect cancels it, the underlying Read errors, the +// scan loop exits, and the deferred Close releases the upstream stream (no leaked +// apiserver connection). +func relayLogStream(w http.ResponseWriter, r *http.Request, src io.ReadCloser) { + defer src.Close() + + // SSE needs per-event flushing; without a Flusher the bytes buffer and never + // reach the client. net/http's ResponseWriter implements it (so does + // httptest.ResponseRecorder). If somehow absent, fail as a clean 500 BEFORE any + // SSE byte — at this point no header has been written, so the status is free. + flusher, ok := w.(http.Flusher) + if !ok { + writeError(w, r, newError(http.StatusInternalServerError, "internal", + "streaming is unsupported by this server")) + return + } + + h := w.Header() + h.Set("Content-Type", "text/event-stream") + h.Set("Cache-Control", "no-cache") + h.Set("Connection", "keep-alive") + // Defeat proxy buffering (nginx / ingress) so events arrive promptly. + h.Set("X-Accel-Buffering", "no") + w.WriteHeader(http.StatusOK) + flusher.Flush() + + // bufio.Scanner.Scan blocks until a line arrives, so to interleave a periodic + // keep-alive on an idle stream the scan runs in a goroutine that feeds a + // channel, and this loop selects those lines against a ticker. The child + // context — cancelled the moment this function returns — is the leak guard: it + // unblocks the goroutine even if it is parked on the channel send (a client + // going away while a backlog line is mid-handoff), which closing src alone + // would not. w is written ONLY here in the select loop, never by the goroutine, + // so there is a single writer to the ResponseWriter. + ctx, cancel := context.WithCancel(r.Context()) + defer cancel() + + lines := make(chan string) + go func() { + defer close(lines) + scanner := bufio.NewScanner(src) + scanner.Buffer(make([]byte, 0, 64*1024), maxLogLineBytes) + for scanner.Scan() { + select { + case lines <- scanner.Text(): + case <-ctx.Done(): + return + } + } + // Scan returned false: EOF (stream ended) or a Read error (the source's + // context was cancelled on client disconnect, or the upstream closed). + }() + + ticker := time.NewTicker(heartbeatInterval) + defer ticker.Stop() + + for { + select { + case line, ok := <-lines: + if !ok { + // The scan goroutine finished — the stream is over. Return and let + // the deferred Close release the source. + return + } + // One log line → one SSE "data:" event. A write error means the client + // side is gone; stop (the deferred Close tears the upstream down too). + if _, err := fmt.Fprintf(w, "data: %s\n\n", line); err != nil { + return + } + flusher.Flush() + case <-ticker.C: + // No line for a whole interval: emit a comment so the connection stays + // warm past the proxy idle timeout. A write error means the client is + // gone; stop. + if _, err := io.WriteString(w, sseHeartbeat); err != nil { + return + } + flusher.Flush() + case <-ctx.Done(): + // Client disconnected (request context cancelled). Return; defers run. + return + } + } +} + +// serverLogContainer is the container whose logs the read-side relay streams. It +// mirrors the operator's pod container name (internal/operator builders — +// containerName), duplicated here for the same reason rconEndpoint duplicates the +// RCON address convention: the api package does not import internal/operator, so +// the single shared convention is restated with this note rather than coupling +// the app face across the runtime boundary. +const serverLogContainer = "minecraft" + +// defaultLogTailLines bounds the backlog a fresh console attach replays before it +// switches to live follow, so opening the console does not dump an entire pod log +// history. The live tail (chat / join / boot progress) is what matters. +const defaultLogTailLines = 200 + +// K8sLogStreamer is the production LogStreamer: it finds the server's pod by the +// operator's LabelServer selector (robust to the StatefulSet's -0 pod +// naming) and opens a follow stream of its container log via client-go +// (pods/log). It needs a typed kubernetes.Interface clientset, NOT the +// controller-runtime client.Client K8sConsole uses, because the log subresource +// (GetLogs(...).Stream) lives only on the typed CoreV1 client. +// +// INTEGRATION-ONLY: like K8sConsole / K8sCluster this needs a live cluster; it +// compiles here but is exercised only against a real cluster, never by the +// hermetic api_test.go suite. The Oracle verifies the handler + relay layer +// (handleServerConsole, relayLogStream) against a fake LogStreamer. +// +// Security: the read side never touches the RCON Secret or password — it only +// reads pod logs, which is why its RBAC grant is the minimal pods:list (to find +// the pod) + pods/log:get (to read it), and nothing more (internal/platform +// APIMinecraftRole). +type K8sLogStreamer struct { + clientset kubernetes.Interface + namespace string + tailLines int64 +} + +// NewK8sLogStreamer builds a LogStreamer over cs, scoped to namespace. +func NewK8sLogStreamer(cs kubernetes.Interface, namespace string) *K8sLogStreamer { + return &K8sLogStreamer{clientset: cs, namespace: namespace, tailLines: defaultLogTailLines} +} + +// StreamLogs selects the server's running pod by label and opens a follow stream +// of its container log. A missing / stopped server (no running pod) is +// ErrNotFound (the handler maps it to 409 not_running); any failure to open the +// stream collapses to ErrConsoleUnavailable (handler → 503) so no driver detail +// leaks. The stream is opened with ctx, so the handler's request context cancels +// it on client disconnect, unblocking the relay and releasing the connection. +func (k *K8sLogStreamer) StreamLogs(ctx context.Context, name string) (io.ReadCloser, error) { + // Select by the StatefulSet's own selector label (the server name); robust to + // the -0 pod naming. The name is DNS-1123-validated upstream, so it is a + // safe label-selector value. + pods, err := k.clientset.CoreV1().Pods(k.namespace).List(ctx, metav1.ListOptions{ + LabelSelector: v1alpha1.LabelServer + "=" + name, + }) + if err != nil { + return nil, ErrConsoleUnavailable + } + + podName := "" + for i := range pods.Items { + if pods.Items[i].Status.Phase == corev1.PodRunning { + podName = pods.Items[i].Name + break + } + } + if podName == "" { + // No running pod: the server is stopped or not yet scheduled. + return nil, ErrNotFound + } + + tail := k.tailLines + stream, err := k.clientset.CoreV1().Pods(k.namespace).GetLogs(podName, &corev1.PodLogOptions{ + Container: serverLogContainer, + Follow: true, + TailLines: &tail, + }).Stream(ctx) + if err != nil { + return nil, ErrConsoleUnavailable + } + return stream, nil +} + +// K8sBuildLogStreamer is the build-namespace LogStreamer (spec §16, §416 日志流复用 +// §8). It is the read side of the build subsystem: it finds the build Job's Pod by +// the build-id label and follows the kaniko container's log — the build/push +// output an admin watches live as a build runs. It deliberately does NOT reuse +// K8sLogStreamer's PodRunning filter: a Pod running its kaniko *initContainer* is +// Phase=Pending (the main trivy container has not started), so a running filter +// would never match a live build. Trivy's CRITICAL-CVE verdict is the admission +// gate, surfaced via the build status (handleGetBuild), not through this stream. +// +// INTEGRATION-ONLY: like K8sLogStreamer it needs a live cluster — the pods/log +// subresource (GetLogs(...).Stream) lives only on the typed CoreV1 client. The +// Oracle verifies the handler + relay (handleBuildLogs, relayLogStream) against a +// fake LogStreamer (images_test.go). The buildID is validated as a label value by +// the handler BEFORE it reaches here, so the selector is injection-safe. +// +// Known limitation: this stream carries the kaniko container only. While kaniko's +// image is still pulling (Pod Pending, container Waiting) GetLogs errors and the +// handler returns 503 — a transient retry state, not a failure; once kaniko +// starts, its log (plus the tailed backlog) flows. Trivy's per-CVE scan detail is +// never streamed — only its pass/fail verdict reaches the admin, via build status. +// +// Security: like the §8 read side this never touches the build SA token, the RCON +// Secret, or any password — it only lists Pods and reads pod logs, which is the +// minimal felis-api-builds grant (pods:list + pods/log:get, internal/platform +// APIBuildRole). +type K8sBuildLogStreamer struct { + clientset kubernetes.Interface + namespace string + tailLines int64 +} + +// NewK8sBuildLogStreamer builds a build-namespace LogStreamer over cs, scoped to +// namespace. namespace MUST be the same value the Builder renders Jobs into +// (cfg.Registry.BuildNamespace), so the streamer looks where the build Pods +// actually run. +func NewK8sBuildLogStreamer(cs kubernetes.Interface, namespace string) *K8sBuildLogStreamer { + return &K8sBuildLogStreamer{clientset: cs, namespace: namespace, tailLines: defaultLogTailLines} +} + +// StreamLogs selects the build Job's Pod by the build-id label and opens a follow +// stream of the kaniko container's log. No Pod (the build is not yet scheduled, or +// its Pod was garbage-collected after completion) is ErrNotFound (the handler maps +// it to 404); any failure to open the stream collapses to ErrConsoleUnavailable +// (handler → 503) so no driver detail leaks. The stream is opened with ctx, so the +// handler's request context cancels it on client disconnect, unblocking the relay +// and releasing the apiserver connection. +func (k *K8sBuildLogStreamer) StreamLogs(ctx context.Context, buildID string) (io.ReadCloser, error) { + pods, err := k.clientset.CoreV1().Pods(k.namespace).List(ctx, metav1.ListOptions{ + LabelSelector: build.LabelBuildID + "=" + buildID, + }) + if err != nil { + return nil, ErrConsoleUnavailable + } + if len(pods.Items) == 0 { + return nil, ErrNotFound + } + // backoffLimit=0 + RestartPolicyNever (build.BuildJob) means a build Job creates + // at most one Pod, so the first match is the build's Pod. + podName := pods.Items[0].Name + + tail := k.tailLines + stream, err := k.clientset.CoreV1().Pods(k.namespace).GetLogs(podName, &corev1.PodLogOptions{ + Container: build.ContainerKaniko, + Follow: true, + TailLines: &tail, + }).Stream(ctx) + if err != nil { + return nil, ErrConsoleUnavailable + } + return stream, nil +} diff --git a/internal/api/middleware.go b/internal/api/middleware.go new file mode 100644 index 0000000..6f41463 --- /dev/null +++ b/internal/api/middleware.go @@ -0,0 +1,85 @@ +package api + +import ( + "context" + "crypto/rand" + "encoding/hex" + "net/http" +) + +// withRequestID assigns a request id (honoring an inbound X-Request-Id) and +// echoes it on the response and into the context for the error envelope. +func withRequestID(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + id := r.Header.Get("X-Request-Id") + if id == "" { + id = newRequestID() + } + w.Header().Set("X-Request-Id", id) + ctx := context.WithValue(r.Context(), ctxKeyRequestID, id) + next.ServeHTTP(w, r.WithContext(ctx)) + }) +} + +// withRecover turns a panicking handler into a 500 envelope instead of a +// dropped connection. +func withRecover(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + defer func() { + if rec := recover(); rec != nil { + writeError(w, r, newError(http.StatusInternalServerError, "panic", "internal error")) + } + }() + next.ServeHTTP(w, r) + }) +} + +// requireInternal enforces service-token auth for the internal face. It never +// applies Zero Trust (spec §14 red line). +func (a *API) requireInternal(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if err := a.Internal.Authenticate(r); err != nil { + writeError(w, r, errUnauthorized) + return + } + next.ServeHTTP(w, r) + }) +} + +// requireExternal enforces Access-JWT auth for the external face and stashes the +// resolved Principal in the request context. +func (a *API) requireExternal(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + p, err := a.External.Authenticate(r) + if err != nil || p == nil { + writeError(w, r, errUnauthorized) + return + } + ctx := context.WithValue(r.Context(), ctxKeyPrincipal, p) + next.ServeHTTP(w, r.WithContext(ctx)) + }) +} + +// adminOnly gates an external-face handler on the admin Zero-Trust path. The +// Access middleware has already authenticated; this enforces that admin-tier +// operations both carry role=admin and arrived via admin.* (spec §14). +func (a *API) adminOnly(next http.HandlerFunc) http.HandlerFunc { + return func(w http.ResponseWriter, r *http.Request) { + if p := principalFromContext(r.Context()); !p.IsAdmin() { + writeError(w, r, errForbidden) + return + } + next(w, r) + } +} + +// newRequestID returns a short random hex id. crypto/rand never fails on the +// platforms we target; on the impossible error path we fall back to a constant +// so a request still gets a (non-unique) id rather than crashing. +func newRequestID() string { + var b [8]byte + if _, err := rand.Read(b[:]); err != nil { + return "req-unknown" + } + return hex.EncodeToString(b[:]) +} diff --git a/internal/api/openapi_test.go b/internal/api/openapi_test.go new file mode 100644 index 0000000..e91bcaa --- /dev/null +++ b/internal/api/openapi_test.go @@ -0,0 +1,198 @@ +package api + +import ( + "os" + "path/filepath" + "sort" + "strings" + "testing" + + "sigs.k8s.io/yaml" +) + +// This is the §28 #7 OpenAPI parity gate. docs/openapi.yaml is checked against +// the routes the handlers are actually built from — the same internalAPIRoutes / +// externalAPIRoutes tables in api.go — in BOTH directions and including each +// route's Zero-Trust tier. So three things cannot silently drift: +// +// - a route served but undocumented (or documented but unserved) fails the build; +// - a route whose documented face set disagrees with where it is mounted fails; +// - a route whose documented tier disagrees with its computed tier fails — this +// is the §14 four-power guard: an admin route quietly demoted to app, or an +// internal service route re-faced as external, breaks `go test ./...`. +// +// http.ServeMux exposes no way to enumerate its patterns, so a doc-vs-mux probe is +// impossible; reading the shared route table is the only exact check (api.go). + +// oasKnownMethods is the set of OpenAPI path-item keys that denote operations. +// Any other key (summary, parameters, description, servers, ...) is ignored. +var oasKnownMethods = map[string]bool{ + "get": true, "post": true, "put": true, "patch": true, + "delete": true, "head": true, "options": true, "trace": true, +} + +// oasDoc is the slice of docs/openapi.yaml this test reads. Body schemas and the +// rest of the document are intentionally not modelled — only the surface the +// handlers must agree with. +type oasDoc struct { + OpenAPI string `json:"openapi"` + Paths map[string]map[string]oasOp `json:"paths"` +} + +type oasOp struct { + Faces []string `json:"x-felis-face"` + Tier string `json:"x-felis-tier"` +} + +// oasFacet is the classification of one {method, path}: which face(s) serve it +// and at which Zero-Trust tier. +type oasFacet struct { + faces map[string]bool + tier string +} + +func TestOpenAPIMatchesServedRoutes(t *testing.T) { + served := oasServedFacets(t) + documented := oasDocumentedFacets(t) + + // 1. Exact {method, path} set parity, both directions. + for key := range served { + if _, ok := documented[key]; !ok { + t.Errorf("route SERVED but not documented in docs/openapi.yaml: %s", key) + } + } + for key := range documented { + if _, ok := served[key]; !ok { + t.Errorf("route DOCUMENTED in docs/openapi.yaml but not served: %s", key) + } + } + + // 2. For every shared route, face set and tier must agree exactly. + for key, s := range served { + d, ok := documented[key] + if !ok { + continue + } + if !oasSameSet(s.faces, d.faces) { + t.Errorf("%s: x-felis-face mismatch — served %v, documented %v", + key, oasSortedKeys(s.faces), oasSortedKeys(d.faces)) + } + if s.tier != d.tier { + t.Errorf("%s: x-felis-tier mismatch — served %q, documented %q", key, s.tier, d.tier) + } + } +} + +// oasServedFacets derives the live route surface from the route tables the +// handlers are built from, keyed by "METHOD /path". The tier is computed the same +// way buildFace treats the route: Public -> public; internal-face -> service; +// external-face Admin -> admin; otherwise app. /healthz is added by both tables, +// so its face set merges to {internal, external} and its tier must agree (public). +func oasServedFacets(t *testing.T) map[string]oasFacet { + t.Helper() + a := &API{} + out := map[string]oasFacet{} + add := func(method, pattern, face, tier string) { + key := method + " " + pattern + f, ok := out[key] + if !ok { + f = oasFacet{faces: map[string]bool{}} + } + f.faces[face] = true + if f.tier != "" && f.tier != tier { + t.Fatalf("%s: route table assigns conflicting tiers %q and %q", key, f.tier, tier) + } + f.tier = tier + out[key] = f + } + for _, rt := range a.internalAPIRoutes() { + tier := "service" + if rt.Public { + tier = "public" + } + add(rt.Method, rt.Pattern, "internal", tier) + } + for _, rt := range a.externalAPIRoutes() { + var tier string + switch { + case rt.Public: + tier = "public" + case rt.Admin: + tier = "admin" + default: + tier = "app" + } + add(rt.Method, rt.Pattern, "external", tier) + } + return out +} + +// oasDocumentedFacets parses docs/openapi.yaml into the same shape. An operation +// missing x-felis-face or x-felis-tier is a failure, not a skip: a new route +// cannot be documented without classifying its face and tier. +func oasDocumentedFacets(t *testing.T) map[string]oasFacet { + t.Helper() + path := filepath.Join("..", "..", "docs", "openapi.yaml") + raw, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read %s: %v", path, err) + } + var doc oasDoc + if err := yaml.Unmarshal(raw, &doc); err != nil { + t.Fatalf("parse %s: %v", path, err) + } + if !strings.HasPrefix(doc.OpenAPI, "3.1") { + t.Fatalf("%s: expected OpenAPI 3.1, got openapi: %q", path, doc.OpenAPI) + } + + out := map[string]oasFacet{} + for rawPath, item := range doc.Paths { + for method, op := range item { + if !oasKnownMethods[strings.ToLower(method)] { + continue + } + key := strings.ToUpper(method) + " " + rawPath + if len(op.Faces) == 0 { + t.Errorf("%s: missing x-felis-face", key) + continue + } + if op.Tier == "" { + t.Errorf("%s: missing x-felis-tier", key) + continue + } + faces := map[string]bool{} + for _, f := range op.Faces { + if f != "internal" && f != "external" { + t.Errorf("%s: x-felis-face has unknown face %q", key, f) + } + faces[f] = true + } + if _, dup := out[key]; dup { + t.Errorf("%s: documented more than once", key) + } + out[key] = oasFacet{faces: faces, tier: op.Tier} + } + } + return out +} + +func oasSameSet(a, b map[string]bool) bool { + if len(a) != len(b) { + return false + } + for k := range a { + if !b[k] { + return false + } + } + return true +} + +func oasSortedKeys(m map[string]bool) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + sort.Strings(out) + return out +} diff --git a/internal/api/pgrepo.go b/internal/api/pgrepo.go new file mode 100644 index 0000000..3b033f1 --- /dev/null +++ b/internal/api/pgrepo.go @@ -0,0 +1,359 @@ +package api + +import ( + "context" + "database/sql" + "errors" + "fmt" + "time" +) + +// PGRepo is the production Repo backed by Postgres (spec §6). It owns only the +// business projection the CRD cannot express. The SQL here is exercised by +// integration tests against a live database, not the hermetic api_test.go suite. +type PGRepo struct { + db *sql.DB +} + +// NewPGRepo wraps an existing pool (from store.PostgresDriver.DB()). +func NewPGRepo(db *sql.DB) *PGRepo { return &PGRepo{db: db} } + +func (p *PGRepo) ServerBySubdomain(ctx context.Context, subdomain string) (*ServerRecord, error) { + const q = `SELECT s.name, sa.subdomain, COALESCE(s.owner_id, ''), COALESCE(s.cached_phase, '') + FROM server_aliases sa JOIN servers s ON s.name = sa.server_name + WHERE sa.subdomain = $1 AND s.deleted_at IS NULL` + var r ServerRecord + switch err := p.db.QueryRowContext(ctx, q, subdomain).Scan(&r.Name, &r.Subdomain, &r.OwnerID, &r.CachedPhase); { + case errors.Is(err, sql.ErrNoRows): + return nil, ErrNotFound + case err != nil: + return nil, err + } + return &r, nil +} + +func (p *PGRepo) ServerByName(ctx context.Context, name string) (*ServerRecord, error) { + const q = `SELECT s.name, COALESCE(sa.subdomain, ''), COALESCE(s.owner_id, ''), COALESCE(s.cached_phase, '') + FROM servers s LEFT JOIN server_aliases sa ON sa.server_name = s.name + WHERE s.name = $1 AND s.deleted_at IS NULL` + var r ServerRecord + switch err := p.db.QueryRowContext(ctx, q, name).Scan(&r.Name, &r.Subdomain, &r.OwnerID, &r.CachedPhase); { + case errors.Is(err, sql.ErrNoRows): + return nil, ErrNotFound + case err != nil: + return nil, err + } + return &r, nil +} + +func (p *PGRepo) IsLinked(ctx context.Context, userID string) (bool, error) { + var ok bool + err := p.db.QueryRowContext(ctx, + `SELECT EXISTS(SELECT 1 FROM account_links WHERE user_id = $1)`, userID).Scan(&ok) + return ok, err +} + +// CreateLinkCode persists a one-time link code for a verified in-game UUID (spec +// §10). expires_at is supplied by the caller (the API clock + TTL) so expiry is +// driven by one authoritative clock. The code is a PRIMARY KEY; a collision on +// the crypto/rand value is astronomically unlikely but surfaces as a plain +// driver error (the caller can retry) rather than being masked here. +func (p *PGRepo) CreateLinkCode(ctx context.Context, code, mcUUID string, expiresAt time.Time) error { + _, err := p.db.ExecContext(ctx, + `INSERT INTO account_link_codes (code, mc_uuid, expires_at) VALUES ($1, $2, $3)`, + code, mcUUID, expiresAt) + return err +} + +// VerifyLinkCode consumes a code for userID and writes the account_links binding +// in one transaction (spec §10). The different-user conflict is detected by a +// guarded SELECT inside the tx rather than by inspecting a driver-specific unique +// violation, so the logic is portable. On the conflict path the tx rolls back, so +// the code is NOT consumed — a wrong user must not be able to burn the real +// owner's pending code. The UNIQUE(mc_uuid) constraint is the last-resort guard +// against a concurrent racer that passed the SELECT; that loses to a 500, which +// is acceptable for this integration-only path. +func (p *PGRepo) VerifyLinkCode(ctx context.Context, userID, code string, now time.Time) (string, error) { + tx, err := p.db.BeginTx(ctx, nil) + if err != nil { + return "", err + } + defer tx.Rollback() //nolint:errcheck // no-op after commit + + var mcUUID string + switch err := tx.QueryRowContext(ctx, + `SELECT mc_uuid FROM account_link_codes WHERE code = $1 AND expires_at > $2`, + code, now).Scan(&mcUUID); { + case errors.Is(err, sql.ErrNoRows): + return "", ErrLinkCodeInvalid + case err != nil: + return "", err + } + + // If this UUID is already linked, only the same user may re-verify (idempotent); + // a different user is a conflict and must not consume the code. + var existingUser string + switch err := tx.QueryRowContext(ctx, + `SELECT user_id FROM account_links WHERE mc_uuid = $1`, mcUUID).Scan(&existingUser); { + case errors.Is(err, sql.ErrNoRows): + // not yet linked — fall through to insert + case err != nil: + return "", err + default: + if existingUser != userID { + return "", ErrConflict + } + } + + if _, err := tx.ExecContext(ctx, + `INSERT INTO account_links (user_id, mc_uuid) VALUES ($1, $2) + ON CONFLICT (user_id, mc_uuid) DO NOTHING`, userID, mcUUID); err != nil { + return "", fmt.Errorf("write account link: %w", err) + } + if _, err := tx.ExecContext(ctx, + `DELETE FROM account_link_codes WHERE code = $1`, code); err != nil { + return "", fmt.Errorf("consume link code: %w", err) + } + if err := tx.Commit(); err != nil { + return "", err + } + return mcUUID, nil +} + +// QuotaAvailable treats a missing quota row or a NULL max_servers as unlimited; +// otherwise it compares the live owned-server count against the cap (spec §9.3). +func (p *PGRepo) QuotaAvailable(ctx context.Context, userID string) (bool, error) { + var maxServers sql.NullInt64 + switch err := p.db.QueryRowContext(ctx, + `SELECT max_servers FROM quotas WHERE user_id = $1`, userID).Scan(&maxServers); { + case errors.Is(err, sql.ErrNoRows): + return true, nil + case err != nil: + return false, err + } + if !maxServers.Valid { + return true, nil + } + var n int64 + if err := p.db.QueryRowContext(ctx, + `SELECT count(*) FROM servers WHERE owner_id = $1 AND deleted_at IS NULL`, userID).Scan(&n); err != nil { + return false, err + } + return n < maxServers.Int64, nil +} + +// ClaimServer performs the atomic ownership transfer (spec §9.3). A missing +// server is ErrNotFound; an existing-but-owned server yields claimed=false so the +// handler can answer 409. +func (p *PGRepo) ClaimServer(ctx context.Context, name, userID string) (bool, error) { + var exists bool + if err := p.db.QueryRowContext(ctx, + `SELECT EXISTS(SELECT 1 FROM servers WHERE name = $1 AND deleted_at IS NULL)`, name).Scan(&exists); err != nil { + return false, err + } + if !exists { + return false, ErrNotFound + } + res, err := p.db.ExecContext(ctx, + `UPDATE servers SET owner_id = $2, claimed_at = now() WHERE name = $1 AND owner_id IS NULL AND deleted_at IS NULL`, + name, userID) + if err != nil { + return false, err + } + n, err := res.RowsAffected() + if err != nil { + return false, err + } + return n == 1, nil +} + +func (p *PGRepo) UserInAllowlist(ctx context.Context, name, userID string) (bool, error) { + const q = `SELECT EXISTS( + SELECT 1 FROM server_allowlist sa JOIN account_links al ON al.mc_uuid = sa.mc_uuid + WHERE sa.server_name = $1 AND al.user_id = $2)` + var ok bool + err := p.db.QueryRowContext(ctx, q, name, userID).Scan(&ok) + return ok, err +} + +// UUIDInAllowlist is the internal-face allowlist check keyed by the in-game UUID +// directly (spec §9.4). The server_allowlist table is UUID-keyed, so the +// velocity-driven wake — which knows the joining player only by their online-mode +// UUID — needs no account_links bridge (contrast UserInAllowlist). +func (p *PGRepo) UUIDInAllowlist(ctx context.Context, name, mcUUID string) (bool, error) { + const q = `SELECT EXISTS( + SELECT 1 FROM server_allowlist WHERE server_name = $1 AND mc_uuid = $2)` + var ok bool + err := p.db.QueryRowContext(ctx, q, name, mcUUID).Scan(&ok) + return ok, err +} + +// UserByMCUUID resolves a verified in-game UUID to its linked user_id (spec §10 +// account_links), or ErrNotFound when the UUID is not linked to any account. +func (p *PGRepo) UserByMCUUID(ctx context.Context, mcUUID string) (string, error) { + var userID string + switch err := p.db.QueryRowContext(ctx, + `SELECT user_id FROM account_links WHERE mc_uuid = $1`, mcUUID).Scan(&userID); { + case errors.Is(err, sql.ErrNoRows): + return "", ErrNotFound + case err != nil: + return "", err + } + return userID, nil +} + +// RecordJoin renews activity and auto-appends the UUID to the allowlist in one +// transaction (spec §7, §9.4). A missing server is ErrNotFound. +func (p *PGRepo) RecordJoin(ctx context.Context, name, mcUUID string) error { + tx, err := p.db.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() //nolint:errcheck // no-op after commit + + res, err := tx.ExecContext(ctx, + `UPDATE servers SET last_active_at = now(), warned_3d_at = NULL, warned_1d_at = NULL + WHERE name = $1 AND deleted_at IS NULL`, name) + if err != nil { + return err + } + n, err := res.RowsAffected() + if err != nil { + return err + } + if n == 0 { + return ErrNotFound + } + if _, err := tx.ExecContext(ctx, + `INSERT INTO server_allowlist (server_name, mc_uuid) VALUES ($1, $2) ON CONFLICT DO NOTHING`, + name, mcUUID); err != nil { + return fmt.Errorf("append allowlist: %w", err) + } + return tx.Commit() +} + +func (p *PGRepo) MyServers(ctx context.Context, userID string) ([]MyServerView, error) { + const q = `SELECT s.name, COALESCE(sa.subdomain, ''), + (s.owner_id = $1) AS owned, (s.owner_id IS NULL) AS claimable, COALESCE(s.cached_phase, '') + FROM servers s LEFT JOIN server_aliases sa ON sa.server_name = s.name + WHERE s.deleted_at IS NULL AND (s.owner_id = $1 OR s.owner_id IS NULL) + ORDER BY s.name` + rows, err := p.db.QueryContext(ctx, q, userID) + if err != nil { + return nil, err + } + defer rows.Close() + var out []MyServerView + for rows.Next() { + var v MyServerView + if err := rows.Scan(&v.Name, &v.Subdomain, &v.Owned, &v.Claimable, &v.Phase); err != nil { + return nil, err + } + out = append(out, v) + } + return out, rows.Err() +} + +// SeedServer inserts the business rows backing a newly created server (spec +// §15): the servers row (owner_id left NULL — the server is created unowned and +// claimed later, spec §9.3) and its subdomain alias. Both inserts are +// ON CONFLICT DO NOTHING so a retried create is idempotent. The alias subdomain +// is a PRIMARY KEY, so a no-op insert means it was already bound; we then +// confirm it resolves to this server and return ErrConflict otherwise, letting +// the create handler answer 409 before it touches the CRD. +func (p *PGRepo) SeedServer(ctx context.Context, name, subdomain string) error { + tx, err := p.db.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() //nolint:errcheck // no-op after commit + + if _, err := tx.ExecContext(ctx, + `INSERT INTO servers (name) VALUES ($1) ON CONFLICT DO NOTHING`, name); err != nil { + return fmt.Errorf("seed server row: %w", err) + } + if _, err := tx.ExecContext(ctx, + `INSERT INTO server_aliases (subdomain, server_name) VALUES ($1, $2) ON CONFLICT DO NOTHING`, + subdomain, name); err != nil { + return fmt.Errorf("seed subdomain alias: %w", err) + } + var boundTo string + if err := tx.QueryRowContext(ctx, + `SELECT server_name FROM server_aliases WHERE subdomain = $1`, subdomain).Scan(&boundTo); err != nil { + return fmt.Errorf("confirm subdomain alias: %w", err) + } + if boundTo != name { + return ErrConflict + } + return tx.Commit() +} + +// AllBackups lists every present world backup, newest first (spec §7 GET +// /backups, admin scope; world_backups in §22). Only status='present' rows are +// listed — an expired or deleted backup is gone (spec §466). +func (p *PGRepo) AllBackups(ctx context.Context) ([]BackupView, error) { + const q = `SELECT id, server_name, COALESCE(former_owner, ''), COALESCE(size_bytes, 0), + reason, status, created_at, expires_at + FROM world_backups WHERE status = 'present' ORDER BY created_at DESC` + rows, err := p.db.QueryContext(ctx, q) + if err != nil { + return nil, err + } + return scanBackupViews(rows) +} + +// BackupsForUser lists the present world backups of worlds the user formerly +// owned, newest first (spec §7 GET /backups, former_owner scope). A NULL +// former_owner never matches a user id, so orphaned backups stay admin-only. +func (p *PGRepo) BackupsForUser(ctx context.Context, userID string) ([]BackupView, error) { + const q = `SELECT id, server_name, COALESCE(former_owner, ''), COALESCE(size_bytes, 0), + reason, status, created_at, expires_at + FROM world_backups WHERE status = 'present' AND former_owner = $1 ORDER BY created_at DESC` + rows, err := p.db.QueryContext(ctx, q, userID) + if err != nil { + return nil, err + } + return scanBackupViews(rows) +} + +// scanBackupViews drains a world_backups result set into BackupViews. backup_ref +// is intentionally not selected — it never leaves the server (spec §286 principle). +func scanBackupViews(rows *sql.Rows) ([]BackupView, error) { + defer rows.Close() + var out []BackupView + for rows.Next() { + var v BackupView + if err := rows.Scan(&v.ID, &v.ServerName, &v.FormerOwner, &v.SizeBytes, + &v.Reason, &v.Status, &v.CreatedAt, &v.ExpiresAt); err != nil { + return nil, err + } + out = append(out, v) + } + return out, rows.Err() +} + +// LatestBackup returns the most recent present backup for a server (spec §466 +// restore), or ErrNotFound. Unlike the list queries this selects backup_ref — the +// caller (the restore handler) hands it to the Restorer and never serializes it. +func (p *PGRepo) LatestBackup(ctx context.Context, serverName string) (*BackupRecord, error) { + const q = `SELECT id, server_name, COALESCE(former_owner, ''), backup_ref, COALESCE(size_bytes, 0) + FROM world_backups WHERE server_name = $1 AND status = 'present' + ORDER BY created_at DESC LIMIT 1` + var b BackupRecord + switch err := p.db.QueryRowContext(ctx, q, serverName).Scan( + &b.ID, &b.ServerName, &b.FormerOwner, &b.BackupRef, &b.SizeBytes); { + case errors.Is(err, sql.ErrNoRows): + return nil, ErrNotFound + case err != nil: + return nil, err + } + return &b, nil +} + +func (p *PGRepo) Audit(ctx context.Context, e AuditEntry) error { + _, err := p.db.ExecContext(ctx, + `INSERT INTO audit_logs (actor, source, action, server_name, request_id) + VALUES ($1, $2, $3, NULLIF($4, ''), NULLIF($5, ''))`, + e.Actor, e.Source, e.Action, e.ServerName, e.RequestID) + return err +} diff --git a/internal/api/repo.go b/internal/api/repo.go new file mode 100644 index 0000000..ad63908 --- /dev/null +++ b/internal/api/repo.go @@ -0,0 +1,142 @@ +package api + +import ( + "context" + "time" +) + +// ServerRecord is the business-layer projection of a server (spec §6 servers / +// server_aliases). It carries only what the CRD cannot express — ownership and +// the cached subdomain alias — never authoritative lifecycle fields. +type ServerRecord struct { + Name string + Subdomain string + // OwnerID is the claiming user, or "" when the server is unclaimed. + OwnerID string + // CachedPhase is the non-authoritative phase projection used for fast lists. + CachedPhase string +} + +// MyServerView is a row of GET /api/v1/me/servers: a server the caller owns, +// may auto-start, or may claim. +type MyServerView struct { + Name string `json:"name"` + Subdomain string `json:"subdomain"` + Owned bool `json:"owned"` + Claimable bool `json:"claimable"` + Phase string `json:"phase,omitempty"` +} + +// AuditEntry is one row written to audit_logs (spec §6). The actor is the Access +// email for human callers and the component name for internal callers. +type AuditEntry struct { + Actor string + Source string + Action string + ServerName string + RequestID string +} + +// BackupView is one row of GET /api/v1/backups (spec §7, world_backups in §22). +// It deliberately omits backup_ref — the opaque internal storage handle (a tar +// path / VolumeSnapshot name / Longhorn URL, spec §19) — on the same principle +// that keeps the RCON password server-side (spec §286): the client lists backups +// and triggers a restore by server, never by handle. +type BackupView struct { + ID string `json:"id"` + ServerName string `json:"server_name"` + FormerOwner string `json:"former_owner,omitempty"` + SizeBytes int64 `json:"size_bytes"` + Reason string `json:"reason"` + Status string `json:"status"` + CreatedAt time.Time `json:"created_at"` + ExpiresAt time.Time `json:"expires_at"` +} + +// BackupRecord is the server-side view of a backup used to drive a restore (spec +// §466). It carries the opaque backup_ref the Restorer needs and the former_owner +// the restore authorization compares against — neither is ever serialized to the +// client (contrast BackupView). +type BackupRecord struct { + ID string + ServerName string + FormerOwner string + BackupRef string + SizeBytes int64 +} + +// Repo is the business-layer data access the API depends on. It is an interface +// so handlers are tested against an in-memory fake; the Postgres implementation +// (pgRepo) is integration-tested only — it requires a live database. +type Repo interface { + // ServerBySubdomain resolves a subdomain alias to its server, or ErrNotFound. + ServerBySubdomain(ctx context.Context, subdomain string) (*ServerRecord, error) + // ServerByName loads a server's business projection, or ErrNotFound. + ServerByName(ctx context.Context, name string) (*ServerRecord, error) + // IsLinked reports whether the user has a verified MC account link (spec §10), + // a precondition for every ownership operation. + IsLinked(ctx context.Context, userID string) (bool, error) + // CreateLinkCode mints a one-time account-link code for an in-game player + // (spec §10: 游戏内 /link → 生成一次性码). The code is born knowing only the + // verified mc_uuid (online-mode=true established it); a web user binds it to + // their user_id later via VerifyLinkCode. This is internal-face only — the web + // has no verified UUID to mint against (the account_link_codes schema has no + // user_id column, which forces the in-game origin). + CreateLinkCode(ctx context.Context, code, mcUUID string, expiresAt time.Time) error + // VerifyLinkCode consumes a non-expired code for the logged-in user and writes + // the account_links binding, atomically (spec §10: 网页 verify 填码 → 写 + // account_links). It returns the bound mc_uuid. A missing or expired code → + // ErrLinkCodeInvalid; a uuid already linked to a *different* user → ErrConflict; + // re-verifying the same (user, uuid) pair is idempotent. now is the API clock so + // expiry is testable. + VerifyLinkCode(ctx context.Context, userID, code string, now time.Time) (mcUUID string, err error) + // QuotaAvailable reports whether the user is under their max_servers quota + // (spec §9.3 step ②, evaluated before provisioning). + QuotaAvailable(ctx context.Context, userID string) (bool, error) + // ClaimServer atomically sets owner_id where it is currently NULL and returns + // whether a row changed. false means the server was already claimed (spec §9.3: + // 0 rows → 409). + ClaimServer(ctx context.Context, name, userID string) (bool, error) + // UserInAllowlist reports whether the user's linked UUID is on the server + // allowlist (spec §9.4). + UserInAllowlist(ctx context.Context, name, userID string) (bool, error) + // UUIDInAllowlist reports whether an in-game UUID is on the server allowlist + // (spec §9.4). It is the internal-face symmetric of UserInAllowlist: velocity + // drives a domain-autostart wake knowing the joining player only by their + // online-mode UUID, never a web user_id, so the allowlist gate for that path + // must be keyed by UUID directly (the allowlist table is UUID-keyed; the + // account_links join in UserInAllowlist only exists to bridge the web side). + UUIDInAllowlist(ctx context.Context, name, mcUUID string) (bool, error) + // UserByMCUUID resolves a verified in-game UUID to the user_id it is linked to + // (spec §10 account_links), or ErrNotFound when the UUID is not linked. The + // internal-face wake uses it to apply the owner bypass for a player known only + // by UUID; an unlinked UUID simply falls through to the autostartPolicy gate. + UserByMCUUID(ctx context.Context, mcUUID string) (userID string, err error) + // RecordJoin updates last_active_at, clears reaper warnings, and auto-appends + // the UUID to the allowlist (spec §7 join-event, §9.4). + RecordJoin(ctx context.Context, name, mcUUID string) error + // MyServers lists the servers a user owns or may claim. + MyServers(ctx context.Context, userID string) ([]MyServerView, error) + // AllBackups lists every present world backup, newest first (spec §7 GET + // /backups, admin scope). Expired/deleted rows are never returned. + AllBackups(ctx context.Context) ([]BackupView, error) + // BackupsForUser lists the present world backups of worlds the user formerly + // owned, newest first (spec §7 GET /backups, former_owner scope). The + // former_owner column is stamped when the world is archived at release time, so + // a user sees their own released worlds even after the server is re-seeded. + BackupsForUser(ctx context.Context, userID string) ([]BackupView, error) + // LatestBackup returns the most recent present backup for a server, or + // ErrNotFound when none exists (spec §466 restore). The returned BackupRecord + // carries the server-side backup_ref + former_owner the restore path needs; the + // client never sees them. + LatestBackup(ctx context.Context, serverName string) (*BackupRecord, error) + // SeedServer inserts the business-layer rows for a newly created server (spec + // §15): a servers row (owner_id NULL — claimed later, spec §9.3) and its + // subdomain alias, both idempotent. It returns ErrConflict if the subdomain is + // already bound to a different server, so the create handler can fail before + // touching the CRD. ClaimServer requires this row to exist, so a CRD-only + // server would be unclaimable — the create path must seed here first. + SeedServer(ctx context.Context, name, subdomain string) error + // Audit appends one audit row. + Audit(ctx context.Context, e AuditEntry) error +} diff --git a/internal/api/restore_wire_test.go b/internal/api/restore_wire_test.go new file mode 100644 index 0000000..0119e2c --- /dev/null +++ b/internal/api/restore_wire_test.go @@ -0,0 +1,13 @@ +package api + +import ( + "felis.lolicon.best/internal/restore" +) + +// Compile-time proof that the production restore executor satisfies the API's +// Restorer interface. The dependency points one way only: api defines the +// narrow Restorer port and never imports the restore package in production +// (handlers_backups.go depends on the interface); this test is the single place +// the concrete type and the port are pinned together, mirroring how images_test +// pins build.Builder to ImageBuilder. +var _ Restorer = (*restore.Restorer)(nil) diff --git a/internal/api/restorer.go b/internal/api/restorer.go new file mode 100644 index 0000000..575a546 --- /dev/null +++ b/internal/api/restorer.go @@ -0,0 +1,26 @@ +package api + +import "context" + +// Restorer starts a world restore from a stored backup (spec §466: former_owner +// 3mo 内重新 claim → restore PVC). Restore only STARTS the work: recreating the +// per-server world PVC and extracting the archive into it is pod-filesystem work +// felis-api cannot do in-process — the world PVC is RWO and its lifecycle is owned +// by the operator's StatefulSet, so the API process has nothing to mount at +// request time. The production implementation therefore hands off to a restore +// Job, exactly the way internal/build hands an image build to a Kaniko Job. The +// call returns once the restore is enqueued, so the handler answers 202 +// (restoring), never claiming the world is already back. +// +// It returns ErrNotFound if the server is unknown to the execution backend; any +// other error is an internal failure (the handler maps it to 500). +// +// It is an interface so the handler is tested against a fake (api_test.go). The +// production executor — a restore Job mirroring internal/build's jobspec + weak-SA +// isolation — is integration-only and is a deliberate follow-up: until it is wired +// the API.Restorer is nil and POST /servers/{name}/restore-backup reports 503, so +// the restore authorization boundary is exercised without shipping a stub that +// cannot run in the real cluster topology. +type Restorer interface { + Restore(ctx context.Context, serverName, backupRef string) error +} diff --git a/internal/api/submissions.go b/internal/api/submissions.go new file mode 100644 index 0000000..6d3871e --- /dev/null +++ b/internal/api/submissions.go @@ -0,0 +1,187 @@ +package api + +import ( + "context" + "errors" + "net/http" + + "felis.lolicon.best/internal/submit" +) + +// SubmissionService is the user-modpack approval lane the API depends on (a +// user-directed extension over the §16 build subsystem — see the internal/submit +// package doc for its provenance). It is an interface so the submission handlers +// are unit-tested against a fake; the production implementation is +// *submit.Manager. The split mirrors +// ImageBuilder: an admin builds an image directly (POST /images/build), whereas +// an ordinary user may only SUBMIT a modpack here and an admin must approve it +// before anything is built. The approval routes are admin-tier (adminOnly runs +// before the handler); the /me/submissions routes are app-tier and scope to the +// authenticated principal. +// +// The identity that matters is never taken from the body: Create stamps +// SubmittedBy from the principal and the handler ignores any submitted_by a +// client tries to send (decodeJSON rejects it as an unknown field), and Approve/ +// Reject record the reviewer from the admin principal's email. +type SubmissionService interface { + // Create records a pending_review submission. It starts NO build (the whole + // point of the lane — nothing is built until an admin approves). + Create(ctx context.Context, req submit.CreateRequest) (*submit.Submission, error) + // ListBy returns one user's submissions, newest first (the "my uploads" view). + ListBy(ctx context.Context, submittedBy string) ([]submit.Submission, error) + // List returns every submission, newest first (the admin review queue). + List(ctx context.Context) ([]submit.Submission, error) + // Approve is the admin gate: it claims pending_review -> approved (CAS) and the + // winner starts the SAME Trivy-gated build as an admin's direct build. + Approve(ctx context.Context, id, reviewedBy string) (*submit.Submission, error) + // Reject is the admin's other verdict: pending_review -> rejected with a + // required reason; it starts no build. + Reject(ctx context.Context, id, reviewedBy, reason string) (*submit.Submission, error) +} + +// createSubmissionRequest is the POST /me/submissions body. The user +// supplies ONLY a human-friendly label; identity comes from the Access principal +// and the build inputs (image/context refs) are platform-derived, never from the +// body. decodeJSON rejects unknown fields, so a client cannot smuggle a +// submitted_by or a context_ref through this endpoint. +type createSubmissionRequest struct { + DisplayName string `json:"display_name"` +} + +// rejectSubmissionRequest is the POST /submissions/{id}/reject body. A reason is +// required (the submit layer rejects an empty one with 400). +type rejectSubmissionRequest struct { + Reason string `json:"reason"` +} + +// handleCreateSubmission records a new pending_review submission (app-tier). The +// submitter is the authenticated principal's id — never the body — so a user can +// only ever file an upload under their own identity. +func (a *API) handleCreateSubmission(w http.ResponseWriter, r *http.Request) { + if a.Submissions == nil { + writeError(w, r, errSubmissionsUnavailable) + return + } + p := principalFromContext(r.Context()) + var body createSubmissionRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + sub, err := a.Submissions.Create(r.Context(), submit.CreateRequest{ + DisplayName: body.DisplayName, + SubmittedBy: p.UserID, + }) + if err != nil { + writeSubmitError(w, r, err) + return + } + a.audit(r, p.Email, "submission.create", sub.ID) + writeJSON(w, http.StatusCreated, sub) +} + +// handleMySubmissions lists the caller's own submissions (app-tier). It scopes +// strictly to the principal's id; there is no parameter that could widen the +// query to another user's uploads. +func (a *API) handleMySubmissions(w http.ResponseWriter, r *http.Request) { + if a.Submissions == nil { + writeError(w, r, errSubmissionsUnavailable) + return + } + p := principalFromContext(r.Context()) + subs, err := a.Submissions.ListBy(r.Context(), p.UserID) + if err != nil { + writeSubmitError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"submissions": subs}) +} + +// handleListSubmissions is the admin review queue: every submission across all +// users, newest first (admin-tier — it reads other users' uploads, so it gates +// on the admin Zero-Trust path via adminOnly). +func (a *API) handleListSubmissions(w http.ResponseWriter, r *http.Request) { + if a.Submissions == nil { + writeError(w, r, errSubmissionsUnavailable) + return + } + subs, err := a.Submissions.List(r.Context()) + if err != nil { + writeSubmitError(w, r, err) + return + } + writeJSON(w, http.StatusOK, map[string]any{"submissions": subs}) +} + +// handleApproveSubmission is the admin approve gate (admin-tier). The reviewer is +// the admin principal's email; the build inputs are derived inside the submit +// layer, so this handler forwards no client-controlled build parameter. +func (a *API) handleApproveSubmission(w http.ResponseWriter, r *http.Request) { + if a.Submissions == nil { + writeError(w, r, errSubmissionsUnavailable) + return + } + p := principalFromContext(r.Context()) + id := r.PathValue("id") + sub, err := a.Submissions.Approve(r.Context(), id, p.Email) + if err != nil { + writeSubmitError(w, r, err) + return + } + a.audit(r, p.Email, "submission.approve", sub.ID) + writeJSON(w, http.StatusOK, sub) +} + +// handleRejectSubmission records an admin rejection with a required reason +// (admin-tier). It starts no build. +func (a *API) handleRejectSubmission(w http.ResponseWriter, r *http.Request) { + if a.Submissions == nil { + writeError(w, r, errSubmissionsUnavailable) + return + } + p := principalFromContext(r.Context()) + id := r.PathValue("id") + var body rejectSubmissionRequest + if err := decodeJSON(w, r, &body); err != nil { + writeError(w, r, err) + return + } + sub, err := a.Submissions.Reject(r.Context(), id, p.Email, body.Reason) + if err != nil { + writeSubmitError(w, r, err) + return + } + a.audit(r, p.Email, "submission.reject", sub.ID) + writeJSON(w, http.StatusOK, sub) +} + +// errSubmissionsUnavailable is returned when the approval lane is not configured +// on this api instance (a nil Submissions service), so the admin/app boundary is +// still exercised before the subsystem is wired in. +var errSubmissionsUnavailable = newError(http.StatusServiceUnavailable, "submissions_unavailable", + "modpack submission subsystem is not configured") + +// writeSubmitError maps submit-package errors onto HTTP status codes. Only the +// three business sentinels are client-facing: a validation failure is 400, a +// missing submission is 404, an already-reviewed submission is 409. Everything +// else — including a build.ErrInvalid raised by the pre-CAS build.Validate (a +// platform registry/context MISCONFIGURATION, never client input, since every +// build input is platform-derived) and a post-CAS Submit hand-off failure — is a +// server-side fault that collapses to 500 via writeError. The lane deliberately +// does not surface those as 4xx: the client did nothing wrong. +func writeSubmitError(w http.ResponseWriter, r *http.Request, err error) { + switch { + case errors.Is(err, submit.ErrInvalid): + writeError(w, r, newError(http.StatusBadRequest, "bad_request", "%s", err.Error())) + case errors.Is(err, submit.ErrNotFound): + writeError(w, r, newError(http.StatusNotFound, "not_found", "submission not found")) + case errors.Is(err, submit.ErrAlreadyReviewed): + writeError(w, r, newError(http.StatusConflict, "already_reviewed", + "submission has already been reviewed")) + default: + writeError(w, r, err) + } +} + +// Compile-time proof that the production Manager satisfies the API interface. +var _ SubmissionService = (*submit.Manager)(nil) diff --git a/internal/api/submissions_test.go b/internal/api/submissions_test.go new file mode 100644 index 0000000..d5b96f5 --- /dev/null +++ b/internal/api/submissions_test.go @@ -0,0 +1,300 @@ +package api + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "testing" + + "felis.lolicon.best/internal/submit" +) + +// fakeSubmissions is an in-memory SubmissionService for the handler tests. Each +// *Err field injects a canned outcome; the recorders let a test assert exactly +// what the handler forwarded (the point of the owner-scoping checks: the +// submitter and reviewer must come from the principal, never the body). +type fakeSubmissions struct { + created *submit.CreateRequest + createErr error + listedBy string + byResult []submit.Submission + byErr error + listed []submit.Submission + listErr error + approvedID string + approvedBy string + approveErr error + rejectedID string + rejectedBy string + rejectReas string + rejectErr error +} + +func (f *fakeSubmissions) Create(_ context.Context, req submit.CreateRequest) (*submit.Submission, error) { + if f.createErr != nil { + return nil, f.createErr + } + cp := req + f.created = &cp + return &submit.Submission{ID: "sub-1", SubmittedBy: req.SubmittedBy, + DisplayName: req.DisplayName, Status: submit.StatusPendingReview}, nil +} + +func (f *fakeSubmissions) ListBy(_ context.Context, submittedBy string) ([]submit.Submission, error) { + f.listedBy = submittedBy + return f.byResult, f.byErr +} + +func (f *fakeSubmissions) List(_ context.Context) ([]submit.Submission, error) { + return f.listed, f.listErr +} + +func (f *fakeSubmissions) Approve(_ context.Context, id, reviewedBy string) (*submit.Submission, error) { + f.approvedID, f.approvedBy = id, reviewedBy + if f.approveErr != nil { + return nil, f.approveErr + } + return &submit.Submission{ID: id, Status: submit.StatusApproved, ReviewedBy: reviewedBy, BuildID: "bld-1"}, nil +} + +func (f *fakeSubmissions) Reject(_ context.Context, id, reviewedBy, reason string) (*submit.Submission, error) { + f.rejectedID, f.rejectedBy, f.rejectReas = id, reviewedBy, reason + if f.rejectErr != nil { + return nil, f.rejectErr + } + return &submit.Submission{ID: id, Status: submit.StatusRejected, ReviewedBy: reviewedBy, RejectReason: reason}, nil +} + +// appSubAPI wires a submissions service behind an ordinary user principal (the +// app tier — /me/submissions). user@example.net / .test are deliberately not the +// deployment domain. +func appSubAPI(s SubmissionService) *API { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.Submissions = s + api.External = staticExternal{p: &Principal{UserID: "user-7", Email: "user@example.net", Role: "user"}} + return api +} + +// adminSubAPI wires a submissions service behind an admin principal (the admin +// tier — /submissions approve/reject/list). +func adminSubAPI(s SubmissionService) *API { + api := newTestAPI(newFakeRepo(), newFakeCluster()) + api.Submissions = s + api.External = staticExternal{p: &Principal{UserID: "admin-1", Email: "admin@example.net", + Role: "admin", ViaAdminAccess: true}} + return api +} + +// The submitter is the principal, never the body: a valid create stamps the +// authenticated user's id onto the submission. +func TestCreateSubmissionStampsPrincipal(t *testing.T) { + fs := &fakeSubmissions{} + api := appSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/me/submissions", `{"display_name":"My Pack"}`, nil) + if w.Code != http.StatusCreated { + t.Fatalf("code = %d, want 201 (%s)", w.Code, w.Body.String()) + } + if fs.created == nil { + t.Fatal("Create was not called") + } + if fs.created.SubmittedBy != "user-7" { + t.Errorf("submitted_by = %q, want the principal id user-7", fs.created.SubmittedBy) + } + if fs.created.DisplayName != "My Pack" { + t.Errorf("display_name = %q", fs.created.DisplayName) + } +} + +// A client cannot smuggle a submitted_by through the body: decodeJSON rejects the +// unknown field with 400 and the handler never reaches Create. +func TestCreateSubmissionRejectsBodySubmittedBy(t *testing.T) { + fs := &fakeSubmissions{} + api := appSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/me/submissions", + `{"display_name":"X","submitted_by":"someone-else"}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400 (%s)", w.Code, w.Body.String()) + } + if fs.created != nil { + t.Errorf("Create must not be called for an unknown-field body; got %+v", fs.created) + } +} + +// A validation failure from the submit layer surfaces as 400 bad_request. +func TestCreateSubmissionValidationIs400(t *testing.T) { + fs := &fakeSubmissions{createErr: fmt.Errorf("%w: display name is required", submit.ErrInvalid)} + api := appSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/me/submissions", `{"display_name":""}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400 (%s)", w.Code, w.Body.String()) + } + if got := decodeErr(t, w); got != "bad_request" { + t.Errorf("error code = %q, want bad_request", got) + } +} + +// The "my uploads" list scopes strictly to the principal's id — there is no +// parameter that could widen it to another user's submissions. +func TestMySubmissionsScopesToPrincipal(t *testing.T) { + fs := &fakeSubmissions{byResult: []submit.Submission{ + {ID: "sub-1", SubmittedBy: "user-7", DisplayName: "p", Status: submit.StatusPendingReview}, + }} + api := appSubAPI(fs) + w := do(api.ExternalHandler(), "GET", "/api/v1/me/submissions", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if fs.listedBy != "user-7" { + t.Errorf("ListBy scoped to %q, want the principal id user-7", fs.listedBy) + } + var got map[string][]submit.Submission + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v", err) + } + if len(got["submissions"]) != 1 { + t.Fatalf("submissions = %d, want 1", len(got["submissions"])) + } +} + +// Every /submissions route is admin-tier: a plain user is rejected before the +// handler runs. +func TestSubmissionAdminRoutesAreAdminOnly(t *testing.T) { + cases := []struct { + method, target, body string + }{ + {"GET", "/api/v1/submissions", ""}, + {"POST", "/api/v1/submissions/sub-1/approve", ""}, + {"POST", "/api/v1/submissions/sub-1/reject", `{"reason":"no"}`}, + } + for _, c := range cases { + api := adminSubAPI(&fakeSubmissions{}) + // downgrade to a plain user + api.External = staticExternal{p: &Principal{UserID: "u", Role: "user"}} + w := do(api.ExternalHandler(), c.method, c.target, c.body, nil) + if w.Code != http.StatusForbidden { + t.Errorf("%s %s: code = %d, want 403", c.method, c.target, w.Code) + } + } +} + +func TestListSubmissionsAdmin(t *testing.T) { + fs := &fakeSubmissions{listed: []submit.Submission{ + {ID: "sub-1", SubmittedBy: "user-7", Status: submit.StatusPendingReview}, + {ID: "sub-2", SubmittedBy: "user-9", Status: submit.StatusApproved}, + }} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "GET", "/api/v1/submissions", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + var got map[string][]submit.Submission + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatalf("body not JSON: %v", err) + } + if len(got["submissions"]) != 2 { + t.Fatalf("submissions = %d, want 2", len(got["submissions"])) + } +} + +// The reviewer is the admin principal's email, never client input. +func TestApproveSubmission(t *testing.T) { + fs := &fakeSubmissions{} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/submissions/sub-9/approve", "", nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if fs.approvedID != "sub-9" { + t.Errorf("approved id = %q, want sub-9", fs.approvedID) + } + if fs.approvedBy != "admin@example.net" { + t.Errorf("reviewer = %q, want the admin email", fs.approvedBy) + } +} + +func TestApproveSubmissionAlreadyReviewedIs409(t *testing.T) { + fs := &fakeSubmissions{approveErr: submit.ErrAlreadyReviewed} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/submissions/sub-9/approve", "", nil) + if w.Code != http.StatusConflict { + t.Fatalf("code = %d, want 409 (%s)", w.Code, w.Body.String()) + } + if got := decodeErr(t, w); got != "already_reviewed" { + t.Errorf("error code = %q, want already_reviewed", got) + } +} + +func TestApproveSubmissionNotFoundIs404(t *testing.T) { + fs := &fakeSubmissions{approveErr: submit.ErrNotFound} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/submissions/missing/approve", "", nil) + if w.Code != http.StatusNotFound { + t.Fatalf("code = %d, want 404 (%s)", w.Code, w.Body.String()) + } +} + +// A build hand-off failure post-approval is a server-side fault (the client did +// nothing wrong), so it collapses to 500 — never a 4xx. +func TestApproveBuildHandoffFailureIs500(t *testing.T) { + fs := &fakeSubmissions{approveErr: fmt.Errorf("submit: approved but build hand-off failed: %w", + errors.New("cluster unreachable"))} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/submissions/sub-9/approve", "", nil) + if w.Code != http.StatusInternalServerError { + t.Fatalf("code = %d, want 500 (%s)", w.Code, w.Body.String()) + } + if got := decodeErr(t, w); got != "internal" { + t.Errorf("error code = %q, want internal", got) + } +} + +func TestRejectSubmission(t *testing.T) { + fs := &fakeSubmissions{} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/submissions/sub-3/reject", + `{"reason":"contains malware"}`, nil) + if w.Code != http.StatusOK { + t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String()) + } + if fs.rejectedID != "sub-3" { + t.Errorf("rejected id = %q, want sub-3", fs.rejectedID) + } + if fs.rejectedBy != "admin@example.net" { + t.Errorf("reviewer = %q, want the admin email", fs.rejectedBy) + } + if fs.rejectReas != "contains malware" { + t.Errorf("reason = %q", fs.rejectReas) + } +} + +func TestRejectSubmissionValidationIs400(t *testing.T) { + fs := &fakeSubmissions{rejectErr: fmt.Errorf("%w: a reject reason is required", submit.ErrInvalid)} + api := adminSubAPI(fs) + w := do(api.ExternalHandler(), "POST", "/api/v1/submissions/sub-3/reject", `{"reason":""}`, nil) + if w.Code != http.StatusBadRequest { + t.Fatalf("code = %d, want 400 (%s)", w.Code, w.Body.String()) + } + if got := decodeErr(t, w); got != "bad_request" { + t.Errorf("error code = %q, want bad_request", got) + } +} + +// When no Submissions service is configured, the routes report 503 — the app +// route directly, and the admin route only AFTER the admin gate, so the boundary +// is still enforced first. +func TestSubmissionRoutesWithoutServiceAre503(t *testing.T) { + // app tier + app := appSubAPI(nil) + app.Submissions = nil + if w := do(app.ExternalHandler(), "GET", "/api/v1/me/submissions", "", nil); w.Code != http.StatusServiceUnavailable { + t.Fatalf("app route: code = %d, want 503", w.Code) + } + // admin tier + adm := adminSubAPI(nil) + adm.Submissions = nil + if w := do(adm.ExternalHandler(), "GET", "/api/v1/submissions", "", nil); w.Code != http.StatusServiceUnavailable { + t.Fatalf("admin route: code = %d, want 503", w.Code) + } +} diff --git a/internal/api/util.go b/internal/api/util.go new file mode 100644 index 0000000..6f57b5e --- /dev/null +++ b/internal/api/util.go @@ -0,0 +1,23 @@ +package api + +import ( + "encoding/json" + "net/http" +) + +// maxBodyBytes caps request bodies; the API only accepts small JSON documents. +const maxBodyBytes = 1 << 20 // 1 MiB + +// decodeJSON strictly decodes a small request body into v, rejecting unknown +// fields and trailing data so malformed callers fail fast with 400. +func decodeJSON(w http.ResponseWriter, r *http.Request, v any) error { + dec := json.NewDecoder(http.MaxBytesReader(w, r.Body, maxBodyBytes)) + dec.DisallowUnknownFields() + if err := dec.Decode(v); err != nil { + return newError(http.StatusBadRequest, "bad_request", "invalid request body: %v", err) + } + if dec.More() { + return newError(http.StatusBadRequest, "bad_request", "unexpected trailing data in body") + } + return nil +}