Files
Felis/internal/api/handlers_user.go
T

981 lines
37 KiB
Go

package api
import (
"context"
"errors"
"fmt"
"net/http"
"strings"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/naming"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/resource"
)
// handleWake is the single lever (spec §9.1): it authorizes per autostartPolicy,
// applies the cooldown, and flips the CRD desiredState to Running. It does not
// transfer the player — the web flow shows status and a connect hint (spec §9.2).
func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil && !errors.Is(err, ErrNotFound) {
writeError(w, r, err)
return
}
if err := a.authorizeWake(r.Context(), p, info, rec); err != nil {
writeError(w, r, err)
return
}
if !a.limiter().allowed(name, a.WakeCooldown) {
writeError(w, r, newError(http.StatusTooManyRequests, "cooldown", "wake is cooling down, retry shortly"))
return
}
// Global running-server cap (spec §9.1). Distinct from the per-server cooldown:
// 503 at_capacity means the cluster is full, not that this server is throttled.
ok, err := a.withinRunningCap(r.Context(), info)
if err != nil {
writeError(w, r, err)
return
}
if !ok {
writeError(w, r, newError(http.StatusServiceUnavailable, "at_capacity",
"the cluster is at its running-server cap; retry once a server stops"))
return
}
// Refused with 409 maintenance_in_progress while a restore, backup or file
// write holds the world volume: starting on a half-written world corrupts it.
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
a.writeLookupError(w, r, err)
return
}
// The wake actually flipped, so consume the per-server cooldown only now: a 503
// at_capacity or the SetDesiredState failure above must not burn it (a player
// held at capacity should retry the instant a slot frees, not wait out a
// cooldown their refused wake never earned).
a.limiter().record(name)
a.audit(r, "wake", name)
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
}
// handleStop flips desiredState to Stopped. Only the owner or an admin may stop a
// server (spec §14: operating someone else's server is admin-tier).
func (a *API) handleStop(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
writeError(w, r, errForbidden)
return
}
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredStopped); err != nil {
writeError(w, r, err)
return
}
a.audit(r, "stop", name)
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Stopped"})
}
// handleClaim is the atomic claim transaction (spec §9.3): require a verified
// account link, enforce the quota gate, then UPDATE ... WHERE owner_id IS NULL.
// A lost race (0 rows) is 409.
func (a *API) handleClaim(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
// ① verified account link
linked, err := a.Repo.IsLinked(r.Context(), p.UserID)
if err != nil {
writeError(w, r, err)
return
}
if !linked {
writeError(w, r, newError(http.StatusPreconditionFailed, "not_linked",
"link your Minecraft account before claiming (see /api/v1/account/link/start)"))
return
}
// ② quota gate, evaluated before the ownership write. All four dimensions
// (servers, CPU, memory, storage) are checked against the user's quota caps
// using the PG-resident resource cache (spec §9.3, §22). The server being
// claimed has owner_id=NULL so it is not yet in the per-owner aggregate.
res, err := a.Repo.ServerResources(r.Context(), name)
if err != nil {
writeError(w, r, err)
return
}
ok, err := a.Repo.QuotaCheck(r.Context(), p.UserID, "", res)
if err != nil {
writeError(w, r, err)
return
}
if !ok {
writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted"))
return
}
// ③ atomic claim
claimed, err := a.Repo.ClaimServer(r.Context(), name, p.UserID)
if err != nil {
// The atomic gate re-checks quota under the per-user lock (audit #4): a
// concurrent claim that spent the last slot surfaces here, with the same
// 403 the pre-check gives sequentially.
if errors.Is(err, ErrQuotaExceeded) {
writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted"))
return
}
a.writeLookupError(w, r, err)
return
}
if !claimed {
writeError(w, r, newError(http.StatusConflict, "already_claimed", "server is already claimed"))
return
}
// ④ audit. Allowlist population happens on first successful join (spec §9.4).
a.audit(r, "claim", name)
writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true})
}
// handleStatus returns the CRD status view (spec §7 GET /servers/{name}/status).
func (a *API) handleStatus(w http.ResponseWriter, r *http.Request) {
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
// The internal face (no principal) is the platform itself. On the external
// face the full record (image, resources, endpoint) is the owner's and the
// staff's; anyone else signed in sees what the game's own server list shows.
if p := principalFromContext(r.Context()); p != nil {
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil && !errors.Is(err, ErrNotFound) {
writeError(w, r, err)
return
}
if !a.isOwnerOrAdmin(p, rec) {
info = publicServerInfo(info)
}
}
writeJSON(w, http.StatusOK, info)
}
// publicServerInfo keeps the fields any signed-in caller may see of a server
// that is not theirs.
func publicServerInfo(s *ServerInfo) *ServerInfo {
return &ServerInfo{
Name: s.Name,
Subdomain: s.Subdomain,
DisplayName: s.DisplayName,
Phase: s.Phase,
Ready: s.Ready,
PlayersOnline: s.PlayersOnline,
PlayersMax: s.PlayersMax,
}
}
// handleMe returns the calling principal's own identity (spec §14 tiering). The
// panel reads it once at boot to decide which navigation surfaces to render:
// the User-Side for everyone, the Admin/SysAdmin sides only when is_admin. This
// is UX truth, NOT a security control — every admin route is independently gated
// by adminOnly + Principal.IsAdmin() server-side, so hiding a nav item never
// widens access. is_admin is computed here as IsAdmin() (Role=="admin" AND the
// admin Access path), so the client never re-derives the graded-ZT rule. App-tier:
// a principal reads only its OWN identity — the response is sourced entirely from
// the verified token, no lookup escapes it.
func (a *API) handleMe(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
emailVerified := false
if u, err := a.Repo.UserByID(r.Context(), p.UserID); err == nil {
emailVerified = u.EmailVerified
}
writeJSON(w, http.StatusOK, map[string]any{
"user_id": p.UserID,
"email": p.Email,
"role": p.Role,
"is_admin": p.IsAdmin(),
"is_owner": p.IsOwner(),
"email_verified": emailVerified,
})
}
// handleMyServers lists what the caller owns or may claim (spec §7 GET /me/servers).
func (a *API) handleMyServers(w http.ResponseWriter, r *http.Request) {
p := principalFromContext(r.Context())
servers, err := a.Repo.MyServers(r.Context(), p.UserID)
if err != nil {
writeError(w, r, err)
return
}
// The live fields are presentational and best-effort, mirroring handleFleet's
// owner join: the list exists for ownership/claim state, so a cluster hiccup
// must degrade to 0/0 counts and the cached phase, never 500 the whole list.
// The CRD status is the only source of live state (spec §1) — Postgres never
// stores counts. Owner detail (desired state, autostart policy, whether the
// count is readable) joins only onto rows the caller owns; the panel needs
// playerCountUnknown there to ask before a stop that may drop players.
if infos, err := a.Cluster.ListServers(r.Context()); err == nil {
byName := make(map[string]ServerInfo, len(infos))
for _, s := range infos {
byName[s.Name] = s
}
for i := range servers {
info, ok := byName[servers[i].Name]
if !ok {
continue
}
v := &servers[i]
v.PlayersOnline = info.PlayersOnline
v.PlayersMax = info.PlayersMax
v.DisplayName = info.DisplayName
if info.Phase != "" {
v.Phase = info.Phase
}
if v.Owned {
v.DesiredState = info.DesiredState
v.AutostartPolicy = info.AutostartPolicy
v.PlayerCountUnknown = info.PlayerCountUnknown
}
}
}
writeJSON(w, http.StatusOK, map[string]any{"servers": servers})
}
// handleFleet is the SysAdmin cockpit's fleet-wide read: the lifecycle view of
// EVERY MinecraftServer, from CRD + status. It is the admin-tier counterpart of
// the app-tier handleMyServers — where /me/servers scopes to the caller, this
// returns the whole fleet, so it gates on the admin Zero-Trust path via adminOnly.
//
// This is a frontend-cockpit-driven extension (the SysAdmin FleetTable in
// panel/DESIGN-WEB-3SIDES.md), NOT a spec §7 route: §7 lists only the internal
// velocity pull (GET /servers, service-tier) and the app-tier GET /me/servers,
// neither of which is an external admin read. It reuses Cluster.ListServers (the
// same CRD-truth source as the velocity pull, §1) but is a DISTINCT handler so
// each route's provenance and tier stay honest, and so the two never share a
// {method, path} key — the OpenAPI parity test forbids one path carrying both the
// service and admin tiers across faces. Lifecycle is read from the CRD (§1); the
// one business field the cockpit needs — the owner — is joined READ-ONLY from
// Postgres at request time (§6 business authority) purely for display. This keeps
// §1 honest: owner is never written back to the CRD and the CRD is never treated
// as its source; the two stores keep their split, the read just renders both.
func (a *API) handleFleet(w http.ResponseWriter, r *http.Request) {
servers, err := a.Cluster.ListServers(r.Context())
if err != nil {
writeError(w, r, err)
return
}
// Ownership is best-effort. The cockpit exists for the lifecycle view, so a
// Postgres hiccup must never 500 the whole fleet; the rows say the owner is
// unknown instead, and none offers a claim that may already be taken.
p := principalFromContext(r.Context())
owners, err := a.Repo.ServerOwners(r.Context())
unknown := err != nil
views := make([]fleetServerView, len(servers))
for i, s := range servers {
o, known := owners[s.Name]
system := naming.IsSystemServer(s.Name)
views[i] = fleetServerView{ServerInfo: s, Owner: o.Owner, System: system,
Owned: o.OwnerID != "" && o.OwnerID == p.UserID,
Claimable: known && o.OwnerID == "" && !system,
OwnerUnknown: unknown}
}
writeJSON(w, http.StatusOK, map[string]any{"servers": views})
}
// fleetServerView is one row of the SysAdmin cockpit's fleet read: the CRD
// lifecycle view (ServerInfo, §1 authority) with the owner's display identity
// joined alongside. The embed keeps every lifecycle field flat in the JSON so the
// shape is a strict superset of ServerInfo.
type fleetServerView struct {
ServerInfo
// Owner is the claiming user's display identity (email, or username when the
// address is absent), or "" when the server is unclaimed or the owner lookup
// failed (OwnerUnknown tells the two apart).
Owner string `json:"owner,omitempty"`
// Owned is true when the caller claimed this server, decided by account id so
// an owner without an email is still recognized.
Owned bool `json:"owned"`
// Claimable is true for a live, unclaimed, non-system server: the same rule
// ClaimServer enforces. It is false whenever ownership is unknown.
Claimable bool `json:"claimable"`
// OwnerUnknown is true when the best-effort owner lookup failed, so an empty
// Owner says nothing about whether the server is claimed.
OwnerUnknown bool `json:"ownerUnknown,omitempty"`
// System marks a platform-provisioned system service (the login gate and the
// lobby, naming.IsSystemServer). Their names are reserved, so every per-server
// API route rejects them — the cockpit must render them read-only rather than
// offer claim/wake/stop/console actions that would answer 400.
System bool `json:"system,omitempty"`
}
// createServerRequest is the structured §15 create-server form. This is the
// ONLY way to create a server from the Web: every field is a typed, validated
// value and decodeJSON rejects unknown fields, so a caller can never smuggle
// free-form YAML or raw CRD fields through this endpoint.
type createServerRequest struct {
Name string `json:"name"`
Subdomain string `json:"subdomain"`
DisplayName string `json:"displayName,omitempty"`
Image string `json:"image"`
Memory string `json:"memory"`
Storage string `json:"storage"`
AutostartPolicy string `json:"autostartPolicy,omitempty"`
Resources *resourceRequest `json:"resources,omitempty"`
}
// resourceRequest is the optional override block. The required top-level memory
// already sets the pod memory limit+request (the §22 ceiling); these fields let
// an admin widen/narrow the cgroup envelope. Each is a Kubernetes quantity
// string ("500m", "2", "1Gi").
type resourceRequest struct {
CPU string `json:"cpu,omitempty"`
CPURequest string `json:"cpuRequest,omitempty"`
Memory string `json:"memory,omitempty"`
MemoryRequest string `json:"memoryRequest,omitempty"`
}
// handleCreateServer (spec §15) is the admin-tier structured create flow:
// validate the form, admit the image against the whitelist, seed the business
// rows, then create the MinecraftServer CRD cold (DesiredState=Stopped) and
// unowned. There is no free-YAML path — the request is a typed form.
func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
// The image whitelist lives in the build subsystem; with no Builder there is
// no admission source, so create cannot run safely.
if a.Builder == nil {
writeError(w, r, errBuildUnavailable)
return
}
var body createServerRequest
if err := decodeJSON(w, r, &body); err != nil {
writeError(w, r, err)
return
}
// Server name and subdomain both obey the §22 portability rule and the
// reservation list.
if err := naming.ValidateServerName(body.Name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
if err := naming.ValidateServerName(body.Subdomain); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_subdomain", "invalid subdomain: %v", err))
return
}
policy, err := parseAutostartPolicy(body.AutostartPolicy)
if err != nil {
writeError(w, r, err)
return
}
// Memory is required: it is both the JVM heap hint and the default pod memory
// limit+request. resolveResources fails closed if the §22 ceiling would be
// zero, so a CRD is never written without a concrete memory limit.
javaMemory, resources, err := resolveResources(body.Memory, body.Resources)
if err != nil {
writeError(w, r, err)
return
}
storage, err := parseStorageSize(body.Storage)
if err != nil {
writeError(w, r, err)
return
}
// Image admission is data-driven (the whitelist), never a free image string.
if strings.TrimSpace(body.Image) == "" {
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "image is required"))
return
}
admitted, err := a.Builder.ImageAdmitted(r.Context(), body.Image)
if err != nil {
writeError(w, r, err)
return
}
if !admitted {
writeError(w, r, newError(http.StatusBadRequest, "image_not_whitelisted",
"image %q is not on the whitelist", body.Image))
return
}
// The spec keeps the digest the tag names now, not the tag: the world is
// created on this build and stays on it until an admin changes the image.
image, err := a.pinImage(r.Context(), body.Image)
if err != nil {
writeError(w, r, err)
return
}
// Quota is intentionally NOT enforced here. §15 creates an UNOWNED server
// (owner_id NULL); the per-user quota is charged at claim time (spec §9.3 /
// §22). The create path is quota-free by design, not by oversight.
// Reject a duplicate subdomain before any write. The CRD list is the
// lifecycle source; SeedServer re-checks the PG alias atomically below.
switch _, err := a.Cluster.GetBySubdomain(r.Context(), body.Subdomain); {
case err == nil:
writeError(w, r, newError(http.StatusConflict, "subdomain_taken",
"subdomain %q is already in use", body.Subdomain))
return
case !errors.Is(err, ErrNotFound):
writeError(w, r, err)
return
}
// Reject a duplicate server NAME before any write too. Without this, a
// dup-name create (fresh subdomain) would reach SeedServer, which would bind
// the new alias onto the PRE-EXISTING server and commit it — then CreateServer
// fails 409 but the stray alias persists. Checking the CRD here keeps the
// failed create side-effect-free. The narrow concurrent same-name race stays
// inside the documented non-transactional tradeoff, backstopped by
// CreateServer's AlreadyExists→409 below.
switch _, err := a.Cluster.GetServer(r.Context(), body.Name); {
case err == nil:
writeError(w, r, newError(http.StatusConflict, "already_exists",
"a server named %q already exists", body.Name))
return
case !errors.Is(err, ErrNotFound):
writeError(w, r, err)
return
}
// Seed the business rows FIRST (servers + alias). ClaimServer needs the row,
// so a CRD-only server would be unclaimable. PG-first means a later CRD
// failure leaves a claimable ghost row — acceptable, not transactional.
// The resource cache (cpuMilli, memoryMB, storageMB) is seeded alongside so
// QuotaCheck can aggregate per-owner usage without cross-system CRD reads.
cpuMilli := quantityToMilli(resources.Limits[corev1.ResourceCPU])
memMB := quantityToMB(resources.Limits[corev1.ResourceMemory])
storQ, _ := resource.ParseQuantity(storage)
storMB := quantityToMB(storQ)
if err := a.Repo.SeedServer(r.Context(), body.Name, body.Subdomain, cpuMilli, memMB, storMB); err != nil {
if errors.Is(err, ErrConflict) {
writeError(w, r, newError(http.StatusConflict, "subdomain_taken",
"subdomain %q is already in use", body.Subdomain))
return
}
writeError(w, r, err)
return
}
in := CreateServerInput{
Name: body.Name,
Subdomain: body.Subdomain,
DisplayName: body.DisplayName,
Image: image,
JavaMemory: javaMemory,
StorageSize: storage,
AutostartPolicy: policy,
Resources: resources,
}
if err := a.Cluster.CreateServer(r.Context(), in); err != nil {
if errors.Is(err, ErrConflict) {
writeError(w, r, newError(http.StatusConflict, "already_exists",
"a server named %q already exists", body.Name))
return
}
writeError(w, r, err)
return
}
a.audit(r, "server.create", body.Name)
writeJSON(w, http.StatusCreated, map[string]any{
"name": body.Name,
"subdomain": body.Subdomain,
"desiredState": string(v1alpha1.DesiredStopped),
})
}
// parseAutostartPolicy maps the form value to a CRD policy. An empty value
// defaults to the safest policy (ownerOnly); any other unknown value is a 400.
func parseAutostartPolicy(s string) (v1alpha1.AutostartPolicy, error) {
switch s {
case "":
return v1alpha1.AutostartOwnerOnly, nil
case string(v1alpha1.AutostartOwnerOnly):
return v1alpha1.AutostartOwnerOnly, nil
case string(v1alpha1.AutostartPublic):
return v1alpha1.AutostartPublic, nil
case string(v1alpha1.AutostartAllowlist):
return v1alpha1.AutostartAllowlist, nil
default:
return "", newError(http.StatusBadRequest, "bad_request",
"invalid autostartPolicy %q (want ownerOnly, public, or allowlist)", s)
}
}
// resolveResources turns the required memory string and the optional override
// block into the pod resource requirements. The top-level memory seeds both the
// memory limit and request; the override block may widen CPU and memory. It
// fails closed: the returned limit's memory is guaranteed non-zero so the §22
// ceiling is never absent from the CRD. The first return value is the derived
// JVM max-heap string (JavaMemory / -Xmx), computed from the FINAL memory limit
// — NOT the raw form value — so an overridden ceiling is honored and the heap
// stays below the cgroup limit (see deriveJavaHeap).
func resolveResources(memory string, rr *resourceRequest) (string, corev1.ResourceRequirements, error) {
memQ, err := parsePositiveQuantity(memory, "memory")
if err != nil {
return "", corev1.ResourceRequirements{}, err
}
limits := corev1.ResourceList{corev1.ResourceMemory: memQ}
requests := corev1.ResourceList{corev1.ResourceMemory: memQ}
if rr != nil {
if rr.Memory != "" {
q, err := parsePositiveQuantity(rr.Memory, "resources.memory")
if err != nil {
return "", corev1.ResourceRequirements{}, err
}
limits[corev1.ResourceMemory] = q
}
if rr.MemoryRequest != "" {
q, err := parsePositiveQuantity(rr.MemoryRequest, "resources.memoryRequest")
if err != nil {
return "", corev1.ResourceRequirements{}, err
}
requests[corev1.ResourceMemory] = q
}
if rr.CPU != "" {
q, err := parsePositiveQuantity(rr.CPU, "resources.cpu")
if err != nil {
return "", corev1.ResourceRequirements{}, err
}
limits[corev1.ResourceCPU] = q
}
if rr.CPURequest != "" {
q, err := parsePositiveQuantity(rr.CPURequest, "resources.cpuRequest")
if err != nil {
return "", corev1.ResourceRequirements{}, err
}
requests[corev1.ResourceCPU] = q
}
}
// A request that exceeds its limit is rejected by Kubernetes; fail fast here
// with a clear 400 instead of letting the CRD write bounce.
if memReq, memLim := requests[corev1.ResourceMemory], limits[corev1.ResourceMemory]; memReq.Cmp(memLim) > 0 {
return "", corev1.ResourceRequirements{}, newError(http.StatusBadRequest, "bad_request",
"memory request %s exceeds limit %s", memReq.String(), memLim.String())
}
if cpuReq, hasReq := requests[corev1.ResourceCPU]; hasReq {
if cpuLim, hasLim := limits[corev1.ResourceCPU]; hasLim && cpuReq.Cmp(cpuLim) > 0 {
return "", corev1.ResourceRequirements{}, newError(http.StatusBadRequest, "bad_request",
"cpu request %s exceeds limit %s", cpuReq.String(), cpuLim.String())
}
}
// §22 fail-closed: never hand the operator a CRD without a concrete memory
// ceiling. This cannot trigger given the positive memQ above, but the assert
// guarantees the invariant survives future edits.
memLim, ok := limits[corev1.ResourceMemory]
if !ok || memLim.IsZero() {
return "", corev1.ResourceRequirements{}, newError(http.StatusInternalServerError, "internal",
"refusing to create a server without a memory ceiling")
}
return deriveJavaHeap(memLim), corev1.ResourceRequirements{Limits: limits, Requests: requests}, nil
}
// deriveJavaHeap converts the pod memory ceiling into a JVM max-heap string
// (JavaMemory → JAVA_MEMORY → -Xmx). Two reasons the raw K8s quantity cannot be
// forwarded as-is:
//
// - Format: the JVM's -Xmx accepts k/m/g (1024-based) suffixes, NOT the
// Kubernetes "Ki/Mi/Gi" forms. "-Xmx2Gi" fails to start the JVM, so we emit
// a plain "<N>M" value, which both -Xmx and the container entrypoint accept.
// - Headroom: metaspace, thread stacks, Netty direct buffers and GC structures
// live OUTSIDE the heap. Setting -Xmx to the full cgroup limit guarantees an
// eventual OOMKill, so we reserve off-heap room (the larger of 512Mi or 25%,
// capped at half the limit) and size the heap to what remains.
//
// This is a sane default the operator/runtime may later refine; it is purely a
// derivation of the §22 ceiling and never exceeds it.
func deriveJavaHeap(limit resource.Quantity) string {
const mib = int64(1024 * 1024)
bytes := limit.Value()
reserve := bytes / 4
if floor := 512 * mib; reserve < floor {
reserve = floor
}
if half := bytes / 2; reserve > half {
reserve = half
}
heapMiB := (bytes - reserve) / mib
if heapMiB < 1 {
heapMiB = 1
}
return fmt.Sprintf("%dM", heapMiB)
}
// parseStorageSize validates the required storage size and returns the
// canonical quantity string for the PVC (spec §15).
func parseStorageSize(s string) (string, error) {
q, err := parsePositiveQuantity(s, "storage")
if err != nil {
return "", err
}
return q.String(), nil
}
// parsePositiveQuantity parses a Kubernetes quantity string and rejects any
// non-positive value with a 400 naming the offending field.
func parsePositiveQuantity(s, field string) (resource.Quantity, error) {
q, err := resource.ParseQuantity(s)
if err != nil {
return resource.Quantity{}, newError(http.StatusBadRequest, "bad_request",
"invalid %s quantity %q: %v", field, s, err)
}
if q.Sign() <= 0 {
return resource.Quantity{}, newError(http.StatusBadRequest, "bad_request",
"%s must be a positive quantity", field)
}
return q, nil
}
// patchServerRequest is the structured §7 PATCH /servers/{name} form. Like the
// §15 create form it is a CLOSED set of typed fields (decodeJSON rejects unknown
// fields), so an admin can never smuggle raw CRD/YAML knobs through a patch.
// Every field is a pointer: a nil pointer means "absent — leave unchanged",
// which a plain zero value could not distinguish from "set to empty". It mutates
// only CRD-authoritative spec fields (spec §22); it deliberately has no field for
// the dual-write routing identity (name is the immutable object key; subdomain
// would desync the Postgres alias) nor for the world PVC size (see below).
type patchServerRequest struct {
DisplayName *string `json:"displayName,omitempty"`
AutostartPolicy *string `json:"autostartPolicy,omitempty"`
Image *string `json:"image,omitempty"`
// ConfirmImageChange acknowledges that a new image opens the world with
// whatever Minecraft version it carries. Chunks a newer version has upgraded
// cannot be read by the older one again, so without it an image change that
// would actually move the server is refused (image_change_unconfirmed).
ConfirmImageChange bool `json:"confirmImageChange,omitempty"`
Memory *string `json:"memory,omitempty"`
Resources *resourceRequest `json:"resources,omitempty"`
// IdleStopSeconds sets idle auto-stop: 0 turns it off, otherwise the server
// stops after that many seconds with nobody online (60 to 86400).
IdleStopSeconds *int32 `json:"idleStopSeconds,omitempty"`
// Storage is recognized only so the endpoint can reject it with a precise
// reason rather than an opaque "unknown field": a StatefulSet's PVC capacity
// is immutable except for storage-class-gated expansion, which this build does
// not orchestrate. Accepting it would write a CRD change the operator cannot
// honor, so it is refused (storage_immutable) instead of silently dropped.
Storage *string `json:"storage,omitempty"`
}
// The idle auto-stop range an admin may pick through PATCH /servers/{name}.
const (
minIdleStopSeconds = 60
maxIdleStopSeconds = 86400
)
// handlePatchServer (spec §7 PATCH /servers/{name}) is the admin-tier spec
// mutation: it validates the structured form, re-admits any new image against the
// whitelist, re-derives the §22 memory ceiling, and applies a merge patch to the
// MinecraftServer CRD. Only CRD-authoritative fields move; the business layer
// (Postgres) is untouched, so the two never desync (spec §22). The admin gate is
// the adminOnly wrapper in routing — every caller here is already an admin.
func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
name := r.PathValue("name")
if err := naming.ValidateServerName(name); err != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
return
}
var body patchServerRequest
if err := decodeJSON(w, r, &body); err != nil {
writeError(w, r, err)
return
}
// An empty patch is a client mistake, not a no-op success.
if body.DisplayName == nil && body.AutostartPolicy == nil && body.Image == nil &&
body.Memory == nil && body.Resources == nil && body.Storage == nil && body.IdleStopSeconds == nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
"patch must set at least one field"))
return
}
if body.Storage != nil {
writeError(w, r, newError(http.StatusBadRequest, "storage_immutable",
"storage size cannot be changed through this endpoint (PVC capacity is immutable)"))
return
}
// Build the resolved patch field-by-field, validating each present field with
// the SAME helpers the create form uses. `changed` records what actually moves
// so the response and audit name the real mutation.
var patch ServerSpecPatch
// Non-nil so a patch that moves nothing (re-picking the image a server is
// already pinned to) still answers "patched": [], as the API documents.
changed := []string{}
// imageFrom is the image a confirmed image change replaced, for the audit row.
var imageFrom string
if body.DisplayName != nil {
patch.DisplayName = body.DisplayName
changed = append(changed, "displayName")
}
if body.AutostartPolicy != nil {
// Unlike create, an explicit empty policy is rejected rather than defaulted:
// a patch states an intent, so "" is ambiguous, not "the safe default".
if *body.AutostartPolicy == "" {
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
"autostartPolicy cannot be empty"))
return
}
policy, err := parseAutostartPolicy(*body.AutostartPolicy)
if err != nil {
writeError(w, r, err)
return
}
patch.AutostartPolicy = &policy
changed = append(changed, "autostartPolicy")
}
if body.IdleStopSeconds != nil {
// A minute is the floor: below it a player who drops for a reconnect
// finds the server stopping under them. A day is the ceiling; longer is
// what "off" is for.
if s := *body.IdleStopSeconds; s != 0 && (s < minIdleStopSeconds || s > maxIdleStopSeconds) {
writeError(w, r, newError(http.StatusBadRequest, "bad_idle_stop",
"idleStopSeconds must be 0 (off) or between %d and %d", minIdleStopSeconds, maxIdleStopSeconds))
return
}
patch.IdleStopSeconds = body.IdleStopSeconds
changed = append(changed, "idleStopSeconds")
}
if body.Image != nil {
// A new image must be re-admitted against the whitelist, exactly as create
// does — admission is the only source of a legal image. With no Builder
// there is no whitelist to check against, so the change cannot run safely.
if a.Builder == nil {
writeError(w, r, errBuildUnavailable)
return
}
if strings.TrimSpace(*body.Image) == "" {
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "image cannot be empty"))
return
}
admitted, err := a.Builder.ImageAdmitted(r.Context(), *body.Image)
if err != nil {
writeError(w, r, err)
return
}
if !admitted {
writeError(w, r, newError(http.StatusBadRequest, "image_not_whitelisted",
"image %q is not on the whitelist", *body.Image))
return
}
image, err := a.pinImage(r.Context(), *body.Image)
if err != nil {
writeError(w, r, err)
return
}
info, err := a.Cluster.GetServer(r.Context(), name)
if err != nil {
a.writeLookupError(w, r, err)
return
}
// Re-picking the tag a server was created from resolves to that tag's
// newest build, which is as much a version move as picking another image.
// Only a pin that lands on exactly the current image is no change at all.
if image != info.Image {
if !body.ConfirmImageChange {
writeError(w, r, newError(http.StatusConflict, "image_change_unconfirmed",
"changing the image from %q to %q opens this world with the new image's Minecraft version, "+
"and chunks it upgrades cannot be opened by the old one again; back the world up first, "+
"then resend with confirmImageChange", info.Image, image))
return
}
patch.Image = &image
changed = append(changed, "image")
imageFrom = info.Image
}
}
// Memory and the resource overrides move together: resolveResources derives the
// JVM heap and the §22 non-zero ceiling from the FINAL memory limit, and the
// override block is meaningless without that base. A resources-only patch has no
// base ceiling to widen (this endpoint does not read the current spec back), so
// it is rejected rather than guessed.
var (
newResources corev1.ResourceRequirements
resUpdated bool
)
if body.Memory != nil {
javaMemory, resources, err := resolveResources(*body.Memory, body.Resources)
if err != nil {
writeError(w, r, err)
return
}
patch.JavaMemory = &javaMemory
patch.Resources = &resources
changed = append(changed, "memory")
if body.Resources != nil {
changed = append(changed, "resources")
}
newResources = resources
resUpdated = true
} else if body.Resources != nil {
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
"resources overrides require memory to be set in the same patch"))
return
}
// Resource-cache consistency + quota enforcement (spec §9.3 / §22): every
// resource-mutating patch must update the cached columns so QuotaCheck
// can aggregate per-owner usage without cross-system CRD reads. For OWNED
// servers the owner's cumulative usage must also stay within their quota caps.
if resUpdated {
newCPU := quantityToMilli(newResources.Limits[corev1.ResourceCPU])
newMemMB := quantityToMB(newResources.Limits[corev1.ResourceMemory])
rec, err := a.Repo.ServerByName(r.Context(), name)
if err != nil && !errors.Is(err, ErrNotFound) {
writeError(w, r, err)
return
}
if rec != nil && rec.OwnerID != "" {
ok, err := a.Repo.QuotaCheck(r.Context(), rec.OwnerID, name,
ResourceSpec{CPUMilli: newCPU, MemoryMB: newMemMB})
if err != nil {
writeError(w, r, err)
return
}
if !ok {
writeError(w, r, newError(http.StatusForbidden, "quota_exceeded",
"this change would exceed the server owner's resource quota"))
return
}
}
if err := a.Cluster.PatchServerSpec(r.Context(), name, patch); err != nil {
a.writeLookupError(w, r, err)
return
}
// A resource patch cannot change storage, so its cached contribution must
// be preserved: passing 0 would silently zero the storage dimension of the
// owner's four-cap aggregate (the cached columns are its only input).
storMB := 0
if rec != nil {
cur, err := a.Repo.ServerResources(r.Context(), name)
if err != nil {
writeError(w, r, err)
return
}
storMB = cur.StorageMB
}
_ = a.Repo.UpdateServerResources(r.Context(), name, newCPU, newMemMB, storMB)
} else {
if err := a.Cluster.PatchServerSpec(r.Context(), name, patch); err != nil {
a.writeLookupError(w, r, err)
return
}
}
if patch.Image != nil {
a.auditImageChange(r, name, imageFrom, *patch.Image)
} else {
a.audit(r, "server.patch", name)
}
writeJSON(w, http.StatusOK, map[string]any{
"name": name,
"patched": changed,
})
}
// ---- authorization helpers ----
// authorizeWake applies the autostartPolicy gate (spec §9.4). The owner and any
// admin may always wake; otherwise the policy decides. An empty/unknown policy
// fails safe (owner-only).
func (a *API) authorizeWake(ctx context.Context, p *Principal, info *ServerInfo, rec *ServerRecord) error {
if p.IsAdmin() {
return nil
}
if rec != nil && rec.OwnerID != "" && rec.OwnerID == p.UserID {
return nil
}
switch info.AutostartPolicy {
case string(v1alpha1.AutostartPublic):
return nil
case string(v1alpha1.AutostartAllowlist):
ok, err := a.Repo.UserInAllowlist(ctx, info.Name, p.UserID)
if err != nil {
return err
}
if ok {
return nil
}
return errForbidden
default: // ownerOnly or unset → only owner/admin, already handled above
return errForbidden
}
}
// isOwnerOrAdmin reports whether p owns rec or is an admin.
func (a *API) isOwnerOrAdmin(p *Principal, rec *ServerRecord) bool {
if p.IsAdmin() {
return true
}
return rec != nil && rec.OwnerID != "" && rec.OwnerID == p.UserID
}
// quantityToMilli converts a K8s resource.Quantity to millicores (e.g. "2"→2000,
// "500m"→500). A zero/unset quantity returns 0.
func quantityToMilli(q resource.Quantity) int {
if q.IsZero() {
return 0
}
return int(q.MilliValue())
}
// quantityToMB converts a K8s resource.Quantity to whole megabytes, rounding up
// (e.g. "4Gi"→4096, "1G"→1000). A zero/unset quantity returns 0.
func quantityToMB(q resource.Quantity) int {
if q.IsZero() {
return 0
}
mb := q.Value() / (1024 * 1024)
if mb < 1 {
return 1
}
return int(mb)
}