package main import ( "bytes" "context" "fmt" "time" "felis.lolicon.best/internal/apis/felis/v1alpha1" "felis.lolicon.best/internal/naming" corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/api/meta" "k8s.io/apimachinery/pkg/api/resource" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" clientgoscheme "k8s.io/client-go/kubernetes/scheme" "k8s.io/client-go/tools/clientcmd" "k8s.io/client-go/util/retry" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" ) // System services are the always-on backends Felis provisions for itself after // setup: the login limbo (LOOHP/Limbo auth gate) and the lobby (Paper + the // felis-paper /menu hub). Unlike a user server they are created Running, are // exempt from the world reaper, and carry reserved names — so they take the // ValidateSystemServerName admission path rather than the user ValidateServerName. // // The routing topology and the one security invariant they encode: // // connect → login (auth gate, front door) → lobby (/menu hub) → target backend // // A stopped/starting server's fallback must land on the LOGIN gate, never on the // lobby: falling back to the lobby would drop an unauthenticated player past the // gate. So login has NO fallback (if it is down we refuse the connection rather // than route onward) and everything else — the lobby included — falls back to // login. "Rather have login unreachable than abandon authentication." // systemServerSpec is the small, explicit shape a system service is built from. // It is intentionally narrower than the user applyRequest: no autostart choice // (always public), no RCON, no resource overrides — a system service is uniform // by construction so the invariants above cannot be configured away. type systemServerSpec struct { name string subdomain string displayName string image string memory string // container memory limit == request (§22 ceiling) storage string // world PVC size fallbackServer string // "" = none (refuse when down); never the lobby healthHTTPPort int32 // > 0 → gate readiness on an HTTP health endpoint // rcon opts a system service into the RCON write channel. It is per-service and // NOT a default, because enabling it on a backend that runs no RCON listener is // actively destructive rather than merely useless: the operator gates readiness // on the probe, so the server would never leave Starting and would eventually be // marked Failed. The login limbo is exactly that case (LOOHP/Limbo has no RCON), // and it is the front door — taking it down locks everyone out. rcon bool // env are extra plain (non-secret) environment variables baked into the pod. // System-service configuration derived from the deployment (the internal API // URL, root domain, lobby name) rides here; secrets never do — the service // token is injected by the operator via secretKeyRef, not as a literal value. env []v1alpha1.EnvVar } // systemRcon renders the RCON block for a system service. The secret name comes // from naming.RconSecretName — the same convention felis-api writes for user // servers and the operator provisions against — so a system service is not a // second, parallel way of doing this. Port is left 0 so the operator's default is // the only place the number lives. func systemRcon(in systemServerSpec) v1alpha1.RconSpec { if !in.rcon { return v1alpha1.RconSpec{Enabled: false} } return v1alpha1.RconSpec{ Enabled: true, SecretRef: v1alpha1.SecretKeyRef{ Name: naming.RconSecretName(in.name), Key: naming.RconSecretKey, }, } } // felisLimboHealthPort is the port the felis-limbo readiness plugin serves its // HTTP health endpoint on. The login system service gates pod readiness on it so // "the limbo has finished starting" — not merely "the game socket is bound" — // is what marks it Ready. const felisLimboHealthPort int32 = 8080 // The felis-limbo login plugin reads its deployment configuration from these // environment variables (env wins over its felis-link.properties template). The // non-secret four are baked into the login pod's Spec.Env here at provision time // (they derive from the deployment: the internal API URL, the root domain, the // resolved panel host, the lobby server name); the service-token secret is injected // separately by the operator via secretKeyRef. Without the token the plugin // fail-safes to readiness-only, so a login pod that has the URL/domain but not yet // the token is safe (it simply does not authenticate) rather than broken. const ( envAPIBaseURL = naming.EnvAPIBaseURL envRootDomain = "FELIS_ROOT_DOMAIN" envPanelHostname = "FELIS_PANEL_HOSTNAME" envLobbyServer = "FELIS_LOBBY_SERVER" ) // buildSystemServer constructs an always-on, reaper-exempt MinecraftServer from // a systemServerSpec. It is a pure function (no K8s, no I/O) so it is unit // testable without a cluster. Unlike buildMinecraftServerFromApplyRequest it: // - permits reserved names (login/lobby) via ValidateSystemServerName, // - sets DesiredState=Running (the service is up the moment it exists), // - sets ReaperExempt=true and AutostartPolicy=public, // - enables RCON only where the image actually serves it (in.rcon): the lobby // is Paper and needs the write channel like any user server, while the login // limbo has no RCON listener at all and gates readiness on pod TCP/HTTP // health instead — see the operator reconciler. func buildSystemServer(in systemServerSpec, namespace string) (*v1alpha1.MinecraftServer, error) { if err := naming.ValidateSystemServerName(in.name); err != nil { return nil, fmt.Errorf("invalid name: %w", err) } if err := naming.ValidateSystemServerName(in.subdomain); err != nil { return nil, fmt.Errorf("invalid subdomain: %w", err) } if in.image == "" { return nil, fmt.Errorf("image is required for system server %q", in.name) } // A system service must never fall back onto the lobby: that would route an // unauthenticated player past the login gate. Refuse to build one that does, // rather than silently ship the bypass. if in.fallbackServer == naming.SystemLobbyServer { return nil, fmt.Errorf("system server %q must not fall back to the lobby (%q) — it would bypass the login gate; fall back to %q or leave it empty", in.name, naming.SystemLobbyServer, naming.SystemLoginServer) } memQ, err := resource.ParseQuantity(in.memory) if err != nil { return nil, fmt.Errorf("invalid memory %q for %q: %w", in.memory, in.name, err) } if memQ.Sign() <= 0 { return nil, fmt.Errorf("memory must be positive for %q", in.name) } storageQ, err := resource.ParseQuantity(in.storage) if err != nil { return nil, fmt.Errorf("invalid storage %q for %q: %w", in.storage, in.name, err) } if storageQ.Sign() <= 0 { return nil, fmt.Errorf("storage must be positive for %q", in.name) } limits := corev1.ResourceList{corev1.ResourceMemory: memQ} requests := corev1.ResourceList{corev1.ResourceMemory: memQ} return &v1alpha1.MinecraftServer{ ObjectMeta: metav1.ObjectMeta{ Name: in.name, Namespace: namespace, Labels: map[string]string{ v1alpha1.LabelSystemRole: in.name, }, }, Spec: v1alpha1.MinecraftServerSpec{ Subdomain: in.subdomain, DisplayName: in.displayName, Image: in.image, JavaMemory: deriveApplyJavaHeap(memQ), DesiredState: v1alpha1.DesiredRunning, AutostartPolicy: v1alpha1.AutostartPublic, ReaperExempt: true, FallbackServer: in.fallbackServer, // Behind the Velocity proxy (which enforces online-mode and modern // forwarding), backends run offline-mode; the proxy is the one place // online-mode is true (spec §8, §11). OnlineMode: false, Rcon: systemRcon(in), Storage: v1alpha1.StorageSpec{Size: storageQ.String()}, Resources: corev1.ResourceRequirements{Limits: limits, Requests: requests}, Startup: v1alpha1.StartupSpec{HealthHTTPPort: in.healthHTTPPort}, Env: in.env, }, }, nil } // loginSystemServer is the LOOHP/Limbo auth gate. It is the front door and the // only safe fallback, so it carries no fallback of its own: if it is down the // proxy refuses the connection rather than routing onward past authentication. // // apiBaseURL is the felis-api internal face the login plugin authenticates to; // panelHostname is the resolved console/panel host the plugin links players at (the // single source of truth for that host — see defaultPanelHostname), and rootDomain // is kept for the plugin's own console. fallback when the panel env is absent // (an older operator). All three are baked in as plain env. The service token is NOT // passed here — the operator injects it via secretKeyRef so the credential never // lands in the CRD. func loginSystemServer(image, namespace, apiBaseURL, rootDomain, panelHostname string) (*v1alpha1.MinecraftServer, error) { return buildSystemServer(systemServerSpec{ name: naming.SystemLoginServer, subdomain: naming.SystemLoginServer, displayName: "Login", image: image, memory: "512Mi", storage: "1Gi", fallbackServer: "", // none — refuse if the gate is down healthHTTPPort: felisLimboHealthPort, env: []v1alpha1.EnvVar{ {Name: envAPIBaseURL, Value: apiBaseURL}, {Name: envRootDomain, Value: rootDomain}, {Name: envPanelHostname, Value: panelHostname}, {Name: envLobbyServer, Value: naming.SystemLobbyServer}, }, }, namespace) } // lobbySystemServer is the post-auth /menu hub (Paper + felis-paper). It falls // back to the login gate — never to itself and never onward — so a lobby that is // briefly down still routes players through authentication first. func lobbySystemServer(image, namespace string) (*v1alpha1.MinecraftServer, error) { return buildSystemServer(systemServerSpec{ name: naming.SystemLobbyServer, subdomain: naming.SystemLobbyServer, displayName: "Lobby", image: image, memory: "1Gi", storage: "2Gi", fallbackServer: naming.SystemLoginServer, // Paper serves RCON, and the lobby is administered through the panel like any // other server — online players, console, permissions all ride this channel. rcon: true, }, namespace) } // buildSystemServerClient builds a controller-runtime client for the setup-time // system-service provisioner. It is deliberately best-effort and never calls // ctrl.SetupSignalHandler (setup owns its own context): it first honours the // standard resolution (in-cluster, then $KUBECONFIG / --kubeconfig, then // ~/.kube/config) and, failing that, falls back to the k3s admin kubeconfig the // host bootstrap writes at hostBootstrapKubeconfigPath — the common case when // setup runs as root directly on a single-node control-plane host. A returned // error is not fatal to setup; the caller degrades to printed guidance. func buildSystemServerClient() (client.Client, error) { scheme := runtime.NewScheme() if err := clientgoscheme.AddToScheme(scheme); err != nil { return nil, err } if err := v1alpha1.AddToScheme(scheme); err != nil { return nil, err } cfg, err := ctrl.GetConfig() if err != nil { cfg, err = clientcmd.BuildConfigFromFlags("", hostBootstrapKubeconfigPath) if err != nil { return nil, fmt.Errorf("no reachable kubeconfig (tried in-cluster/$KUBECONFIG/~/.kube and %s): %w", hostBootstrapKubeconfigPath, err) } } return client.New(cfg, client.Options{Scheme: scheme}) } // systemServerOutcome records what ensureSystemServers did with one service so // setup can report it without the provisioner deciding on the output format. type systemServerOutcome struct { name string created bool // true = we created it this run updated bool // true = we refreshed an existing replica from the source available bool // true = the required object now exists skipped string // non-empty = why it was skipped (image unset / already exists) err error // non-nil = create failed changes []string // converge only: the fields this pass filled } // systemServerPlan is one system service in the provisioner's table: its name, // the image config gives it, and the pure builder for its desired CR. type systemServerPlan struct { name string image string build func(image, namespace string) (*v1alpha1.MinecraftServer, error) } // systemServerPlans is the single description of the login+lobby pair, shared by // ensureSystemServers (create-if-absent) and convergeSystemServers (field fill). func systemServerPlans(loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname string) []systemServerPlan { return []systemServerPlan{ {name: naming.SystemLoginServer, image: loginImage, build: func(image, ns string) (*v1alpha1.MinecraftServer, error) { return loginSystemServer(image, ns, apiBaseURL, rootDomain, panelHostname) }}, {name: naming.SystemLobbyServer, image: lobbyImage, build: lobbySystemServer}, } } // ensureSystemServers idempotently creates the login and lobby system services. // It create-if-absent per service: an existing CRD is left untouched (so an // operator's later edits to a system service survive re-runs of setup), a // service whose image is unset in config is skipped with a reason, and any other // service is created. It never deletes or overwrites. The caller supplies the // K8s client and namespace; this function performs no signal-handler or client // setup of its own. func ensureSystemServers(ctx context.Context, cl client.Client, namespace, loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname string) []systemServerOutcome { plans := systemServerPlans(loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname) outcomes := make([]systemServerOutcome, 0, len(plans)) for _, p := range plans { if p.image == "" { outcomes = append(outcomes, systemServerOutcome{name: p.name, skipped: "image not configured"}) continue } ms, err := p.build(p.image, namespace) if err != nil { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err}) continue } // Create-if-absent: check first so an existing service is reported as a // deliberate skip rather than an AlreadyExists error. Never adopt a // legacy user server that happens to occupy a reserved system name. var existing v1alpha1.MinecraftServer getErr := cl.Get(ctx, client.ObjectKeyFromObject(ms), &existing) if getErr == nil { if existing.Labels[v1alpha1.LabelSystemRole] != p.name { outcomes = append(outcomes, systemServerOutcome{ name: p.name, err: fmt.Errorf( "existing MinecraftServer %s/%s is not marked as the Felis %q system role; remove or rename it, then rerun setup", namespace, p.name, p.name, ), }) continue } refreshed, err := refreshDerivedEnv(ctx, cl, &existing, ms) if err != nil { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err}) continue } skipped := "already exists" if refreshed { skipped = "already exists; refreshed the console hostnames it points players at" } outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, skipped: skipped}) continue } if !apierrors.IsNotFound(getErr) { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: getErr}) continue } if err := cl.Create(ctx, ms); err != nil { if apierrors.IsAlreadyExists(err) { // Close the Get/Create race without trusting the object that won it. var raced v1alpha1.MinecraftServer if getErr := cl.Get(ctx, client.ObjectKeyFromObject(ms), &raced); getErr != nil { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: getErr}) continue } if raced.Labels[v1alpha1.LabelSystemRole] != p.name { outcomes = append(outcomes, systemServerOutcome{ name: p.name, err: fmt.Errorf( "concurrent MinecraftServer %s/%s is not marked as the Felis %q system role; refusing to adopt it", namespace, p.name, p.name, ), }) continue } outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, skipped: "already exists"}) continue } outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err}) continue } outcomes = append(outcomes, systemServerOutcome{name: p.name, created: true, available: true}) } return outcomes } // derivedSystemEnv are the system-server env vars whose values setup computes from // config rather than inventing. They are the exception to create-if-absent, and the // exception is narrow on purpose. // // Everything else on an existing system service is left alone so an operator's edits // survive a re-run — but these are not the operator's to own, they are a copy of // config that goes stale the moment config changes. That is not hypothetical: after // a root-domain change the login gate keeps handing every joining player a console // link built from the OLD domain, which is the one screen an unauthenticated player // is guaranteed to see. Nothing else in the install rewrites them, so a re-run of // setup is the only chance they get to catch up. var derivedSystemEnv = map[string]bool{ envAPIBaseURL: true, envRootDomain: true, envPanelHostname: true, } // refreshDerivedEnv converges the config-derived env of an existing system server // onto what setup just computed, and reports whether anything actually changed. // // It only ever overwrites a name that is already present with a different value, and // only for the names above: env the operator added by hand is untouched, and a name // missing from the live object is left missing rather than added back, since a // deliberate removal is indistinguishable from drift and re-adding it would fight the // operator every run. func refreshDerivedEnv(ctx context.Context, cl client.Client, existing, desired *v1alpha1.MinecraftServer) (bool, error) { want := derivedEnvWanted(desired) changed, err := patchOnConflictRetry(ctx, cl, existing, func() bool { changed := false for i, e := range existing.Spec.Env { if v, ok := want[e.Name]; ok && v != e.Value { existing.Spec.Env[i].Value = v changed = true } } return changed }) if err != nil { return false, fmt.Errorf("refresh %s env: %w", existing.Name, err) } return changed, nil } // patchOnConflictRetry applies mutate to obj and sends only the difference, as a // merge patch that carries the resourceVersion obj was read at. The operator writes // status and felis-api patches spec.idle on these same objects, so a write can land // between setup's read and its patch: the apiserver then answers 409, and this // re-reads obj and runs mutate again on the fresh copy, up to retry.DefaultRetry's // five attempts. The pinned resourceVersion is what keeps a list field such as // spec.env safe — a merge patch replaces a list whole, and without the lock an // entry added concurrently would be dropped. mutate reports whether it changed // anything; nothing is sent when it did not. obj holds the stored object after. func patchOnConflictRetry(ctx context.Context, cl client.Client, obj client.Object, mutate func() bool) (bool, error) { key := client.ObjectKeyFromObject(obj) changed, reread := false, false err := retry.RetryOnConflict(retry.DefaultRetry, func() error { if reread { if err := cl.Get(ctx, key, obj); err != nil { return err } } reread = true base := obj.DeepCopyObject().(client.Object) if changed = mutate(); !changed { return nil } return cl.Patch(ctx, obj, client.MergeFromWithOptions(base, client.MergeFromWithOptimisticLock{})) }) return changed, err } // derivedEnvWanted maps the derived env keys of desired onto their values. func derivedEnvWanted(desired *v1alpha1.MinecraftServer) map[string]string { want := make(map[string]string, len(derivedSystemEnv)) for _, e := range desired.Spec.Env { if derivedSystemEnv[e.Name] { want[e.Name] = e.Value } } return want } // convergeSystemServers is the explicit convergence pass over already-installed // system servers (#1). ensureSystemServers is create-if-absent by design — an // existing CR is left alone so a re-run cannot clobber an operator's edits — and // that leaves no path for a field the DESIRED spec gained after the install: // spec.rcon (the lobby's write channel), spec.startup.healthHTTPPort (the login // gate's readiness probe), or a config-derived env key that did not exist yet. // Such fields sit at their zero value forever while re-running setup reports // success, which is exactly the reported "configuration updates never reach an // installed deployment" symptom. // // This pass fills exactly those zero-value fields and the config-derived env keys, // and nothing else: a field already holding a non-zero value is the operator's and // is never overwritten. It is an explicit command rather than an implicit step of // setup because some fills need an ordering only the operator knows — enabling // RCON or the HTTP readiness gate on a server whose image predates the listener // would hold that server in Starting until it was marked Failed. Rebuild (or // upgrade) the images first, then run this. func convergeSystemServers(ctx context.Context, cl client.Client, namespace, loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname string) []systemServerOutcome { outcomes := make([]systemServerOutcome, 0, 2) for _, p := range systemServerPlans(loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname) { if p.image == "" { outcomes = append(outcomes, systemServerOutcome{name: p.name, skipped: "image not configured"}) continue } desired, err := p.build(p.image, namespace) if err != nil { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err}) continue } var existing v1alpha1.MinecraftServer switch err := cl.Get(ctx, client.ObjectKeyFromObject(desired), &existing); { case apierrors.IsNotFound(err): outcomes = append(outcomes, systemServerOutcome{name: p.name, skipped: "not present — run `sudo felis setup` first"}) continue case err != nil: outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err}) continue } if existing.Labels[v1alpha1.LabelSystemRole] != p.name { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: fmt.Errorf( "existing MinecraftServer %s/%s is not marked as the Felis %q system role; refusing to converge it", namespace, p.name, p.name, )}) continue } var changes []string changed, err := patchOnConflictRetry(ctx, cl, &existing, func() bool { changes = nil if existing.Spec.Rcon == (v1alpha1.RconSpec{}) && desired.Spec.Rcon != (v1alpha1.RconSpec{}) { existing.Spec.Rcon = desired.Spec.Rcon changes = append(changes, "spec.rcon") } if existing.Spec.Startup.HealthHTTPPort == 0 && desired.Spec.Startup.HealthHTTPPort != 0 { existing.Spec.Startup.HealthHTTPPort = desired.Spec.Startup.HealthHTTPPort changes = append(changes, "spec.startup.healthHTTPPort") } changes = append(changes, convergeDerivedEnv(&existing, desired)...) return len(changes) > 0 }) if err != nil { outcomes = append(outcomes, systemServerOutcome{name: p.name, err: fmt.Errorf("converge %s: %w", p.name, err)}) continue } if !changed { outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, skipped: "already converged"}) continue } outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, updated: true, changes: changes}) } return outcomes } // convergeDerivedEnv makes the config-derived env match the desired values: a key // whose value drifted is overwritten, and a key missing entirely is added. This is // the wider half of the same explicit pass — refreshDerivedEnv's present-only loop // can never introduce a NEW key, which is how a derived key added after an install // never reached it at all. func convergeDerivedEnv(existing, desired *v1alpha1.MinecraftServer) []string { want := derivedEnvWanted(desired) var changes []string present := make(map[string]bool, len(existing.Spec.Env)) for i := range existing.Spec.Env { e := &existing.Spec.Env[i] present[e.Name] = true if v, ok := want[e.Name]; ok && v != e.Value { e.Value = v changes = append(changes, "env "+e.Name) } } for _, e := range desired.Spec.Env { if !derivedSystemEnv[e.Name] || present[e.Name] { continue } existing.Spec.Env = append(existing.Spec.Env, e) changes = append(changes, "env "+e.Name) } return changes } // The login gate is a hard prerequisite of the Owner bind, so setup waits for it // rather than racing it. The ceiling covers a cold image pull on a fresh node; // the poll is fast enough that a warm start feels immediate. const ( loginGateReadyTimeout = 5 * time.Minute loginGatePollInterval = 3 * time.Second ) // awaitLoginGateReady blocks until the login system server reports status.ready. // // The Owner claims their seat by JOINING the game and running /link, so the gate // being up is not a nicety — it is the precondition for the very next thing setup // asks of the operator. progress is called on each phase change so the caller can // show movement during a cold image pull; it may be nil. func awaitLoginGateReady(ctx context.Context, cl client.Client, namespace string, timeout, poll time.Duration, progress func(v1alpha1.Phase)) error { key := client.ObjectKey{Namespace: namespace, Name: naming.SystemLoginServer} deadline := time.Now().Add(timeout) last := v1alpha1.Phase("") for { var ms v1alpha1.MinecraftServer switch err := cl.Get(ctx, key, &ms); { case err == nil: if ms.Status.Ready { return nil } if ms.Status.Phase != last { last = ms.Status.Phase if progress != nil { progress(last) } } // The operator only marks Failed once its OWN startup deadline has already // elapsed, so Failed is a settled verdict rather than a transient — sitting // out the rest of our timeout on top of it would only hide the reason. if ms.Status.Phase == v1alpha1.PhaseFailed { return fmt.Errorf("the login gate failed to start: %s", readyConditionMessage(&ms)) } case !apierrors.IsNotFound(err): return err } if !time.Now().Before(deadline) { return fmt.Errorf("timed out after %s waiting for the login gate to become ready (last phase: %s)", timeout, phaseOrPending(last)) } select { case <-ctx.Done(): return ctx.Err() case <-time.After(poll): } } } // readyConditionMessage is the operator's own account of why the gate is not // ready — far more useful to an operator than "phase: Failed". func readyConditionMessage(ms *v1alpha1.MinecraftServer) string { if c := meta.FindStatusCondition(ms.Status.Conditions, v1alpha1.ConditionReady); c != nil && c.Message != "" { return c.Message } return "no Ready condition was reported" } // phaseOrPending names the empty phase, which means the operator has not // reconciled the server yet (commonly: the operator itself is not running). func phaseOrPending(p v1alpha1.Phase) string { if p == "" { return "not yet reconciled — is the felis operator running?" } return string(p) } // provisionSecretReplicas copies the Secrets workload pods mount from the control // namespace into the namespaces those pods run in. The proxy's felis-service-token // is not among them: it lives in the control namespace and on the host, and a copy // anywhere else would open every game route to whoever reads that namespace. func provisionSecretReplicas(ctx context.Context, cl client.Client, controlNS, minecraftNS, buildNS string) []systemServerOutcome { return []systemServerOutcome{ // refresh=true for both caller tokens: the control namespace holds the // current value and `felis rotate-token` replaces it there, so a replica // that differs is stale and the login gate would be turned away with it. ensureSecretReplica(ctx, cl, controlNS, minecraftNS, naming.LimboTokenSecretName, naming.ServiceTokenSecretKey, "limbo-token", "minecraft ns", true), ensureSecretReplica(ctx, cl, controlNS, minecraftNS, naming.ForwardingSecretName, naming.ForwardingSecretKey, "forwarding-secret", "minecraft ns", false), // refresh=true: felis-config is the rendered config, not a credential. The // backup/restore/fileedit Jobs and the reaper mount this copy, so a re-run // must update it when the control plane's render has moved on (a stale copy // e.g. keeps an old database URL after a credential rotation). ensureSecretReplica(ctx, cl, controlNS, minecraftNS, "felis-config", "felis.toml", "config", "minecraft ns", true), // The reaper's pre-reap warning emails authenticate with the same relay // password felis-api uses; the reaper pod runs in the minecraft namespace, // where a secretKeyRef resolves only against a local mirror. Skipped while // the relay is not configured yet — the "configure email" screen refreshes // both mirrors when it applies. ensureSecretReplica(ctx, cl, controlNS, minecraftNS, "felis-smtp", "password", "smtp", "minecraft ns", false), // The build namespace needs the build token: the build Job's fetch // initContainer reads the submission context from the internal face, and // that is all this token opens. Best effort — a deployment that only // installs the control plane simply never builds a user submission. ensureSecretReplica(ctx, cl, controlNS, buildNS, naming.BuildTokenSecretName, naming.ServiceTokenSecretKey, "build-token", "felis-build ns", true), } } // ensureSecretReplica copies one Secret from the control namespace into a workload // namespace (minecraft — or the build namespace, whose fetch initContainer reads the // context from the felis-api internal face with the build token) so a pod can mount it // via secretKeyRef. A secretKeyRef is namespace-local, but those workloads do not run // beside the control plane — so without this replica the secretKeyRef would dangle and // wedge the pod in CreateContainerConfigError. // // Several Secrets need it, for different reasons: the caller tokens of the login // limbo and the build Pod's context fetch (both authenticate to the felis-api // internal face, each with its own token), // the Velocity modern-forwarding secret (every backend — it is how a backend knows // a login really came from the proxy, and so that the player's UUID is Mojang-verified // rather than offline-derived), and the SMTP relay password (the reaper's pre-reap // warning emails; the felis-config mirror is what carries [smtp] into its pod). // // It is create-if-absent: an existing replica is left untouched so a hand-rotated // value in the workload namespace is never clobbered (to rotate, delete the replica // and re-run setup). Best-effort like the rest of the provisioner: a missing source or // a create failure degrades to a reported outcome, never a hard setup failure. It // copies only Type and Data — never labels/annotations/ownerRefs — so the replica // carries no accidental GC owner or managed-by lineage. // // refreshExisting switches a replica to refresh-in-place from the control namespace. // The felis-config mirror uses it because that Secret is a rendered config and the // workload Jobs that mount it (backup/restore/fileedit) plus the reaper silently // misbehave on a stale copy — e.g. after a database credential rotation the control // plane moves on while every backup Job keeps failing auth. The caller tokens use it // because `felis rotate-token` replaces them in the control namespace, which makes a // differing replica stale by definition. The forwarding and SMTP Secrets keep the // never-overwrite rule. func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace, minecraftNamespace, secretName, secretKey, label, where string, refreshExisting bool) systemServerOutcome { name := label + " (" + where + ")" validate := func(secret *corev1.Secret, location, skipped string) systemServerOutcome { if len(secret.Data[secretKey]) == 0 { return systemServerOutcome{name: name, skipped: fmt.Sprintf( "Secret %s/%s has no non-empty %q key", location, secretName, secretKey)} } return systemServerOutcome{name: name, available: true, skipped: skipped} } // refreshFromControl updates an existing replica from the control-namespace source // when the rendered key differs. Only the felis-config mirror opts in. refreshFromControl := func(existing *corev1.Secret) systemServerOutcome { var src corev1.Secret if err := cl.Get(ctx, client.ObjectKey{Namespace: controlNamespace, Name: secretName}, &src); err != nil { if apierrors.IsNotFound(err) { return systemServerOutcome{name: name, skipped: fmt.Sprintf( "source Secret %s/%s not found — provision it (deploy/bootstrap.sh), then re-run setup", controlNamespace, secretName)} } return systemServerOutcome{name: name, err: err} } if out := validate(&src, controlNamespace, ""); !out.available { return out } changed, err := patchOnConflictRetry(ctx, cl, existing, func() bool { if bytes.Equal(existing.Data[secretKey], src.Data[secretKey]) { return false } if existing.Data == nil { existing.Data = map[string][]byte{} } existing.Data[secretKey] = src.Data[secretKey] return true }) if err != nil { return systemServerOutcome{name: name, err: err} } if !changed { return validate(existing, minecraftNamespace, "already current") } return systemServerOutcome{name: name, updated: true, available: true} } if controlNamespace == minecraftNamespace { // Same namespace needs no replica, but the source still has to exist. var existing corev1.Secret err := cl.Get(ctx, client.ObjectKey{Namespace: minecraftNamespace, Name: secretName}, &existing) if err == nil { return validate(&existing, minecraftNamespace, "control and minecraft namespaces coincide") } if apierrors.IsNotFound(err) { return systemServerOutcome{name: name, skipped: fmt.Sprintf( "source Secret %s/%s not found — provision it (deploy/bootstrap.sh), then re-run setup", controlNamespace, secretName)} } return systemServerOutcome{name: name, err: err} } // Never overwrite an existing replica (it may hold a rotated value). var existing corev1.Secret getErr := cl.Get(ctx, client.ObjectKey{Namespace: minecraftNamespace, Name: secretName}, &existing) if getErr == nil { if refreshExisting { return refreshFromControl(&existing) } return validate(&existing, minecraftNamespace, "already exists") } if !apierrors.IsNotFound(getErr) { return systemServerOutcome{name: name, err: getErr} } // Read the source of truth from the control namespace. var src corev1.Secret if err := cl.Get(ctx, client.ObjectKey{Namespace: controlNamespace, Name: secretName}, &src); err != nil { if apierrors.IsNotFound(err) { return systemServerOutcome{name: name, skipped: fmt.Sprintf( "source Secret %s/%s not found — provision it (deploy/bootstrap.sh), then re-run setup", controlNamespace, secretName)} } return systemServerOutcome{name: name, err: err} } if out := validate(&src, controlNamespace, ""); !out.available { return out } replica := &corev1.Secret{ ObjectMeta: metav1.ObjectMeta{Name: secretName, Namespace: minecraftNamespace}, Type: src.Type, Data: src.Data, } if err := cl.Create(ctx, replica); err != nil { if apierrors.IsAlreadyExists(err) { if getErr := cl.Get(ctx, client.ObjectKey{Namespace: minecraftNamespace, Name: secretName}, &existing); getErr != nil { return systemServerOutcome{name: name, err: getErr} } if refreshExisting { return refreshFromControl(&existing) } return validate(&existing, minecraftNamespace, "already exists") } return systemServerOutcome{name: name, err: err} } return systemServerOutcome{name: name, created: true, available: true} } func requiredProvisioningError(outcomes []systemServerOutcome) error { required := map[string]struct{}{ "limbo-token (minecraft ns)": {}, "forwarding-secret (minecraft ns)": {}, naming.SystemLoginServer: {}, } for _, o := range outcomes { if o.err != nil { return fmt.Errorf("%s: %w", o.name, o.err) } if _, ok := required[o.name]; ok && !o.available { reason := o.skipped if reason == "" { reason = "object was not created" } return fmt.Errorf("%s unavailable: %s", o.name, reason) } } return nil }