package platform import ( "crypto/sha256" "encoding/hex" "fmt" "time" "felis.lolicon.best/internal/naming" "felis.lolicon.best/internal/placement" appsv1 "k8s.io/api/apps/v1" batchv1 "k8s.io/api/batch/v1" corev1 "k8s.io/api/core/v1" "k8s.io/apimachinery/pkg/api/resource" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/util/intstr" ) // This file renders the running control-plane workloads — the felis-api and // felis-operator Deployments, the in-cluster image registry (Deployment + // Service + PVC), and (when configured) the reaper CronJob. They are what make // the RBAC and NetworkPolicy fence MEAN something: each Deployment runs as its // matching control-plane SA (so the namespaced Roles actually bind to a workload) // and stamps the recommended labels the RCON NetworkPolicy peer selects on (so // the api console / operator prober can reach RCON through the fence). // workloads_test.go evaluates those correspondences with the SAME selector // machinery K8s uses, because no cluster runs here. // // The reaper CronJob (spec §18 three-clock retention) reaps worlds ONLY when the // storage topology is supplied — WorldsHostPath + BackupPVC + ArchiveLocalPath, // gated by reaperEnabled. World reaping is opt-in-when-configured rather than always-on // because archiving idle worlds means mounting where the worlds physically live, // and the spec keeps that open (§18/§19: tarLocal-on-local-path is the starter, // Longhorn/snapshot the documented evolution, and they do not share a mount // model). The starter model — a node-local worlds-root, exposed through a static // hostPath PV and mounted read-only — is the only one coherent with the operator's per-server // ReadWriteOnce world PVCs (a shared RWX worlds mount would contradict them), so // that is what renders; when the trio is absent no world is ever reaped, which is // the fail-safe choice for a workload that deletes PVCs. With only the archive // store configured (BackupPVC + ArchiveLocalPath) the CronJob still renders, in // a retention-only shape that expires, reads back and sweeps backups and never // touches a world (retentionEnabled). The reaper resolver // (cmd/felis/reaper.resolveWorldDir) finds worlds either as / // or in the stock local-path layout k3s writes under its storage root // (__, read from the live PVC), so pointing // --worlds-host-path at /var/lib/rancher/k3s/storage works on a default install — // see the WorldsHostPath field doc. Whether the tar finds a world still depends on // the hosting node, and is not provable without a cluster. No nodeSelector is set // unless ReaperNode names one: the single-node starter pins the worlds to one node // implicitly, while a multi-node deployment passes --reaper-node (rendered as a // kubernetes.io/hostname selector and PV node affinity) or the CronJob could // schedule on a node where the worlds-root is empty. const ( // configSecretName and the caller token Secrets (naming.CallerTokens) are // referenced BY NAME and NEVER rendered into the bundle: felis.toml carries the // database URL (a credential) and each token is a credential, so writing any of // them into a checked-in manifest is a hard red line. The deployment provisions // these Secrets out-of-band before applying these workloads. configSecretName = "felis-config" configSecretKey = "felis.toml" configMountPath = "/etc/felis" configFilePath = "/etc/felis/felis.toml" felisBinaryPath = "/usr/local/bin/felis" // The key every caller token Secret stores its value under (internal/naming). serviceTokenSecretKey = naming.ServiceTokenSecretKey // Ports, single-sourced with the entrypoints (cmd/felis). The api external // port must match server.listen in felis.toml (default 0.0.0.0:8080); that // agreement lives in the out-of-band config Secret and cannot be enforced here. apiExternalPort int32 = 8080 apiInternalPort int32 = 8081 apiHTTPSPort int32 = 8443 operatorMetricsPort int32 = 8080 // operatorHealthPort must match cmd/felis/operator.go's // --health-probe-bind-address default (and the arg rendered below): it is the // only listener the operator Deployment's probes can dial — :8080 is the // metrics server, which serves no health endpoints. operatorHealthPort int32 = 8081 apiTLSSecretName = "felis-api-tls" apiTLSMountPath = "/etc/felis/tls" registryName = "registry" registryDataPath = "/var/lib/registry" registryStorageSize = "10Gi" // registryLoopbackHost is the interface the registry's hostPort binds. Node-level // containerd is the only client that needs it: it cannot reach the registry Service // VIP (the live stack answered "Empty reply" to it), so the node's registries.yaml // mirror rewrites registry..svc: onto http://127.0.0.1: and the // pull lands here. Loopback-only is deliberate — the registry serves plain HTTP // and must never be reachable off the node. registryLoopbackHost = "127.0.0.1" // registryGateName is the write-authorization sidecar in the registry pod, and // registryAuth* mount its per-principal token files (naming.RegistryAuthSecretName). registryGateName = "registry-gate" registryAuthVolume = "registry-auth" registryAuthMountPath = "/etc/felis-registry-auth" // registryGCName is the garbage-collection sidecar, and registryMaint* the // emptyDir where the gate keeps an open read-only window across its own restart // (registrygate.SetMaintenanceState). registryGCInterval is how often a sweep // runs; the pruner in felis-api deletes manifests between sweeps. registryGCName = "registry-gc" registryMaintVolume = "maint" registryMaintMountPath = "/run/felis-maint" registryGCInterval = 24 * time.Hour configVolume = "config" tmpVolume = "tmp" registryVolume = "data" worldsVolume = "worlds" backupVolume = "backup" uploadsVolume = "uploads" // uploads* wire the user-uploads build-context store into felis-api. The PVC // backs a LOCAL user_uploads_context — durable across pod restarts and // fsGroup-writable by the non-root pod (unlike a root-owned hostPath). It is // always rendered but only written to when uploads are local. The S3 secret/env // carry credentials for an s3:// user_uploads_context and are OPTIONAL, so a // local-storage install (no such Secret) still starts. uploadsPVCName = "felis-uploads" uploadsStorageSize = "5Gi" // backupStorageSize is the world-archive store's default capacity. It is a // starter-sized floor (registry 10Gi, uploads 5Gi sit beside it): archives are // compressed worlds and a fresh one is ~200MB, so this holds many while leaving // the growth knob (resize the PVC / move to a snapshot backend, spec §19) to the // operator. There is no storageClassName: the cluster default is the only safe // binding, exactly like registryPVC. backupStorageSize = "10Gi" // UploadsLocalPath is the in-pod mount of the uploads PVC; a local // user_uploads_context points here so the derived context ref and the on-disk // write location agree. Exported so the setup wizard stamps it into felis.toml. UploadsLocalPath = "/var/lib/felis/uploads" // UploadsS3SecretName is the out-of-band Secret carrying the S3 credentials for // an s3:// user_uploads_context. felis-api mounts its keys into env (optionally) // and the setup wizard creates it, keeping a host copy in /etc/felis that every // installer run applies it from again. Like configSecretName it is NEVER rendered // into the bundle — the credentials are the same red line. UploadsS3SecretName = "felis-uploads-s3" UploadsS3SecretAccessKey = "access_key_id" UploadsS3SecretSecretKey = "secret_access_key" // UploadsS3AccessKeyEnv / UploadsS3SecretKeyEnv are the env vars felis-api reads // the S3 credentials from; registry.s3.access_key_ref / secret_key_ref default to // these names. The deployment injects them from UploadsS3SecretName (optional). UploadsS3AccessKeyEnv = "FELIS_UPLOADS_S3_ACCESS_KEY" UploadsS3SecretKeyEnv = "FELIS_UPLOADS_S3_SECRET_KEY" // SMTPSecretName is the out-of-band Secret carrying the [smtp] relay password. // Same red line as the S3 credentials: the setup wizard's "configure email" // step creates it (and keeps a host copy in /etc/felis that every installer run // applies it from again), felis-api reads it via SMTPPasswordEnv (optionally — a // mailer-less install has no such Secret and still starts, falling back to // logging codes), and it is never rendered into the bundle. SMTPSecretName = "felis-smtp" SMTPSecretPasswordKey = "password" // SMTPPasswordEnv is the env var felis-api reads the relay password from; // [smtp] password_ref defaults to this name. SMTPPasswordEnv = "FELIS_SMTP_PASSWORD" // RegistryPruneTokenEnv carries the registry gate's prune principal token into // felis-api, whose pruner deletes the manifests nothing references // (internal/registryprune). It comes from the prune key of // naming.RegistryAuthSecretName, optionally: that Secret lives in the registry // namespace, which is the control namespace on every install the bootstrap // makes, and an install that splits them or predates the key runs without the // pruner. RegistryPruneTokenEnv = "FELIS_REGISTRY_PRUNE_TOKEN" registryPruneTokenKey = "prune" // worldsMountPath is where the reaper CronJob mounts the worlds-root (read-only). // It is the default of `felis reaper --worlds-root`; the resolver then reads each // world at /. Single-sourced with cmd/felis/reaper.go. worldsMountPath = "/worlds" // reaperSchedule is the daily retention cadence (spec §18: a daily batch). 04:00 // is an off-peak window; the reaper itself is idempotent and run-once, so the // exact minute is not load-bearing. ConcurrencyPolicy=Forbid keeps a slow run // from overlapping the next day's. reaperSchedule = "0 4 * * *" // reaperStartingDeadlineSeconds bounds how late a missed run may still start (a // controller outage at 04:00 shouldn't silently skip retention) without letting // a long backlog pile up. reaperActiveDeadlineSeconds caps a single run so a // wedged archive can't hold the Forbid lock forever. reaperStartingDeadlineSeconds int64 = 300 reaperActiveDeadlineSeconds int64 = 3600 reaperBackoffLimit int32 = 2 reaperHistoryLimit int32 = 3 // nonRootUID matches the tree's non-root identity convention // (internal/restore.defaultRunAsID). nonRootUID int64 = 1000 ) // UploadsPVCName is felis-api's submission uploads PVC in the control-plane // namespace; the host-side off-site copy reads and restores it. const UploadsPVCName = uploadsPVCName // The Secrets felis-api mounts its config and panel certificate from, which the // host rewrites when the install moves to a new root domain (felis domain set). const ( ConfigSecretName = configSecretName ConfigSecretKey = configSecretKey APITLSSecretName = apiTLSSecretName ) // ControlPlaneUID is the uid and gid the control-plane pods run as, which own // what they write to their volumes; a host-side restore into one of them // writes as it. const ControlPlaneUID = int(nonRootUID) // APIInternalServiceName is the ClusterIP Service that fronts the felis-api // internal face (8081). It is SEPARATE from the external NodePort Service (SAAPI) // on purpose — see apiInternalService. The login pod resolves it by cross-namespace // DNS; the on-node break-glass console resolves its ClusterIP and dials it directly. const APIInternalServiceName = SAAPI + "-internal" // OperatorMetricsServiceName is the ClusterIP Service in front of the operator's // /metrics, the scrape target for felis_servers_total, felis_server_phase, // felis_start_duration_seconds and controller-runtime's reconcile series. const OperatorMetricsServiceName = SAOperator + "-metrics" // scrapeAnnotations mark a Service for a Prometheus that discovers targets with // the common prometheus.io/* convention (kubernetes_sd role: endpoints). func scrapeAnnotations(port int32) map[string]string { return map[string]string{ "prometheus.io/scrape": "true", "prometheus.io/port": fmt.Sprint(port), "prometheus.io/path": "/metrics", } } // APIInternalPort is the felis-api internal-face port, exported for the on-node // console which builds http://:APIInternalPort after a Service lookup. const APIInternalPort = apiInternalPort // InternalAPIBaseURL returns the in-cluster base URL of the felis-api INTERNAL // face for a caller in another namespace — specifically the login system server, // which dials it with the service token to mint bind codes and poll link status. // It single-sources the internal Service name (APIInternalServiceName, in the // control namespace) and the internal port with the Deployment/Service above, so a // rename or port change here can never drift from what the login pod is told to // call. Cross-namespace DNS is always resolvable, and apiInternalService actually // programs 8081 on that ClusterIP; reachability additionally depends on there being // no fence in the way (today neither the minecraft-ns egress nor the control-ns // ingress is policy-locked, so the path is open — see internal/platform/netpol.go). func InternalAPIBaseURL(controlNamespace string) string { return fmt.Sprintf("http://%s.%s.svc.cluster.local:%d", APIInternalServiceName, controlNamespace, apiInternalPort) } // Workloads renders the running control-plane: the database (Deployment + // Service, postgres.go), the felis-api Deployment, the felis-operator // Deployment, and the in-cluster registry (Deployment + Service + PVC), the // world-archive PVC when p.BackupPVC names it (it backs the // backup/restore Jobs and the reaper), plus the reaper CronJob when // reaperEnabled(p). Every pod template carries the built-in // system-cluster-critical PriorityClass (controlPlanePriorityName), the // node-pressure eviction shield. FelisImage is required — `felis manifests` // enforces it (fail-loud), so a rendered bundle always names a concrete image. func Workloads(p Params) []Object { p = p.withDefaults() objs := []Object{ postgresDeployment(p), postgresService(p), APIDeployment(p), apiService(p), apiInternalService(p), OperatorDeployment(p), operatorMetricsService(p), registryDeployment(p), registryService(p), registryPVC(p), uploadsPVC(p), } if p.BackupPVC != "" { objs = append(objs, backupPVC(p)) } switch { case p.Distributed && retentionEnabled(p): objs = append(objs, reaperCronJob(p)) case reaperEnabled(p): objs = append(objs, worldsRootPV(p), worldsRootPVC(p), reaperCronJob(p)) case retentionEnabled(p): objs = append(objs, reaperCronJob(p)) } if p.Distributed { objs = append(objs, ArchiveDeployment(p), ArchiveService(p)) } return objs } // controlPlanePriorityName is the PriorityClass every control-plane workload runs // under (api, operator, reaper, registry): the BUILT-IN system-cluster-critical, // not a class we render. Two live findings force it: // // - ordering alone is not enough. With the disk at k3s's eviction threshold // (nodefs.available<5%) a drill watched the kubelet evict the game pods first // — and then the api, operator and registry too, because evicting them was // never what would reclaim the disk. The images live only in the node's // containerd (air-gapped), so once their pods were gone the kubelet GC'd them // and recovery degenerated into ImagePullBackOff plus a manual re-import. // - a class we render cannot reach the critical threshold: user-defined // PriorityClasses are capped at 1e9, while kubelet's // `IsCriticalPodBasedOnPriority` — and with it the eviction manager's // "cannot evict a critical pod" refusal — needs ≥ 2e9. Only the built-in // classes carry those values. // // system-cluster-critical keeps the panel/console alive through a full-disk // incident (which is what lets an operator see it and act) and makes the control // plane schedule before game pods after a reboot. Its preemptionPolicy is // PreemptLowerPriority (built-in classes fix it; the API rejects a per-pod // override), so a control-plane pod that cannot fit MAY preempt a game pod: // accepted deliberately — the management plane must be placeable, and the // alternative is an unreachable box. Game pods stay at the default 0 and are // still the first casualties of node pressure. const controlPlanePriorityName = "system-cluster-critical" // reaperEnabled reports whether the retention CronJob should render. It needs all // three storage coordinates: WorldsHostPath (where worlds live, mounted to read // them), BackupPVC (where archives are written) and ArchiveLocalPath (the mount // path that must equal [archive] local_path so absolute archive refs resolve). // Any missing ⇒ no CronJob (fail-safe). `felis manifests` enforces the trio // together so a partial configuration fails loudly rather than silently dropping // retention here. func reaperEnabled(p Params) bool { return (p.WorldsHostPath != "" || p.Distributed) && retentionEnabled(p) } // retentionEnabled reports whether the archive store can be looked after: the // backup PVC and the path it must be mounted at. Without a worlds root the // same CronJob renders in its retention-only shape (see reaperCronJob), so // backups past their expiry still leave the store on an install that never // reaps worlds; without it they would pile up until the node's disk filled. func retentionEnabled(p Params) bool { return p.BackupPVC != "" && p.ArchiveLocalPath != "" } // APIDeployment renders the felis-api Deployment (spec §7). It runs as the // felis-api SA (so APIMinecraftRole/APIBuildRole bind to a real workload) and // carries controlPlanePodLabels(api), which the allow-rcon NetworkPolicy peer // selects — that correspondence lets the console reach server RCON through the // fence and is asserted in workloads_test.go. // // felis.toml is mounted read-only from a Secret (it carries the database URL, a // credential, so it must never be a ConfigMap); each internal caller's token // (naming.CallerTokens) comes from its own Secret by reference. FELIS_IMAGE is the felis image itself, so the // restore executor launches `felis restore` with the same image. FELIS_BACKUP_PVC // is rendered only when a backup PVC is named — otherwise the restore endpoint // degrades to 503 rather than enqueuing a Job that cannot mount its backup. // // The uploads PVC is mounted read-write at UploadsLocalPath for a local // user_uploads_context, and the two optional S3 credential env vars // (UploadsS3*Env, from the felis-uploads-s3 Secret) feed an s3:// one — the two // storage backends the setup wizard chooses between. func APIDeployment(p Params) *appsv1.Deployment { p = p.withDefaults() var env []corev1.EnvVar // Every internal caller's token, each from its own Secret. Required: the // installer applies all four before this Deployment, and a missing one should // stall the rollout on the old pods rather than start an api that turns that // caller away. for _, ct := range naming.CallerTokens { env = append(env, corev1.EnvVar{ Name: ct.APIEnv, ValueFrom: &corev1.EnvVarSource{ SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: ct.Secret}, Key: serviceTokenSecretKey, }, }, }) } env = append(env, corev1.EnvVar{Name: "FELIS_IMAGE", Value: p.FelisImage}, // Passkey origins must include the same public port the Service exposes. corev1.EnvVar{Name: "FELIS_PANEL_NODEPORT", Value: fmt.Sprint(p.PanelNodePort)}, // The api's own internal-face base URL, so it derives the submission // context URLs that build Pods fetch through it. Same value the login gate // is handed; one address for one face. corev1.EnvVar{Name: naming.EnvAPIBaseURL, Value: InternalAPIBaseURL(p.ControlNamespace)}, ) if p.BackupPVC != "" { env = append(env, corev1.EnvVar{Name: "FELIS_BACKUP_PVC", Value: p.BackupPVC}) } // S3 credentials for an s3:// user_uploads_context, sourced from the // felis-uploads-s3 Secret. Optional: a local-storage install has no such Secret, // and marking these optional lets the pod start anyway (the store falls back to // the uploads PVC). The setup wizard creates the Secret and rolls the API when // the operator picks S3 storage. optional := boolPtr(true) env = append(env, corev1.EnvVar{Name: UploadsS3AccessKeyEnv, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: UploadsS3SecretName}, Key: UploadsS3SecretAccessKey, Optional: optional, }}}, corev1.EnvVar{Name: UploadsS3SecretKeyEnv, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: UploadsS3SecretName}, Key: UploadsS3SecretSecretKey, Optional: optional, }}}, // The [smtp] relay password, same optional-Secret pattern: absent until the // setup wizard's "configure email" step creates felis-smtp. corev1.EnvVar{Name: SMTPPasswordEnv, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: SMTPSecretName}, Key: SMTPSecretPasswordKey, Optional: optional, }}}, corev1.EnvVar{Name: RegistryPruneTokenEnv, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: naming.RegistryAuthSecretName}, Key: registryPruneTokenKey, Optional: optional, }}}, ) container := corev1.Container{ Name: ComponentAPI, Image: p.FelisImage, Command: []string{felisBinaryPath, "api"}, Args: []string{ "--config", configFilePath, "--internal-addr", fmt.Sprintf(":%d", apiInternalPort), "--https-addr", fmt.Sprintf(":%d", apiHTTPSPort), "--tls-cert", apiTLSMountPath + "/tls.crt", "--tls-key", apiTLSMountPath + "/tls.key", }, Env: env, Ports: []corev1.ContainerPort{ {Name: "external", ContainerPort: apiExternalPort, Protocol: corev1.ProtocolTCP}, {Name: "https", ContainerPort: apiHTTPSPort, Protocol: corev1.ProtocolTCP}, {Name: "internal", ContainerPort: apiInternalPort, Protocol: corev1.ProtocolTCP}, }, VolumeMounts: []corev1.VolumeMount{ {Name: configVolume, MountPath: configMountPath, ReadOnly: true}, {Name: "tls", MountPath: apiTLSMountPath, ReadOnly: true}, {Name: tmpVolume, MountPath: "/tmp"}, // Read-WRITE: local uploads land here (an s3:// store bypasses it). The // hardened container root is read-only, so this PVC mount is where a local // LocalContextStore can persist a submitted context. {Name: uploadsVolume, MountPath: UploadsLocalPath}, }, // Probes dial the INTERNAL face (8081), the only listener carrying both // /healthz and /readyz (the external face deliberately serves liveness // only), and the face kubelet can reach without any Zero Trust hop. // Readiness = /readyz (DB + K8s API round-trip): a not-ready answer only // pulls the pod out of Service endpoints. Liveness = the cheap /healthz — // pointing it at /readyz would restart the api on every DB blip. ReadinessProbe: &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{ Path: "/readyz", Port: intstr.FromInt32(apiInternalPort), }}, InitialDelaySeconds: 5, PeriodSeconds: 10, TimeoutSeconds: 3, FailureThreshold: 3, }, LivenessProbe: &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{ Path: "/healthz", Port: intstr.FromInt32(apiInternalPort), }}, InitialDelaySeconds: 10, PeriodSeconds: 10, TimeoutSeconds: 3, FailureThreshold: 3, }, Resources: controlPlaneResources(), SecurityContext: hardenedContainerSecurityContext(), } volumes := []corev1.Volume{ { Name: configVolume, VolumeSource: corev1.VolumeSource{ Secret: &corev1.SecretVolumeSource{SecretName: configSecretName}, }, }, { Name: "tls", VolumeSource: corev1.VolumeSource{ Secret: &corev1.SecretVolumeSource{SecretName: apiTLSSecretName}, }, }, {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, { Name: uploadsVolume, VolumeSource: corev1.VolumeSource{ PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: uploadsPVCName}, }, }, } return controlPlaneDeployment(p, SAAPI, container, volumes) } // apiService exposes the built-in HTTPS panel/API origin as a stable NodePort. func apiService(p Params) *corev1.Service { p = p.withDefaults() labels := controlPlanePodLabels(ComponentAPI) return &corev1.Service{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Service"}, ObjectMeta: metav1.ObjectMeta{Name: SAAPI, Namespace: p.ControlNamespace, Labels: labels}, Spec: corev1.ServiceSpec{ Type: corev1.ServiceTypeNodePort, Selector: labels, Ports: []corev1.ServicePort{{ Name: "https", Port: 443, TargetPort: intstr.FromString("https"), NodePort: p.PanelNodePort, Protocol: corev1.ProtocolTCP, }}, }, } } // apiInternalService fronts the felis-api INTERNAL face (service-token, no Zero // Trust) on a ClusterIP-only Service, kept SEPARATE from the external NodePort // apiService on purpose: a NodePort Service allocates a node port for EVERY declared // port with no per-port opt-out, so folding 8081 into apiService would publish the // no-Zero-Trust internal face on every node's external IP — a hard red line for a // face whose only guard is the bearer service token. A distinct ClusterIP Service // exposes 8081 in-cluster only: reachable by the login pod (cross-namespace DNS to // APIInternalServiceName) and, on the k3s node, by the break-glass console dialing // this Service's ClusterIP. Without it the felis-api DNS name has no 8081 port and // every internal-face call silently fails to connect. func apiInternalService(p Params) *corev1.Service { p = p.withDefaults() labels := controlPlanePodLabels(ComponentAPI) return &corev1.Service{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Service"}, // The internal face also serves /metrics (felis-api's felis_* series). ObjectMeta: metav1.ObjectMeta{Name: APIInternalServiceName, Namespace: p.ControlNamespace, Labels: labels, Annotations: scrapeAnnotations(apiInternalPort)}, Spec: corev1.ServiceSpec{ Type: corev1.ServiceTypeClusterIP, Selector: labels, Ports: []corev1.ServicePort{{ Name: "internal", Port: apiInternalPort, TargetPort: intstr.FromString("internal"), Protocol: corev1.ProtocolTCP, }}, }, } } // operatorMetricsService gives the operator's metrics listener a stable scrape // target. ClusterIP only: /metrics is unauthenticated, like felis-api's. func operatorMetricsService(p Params) *corev1.Service { p = p.withDefaults() labels := controlPlanePodLabels(ComponentOperator) return &corev1.Service{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Service"}, ObjectMeta: metav1.ObjectMeta{Name: OperatorMetricsServiceName, Namespace: p.ControlNamespace, Labels: labels, Annotations: scrapeAnnotations(operatorMetricsPort)}, Spec: corev1.ServiceSpec{ Type: corev1.ServiceTypeClusterIP, Selector: labels, Ports: []corev1.ServicePort{{ Name: "metrics", Port: operatorMetricsPort, TargetPort: intstr.FromString("metrics"), Protocol: corev1.ProtocolTCP, }}, }, } } // OperatorDeployment renders the felis-operator Deployment (spec §5). It runs as // the felis-operator SA and carries controlPlanePodLabels(operator), the second // pod the allow-rcon peer admits (the readiness prober dials RCON). It takes NO // config Secret: the operator reads everything from flags + the in-cluster API, // so it never loads the database URL — a deliberately smaller attack surface than // the api. Its secrets:get in the minecraft namespace is by name and uncached // (no list, no informer), yet namespaced RBAC cannot exclude a name, so a // compromised operator could still fetch the felis-config mirror the Jobs and // the reaper mount there. It watches the minecraft namespace (--namespace) while running in the // control namespace, exactly the split cmd/felis/operator.go documents. func OperatorDeployment(p Params) *appsv1.Deployment { p = p.withDefaults() container := corev1.Container{ Name: ComponentOperator, Image: p.FelisImage, Command: []string{felisBinaryPath, "operator"}, Args: []string{ "--namespace", p.MinecraftNamespace, "--metrics-bind-address", fmt.Sprintf(":%d", operatorMetricsPort), "--health-probe-bind-address", fmt.Sprintf(":%d", operatorHealthPort), }, // FELIS_IMAGE names this same image so the operator can run it as the // forwarding-config initContainer it injects into user servers (it must // name an image to run, and its own is the one image guaranteed present). Env: []corev1.EnvVar{ {Name: "FELIS_IMAGE", Value: p.FelisImage}, }, Ports: []corev1.ContainerPort{ {Name: "metrics", ContainerPort: operatorMetricsPort, Protocol: corev1.ProtocolTCP}, {Name: "health", ContainerPort: operatorHealthPort, Protocol: corev1.ProtocolTCP}, }, VolumeMounts: []corev1.VolumeMount{ {Name: tmpVolume, MountPath: "/tmp"}, }, // controller-runtime serves /healthz and /readyz on the health listener // (cmd/felis/operator.go): /healthz fails while a reconcile pass is stuck, // so liveness restarts a wedged operator; /readyz waits for the informer // caches. Neither checks a dependency, so an API blip restarts nothing. ReadinessProbe: &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{ Path: "/readyz", Port: intstr.FromInt32(operatorHealthPort), }}, InitialDelaySeconds: 5, PeriodSeconds: 10, TimeoutSeconds: 3, FailureThreshold: 3, }, LivenessProbe: &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{ Path: "/healthz", Port: intstr.FromInt32(operatorHealthPort), }}, InitialDelaySeconds: 10, PeriodSeconds: 10, TimeoutSeconds: 3, FailureThreshold: 3, }, Resources: controlPlaneResources(), SecurityContext: hardenedContainerSecurityContext(), } volumes := []corev1.Volume{ {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, } return controlPlaneDeployment(p, SAOperator, container, volumes) } // reaperCronJob renders the world-retention CronJob (spec §18). It runs the // `felis reaper` run-once entrypoint on a daily cadence — the CronJob, not the // process, owns scheduling, so the reaper stays idempotent and restart-safe. // // Identity & reach. It runs as SAReaper (felis-reaper), the only SA holding // minecraftservers:[get,patch] + persistentvolumeclaims:delete in the minecraft // namespace (rbac.go ReaperRole). Unlike the build/restore/registry pods — which // set AutomountServiceAccountToken=false because they never call the K8s API — the // reaper LEGITIMATELY patches MinecraftServers (flip desiredState) and deletes // world PVCs, so its SA token is left to auto-mount (nil). Its pod carries // controlPlanePodLabels(ComponentReaper); reaper is deliberately excluded from the // RCON NetworkPolicy peer (component ∉ {api,operator}), asserted in // workloads_test.go — the reaper never opens an RCON connection. // // Mounts (the storage crux). felis.toml is mounted read-only from the config // Secret (it carries the DB URL). The worlds-root is the node directory behind the // static worldsRootPV, claimed by worldsRootPVC and mounted READ-ONLY at /worlds // (why a PV rather than an inline hostPath: see worldsRootPV). The reaper only // reads worlds to tar them; deleting a world // is a K8s API call (DeletePVC), never an rm, so the mount never needs write. The // backup PVC is mounted READ-WRITE at p.ArchiveLocalPath — which MUST equal // felis.toml [archive] local_path, because tarLocal writes archive refs as absolute // paths under it and the restore Job later mounts the same PVC at the same path to // resolve them (see the ArchiveLocalPath field doc). A /tmp emptyDir absorbs writes // under the read-only root filesystem. The PV's hostPath type Directory fails the // pod loud if the worlds-root is absent, rather than silently creating an empty dir // and archiving nothing. // // Retention only. With no WorldsHostPath the same CronJob runs `felis reaper // --retention-only`: it expires, reads back and sweeps the archive store and // never looks at a server. It then mounts no worlds root, reads no SMTP // password (it sends no warnings) and gets no service account token, since it // never calls the K8s API. It keeps the reaper's root + DAC_OVERRIDE identity: // the archives it reads back and deletes were written by that identity, and by // the backup Jobs. // // Pre-conditions are the caller's: reaperCronJob assumes retentionEnabled(p) — // it dereferences BackupPVC / ArchiveLocalPath without re-checking. func reaperCronJob(p Params) *batchv1.CronJob { p = p.withDefaults() labels := controlPlanePodLabels(ComponentReaper) container := corev1.Container{ Name: ComponentReaper, Image: p.FelisImage, Command: []string{felisBinaryPath, "reaper"}, Args: []string{"--config", configFilePath, "--retention-only"}, VolumeMounts: []corev1.VolumeMount{ {Name: configVolume, MountPath: configMountPath, ReadOnly: true}, {Name: backupVolume, MountPath: p.ArchiveLocalPath}, {Name: tmpVolume, MountPath: "/tmp"}, }, Resources: controlPlaneResources(), SecurityContext: reaperContainerSecurityContext(), } volumes := []corev1.Volume{ { Name: configVolume, VolumeSource: corev1.VolumeSource{ Secret: &corev1.SecretVolumeSource{SecretName: configSecretName}, }, }, { Name: backupVolume, VolumeSource: corev1.VolumeSource{ PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: p.BackupPVC}, }, }, {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, } if p.WorldsHostPath != "" { container.Args = []string{"--config", configFilePath, "--worlds-root", worldsMountPath} // The [smtp] relay password for pre-reap warning emails — same optional // Secret felis-api reads. Namespace caveat: a secretKeyRef is // namespace-local, so this resolves against the minecraft-ns felis-smtp // mirror that the "configure email" screen refreshes (the felis-config // mirror it also refreshes is what puts [smtp] in this pod's config). // Absent Secret ⇒ empty env ⇒ the reaper logs suppressed warnings // instead of stamping them (never a failed pod). container.Env = []corev1.EnvVar{ {Name: SMTPPasswordEnv, ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{ LocalObjectReference: corev1.LocalObjectReference{Name: SMTPSecretName}, Key: SMTPSecretPasswordKey, Optional: boolPtr(true), }}}, } container.VolumeMounts = append(container.VolumeMounts, corev1.VolumeMount{Name: worldsVolume, MountPath: worldsMountPath, ReadOnly: true}) volumes = append(volumes, corev1.Volume{ Name: worldsVolume, VolumeSource: corev1.VolumeSource{ PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ ClaimName: worldsRootName(p), ReadOnly: true, }, }, }) } if p.Distributed { container.Args = []string{"--config", configFilePath} container.Env = append(container.Env, distributedEnv(p)...) container.Env = append(container.Env, corev1.EnvVar{Name: "FELIS_IMAGE", Value: p.FelisImage}) } return &batchv1.CronJob{ TypeMeta: metav1.TypeMeta{APIVersion: "batch/v1", Kind: "CronJob"}, // The CronJob lives in the MINECRAFT namespace: a Pod can only mount PVCs // from its own namespace and the backup PVC is provisioned there alongside // the backup Jobs. Placed under ControlNamespace it could never schedule // (FailedScheduling: persistentvolumeclaim not found) in any stock install; // the reaper Role/RoleBinding were already minecraft-scoped for the same // reason, and the minecraft felis-config replica (felis setup) supplies the // config mount. ObjectMeta: metav1.ObjectMeta{Name: SAReaper, Namespace: p.MinecraftNamespace, Labels: labels}, Spec: batchv1.CronJobSpec{ Schedule: reaperSchedule, ConcurrencyPolicy: batchv1.ForbidConcurrent, StartingDeadlineSeconds: int64Ptr(reaperStartingDeadlineSeconds), SuccessfulJobsHistoryLimit: int32Ptr(reaperHistoryLimit), FailedJobsHistoryLimit: int32Ptr(reaperHistoryLimit), JobTemplate: batchv1.JobTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: labels}, Spec: batchv1.JobSpec{ BackoffLimit: int32Ptr(reaperBackoffLimit), ActiveDeadlineSeconds: int64Ptr(reaperActiveDeadlineSeconds), Template: corev1.PodTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: labels}, Spec: reaperPodSpec(p, container, volumes), }, }, }, }, } } // worldsRootName names the static PV and its PVC. The PV's hostPath and node // affinity are immutable once created, so the name carries a digest of them: a // re-install that moves the worlds-root renders a fresh pair (and the reaper // follows it) instead of an apply the API server would refuse. The namespace is in // the digest because a PV is cluster-scoped and two installs must not collide. func worldsRootName(p Params) string { sum := sha256.Sum256([]byte(p.MinecraftNamespace + "\x00" + p.WorldsHostPath + "\x00" + p.ReaperNode)) return "felis-worlds-root-" + hex.EncodeToString(sum[:4]) } // worldsRootStorage is the nominal size both halves of the static pair declare. A // hostPath volume enforces no quota; the PVC only has to request no more than the // PV offers for the two to bind. const worldsRootStorage = "1Gi" // worldsRootPV exposes the node's worlds-root to the reaper as a pre-bound static // PersistentVolume. The reaper's pod then mounts a PVC, not a hostPath, and that is // what lets the minecraft namespace enforce the PodSecurity baseline profile (see // minecraftPodSecurityLabels): baseline refuses any pod with an inline hostPath // volume, whoever submits it. The node path is still reachable, but only through // this PV, which is cluster-scoped and so something only a cluster admin — the // installer applying this bundle — can create. A game server, file-editor or // backup pod in the namespace cannot name a node path of its own. // // Retain keeps a deleted claim from ever handing the path to a reclaimer, the // empty storageClassName keeps the default provisioner out of it, and the claimRef // reserves it for exactly worldsRootPVC. With ReaperNode set the PV carries the // same node pin as the reaper pod, since the directory exists only on that node. func worldsRootPV(p Params) *corev1.PersistentVolume { p = p.withDefaults() name := worldsRootName(p) hostPathDir := corev1.HostPathDirectory pv := &corev1.PersistentVolume{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "PersistentVolume"}, ObjectMeta: metav1.ObjectMeta{Name: name, Labels: controlPlanePodLabels(ComponentReaper)}, Spec: corev1.PersistentVolumeSpec{ Capacity: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse(worldsRootStorage)}, AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, PersistentVolumeReclaimPolicy: corev1.PersistentVolumeReclaimRetain, StorageClassName: "", ClaimRef: &corev1.ObjectReference{Namespace: p.MinecraftNamespace, Name: name}, PersistentVolumeSource: corev1.PersistentVolumeSource{ HostPath: &corev1.HostPathVolumeSource{Path: p.WorldsHostPath, Type: &hostPathDir}, }, }, } if p.ReaperNode != "" { pv.Spec.NodeAffinity = &corev1.VolumeNodeAffinity{Required: &corev1.NodeSelector{ NodeSelectorTerms: []corev1.NodeSelectorTerm{{MatchExpressions: []corev1.NodeSelectorRequirement{{ Key: "kubernetes.io/hostname", Operator: corev1.NodeSelectorOpIn, Values: []string{p.ReaperNode}, }}}}, }} } return pv } // worldsRootPVC is the reaper's claim on worldsRootPV, bound by name both ways. func worldsRootPVC(p Params) *corev1.PersistentVolumeClaim { p = p.withDefaults() name := worldsRootName(p) noClass := "" return &corev1.PersistentVolumeClaim{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "PersistentVolumeClaim"}, ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: p.MinecraftNamespace, Labels: controlPlanePodLabels(ComponentReaper)}, Spec: corev1.PersistentVolumeClaimSpec{ AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, StorageClassName: &noClass, VolumeName: name, Resources: corev1.VolumeResourceRequirements{ Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse(worldsRootStorage)}, }, }, } } // reaperPodSpec is the reaper Job's pod template. It lives apart from the CronJob // literal only so the optional node pin is one visible branch: with ReaperNode // set the pod carries a kubernetes.io/hostname selector, keeping the reaper on // the node that actually holds the worlds hostPath on a multi-node cluster. // // Only the reaping shape runs as SAReaper. The retention-only shape never calls the // API, so it gets no token and runs as the namespace's default ServiceAccount: // felis-reaper is rendered only alongside its Role (rbac.go), and a pod naming a // ServiceAccount that does not exist is never created. Every stock install renders // this shape, and its Jobs used to hit their deadline without a single pod. // // "default" is spelled out. An install upgraded from the broken shape still carries // the deprecated serviceAccount: felis-reaper the API server copied into the // template, and an empty serviceAccountName is filled back in from it. func reaperPodSpec(p Params, container corev1.Container, volumes []corev1.Volume) corev1.PodSpec { spec := corev1.PodSpec{ PriorityClassName: controlPlanePriorityName, RestartPolicy: corev1.RestartPolicyNever, SecurityContext: reaperPodSecurityContext(), Containers: []corev1.Container{container}, Volumes: volumes, } if p.ControllerNode != "" { spec.NodeSelector = controllerSelector(p) } else if p.ReaperNode != "" { spec.NodeSelector = map[string]string{"kubernetes.io/hostname": p.ReaperNode} } if p.WorldsHostPath != "" || p.Distributed { spec.ServiceAccountName = SAReaper } else { spec.ServiceAccountName = "default" spec.AutomountServiceAccountToken = boolPtr(false) } return spec } // controlPlaneDeployment assembles a single-replica control-plane Deployment. The // Deployment, its selector, and the pod template all carry // controlPlanePodLabels(component) so the three agree (a selector mismatch would // leave pods unmanaged). The component is taken from the container name, which is // the component value for both control-plane workloads. // // Strategy is Recreate, not RollingUpdate: neither the operator nor the api wires // leader election, so a RollingUpdate's maxSurge overlap would briefly run two // instances — two reconcilers fighting, or two processes binding the same ports. // Recreate guarantees the old pod is gone before the new one starts. func controlPlaneDeployment(p Params, sa string, container corev1.Container, volumes []corev1.Volume) *appsv1.Deployment { labels := controlPlanePodLabels(container.Name) if p.Distributed { if sa == SAOperator { container.Env = append(container.Env, corev1.EnvVar{Name: "FELIS_DISTRIBUTED", Value: "true"}, corev1.EnvVar{Name: "FELIS_CONTROLLER_NODE", Value: p.ControllerNode}, corev1.EnvVar{Name: "FELIS_EGRESS_PROBE", Value: p.EgressProbe}) } else { container.Env = append(container.Env, distributedEnv(p)...) } } return &appsv1.Deployment{ TypeMeta: metav1.TypeMeta{APIVersion: "apps/v1", Kind: "Deployment"}, ObjectMeta: metav1.ObjectMeta{Name: sa, Namespace: p.ControlNamespace, Labels: labels}, Spec: appsv1.DeploymentSpec{ Replicas: int32Ptr(1), Strategy: appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}, Selector: &metav1.LabelSelector{MatchLabels: labels}, Template: corev1.PodTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: labels}, Spec: corev1.PodSpec{ NodeSelector: controllerSelector(p), ServiceAccountName: sa, PriorityClassName: controlPlanePriorityName, SecurityContext: hardenedPodSecurityContext(), Containers: []corev1.Container{container}, Volumes: volumes, }, }, }, } } // registryDeployment renders the in-cluster Distribution registry (spec §16). // Build Jobs push to registry..svc:, the destination the build // egress NetworkPolicy opens — so this Deployment+Service+PVC make that policy // target real. The registry never calls the K8s API, so its token auto-mount is // disabled (matching the weak build/restore SA hygiene). // // The pod has three containers. registry:2 itself has no auth and listens on the // pod's loopback only (registryUpstreamPort), so nothing outside the pod can reach // it directly. The gate sidecar (felis registry-gate, internal/registrygate) owns // the registry port: reads pass anonymously, writes need the platform or build // credential from the registry-auth Secret, and the build credential cannot touch // the platform's own repositories. Before the gate any pod that could reach the // registry could overwrite felis/felis. The registry-gc sidecar reclaims the // blobs of deleted manifests inside a read-only window the gate grants // (registryGCScript). // // The gate's port also carries a loopback hostPort (registryLoopbackHost): it is // the node-side pull path. The node's containerd cannot dial the Service VIP, so // deploy/bootstrap.sh writes a registries.yaml mirror rewriting // registry..svc: onto http://127.0.0.1:, and that request // arrives at this hostPort — which is what lets kubelet re-pull a garbage-collected // platform image without an operator re-import. The gate runs the felis image, so // the installer pins that image in containerd: the registry must never depend on // pulling its own gate from itself. func registryDeployment(p Params) *appsv1.Deployment { p = p.withDefaults() labels := registryLabels() upstreamPort := registryUpstreamPort(p) registry := corev1.Container{ Name: registryName, Image: p.RegistryImage, Env: []corev1.EnvVar{ {Name: "REGISTRY_HTTP_ADDR", Value: fmt.Sprintf("127.0.0.1:%d", upstreamPort)}, // Manifest DELETE is how felis-api's pruner releases an image; the gate // lets only the prune and platform principals send it. {Name: "REGISTRY_STORAGE_DELETE_ENABLED", Value: "true"}, // The in-memory blob descriptor cache outlives a garbage-collect run: a // blob the sweep deleted would still answer HEAD, a push would skip // uploading it, and the manifest pushed after it would name a blob that // is gone. Any value other than inmemory/redis turns the cache off // (registry 2.8 logs "unknown cache type ... caching disabled"). {Name: "REGISTRY_STORAGE_CACHE_BLOBDESCRIPTOR", Value: "none"}, }, VolumeMounts: []corev1.VolumeMount{ {Name: registryVolume, MountPath: registryDataPath}, {Name: tmpVolume, MountPath: "/tmp"}, }, Resources: registryResources(), SecurityContext: hardenedContainerSecurityContext(), } gate := corev1.Container{ Name: registryGateName, Image: p.FelisImage, Command: []string{felisBinaryPath, "registry-gate"}, Args: []string{ fmt.Sprintf("--listen=:%d", p.RegistryPort), fmt.Sprintf("--upstream=http://127.0.0.1:%d", upstreamPort), "--auth-dir=" + registryAuthMountPath, fmt.Sprintf("--maint-listen=127.0.0.1:%d", registryMaintPort(p)), "--maint-dir=" + registryMaintMountPath, // The manifest index felis-api's pruner reads (registrygate/index.go). "--data-dir=" + registryDataPath, }, Ports: []corev1.ContainerPort{ { Name: registryName, ContainerPort: p.RegistryPort, Protocol: corev1.ProtocolTCP, // The node-side pull path: containerd's mirror rewrites the Service // name onto 127.0.0.1: (see the doc comment above). HostPort: p.RegistryPort, HostIP: registryLoopbackHost, }, }, VolumeMounts: []corev1.VolumeMount{ {Name: registryAuthVolume, MountPath: registryAuthMountPath, ReadOnly: true}, {Name: registryMaintVolume, MountPath: registryMaintMountPath}, {Name: registryVolume, MountPath: registryDataPath, ReadOnly: true}, }, // /healthz answers 200 only while registry:2 answers GET /v2/ on loopback, // so a registry whose storage broke shows up as an unready pod instead of a // "Running" one every push fails against. Liveness checks the gate alone: a // failing registry is not fixed by restarting the gate in front of it. ReadinessProbe: &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{ Path: "/healthz", Port: intstr.FromString(registryName), }}, InitialDelaySeconds: 5, PeriodSeconds: 10, TimeoutSeconds: 5, FailureThreshold: 3, }, LivenessProbe: &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{ Path: "/livez", Port: intstr.FromString(registryName), }}, InitialDelaySeconds: 10, PeriodSeconds: 10, TimeoutSeconds: 3, FailureThreshold: 3, }, Resources: registryGateResources(), SecurityContext: hardenedContainerSecurityContext(), } return &appsv1.Deployment{ TypeMeta: metav1.TypeMeta{APIVersion: "apps/v1", Kind: "Deployment"}, ObjectMeta: metav1.ObjectMeta{Name: registryName, Namespace: p.RegistryNamespace, Labels: labels}, Spec: appsv1.DeploymentSpec{ Replicas: int32Ptr(1), Strategy: appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}, Selector: &metav1.LabelSelector{MatchLabels: labels}, Template: corev1.PodTemplateSpec{ ObjectMeta: metav1.ObjectMeta{Labels: labels}, Spec: corev1.PodSpec{ NodeSelector: controllerSelector(p), AutomountServiceAccountToken: boolPtr(false), PriorityClassName: controlPlanePriorityName, SecurityContext: hardenedPodSecurityContext(), Containers: []corev1.Container{registry, gate, registryGCContainer(p)}, Volumes: []corev1.Volume{ { Name: registryVolume, VolumeSource: corev1.VolumeSource{ PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: registryName}, }, }, {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, {Name: registryMaintVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, { Name: registryAuthVolume, VolumeSource: corev1.VolumeSource{Secret: &corev1.SecretVolumeSource{ SecretName: naming.RegistryAuthSecretName, // Optional: without the Secret the gate starts with no // principals, so pulls keep working and every write is // refused — never a registry that cannot start. Optional: boolPtr(true), DefaultMode: int32Ptr(0o440), }}, }, }, }, }, }, } } // registryUpstreamPort is the loopback port registry:2 listens on behind the gate: // the next port after the public one. func registryUpstreamPort(p Params) int32 { return p.RegistryPort + 1 } // registryMaintPort is the gate's loopback-only maintenance listener, the one // after the upstream port. func registryMaintPort(p Params) int32 { return p.RegistryPort + 2 } // registryGCScript is the registry-gc sidecar's loop. Once per interval (the last // run is stamped on the data volume, so a pod restart does not reset the clock) it // asks the gate for a read-only window, waiting up to 30 minutes for pushes to go // quiet, runs registry garbage-collect, and hands the window back. The window is // a lease, so a sidecar killed mid-sweep leaves the registry writable again within // the hour. // // No --delete-untagged: a server's spec pins its image by digest, and the tag it // was created from moves with every rebuild, so an untagged manifest may be the // exact build a sleeping server boots. Manifests go only when felis-api's pruner // has found no whitelist entry, server or recent build naming them and deleted // them; this sweep then frees the blobs nothing references any more. const registryGCScript = `set -u # sh as PID 1 ignores SIGTERM unless it traps it, and runs the trap only once the # foreground child exits: sleep in the background and wait, so a pod delete does # not sit out the grace period. trap 'exit 0' TERM nap() { sleep "$1" & wait $!; } maint="http://127.0.0.1:${FELIS_GC_MAINT_PORT}" stamp=/var/lib/registry/.felis-last-gc while :; do now=$(date +%s) last=$(cat "$stamp" 2>/dev/null || echo 0) case "$last" in ''|*[!0-9]*) last=0 ;; esac if [ $((now - last)) -ge "$FELIS_GC_INTERVAL_SECONDS" ]; then tries=0 until wget -q -O /dev/null --post-data '' "$maint/readonly?lease=3600"; do tries=$((tries + 1)) [ "$tries" -ge 180 ] && break nap 10 done if [ "$tries" -lt 180 ]; then echo "felis-gc: registry is read-only; collecting" if registry garbage-collect /etc/docker/registry/config.yml > /tmp/gc.log 2>&1; then date +%s > "$stamp" echo "felis-gc: done ($(grep -c 'blob eligible for deletion' /tmp/gc.log) blob(s) deleted)" else echo "felis-gc: garbage-collect failed:" >&2 tail -n 20 /tmp/gc.log >&2 fi wget -q -O /dev/null --post-data '' "$maint/readwrite" \ || echo "felis-gc: could not hand the read-only window back; it lapses with its lease" >&2 else echo "felis-gc: pushes never went quiet for 30 minutes; trying again later" >&2 fi fi nap 600 done ` // registryGCContainer renders the garbage-collection sidecar. It runs the // registry image (garbage-collect is a subcommand of the registry binary) against // the same data volume, under the same non-root identity and read-only root. func registryGCContainer(p Params) corev1.Container { return corev1.Container{ Name: registryGCName, Image: p.RegistryImage, Command: []string{"/bin/sh", "-c", registryGCScript}, Env: []corev1.EnvVar{ {Name: "FELIS_GC_MAINT_PORT", Value: fmt.Sprint(registryMaintPort(p))}, {Name: "FELIS_GC_INTERVAL_SECONDS", Value: fmt.Sprint(int64(registryGCInterval / time.Second))}, }, VolumeMounts: []corev1.VolumeMount{ {Name: registryVolume, MountPath: registryDataPath}, {Name: tmpVolume, MountPath: "/tmp"}, }, Resources: corev1.ResourceRequirements{ Requests: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("10m"), corev1.ResourceMemory: resource.MustParse("16Mi"), }, Limits: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("500m"), corev1.ResourceMemory: resource.MustParse("512Mi"), }, }, SecurityContext: hardenedContainerSecurityContext(), } } // registryGateResources sizes the gate sidecar: a streaming reverse proxy that // holds no layer in memory. func registryGateResources() corev1.ResourceRequirements { return corev1.ResourceRequirements{ Requests: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("20m"), corev1.ResourceMemory: resource.MustParse("32Mi"), }, Limits: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("500m"), corev1.ResourceMemory: resource.MustParse("128Mi"), }, } } // registryService renders the ClusterIP Service that gives the registry its // pinned DNS name registry..svc: — hardcoded across the build // subsystem and config. Its selector matches the registry pod labels; because // those labels are NOT part-of=felis-control-plane, the registry is invisible to // the RCON NetworkPolicy peer. func registryService(p Params) *corev1.Service { p = p.withDefaults() labels := registryLabels() return &corev1.Service{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Service"}, ObjectMeta: metav1.ObjectMeta{Name: registryName, Namespace: p.RegistryNamespace, Labels: labels}, Spec: corev1.ServiceSpec{ Type: corev1.ServiceTypeClusterIP, Selector: labels, Ports: []corev1.ServicePort{{ Name: registryName, Port: p.RegistryPort, TargetPort: intstr.FromInt32(p.RegistryPort), Protocol: corev1.ProtocolTCP, }}, }, } } // registryPVC renders the registry's data volume (RWO). No storageClassName is // set, so it binds the cluster's default class — pinning one would be a guess. func registryPVC(p Params) *corev1.PersistentVolumeClaim { p = p.withDefaults() return &corev1.PersistentVolumeClaim{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "PersistentVolumeClaim"}, ObjectMeta: metav1.ObjectMeta{Name: registryName, Namespace: p.RegistryNamespace, Labels: registryLabels()}, Spec: corev1.PersistentVolumeClaimSpec{ AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, Resources: corev1.VolumeResourceRequirements{ Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse(p.RegistryStorage)}, }, }, } } // uploadsPVC renders the felis-api user-uploads PVC (spec §16 build-context input // domain). It backs a LOCAL user_uploads_context: felis-api mounts it read-write // at UploadsLocalPath and LocalContextStore writes each submission's context there. // It is always rendered (an s3:// install simply never writes to it) and carries // control-plane labels so it reads as part of felis-api's storage. ReadWriteOnce is // the fail-safe access mode: the single-node starter binds it to felis-api's node, // and a future Kaniko-read integration mounts the same PVC on that node. func uploadsPVC(p Params) *corev1.PersistentVolumeClaim { p = p.withDefaults() return &corev1.PersistentVolumeClaim{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "PersistentVolumeClaim"}, ObjectMeta: metav1.ObjectMeta{Name: uploadsPVCName, Namespace: p.ControlNamespace, Labels: controlPlanePodLabels(ComponentAPI)}, Spec: corev1.PersistentVolumeClaimSpec{ AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, Resources: corev1.VolumeResourceRequirements{ Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse(p.UploadsStorage)}, }, }, } } // backupPVC renders the world-archive store (spec §18/§19): the PVC the backup // and restore Jobs mount read-write, and the one the reaper CronJob writes // archives into. It renders ONLY when p.BackupPVC names it, so the same value // gates the PVC and the FELIS_BACKUP_PVC env on the felis-api Deployment — with // no name there is no PVC, no env, and the backup/restore endpoints keep // answering an honest 503 rather than enqueuing a Job that cannot mount its // backup. It lives in the Minecraft namespace because every pod that mounts it // runs there (a Pod can only mount PVCs from its own namespace; the reaper // CronJob is rendered there for the same reason). No storageClassName: binding // the cluster default is the only safe default. func backupPVC(p Params) *corev1.PersistentVolumeClaim { p = p.withDefaults() return &corev1.PersistentVolumeClaim{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "PersistentVolumeClaim"}, ObjectMeta: metav1.ObjectMeta{Name: p.BackupPVC, Namespace: p.MinecraftNamespace, Labels: controlPlanePodLabels(ComponentAPI)}, Spec: corev1.PersistentVolumeClaimSpec{ AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, Resources: corev1.VolumeResourceRequirements{ Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse(p.BackupStorage)}, }, }, } } // VolumeBinderPod mounts pvc and exits at once. A WaitForFirstConsumer volume // (k3s local-path) has no directory until a pod uses it, and on a rebuilt node // nothing has yet, so `felis offsite fetch-worlds` runs this to have the // archive volume provisioned before it writes the archives back into it. It // runs as the control-plane identity, which the archive volume's files already // belong to (fsGroup). func VolumeBinderPod(ns, pvc, image string) *corev1.Pod { return &corev1.Pod{ TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Pod"}, ObjectMeta: metav1.ObjectMeta{ GenerateName: "felis-bind-" + pvc + "-", Namespace: ns, Labels: map[string]string{LabelName: appName, LabelComponent: "volume-binder"}, }, Spec: corev1.PodSpec{ RestartPolicy: corev1.RestartPolicyNever, AutomountServiceAccountToken: boolPtr(false), SecurityContext: hardenedPodSecurityContext(), Containers: []corev1.Container{{ Name: "bind", Image: image, Command: []string{felisBinaryPath, "version"}, SecurityContext: hardenedContainerSecurityContext(), VolumeMounts: []corev1.VolumeMount{{Name: "archives", MountPath: "/archives"}}, Resources: corev1.ResourceRequirements{ Requests: corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("10m"), corev1.ResourceMemory: resource.MustParse("16Mi")}, Limits: corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("100m"), corev1.ResourceMemory: resource.MustParse("64Mi")}, }, }}, Volumes: []corev1.Volume{{ Name: "archives", VolumeSource: corev1.VolumeSource{PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: pvc}}, }}, }, } } // registryLabels are the registry's recommended labels. Note the absence of // part-of=felis-control-plane: that is what keeps the registry out of the RCON // NetworkPolicy peer's reach (asserted in workloads_test.go). func registryLabels() map[string]string { return map[string]string{ LabelName: appName, LabelComponent: ComponentRegistry, } } // controlPlaneResources are conservative starting requests/limits. They bound // resource exhaustion (a security-hygiene baseline) without claiming to be tuned // for production load — that is a deployment concern. func controlPlaneResources() corev1.ResourceRequirements { return corev1.ResourceRequirements{ Requests: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("50m"), corev1.ResourceMemory: resource.MustParse("64Mi"), }, Limits: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("500m"), corev1.ResourceMemory: resource.MustParse("256Mi"), }, } } // registryResources sizes the registry for what it actually does: it is the pull // source for every image the node runs and the push target of every build, so its // limits are not the control plane's. The memory limit is LOAD-BEARING, not // tuning: a live drill pushing a 475MB layer into a 256Mi registry (the old // control-plane template) OOM-killed the registry mid-upload — dmesg oom-kill, // oom_score_adj 989, the push failed — and the identical push completed in // 2 seconds once the limit was 2Gi. Large layers (Kaniko-built modpacks, the game // images) are exactly that case, so 2Gi is the floor this renderer stands on. func registryResources() corev1.ResourceRequirements { return corev1.ResourceRequirements{ Requests: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("50m"), corev1.ResourceMemory: resource.MustParse("64Mi"), }, Limits: corev1.ResourceList{ corev1.ResourceCPU: resource.MustParse("1"), corev1.ResourceMemory: resource.MustParse("2Gi"), }, } } // hardenedPodSecurityContext is the pod-level hardening shared by the // control-plane workloads (api, operator, registry — the world-touching reaper // uses reaperPodSecurityContext instead): run as a fixed non-root uid/gid with a // matching fsGroup (so the registry can write its group-owned PVC) and the // RuntimeDefault seccomp profile. // // SHAPE-ASSERTED, runtime-unverified: this asserts the images can run as // nonRootUID. The felis image is built to; registry:2 (CNCF Distribution) can, // with the data PVC fsGroup-owned. Without a cluster the actual start-up is not // proven here. func hardenedPodSecurityContext() *corev1.PodSecurityContext { return &corev1.PodSecurityContext{ RunAsNonRoot: boolPtr(true), RunAsUser: int64Ptr(nonRootUID), RunAsGroup: int64Ptr(nonRootUID), FSGroup: int64Ptr(nonRootUID), SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault}, } } // reaperPodSecurityContext is the reaper's Pod identity: ROOT, deliberately NOT // the control-plane's non-root uid. Its worlds mount IS the live storage root, // and the world directories beneath it (and the files inside them) are written // by the game uid (naming.GameUID) — or by root, in a world an older release // wrote — with Paper's mode-0600 saves (level.dat) included, while the storage // root that holds each world directory belongs to root. Only root with DAC // override (granted on the container below) can both archive every world and // remove its directory from that root; the uid-1000 control-plane convention // failed them with `permission denied` (verified live). func reaperPodSecurityContext() *corev1.PodSecurityContext { return &corev1.PodSecurityContext{ RunAsNonRoot: boolPtr(false), RunAsUser: int64Ptr(0), RunAsGroup: int64Ptr(0), SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault}, } } // reaperContainerSecurityContext is hardenedContainerSecurityContext plus // DAC_OVERRIDE: with root's caps dropped, root can read only files it owns, and // a world may have been written by a game image whose UID is neither root nor // ours. DAC_OVERRIDE restores exactly the file-mode bypass the archive needs. func reaperContainerSecurityContext() *corev1.SecurityContext { sc := hardenedContainerSecurityContext() sc.Capabilities = &corev1.Capabilities{ Drop: []corev1.Capability{"ALL"}, Add: []corev1.Capability{"DAC_OVERRIDE"}, } return sc } // hardenedContainerSecurityContext mirrors the build/restore Job containers: no // privilege, no escalation, read-only root filesystem (all writes go to the // mounted volumes — the config/data mounts and the /tmp emptyDir), drop ALL // capabilities. readOnlyRootFilesystem is shape-asserted, not runtime-proven. func hardenedContainerSecurityContext() *corev1.SecurityContext { return &corev1.SecurityContext{ Privileged: boolPtr(false), AllowPrivilegeEscalation: boolPtr(false), ReadOnlyRootFilesystem: boolPtr(true), Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}}, } } func boolPtr(b bool) *bool { return &b } func int32Ptr(i int32) *int32 { return &i } func int64Ptr(i int64) *int64 { return &i } func controllerSelector(p Params) map[string]string { if p.ControllerNode == "" { return nil } return map[string]string{placement.LabelIdentity: p.ControllerNode, placement.LabelRole: placement.RoleController} }