594 lines
25 KiB
Go
594 lines
25 KiB
Go
package operator
|
|
|
|
import (
|
|
"time"
|
|
|
|
"fmt"
|
|
"strconv"
|
|
|
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
|
"felis.lolicon.best/internal/naming"
|
|
"felis.lolicon.best/internal/placement"
|
|
appsv1 "k8s.io/api/apps/v1"
|
|
corev1 "k8s.io/api/core/v1"
|
|
"k8s.io/apimachinery/pkg/api/resource"
|
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
|
"k8s.io/apimachinery/pkg/util/intstr"
|
|
)
|
|
|
|
// envServiceToken is the environment variable the felis-limbo login plugin reads
|
|
// its internal-API bearer credential from. It is injected ONLY into the login
|
|
// system server (see buildEnv), sourced from a Secret, never a literal.
|
|
const envServiceToken = "FELIS_SERVICE_TOKEN"
|
|
|
|
// envForwardingSecret is the environment variable a backend reads the Velocity
|
|
// modern-forwarding secret from. Unlike the service token it goes to EVERY backend
|
|
// (see buildEnv), because Velocity's forwarding mode is proxy-wide.
|
|
const envForwardingSecret = "FELIS_FORWARDING_SECRET"
|
|
|
|
// Workload constants shared by the builders.
|
|
const (
|
|
// GamePort is the Minecraft TCP port the proxy and readiness probe target.
|
|
GamePort int32 = 25565
|
|
|
|
// DefaultRconPort is the RCON port used when a server does not override
|
|
// spec.rcon.port (see rconPort). It is exported because the platform package's
|
|
// allow-rcon NetworkPolicy opens this port for the control plane — sharing the
|
|
// constant keeps the policy port and the container's default RCON port a single
|
|
// source of truth, so the operator's prober can always reach a default-port
|
|
// server through the fence.
|
|
DefaultRconPort int32 = 25575
|
|
|
|
containerName = "minecraft"
|
|
dataVolumeName = "world"
|
|
dataMountPath = "/data"
|
|
|
|
// felisBinaryPath is where the felis image installs its binary; the
|
|
// forwarding-config initContainer invokes it by absolute path (matches
|
|
// platform.felisBinaryPath — the same image, the same install location).
|
|
felisBinaryPath = "/usr/local/bin/felis"
|
|
|
|
// ManagedByValue / ComponentValue are the values of the LabelManagedBy /
|
|
// LabelComponent labels stamped on every per-server pod (see labelsFor). They
|
|
// are exported because the platform package's minecraft-namespace
|
|
// NetworkPolicies select server pods by exactly these labels — keeping the
|
|
// selector and the pod labels a single source of truth, so an isolation policy
|
|
// can never silently stop matching the pods it is meant to fence.
|
|
ManagedByValue = "felis-operator"
|
|
ComponentValue = "server"
|
|
|
|
defaultGraceSeconds int64 = 300
|
|
defaultStorageSize string = "8Gi"
|
|
)
|
|
|
|
// selectorFor returns the immutable selector labels (a StatefulSet selector
|
|
// must never change after creation, so it carries only the server identity).
|
|
func selectorFor(server *v1alpha1.MinecraftServer) map[string]string {
|
|
return map[string]string{v1alpha1.LabelServer: server.Name}
|
|
}
|
|
|
|
// labelsFor returns the full label set applied to managed objects.
|
|
func labelsFor(server *v1alpha1.MinecraftServer) map[string]string {
|
|
return map[string]string{
|
|
v1alpha1.LabelServer: server.Name,
|
|
v1alpha1.LabelManagedBy: ManagedByValue,
|
|
v1alpha1.LabelComponent: ComponentValue,
|
|
}
|
|
}
|
|
|
|
// podLabelsFor is labelsFor plus the setup-owned system-role label, copied onto
|
|
// the pod so the platform's NetworkPolicies can tell the login gate apart from a
|
|
// user server (internal/platform loginToInternalAPI). Only the template carries it:
|
|
// the StatefulSet selector is immutable and stays selectorFor.
|
|
func podLabelsFor(server *v1alpha1.MinecraftServer) map[string]string {
|
|
l := labelsFor(server)
|
|
if role := server.Labels[v1alpha1.LabelSystemRole]; role != "" {
|
|
l[v1alpha1.LabelSystemRole] = role
|
|
}
|
|
return l
|
|
}
|
|
|
|
func headlessServiceName(name string) string { return name + "-hl" }
|
|
|
|
// rconPort resolves the RCON port, defaulting to the conventional DefaultRconPort.
|
|
func rconPort(server *v1alpha1.MinecraftServer) int32 {
|
|
if server.Spec.Rcon.Port > 0 {
|
|
return server.Spec.Rcon.Port
|
|
}
|
|
return DefaultRconPort
|
|
}
|
|
|
|
// graceSeconds resolves the pod termination grace period (spec §7).
|
|
func graceSeconds(server *v1alpha1.MinecraftServer) int64 {
|
|
if server.Spec.Lifecycle.TerminationGracePeriodSeconds > 0 {
|
|
return server.Spec.Lifecycle.TerminationGracePeriodSeconds
|
|
}
|
|
return defaultGraceSeconds
|
|
}
|
|
|
|
// rconAddress is the in-cluster RCON endpoint the operator probes for readiness.
|
|
func rconAddress(server *v1alpha1.MinecraftServer) string {
|
|
return fmt.Sprintf("%s.%s.svc.cluster.local:%d", server.Name, server.Namespace, rconPort(server))
|
|
}
|
|
|
|
// buildHeadlessService backs the StatefulSet's stable network identity.
|
|
func buildHeadlessService(server *v1alpha1.MinecraftServer) *corev1.Service {
|
|
svc := &corev1.Service{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: headlessServiceName(server.Name),
|
|
Namespace: server.Namespace,
|
|
Labels: labelsFor(server),
|
|
},
|
|
Spec: corev1.ServiceSpec{
|
|
ClusterIP: corev1.ClusterIPNone,
|
|
Selector: selectorFor(server),
|
|
Ports: servicePorts(server),
|
|
},
|
|
}
|
|
return svc
|
|
}
|
|
|
|
// buildClientService is the stable ClusterIP the proxy and operator dial.
|
|
func buildClientService(server *v1alpha1.MinecraftServer) *corev1.Service {
|
|
return &corev1.Service{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: server.Name,
|
|
Namespace: server.Namespace,
|
|
Labels: labelsFor(server),
|
|
},
|
|
Spec: corev1.ServiceSpec{
|
|
Selector: selectorFor(server),
|
|
Ports: servicePorts(server),
|
|
},
|
|
}
|
|
}
|
|
|
|
func servicePorts(server *v1alpha1.MinecraftServer) []corev1.ServicePort {
|
|
ports := []corev1.ServicePort{{
|
|
Name: "game",
|
|
Port: GamePort,
|
|
TargetPort: intstr.FromInt32(GamePort),
|
|
Protocol: corev1.ProtocolTCP,
|
|
}}
|
|
if server.Spec.Rcon.Enabled {
|
|
p := rconPort(server)
|
|
ports = append(ports, corev1.ServicePort{
|
|
Name: "rcon",
|
|
Port: p,
|
|
TargetPort: intstr.FromInt32(p),
|
|
Protocol: corev1.ProtocolTCP,
|
|
})
|
|
}
|
|
return ports
|
|
}
|
|
|
|
// readinessProbe selects the pod readiness probe. By default it is a plain TCP
|
|
// check on the game port; when the server declares an HTTP health port
|
|
// (StartupSpec.HealthHTTPPort > 0) it becomes an HTTP GET on that port, so an
|
|
// RCON-less loader's own "started" signal — not the mere fact that the game
|
|
// socket is bound — gates readiness. Timings are identical across both modes.
|
|
func readinessProbe(server *v1alpha1.MinecraftServer) *corev1.Probe {
|
|
return &corev1.Probe{
|
|
ProbeHandler: healthHandler(server),
|
|
InitialDelaySeconds: 20,
|
|
PeriodSeconds: 10,
|
|
FailureThreshold: 6,
|
|
}
|
|
}
|
|
|
|
// startupProbe holds liveness off until the server first answers. Its budget
|
|
// outlasts the operator's startup timeout plus the first auto-restart backoff,
|
|
// so a slow first world generation is the operator's to judge (a counted,
|
|
// bounded pod restart) and never a kubelet restart loop.
|
|
func startupProbe(server *v1alpha1.MinecraftServer) *corev1.Probe {
|
|
const period = 10
|
|
budget := startupTimeout(server) + autoRestartBaseBackoff
|
|
return &corev1.Probe{
|
|
ProbeHandler: healthHandler(server),
|
|
PeriodSeconds: period,
|
|
FailureThreshold: int32((budget+period*time.Second-1)/(period*time.Second)) + 1,
|
|
}
|
|
}
|
|
|
|
// livenessProbe restarts a server that stopped answering for two minutes: long
|
|
// enough to ride out a world save or a lag spike.
|
|
func livenessProbe(server *v1alpha1.MinecraftServer) *corev1.Probe {
|
|
return &corev1.Probe{
|
|
ProbeHandler: healthHandler(server),
|
|
PeriodSeconds: 20,
|
|
TimeoutSeconds: 5,
|
|
FailureThreshold: 6,
|
|
}
|
|
}
|
|
|
|
// healthHandler is a plain TCP check on the game port, or an HTTP GET on the
|
|
// loader's health endpoint when StartupSpec.HealthHTTPPort is set.
|
|
func healthHandler(server *v1alpha1.MinecraftServer) corev1.ProbeHandler {
|
|
if hp := server.Spec.Startup.HealthHTTPPort; hp > 0 {
|
|
path := server.Spec.Startup.HealthHTTPPath
|
|
if path == "" {
|
|
path = "/healthz"
|
|
}
|
|
return corev1.ProbeHandler{HTTPGet: &corev1.HTTPGetAction{Path: path, Port: intstr.FromInt32(hp)}}
|
|
}
|
|
return corev1.ProbeHandler{TCPSocket: &corev1.TCPSocketAction{Port: intstr.FromInt32(GamePort)}}
|
|
}
|
|
|
|
// buildStatefulSet renders the workload for replicas in {0,1}. Its half of
|
|
// graceful shutdown is terminationGracePeriodSeconds, the time the server gets to
|
|
// save on SIGTERM; the reconciler flushes the world over RCON before it scales to
|
|
// zero (saveBeforeStop).
|
|
func buildStatefulSet(server *v1alpha1.MinecraftServer, replicas int32, felisImage string, gateProbe ...string) (*appsv1.StatefulSet, error) {
|
|
storageSize := server.Spec.Storage.Size
|
|
if storageSize == "" {
|
|
storageSize = defaultStorageSize
|
|
}
|
|
storageQty, err := resource.ParseQuantity(storageSize)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("invalid storage size %q: %w", storageSize, err)
|
|
}
|
|
|
|
container := corev1.Container{
|
|
Name: containerName,
|
|
Image: server.Spec.Image,
|
|
Resources: server.Spec.Resources,
|
|
Env: buildEnv(server),
|
|
Ports: []corev1.ContainerPort{
|
|
{Name: "game", ContainerPort: GamePort, Protocol: corev1.ProtocolTCP},
|
|
},
|
|
VolumeMounts: []corev1.VolumeMount{
|
|
{Name: dataVolumeName, MountPath: dataMountPath},
|
|
},
|
|
// Readiness defaults to a plain TCP check (spec §5: readinessProbe is only
|
|
// tcpSocket; the RCON gate is enforced by the operator, not the kubelet).
|
|
// An RCON-less loader may instead publish an HTTP health endpoint (see
|
|
// StartupSpec.HealthHTTPPort) that reports true readiness — used below when
|
|
// set.
|
|
ReadinessProbe: readinessProbe(server),
|
|
StartupProbe: startupProbe(server),
|
|
LivenessProbe: livenessProbe(server),
|
|
// The server runs untrusted plugins, so it keeps no capability and can never
|
|
// regain one. The root filesystem stays writable: an arbitrary Paper image
|
|
// may unpack its runtime or write temp files outside /data.
|
|
SecurityContext: hardenedContainerSecurityContext(false),
|
|
}
|
|
if hp := server.Spec.Startup.HealthHTTPPort; hp > 0 {
|
|
container.Ports = append(container.Ports, corev1.ContainerPort{
|
|
Name: "health", ContainerPort: hp, Protocol: corev1.ProtocolTCP,
|
|
})
|
|
}
|
|
if len(server.Spec.Args) > 0 {
|
|
container.Args = append([]string(nil), server.Spec.Args...)
|
|
}
|
|
if server.Spec.Rcon.Enabled {
|
|
container.Ports = append(container.Ports, corev1.ContainerPort{
|
|
Name: "rcon", ContainerPort: rconPort(server), Protocol: corev1.ProtocolTCP,
|
|
})
|
|
}
|
|
|
|
// Every server first hands its world volume to the game uid (prepareDataInitContainer),
|
|
// since the pod runs as that uid and a world an older root-run release wrote would
|
|
// otherwise be read-only to it. An arbitrary user Paper image then gets the forwarding
|
|
// config written for it (it does not consume FELIS_FORWARDING_SECRET itself).
|
|
// The lobby uses the same merge so custom settings survive; the login Limbo
|
|
// handles its own properties format. Every server then waits for its egress fence.
|
|
// Without a felis image name there is nothing to run any step with.
|
|
var initContainers []corev1.Container
|
|
if felisImage != "" {
|
|
initContainers = append(initContainers, prepareDataInitContainer(felisImage))
|
|
if server.Labels[v1alpha1.LabelSystemRole] != naming.SystemLoginServer {
|
|
initContainers = append(initContainers, forwardingInitContainer(felisImage))
|
|
if server.Labels[v1alpha1.LabelSystemRole] == naming.SystemLobbyServer {
|
|
container.Env = append(container.Env, corev1.EnvVar{Name: "FELIS_MANAGED_FORWARDING", Value: "true"})
|
|
}
|
|
}
|
|
gate := egressGateInitContainer(felisImage)
|
|
if server.Spec.NodeName != "" || (len(gateProbe) > 0 && gateProbe[0] != "") {
|
|
gate.Command = append(gate.Command[:len(gate.Command)-1], "--positive-probe", "kube-dns.kube-system.svc:53")
|
|
if len(gateProbe) > 0 && gateProbe[0] != "" {
|
|
gate.Command = append(gate.Command, "--probe", gateProbe[0])
|
|
}
|
|
}
|
|
initContainers = append(initContainers, gate)
|
|
}
|
|
|
|
grace := graceSeconds(server)
|
|
pvc := corev1.PersistentVolumeClaim{
|
|
ObjectMeta: metav1.ObjectMeta{Name: dataVolumeName},
|
|
Spec: corev1.PersistentVolumeClaimSpec{
|
|
AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce},
|
|
Resources: corev1.VolumeResourceRequirements{
|
|
Requests: corev1.ResourceList{corev1.ResourceStorage: storageQty},
|
|
},
|
|
},
|
|
}
|
|
if sc := server.Spec.Storage.StorageClassName; sc != "" {
|
|
pvc.Spec.StorageClassName = &sc
|
|
}
|
|
|
|
sts := &appsv1.StatefulSet{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: server.Name,
|
|
Namespace: server.Namespace,
|
|
Labels: labelsFor(server),
|
|
},
|
|
Spec: appsv1.StatefulSetSpec{
|
|
Replicas: &replicas,
|
|
ServiceName: headlessServiceName(server.Name),
|
|
Selector: &metav1.LabelSelector{MatchLabels: selectorFor(server)},
|
|
Template: corev1.PodTemplateSpec{
|
|
ObjectMeta: metav1.ObjectMeta{Labels: podLabelsFor(server)},
|
|
Spec: corev1.PodSpec{
|
|
TerminationGracePeriodSeconds: &grace,
|
|
InitContainers: initContainers,
|
|
Containers: []corev1.Container{container},
|
|
// A Minecraft server runs untrusted user worlds and plugins and
|
|
// has no business calling the K8s API, so its pod must NOT carry the
|
|
// default ServiceAccount token: a compromised plugin could otherwise
|
|
// authenticate as the namespace default SA (spec §21: user servers
|
|
// default to no SA-token mount). The pod keeps the default SA but
|
|
// with automounting explicitly disabled.
|
|
AutomountServiceAccountToken: boolPtr(false),
|
|
SecurityContext: gamePodSecurityContext(),
|
|
},
|
|
},
|
|
VolumeClaimTemplates: []corev1.PersistentVolumeClaim{pvc},
|
|
// The world outlives its StatefulSet: deleting the CR (and with it,
|
|
// by owner reference, the StatefulSet) keeps the claim, so a CR that
|
|
// comes back under the same name mounts the same world. The reaper is
|
|
// the one path that deletes a world, after its final backup. Stated
|
|
// here although it is the API default, so the semantics never ride on
|
|
// a default.
|
|
PersistentVolumeClaimRetentionPolicy: &appsv1.StatefulSetPersistentVolumeClaimRetentionPolicy{
|
|
WhenDeleted: appsv1.RetainPersistentVolumeClaimRetentionPolicyType,
|
|
WhenScaled: appsv1.RetainPersistentVolumeClaimRetentionPolicyType,
|
|
},
|
|
},
|
|
}
|
|
if server.Spec.Storage.ClaimName != "" {
|
|
sts.Spec.VolumeClaimTemplates = nil
|
|
sts.Spec.Template.Spec.Volumes = append(sts.Spec.Template.Spec.Volumes, corev1.Volume{Name: dataVolumeName, VolumeSource: corev1.VolumeSource{PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: server.WorldPVC()}}})
|
|
}
|
|
if server.Spec.NodeName != "" {
|
|
sts.Spec.Template.Spec.NodeSelector = map[string]string{placement.LabelIdentity: server.Spec.NodeName}
|
|
}
|
|
return sts, nil
|
|
}
|
|
|
|
// buildEnv assembles the container environment: heap sizing, user-supplied
|
|
// vars, and the RCON_* pair (password sourced from the referenced Secret, never
|
|
// inlined into the CRD).
|
|
func buildEnv(server *v1alpha1.MinecraftServer) []corev1.EnvVar {
|
|
var env []corev1.EnvVar
|
|
if mem := server.Spec.JavaMemory; mem != "" {
|
|
env = append(env, corev1.EnvVar{Name: "JAVA_MEMORY", Value: mem})
|
|
}
|
|
if len(server.Spec.JavaFlags) > 0 {
|
|
env = append(env, corev1.EnvVar{Name: "JAVA_FLAGS", Value: joinFlags(server.Spec.JavaFlags)})
|
|
}
|
|
for _, e := range server.Spec.Env {
|
|
env = append(env, corev1.EnvVar{Name: e.Name, Value: e.Value})
|
|
}
|
|
if server.Spec.Rcon.Enabled && server.Spec.Rcon.SecretRef.Name != "" {
|
|
env = append(env,
|
|
corev1.EnvVar{Name: "RCON_PORT", Value: strconv.Itoa(int(rconPort(server)))},
|
|
corev1.EnvVar{Name: "RCON_PASSWORD", ValueFrom: &corev1.EnvVarSource{
|
|
SecretKeyRef: &corev1.SecretKeySelector{
|
|
LocalObjectReference: corev1.LocalObjectReference{Name: server.Spec.Rcon.SecretRef.Name},
|
|
Key: server.Spec.Rcon.SecretRef.Key,
|
|
},
|
|
}},
|
|
)
|
|
}
|
|
// The login system server is the ONE workload that authenticates to the
|
|
// felis-api internal face (its felis-limbo plugin mints bind codes and polls
|
|
// link status), so it — and only it — receives a token: felis-limbo-token,
|
|
// which the api serves on those routes alone. Injected
|
|
// from a Secret in this namespace, never inlined into the CRD (the same
|
|
// discipline as RCON_PASSWORD above; the CRD's EnvVar type has no valueFrom
|
|
// precisely so a user server cannot mount an arbitrary secret). Require both
|
|
// the reserved name and the setup-owned system-role label: the label prevents
|
|
// a legacy user server named "login" from receiving the token after upgrade.
|
|
// The Secret must exist in this (minecraft) namespace; the installer applies it
|
|
// there and `felis setup` replicates it from the control namespace.
|
|
if server.Name == naming.SystemLoginServer &&
|
|
server.Labels[v1alpha1.LabelSystemRole] == naming.SystemLoginServer {
|
|
env = append(env, corev1.EnvVar{
|
|
Name: envServiceToken,
|
|
ValueFrom: &corev1.EnvVarSource{
|
|
SecretKeyRef: &corev1.SecretKeySelector{
|
|
LocalObjectReference: corev1.LocalObjectReference{Name: naming.LimboTokenSecretName},
|
|
Key: naming.ServiceTokenSecretKey,
|
|
},
|
|
},
|
|
})
|
|
}
|
|
// The Velocity modern-forwarding secret goes to EVERY backend, system and user
|
|
// alike — not because user servers are trusted, but because Velocity's forwarding
|
|
// mode is one proxy-wide setting: with it on, a backend that cannot verify the
|
|
// signed handshake rejects every login the proxy sends it. Withholding the secret
|
|
// from user servers would not harden them, it would simply make them unjoinable.
|
|
// It is the backend's proof that a login really came from the proxy (and so that
|
|
// the player's UUID is Mojang-verified, not offline-derived) — the pod-level fence
|
|
// against bypassing the proxy is the NetworkPolicy, not this value's secrecy.
|
|
//
|
|
// The Felis-built images (deploy/limbo, deploy/lobby) read this in their
|
|
// entrypoints. An arbitrary user Paper image does NOT — so the operator also runs
|
|
// a forwarding-config initContainer (see forwardingInitContainer) that writes the
|
|
// Velocity block into the shared world volume before the server starts, making a
|
|
// stock Paper image joinable without modifying it.
|
|
env = append(env, forwardingSecretEnvVar())
|
|
return env
|
|
}
|
|
|
|
// forwardingSecretEnvVar sources FELIS_FORWARDING_SECRET from the Secret the setup
|
|
// provisioner replicas into this namespace. Optional so a cluster whose proxy is
|
|
// not in modern mode — no Secret provisioned — still schedules its pods instead of
|
|
// wedging them all in CreateContainerConfigError; the init and lobby/limbo
|
|
// entrypoints treat an empty value as "not in modern mode" and leave config alone.
|
|
func forwardingSecretEnvVar() corev1.EnvVar {
|
|
return corev1.EnvVar{
|
|
Name: envForwardingSecret,
|
|
ValueFrom: &corev1.EnvVarSource{
|
|
SecretKeyRef: &corev1.SecretKeySelector{
|
|
LocalObjectReference: corev1.LocalObjectReference{Name: naming.ForwardingSecretName},
|
|
Key: naming.ForwardingSecretKey,
|
|
Optional: boolPtr(true),
|
|
},
|
|
},
|
|
}
|
|
}
|
|
|
|
// forwardingInitContainer writes Velocity modern-forwarding config into the shared
|
|
// world volume before the server container starts, so an arbitrary Paper image Felis
|
|
// did NOT build becomes joinable behind the proxy without being modified. It runs the
|
|
// felis image's `init-forwarding` subcommand, which merges the proxies.velocity block
|
|
// into config/paper-global.yml and forces online-mode=false in server.properties.
|
|
//
|
|
// It runs as the game uid like the server container (the pod securityContext), after
|
|
// prepareDataInitContainer has handed the volume to that uid, so it needs no privilege
|
|
// at all: no capability, a read-only root filesystem, and the files it writes are
|
|
// owned by the very uid that rewrites them on boot.
|
|
//
|
|
// User Paper servers and the lobby share this merge. The login Limbo handles
|
|
// its own properties format in its entrypoint.
|
|
func forwardingInitContainer(felisImage string) corev1.Container {
|
|
return corev1.Container{
|
|
Name: "init-forwarding",
|
|
Image: felisImage,
|
|
Command: []string{felisBinaryPath, "init-forwarding"},
|
|
Env: []corev1.EnvVar{forwardingSecretEnvVar()},
|
|
VolumeMounts: []corev1.VolumeMount{
|
|
{Name: dataVolumeName, MountPath: dataMountPath},
|
|
},
|
|
Resources: initContainerResources(),
|
|
SecurityContext: hardenedContainerSecurityContext(true),
|
|
}
|
|
}
|
|
|
|
// egressGateWait bounds how long a server's start waits for its egress fence.
|
|
// kube-router programs a new pod's policy within a second; the rest is margin for
|
|
// a loaded node.
|
|
const egressGateWait = 30 * time.Second
|
|
|
|
// egressGateInitContainer holds the server image back until the pod's egress fence
|
|
// (platform.ServerEgressPolicies) is in effect. The policy is programmed after the
|
|
// pod starts, and live on k3s a new server-labelled pod reached felis-api's
|
|
// internal face on its first request; the server image and the plugins it loads
|
|
// are the owner's code and must never run inside that window. The gate is `felis
|
|
// egress-gate`, which the build pod runs first for the same reason.
|
|
//
|
|
// It fails open where the build's gate fails closed. An operator's
|
|
// --server-egress-allow-cidr can cover the node the Kubernetes API Service leads
|
|
// to, and then the probe answers forever while the fence stands; refusing would
|
|
// take every server down on a legitimate install. After egressGateWait the policy
|
|
// has landed if it ever will, so letting the server on costs nothing the fence
|
|
// would have given, and the gate's log says why the start was slow.
|
|
func egressGateInitContainer(felisImage string) corev1.Container {
|
|
return corev1.Container{
|
|
Name: "egress-gate",
|
|
Image: felisImage,
|
|
Command: []string{felisBinaryPath, "egress-gate",
|
|
"--wait", egressGateWait.String(), "--fail-open"},
|
|
Resources: initContainerResources(),
|
|
SecurityContext: hardenedContainerSecurityContext(true),
|
|
}
|
|
}
|
|
|
|
// prepareDataInitContainer runs `felis init-volume`, which chowns every world-volume
|
|
// entry not already owned by naming.GameUID:GameGID. It is the one container in the
|
|
// pod that runs as root, and it holds only what a chown walk needs: CHOWN to change
|
|
// an owner and DAC_OVERRIDE to descend into a directory some other uid left at 0700.
|
|
// Both are inside the PodSecurity baseline profile; everything else is dropped, the
|
|
// root filesystem is read-only, and it exits before the server container starts.
|
|
//
|
|
// fsGroup (gamePodSecurityContext) alone would not do: kubelet skips it for hostPath
|
|
// volumes, which is what a k3s local-path PV is underneath, and it only fixes the
|
|
// group besides.
|
|
func prepareDataInitContainer(felisImage string) corev1.Container {
|
|
return corev1.Container{
|
|
Name: "prepare-data",
|
|
Image: felisImage,
|
|
Command: []string{felisBinaryPath, "init-volume", "--data", dataMountPath},
|
|
VolumeMounts: []corev1.VolumeMount{
|
|
{Name: dataVolumeName, MountPath: dataMountPath},
|
|
},
|
|
Resources: initContainerResources(),
|
|
SecurityContext: &corev1.SecurityContext{
|
|
RunAsUser: int64Ptr(0),
|
|
RunAsGroup: int64Ptr(0),
|
|
RunAsNonRoot: boolPtr(false),
|
|
Privileged: boolPtr(false),
|
|
AllowPrivilegeEscalation: boolPtr(false),
|
|
ReadOnlyRootFilesystem: boolPtr(true),
|
|
Capabilities: &corev1.Capabilities{
|
|
Drop: []corev1.Capability{"ALL"},
|
|
Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE"},
|
|
},
|
|
},
|
|
}
|
|
}
|
|
|
|
// gamePodSecurityContext pins every container in a server pod to the game uid,
|
|
// whatever USER its image declares, and to the runtime's default seccomp filter.
|
|
// fsGroup makes a volume type that supports ownership management group-writable
|
|
// for that uid; OnRootMismatch keeps kubelet from re-walking a large world on every
|
|
// start once the volume root already carries the group.
|
|
func gamePodSecurityContext() *corev1.PodSecurityContext {
|
|
onRootMismatch := corev1.FSGroupChangeOnRootMismatch
|
|
return &corev1.PodSecurityContext{
|
|
RunAsNonRoot: boolPtr(true),
|
|
RunAsUser: int64Ptr(naming.GameUID),
|
|
RunAsGroup: int64Ptr(naming.GameGID),
|
|
FSGroup: int64Ptr(naming.GameGID),
|
|
FSGroupChangePolicy: &onRootMismatch,
|
|
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
|
|
}
|
|
}
|
|
|
|
// hardenedContainerSecurityContext drops every capability and forbids gaining one
|
|
// back through a setuid binary. readOnlyRoot is set for the felis-image containers,
|
|
// which write nothing outside the world volume.
|
|
func hardenedContainerSecurityContext(readOnlyRoot bool) *corev1.SecurityContext {
|
|
sc := &corev1.SecurityContext{
|
|
Privileged: boolPtr(false),
|
|
AllowPrivilegeEscalation: boolPtr(false),
|
|
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
|
}
|
|
if readOnlyRoot {
|
|
sc.ReadOnlyRootFilesystem = boolPtr(true)
|
|
}
|
|
return sc
|
|
}
|
|
|
|
// initContainerResources bounds the felis-image initContainers. Two are short file
|
|
// walks and the third a dial loop; the memory ceiling stops a pathological volume from taking the node's
|
|
// memory with it, and no CPU limit keeps a large world's chown from being throttled
|
|
// into the pod's start-up time.
|
|
func initContainerResources() corev1.ResourceRequirements {
|
|
return corev1.ResourceRequirements{
|
|
Requests: corev1.ResourceList{
|
|
corev1.ResourceCPU: resource.MustParse("10m"),
|
|
corev1.ResourceMemory: resource.MustParse("32Mi"),
|
|
},
|
|
Limits: corev1.ResourceList{
|
|
corev1.ResourceMemory: resource.MustParse("128Mi"),
|
|
},
|
|
}
|
|
}
|
|
|
|
func boolPtr(b bool) *bool { return &b }
|
|
|
|
func int64Ptr(i int64) *int64 { return &i }
|
|
|
|
func joinFlags(flags []string) string {
|
|
out := ""
|
|
for i, f := range flags {
|
|
if i > 0 {
|
|
out += " "
|
|
}
|
|
out += f
|
|
}
|
|
return out
|
|
}
|