fix(platform): control plane runs system-cluster-critical, so eviction refuses it (#8)
Following the first shield attempt (custom class, value 1e6) a live drill showed the limit: kubelet evicted the game pods and then the api, operator and registry anyway — evicting them was never what reclaimed the disk — and with the images containerd-only, the GC stage left everything in ImagePullBackOff. A custom class cannot be raised past 1e9 (the API caps user-defined values), while kubelet's eviction refusal needs >= 2e9, so the control plane now uses the built-in system-cluster-critical. Re-drilled: disk filled to 1.7G free -> login/lobby evicted, and kubelet logged "cannot evict a critical pod" for felis-api/operator/registry, which stayed Running throughout. Recovery facts now in troubleshooting 13b: the DiskPressure condition lingers ~5m after space is freed (--eviction-pressure-transition-period), and game images GC'd while their pods were evicted need the documented re-import (verified: 25s to Running).
This commit is contained in:
5 files changed
+72
-77
No files matched your search
@@ -34,9 +34,10 @@ type Object interface {
|
||||
// Deployments and the in-cluster registry (Deployment + Service + PVC), which
|
||||
// make the SAs and NetworkPolicy peers refer to something real, plus the
|
||||
// world-archive PVC that backs backup/restore when a backup PVC is named (see
|
||||
// workloads.go). Workloads also carries the one cluster-scoped object, the
|
||||
// felis-control-plane PriorityClass (node-pressure eviction shield — a
|
||||
// PriorityClass is not namespaced by design); everything else is namespaced.
|
||||
// workloads.go). Every object here is namespaced; the control plane's
|
||||
// node-pressure eviction shield is the BUILT-IN system-cluster-critical
|
||||
// PriorityClass the pod templates reference (workloads.go controlPlanePriorityName),
|
||||
// not an object this bundle renders.
|
||||
// The reaper CronJob is also part of Workloads, rendered only when the retention
|
||||
// storage topology is supplied (WorldsHostPath + BackupPVC + ArchiveLocalPath —
|
||||
// workloads.go documents the gate and the shape-asserted hostPath caveat). The
|
||||
|
||||
@@ -7,7 +7,6 @@ import (
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
schedulingv1 "k8s.io/api/scheduling/v1"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/util/intstr"
|
||||
@@ -177,14 +176,13 @@ func InternalAPIBaseURL(controlNamespace string) string {
|
||||
// felis-operator Deployment, and the in-cluster registry (Deployment + Service +
|
||||
// PVC), the world-archive PVC when p.BackupPVC names it (it backs the
|
||||
// backup/restore Jobs and the reaper), plus the reaper CronJob when
|
||||
// reaperEnabled(p). Every pod template carries the felis-control-plane
|
||||
// PriorityClass (controlPlanePriorityClass below), the node-pressure eviction
|
||||
// shield. FelisImage is required — `felis manifests` enforces it (fail-loud), so
|
||||
// a rendered bundle always names a concrete image.
|
||||
// reaperEnabled(p). Every pod template carries the built-in
|
||||
// system-cluster-critical PriorityClass (controlPlanePriorityName), the
|
||||
// node-pressure eviction shield. FelisImage is required — `felis manifests`
|
||||
// enforces it (fail-loud), so a rendered bundle always names a concrete image.
|
||||
func Workloads(p Params) []Object {
|
||||
p = p.withDefaults()
|
||||
objs := []Object{
|
||||
controlPlanePriorityClass(),
|
||||
APIDeployment(p),
|
||||
apiService(p),
|
||||
apiInternalService(p),
|
||||
@@ -204,35 +202,30 @@ func Workloads(p Params) []Object {
|
||||
}
|
||||
|
||||
// controlPlanePriorityName is the PriorityClass every control-plane workload runs
|
||||
// under (api, operator, reaper, registry). Node-pressure eviction (a full disk
|
||||
// being the realistic case on a game box) removes pods in ASCENDING priority, and
|
||||
// a classless pod is priority 0 — the same as the game servers, which are exactly
|
||||
// the pods a burst of joins has just filled the node with. A full disk then takes
|
||||
// the api/operator down too, and recovery needs a human: with no reachable
|
||||
// registry on an air-gapped box the images are gone, so the fix is a re-run of the
|
||||
// installer to re-import them. The class is an eviction shield, not a preemption
|
||||
// lever: PreemptionPolicy=Never, so a busy node never loses a running game server
|
||||
// merely to schedule the api. Value 1,000,000 sits above every game pod (0) and
|
||||
// far below the kubelet's system-reserved classes (2,000,000,000).
|
||||
const (
|
||||
controlPlanePriorityName = "felis-control-plane"
|
||||
controlPlanePriorityValue int32 = 1_000_000
|
||||
controlPlanePriorityReason = "Felis control plane (api/operator/reaper/registry) survives node-pressure eviction before game servers"
|
||||
)
|
||||
|
||||
// controlPlanePriorityClass renders the cluster-scoped PriorityClass the
|
||||
// control-plane pod templates reference (the ONE cluster-scoped object in the
|
||||
// bundle; a PriorityClass is not namespaced by design).
|
||||
func controlPlanePriorityClass() *schedulingv1.PriorityClass {
|
||||
never := corev1.PreemptNever
|
||||
return &schedulingv1.PriorityClass{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "scheduling.k8s.io/v1", Kind: "PriorityClass"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: controlPlanePriorityName},
|
||||
Value: controlPlanePriorityValue,
|
||||
PreemptionPolicy: &never,
|
||||
Description: controlPlanePriorityReason,
|
||||
}
|
||||
}
|
||||
// under (api, operator, reaper, registry): the BUILT-IN system-cluster-critical,
|
||||
// not a class we render. Two live findings force it:
|
||||
//
|
||||
// - ordering alone is not enough. With the disk at k3s's eviction threshold
|
||||
// (nodefs.available<5%) a drill watched the kubelet evict the game pods first
|
||||
// — and then the api, operator and registry too, because evicting them was
|
||||
// never what would reclaim the disk. The images live only in the node's
|
||||
// containerd (air-gapped), so once their pods were gone the kubelet GC'd them
|
||||
// and recovery degenerated into ImagePullBackOff plus a manual re-import.
|
||||
// - a class we render cannot reach the critical threshold: user-defined
|
||||
// PriorityClasses are capped at 1e9, while kubelet's
|
||||
// `IsCriticalPodBasedOnPriority` — and with it the eviction manager's
|
||||
// "cannot evict a critical pod" refusal — needs ≥ 2e9. Only the built-in
|
||||
// classes carry those values.
|
||||
//
|
||||
// system-cluster-critical keeps the panel/console alive through a full-disk
|
||||
// incident (which is what lets an operator see it and act) and makes the control
|
||||
// plane schedule before game pods after a reboot. Its preemptionPolicy is
|
||||
// PreemptLowerPriority (built-in classes fix it; the API rejects a per-pod
|
||||
// override), so a control-plane pod that cannot fit MAY preempt a game pod:
|
||||
// accepted deliberately — the management plane must be placeable, and the
|
||||
// alternative is an unreachable box. Game pods stay at the default 0 and are
|
||||
// still the first casualties of node pressure.
|
||||
const controlPlanePriorityName = "system-cluster-critical"
|
||||
|
||||
// reaperEnabled reports whether the retention CronJob should render. It needs all
|
||||
// three storage coordinates: WorldsHostPath (where worlds live, mounted to read
|
||||
|
||||
@@ -6,7 +6,6 @@ import (
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
schedulingv1 "k8s.io/api/scheduling/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/labels"
|
||||
)
|
||||
@@ -460,8 +459,8 @@ func TestRegistry_DeploymentServicePVC(t *testing.T) {
|
||||
// internal Service must be present or the login pod's felis-api:8081 path is dead.
|
||||
func TestWorkloads_BundleContents(t *testing.T) {
|
||||
objs := Workloads(testParams())
|
||||
if len(objs) != 9 {
|
||||
t.Fatalf("Workloads returned %d objects, want 9", len(objs))
|
||||
if len(objs) != 8 {
|
||||
t.Fatalf("Workloads returned %d objects, want 8", len(objs))
|
||||
}
|
||||
var haveInternalSvc bool
|
||||
for _, o := range objs {
|
||||
@@ -479,24 +478,20 @@ func TestWorkloads_BundleContents(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestControlPlanePriorityClass pins the node-pressure eviction shield: every
|
||||
// control-plane pod template (api/operator/reaper/registry) references the one
|
||||
// PriorityClass, which outranks the default-0 game pods by eviction order while
|
||||
// never preempting them, and is the bundle's only cluster-scoped object. Without
|
||||
// this, a full disk evicts the api alongside the game pods and (no reachable
|
||||
// registry on an air-gapped box) recovery needs a human re-importing images.
|
||||
// control-plane pod template (api/operator/reaper/registry) runs under the
|
||||
// BUILT-IN system-cluster-critical class (value 2e9), at which kubelet's
|
||||
// eviction manager refuses to evict the pod. A live drill showed the whole
|
||||
// cascade with plain ordering: disk pressure evicted the game pods and then the
|
||||
// control plane, whose images (air-gapped, containerd-only) were GC'd →
|
||||
// ImagePullBackOff plus a manual re-import. User-defined classes are capped at
|
||||
// 1e9 (API-enforced), so the built-in class is the only way to reach the
|
||||
// critical threshold.
|
||||
func TestControlPlanePriorityClass(t *testing.T) {
|
||||
objs := Workloads(reaperParams()) // include the reaper CronJob's pod template
|
||||
|
||||
var class *schedulingv1.PriorityClass
|
||||
clusterScoped := 0
|
||||
deployments, cronJobs := 0, 0
|
||||
for _, o := range objs {
|
||||
if o.GetNamespace() == "" {
|
||||
clusterScoped++
|
||||
}
|
||||
switch v := o.(type) {
|
||||
case *schedulingv1.PriorityClass:
|
||||
class = v
|
||||
case *appsv1.Deployment:
|
||||
deployments++
|
||||
if v.Spec.Template.Spec.PriorityClassName != controlPlanePriorityName {
|
||||
@@ -509,25 +504,13 @@ func TestControlPlanePriorityClass(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
if class == nil {
|
||||
t.Fatal("Workloads must render the control-plane PriorityClass")
|
||||
}
|
||||
if class.Name != controlPlanePriorityName || class.Value != controlPlanePriorityValue {
|
||||
t.Errorf("class = %s/%d, want %s/%d", class.Name, class.Value, controlPlanePriorityName, controlPlanePriorityValue)
|
||||
}
|
||||
if class.PreemptionPolicy == nil || *class.PreemptionPolicy != corev1.PreemptNever {
|
||||
t.Errorf("class preemptionPolicy = %v, want Never (an eviction shield, never a lever against running game servers)", class.PreemptionPolicy)
|
||||
}
|
||||
if clusterScoped != 1 {
|
||||
t.Errorf("bundle has %d cluster-scoped objects, want exactly the PriorityClass", clusterScoped)
|
||||
}
|
||||
if deployments != 3 || cronJobs != 1 {
|
||||
t.Errorf("scanned %d deployments / %d cronjobs, want 3 / 1 — a pod template escaped the class check", deployments, cronJobs)
|
||||
}
|
||||
// The class must outrank the default game-pod priority (0); equality would make
|
||||
// the eviction order nondeterministic between the api and a full game server.
|
||||
if controlPlanePriorityValue <= 0 {
|
||||
t.Errorf("control-plane priority %d must exceed the default 0 game-pod priority", controlPlanePriorityValue)
|
||||
// The name must be the built-in critical class: any custom class is capped at
|
||||
// 1e9 by the API server and would be evictable.
|
||||
if controlPlanePriorityName != "system-cluster-critical" {
|
||||
t.Errorf("control-plane priority class = %q; only the built-in critical classes reach the 2e9 eviction-refusal threshold", controlPlanePriorityName)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user