fix(operator): 服务器 pod 先等出站围栏生效再启动镜像,新 pod 第一个请求不再能连到内网

This commit is contained in:
Lemon-miaow committed 2026-09-27 20:10:41 +08:00
1 parent 051cc1c9f5
commit ab9a6f0bdb
5 files changed
+160 -20

No files matched your search

+22 -11
View File
@@ -17,23 +17,28 @@ var (
egressPollInterval = 200 * time.Millisecond
)
// cmdEgressGate is the first initContainer of every build pod. The pod's
// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s:
// a build-labelled pod reached the internet and the Kubernetes API for its first
// ~0.7 s), so the gate dials a destination the policy denies until it stops
// answering, and only then lets the pod's next container, eventually the
// untrusted Dockerfile, start.
// cmdEgressGate is the first initContainer of every build pod and the last of
// every game server pod. A pod's NetworkPolicy is programmed asynchronously after
// the pod starts (live on k3s: a build-labelled pod reached the internet and the
// Kubernetes API for its first ~0.7 s, a server-labelled one felis-api's internal
// face on its first request), so the gate dials a destination the policy denies
// until it stops answering, and only then lets the pod's next container, the
// untrusted Dockerfile or server image, start.
//
// The default probe is the Kubernetes API Service, which the kubelet names in
// every pod's environment and the build policy never admits. A probe that still
// answers after --wait means the policy is not enforced at all (a CNI without
// every pod's environment and neither policy admits. A probe that still answers
// after --wait means the policy is not enforced at all (a CNI without
// NetworkPolicy support, or k3s run with --disable-network-policy), and the
// build fails closed.
// build fails closed. A server passes --fail-open: an operator's
// --server-egress-allow-cidr may cover the node the API Service leads to, so a
// probe that keeps answering does not prove the fence is missing, and by then
// the policy has had --wait to land.
func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError)
fs.SetOutput(stderr)
probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused")
probe := fs.String("probe", "", "host:port the pod's NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the gate gives up")
failOpen := fs.Bool("fail-open", false, "when --wait runs out, warn and let the pod go on instead of refusing it")
if err := fs.Parse(args); err != nil {
return 2
}
@@ -56,6 +61,12 @@ func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
}
_ = conn.Close()
if time.Since(start) >= *wait {
if *failOpen {
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s; starting anyway. Either this namespace's "+
"NetworkPolicy is not enforced (a CNI without NetworkPolicy support, or k3s started with "+
"--disable-network-policy), or an allowed CIDR admits the address behind it\n", *probe, *wait)
return 0
}
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+
"(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n",
*probe, *wait)
+34
View File
@@ -75,6 +75,40 @@ func TestEgressGateRefusesAnOpenNetwork(t *testing.T) {
}
}
// A server's gate waits out --wait all the same, then lets the pod start with a
// warning in its log.
func TestEgressGateFailOpenWaitsThenWarns(t *testing.T) {
shrinkEgressGate(t)
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
defer ln.Close()
go func() {
for {
c, err := ln.Accept()
if err != nil {
return
}
_ = c.Close()
}
}()
var out, errb bytes.Buffer
start := time.Now()
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "150ms", "--fail-open"}, &out, &errb); code != 0 {
t.Fatalf("exit %d, want 0: %s", code, errb.String())
}
if waited := time.Since(start); waited < 150*time.Millisecond {
t.Errorf("gave up after %s, before --wait ran out", waited)
}
if !strings.Contains(errb.String(), ln.Addr().String()+" still answers after 150ms; starting anyway") {
t.Errorf("stderr = %q", errb.String())
}
if out.Len() != 0 {
t.Errorf("stdout = %q, want nothing: the lock was never seen", out.String())
}
}
func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) {
shrinkEgressGate(t)
t.Setenv("KUBERNETES_SERVICE_HOST", "")
+25
View File
@@ -118,6 +118,31 @@ kubectl describe pod <pod> # look at Events + container State
to keep its state under `/data`. [GO-TESTED: `TestBuildStatefulSetRunsGameAsNonRoot`,
`TestChownTreeHandsOverMismatchedEntries`; INTEGRATION-ONLY for the walk on a
live volume.]
- **Every start sits about 30 s in its last `Init:` step** → the `egress-gate`
initContainer (`felis egress-gate`) holds the server image until the pod's
egress fence (`felis-server-egress`) is in effect: kube-router programs a new
pod's policy a moment after it starts, and a server-labelled pod reached
felis-api's internal face on its first request. It dials the Kubernetes API
Service, which no server is admitted to, and normally passes within a second.
`kubectl logs <pod> -c egress-gate` says which way it went:
```
felis egress-gate: 10.43.0.1:443 is unreachable after 201ms (...); the egress lock is in effect
felis egress-gate: 10.43.0.1:443 still answers after 30s; starting anyway. ...
```
The second line means the probe answered for the whole wait. Either the cluster
runs without NetworkPolicy enforcement (a CNI without it, or k3s started with
`--disable-network-policy`), which leaves every server off the fence, or a
`--server-egress-allow-cidr` covers the node the API Service leads to, which
admits servers to the node's PostgreSQL and kube API port as well. Narrow the
CIDR to the host the servers need. The gate lets the server start after the
wait either way, where the build's gate refuses. [GO-TESTED:
`TestBuildStatefulSetGatesEgress`, `TestEgressGateFailOpenWaitsThenWarns`.
VM-TESTED on k3s: without the gate a server-labelled pod's first request reached
`felis-api-internal:8081`; behind it, three pods' first requests were refused
after gates of about 200 ms, and a pod no policy selects started after the 5 s
wait it was given, with the warning.]
- **`FailedCreate … violates PodSecurity "baseline"`** on the StatefulSet or a
Job → the minecraft namespace enforces the PodSecurity `baseline` profile
(`pod-security.kubernetes.io/enforce=baseline`, set by the install bundle).
+34 -3
View File
@@ -270,13 +270,15 @@ func buildStatefulSet(server *v1alpha1.MinecraftServer, replicas int32, felisIma
// otherwise be read-only to it. An arbitrary user Paper image then gets the forwarding
// config written for it (it does not consume FELIS_FORWARDING_SECRET itself); system
// servers (login/lobby) are Felis-built and handle forwarding in their own
// entrypoints. Without a felis image name there is nothing to run either step with.
// entrypoints. Last, every server waits for its egress fence (egressGateInitContainer).
// Without a felis image name there is nothing to run any step with.
var initContainers []corev1.Container
if felisImage != "" {
initContainers = append(initContainers, prepareDataInitContainer(felisImage))
if server.Labels[v1alpha1.LabelSystemRole] == "" {
initContainers = append(initContainers, forwardingInitContainer(felisImage))
}
initContainers = append(initContainers, egressGateInitContainer(felisImage))
}
grace := graceSeconds(server)
@@ -446,6 +448,35 @@ func forwardingInitContainer(felisImage string) corev1.Container {
}
}
// egressGateWait bounds how long a server's start waits for its egress fence.
// kube-router programs a new pod's policy within a second; the rest is margin for
// a loaded node.
const egressGateWait = 30 * time.Second
// egressGateInitContainer holds the server image back until the pod's egress fence
// (platform.ServerEgressPolicies) is in effect. The policy is programmed after the
// pod starts, and live on k3s a new server-labelled pod reached felis-api's
// internal face on its first request; the server image and the plugins it loads
// are the owner's code and must never run inside that window. The gate is `felis
// egress-gate`, which the build pod runs first for the same reason.
//
// It fails open where the build's gate fails closed. An operator's
// --server-egress-allow-cidr can cover the node the Kubernetes API Service leads
// to, and then the probe answers forever while the fence stands; refusing would
// take every server down on a legitimate install. After egressGateWait the policy
// has landed if it ever will, so letting the server on costs nothing the fence
// would have given, and the gate's log says why the start was slow.
func egressGateInitContainer(felisImage string) corev1.Container {
return corev1.Container{
Name: "egress-gate",
Image: felisImage,
Command: []string{felisBinaryPath, "egress-gate",
"--wait", egressGateWait.String(), "--fail-open"},
Resources: initContainerResources(),
SecurityContext: hardenedContainerSecurityContext(true),
}
}
// prepareDataInitContainer runs `felis init-volume`, which chowns every world-volume
// entry not already owned by naming.GameUID:GameGID. It is the one container in the
// pod that runs as root, and it holds only what a chown walk needs: CHOWN to change
@@ -512,8 +543,8 @@ func hardenedContainerSecurityContext(readOnlyRoot bool) *corev1.SecurityContext
return sc
}
// initContainerResources bounds the two felis-image initContainers. Both are short
// file walks; the memory ceiling stops a pathological volume from taking the node's
// initContainerResources bounds the felis-image initContainers. Two are short file
// walks and the third a dial loop; the memory ceiling stops a pathological volume from taking the node's
// memory with it, and no CPU limit keeps a large world's chown from being throttled
// into the pod's start-up time.
func initContainerResources() corev1.ResourceRequirements {
+45 -6
View File
@@ -1,6 +1,7 @@
package operator
import (
"slices"
"testing"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
@@ -58,7 +59,7 @@ func TestReadinessProbeHTTPCustomPath(t *testing.T) {
// A user server (no system-role label) gets the forwarding-config initContainer after
// prepare-data, running the felis image and mounting the world volume. A system
// server gets only prepare-data, and a build with no felis image name gets neither.
// server gets no forwarding step, and a build with no felis image name gets no step.
func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
user := &v1alpha1.MinecraftServer{}
user.Spec.Storage.Size = "1Gi"
@@ -68,8 +69,8 @@ func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
t.Fatalf("buildStatefulSet: %v", err)
}
inits := sts.Spec.Template.Spec.InitContainers
if len(inits) != 2 || inits[0].Name != "prepare-data" || inits[1].Name != "init-forwarding" {
t.Fatalf("want [prepare-data init-forwarding], got %+v", inits)
if len(inits) != 3 || inits[0].Name != "prepare-data" || inits[1].Name != "init-forwarding" || inits[2].Name != "egress-gate" {
t.Fatalf("want [prepare-data init-forwarding egress-gate], got %+v", inits)
}
ic := inits[1]
if ic.Image != "felis:demo" {
@@ -113,13 +114,51 @@ func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
}
// System server handles forwarding in its own entrypoint, but its world still
// needs handing to the game uid.
// needs handing to the game uid and its image waits for the fence all the same.
sys := &v1alpha1.MinecraftServer{}
sys.Spec.Storage.Size = "1Gi"
sys.Labels = map[string]string{v1alpha1.LabelSystemRole: "lobby"}
sysSts, _ := buildStatefulSet(sys, 1, "felis:demo")
if got := sysSts.Spec.Template.Spec.InitContainers; len(got) != 1 || got[0].Name != "prepare-data" {
t.Errorf("system server must get only prepare-data, got %+v", got)
if got := sysSts.Spec.Template.Spec.InitContainers; len(got) != 2 || got[0].Name != "prepare-data" || got[1].Name != "egress-gate" {
t.Errorf("system server must get [prepare-data egress-gate], got %+v", got)
}
}
// Every server's image waits behind the egress gate, the last step before it: the
// pod's NetworkPolicy is programmed after the pod starts. The gate runs the felis
// image with nothing but the dial it needs, and it lets the server start after
// egressGateWait, where the build's gate refuses.
func TestBuildStatefulSetGatesEgress(t *testing.T) {
for _, role := range []string{"", naming.SystemLoginServer, "lobby"} {
server := &v1alpha1.MinecraftServer{}
server.Spec.Storage.Size = "1Gi"
if role != "" {
server.Labels = map[string]string{v1alpha1.LabelSystemRole: role}
}
sts, err := buildStatefulSet(server, 1, "felis:demo")
if err != nil {
t.Fatalf("buildStatefulSet: %v", err)
}
inits := sts.Spec.Template.Spec.InitContainers
gate := inits[len(inits)-1]
if gate.Name != "egress-gate" || gate.Image != "felis:demo" {
t.Fatalf("role %q: last init container = %s (%s), want egress-gate on the felis image", role, gate.Name, gate.Image)
}
want := []string{felisBinaryPath, "egress-gate", "--wait", "30s", "--fail-open"}
if !slices.Equal(gate.Command, want) || len(gate.Args) != 0 {
t.Errorf("role %q: gate runs %v %v, want %v", role, gate.Command, gate.Args, want)
}
if sc := gate.SecurityContext; sc == nil || sc.RunAsUser != nil || !dropsAll(sc) || len(sc.Capabilities.Add) != 0 ||
sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem ||
sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
t.Errorf("role %q: the gate must run unprivileged as the pod uid, got %+v", role, sc)
}
if len(gate.VolumeMounts) != 0 || len(gate.Env) != 0 || len(gate.EnvFrom) != 0 {
t.Errorf("role %q: the gate needs no volume and no secret, got mounts %+v env %+v %+v", role, gate.VolumeMounts, gate.Env, gate.EnvFrom)
}
if gate.Resources.Limits.Memory().IsZero() {
t.Errorf("role %q: the gate must carry a memory limit", role)
}
}
}