fix(operator): 服务器 pod 先等出站围栏生效再启动镜像,新 pod 第一个请求不再能连到内网
This commit is contained in:
5 files changed
+160
-20
No files matched your search
+22
-11
@@ -17,23 +17,28 @@ var (
|
||||
egressPollInterval = 200 * time.Millisecond
|
||||
)
|
||||
|
||||
// cmdEgressGate is the first initContainer of every build pod. The pod's
|
||||
// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s:
|
||||
// a build-labelled pod reached the internet and the Kubernetes API for its first
|
||||
// ~0.7 s), so the gate dials a destination the policy denies until it stops
|
||||
// answering, and only then lets the pod's next container, eventually the
|
||||
// untrusted Dockerfile, start.
|
||||
// cmdEgressGate is the first initContainer of every build pod and the last of
|
||||
// every game server pod. A pod's NetworkPolicy is programmed asynchronously after
|
||||
// the pod starts (live on k3s: a build-labelled pod reached the internet and the
|
||||
// Kubernetes API for its first ~0.7 s, a server-labelled one felis-api's internal
|
||||
// face on its first request), so the gate dials a destination the policy denies
|
||||
// until it stops answering, and only then lets the pod's next container, the
|
||||
// untrusted Dockerfile or server image, start.
|
||||
//
|
||||
// The default probe is the Kubernetes API Service, which the kubelet names in
|
||||
// every pod's environment and the build policy never admits. A probe that still
|
||||
// answers after --wait means the policy is not enforced at all (a CNI without
|
||||
// every pod's environment and neither policy admits. A probe that still answers
|
||||
// after --wait means the policy is not enforced at all (a CNI without
|
||||
// NetworkPolicy support, or k3s run with --disable-network-policy), and the
|
||||
// build fails closed.
|
||||
// build fails closed. A server passes --fail-open: an operator's
|
||||
// --server-egress-allow-cidr may cover the node the API Service leads to, so a
|
||||
// probe that keeps answering does not prove the fence is missing, and by then
|
||||
// the policy has had --wait to land.
|
||||
func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
|
||||
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused")
|
||||
probe := fs.String("probe", "", "host:port the pod's NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
|
||||
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the gate gives up")
|
||||
failOpen := fs.Bool("fail-open", false, "when --wait runs out, warn and let the pod go on instead of refusing it")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -56,6 +61,12 @@ func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
_ = conn.Close()
|
||||
if time.Since(start) >= *wait {
|
||||
if *failOpen {
|
||||
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s; starting anyway. Either this namespace's "+
|
||||
"NetworkPolicy is not enforced (a CNI without NetworkPolicy support, or k3s started with "+
|
||||
"--disable-network-policy), or an allowed CIDR admits the address behind it\n", *probe, *wait)
|
||||
return 0
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+
|
||||
"(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n",
|
||||
*probe, *wait)
|
||||
|
||||
@@ -75,6 +75,40 @@ func TestEgressGateRefusesAnOpenNetwork(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A server's gate waits out --wait all the same, then lets the pod start with a
|
||||
// warning in its log.
|
||||
func TestEgressGateFailOpenWaitsThenWarns(t *testing.T) {
|
||||
shrinkEgressGate(t)
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer ln.Close()
|
||||
go func() {
|
||||
for {
|
||||
c, err := ln.Accept()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
_ = c.Close()
|
||||
}
|
||||
}()
|
||||
var out, errb bytes.Buffer
|
||||
start := time.Now()
|
||||
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "150ms", "--fail-open"}, &out, &errb); code != 0 {
|
||||
t.Fatalf("exit %d, want 0: %s", code, errb.String())
|
||||
}
|
||||
if waited := time.Since(start); waited < 150*time.Millisecond {
|
||||
t.Errorf("gave up after %s, before --wait ran out", waited)
|
||||
}
|
||||
if !strings.Contains(errb.String(), ln.Addr().String()+" still answers after 150ms; starting anyway") {
|
||||
t.Errorf("stderr = %q", errb.String())
|
||||
}
|
||||
if out.Len() != 0 {
|
||||
t.Errorf("stdout = %q, want nothing: the lock was never seen", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) {
|
||||
shrinkEgressGate(t)
|
||||
t.Setenv("KUBERNETES_SERVICE_HOST", "")
|
||||
|
||||
@@ -118,6 +118,31 @@ kubectl describe pod <pod> # look at Events + container State
|
||||
to keep its state under `/data`. [GO-TESTED: `TestBuildStatefulSetRunsGameAsNonRoot`,
|
||||
`TestChownTreeHandsOverMismatchedEntries`; INTEGRATION-ONLY for the walk on a
|
||||
live volume.]
|
||||
- **Every start sits about 30 s in its last `Init:` step** → the `egress-gate`
|
||||
initContainer (`felis egress-gate`) holds the server image until the pod's
|
||||
egress fence (`felis-server-egress`) is in effect: kube-router programs a new
|
||||
pod's policy a moment after it starts, and a server-labelled pod reached
|
||||
felis-api's internal face on its first request. It dials the Kubernetes API
|
||||
Service, which no server is admitted to, and normally passes within a second.
|
||||
`kubectl logs <pod> -c egress-gate` says which way it went:
|
||||
|
||||
```
|
||||
felis egress-gate: 10.43.0.1:443 is unreachable after 201ms (...); the egress lock is in effect
|
||||
felis egress-gate: 10.43.0.1:443 still answers after 30s; starting anyway. ...
|
||||
```
|
||||
|
||||
The second line means the probe answered for the whole wait. Either the cluster
|
||||
runs without NetworkPolicy enforcement (a CNI without it, or k3s started with
|
||||
`--disable-network-policy`), which leaves every server off the fence, or a
|
||||
`--server-egress-allow-cidr` covers the node the API Service leads to, which
|
||||
admits servers to the node's PostgreSQL and kube API port as well. Narrow the
|
||||
CIDR to the host the servers need. The gate lets the server start after the
|
||||
wait either way, where the build's gate refuses. [GO-TESTED:
|
||||
`TestBuildStatefulSetGatesEgress`, `TestEgressGateFailOpenWaitsThenWarns`.
|
||||
VM-TESTED on k3s: without the gate a server-labelled pod's first request reached
|
||||
`felis-api-internal:8081`; behind it, three pods' first requests were refused
|
||||
after gates of about 200 ms, and a pod no policy selects started after the 5 s
|
||||
wait it was given, with the warning.]
|
||||
- **`FailedCreate … violates PodSecurity "baseline"`** on the StatefulSet or a
|
||||
Job → the minecraft namespace enforces the PodSecurity `baseline` profile
|
||||
(`pod-security.kubernetes.io/enforce=baseline`, set by the install bundle).
|
||||
|
||||
@@ -270,13 +270,15 @@ func buildStatefulSet(server *v1alpha1.MinecraftServer, replicas int32, felisIma
|
||||
// otherwise be read-only to it. An arbitrary user Paper image then gets the forwarding
|
||||
// config written for it (it does not consume FELIS_FORWARDING_SECRET itself); system
|
||||
// servers (login/lobby) are Felis-built and handle forwarding in their own
|
||||
// entrypoints. Without a felis image name there is nothing to run either step with.
|
||||
// entrypoints. Last, every server waits for its egress fence (egressGateInitContainer).
|
||||
// Without a felis image name there is nothing to run any step with.
|
||||
var initContainers []corev1.Container
|
||||
if felisImage != "" {
|
||||
initContainers = append(initContainers, prepareDataInitContainer(felisImage))
|
||||
if server.Labels[v1alpha1.LabelSystemRole] == "" {
|
||||
initContainers = append(initContainers, forwardingInitContainer(felisImage))
|
||||
}
|
||||
initContainers = append(initContainers, egressGateInitContainer(felisImage))
|
||||
}
|
||||
|
||||
grace := graceSeconds(server)
|
||||
@@ -446,6 +448,35 @@ func forwardingInitContainer(felisImage string) corev1.Container {
|
||||
}
|
||||
}
|
||||
|
||||
// egressGateWait bounds how long a server's start waits for its egress fence.
|
||||
// kube-router programs a new pod's policy within a second; the rest is margin for
|
||||
// a loaded node.
|
||||
const egressGateWait = 30 * time.Second
|
||||
|
||||
// egressGateInitContainer holds the server image back until the pod's egress fence
|
||||
// (platform.ServerEgressPolicies) is in effect. The policy is programmed after the
|
||||
// pod starts, and live on k3s a new server-labelled pod reached felis-api's
|
||||
// internal face on its first request; the server image and the plugins it loads
|
||||
// are the owner's code and must never run inside that window. The gate is `felis
|
||||
// egress-gate`, which the build pod runs first for the same reason.
|
||||
//
|
||||
// It fails open where the build's gate fails closed. An operator's
|
||||
// --server-egress-allow-cidr can cover the node the Kubernetes API Service leads
|
||||
// to, and then the probe answers forever while the fence stands; refusing would
|
||||
// take every server down on a legitimate install. After egressGateWait the policy
|
||||
// has landed if it ever will, so letting the server on costs nothing the fence
|
||||
// would have given, and the gate's log says why the start was slow.
|
||||
func egressGateInitContainer(felisImage string) corev1.Container {
|
||||
return corev1.Container{
|
||||
Name: "egress-gate",
|
||||
Image: felisImage,
|
||||
Command: []string{felisBinaryPath, "egress-gate",
|
||||
"--wait", egressGateWait.String(), "--fail-open"},
|
||||
Resources: initContainerResources(),
|
||||
SecurityContext: hardenedContainerSecurityContext(true),
|
||||
}
|
||||
}
|
||||
|
||||
// prepareDataInitContainer runs `felis init-volume`, which chowns every world-volume
|
||||
// entry not already owned by naming.GameUID:GameGID. It is the one container in the
|
||||
// pod that runs as root, and it holds only what a chown walk needs: CHOWN to change
|
||||
@@ -512,8 +543,8 @@ func hardenedContainerSecurityContext(readOnlyRoot bool) *corev1.SecurityContext
|
||||
return sc
|
||||
}
|
||||
|
||||
// initContainerResources bounds the two felis-image initContainers. Both are short
|
||||
// file walks; the memory ceiling stops a pathological volume from taking the node's
|
||||
// initContainerResources bounds the felis-image initContainers. Two are short file
|
||||
// walks and the third a dial loop; the memory ceiling stops a pathological volume from taking the node's
|
||||
// memory with it, and no CPU limit keeps a large world's chown from being throttled
|
||||
// into the pod's start-up time.
|
||||
func initContainerResources() corev1.ResourceRequirements {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package operator
|
||||
|
||||
import (
|
||||
"slices"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
@@ -58,7 +59,7 @@ func TestReadinessProbeHTTPCustomPath(t *testing.T) {
|
||||
|
||||
// A user server (no system-role label) gets the forwarding-config initContainer after
|
||||
// prepare-data, running the felis image and mounting the world volume. A system
|
||||
// server gets only prepare-data, and a build with no felis image name gets neither.
|
||||
// server gets no forwarding step, and a build with no felis image name gets no step.
|
||||
func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
|
||||
user := &v1alpha1.MinecraftServer{}
|
||||
user.Spec.Storage.Size = "1Gi"
|
||||
@@ -68,8 +69,8 @@ func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
|
||||
t.Fatalf("buildStatefulSet: %v", err)
|
||||
}
|
||||
inits := sts.Spec.Template.Spec.InitContainers
|
||||
if len(inits) != 2 || inits[0].Name != "prepare-data" || inits[1].Name != "init-forwarding" {
|
||||
t.Fatalf("want [prepare-data init-forwarding], got %+v", inits)
|
||||
if len(inits) != 3 || inits[0].Name != "prepare-data" || inits[1].Name != "init-forwarding" || inits[2].Name != "egress-gate" {
|
||||
t.Fatalf("want [prepare-data init-forwarding egress-gate], got %+v", inits)
|
||||
}
|
||||
ic := inits[1]
|
||||
if ic.Image != "felis:demo" {
|
||||
@@ -113,13 +114,51 @@ func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
|
||||
}
|
||||
|
||||
// System server handles forwarding in its own entrypoint, but its world still
|
||||
// needs handing to the game uid.
|
||||
// needs handing to the game uid and its image waits for the fence all the same.
|
||||
sys := &v1alpha1.MinecraftServer{}
|
||||
sys.Spec.Storage.Size = "1Gi"
|
||||
sys.Labels = map[string]string{v1alpha1.LabelSystemRole: "lobby"}
|
||||
sysSts, _ := buildStatefulSet(sys, 1, "felis:demo")
|
||||
if got := sysSts.Spec.Template.Spec.InitContainers; len(got) != 1 || got[0].Name != "prepare-data" {
|
||||
t.Errorf("system server must get only prepare-data, got %+v", got)
|
||||
if got := sysSts.Spec.Template.Spec.InitContainers; len(got) != 2 || got[0].Name != "prepare-data" || got[1].Name != "egress-gate" {
|
||||
t.Errorf("system server must get [prepare-data egress-gate], got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Every server's image waits behind the egress gate, the last step before it: the
|
||||
// pod's NetworkPolicy is programmed after the pod starts. The gate runs the felis
|
||||
// image with nothing but the dial it needs, and it lets the server start after
|
||||
// egressGateWait, where the build's gate refuses.
|
||||
func TestBuildStatefulSetGatesEgress(t *testing.T) {
|
||||
for _, role := range []string{"", naming.SystemLoginServer, "lobby"} {
|
||||
server := &v1alpha1.MinecraftServer{}
|
||||
server.Spec.Storage.Size = "1Gi"
|
||||
if role != "" {
|
||||
server.Labels = map[string]string{v1alpha1.LabelSystemRole: role}
|
||||
}
|
||||
sts, err := buildStatefulSet(server, 1, "felis:demo")
|
||||
if err != nil {
|
||||
t.Fatalf("buildStatefulSet: %v", err)
|
||||
}
|
||||
inits := sts.Spec.Template.Spec.InitContainers
|
||||
gate := inits[len(inits)-1]
|
||||
if gate.Name != "egress-gate" || gate.Image != "felis:demo" {
|
||||
t.Fatalf("role %q: last init container = %s (%s), want egress-gate on the felis image", role, gate.Name, gate.Image)
|
||||
}
|
||||
want := []string{felisBinaryPath, "egress-gate", "--wait", "30s", "--fail-open"}
|
||||
if !slices.Equal(gate.Command, want) || len(gate.Args) != 0 {
|
||||
t.Errorf("role %q: gate runs %v %v, want %v", role, gate.Command, gate.Args, want)
|
||||
}
|
||||
if sc := gate.SecurityContext; sc == nil || sc.RunAsUser != nil || !dropsAll(sc) || len(sc.Capabilities.Add) != 0 ||
|
||||
sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem ||
|
||||
sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation {
|
||||
t.Errorf("role %q: the gate must run unprivileged as the pod uid, got %+v", role, sc)
|
||||
}
|
||||
if len(gate.VolumeMounts) != 0 || len(gate.Env) != 0 || len(gate.EnvFrom) != 0 {
|
||||
t.Errorf("role %q: the gate needs no volume and no secret, got mounts %+v env %+v %+v", role, gate.VolumeMounts, gate.Env, gate.EnvFrom)
|
||||
}
|
||||
if gate.Resources.Limits.Memory().IsZero() {
|
||||
t.Errorf("role %q: the gate must carry a memory limit", role)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user