feat(netpol): 锁定游戏服出站并为 registry 加入站围栏
This commit is contained in:
13 files changed
+620
-37
No files matched your search
+10
-1
@@ -54,6 +54,10 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
fs.Var(&velocityCIDRs, "velocity-cidr", "CIDR of a Velocity proxy host allowed to reach game port 25565 (repeatable, REQUIRED)")
|
||||
var packageCIDRs multiFlag
|
||||
fs.Var(&packageCIDRs, "package-cidr", "CIDR of a package mirror build Pods may reach (repeatable; default none = no internet egress)")
|
||||
var serverDenyCIDRs multiFlag
|
||||
fs.Var(&serverDenyCIDRs, "server-egress-deny-cidr", "extra CIDR game server pods may never reach, e.g. the node's public address (repeatable)")
|
||||
var serverAllowCIDRs multiFlag
|
||||
fs.Var(&serverAllowCIDRs, "server-egress-allow-cidr", "private CIDR game server pods may reach despite the private-range block, e.g. a LAN database (repeatable)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -74,7 +78,9 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
"(the felis-api/operator Deployments run it and it is passed through as FELIS_IMAGE, e.g. --felis-image registry.felis.svc:5000/felis:v1)")
|
||||
return 2
|
||||
}
|
||||
for _, cidr := range append(append([]string{}, velocityCIDRs...), packageCIDRs...) {
|
||||
allCIDRs := append(append([]string{}, velocityCIDRs...), packageCIDRs...)
|
||||
allCIDRs = append(append(allCIDRs, serverDenyCIDRs...), serverAllowCIDRs...)
|
||||
for _, cidr := range allCIDRs {
|
||||
if _, _, err := net.ParseCIDR(cidr); err != nil {
|
||||
fmt.Fprintf(stderr, "felis manifests: invalid CIDR %q: %v\n", cidr, err)
|
||||
return 2
|
||||
@@ -148,6 +154,9 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
ArchiveLocalPath: *archiveLocalPath,
|
||||
VelocityCIDRs: []string(velocityCIDRs),
|
||||
PackageSourceCIDRs: []string(packageCIDRs),
|
||||
|
||||
ServerEgressDenyCIDRs: []string(serverDenyCIDRs),
|
||||
ServerEgressAllowCIDRs: []string(serverAllowCIDRs),
|
||||
})
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis manifests: render: %v\n", err)
|
||||
|
||||
+89
-2
@@ -299,6 +299,7 @@ die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
TEMP_PATHS=()
|
||||
DOCKER_CONTAINERS=()
|
||||
REGISTRY_DOCKER_CONFIG=""
|
||||
PKG_TIMERS_TO_RESTORE=()
|
||||
|
||||
on_error() {
|
||||
@@ -359,6 +360,40 @@ apply_felis_config_secrets() {
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
}
|
||||
|
||||
# The registry gate reads one token file per principal from felis-registry-auth
|
||||
# (registry namespace = control namespace); a build Job's push container reads the
|
||||
# build principal's username/password from felis-registry-push in the build
|
||||
# namespace. Values go through 0600 temp files, never kubectl's argv.
|
||||
apply_registry_secrets() {
|
||||
local dir
|
||||
dir="$(umask 077; mktemp -d)"
|
||||
remember_temp "$dir"
|
||||
printf '%s' "$REGISTRY_PLATFORM_TOKEN" > "${dir}/platform"
|
||||
printf '%s' "$REGISTRY_BUILD_TOKEN" > "${dir}/build"
|
||||
printf '%s' build > "${dir}/username"
|
||||
kube -n "$CONTROL_NS" create secret generic felis-registry-auth \
|
||||
--from-file=platform="${dir}/platform" \
|
||||
--from-file=build="${dir}/build" \
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
kube -n "$BUILD_NS" create secret generic felis-registry-push \
|
||||
--from-file=username="${dir}/username" \
|
||||
--from-file=password="${dir}/build" \
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
rm -rf -- "$dir"
|
||||
}
|
||||
|
||||
# node_global_cidrs prints one host-length CIDR per global address on this node.
|
||||
# Game server egress already excludes every private range; this adds the node's
|
||||
# public addresses, which would otherwise let a server dial the panel NodePort,
|
||||
# SSH, or anything else the host serves on them.
|
||||
node_global_cidrs() {
|
||||
command -v ip >/dev/null 2>&1 || return 0
|
||||
ip -o addr show scope global 2>/dev/null | awk '
|
||||
$3 == "inet" { split($4, a, "/"); print a[1] "/32" }
|
||||
$3 == "inet6" { split($4, a, "/"); print a[1] "/128" }
|
||||
' | sort -u
|
||||
}
|
||||
|
||||
as_postgres() {
|
||||
if command -v runuser >/dev/null 2>&1; then
|
||||
runuser -u postgres -- "$@"
|
||||
@@ -933,6 +968,30 @@ import_registry_image() {
|
||||
systemctl stop docker docker.socket 2>/dev/null || true
|
||||
}
|
||||
|
||||
# The registry pod runs registry:2 and, as its registry-gate sidecar, the felis
|
||||
# image — neither of which can be pulled from the registry they make up. A kubelet
|
||||
# image GC that collected either would leave the registry, and every pull through
|
||||
# it, dead until someone re-imported by hand. containerd reports an image labelled
|
||||
# io.cri-containerd.pinned=pinned as pinned over CRI, and kubelet's image GC never
|
||||
# removes a pinned image. Older felis/felis tags are unpinned first, so upgrades
|
||||
# do not pile up pinned images forever.
|
||||
pin_registry_images() {
|
||||
local ref
|
||||
while read -r ref; do
|
||||
case "$ref" in
|
||||
"$FELIS_IMAGE") ;;
|
||||
*/felis/felis:*) k3s_cmd ctr images label "$ref" io.cri-containerd.pinned= >/dev/null 2>&1 || true ;;
|
||||
esac
|
||||
done < <(k3s_cmd ctr images ls -q 2>/dev/null || true)
|
||||
for ref in "$FELIS_IMAGE" docker.io/library/registry:2; do
|
||||
if k3s_cmd ctr images label "$ref" io.cri-containerd.pinned=pinned >/dev/null 2>&1; then
|
||||
ok "pinned ${ref} in containerd (exempt from kubelet image GC)"
|
||||
else
|
||||
warn "could not pin ${ref} in containerd: if the kubelet's image GC collects it, the registry pod cannot restart until it is re-imported (docs/troubleshooting.md §8e)"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Source/binary + image build + containerd import
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -2042,6 +2101,12 @@ load_or_make_secrets() {
|
||||
# the username — i.e. anyone could join as anyone, the Owner included. Same value on
|
||||
# the proxy (forwarding.secret) and in every backend pod (felis-forwarding-secret).
|
||||
FORWARDING_SECRET="${FORWARDING_SECRET:-$(openssl rand -hex 32)}"
|
||||
# Registry write credentials, one per principal the registry gate knows
|
||||
# (internal/registrygate): platform pushes the installer's own images and the
|
||||
# Trivy DB mirrors, build is what a build Job's push container presents and may
|
||||
# never write under felis/ or mirror/. Reads stay anonymous.
|
||||
REGISTRY_PLATFORM_TOKEN="${REGISTRY_PLATFORM_TOKEN:-$(openssl rand -hex 32)}"
|
||||
REGISTRY_BUILD_TOKEN="${REGISTRY_BUILD_TOKEN:-$(openssl rand -hex 32)}"
|
||||
(
|
||||
umask 077
|
||||
cat > "$SECRETS_ENV" <<EOF
|
||||
@@ -2049,6 +2114,8 @@ DB_PASSWORD=${DB_PASSWORD}
|
||||
SERVICE_TOKEN=${SERVICE_TOKEN}
|
||||
SESSION_SECRET=${SESSION_SECRET}
|
||||
FORWARDING_SECRET=${FORWARDING_SECRET}
|
||||
REGISTRY_PLATFORM_TOKEN=${REGISTRY_PLATFORM_TOKEN}
|
||||
REGISTRY_BUILD_TOKEN=${REGISTRY_BUILD_TOKEN}
|
||||
EOF
|
||||
)
|
||||
chmod 0600 "$SECRETS_ENV"
|
||||
@@ -2324,7 +2391,7 @@ deploy_bundle() {
|
||||
kube create namespace "$ns" --dry-run=client -o yaml | kube apply -f -
|
||||
done
|
||||
|
||||
log "provisioning felis-config + felis-service-token + felis-forwarding-secret + panel TLS secrets (out-of-band, never in the bundle)"
|
||||
log "provisioning felis-config + felis-service-token + felis-forwarding-secret + registry credentials + panel TLS secrets (out-of-band, never in the bundle)"
|
||||
apply_felis_config_secrets
|
||||
apply_literal_secret "$CONTROL_NS" felis-service-token token "$SERVICE_TOKEN"
|
||||
# The build namespace needs the same token: the build Job's fetch initContainer
|
||||
@@ -2337,6 +2404,7 @@ deploy_bundle() {
|
||||
# forwarding mode is one proxy-wide setting — a backend that does not speak it is not
|
||||
# "less secure", it is unjoinable.
|
||||
apply_literal_secret "$CONTROL_NS" felis-forwarding-secret secret "$FORWARDING_SECRET"
|
||||
apply_registry_secrets
|
||||
kube -n "$CONTROL_NS" create secret tls felis-api-tls \
|
||||
--cert="$PANEL_TLS_CERT" \
|
||||
--key="$PANEL_TLS_KEY" \
|
||||
@@ -2348,6 +2416,10 @@ deploy_bundle() {
|
||||
--panel-node-port "$FELIS_PANEL_NODEPORT"
|
||||
--velocity-cidr "${NODE_IP}/32"
|
||||
)
|
||||
local cidr
|
||||
while read -r cidr; do
|
||||
[ -n "$cidr" ] && manifest_args+=(--server-egress-deny-cidr "$cidr")
|
||||
done < <(node_global_cidrs)
|
||||
# Backups are on by default (the renderer's own default names felis-backups); an emptied
|
||||
# FELIS_BACKUP_PVC asks for the no-backup shape explicitly, and a custom name must be
|
||||
# passed through or the api would advertise a PVC the bundle never created.
|
||||
@@ -2422,10 +2494,22 @@ push_image_to_registry() {
|
||||
esac
|
||||
log "mirroring ${ref} into the internal registry"
|
||||
docker tag "$ref" "$push_ref" || die "could not tag ${ref} as ${push_ref} — is docker healthy?"
|
||||
docker push "$push_ref" || die "could not mirror ${ref} into the internal registry — check the registry Deployment/pod and its PVC"
|
||||
docker --config "$REGISTRY_DOCKER_CONFIG" push "$push_ref" || die "could not mirror ${ref} into the internal registry — check the registry Deployment/pod (both the registry and registry-gate containers) and its PVC"
|
||||
docker rmi "$push_ref" >/dev/null 2>&1 || true
|
||||
}
|
||||
|
||||
# registry_docker_login logs a throwaway docker config into the registry gate as
|
||||
# the platform principal: writes are refused anonymously, and this identity is
|
||||
# the only one allowed under felis/. The config lives in a 0700 temp dir that the
|
||||
# EXIT trap removes, so the token never lands in root's ~/.docker.
|
||||
registry_docker_login() {
|
||||
REGISTRY_DOCKER_CONFIG="$(umask 077; mktemp -d)"
|
||||
remember_temp "$REGISTRY_DOCKER_CONFIG"
|
||||
printf '%s' "$REGISTRY_PLATFORM_TOKEN" | docker --config "$REGISTRY_DOCKER_CONFIG" \
|
||||
login --username platform --password-stdin "$REGISTRY_PUSH_HOST" >/dev/null \
|
||||
|| die "could not log in to the internal registry at ${REGISTRY_PUSH_HOST} as platform — check the registry-gate container's log and the felis-registry-auth Secret"
|
||||
}
|
||||
|
||||
# Every image this installer builds is hosted in the registry, so the copies it
|
||||
# imported into containerd are a first-boot cache, not the only copy: kubelet
|
||||
# re-pulls from the registry after any image GC. Runs AFTER deploy_bundle — the
|
||||
@@ -2439,6 +2523,7 @@ push_image_to_registry() {
|
||||
push_images_to_registry() {
|
||||
local img
|
||||
systemctl start docker
|
||||
registry_docker_login
|
||||
for img in "$FELIS_IMAGE" "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do
|
||||
[ -n "$img" ] || continue
|
||||
push_image_to_registry "$img"
|
||||
@@ -2871,6 +2956,8 @@ main() {
|
||||
fetch_source
|
||||
fi
|
||||
build_image
|
||||
# After build_image imported the felis image: the registry pod's gate runs it.
|
||||
pin_registry_images
|
||||
build_game_stack
|
||||
install_postgres
|
||||
configure_postgres
|
||||
|
||||
@@ -670,6 +670,7 @@ run_bundle_flags() { # backup-pvc worlds-host-path
|
||||
kube() { cat; }
|
||||
myManifests() { printf "%s\n" "$@"; }
|
||||
setfacl() { printf "SETFACL %s\n" "$*"; }
|
||||
node_global_cidrs() { printf "203.0.113.7/32\n2001:db8::7/128\n"; }
|
||||
run_bundle() {
|
||||
'"$mblock"'
|
||||
}
|
||||
@@ -685,6 +686,27 @@ esac
|
||||
|
||||
out="$(run_bundle_flags '' '')"
|
||||
expect "an emptied FELIS_BACKUP_PVC is the explicit no-backup shape" "--backup-pvc=" "$out"
|
||||
# Game server egress excludes every private range already; the node's own public
|
||||
# addresses must be excluded too or a server can dial the panel NodePort on them.
|
||||
expect "every global node address is denied to game server egress (v4)" "--server-egress-deny-cidr
|
||||
203.0.113.7/32" "$out"
|
||||
expect "every global node address is denied to game server egress (v6)" "--server-egress-deny-cidr
|
||||
2001:db8::7/128" "$out"
|
||||
|
||||
ngblock="$(awk '/^node_global_cidrs\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$ngblock" ] || { echo "FAIL: no node_global_cidrs found in $BS"; exit 1; }
|
||||
out="$(bash -c '
|
||||
ip() {
|
||||
printf "2: eth0 inet 203.0.113.7/24 brd 203.0.113.255 scope global eth0\\ valid_lft forever\n"
|
||||
printf "2: eth0 inet6 2001:db8::7/64 scope global dynamic\\ valid_lft 86000sec\n"
|
||||
}
|
||||
'"$ngblock"'
|
||||
node_global_cidrs')"
|
||||
expect "a global v4 address becomes a /32" "203.0.113.7/32" "$out"
|
||||
expect "a global v6 address becomes a /128" "2001:db8::7/128" "$out"
|
||||
case "$out" in
|
||||
*/24*|*/64*) echo "FAIL: node_global_cidrs must deny the address, not its whole subnet"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
out="$(run_bundle_flags felis-backups /var/lib/rancher/k3s/storage)"
|
||||
expect "enabling retention passes the worlds root" "--worlds-host-path
|
||||
@@ -758,7 +780,7 @@ pblock="$(awk '/^push_image_to_registry\(\) \{/,/^}/' "$BS")"
|
||||
|
||||
run_push() { # ref [docker-push-exit]
|
||||
REF="$1" PUSH_EXIT="${2:-0}" \
|
||||
REGISTRY_URL=registry.felis.svc:5000 REGISTRY_PUSH_HOST=127.0.0.1:5000 \
|
||||
REGISTRY_URL=registry.felis.svc:5000 REGISTRY_PUSH_HOST=127.0.0.1:5000 REGISTRY_DOCKER_CONFIG=/cfg \
|
||||
bash -c '
|
||||
log() { printf "LOG: %s\n" "$*"; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
@@ -766,9 +788,9 @@ run_push() { # ref [docker-push-exit]
|
||||
ok() { :; }
|
||||
systemctl() { :; }
|
||||
docker() {
|
||||
case "$1" in
|
||||
push) printf "DOCKER %s\n" "$*"; return "$PUSH_EXIT" ;;
|
||||
*) printf "DOCKER %s\n" "$*" ;;
|
||||
printf "DOCKER %s\n" "$*"
|
||||
case " $* " in
|
||||
*" push "*) return "$PUSH_EXIT" ;;
|
||||
esac
|
||||
}
|
||||
'"$pblock"'
|
||||
@@ -778,7 +800,7 @@ run_push() { # ref [docker-push-exit]
|
||||
out="$(run_push registry.felis.svc:5000/felis/felis:demo)"
|
||||
expect "a registry ref is re-tagged onto the node loopback endpoint" \
|
||||
"DOCKER tag registry.felis.svc:5000/felis/felis:demo 127.0.0.1:5000/felis/felis:demo" "$out"
|
||||
expect "and pushed to exactly that endpoint" "DOCKER push 127.0.0.1:5000/felis/felis:demo" "$out"
|
||||
expect "and pushed to exactly that endpoint, with the platform login" "DOCKER --config /cfg push 127.0.0.1:5000/felis/felis:demo" "$out"
|
||||
|
||||
out="$(run_push registry.felis.svc:50000/felis/felis:demo)"
|
||||
expect "a ref outside the registry is refused with a warning" "WARN: not mirroring" "$out"
|
||||
@@ -798,15 +820,70 @@ out="$(
|
||||
FELIS_IMAGE=a FELIS_LIMBO_IMAGE=b FELIS_LOBBY_IMAGE=c FELIS_PAPER_IMAGE=d bash -c '
|
||||
systemctl() { printf "SYSTEMCTL %s\n" "$*"; }
|
||||
push_image_to_registry() { printf "PUSH %s\n" "$1"; }
|
||||
registry_docker_login() { printf "LOGIN\n"; }
|
||||
'"$wiblock"'
|
||||
push_images_to_registry'
|
||||
)"
|
||||
expect "the batch logs in to the registry gate before pushing" "SYSTEMCTL start docker
|
||||
LOGIN
|
||||
PUSH a" "$out"
|
||||
starts="$(printf '%s\n' "$out" | grep -c 'SYSTEMCTL start docker')"
|
||||
stops="$(printf '%s\n' "$out" | grep -c 'SYSTEMCTL stop docker')"
|
||||
[ "$starts" = 1 ] && [ "$stops" = 1 ] && [ "$(printf '%s\n' "$out" | grep -c '^PUSH')" = 4 ] \
|
||||
&& echo "PASS the batch wraps all four pushes in ONE docker start/stop" \
|
||||
|| { echo "FAIL: expected 1 start / 1 stop / 4 pushes, got:"; printf '%s\n' "$out"; fails=$((fails + 1)); }
|
||||
|
||||
# The registry refuses anonymous writes, and the platform token must never reach
|
||||
# docker's argv (ps) or root's ~/.docker: stdin into a throwaway --config dir.
|
||||
lblock="$(awk '/^registry_docker_login\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$lblock" ] || { echo "FAIL: no registry_docker_login found in $BS"; exit 1; }
|
||||
calls="$(mktemp)"
|
||||
out="$(
|
||||
CALLS="$calls" REGISTRY_PLATFORM_TOKEN=s3cret REGISTRY_PUSH_HOST=127.0.0.1:5000 bash -c '
|
||||
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||
remember_temp() { printf "TEMP %s\n" "$1"; }
|
||||
docker() { printf "DOCKER %s STDIN=%s\n" "$*" "$(cat)" >>"$CALLS"; }
|
||||
'"$lblock"'
|
||||
registry_docker_login
|
||||
rm -rf "$REGISTRY_DOCKER_CONFIG"'
|
||||
)$(printf '\n'; cat "$calls")"
|
||||
rm -f "$calls"
|
||||
expect "the installer logs in as the platform principal via stdin" "login --username platform --password-stdin 127.0.0.1:5000 STDIN=s3cret" "$out"
|
||||
case "$(printf '%s\n' "$out" | grep '^DOCKER')" in
|
||||
*"DOCKER --config /"*) echo "PASS the login writes a throwaway docker config" ;;
|
||||
*) echo "FAIL: registry_docker_login must use a --config temp dir, got: $out"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
case "$out" in
|
||||
*"--password s3cret"*|*"-p s3cret"*) echo "FAIL: the registry token reached docker argv"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
expect "the throwaway config is registered for EXIT cleanup" "TEMP /" "$out"
|
||||
|
||||
# The registry pod's two images can only come from containerd's own store: pin
|
||||
# both against kubelet image GC, and unpin a previous felis tag.
|
||||
pnblock="$(awk '/^pin_registry_images\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$pnblock" ] || { echo "FAIL: no pin_registry_images found in $BS"; exit 1; }
|
||||
calls="$(mktemp)"
|
||||
out="$(
|
||||
CALLS="$calls" FELIS_IMAGE=registry.felis.svc:5000/felis/felis:v2 bash -c '
|
||||
ok() { printf "OK: %s\n" "$*"; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
k3s_cmd() {
|
||||
case "$*" in
|
||||
"ctr images ls -q") printf "registry.felis.svc:5000/felis/felis:v1\nregistry.felis.svc:5000/felis/felis:v2\ndocker.io/library/registry:2\nregistry.felis.svc:5000/felis/limbo:demo\n" ;;
|
||||
*) printf "CTR %s\n" "$*" >>"$CALLS" ;;
|
||||
esac
|
||||
}
|
||||
'"$pnblock"'
|
||||
pin_registry_images'
|
||||
)$(printf '\n'; cat "$calls")"
|
||||
rm -f "$calls"
|
||||
expect "the running felis image is pinned" "CTR ctr images label registry.felis.svc:5000/felis/felis:v2 io.cri-containerd.pinned=pinned" "$out"
|
||||
expect "registry:2 is pinned" "CTR ctr images label docker.io/library/registry:2 io.cri-containerd.pinned=pinned" "$out"
|
||||
expect "a previous felis tag is unpinned" "CTR ctr images label registry.felis.svc:5000/felis/felis:v1 io.cri-containerd.pinned=" "$out"
|
||||
case "$out" in
|
||||
*"limbo:demo io.cri"*) echo "FAIL: only the registry pod's images may be pinned or unpinned"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
# --- the registry's own image must not be re-pulled on every run --------------------------
|
||||
iblock="$(awk '/^import_registry_image\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$iblock" ] || { echo "FAIL: no import_registry_image found in $BS"; exit 1; }
|
||||
|
||||
@@ -143,11 +143,14 @@ set them by hand:
|
||||
pod's internal port 8081. That Service is deliberately separate from the external
|
||||
NodePort `felis-api` (443) so the no-Zero-Trust internal face is never published on
|
||||
a node's external IP.
|
||||
- **NetworkPolicy:** none is required today — neither the minecraft-namespace egress
|
||||
nor the control-namespace ingress is policy-locked, so the login pod's call to the
|
||||
API internal port is reachable. If a future deployment adds a minecraft egress lock
|
||||
or a control-namespace ingress fence, it must also open the login-pod →
|
||||
felis-api-internal (8081) path.
|
||||
- **NetworkPolicy:** the minecraft namespace is egress-locked
|
||||
(`felis-server-egress`: DNS plus the public internet, every private range
|
||||
excluded), so the internal API is unreachable from a game server by default.
|
||||
`felis-login-to-internal-api` opens exactly the login pod → felis-api (8081) path,
|
||||
selecting on the reserved `login` name AND the setup-owned
|
||||
`felis.lolicon.best/system-role=login` label the operator copies onto the pod — the
|
||||
same pair that decides who receives `FELIS_SERVICE_TOKEN`, so a user server cannot
|
||||
match it by picking a name.
|
||||
|
||||
The Velocity gate/lobby wiring is printed by `felis setup` and enforces the
|
||||
invariant: fresh connections hit `login` first, and only an authenticated release
|
||||
|
||||
@@ -60,9 +60,9 @@ A grep across `*.md` and `*.go` returns both sets; only the Go ones are seams.
|
||||
(recipe in docs/troubleshooting.md §8e); `trivy_java_db_repository` does the
|
||||
same for the Java DB, which Trivy fetches so soon as the scanned image contains
|
||||
a jar — i.e. for every real modpack build. Left unset on an egress-locked box
|
||||
the scan step fails closed — Kaniko pushes, Trivy exits on the DB download —
|
||||
which is the correct fail direction but leaves the build unfinished, so the
|
||||
mirrors are part of a production build install.
|
||||
the scan step fails closed — Trivy exits on the DB download before anything is
|
||||
pushed — which is the correct fail direction but leaves every build unfinished,
|
||||
so the mirrors are part of a production build install.
|
||||
|
||||
## Built; only its I/O is unverifiable from this repo
|
||||
|
||||
@@ -132,8 +132,8 @@ worth revisiting.
|
||||
|
||||
## Recorded outside the code
|
||||
|
||||
- `deploy/limbo/README.md:139` — no NetworkPolicy locks the minecraft-namespace
|
||||
egress or the control-namespace ingress today, which is why the login pod reaches
|
||||
`felis-api-internal:8081`. This is a conditional obligation rather than a seam: if
|
||||
a future deployment adds either lock, it must also open that path. Spec v4.1 §21
|
||||
asks for those policies; `cmd/felis/manifests.go` renders the game-port one.
|
||||
- The minecraft-namespace egress is locked (`felis-server-egress`, DNS plus the
|
||||
public internet with every private range and the node's own global addresses
|
||||
excluded) and `felis-login-to-internal-api` opens the one platform path a game pod
|
||||
needs — login → felis-api:8081. Any new in-cluster service a game server must call
|
||||
needs its own allow policy next to that one (`internal/platform/netpol.go`).
|
||||
+48
-14
@@ -386,14 +386,21 @@ internal registry.
|
||||
|
||||
`reconcileBuilds` polls the Job; a Job reaching `Failed` is surfaced via
|
||||
`writeBuildError` (JobPhase→Failed). [GO-TESTED for the mapping.] The underlying
|
||||
cause — a kaniko build error or the **Trivy CRITICAL-CVE gate** failing the build
|
||||
before push (spec §16) — is in the Job's pod logs and is [INTEGRATION-ONLY].
|
||||
Inspect:
|
||||
cause — a kaniko build error, the **Trivy CRITICAL-CVE gate** failing the build
|
||||
(spec §16), or the final push — is in the Job's pod logs and is
|
||||
[INTEGRATION-ONLY]. The pod runs `kaniko` (builds a tarball, never pushes) and
|
||||
`trivy` (scans that tarball) as init containers, then `push` — so a CVE-rejected
|
||||
image never reaches the registry. Inspect every step:
|
||||
|
||||
```
|
||||
kubectl logs -n felis-build job/<build-job>
|
||||
kubectl logs -n felis-build job/<build-job> --all-containers --prefix
|
||||
```
|
||||
|
||||
A `push` that fails with `403` means the target repository is under `felis/` or
|
||||
`mirror/` — the registry gate reserves those for the platform (§9); `401` means
|
||||
the `felis-registry-push` Secret in `felis-build` is missing or stale (re-run the
|
||||
installer).
|
||||
|
||||
### 8e. Build Pods never start: executor images and air-gapped installs
|
||||
|
||||
The build Job runs Kaniko and Trivy from external registries by default
|
||||
@@ -422,9 +429,13 @@ build_mem_limit = "4Gi"
|
||||
Mirror the executor images into the registry once. On the node itself, push
|
||||
through the loopback hostPort the registry Deployment binds (docker treats
|
||||
`127.0.0.1` as insecure by default; the installer leaves the daemon stopped, so
|
||||
`sudo systemctl start docker` first):
|
||||
`sudo systemctl start docker` first). The registry takes writes only from an
|
||||
authenticated principal, and `mirror/` only from `platform`, so log in with the
|
||||
platform token first:
|
||||
|
||||
```sh
|
||||
kubectl -n felis get secret felis-registry-auth -o jsonpath='{.data.platform}' | base64 -d \
|
||||
| docker login --username platform --password-stdin 127.0.0.1:5000
|
||||
docker pull gcr.io/kaniko-project/executor:v1.24.0 # any versions you pin
|
||||
docker pull aquasec/trivy:0.74.0
|
||||
docker pull mirror.gcr.io/aquasec/trivy-java-db:1
|
||||
@@ -434,6 +445,7 @@ docker tag mirror.gcr.io/aquasec/trivy-java-db:1 127.0.0.1:5000/mirror/trivy-ja
|
||||
docker push 127.0.0.1:5000/mirror/kaniko-executor:v1.24.0
|
||||
docker push 127.0.0.1:5000/mirror/trivy:0.74.0
|
||||
docker push 127.0.0.1:5000/mirror/trivy-java-db:1
|
||||
docker logout 127.0.0.1:5000
|
||||
```
|
||||
|
||||
From another machine, port-forward the registry instead (`kubectl -n felis
|
||||
@@ -460,20 +472,19 @@ Unset fields keep the defaults.
|
||||
`trivy_db_repository` is not optional on an egress-locked box. Trivy fetches its
|
||||
vulnerability DB from `mirror.gcr.io`/`ghcr.io` unless told otherwise, and the
|
||||
build egress policy denies those hosts — so the scan step fails closed
|
||||
(`failed to download vulnerability DB`) and NO build ever completes, even though
|
||||
Kaniko pushed the image. Mirror the DB into the internal registry once:
|
||||
(`failed to download vulnerability DB`), nothing is pushed, and NO build ever
|
||||
completes. Mirror the DB into the internal registry once:
|
||||
|
||||
```
|
||||
# On the node (docker treats 127.0.0.1 as insecure by default), or through the
|
||||
# port-forward above:
|
||||
# port-forward above, logged in as platform (see the block above):
|
||||
# docker pull mirror.gcr.io/aquasec/trivy-db:2
|
||||
# docker tag mirror.gcr.io/aquasec/trivy-db:2 127.0.0.1:5000/mirror/trivy-db:2
|
||||
# docker push 127.0.0.1:5000/mirror/trivy-db:2
|
||||
```
|
||||
|
||||
The Job's Trivy container already runs with `--insecure`, so the internal
|
||||
registry's plain HTTP works for the DB pull exactly as it does for the scanned
|
||||
image. Re-mirror the tag periodically (Trivy refreshes the DB several times a
|
||||
The Job's Trivy container runs with `--insecure`, so the internal registry's plain
|
||||
HTTP works for the DB pull; reads need no credential. Re-mirror the tag periodically (Trivy refreshes the DB several times a
|
||||
day upstream; a stale mirror only means stale CVE data, never a failed gate).
|
||||
|
||||
`trivy_java_db_repository` is the same story one step lazier: Trivy downloads
|
||||
@@ -510,6 +521,27 @@ control namespace (or `--registry-namespace`):
|
||||
control-plane pods' 256Mi — a live 475MB-layer push OOM-killed the 256Mi
|
||||
template mid-upload (audit #46). Very large layers need headroom here, not
|
||||
more CPU.
|
||||
- **Write authorization:** registry:2 listens on the pod's loopback only; the
|
||||
`registry-gate` sidecar (`felis registry-gate`, the felis image) owns the port and
|
||||
the hostPort. Reads are anonymous — containerd, Kaniko and Trivy pull without a
|
||||
credential — but an anonymous `GET /v2/` answers `401 Basic` so docker knows to
|
||||
send the credential on a push. Every write needs HTTP basic auth against a token
|
||||
in the `felis-registry-auth` Secret: `platform` may write anything (the
|
||||
installer's own images, the `mirror/` DB copies); `build` (a build Job's `push`
|
||||
container, via `felis-registry-push` in `felis-build`) may write anything outside
|
||||
`felis/` and `mirror/` and may never delete. A missing Secret leaves the registry
|
||||
read-only rather than down. The tokens persist in `/etc/felis/secrets.env`;
|
||||
rotating one means editing it there and re-running the installer, then
|
||||
`kubectl -n felis rollout restart deployment/registry` (the gate reads its tokens
|
||||
at start).
|
||||
- **Who can connect:** `felis-registry-ingress` admits only the `felis-build`
|
||||
namespace to the registry pod. Node-local traffic (containerd pulls, the
|
||||
installer's pushes through the hostPort) is always allowed by Kubernetes; game
|
||||
servers cannot reach it at all (`felis-server-egress`).
|
||||
- **GC pinning:** the registry pod's own images (registry:2 and the felis image
|
||||
its gate runs) cannot be pulled from the registry they make up, so the installer
|
||||
labels both `io.cri-containerd.pinned=pinned` in containerd and kubelet's image
|
||||
GC never collects them. Check with `k3s ctr images ls | grep pinned`.
|
||||
- **Selector quirk worth knowing:** the registry Service selector is only
|
||||
`name + component=registry` — it deliberately lacks the
|
||||
`part-of=felis-control-plane` label, so the registry is *invisible* to the
|
||||
@@ -812,15 +844,17 @@ If a pull does NOT come back:
|
||||
`/var/lib/rancher/k3s/storage`).
|
||||
2. Check the registry: `kubectl -n felis get pods -l
|
||||
app.kubernetes.io/component=registry` and, on the node,
|
||||
`curl -s http://127.0.0.1:5000/v2/` (expect `{}`).
|
||||
`curl -s -o /dev/null -w '%{http_code}\n' http://127.0.0.1:5000/healthz`
|
||||
(expect `200`: the registry gate answers it only while registry:2 behind it
|
||||
does). An anonymous `GET /v2/` answers `401` by design — see §9.
|
||||
3. Check the mirror file: `/etc/rancher/k3s/registries.yaml` must map
|
||||
`registry.felis.svc:5000` to `http://127.0.0.1:5000`. Missing or changed:
|
||||
re-run the installer (it rewrites the file and restarts k3s only when the
|
||||
content changed).
|
||||
4. Re-mirror a tag the registry does not have (hand-built images were never
|
||||
pushed): `sudo systemctl start docker` (the installer leaves the daemon
|
||||
stopped), then `docker tag <ref> 127.0.0.1:5000/<repo>:<tag> && docker push
|
||||
127.0.0.1:5000/<repo>:<tag>`.
|
||||
stopped), log in as `platform` (§8e), then `docker tag <ref>
|
||||
127.0.0.1:5000/<repo>:<tag> && docker push 127.0.0.1:5000/<repo>:<tag>`.
|
||||
|
||||
For an image that is in neither place, the old fallback still stands: re-run the
|
||||
installer (it rebuilds/re-imports from the local Docker store AND mirrors into
|
||||
|
||||
@@ -73,6 +73,18 @@ func labelsFor(server *v1alpha1.MinecraftServer) map[string]string {
|
||||
}
|
||||
}
|
||||
|
||||
// podLabelsFor is labelsFor plus the setup-owned system-role label, copied onto
|
||||
// the pod so the platform's NetworkPolicies can tell the login gate apart from a
|
||||
// user server (internal/platform loginToInternalAPI). Only the template carries it:
|
||||
// the StatefulSet selector is immutable and stays selectorFor.
|
||||
func podLabelsFor(server *v1alpha1.MinecraftServer) map[string]string {
|
||||
l := labelsFor(server)
|
||||
if role := server.Labels[v1alpha1.LabelSystemRole]; role != "" {
|
||||
l[v1alpha1.LabelSystemRole] = role
|
||||
}
|
||||
return l
|
||||
}
|
||||
|
||||
func headlessServiceName(name string) string { return name + "-hl" }
|
||||
|
||||
// rconPort resolves the RCON port, defaulting to the conventional DefaultRconPort.
|
||||
@@ -274,7 +286,7 @@ func buildStatefulSet(server *v1alpha1.MinecraftServer, replicas int32, felisIma
|
||||
ServiceName: headlessServiceName(server.Name),
|
||||
Selector: &metav1.LabelSelector{MatchLabels: selectorFor(server)},
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: labelsFor(server)},
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: podLabelsFor(server)},
|
||||
Spec: corev1.PodSpec{
|
||||
TerminationGracePeriodSeconds: &grace,
|
||||
InitContainers: initContainers,
|
||||
|
||||
@@ -119,6 +119,40 @@ func TestBuildStatefulSetForwardingInitContainer(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The system-role label travels onto the pod: the platform's NetworkPolicies select
|
||||
// on it (felis-login-to-internal-api opens 8081 only to name=login AND
|
||||
// system-role=login), and a pod without it would be fenced off the one service it
|
||||
// exists to call. The selector stays the immutable labelsFor subset, so existing
|
||||
// StatefulSets roll instead of failing to update.
|
||||
func TestBuildStatefulSetCopiesSystemRoleOntoPods(t *testing.T) {
|
||||
login := &v1alpha1.MinecraftServer{}
|
||||
login.Name = naming.SystemLoginServer
|
||||
login.Labels = map[string]string{v1alpha1.LabelSystemRole: naming.SystemLoginServer}
|
||||
sts, err := buildStatefulSet(login, 1, "felis:demo")
|
||||
if err != nil {
|
||||
t.Fatalf("buildStatefulSet: %v", err)
|
||||
}
|
||||
pod := sts.Spec.Template.Labels
|
||||
if pod[v1alpha1.LabelSystemRole] != naming.SystemLoginServer {
|
||||
t.Errorf("pod labels = %v, want %s=%s", pod, v1alpha1.LabelSystemRole, naming.SystemLoginServer)
|
||||
}
|
||||
for k, v := range sts.Spec.Selector.MatchLabels {
|
||||
if pod[k] != v {
|
||||
t.Errorf("selector %s=%s does not match the pod template", k, v)
|
||||
}
|
||||
}
|
||||
if _, ok := sts.Spec.Selector.MatchLabels[v1alpha1.LabelSystemRole]; ok {
|
||||
t.Error("the system role must stay out of the (immutable) selector")
|
||||
}
|
||||
|
||||
user := &v1alpha1.MinecraftServer{}
|
||||
user.Name = "survival"
|
||||
userSts, _ := buildStatefulSet(user, 1, "felis:demo")
|
||||
if _, ok := userSts.Spec.Template.Labels[v1alpha1.LabelSystemRole]; ok {
|
||||
t.Errorf("a user server pod must carry no system role, got %v", userSts.Spec.Template.Labels)
|
||||
}
|
||||
}
|
||||
|
||||
// A server with a health port also exposes it as a named container port so the
|
||||
// kubelet can reach it.
|
||||
func TestBuildStatefulSetAddsHealthPort(t *testing.T) {
|
||||
|
||||
@@ -89,10 +89,15 @@ func Objects(p Params) []Object {
|
||||
buildNP.TypeMeta = metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"}
|
||||
objs = append(objs, buildNP)
|
||||
|
||||
// Minecraft-namespace ingress fence (default-deny + RCON + game).
|
||||
// Minecraft-namespace ingress fence (default-deny + RCON + game), the server
|
||||
// egress fence, and the registry's ingress fence.
|
||||
for _, np := range MinecraftNetworkPolicies(p) {
|
||||
objs = append(objs, np)
|
||||
}
|
||||
for _, np := range ServerEgressPolicies(p) {
|
||||
objs = append(objs, np)
|
||||
}
|
||||
objs = append(objs, RegistryIngressPolicy(p))
|
||||
|
||||
// The running control-plane the fence protects: felis-api/operator Deployments
|
||||
// (which bind the SAs to workloads and stamp the RCON-peer labels) and the
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
@@ -21,6 +22,33 @@ func TestObjects_EveryDocHasTypeMeta(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestObjects_CarryTheFences checks the bundle ships every NetworkPolicy the
|
||||
// security model counts on, each in the namespace it guards. Rendering one is
|
||||
// worth nothing if Objects forgets to include it.
|
||||
func TestObjects_CarryTheFences(t *testing.T) {
|
||||
want := map[string]string{
|
||||
"felis-default-deny-ingress": "minecraft",
|
||||
"felis-server-egress": "minecraft",
|
||||
"felis-login-to-internal-api": "minecraft",
|
||||
"felis-registry-ingress": "felis",
|
||||
}
|
||||
for _, obj := range Objects(testParams()) {
|
||||
np, ok := obj.(*networkingv1.NetworkPolicy)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if ns, expected := want[np.Name]; expected {
|
||||
if np.Namespace != ns {
|
||||
t.Errorf("%s in namespace %q, want %q", np.Name, np.Namespace, ns)
|
||||
}
|
||||
delete(want, np.Name)
|
||||
}
|
||||
}
|
||||
for name := range want {
|
||||
t.Errorf("bundle is missing NetworkPolicy %s", name)
|
||||
}
|
||||
}
|
||||
|
||||
// TestObjects_NamespacesLabeled checks the three namespaces are rendered with the
|
||||
// immutable name label the NetworkPolicy namespaceSelectors key on.
|
||||
func TestObjects_NamespacesLabeled(t *testing.T) {
|
||||
|
||||
@@ -98,6 +98,16 @@ type Params struct {
|
||||
// Pods (spec §16). Empty means no internet egress at all — the locked-down
|
||||
// default the build subsystem already enforces.
|
||||
PackageSourceCIDRs []string
|
||||
// ServerEgressDenyCIDRs are extra destinations game server pods may never
|
||||
// reach, on top of the private, link-local and loopback ranges the server
|
||||
// egress policy always excludes. The installer passes the node's own global
|
||||
// addresses: a node with a public IP would otherwise be reachable from a
|
||||
// tenant's plugin on every host port (PostgreSQL, the kube API, kubelet).
|
||||
ServerEgressDenyCIDRs []string
|
||||
// ServerEgressAllowCIDRs are private destinations game servers MAY reach
|
||||
// despite that exclusion, e.g. a LAN database a server's plugin uses. Empty by
|
||||
// default: a tenant's code has no business on the operator's network.
|
||||
ServerEgressAllowCIDRs []string
|
||||
// FelisImage is the container image the felis-api and felis-operator
|
||||
// Deployments run (the multi-call `felis` binary). It has NO default and no
|
||||
// safe guess: `felis manifests` REQUIRES --felis-image and refuses to render
|
||||
|
||||
@@ -1,7 +1,11 @@
|
||||
package platform
|
||||
|
||||
import (
|
||||
"net"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/build"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
@@ -124,6 +128,135 @@ func allowGameFromVelocity(p Params) *networkingv1.NetworkPolicy {
|
||||
return netpol("felis-allow-game-from-velocity", p.MinecraftNamespace, serverPodSelector(), ingress)
|
||||
}
|
||||
|
||||
// serverEgressExceptV4 / V6 are the destinations a game server never reaches
|
||||
// through the internet rule: every private, shared, link-local, loopback,
|
||||
// multicast and reserved range. They cover the pod and Service CIDRs of any stock
|
||||
// k3s/k8s install (10.42/16, 10.43/16), the node's private addresses, cloud
|
||||
// metadata (169.254.169.254) and the operator's LAN.
|
||||
var (
|
||||
serverEgressExceptV4 = []string{
|
||||
"0.0.0.0/8", "10.0.0.0/8", "100.64.0.0/10", "127.0.0.0/8", "169.254.0.0/16",
|
||||
"172.16.0.0/12", "192.168.0.0/16", "224.0.0.0/4", "240.0.0.0/4",
|
||||
}
|
||||
serverEgressExceptV6 = []string{"::1/128", "fc00::/7", "fe80::/10", "ff00::/8"}
|
||||
)
|
||||
|
||||
// ServerEgressPolicies render the egress fence for game server pods. A server runs
|
||||
// code its owner chose — plugins, mods, a whole image — so the namespace used to
|
||||
// be a launch pad: any server could push to the unauthenticated registry, dial
|
||||
// felis-api's internal face, PostgreSQL on the node, or the kube API. Now:
|
||||
//
|
||||
// - every server may resolve names and reach the public internet (plugin
|
||||
// updates, resource packs, web maps) and nothing private;
|
||||
// - the login system server additionally reaches felis-api's internal face,
|
||||
// the one platform service it is built to call.
|
||||
//
|
||||
// Policies are additive, so the login pod gets the union of both.
|
||||
func ServerEgressPolicies(p Params) []*networkingv1.NetworkPolicy {
|
||||
p = p.withDefaults()
|
||||
return []*networkingv1.NetworkPolicy{serverEgress(p), loginToInternalAPI(p)}
|
||||
}
|
||||
|
||||
func serverEgress(p Params) *networkingv1.NetworkPolicy {
|
||||
v4 := append([]string{}, serverEgressExceptV4...)
|
||||
v6 := append([]string{}, serverEgressExceptV6...)
|
||||
for _, c := range p.ServerEgressDenyCIDRs {
|
||||
_, n, err := net.ParseCIDR(c)
|
||||
if err != nil {
|
||||
continue // `felis manifests` rejects these before rendering
|
||||
}
|
||||
if n.IP.To4() != nil {
|
||||
v4 = append(v4, n.String())
|
||||
} else {
|
||||
v6 = append(v6, n.String())
|
||||
}
|
||||
}
|
||||
peers := []networkingv1.NetworkPolicyPeer{
|
||||
{IPBlock: &networkingv1.IPBlock{CIDR: "0.0.0.0/0", Except: v4}},
|
||||
{IPBlock: &networkingv1.IPBlock{CIDR: "::/0", Except: v6}},
|
||||
}
|
||||
for _, c := range p.ServerEgressAllowCIDRs {
|
||||
if _, n, err := net.ParseCIDR(c); err == nil {
|
||||
peers = append(peers, networkingv1.NetworkPolicyPeer{IPBlock: &networkingv1.IPBlock{CIDR: n.String()}})
|
||||
}
|
||||
}
|
||||
udp, tcp := corev1.ProtocolUDP, corev1.ProtocolTCP
|
||||
dns := intstr.FromInt32(53)
|
||||
return &networkingv1.NetworkPolicy{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-server-egress", Namespace: p.MinecraftNamespace},
|
||||
Spec: networkingv1.NetworkPolicySpec{
|
||||
PodSelector: serverPodSelector(),
|
||||
PolicyTypes: []networkingv1.PolicyType{networkingv1.PolicyTypeEgress},
|
||||
Egress: []networkingv1.NetworkPolicyEgressRule{
|
||||
// Name resolution through the cluster resolver: the DNS Service
|
||||
// sits inside 10/8, which the internet rule excludes.
|
||||
{
|
||||
To: []networkingv1.NetworkPolicyPeer{build.ClusterDNSPeer()},
|
||||
Ports: []networkingv1.NetworkPolicyPort{{Protocol: &udp, Port: &dns}, {Protocol: &tcp, Port: &dns}},
|
||||
},
|
||||
{To: peers},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// loginToInternalAPI opens felis-api's internal face (8081) to the login system
|
||||
// server only. The selector needs both the reserved name and the setup-owned
|
||||
// system-role label the operator copies onto that pod — the same pair that decides
|
||||
// who receives FELIS_SERVICE_TOKEN (internal/operator buildEnv), so a user server
|
||||
// can never match it by picking a name.
|
||||
func loginToInternalAPI(p Params) *networkingv1.NetworkPolicy {
|
||||
tcp := corev1.ProtocolTCP
|
||||
port := intstr.FromInt32(apiInternalPort)
|
||||
sel := serverPodSelector()
|
||||
sel.MatchLabels[v1alpha1.LabelServer] = naming.SystemLoginServer
|
||||
sel.MatchLabels[v1alpha1.LabelSystemRole] = naming.SystemLoginServer
|
||||
return &networkingv1.NetworkPolicy{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-login-to-internal-api", Namespace: p.MinecraftNamespace},
|
||||
Spec: networkingv1.NetworkPolicySpec{
|
||||
PodSelector: sel,
|
||||
PolicyTypes: []networkingv1.PolicyType{networkingv1.PolicyTypeEgress},
|
||||
Egress: []networkingv1.NetworkPolicyEgressRule{{
|
||||
To: []networkingv1.NetworkPolicyPeer{{
|
||||
NamespaceSelector: &metav1.LabelSelector{
|
||||
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.ControlNamespace},
|
||||
},
|
||||
PodSelector: &metav1.LabelSelector{MatchLabels: map[string]string{
|
||||
LabelPartOf: controlPlanePartOf,
|
||||
LabelComponent: ComponentAPI,
|
||||
}},
|
||||
}},
|
||||
Ports: []networkingv1.NetworkPolicyPort{{Protocol: &tcp, Port: &port}},
|
||||
}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// RegistryIngressPolicy fences the registry pod: only build pods reach its port.
|
||||
// Everything else that uses the registry runs on the node — containerd's pulls and
|
||||
// the installer's pushes both arrive through the loopback hostPort — and Kubernetes
|
||||
// never blocks resident-node traffic. Write authorization is the gate's job; this
|
||||
// policy keeps every other pod from even trying.
|
||||
func RegistryIngressPolicy(p Params) *networkingv1.NetworkPolicy {
|
||||
p = p.withDefaults()
|
||||
tcp := corev1.ProtocolTCP
|
||||
port := intstr.FromInt32(p.RegistryPort)
|
||||
np := netpol("felis-registry-ingress", p.RegistryNamespace,
|
||||
metav1.LabelSelector{MatchLabels: registryLabels()},
|
||||
[]networkingv1.NetworkPolicyIngressRule{{
|
||||
From: []networkingv1.NetworkPolicyPeer{{
|
||||
NamespaceSelector: &metav1.LabelSelector{
|
||||
MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.BuildNamespace},
|
||||
},
|
||||
}},
|
||||
Ports: []networkingv1.NetworkPolicyPort{{Protocol: &tcp, Port: &port}},
|
||||
}},
|
||||
)
|
||||
return np
|
||||
}
|
||||
|
||||
// netpol assembles an ingress-only NetworkPolicy. A nil/empty ingress slice with
|
||||
// PolicyTypeIngress is the canonical "deny all ingress" shape.
|
||||
func netpol(name, ns string, sel metav1.LabelSelector, ingress []networkingv1.NetworkPolicyIngressRule) *networkingv1.NetworkPolicy {
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"felis.lolicon.best/internal/operator"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/labels"
|
||||
)
|
||||
|
||||
func npByName(t *testing.T, nps []*networkingv1.NetworkPolicy, name string) *networkingv1.NetworkPolicy {
|
||||
@@ -190,3 +191,153 @@ func assertSinglePort(t *testing.T, ports []networkingv1.NetworkPolicyPort, want
|
||||
t.Errorf("protocol = %v, want TCP", ports[0].Protocol)
|
||||
}
|
||||
}
|
||||
|
||||
// TestServerEgress_OnlyDNSAndPublicInternet pins the egress fence on game server
|
||||
// pods: DNS by port, the public internet by address, and every private range
|
||||
// carved out of it — the pod and Service CIDRs, the node's LAN, metadata.
|
||||
func TestServerEgress_OnlyDNSAndPublicInternet(t *testing.T) {
|
||||
p := testParams()
|
||||
p.ServerEgressDenyCIDRs = []string{"203.0.113.7/32", "2001:db8::1/128"}
|
||||
p.ServerEgressAllowCIDRs = []string{"10.9.8.0/24"}
|
||||
eg := npByName(t, ServerEgressPolicies(p), "felis-server-egress")
|
||||
|
||||
if !hasPolicyType(eg, networkingv1.PolicyTypeEgress) || hasPolicyType(eg, networkingv1.PolicyTypeIngress) {
|
||||
t.Fatalf("server egress policy types = %v, want Egress only (ingress stays with the default-deny set)", eg.Spec.PolicyTypes)
|
||||
}
|
||||
if !selectorEquals(eg.Spec.PodSelector, serverPodSelector()) {
|
||||
t.Errorf("server egress selects %v, want every server pod", eg.Spec.PodSelector)
|
||||
}
|
||||
if len(eg.Spec.Egress) != 2 {
|
||||
t.Fatalf("egress rules = %d, want DNS + internet", len(eg.Spec.Egress))
|
||||
}
|
||||
dns := eg.Spec.Egress[0]
|
||||
if len(dns.To) != 1 || dns.To[0].PodSelector == nil || dns.To[0].PodSelector.MatchLabels["k8s-app"] != "kube-dns" || len(dns.Ports) != 2 {
|
||||
t.Errorf("DNS rule = %+v, want port 53 udp+tcp to the cluster DNS pods", dns)
|
||||
}
|
||||
for _, port := range dns.Ports {
|
||||
if port.Port == nil || port.Port.IntValue() != 53 {
|
||||
t.Errorf("DNS rule port = %v, want 53", port.Port)
|
||||
}
|
||||
}
|
||||
|
||||
net := eg.Spec.Egress[1]
|
||||
if len(net.Ports) != 0 {
|
||||
t.Errorf("internet rule must not be port-restricted, got %v", net.Ports)
|
||||
}
|
||||
blocks := map[string][]string{}
|
||||
for _, peer := range net.To {
|
||||
if peer.IPBlock == nil || peer.PodSelector != nil || peer.NamespaceSelector != nil {
|
||||
t.Fatalf("internet rule peer %+v must be a bare ipBlock", peer)
|
||||
}
|
||||
blocks[peer.IPBlock.CIDR] = peer.IPBlock.Except
|
||||
}
|
||||
for _, want := range []string{"10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16", "100.64.0.0/10", "169.254.0.0/16", "127.0.0.0/8", "203.0.113.7/32"} {
|
||||
if !contains(blocks["0.0.0.0/0"], want) {
|
||||
t.Errorf("0.0.0.0/0 except = %v, missing %s", blocks["0.0.0.0/0"], want)
|
||||
}
|
||||
}
|
||||
for _, want := range []string{"fc00::/7", "fe80::/10", "::1/128", "2001:db8::1/128"} {
|
||||
if !contains(blocks["::/0"], want) {
|
||||
t.Errorf("::/0 except = %v, missing %s", blocks["::/0"], want)
|
||||
}
|
||||
}
|
||||
if except, ok := blocks["10.9.8.0/24"]; !ok || len(except) != 0 {
|
||||
t.Errorf("allow CIDR 10.9.8.0/24 must be its own peer, blocks=%v", blocks)
|
||||
}
|
||||
if len(blocks) != 3 {
|
||||
t.Errorf("internet rule peers = %v, want v4 + v6 + one allow CIDR", blocks)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoginToInternalAPI_SelectsOnlyTheSystemLoginPod pins the one hole in the
|
||||
// server egress fence: felis-api's internal face on 8081, for the pod that is both
|
||||
// named login AND carries the setup-owned system-role label.
|
||||
func TestLoginToInternalAPI_SelectsOnlyTheSystemLoginPod(t *testing.T) {
|
||||
p := testParams().withDefaults()
|
||||
np := npByName(t, ServerEgressPolicies(p), "felis-login-to-internal-api")
|
||||
sel, err := metav1.LabelSelectorAsSelector(&np.Spec.PodSelector)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
server := map[string]string{
|
||||
v1alpha1.LabelManagedBy: operator.ManagedByValue,
|
||||
v1alpha1.LabelComponent: operator.ComponentValue,
|
||||
}
|
||||
with := func(extra map[string]string) labels.Set {
|
||||
m := labels.Set{}
|
||||
for k, v := range server {
|
||||
m[k] = v
|
||||
}
|
||||
for k, v := range extra {
|
||||
m[k] = v
|
||||
}
|
||||
return m
|
||||
}
|
||||
if !sel.Matches(with(map[string]string{v1alpha1.LabelServer: "login", v1alpha1.LabelSystemRole: "login"})) {
|
||||
t.Error("the system login pod must match")
|
||||
}
|
||||
for name, l := range map[string]map[string]string{
|
||||
"user server": {v1alpha1.LabelServer: "survival"},
|
||||
"login name without the role": {v1alpha1.LabelServer: "login"},
|
||||
"role label on another server": {v1alpha1.LabelServer: "survival", v1alpha1.LabelSystemRole: "login"},
|
||||
"lobby": {v1alpha1.LabelServer: "lobby", v1alpha1.LabelSystemRole: "lobby"},
|
||||
} {
|
||||
if sel.Matches(with(l)) {
|
||||
t.Errorf("%s must not reach felis-api's internal face", name)
|
||||
}
|
||||
}
|
||||
|
||||
if len(np.Spec.Egress) != 1 || len(np.Spec.Egress[0].To) != 1 {
|
||||
t.Fatalf("login egress shape = %+v, want one rule with one peer", np.Spec.Egress)
|
||||
}
|
||||
rule := np.Spec.Egress[0]
|
||||
assertSinglePort(t, rule.Ports, int(apiInternalPort))
|
||||
peer := rule.To[0]
|
||||
if peer.NamespaceSelector == nil || peer.NamespaceSelector.MatchLabels["kubernetes.io/metadata.name"] != p.ControlNamespace {
|
||||
t.Errorf("login egress namespace = %v, want %s", peer.NamespaceSelector, p.ControlNamespace)
|
||||
}
|
||||
if peer.PodSelector == nil || !mapSelectorMatches(peer.PodSelector.MatchLabels, APIDeployment(p).Spec.Template.Labels) {
|
||||
t.Errorf("login egress pod selector %v must select the api pod", peer.PodSelector)
|
||||
}
|
||||
if mapSelectorMatches(peer.PodSelector.MatchLabels, OperatorDeployment(p).Spec.Template.Labels) {
|
||||
t.Error("login egress must not reach the operator")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRegistryIngress_BuildNamespaceOnly pins who may dial the registry pod: build
|
||||
// pods, on the registry port. Game servers and the control plane never pull
|
||||
// through the Service — containerd pulls over the node's loopback hostPort.
|
||||
func TestRegistryIngress_BuildNamespaceOnly(t *testing.T) {
|
||||
p := testParams().withDefaults()
|
||||
np := RegistryIngressPolicy(p)
|
||||
if np.Namespace != p.RegistryNamespace {
|
||||
t.Errorf("registry ingress namespace = %s, want %s", np.Namespace, p.RegistryNamespace)
|
||||
}
|
||||
if !mapSelectorMatches(np.Spec.PodSelector.MatchLabels, registryDeployment(p).Spec.Template.Labels) {
|
||||
t.Errorf("registry ingress selector %v does not select the registry pod", np.Spec.PodSelector)
|
||||
}
|
||||
if mapSelectorMatches(np.Spec.PodSelector.MatchLabels, APIDeployment(p).Spec.Template.Labels) {
|
||||
t.Error("registry ingress must not also fence the api pod")
|
||||
}
|
||||
if len(np.Spec.Ingress) != 1 || len(np.Spec.Ingress[0].From) != 1 {
|
||||
t.Fatalf("registry ingress shape = %+v, want one rule, one peer", np.Spec.Ingress)
|
||||
}
|
||||
peer := np.Spec.Ingress[0].From[0]
|
||||
if peer.PodSelector != nil || peer.IPBlock != nil || peer.NamespaceSelector == nil ||
|
||||
peer.NamespaceSelector.MatchLabels["kubernetes.io/metadata.name"] != p.BuildNamespace {
|
||||
t.Errorf("registry ingress peer = %+v, want the whole %s namespace", peer, p.BuildNamespace)
|
||||
}
|
||||
assertSinglePort(t, np.Spec.Ingress[0].Ports, int(p.RegistryPort))
|
||||
}
|
||||
|
||||
func selectorEquals(a, b metav1.LabelSelector) bool {
|
||||
if len(a.MatchLabels) != len(b.MatchLabels) || len(a.MatchExpressions)+len(b.MatchExpressions) != 0 {
|
||||
return false
|
||||
}
|
||||
for k, v := range a.MatchLabels {
|
||||
if b.MatchLabels[k] != v {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
Reference in new issue
Block a user