feat(platform): 渲染 k3s 内的 felis-postgres 及其入站隔离
This commit is contained in:
9 files changed
+770
-15
No files matched your search
@@ -111,6 +111,11 @@ func Objects(p Params) []Object {
|
||||
}
|
||||
objs = append(objs, RegistryIngressPolicy(p))
|
||||
|
||||
// The database's pg_hba.conf and ingress fence. deploy/bootstrap.sh applies
|
||||
// the database alone first (PostgresObjects); rendering it here as well keeps
|
||||
// the full apply from drifting it.
|
||||
objs = append(objs, postgresHBAConfig(p), PostgresIngressPolicy(p))
|
||||
|
||||
// The running control-plane the fence protects: felis-api/operator Deployments
|
||||
// (which bind the SAs to workloads and stamp the RCON-peer labels) and the
|
||||
// in-cluster registry (Deployment + Service + PVC) the build egress policy
|
||||
@@ -124,8 +129,13 @@ func Objects(p Params) []Object {
|
||||
// `---`-separated form kubectl apply consumes). It is the verifiable source of
|
||||
// truth a Helm chart would otherwise only re-encode.
|
||||
func RenderYAML(p Params) ([]byte, error) {
|
||||
return RenderObjects(Objects(p))
|
||||
}
|
||||
|
||||
// RenderObjects marshals objs into one multi-document YAML stream, in order.
|
||||
func RenderObjects(objs []Object) ([]byte, error) {
|
||||
var buf bytes.Buffer
|
||||
for i, obj := range Objects(p) {
|
||||
for i, obj := range objs {
|
||||
if i > 0 {
|
||||
buf.WriteString("---\n")
|
||||
}
|
||||
|
||||
@@ -39,6 +39,11 @@ const (
|
||||
// a control-plane identity, so the RCON NetworkPolicy peer (which requires
|
||||
// part-of=felis-control-plane) can never select it.
|
||||
ComponentRegistry = "registry"
|
||||
|
||||
// ComponentPostgres labels the control-plane database. Like the registry it
|
||||
// is a supporting workload and NOT part-of=felis-control-plane, so no
|
||||
// control-plane peer selector (RCON, the internal API) can select it.
|
||||
ComponentPostgres = "postgres"
|
||||
)
|
||||
|
||||
// Service-account names. The control-plane SAs (api/operator/reaper) are bound to
|
||||
@@ -70,6 +75,15 @@ const (
|
||||
// for humans. deploy/bootstrap.sh's REGISTRY_IMAGE caches and GC-pins this exact
|
||||
// ref and must name the same one (TestBootstrapPinsTheRegistryImage).
|
||||
defaultRegistryImage = "docker.io/library/registry:2.8.3@sha256:a3d8aaa63ed8681a604f1dea0aa03f100d5895b6a58ace528858a7b332415373"
|
||||
|
||||
// defaultPostgresImage is the official PostgreSQL image the control-plane
|
||||
// database runs (postgres.go). Pinned by digest for the same reason as the
|
||||
// registry, and more: a re-tag that moved the major would start an empty
|
||||
// cluster in a fresh <major>/docker directory beside the real one.
|
||||
// deploy/bootstrap.sh's POSTGRES_IMAGE caches and GC-pins this exact ref
|
||||
// (TestBootstrapPinsThePostgresImage) and refuses to start it over data
|
||||
// another major wrote.
|
||||
defaultPostgresImage = "docker.io/library/postgres:18.6-trixie@sha256:5a5a84b19854a9ffaa54082c166ff4ec27473a361e496e5ea167f298f2da9722"
|
||||
)
|
||||
|
||||
// Params parameterises the install bundle. Namespaces and the registry location
|
||||
@@ -126,6 +140,9 @@ type Params struct {
|
||||
FelisImage string
|
||||
// RegistryImage is the in-cluster registry image. Defaults to registry 2.8.3, by digest.
|
||||
RegistryImage string
|
||||
// PostgresImage is the control-plane database image. Defaults to PostgreSQL
|
||||
// 18.6, by digest.
|
||||
PostgresImage string
|
||||
// BackupPVC is the name of the world-archive PersistentVolumeClaim. The bundle
|
||||
// RENDERS this PVC (backupPVC in workloads.go, Minecraft namespace — where every
|
||||
// pod that mounts it runs) and felis-api advertises the name to its backup/restore
|
||||
@@ -236,6 +253,9 @@ func (p Params) withDefaults() Params {
|
||||
if p.RegistryImage == "" {
|
||||
p.RegistryImage = defaultRegistryImage
|
||||
}
|
||||
if p.PostgresImage == "" {
|
||||
p.PostgresImage = defaultPostgresImage
|
||||
}
|
||||
if p.RegistryStorage == "" {
|
||||
p.RegistryStorage = registryStorageSize
|
||||
}
|
||||
|
||||
@@ -0,0 +1,295 @@
|
||||
package platform
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"felis.lolicon.best/internal/backupjob"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/util/intstr"
|
||||
)
|
||||
|
||||
// The control-plane database. It used to be the host's own PostgreSQL, which the
|
||||
// installer had to find in the distro's repositories, initialise, open a hole in
|
||||
// the firewall for and keep in step with the distro's major version — the part of
|
||||
// an install that broke first whenever a host differed from the ones it was tried
|
||||
// on. It is now the official image, pinned by digest (Params.PostgresImage), run
|
||||
// by k3s beside the rest of the control plane.
|
||||
const (
|
||||
// PostgresName names the Deployment, the Service, the NetworkPolicy and the
|
||||
// Secret. deploy/bootstrap.sh and `felis db` reach the pod as
|
||||
// deploy/<PostgresName> in the control namespace.
|
||||
PostgresName = "felis-postgres"
|
||||
// PostgresContainer is the database container `kubectl exec -c` targets.
|
||||
PostgresContainer = "postgres"
|
||||
// PostgresPort is the port the server listens on, in the pod and on the
|
||||
// Service.
|
||||
PostgresPort int32 = 5432
|
||||
// PostgresHostPort is the node-loopback port the host's own tools reach the
|
||||
// database on (`felis migrate`, the watchdog, off-site backups): a hostPort
|
||||
// bound to 127.0.0.1 only, the path the registry's node-side pulls already
|
||||
// take (registryLoopbackHost). It sits clear of 5432 so a host PostgreSQL
|
||||
// left running from an older install does not collide with it.
|
||||
PostgresHostPort int32 = 15432
|
||||
// PostgresDataHostPath is the node directory the cluster lives under,
|
||||
// mounted at postgresDataMount. deploy/bootstrap.sh creates it owned by
|
||||
// postgresUID, mode 0700, before the first apply; `felis uninstall` keeps
|
||||
// it unless --purge. A hostPath instead of a local-path PVC so the data sits
|
||||
// at a fixed, documented place under /var/lib/felis that survives the
|
||||
// Deployment, the namespace and k3s itself being deleted.
|
||||
PostgresDataHostPath = "/var/lib/felis/postgres"
|
||||
// PostgresSuperuserSecret holds the postgres role's password, which the
|
||||
// image's entrypoint needs only to initialise an empty data directory.
|
||||
// deploy/bootstrap.sh creates it once with a random value and nothing reads
|
||||
// it back: pg_hba.conf (postgresHBA) refuses the postgres role over TCP, and
|
||||
// the socket inside the container trusts local connections.
|
||||
PostgresSuperuserSecret = "felis-postgres"
|
||||
// PostgresSuperuserSecretKey is the key in PostgresSuperuserSecret.
|
||||
PostgresSuperuserSecretKey = "superuser-password"
|
||||
// PostgresSocketDir is the image's socket directory, which `felis db` and
|
||||
// deploy/bootstrap.sh connect through under kubectl exec.
|
||||
PostgresSocketDir = "/var/run/postgresql"
|
||||
|
||||
postgresHBAConfigMap = "felis-postgres-hba"
|
||||
postgresHBAMount = "/etc/felis-postgres"
|
||||
// postgresDataMount is the image's VOLUME. The image's PGDATA is
|
||||
// <mount>/<major>/docker, so a new major starts beside the old one's data
|
||||
// instead of on top of it; deploy/bootstrap.sh refuses to run an image whose
|
||||
// major finds only another major's cluster there.
|
||||
postgresDataMount = "/var/lib/postgresql"
|
||||
postgresDataVol = "data"
|
||||
postgresSocketVol = "socket"
|
||||
postgresShmVol = "shm"
|
||||
postgresHBAVol = "hba"
|
||||
// postgresUID is the postgres account the official Debian image creates.
|
||||
// Running as it (instead of root) skips the entrypoint's chown pass, which
|
||||
// is why the installer owns the data directory to it up front.
|
||||
postgresUID int64 = 999
|
||||
)
|
||||
|
||||
// postgresHBA is the pg_hba.conf the server runs under, from a ConfigMap
|
||||
// (hba_file) instead of the one initdb writes into the data directory, so the
|
||||
// rules are the rendered ones on every start, whatever the cluster was
|
||||
// initialised with. The container's socket trusts every role: only something
|
||||
// that can already exec into the pod reaches it. Over TCP the postgres
|
||||
// superuser is refused outright and every other role needs its password; which
|
||||
// pods may open a TCP connection at all is postgresIngressPolicy's business.
|
||||
const postgresHBA = `# Rendered by felis (internal/platform/postgres.go); edits are overwritten.
|
||||
local all all trust
|
||||
host all postgres 0.0.0.0/0 reject
|
||||
host all postgres ::/0 reject
|
||||
host all all 0.0.0.0/0 scram-sha-256
|
||||
host all all ::/0 scram-sha-256
|
||||
`
|
||||
|
||||
// postgresPingCommand answers only once the server that takes connections is
|
||||
// up: the entrypoint's initialisation server listens on no TCP address.
|
||||
var postgresPingCommand = []string{"pg_isready", "-q", "-h", "127.0.0.1", "-p", fmt.Sprint(PostgresPort)}
|
||||
|
||||
// PostgresObjects renders the database: its control namespace, pg_hba.conf, the
|
||||
// ingress fence, the Deployment and the Service. deploy/bootstrap.sh applies
|
||||
// exactly these (`felis manifests --only postgres`) before it runs migrations,
|
||||
// which is before the rest of the bundle can render; Objects includes them
|
||||
// again so a full apply never prunes or drifts them.
|
||||
func PostgresObjects(p Params) []Object {
|
||||
p = p.withDefaults()
|
||||
return []Object{
|
||||
namespaceObject(p.ControlNamespace),
|
||||
postgresHBAConfig(p),
|
||||
PostgresIngressPolicy(p),
|
||||
postgresDeployment(p),
|
||||
postgresService(p),
|
||||
}
|
||||
}
|
||||
|
||||
// postgresLabels are the database's labels. Like the registry it is not
|
||||
// part-of=felis-control-plane, which keeps it out of the RCON peer.
|
||||
func postgresLabels() map[string]string {
|
||||
return map[string]string{
|
||||
LabelName: appName,
|
||||
LabelComponent: ComponentPostgres,
|
||||
}
|
||||
}
|
||||
|
||||
func postgresHBAConfig(p Params) *corev1.ConfigMap {
|
||||
return &corev1.ConfigMap{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "ConfigMap"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: postgresHBAConfigMap, Namespace: p.ControlNamespace, Labels: postgresLabels()},
|
||||
Data: map[string]string{"pg_hba.conf": postgresHBA},
|
||||
}
|
||||
}
|
||||
|
||||
// postgresDeployment renders the single database pod.
|
||||
//
|
||||
// Recreate, one replica: two servers on one data directory corrupt it, and a
|
||||
// rolling update would start the second before stopping the first. The pod
|
||||
// runs as the image's postgres account under the control plane's hardening
|
||||
// (read-only root, no capabilities); everything the server writes lands on the
|
||||
// data hostPath, the socket emptyDir or the /dev/shm emptyDir (dynamic shared
|
||||
// memory, which the runtime's 64Mi default would cap).
|
||||
//
|
||||
// Stopping: the image's STOPSIGNAL is SIGINT (fast shutdown: roll back open
|
||||
// transactions, checkpoint, exit), which containerd sends in place of SIGTERM;
|
||||
// SIGTERM would wait for felis-api's pooled connections to leave and run into
|
||||
// the kill at the end of the grace period. The preStop hook asks for the same
|
||||
// fast shutdown explicitly so the stop does not rest on the runtime honouring
|
||||
// the image's signal.
|
||||
func postgresDeployment(p Params) *appsv1.Deployment {
|
||||
labels := postgresLabels()
|
||||
hostPathDir := corev1.HostPathDirectory
|
||||
probe := func(period, timeout, failures int32) *corev1.Probe {
|
||||
return &corev1.Probe{
|
||||
ProbeHandler: corev1.ProbeHandler{Exec: &corev1.ExecAction{Command: postgresPingCommand}},
|
||||
PeriodSeconds: period,
|
||||
TimeoutSeconds: timeout,
|
||||
FailureThreshold: failures,
|
||||
}
|
||||
}
|
||||
container := corev1.Container{
|
||||
Name: PostgresContainer,
|
||||
Image: p.PostgresImage,
|
||||
// The image's entrypoint initialises an empty data directory, then execs
|
||||
// these arguments.
|
||||
Args: []string{"postgres", "-c", "hba_file=" + postgresHBAMount + "/pg_hba.conf"},
|
||||
Env: []corev1.EnvVar{{
|
||||
Name: "POSTGRES_PASSWORD",
|
||||
ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
|
||||
LocalObjectReference: corev1.LocalObjectReference{Name: PostgresSuperuserSecret},
|
||||
Key: PostgresSuperuserSecretKey,
|
||||
// Only initdb reads it: a cluster that already exists starts
|
||||
// without the Secret.
|
||||
Optional: boolPtr(true),
|
||||
}},
|
||||
}},
|
||||
Ports: []corev1.ContainerPort{{
|
||||
Name: PostgresContainer, ContainerPort: PostgresPort, Protocol: corev1.ProtocolTCP,
|
||||
HostPort: PostgresHostPort, HostIP: registryLoopbackHost,
|
||||
}},
|
||||
VolumeMounts: []corev1.VolumeMount{
|
||||
{Name: postgresDataVol, MountPath: postgresDataMount},
|
||||
{Name: postgresSocketVol, MountPath: PostgresSocketDir},
|
||||
{Name: tmpVolume, MountPath: "/tmp"},
|
||||
{Name: postgresShmVol, MountPath: "/dev/shm"},
|
||||
{Name: postgresHBAVol, MountPath: postgresHBAMount, ReadOnly: true},
|
||||
},
|
||||
// Crash recovery after an unclean stop, or initdb on a slow disk, can
|
||||
// take minutes; liveness waits for the startup probe.
|
||||
StartupProbe: probe(5, 5, 120),
|
||||
ReadinessProbe: probe(10, 5, 3),
|
||||
LivenessProbe: probe(30, 10, 6),
|
||||
Lifecycle: &corev1.Lifecycle{PreStop: &corev1.LifecycleHandler{Exec: &corev1.ExecAction{
|
||||
Command: []string{"/bin/sh", "-c", `pg_ctl -D "$PGDATA" -m fast -w -t 50 stop`},
|
||||
}}},
|
||||
Resources: corev1.ResourceRequirements{
|
||||
Requests: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("50m"),
|
||||
corev1.ResourceMemory: resource.MustParse("256Mi"),
|
||||
},
|
||||
Limits: corev1.ResourceList{
|
||||
corev1.ResourceCPU: resource.MustParse("1"),
|
||||
corev1.ResourceMemory: resource.MustParse("1Gi"),
|
||||
},
|
||||
},
|
||||
SecurityContext: hardenedContainerSecurityContext(),
|
||||
}
|
||||
shmLimit := resource.MustParse("256Mi")
|
||||
return &appsv1.Deployment{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "apps/v1", Kind: "Deployment"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: PostgresName, Namespace: p.ControlNamespace, Labels: labels},
|
||||
Spec: appsv1.DeploymentSpec{
|
||||
Replicas: int32Ptr(1),
|
||||
Strategy: appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType},
|
||||
Selector: &metav1.LabelSelector{MatchLabels: labels},
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: labels},
|
||||
Spec: corev1.PodSpec{
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
EnableServiceLinks: boolPtr(false),
|
||||
PriorityClassName: controlPlanePriorityName,
|
||||
// pg_ctl's -t 50 in the preStop hook fits inside it.
|
||||
TerminationGracePeriodSeconds: int64Ptr(60),
|
||||
SecurityContext: &corev1.PodSecurityContext{
|
||||
RunAsNonRoot: boolPtr(true),
|
||||
RunAsUser: int64Ptr(postgresUID),
|
||||
RunAsGroup: int64Ptr(postgresUID),
|
||||
FSGroup: int64Ptr(postgresUID),
|
||||
SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault},
|
||||
},
|
||||
Containers: []corev1.Container{container},
|
||||
Volumes: []corev1.Volume{
|
||||
{
|
||||
Name: postgresDataVol,
|
||||
VolumeSource: corev1.VolumeSource{HostPath: &corev1.HostPathVolumeSource{
|
||||
Path: PostgresDataHostPath,
|
||||
// Directory, not DirectoryOrCreate: a directory
|
||||
// kubelet made would belong to root, and a missing
|
||||
// one means the installer did not run.
|
||||
Type: &hostPathDir,
|
||||
}},
|
||||
},
|
||||
{Name: postgresSocketVol, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}},
|
||||
{Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}},
|
||||
{Name: postgresShmVol, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
|
||||
Medium: corev1.StorageMediumMemory, SizeLimit: &shmLimit,
|
||||
}}},
|
||||
{Name: postgresHBAVol, VolumeSource: corev1.VolumeSource{ConfigMap: &corev1.ConfigMapVolumeSource{
|
||||
LocalObjectReference: corev1.LocalObjectReference{Name: postgresHBAConfigMap},
|
||||
}}},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// postgresService is the name the pods reach the database by:
|
||||
// felis-postgres.<control-ns>.svc:5432 (the pod config's [database] url).
|
||||
func postgresService(p Params) *corev1.Service {
|
||||
labels := postgresLabels()
|
||||
return &corev1.Service{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Service"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: PostgresName, Namespace: p.ControlNamespace, Labels: labels},
|
||||
Spec: corev1.ServiceSpec{
|
||||
Type: corev1.ServiceTypeClusterIP,
|
||||
Selector: labels,
|
||||
Ports: []corev1.ServicePort{{
|
||||
Name: PostgresContainer,
|
||||
Port: PostgresPort,
|
||||
TargetPort: intstr.FromString(PostgresContainer),
|
||||
Protocol: corev1.ProtocolTCP,
|
||||
}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// PostgresIngressPolicy admits TCP to the database from the pods that open it
|
||||
// and nothing else in the cluster: felis-api, the reaper (which records what it
|
||||
// archives) and the world-backup Jobs (which record each backup). The operator,
|
||||
// restore and file Jobs and every game server hold no database credential and
|
||||
// get no route. The node itself is always admitted by the policy controller,
|
||||
// which is how the host's tools reach PostgresHostPort.
|
||||
func PostgresIngressPolicy(p Params) *networkingv1.NetworkPolicy {
|
||||
p = p.withDefaults()
|
||||
tcp := corev1.ProtocolTCP
|
||||
port := intstr.FromInt32(PostgresPort)
|
||||
inNamespace := func(ns string, pods map[string]string) networkingv1.NetworkPolicyPeer {
|
||||
return networkingv1.NetworkPolicyPeer{
|
||||
NamespaceSelector: &metav1.LabelSelector{MatchLabels: map[string]string{"kubernetes.io/metadata.name": ns}},
|
||||
PodSelector: &metav1.LabelSelector{MatchLabels: pods},
|
||||
}
|
||||
}
|
||||
return netpol(PostgresName+"-ingress", p.ControlNamespace,
|
||||
metav1.LabelSelector{MatchLabels: postgresLabels()},
|
||||
[]networkingv1.NetworkPolicyIngressRule{{
|
||||
From: []networkingv1.NetworkPolicyPeer{
|
||||
inNamespace(p.ControlNamespace, map[string]string{LabelPartOf: controlPlanePartOf, LabelComponent: ComponentAPI}),
|
||||
inNamespace(p.MinecraftNamespace, map[string]string{LabelPartOf: controlPlanePartOf, LabelComponent: ComponentReaper}),
|
||||
inNamespace(p.MinecraftNamespace, backupjob.PodSelector()),
|
||||
},
|
||||
Ports: []networkingv1.NetworkPolicyPort{{Protocol: &tcp, Port: &port}},
|
||||
}},
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,362 @@
|
||||
package platform
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"reflect"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/backupjob"
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
"felis.lolicon.best/internal/restore"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
"k8s.io/apimachinery/pkg/labels"
|
||||
)
|
||||
|
||||
// admits reports whether np lets a pod with podLabels in namespace ns open a
|
||||
// TCP connection to port.
|
||||
func admits(np *networkingv1.NetworkPolicy, ns string, podLabels map[string]string, port int32) bool {
|
||||
for _, rule := range np.Spec.Ingress {
|
||||
portOK := len(rule.Ports) == 0
|
||||
for _, p := range rule.Ports {
|
||||
if (p.Protocol == nil || *p.Protocol == corev1.ProtocolTCP) && p.Port != nil && p.Port.IntVal == port {
|
||||
portOK = true
|
||||
}
|
||||
}
|
||||
if !portOK {
|
||||
continue
|
||||
}
|
||||
for _, peer := range rule.From {
|
||||
if peer.IPBlock != nil {
|
||||
continue
|
||||
}
|
||||
nsOK := peer.NamespaceSelector == nil && ns == np.Namespace
|
||||
if peer.NamespaceSelector != nil {
|
||||
nsOK = labels.SelectorFromSet(peer.NamespaceSelector.MatchLabels).
|
||||
Matches(labels.Set{"kubernetes.io/metadata.name": ns})
|
||||
}
|
||||
podOK := peer.PodSelector == nil ||
|
||||
labels.SelectorFromSet(peer.PodSelector.MatchLabels).Matches(labels.Set(podLabels))
|
||||
if nsOK && podOK {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func podTemplateOf(t *testing.T, objs []Object, kind, name string) corev1.PodTemplateSpec {
|
||||
t.Helper()
|
||||
for _, o := range objs {
|
||||
switch v := o.(type) {
|
||||
case *appsv1.Deployment:
|
||||
if kind == "Deployment" && v.Name == name {
|
||||
return v.Spec.Template
|
||||
}
|
||||
case *batchv1.CronJob:
|
||||
if kind == "CronJob" && v.Name == name {
|
||||
return v.Spec.JobTemplate.Spec.Template
|
||||
}
|
||||
}
|
||||
}
|
||||
t.Fatalf("%s %s not rendered", kind, name)
|
||||
return corev1.PodTemplateSpec{}
|
||||
}
|
||||
|
||||
// TestPostgresIngress_AdmitsExactlyTheDatabaseClients checks the fence against
|
||||
// the labels the pods actually carry: the three workloads that open the
|
||||
// database must get through and every other pod the platform runs must not.
|
||||
// A selector typo would either cut felis-api off from its database or leave a
|
||||
// game server able to try passwords against it.
|
||||
func TestPostgresIngress_AdmitsExactlyTheDatabaseClients(t *testing.T) {
|
||||
p := reaperParams().withDefaults()
|
||||
objs := Objects(p)
|
||||
np := PostgresIngressPolicy(p)
|
||||
|
||||
db := podTemplateOf(t, objs, "Deployment", PostgresName)
|
||||
if !labels.SelectorFromSet(np.Spec.PodSelector.MatchLabels).Matches(labels.Set(db.Labels)) {
|
||||
t.Fatalf("the policy selects %v, which is not the database pod %v", np.Spec.PodSelector.MatchLabels, db.Labels)
|
||||
}
|
||||
|
||||
backup, err := backupjob.BackupJob(backupjob.JobParams{
|
||||
Server: "survival", WorldPVC: "world-survival-0", BackupPVC: "felis-backups",
|
||||
Namespace: p.MinecraftNamespace, Image: "felis:test", ConfigSecret: "felis-config",
|
||||
ConfigMount: "/etc/felis", BackupRoot: "/backups", WorldsRoot: "/world",
|
||||
Deadline: time.Minute, CPULimit: "1", MemLimit: "1Gi",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rst, err := restore.RestoreJob(restore.JobParams{
|
||||
Server: "survival", WorldPVC: "world-survival-0", BackupPVC: "felis-backups",
|
||||
BackupRef: "/backups/survival/x.tar.gz", ArchiveStore: "tarLocal",
|
||||
Namespace: p.MinecraftNamespace, Image: "felis:test", BackupRoot: "/backups",
|
||||
WorldsRoot: "/world", Deadline: time.Minute, CPULimit: "1", MemLimit: "1Gi",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
files, err := fileedit.FilesJob(fileedit.JobParams{
|
||||
Server: "survival", OpID: "deadbeefcafe0001", Op: fileedit.OpRead, Path: "server.properties",
|
||||
WorldPVC: "world-survival-0", Namespace: p.MinecraftNamespace, Image: "felis:test",
|
||||
WorldsRoot: "/data", Deadline: time.Minute, CPULimit: "500m", MemLimit: "256Mi",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
game := map[string]string{
|
||||
v1alpha1.LabelManagedBy: operator.ManagedByValue,
|
||||
v1alpha1.LabelComponent: operator.ComponentValue,
|
||||
}
|
||||
|
||||
for _, c := range []struct {
|
||||
who string
|
||||
ns string
|
||||
labels map[string]string
|
||||
want bool
|
||||
}{
|
||||
{"felis-api", p.ControlNamespace, podTemplateOf(t, objs, "Deployment", "felis-api").Labels, true},
|
||||
{"the reaper", p.MinecraftNamespace, podTemplateOf(t, objs, "CronJob", SAReaper).Labels, true},
|
||||
{"a world-backup Job", p.MinecraftNamespace, backup.Spec.Template.Labels, true},
|
||||
|
||||
{"felis-operator", p.ControlNamespace, podTemplateOf(t, objs, "Deployment", "felis-operator").Labels, false},
|
||||
{"the registry", p.RegistryNamespace, podTemplateOf(t, objs, "Deployment", registryName).Labels, false},
|
||||
{"a restore Job", p.MinecraftNamespace, rst.Spec.Template.Labels, false},
|
||||
{"a files Job", p.MinecraftNamespace, files.Spec.Template.Labels, false},
|
||||
{"a game server", p.MinecraftNamespace, game, false},
|
||||
{"a build Job", p.BuildNamespace, map[string]string{}, false},
|
||||
// felis-api's labels in the wrong namespace: a pod anyone can create in
|
||||
// the Minecraft namespace must not borrow them.
|
||||
{"api labels outside the control namespace", p.MinecraftNamespace, podTemplateOf(t, objs, "Deployment", "felis-api").Labels, false},
|
||||
} {
|
||||
if got := admits(np, c.ns, c.labels, PostgresPort); got != c.want {
|
||||
t.Errorf("%s (%s, %v): admitted = %v, want %v", c.who, c.ns, c.labels, got, c.want)
|
||||
}
|
||||
}
|
||||
if admits(np, p.ControlNamespace, podTemplateOf(t, objs, "Deployment", "felis-api").Labels, PostgresPort+1) {
|
||||
t.Error("the policy admits felis-api on a port other than the database's")
|
||||
}
|
||||
}
|
||||
|
||||
// hbaMethod returns the auth method pg_hba.conf picks for a connection:
|
||||
// PostgreSQL takes the first line whose type, user and address match.
|
||||
func hbaMethod(t *testing.T, hba, connType, user, addr string) string {
|
||||
t.Helper()
|
||||
for _, line := range strings.Split(hba, "\n") {
|
||||
f := strings.Fields(line)
|
||||
if len(f) == 0 || strings.HasPrefix(f[0], "#") {
|
||||
continue
|
||||
}
|
||||
if f[0] != connType || (f[2] != "all" && f[2] != user) {
|
||||
continue
|
||||
}
|
||||
if connType == "local" {
|
||||
return f[3]
|
||||
}
|
||||
prefix, err := netip.ParsePrefix(f[3])
|
||||
if err != nil {
|
||||
t.Fatalf("pg_hba.conf line %q: %v", line, err)
|
||||
}
|
||||
if prefix.Contains(netip.MustParseAddr(addr)) {
|
||||
return f[4]
|
||||
}
|
||||
}
|
||||
return "(no line: rejected)"
|
||||
}
|
||||
|
||||
// TestPostgresHBA_SuperuserOnlyOverTheSocket walks the rendered pg_hba.conf the
|
||||
// way the server does. The superuser's password is a random value nobody keeps,
|
||||
// so TCP must refuse the role outright; a felis role must be asked for its
|
||||
// password from every address family, never trusted.
|
||||
func TestPostgresHBA_SuperuserOnlyOverTheSocket(t *testing.T) {
|
||||
hba := postgresHBAConfig(testParams().withDefaults()).Data["pg_hba.conf"]
|
||||
for _, c := range []struct{ connType, user, addr, want string }{
|
||||
{"local", "postgres", "", "trust"},
|
||||
{"local", "felis", "", "trust"},
|
||||
{"host", "postgres", "10.42.0.7", "reject"},
|
||||
{"host", "postgres", "127.0.0.1", "reject"},
|
||||
{"host", "postgres", "fd00::7", "reject"},
|
||||
{"host", "felis", "10.42.0.7", "scram-sha-256"},
|
||||
{"host", "felis", "127.0.0.1", "scram-sha-256"},
|
||||
{"host", "felis", "fd00::7", "scram-sha-256"},
|
||||
} {
|
||||
if got := hbaMethod(t, hba, c.connType, c.user, c.addr); got != c.want {
|
||||
t.Errorf("%s %s from %q: %s, want %s", c.connType, c.user, c.addr, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPostgresDeployment_RunsTheImageSafely pins what keeps the data directory
|
||||
// intact and the database off the network, against the official image's own
|
||||
// layout (VOLUME /var/lib/postgresql, the postgres account uid 999).
|
||||
func TestPostgresDeployment_RunsTheImageSafely(t *testing.T) {
|
||||
p := testParams().withDefaults()
|
||||
objs := PostgresObjects(p)
|
||||
var dep *appsv1.Deployment
|
||||
var svc *corev1.Service
|
||||
var hba *corev1.ConfigMap
|
||||
for _, o := range objs {
|
||||
switch v := o.(type) {
|
||||
case *appsv1.Deployment:
|
||||
dep = v
|
||||
case *corev1.Service:
|
||||
svc = v
|
||||
case *corev1.ConfigMap:
|
||||
hba = v
|
||||
}
|
||||
}
|
||||
if dep == nil || svc == nil || hba == nil {
|
||||
t.Fatalf("PostgresObjects is missing the Deployment, Service or ConfigMap: %v", objs)
|
||||
}
|
||||
|
||||
// Two servers on one data directory corrupt it.
|
||||
if dep.Spec.Replicas == nil || *dep.Spec.Replicas != 1 || dep.Spec.Strategy.Type != appsv1.RecreateDeploymentStrategyType {
|
||||
t.Errorf("replicas %v strategy %q: want exactly one pod, stopped before its replacement starts", dep.Spec.Replicas, dep.Spec.Strategy.Type)
|
||||
}
|
||||
pod := dep.Spec.Template.Spec
|
||||
if len(pod.Containers) != 1 {
|
||||
t.Fatalf("want one container, got %d", len(pod.Containers))
|
||||
}
|
||||
c := pod.Containers[0]
|
||||
if c.Name != PostgresContainer || c.Image != p.PostgresImage {
|
||||
t.Errorf("container %q runs %q, want %q running %q", c.Name, c.Image, PostgresContainer, p.PostgresImage)
|
||||
}
|
||||
if sc := pod.SecurityContext; sc == nil || sc.RunAsUser == nil || *sc.RunAsUser != 999 || sc.RunAsGroup == nil || *sc.RunAsGroup != 999 {
|
||||
t.Errorf("pod runs as %+v, want the image's postgres account 999:999 (the installer owns the data directory to it)", sc)
|
||||
}
|
||||
if c.SecurityContext == nil || c.SecurityContext.ReadOnlyRootFilesystem == nil || !*c.SecurityContext.ReadOnlyRootFilesystem {
|
||||
t.Error("the database container's root filesystem is writable")
|
||||
}
|
||||
|
||||
volumes := map[string]corev1.Volume{}
|
||||
for _, v := range pod.Volumes {
|
||||
volumes[v.Name] = v
|
||||
}
|
||||
mounts := map[string]corev1.VolumeMount{}
|
||||
for _, m := range c.VolumeMounts {
|
||||
mounts[m.MountPath] = m
|
||||
if _, ok := volumes[m.Name]; !ok {
|
||||
t.Errorf("mount %s names volume %q, which the pod does not declare", m.MountPath, m.Name)
|
||||
}
|
||||
}
|
||||
data, ok := mounts["/var/lib/postgresql"]
|
||||
if !ok {
|
||||
t.Fatal("nothing is mounted at the image's VOLUME /var/lib/postgresql: the cluster would live in the container")
|
||||
}
|
||||
hp := volumes[data.Name].HostPath
|
||||
if hp == nil || hp.Path != PostgresDataHostPath || hp.Type == nil || *hp.Type != corev1.HostPathDirectory {
|
||||
t.Errorf("data volume = %+v, want hostPath %s of type Directory", volumes[data.Name].VolumeSource, PostgresDataHostPath)
|
||||
}
|
||||
for _, path := range []string{PostgresSocketDir, "/tmp", "/dev/shm"} {
|
||||
if _, ok := mounts[path]; !ok {
|
||||
t.Errorf("nothing writable at %s under a read-only root", path)
|
||||
}
|
||||
}
|
||||
|
||||
// hba_file must name a file the ConfigMap mount actually provides.
|
||||
var hbaFile string
|
||||
for i, a := range c.Args {
|
||||
if a == "-c" && i+1 < len(c.Args) && strings.HasPrefix(c.Args[i+1], "hba_file=") {
|
||||
hbaFile = strings.TrimPrefix(c.Args[i+1], "hba_file=")
|
||||
}
|
||||
}
|
||||
if len(c.Args) == 0 || c.Args[0] != "postgres" {
|
||||
t.Errorf("args %q: the entrypoint initialises the cluster only when the first argument is postgres", c.Args)
|
||||
}
|
||||
dir, key := hbaFile[:max(strings.LastIndex(hbaFile, "/"), 0)], hbaFile[strings.LastIndex(hbaFile, "/")+1:]
|
||||
m, ok := mounts[dir]
|
||||
if !ok || volumes[m.Name].ConfigMap == nil || volumes[m.Name].ConfigMap.Name != hba.Name {
|
||||
t.Errorf("hba_file %q is not in the %s ConfigMap's mount", hbaFile, hba.Name)
|
||||
} else if _, ok := hba.Data[key]; !ok {
|
||||
t.Errorf("hba_file %q: the ConfigMap has no key %q", hbaFile, key)
|
||||
}
|
||||
|
||||
// The host's tools get the port on loopback only.
|
||||
if len(c.Ports) != 1 || c.Ports[0].HostPort != PostgresHostPort || c.Ports[0].HostIP != "127.0.0.1" || c.Ports[0].ContainerPort != PostgresPort {
|
||||
t.Errorf("ports = %+v, want %d published on 127.0.0.1:%d and nowhere else", c.Ports, PostgresPort, PostgresHostPort)
|
||||
}
|
||||
// The Service reaches that port on that pod.
|
||||
if !labels.SelectorFromSet(svc.Spec.Selector).Matches(labels.Set(dep.Spec.Template.Labels)) {
|
||||
t.Errorf("Service selector %v misses the pod %v", svc.Spec.Selector, dep.Spec.Template.Labels)
|
||||
}
|
||||
if len(svc.Spec.Ports) != 1 || svc.Spec.Ports[0].Port != PostgresPort || svc.Spec.Ports[0].TargetPort.StrVal != c.Ports[0].Name {
|
||||
t.Errorf("Service ports %+v do not lead to the container's %q port", svc.Spec.Ports, c.Ports[0].Name)
|
||||
}
|
||||
if svc.Spec.Type != corev1.ServiceTypeClusterIP {
|
||||
t.Errorf("Service type %q: the database is for the cluster only", svc.Spec.Type)
|
||||
}
|
||||
|
||||
// The superuser password is read by initdb alone; a restart must not need it.
|
||||
var pw *corev1.EnvVar
|
||||
for i := range c.Env {
|
||||
if c.Env[i].Name == "POSTGRES_PASSWORD" {
|
||||
pw = &c.Env[i]
|
||||
}
|
||||
}
|
||||
if pw == nil || pw.ValueFrom == nil || pw.ValueFrom.SecretKeyRef == nil ||
|
||||
pw.ValueFrom.SecretKeyRef.Name != PostgresSuperuserSecret || pw.ValueFrom.SecretKeyRef.Optional == nil || !*pw.ValueFrom.SecretKeyRef.Optional {
|
||||
t.Errorf("POSTGRES_PASSWORD = %+v, want an optional reference to Secret %s", pw, PostgresSuperuserSecret)
|
||||
}
|
||||
|
||||
// Every probe goes over TCP: during initialisation the entrypoint's
|
||||
// temporary server answers on the socket only, and must not count as up.
|
||||
for name, pr := range map[string]*corev1.Probe{"startup": c.StartupProbe, "readiness": c.ReadinessProbe, "liveness": c.LivenessProbe} {
|
||||
if pr == nil || pr.Exec == nil {
|
||||
t.Errorf("%s probe missing", name)
|
||||
continue
|
||||
}
|
||||
cmd := strings.Join(pr.Exec.Command, " ")
|
||||
if !strings.HasPrefix(cmd, "pg_isready") || !strings.Contains(cmd, "-h 127.0.0.1") {
|
||||
t.Errorf("%s probe %q does not ping the TCP listener", name, cmd)
|
||||
}
|
||||
}
|
||||
if c.StartupProbe != nil && c.StartupProbe.PeriodSeconds*c.StartupProbe.FailureThreshold < 300 {
|
||||
t.Errorf("startup allows %ds: crash recovery or initdb on a slow disk gets killed mid-way",
|
||||
c.StartupProbe.PeriodSeconds*c.StartupProbe.FailureThreshold)
|
||||
}
|
||||
|
||||
// The fast shutdown's own timeout must end before the kubelet's SIGKILL.
|
||||
if c.Lifecycle == nil || c.Lifecycle.PreStop == nil || c.Lifecycle.PreStop.Exec == nil {
|
||||
t.Fatal("no preStop fast shutdown")
|
||||
}
|
||||
stop := strings.Join(c.Lifecycle.PreStop.Exec.Command, " ")
|
||||
tm := regexp.MustCompile(`pg_ctl .*-m fast .*-t (\d+) stop`).FindStringSubmatch(stop)
|
||||
if tm == nil {
|
||||
t.Fatalf("preStop %q is not a timed fast pg_ctl stop", stop)
|
||||
}
|
||||
timeout, _ := strconv.Atoi(tm[1])
|
||||
if pod.TerminationGracePeriodSeconds == nil || int64(timeout) >= *pod.TerminationGracePeriodSeconds {
|
||||
t.Errorf("pg_ctl waits %ds but the grace period is %v", timeout, pod.TerminationGracePeriodSeconds)
|
||||
}
|
||||
}
|
||||
|
||||
// TestObjects_CarryThePostgresObjectsUnchanged: bootstrap applies
|
||||
// PostgresObjects first and the full bundle afterwards. An object that renders
|
||||
// differently in the two would flip back and forth on every install, and the
|
||||
// Deployment would restart the database each time.
|
||||
func TestObjects_CarryThePostgresObjectsUnchanged(t *testing.T) {
|
||||
p := reaperParams()
|
||||
full := map[string]Object{}
|
||||
for _, o := range Objects(p) {
|
||||
full[o.GetObjectKind().GroupVersionKind().Kind+"/"+o.GetNamespace()+"/"+o.GetName()] = o
|
||||
}
|
||||
for _, o := range PostgresObjects(p) {
|
||||
key := o.GetObjectKind().GroupVersionKind().Kind + "/" + o.GetNamespace() + "/" + o.GetName()
|
||||
got, ok := full[key]
|
||||
if !ok {
|
||||
t.Errorf("%s is in PostgresObjects but not in Objects", key)
|
||||
continue
|
||||
}
|
||||
if !reflect.DeepEqual(got, o) {
|
||||
t.Errorf("%s renders differently in Objects and PostgresObjects", key)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -243,9 +243,10 @@ func InternalAPIBaseURL(controlNamespace string) string {
|
||||
return fmt.Sprintf("http://%s.%s.svc.cluster.local:%d", APIInternalServiceName, controlNamespace, apiInternalPort)
|
||||
}
|
||||
|
||||
// Workloads renders the running control-plane: the felis-api Deployment, the
|
||||
// felis-operator Deployment, and the in-cluster registry (Deployment + Service +
|
||||
// PVC), the world-archive PVC when p.BackupPVC names it (it backs the
|
||||
// Workloads renders the running control-plane: the database (Deployment +
|
||||
// Service, postgres.go), the felis-api Deployment, the felis-operator
|
||||
// Deployment, and the in-cluster registry (Deployment + Service + PVC), the
|
||||
// world-archive PVC when p.BackupPVC names it (it backs the
|
||||
// backup/restore Jobs and the reaper), plus the reaper CronJob when
|
||||
// reaperEnabled(p). Every pod template carries the built-in
|
||||
// system-cluster-critical PriorityClass (controlPlanePriorityName), the
|
||||
@@ -254,6 +255,8 @@ func InternalAPIBaseURL(controlNamespace string) string {
|
||||
func Workloads(p Params) []Object {
|
||||
p = p.withDefaults()
|
||||
objs := []Object{
|
||||
postgresDeployment(p),
|
||||
postgresService(p),
|
||||
APIDeployment(p),
|
||||
apiService(p),
|
||||
apiInternalService(p),
|
||||
|
||||
@@ -707,15 +707,15 @@ func TestWorkloads_DeploymentsCarryProbes(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestWorkloads_BundleContents sanity-checks the slice Workloads returns: the two
|
||||
// control-plane Deployments, the api external+internal Services, the operator
|
||||
// metrics Service, and the registry
|
||||
// TestWorkloads_BundleContents sanity-checks the slice Workloads returns: the
|
||||
// database Deployment/Service, the two control-plane Deployments, the api
|
||||
// external+internal Services, the operator metrics Service, and the registry
|
||||
// Deployment/Service/PVC, every one with TypeMeta (so its YAML header renders). The
|
||||
// internal Service must be present or the login pod's felis-api:8081 path is dead.
|
||||
func TestWorkloads_BundleContents(t *testing.T) {
|
||||
objs := Workloads(testParams())
|
||||
if len(objs) != 9 {
|
||||
t.Fatalf("Workloads returned %d objects, want 9", len(objs))
|
||||
if len(objs) != 11 {
|
||||
t.Fatalf("Workloads returned %d objects, want 11", len(objs))
|
||||
}
|
||||
var haveInternalSvc bool
|
||||
for _, o := range objs {
|
||||
@@ -733,7 +733,7 @@ func TestWorkloads_BundleContents(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestControlPlanePriorityClass pins the node-pressure eviction shield: every
|
||||
// control-plane pod template (api/operator/reaper/registry) runs under the
|
||||
// control-plane pod template (postgres/api/operator/reaper/registry) runs under the
|
||||
// BUILT-IN system-cluster-critical class (value 2e9), at which kubelet's
|
||||
// eviction manager refuses to evict the pod. A live drill showed the whole
|
||||
// cascade with plain ordering: disk pressure evicted the game pods and then the
|
||||
@@ -759,8 +759,8 @@ func TestControlPlanePriorityClass(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
if deployments != 3 || cronJobs != 1 {
|
||||
t.Errorf("scanned %d deployments / %d cronjobs, want 3 / 1 — a pod template escaped the class check", deployments, cronJobs)
|
||||
if deployments != 4 || cronJobs != 1 {
|
||||
t.Errorf("scanned %d deployments / %d cronjobs, want 4 / 1 — a pod template escaped the class check", deployments, cronJobs)
|
||||
}
|
||||
// The name must be the built-in critical class: any custom class is capped at
|
||||
// 1e9 by the API server and would be evictable.
|
||||
|
||||
Reference in new issue
Block a user