From 47fcd90f752fe91a6cd860dd62e8dfe345ce8e55 Mon Sep 17 00:00:00 2001 From: Minseong Choi Date: Fri, 26 Jun 2026 23:32:38 +0900 Subject: [PATCH] feat(platform): add node orchestration and the felis entrypoint The platform package that places servers across nodes and wires the operator, build, restore, and reaper subsystems, plus cmd/felis, the single binary that runs them. --- cmd/felis/api.go | 215 +++++++++++ cmd/felis/main.go | 13 + cmd/felis/manifests.go | 134 +++++++ cmd/felis/manifests_test.go | 158 ++++++++ cmd/felis/migrate.go | 58 +++ cmd/felis/operator.go | 75 ++++ cmd/felis/reaper.go | 187 ++++++++++ cmd/felis/restore.go | 83 +++++ cmd/felis/restore_test.go | 140 +++++++ cmd/felis/run.go | 63 ++++ cmd/felis/run_test.go | 86 +++++ internal/platform/bundle.go | 152 ++++++++ internal/platform/bundle_test.go | 141 +++++++ internal/platform/doc.go | 58 +++ internal/platform/identities.go | 178 +++++++++ internal/platform/netpol.go | 140 +++++++ internal/platform/netpol_test.go | 192 ++++++++++ internal/platform/rbac.go | 204 ++++++++++ internal/platform/rbac_test.go | 354 ++++++++++++++++++ internal/platform/workloads.go | 527 ++++++++++++++++++++++++++ internal/platform/workloads_test.go | 555 ++++++++++++++++++++++++++++ 21 files changed, 3713 insertions(+) create mode 100644 cmd/felis/api.go create mode 100644 cmd/felis/main.go create mode 100644 cmd/felis/manifests.go create mode 100644 cmd/felis/manifests_test.go create mode 100644 cmd/felis/migrate.go create mode 100644 cmd/felis/operator.go create mode 100644 cmd/felis/reaper.go create mode 100644 cmd/felis/restore.go create mode 100644 cmd/felis/restore_test.go create mode 100644 cmd/felis/run.go create mode 100644 cmd/felis/run_test.go create mode 100644 internal/platform/bundle.go create mode 100644 internal/platform/bundle_test.go create mode 100644 internal/platform/doc.go create mode 100644 internal/platform/identities.go create mode 100644 internal/platform/netpol.go create mode 100644 internal/platform/netpol_test.go create mode 100644 internal/platform/rbac.go create mode 100644 internal/platform/rbac_test.go create mode 100644 internal/platform/workloads.go create mode 100644 internal/platform/workloads_test.go diff --git a/cmd/felis/api.go b/cmd/felis/api.go new file mode 100644 index 0000000..7a51dfe --- /dev/null +++ b/cmd/felis/api.go @@ -0,0 +1,215 @@ +package main + +import ( + "context" + "flag" + "fmt" + "io" + "net/http" + "os" + "time" + + "felis.lolicon.best/internal/api" + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/build" + "felis.lolicon.best/internal/config" + "felis.lolicon.best/internal/restore" + "felis.lolicon.best/internal/store" + "felis.lolicon.best/internal/submit" + "k8s.io/apimachinery/pkg/runtime" + utilruntime "k8s.io/apimachinery/pkg/util/runtime" + "k8s.io/client-go/kubernetes" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +// cmdAPI runs felis-api: two listeners, two middleware chains (spec §7). The +// internal face (service token) is fully wired. The external face is wired but +// fails closed until an Access JWKS key function is configured — the verifier's +// audience logic is unit-tested (internal/api), the JWKS source is a deployment +// integration point. +func cmdAPI(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("api", flag.ContinueOnError) + fs.SetOutput(stderr) + cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml") + internalAddr := fs.String("internal-addr", ":8081", "internal-face listen address (service token, no Zero Trust)") + if err := fs.Parse(args); err != nil { + return 2 + } + + cfg, err := config.Load(*cfgPath) + if err != nil { + fmt.Fprintf(stderr, "felis api: %v\n", err) + return 1 + } + + ctx := ctrl.SetupSignalHandler() + + drv, err := store.Open(ctx, cfg.Database.URL) + if err != nil { + fmt.Fprintf(stderr, "felis api: open database: %v\n", err) + return 1 + } + defer drv.Close() + + scheme := runtime.NewScheme() + utilruntime.Must(clientgoscheme.AddToScheme(scheme)) + utilruntime.Must(v1alpha1.AddToScheme(scheme)) + // Both clients are built from the SAME rest.Config. The controller-runtime + // client.Client drives CRDs/Secrets/Jobs (cluster, console-write, restore); the + // typed clientset is needed solely for the read-side console, because the + // pods/log subresource (GetLogs(...).Stream) lives only on the typed CoreV1 + // client, not on client.Client (spec §8 读=pods/log follow). + restCfg := ctrl.GetConfigOrDie() + cl, err := client.New(restCfg, client.Options{Scheme: scheme}) + if err != nil { + fmt.Fprintf(stderr, "felis api: build k8s client: %v\n", err) + return 1 + } + clientset, err := kubernetes.NewForConfig(restCfg) + if err != nil { + fmt.Fprintf(stderr, "felis api: build k8s clientset: %v\n", err) + return 1 + } + + token := os.Getenv("FELIS_SERVICE_TOKEN") + if token == "" { + fmt.Fprintln(stderr, "felis api: warning: FELIS_SERVICE_TOKEN unset — internal face will reject all callers") + } + + // Build subsystem (spec §16): the weak-SA build Job runs in the configured + // build namespace and pushes to the internal registry. The build Pod never + // holds DB credentials — felis-api owns the PG store and admits scanned + // images, so the Builder is constructed here with both bindings. + builder := &build.Builder{ + Store: build.NewPGStore(drv.DB()), + Jobs: build.NewK8sJobs(cl, buildConfig(cfg)), + Config: buildConfig(cfg), + } + + // User-modpack approval lane (user-directed extension over §16; see + // internal/submit). An ordinary user may only SUBMIT a + // modpack; an admin must approve it before anything is built, at which point + // the SAME Trivy-gated Builder runs as for an admin's direct build. Registry + // MUST match the Builder's RegistryURL (cfg.Registry.URL) — both are wired from + // the one field here so the lane's pre-CAS validate and the Builder's Submit + // can never disagree about the push target. The blob upload transport that + // populates the derived context ref is deferred (INTEGRATION-ONLY): the + // create→approve→reject state machine is real Postgres truth, but a real + // Kaniko context pull needs that transport in place. + submissions := &submit.Manager{ + Store: submit.NewPGStore(drv.DB()), + Builds: builder, + Registry: cfg.Registry.URL, + ContextStore: cfg.Registry.UserUploadsContext, + } + + // Restore subsystem (spec §7): the weak-SA restore Job mounts the target + // world PVC + the backup PVC and runs `felis restore`. It needs deployment- + // specific values that have no safe default — the felis image to run and the + // backup PVC to mount — so it is wired only when both are supplied. Otherwise + // the Restorer is left nil and the restore endpoint honestly returns 503 + // rather than enqueuing a Job that cannot run. (The archive store no longer + // gates wiring here: config.Validate rejects any recognized-but-unimplemented + // store at load, so by this point cfg.Archive.Store is guaranteed tarLocal.) + var restorer api.Restorer + felisImage, backupPVC := os.Getenv("FELIS_IMAGE"), os.Getenv("FELIS_BACKUP_PVC") + if felisImage != "" && backupPVC != "" { + rcfg := restoreConfig(cfg, felisImage, backupPVC) + restorer = &restore.Restorer{Jobs: restore.NewK8sJobs(cl), Config: rcfg} + } else { + fmt.Fprintln(stderr, "felis api: restore executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — restore endpoint returns 503") + } + + a := &api.API{ + Repo: api.NewPGRepo(drv.DB()), + Cluster: api.NewK8sCluster(cl, cfg.K8s.Namespace), + Console: api.NewK8sConsole(cl, cfg.K8s.Namespace), + Logs: api.NewK8sLogStreamer(clientset, cfg.K8s.Namespace), + // Build-log stream (spec §16) is scoped to the BUILD namespace — the same + // value the Builder renders Jobs into — so it follows where build Pods run. + BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace), + Internal: api.BearerTokenAuth{Token: token}, + Builder: builder, + Restorer: restorer, + Submissions: submissions, + // Keyfunc is intentionally nil: the external face fails closed until a + // JWKS-backed key function is wired (deployment integration point). + External: api.AccessVerifier{Audience: cfg.Auth.AccessJWTAud}, + RootDomain: cfg.Server.RootDomain, + WakeCooldown: 30 * time.Second, + } + fmt.Fprintln(stderr, "felis api: external face fails closed (Access JWKS key function not configured)") + + internalSrv := &http.Server{Addr: *internalAddr, Handler: a.InternalHandler()} + externalSrv := &http.Server{Addr: cfg.Server.Listen, Handler: a.ExternalHandler()} + + errc := make(chan error, 2) + go func() { errc <- internalSrv.ListenAndServe() }() + go func() { errc <- externalSrv.ListenAndServe() }() + fmt.Fprintf(stdout, "felis api: internal=%s external=%s\n", *internalAddr, cfg.Server.Listen) + + // reconcileBuilds drives the scan-gate translation: poll unfinished builds + // and advance any whose Job has reached a terminal phase. GET on a build also + // reconciles it, but this loop converges builds nobody is polling. + go reconcileBuilds(ctx, builder, stderr) + + select { + case <-ctx.Done(): + shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + _ = internalSrv.Shutdown(shutdownCtx) + _ = externalSrv.Shutdown(shutdownCtx) + return 0 + case err := <-errc: + if err != nil && err != http.ErrServerClosed { + fmt.Fprintf(stderr, "felis api: listener exited: %v\n", err) + return 1 + } + return 0 + } +} + +// buildConfig projects felis.toml onto the build subsystem config (spec §16, +// §24). Unset fields fall back to the build package's hardened defaults +// (felis-build namespace + weak SA, 30m deadline, resource limits). +func buildConfig(cfg *config.Config) build.Config { + return build.Config{ + Namespace: cfg.Registry.BuildNamespace, + RegistryURL: cfg.Registry.URL, + } +} + +// restoreConfig projects felis.toml + the deployment-supplied image and backup +// PVC onto the restore subsystem config (spec §7). The runtime identity, mount +// roots, resource limits, and weak SA fall back to the restore package's +// hardened defaults. BackupRoot tracks cfg.Archive.LocalPath because tarLocal +// archive refs are absolute: the restore Pod must mount the backup PVC at the +// same path the reaper wrote archives under, or the stored ref won't resolve. +func restoreConfig(cfg *config.Config, image, backupPVC string) restore.Config { + return restore.Config{ + Namespace: cfg.K8s.Namespace, + Image: image, + BackupPVC: backupPVC, + ArchiveStore: cfg.Archive.Store, + BackupRoot: cfg.Archive.LocalPath, + } +} + +// reconcileBuilds polls unfinished builds on an interval and advances any whose +// Job has reached a terminal phase. It exits when ctx is cancelled. +func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) { + t := time.NewTicker(15 * time.Second) + defer t.Stop() + for { + select { + case <-ctx.Done(): + return + case <-t.C: + if _, err := b.SyncAll(ctx); err != nil { + fmt.Fprintf(stderr, "felis api: build reconcile: %v\n", err) + } + } + } +} diff --git a/cmd/felis/main.go b/cmd/felis/main.go new file mode 100644 index 0000000..bd1015b --- /dev/null +++ b/cmd/felis/main.go @@ -0,0 +1,13 @@ +// Command felis is the single multi-call binary for the platform (spec §25): +// it dispatches to the api, operator, migrate, reaper, and apply subcommands. +// Building one binary keeps the shared packages (scheme, store, config) linked +// once and shipped in a single image. +package main + +import ( + "os" +) + +func main() { + os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) +} diff --git a/cmd/felis/manifests.go b/cmd/felis/manifests.go new file mode 100644 index 0000000..df3c5a9 --- /dev/null +++ b/cmd/felis/manifests.go @@ -0,0 +1,134 @@ +package main + +import ( + "flag" + "fmt" + "io" + "net" + "strings" + + "felis.lolicon.best/internal/platform" +) + +// multiFlag collects a repeatable string flag (e.g. --velocity-cidr a --velocity-cidr b). +type multiFlag []string + +func (m *multiFlag) String() string { return strings.Join(*m, ",") } + +func (m *multiFlag) Set(v string) error { + *m = append(*m, v) + return nil +} + +// cmdManifests renders the control-plane install bundle (spec §21, §22) — +// namespaces, the control-plane identities (SAs + namespaced Roles + +// RoleBindings — felis-api and felis-operator always, plus the destructive +// felis-reaper identity only when the retention reaper is enabled, gated with +// its CronJob), the weak build/restore Job SAs, the build/minecraft +// NetworkPolicies, and the running control-plane workloads (felis-api/operator +// Deployments + the in-cluster registry Deployment/Service/PVC) — as a single +// multi-document YAML stream on stdout, ready for `kubectl apply -f -`. +// +// It is a pure renderer: it never contacts a cluster and holds no credentials. +// --velocity-cidr is REQUIRED because the game NetworkPolicy fails closed without +// it; emitting a bundle whose 25565 ingress admitted no one would silently break +// the server, so the generator refuses rather than guess. +func cmdManifests(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("manifests", flag.ContinueOnError) + fs.SetOutput(stderr) + controlNS := fs.String("control-namespace", platform.DefaultControlNamespace, "namespace the control plane (api/operator/reaper) runs in") + minecraftNS := fs.String("minecraft-namespace", platform.DefaultMinecraftNamespace, "namespace MinecraftServer workloads run in") + buildNS := fs.String("build-namespace", platform.DefaultBuildNamespace, "namespace image-build Jobs run in") + registryNS := fs.String("registry-namespace", "", "namespace of the in-cluster registry (default: control namespace)") + registryPort := fs.Int("registry-port", 5000, "port the in-cluster registry listens on") + felisImage := fs.String("felis-image", "", "container image the felis-api/operator Deployments run, also passed through as FELIS_IMAGE (REQUIRED)") + registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry:2)") + backupPVC := fs.String("backup-pvc", "", "name of the backup PVC advertised to the restore executor via FELIS_BACKUP_PVC (default none = restore endpoint returns 503)") + worldsHostPath := fs.String("worlds-host-path", "", "node directory under which each world PVC is visible as /; enables the reaper CronJob (requires --backup-pvc and --archive-local-path)") + archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path") + var velocityCIDRs multiFlag + fs.Var(&velocityCIDRs, "velocity-cidr", "CIDR of an off-cluster Velocity proxy host allowed to reach game port 25565 (repeatable, REQUIRED)") + var packageCIDRs multiFlag + fs.Var(&packageCIDRs, "package-cidr", "CIDR of a package mirror build Pods may reach (repeatable; default none = no internet egress)") + if err := fs.Parse(args); err != nil { + return 2 + } + + // --velocity-cidr is mandatory: the game policy is fail-closed, so omitting it + // would render a server nobody can reach. Fail loudly at generation time. + if len(velocityCIDRs) == 0 { + fmt.Fprintln(stderr, "felis manifests: at least one --velocity-cidr is required "+ + "(the game NetworkPolicy fails closed without it; pass the Velocity proxy host CIDR, e.g. --velocity-cidr 10.0.0.5/32)") + return 2 + } + + // --felis-image is mandatory: the api/operator Deployments and the FELIS_IMAGE + // passthrough (used to launch the restore Job) have no safe default image. Same + // fail-loud contract as --velocity-cidr. + if *felisImage == "" { + fmt.Fprintln(stderr, "felis manifests: --felis-image is required "+ + "(the felis-api/operator Deployments run it and it is passed through as FELIS_IMAGE, e.g. --felis-image registry.felis.svc:5000/felis:v1)") + return 2 + } + for _, cidr := range append(append([]string{}, velocityCIDRs...), packageCIDRs...) { + if _, _, err := net.ParseCIDR(cidr); err != nil { + fmt.Fprintf(stderr, "felis manifests: invalid CIDR %q: %v\n", cidr, err) + return 2 + } + } + + // Retention/reaper rendering is opt-in and needs all three storage coordinates + // together: where worlds live (to read+archive them), the backup PVC (to write + // archives into), and the path it is mounted at (which MUST equal felis.toml + // [archive] local_path so tarLocal's absolute archive refs resolve). A partial + // configuration is almost certainly an operator mistake, so fail loud rather than + // silently drop retention. Asking for it without the other two is rejected; an + // empty trio renders the bundle WITHOUT the reaper and says so. + if *worldsHostPath != "" { + if *backupPVC == "" || *archiveLocalPath == "" { + fmt.Fprintln(stderr, "felis manifests: --worlds-host-path enables the reaper CronJob and requires "+ + "--backup-pvc and --archive-local-path too (--archive-local-path must equal felis.toml [archive] local_path)") + return 2 + } + // The reaper WILL render. Two deployment preconditions this generator cannot + // check would SILENTLY turn retention into a no-op if unmet — surface them as + // loudly as the fail-closed cases above, so an operator is never left with a + // reaper that reaps an empty directory. (Both are also in the WorldsHostPath + // flag/field docs, but nobody deploying from stdout reads those.) + fmt.Fprintf(stderr, "felis manifests: note: rendering the retention reaper CronJob (worlds hostPath %q). "+ + "Two preconditions are NOT verified here:\n"+ + " - each world PVC must be visible at %s/ on the node: a stock local-path-provisioner lays "+ + "volumes under PV-name paths (.../pvc-__/), so unless the worlds StorageClass is "+ + "arranged to expose /, the reaper tars an empty directory;\n"+ + " - the CronJob sets NO nodeSelector: a single-node starter pins it to the worlds implicitly, but "+ + "on a multi-node cluster you MUST add a nodeSelector for the node holding the worlds, or the reaper "+ + "may schedule where the hostPath is empty.\n", *worldsHostPath, *worldsHostPath) + } else { + fmt.Fprintln(stderr, "felis manifests: note: retention reaper CronJob not rendered "+ + "(pass --worlds-host-path, --backup-pvc and --archive-local-path to enable it)") + } + + out, err := platform.RenderYAML(platform.Params{ + ControlNamespace: *controlNS, + MinecraftNamespace: *minecraftNS, + BuildNamespace: *buildNS, + RegistryNamespace: *registryNS, + RegistryPort: int32(*registryPort), + FelisImage: *felisImage, + RegistryImage: *registryImage, + BackupPVC: *backupPVC, + WorldsHostPath: *worldsHostPath, + ArchiveLocalPath: *archiveLocalPath, + VelocityCIDRs: []string(velocityCIDRs), + PackageSourceCIDRs: []string(packageCIDRs), + }) + if err != nil { + fmt.Fprintf(stderr, "felis manifests: render: %v\n", err) + return 1 + } + if _, err := stdout.Write(out); err != nil { + fmt.Fprintf(stderr, "felis manifests: write: %v\n", err) + return 1 + } + return 0 +} diff --git a/cmd/felis/manifests_test.go b/cmd/felis/manifests_test.go new file mode 100644 index 0000000..5eb981b --- /dev/null +++ b/cmd/felis/manifests_test.go @@ -0,0 +1,158 @@ +package main + +import ( + "bytes" + "strings" + "testing" +) + +// TestManifestsRequiresVelocityCIDR proves the generator refuses to emit a bundle +// without --velocity-cidr (the game policy would otherwise fail closed silently). +func TestManifestsRequiresVelocityCIDR(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{"manifests"}, &out, &errBuf) + if code == 0 { + t.Fatalf("exit code = 0, want nonzero (missing --velocity-cidr)") + } + if !strings.Contains(errBuf.String(), "velocity-cidr") { + t.Errorf("expected a --velocity-cidr error, got %q", errBuf.String()) + } + if out.Len() != 0 { + t.Errorf("no YAML must be written when the flag is missing, got %q", out.String()) + } +} + +// TestManifestsRejectsBadCIDR proves CIDR inputs are validated. --felis-image is +// supplied so the only defect is the CIDR (the felis-image requirement is checked +// before CIDR validation, so omitting it would surface the wrong error). +func TestManifestsRejectsBadCIDR(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "not-a-cidr"}, &out, &errBuf) + if code == 0 { + t.Fatalf("exit code = 0, want nonzero (invalid CIDR)") + } + if !strings.Contains(errBuf.String(), "invalid CIDR") { + t.Errorf("expected an invalid-CIDR error, got %q", errBuf.String()) + } +} + +// TestManifestsRequiresFelisImage proves the generator refuses to emit a bundle +// without --felis-image (the api/operator Deployments have no default image, and +// FELIS_IMAGE has no safe guess). Same fail-loud contract as --velocity-cidr. +func TestManifestsRequiresFelisImage(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{"manifests", "--velocity-cidr", "10.0.0.5/32"}, &out, &errBuf) + if code == 0 { + t.Fatalf("exit code = 0, want nonzero (missing --felis-image)") + } + if !strings.Contains(errBuf.String(), "felis-image") { + t.Errorf("expected a --felis-image error, got %q", errBuf.String()) + } + if out.Len() != 0 { + t.Errorf("no YAML must be written when --felis-image is missing, got %q", out.String()) + } +} + +// TestManifestsRendersBundle proves the happy path: a valid invocation writes a +// multi-doc YAML bundle containing the fence kinds and no cluster-scoped RBAC. +func TestManifestsRendersBundle(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{"manifests", "--felis-image", "registry.felis.svc:5000/felis:v1", "--velocity-cidr", "10.0.0.5/32"}, &out, &errBuf) + if code != 0 { + t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String()) + } + text := out.String() + for _, want := range []string{ + "kind: Namespace", + "kind: ServiceAccount", + "kind: Role", + "kind: RoleBinding", + "kind: NetworkPolicy", + // The running control-plane workloads now in the bundle. + "kind: Deployment", + "kind: Service", + "kind: PersistentVolumeClaim", + "felis-allow-rcon-from-control-plane", + "felis-allow-game-from-velocity", + "10.0.0.5/32", + // The felis image flows through to the Deployments. + "registry.felis.svc:5000/felis:v1", + } { + if !strings.Contains(text, want) { + t.Errorf("rendered bundle missing %q", want) + } + } + if strings.Contains(text, "ClusterRole") { + t.Error("rendered bundle must not contain ClusterRole/ClusterRoleBinding") + } + // Without the retention flags, the reaper CronJob is not rendered and the + // generator says so on stderr. + if strings.Contains(text, "kind: CronJob") { + t.Error("no reaper CronJob must render without --worlds-host-path") + } + if !strings.Contains(errBuf.String(), "not rendered") { + t.Errorf("expected a 'reaper not rendered' notice on stderr, got %q", errBuf.String()) + } +} + +// TestManifestsReaperRequiresTrio proves --worlds-host-path is a fail-loud opt-in: +// asking for the reaper without the backup PVC and its mount path (which must equal +// [archive] local_path) is rejected rather than silently dropping retention. +func TestManifestsReaperRequiresTrio(t *testing.T) { + base := []string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "10.0.0.5/32", "--worlds-host-path", "/var/lib/felis/worlds"} + for _, extra := range [][]string{ + {}, // neither backup-pvc nor archive-local-path + {"--backup-pvc", "felis-backups"}, // missing archive-local-path + {"--archive-local-path", "/backups"}, // missing backup-pvc + } { + var out, errBuf bytes.Buffer + code := run(append(append([]string{}, base...), extra...), &out, &errBuf) + if code == 0 { + t.Fatalf("extra=%v: exit code = 0, want nonzero (incomplete reaper config)", extra) + } + if !strings.Contains(errBuf.String(), "backup-pvc") || !strings.Contains(errBuf.String(), "archive-local-path") { + t.Errorf("extra=%v: expected the trio requirement on stderr, got %q", extra, errBuf.String()) + } + if out.Len() != 0 { + t.Errorf("extra=%v: no YAML must be written on a fail-loud reject, got %q", extra, out.String()) + } + } +} + +// TestManifestsRendersReaper proves the happy path with the full retention trio: +// a batch/v1 CronJob is emitted, named felis-reaper, mounting the backup PVC at the +// supplied archive path. +func TestManifestsRendersReaper(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{ + "manifests", + "--felis-image", "registry.felis.svc:5000/felis:v1", + "--velocity-cidr", "10.0.0.5/32", + "--worlds-host-path", "/var/lib/felis/worlds", + "--backup-pvc", "felis-backups", + "--archive-local-path", "/backups", + }, &out, &errBuf) + if code != 0 { + t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String()) + } + text := out.String() + for _, want := range []string{ + "kind: CronJob", + "name: felis-reaper", + "/var/lib/felis/worlds", // the worlds hostPath + "claimName: felis-backups", + } { + if !strings.Contains(text, want) { + t.Errorf("rendered bundle with reaper missing %q", want) + } + } + // Rendering the reaper must also warn the operator about the two preconditions + // this generator cannot verify (else a misarranged hostPath silently no-ops + // retention): the / arrangement-dependency and the multi-node + // nodeSelector hazard. + for _, want := range []string{"local-path-provisioner", "nodeSelector"} { + if !strings.Contains(errBuf.String(), want) { + t.Errorf("reaper render must warn operators about %q on stderr, got %q", want, errBuf.String()) + } + } +} diff --git a/cmd/felis/migrate.go b/cmd/felis/migrate.go new file mode 100644 index 0000000..4bf752d --- /dev/null +++ b/cmd/felis/migrate.go @@ -0,0 +1,58 @@ +package main + +import ( + "context" + "flag" + "fmt" + "io" + + "felis.lolicon.best/internal/config" + "felis.lolicon.best/internal/store" +) + +// cmdMigrate implements `felis migrate up`: load config, open the database, and +// apply every pending embedded migration under the advisory lock (spec §6). +func cmdMigrate(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("migrate", flag.ContinueOnError) + fs.SetOutput(stderr) + cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml") + if err := fs.Parse(args); err != nil { + return 2 + } + if fs.Arg(0) != "up" { + fmt.Fprintln(stderr, "usage: felis migrate up [-config path]") + return 2 + } + + cfg, err := config.Load(*cfgPath) + if err != nil { + fmt.Fprintf(stderr, "felis migrate: %v\n", err) + return 1 + } + + ctx := context.Background() + drv, err := store.Open(ctx, cfg.Database.URL) + if err != nil { + fmt.Fprintf(stderr, "felis migrate: open database: %v\n", err) + return 1 + } + defer drv.Close() + + migrations, err := store.LoadMigrations() + if err != nil { + fmt.Fprintf(stderr, "felis migrate: load migrations: %v\n", err) + return 1 + } + + applied, err := store.Up(ctx, drv, migrations) + if err != nil { + fmt.Fprintf(stderr, "felis migrate: %v\n", err) + return 1 + } + if len(applied) == 0 { + fmt.Fprintln(stdout, "felis migrate: database already up to date") + } else { + fmt.Fprintf(stdout, "felis migrate: applied %d migration(s): %v\n", len(applied), applied) + } + return 0 +} diff --git a/cmd/felis/operator.go b/cmd/felis/operator.go new file mode 100644 index 0000000..893f978 --- /dev/null +++ b/cmd/felis/operator.go @@ -0,0 +1,75 @@ +package main + +import ( + "flag" + "fmt" + "io" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/operator" + "k8s.io/apimachinery/pkg/runtime" + utilruntime "k8s.io/apimachinery/pkg/util/runtime" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/cache" + metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server" +) + +// cmdOperator runs the MinecraftServer controller-manager (spec §5). It builds +// the scheme, wires the Reconciler with the production RCON prober, and blocks +// on the manager until the process receives a termination signal. +func cmdOperator(args []string, _, stderr io.Writer) int { + fs := flag.NewFlagSet("operator", flag.ContinueOnError) + fs.SetOutput(stderr) + metricsAddr := fs.String("metrics-bind-address", ":8080", "address the metric endpoint binds to") + // namespace MUST equal the [k8s] namespace felis-api is configured with, and + // the deployment manifests (felis manifests) render both from one value. It + // scopes the manager's cache (informers) to a single namespace so the operator + // can run under a namespaced Role instead of cluster-admin (spec §21). The + // default matches config.defaultNamespace, so an unconfigured deployment + // agrees; a mismatch would silently scope the cache to the wrong namespace and + // every reconcile would see zero servers — hence the watched namespace is + // logged at startup so a divergence surfaces immediately rather than silently. + namespace := fs.String("namespace", "minecraft", "namespace to watch; must match felis-api's [k8s] namespace") + if err := fs.Parse(args); err != nil { + return 2 + } + + scheme := runtime.NewScheme() + utilruntime.Must(clientgoscheme.AddToScheme(scheme)) + utilruntime.Must(v1alpha1.AddToScheme(scheme)) + + mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{ + Scheme: scheme, + Metrics: metricsserver.Options{BindAddress: *metricsAddr}, + // Scope every informer to the single watched namespace. Without this the + // cached client (mgr.GetClient) would LIST/WATCH cluster-wide, which a + // namespaced Role cannot grant — the operator would fail closed at runtime + // or, worse, demand cluster-admin. With it, the platform.OperatorRole + // (get/list/watch in one namespace) is exactly sufficient. + Cache: cache.Options{ + DefaultNamespaces: map[string]cache.Config{*namespace: {}}, + }, + }) + if err != nil { + fmt.Fprintf(stderr, "felis operator: create manager: %v\n", err) + return 1 + } + fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace) + + r := &operator.Reconciler{ + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + Prober: operator.RconProber{}, + } + if err := r.SetupWithManager(mgr); err != nil { + fmt.Fprintf(stderr, "felis operator: setup controller: %v\n", err) + return 1 + } + + if err := mgr.Start(ctrl.SetupSignalHandler()); err != nil { + fmt.Fprintf(stderr, "felis operator: manager exited: %v\n", err) + return 1 + } + return 0 +} diff --git a/cmd/felis/reaper.go b/cmd/felis/reaper.go new file mode 100644 index 0000000..96f3cf4 --- /dev/null +++ b/cmd/felis/reaper.go @@ -0,0 +1,187 @@ +package main + +import ( + "flag" + "fmt" + "io" + "path/filepath" + "strconv" + "strings" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/backup" + "felis.lolicon.best/internal/config" + "felis.lolicon.best/internal/reaper" + "felis.lolicon.best/internal/store" + "k8s.io/apimachinery/pkg/runtime" + utilruntime "k8s.io/apimachinery/pkg/util/runtime" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +// cmdReaper runs one pass of the world reaper / retention batch (spec §18). It +// is intentionally run-once-and-exit: a Kubernetes CronJob drives the daily +// cadence, and RunOnce is idempotent and restart-safe, so a missed or retried +// run simply converges. Only the tarLocal archive backend is wired in this +// build; the snapshot backends (§19) are a later integration. +func cmdReaper(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("reaper", flag.ContinueOnError) + fs.SetOutput(stderr) + cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml") + worldsRoot := fs.String("worlds-root", "/worlds", "mount root under which world PVCs are visible (tarLocal: /)") + if err := fs.Parse(args); err != nil { + return 2 + } + + cfg, err := config.Load(*cfgPath) + if err != nil { + fmt.Fprintf(stderr, "felis reaper: %v\n", err) + return 1 + } + + rcfg, err := reaperConfig(cfg) + if err != nil { + fmt.Fprintf(stderr, "felis reaper: %v\n", err) + return 1 + } + + archiver, err := buildArchiver(cfg, *worldsRoot) + if err != nil { + fmt.Fprintf(stderr, "felis reaper: %v\n", err) + return 1 + } + + ctx := ctrl.SetupSignalHandler() + + drv, err := store.Open(ctx, cfg.Database.URL) + if err != nil { + fmt.Fprintf(stderr, "felis reaper: open database: %v\n", err) + return 1 + } + defer drv.Close() + + scheme := runtime.NewScheme() + utilruntime.Must(clientgoscheme.AddToScheme(scheme)) + utilruntime.Must(v1alpha1.AddToScheme(scheme)) + cl, err := client.New(ctrl.GetConfigOrDie(), client.Options{Scheme: scheme}) + if err != nil { + fmt.Fprintf(stderr, "felis reaper: build k8s client: %v\n", err) + return 1 + } + + r := &reaper.Reaper{ + Cfg: rcfg, + Store: reaper.NewPGStore(drv.DB()), + Cluster: reaper.NewK8sCluster(cl, cfg.K8s.Namespace), + Archiver: archiver, + } + + sum, err := r.RunOnce(ctx) + if err != nil { + fmt.Fprintf(stderr, "felis reaper: %v\n", err) + return 1 + } + fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d warned=%d skipped=%d evicted=%d expired=%d\n", + sum.Evaluated, sum.WorldsReaped, sum.Warned, sum.Skipped, sum.EvictedEarly, sum.BackupsExpired) + return 0 +} + +// reaperConfig derives the reaper's retention windows from felis.toml. The 15d +// idle deadline is fixed by §18; only the warning offsets, retention, and the +// store soft-cap are configurable (§24). +func reaperConfig(cfg *config.Config) (reaper.Config, error) { + rc := reaper.DefaultConfig() + if v := cfg.Archive.Retention; v != "" { + d, err := parseSpanDuration(v) + if err != nil { + return rc, fmt.Errorf("[archive] retention %q: %w", v, err) + } + rc.Retention = d + } + if len(cfg.Archive.WarnBefore) > 0 { + offs := make([]time.Duration, 0, len(cfg.Archive.WarnBefore)) + for _, w := range cfg.Archive.WarnBefore { + d, err := parseSpanDuration(w) + if err != nil { + return rc, fmt.Errorf("[archive] warn_before %q: %w", w, err) + } + offs = append(offs, d) + } + rc.WarnBefore = offs + } + if v := cfg.Archive.MaxLocalBytes; v != "" { + b, err := parseByteSize(v) + if err != nil { + return rc, fmt.Errorf("[archive] max_local_bytes %q: %w", v, err) + } + rc.MaxLocalBytes = b + } + return rc, nil +} + +// buildArchiver constructs the WorldArchiver. Only tarLocal is implemented in +// this build; the resolver maps each world PVC to /, the mount +// convention the reaper Job is deployed with. +func buildArchiver(cfg *config.Config, worldsRoot string) (backup.WorldArchiver, error) { + switch cfg.Archive.Store { + case "tarLocal": + return &backup.TarLocal{ + BackupRoot: cfg.Archive.LocalPath, + Resolve: func(pvc string) (string, error) { + return filepath.Join(worldsRoot, pvc), nil + }, + }, nil + default: + return nil, fmt.Errorf("[archive] store %q is not implemented in this build (only tarLocal)", cfg.Archive.Store) + } +} + +// parseSpanDuration parses the human spans used in felis.toml's [archive] table: +// "3mo" (months≈30d), "15d" (days), or any time.ParseDuration unit ("12h"). +func parseSpanDuration(s string) (time.Duration, error) { + s = strings.TrimSpace(s) + switch { + case strings.HasSuffix(s, "mo"): + n, err := strconv.Atoi(strings.TrimSuffix(s, "mo")) + if err != nil { + return 0, err + } + return time.Duration(n) * 30 * 24 * time.Hour, nil + case strings.HasSuffix(s, "d"): + n, err := strconv.Atoi(strings.TrimSuffix(s, "d")) + if err != nil { + return 0, err + } + return time.Duration(n) * 24 * time.Hour, nil + default: + return time.ParseDuration(s) + } +} + +// parseByteSize parses a Kubernetes-style quantity ("200Gi", "10G") into bytes. +// An empty string means unlimited (0). +func parseByteSize(s string) (int64, error) { + s = strings.TrimSpace(s) + if s == "" { + return 0, nil + } + units := []struct { + suffix string + mul int64 + }{ + {"Gi", 1 << 30}, {"Mi", 1 << 20}, {"Ki", 1 << 10}, + {"G", 1_000_000_000}, {"M", 1_000_000}, {"K", 1_000}, + } + for _, u := range units { + if strings.HasSuffix(s, u.suffix) { + n, err := strconv.ParseFloat(strings.TrimSuffix(s, u.suffix), 64) + if err != nil { + return 0, err + } + return int64(n * float64(u.mul)), nil + } + } + return strconv.ParseInt(s, 10, 64) +} diff --git a/cmd/felis/restore.go b/cmd/felis/restore.go new file mode 100644 index 0000000..e1816ac --- /dev/null +++ b/cmd/felis/restore.go @@ -0,0 +1,83 @@ +package main + +import ( + "flag" + "fmt" + "io" + "path/filepath" + "strings" + + "felis.lolicon.best/internal/backup" + ctrl "sigs.k8s.io/controller-runtime" +) + +// cmdRestore is the in-Pod entrypoint the restore Job runs (internal/restore +// renders a Pod whose command is `felis restore`). It extracts a world archive +// from the backup mount into the world mount and exits — it is NOT a +// user-facing command and is never invoked by hand. +// +// It deliberately holds NO database credentials and never calls config.Load: +// the felis-api made the authorization decision and looked up the archive ref; +// this process is the unprivileged hands that move bytes. Its entire input is +// the five flags below, mirroring the four-power isolation the rendered Job +// enforces (a restore Pod must not be able to reach the felis DB or the K8s +// API). All of those guarantees live in the Pod spec; this command only needs +// the archive store and the two mount roots. +func cmdRestore(args []string, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("restore", flag.ContinueOnError) + fs.SetOutput(stderr) + server := fs.String("server", "", "server name being restored (for logging)") + ref := fs.String("ref", "", "absolute path to the archive on the backup mount") + store := fs.String("archive-store", "tarLocal", "archive backend (only tarLocal is implemented)") + backupRoot := fs.String("backup-root", "/backups", "mount path of the backup PVC (archive refs must resolve under it)") + worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC the archive extracts into") + if err := fs.Parse(args); err != nil { + return 2 + } + + if *ref == "" { + fmt.Fprintln(stderr, "felis restore: --ref is required") + return 2 + } + if *store != "tarLocal" { + fmt.Fprintf(stderr, "felis restore: archive store %q is not implemented in this build (only tarLocal)\n", *store) + return 1 + } + // Defense in depth: the ref comes from felis-api (trusted), but this process + // is the one that opens it, so it confirms the ref stays within the backup + // mount. A ref outside it would mean reading an arbitrary host path, which a + // restore Pod must never do. + if !refWithinRoot(*ref, *backupRoot) { + fmt.Fprintf(stderr, "felis restore: ref %q is not under backup root %q\n", *ref, *backupRoot) + return 1 + } + + // The world PVC is mounted directly at worldsRoot, so the resolver returns it + // for any target; the archive ref is absolute and is opened directly. This is + // the same TarLocal the reaper uses to write archives, run in reverse. + archiver := &backup.TarLocal{ + BackupRoot: *backupRoot, + Resolve: func(string) (string, error) { + return *worldsRoot, nil + }, + } + + ctx := ctrl.SetupSignalHandler() + if err := archiver.Restore(ctx, backup.ArchiveRef(*ref), *server); err != nil { + fmt.Fprintf(stderr, "felis restore: %v\n", err) + return 1 + } + fmt.Fprintf(stdout, "felis restore: server=%s restored from %s into %s\n", *server, *ref, *worldsRoot) + return 0 +} + +// refWithinRoot reports whether ref resolves to a path inside root. Both are +// cleaned first so "/backups/../etc/passwd" cannot slip through. +func refWithinRoot(ref, root string) bool { + cleanRoot := filepath.Clean(root) + cleanRef := filepath.Clean(ref) + if cleanRef == cleanRoot { + return true + } + return strings.HasPrefix(cleanRef, cleanRoot+string(filepath.Separator)) +} diff --git a/cmd/felis/restore_test.go b/cmd/felis/restore_test.go new file mode 100644 index 0000000..1f154d7 --- /dev/null +++ b/cmd/felis/restore_test.go @@ -0,0 +1,140 @@ +package main + +import ( + "bytes" + "context" + "os" + "path/filepath" + "strings" + "testing" + + "felis.lolicon.best/internal/backup" +) + +// archiveTempWorld tars a freshly populated world directory with the real +// TarLocal and returns the absolute archive ref plus the backup root it lives +// under, so the restore round-trip below exercises production code end to end +// without a cluster. +func archiveTempWorld(t *testing.T, files map[string]string) (ref string, backupRoot string) { + t.Helper() + srcDir := t.TempDir() + for name, content := range files { + full := filepath.Join(srcDir, filepath.FromSlash(name)) + if err := os.MkdirAll(filepath.Dir(full), 0o750); err != nil { + t.Fatalf("mkdir: %v", err) + } + if err := os.WriteFile(full, []byte(content), 0o640); err != nil { + t.Fatalf("write: %v", err) + } + } + backupRoot = t.TempDir() + ar := &backup.TarLocal{ + BackupRoot: backupRoot, + Resolve: func(string) (string, error) { return srcDir, nil }, + } + got, _, err := ar.Archive(context.Background(), "survival", "world-survival-0") + if err != nil { + t.Fatalf("Archive: %v", err) + } + return string(got), backupRoot +} + +// The restore subcommand must extract the archived world into the target world +// mount — the real reverse of what the reaper wrote. +func TestRestoreSubcommandRoundTrips(t *testing.T) { + files := map[string]string{ + "level.dat": "world-seed", + "region/r.0.0.mca": "chunk-bytes", + "playerdata/uuid.json": "{}", + } + ref, backupRoot := archiveTempWorld(t, files) + + // Seed the target with a stale file the archive does not contain: the restore + // must prune it (replace semantics), not leave it behind. This is the path the + // admin/owner rollback-onto-a-stopped-populated-PVC case actually takes. + targetDir := t.TempDir() + stale := filepath.Join(targetDir, "region", "r.9.9.mca") + if err := os.MkdirAll(filepath.Dir(stale), 0o750); err != nil { + t.Fatalf("mkdir stale: %v", err) + } + if err := os.WriteFile(stale, []byte("griefer-chunk"), 0o640); err != nil { + t.Fatalf("seed stale: %v", err) + } + + var out, errBuf bytes.Buffer + code := run([]string{ + "restore", + "--server", "survival", + "--ref", ref, + "--archive-store", "tarLocal", + "--backup-root", backupRoot, + "--worlds-root", targetDir, + }, &out, &errBuf) + if code != 0 { + t.Fatalf("restore exit = %d, stderr=%q", code, errBuf.String()) + } + + for name, want := range files { + full := filepath.Join(targetDir, filepath.FromSlash(name)) + got, err := os.ReadFile(full) + if err != nil { + t.Errorf("restored file %q missing: %v", name, err) + continue + } + if string(got) != want { + t.Errorf("restored %q = %q, want %q", name, got, want) + } + } + + if _, err := os.Stat(stale); !os.IsNotExist(err) { + t.Errorf("stale file survived restore (err=%v); replace semantics broken", err) + } +} + +// A ref outside the backup mount must be rejected before any file is opened — +// the restore Pod must never read an arbitrary host path. +func TestRestoreSubcommandRejectsRefOutsideBackupRoot(t *testing.T) { + backupRoot := t.TempDir() + outside := filepath.Join(t.TempDir(), "evil.tar.gz") + + var out, errBuf bytes.Buffer + code := run([]string{ + "restore", + "--server", "survival", + "--ref", outside, + "--backup-root", backupRoot, + "--worlds-root", t.TempDir(), + }, &out, &errBuf) + if code != 1 { + t.Fatalf("exit = %d, want 1 for an out-of-root ref", code) + } + if !strings.Contains(errBuf.String(), "not under backup root") { + t.Errorf("expected containment error, got %q", errBuf.String()) + } +} + +// An unimplemented archive store must fail loudly, not silently no-op. +func TestRestoreSubcommandRejectsUnknownStore(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{ + "restore", + "--ref", "/backups/x.tar.gz", + "--archive-store", "s3snapshot", + "--backup-root", "/backups", + }, &out, &errBuf) + if code != 1 { + t.Fatalf("exit = %d, want 1 for an unknown store", code) + } + if !strings.Contains(errBuf.String(), "not implemented") { + t.Errorf("expected not-implemented error, got %q", errBuf.String()) + } +} + +// --ref is mandatory: a missing ref is a usage error, not a panic. +func TestRestoreSubcommandRequiresRef(t *testing.T) { + var out, errBuf bytes.Buffer + code := run([]string{"restore", "--backup-root", "/backups"}, &out, &errBuf) + if code != 2 { + t.Fatalf("exit = %d, want 2 for a missing --ref", code) + } +} diff --git a/cmd/felis/run.go b/cmd/felis/run.go new file mode 100644 index 0000000..32c3dee --- /dev/null +++ b/cmd/felis/run.go @@ -0,0 +1,63 @@ +package main + +import ( + "fmt" + "io" +) + +const usage = `felis — Kubernetes-native Minecraft server orchestration + +Usage: + felis [flags] + +Commands: + migrate up Apply embedded database migrations under an advisory lock + operator Run the MinecraftServer controller-manager + api Run the felis-api HTTP server + reaper Run the world reaper / backup batch + restore Extract a world archive into a world volume (internal Job entrypoint) + manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML + apply Apply a MinecraftServer manifest + +Run "felis -h" for command-specific flags. +` + +// run dispatches a subcommand. It is separate from main so the router is +// testable without spawning a process. +func run(args []string, stdout, stderr io.Writer) int { + if len(args) == 0 { + fmt.Fprint(stderr, usage) + return 2 + } + cmd, rest := args[0], args[1:] + switch cmd { + case "migrate": + return cmdMigrate(rest, stdout, stderr) + case "operator": + return cmdOperator(rest, stdout, stderr) + case "api": + return cmdAPI(rest, stdout, stderr) + case "reaper": + return cmdReaper(rest, stdout, stderr) + case "restore": + return cmdRestore(rest, stdout, stderr) + case "manifests": + return cmdManifests(rest, stdout, stderr) + case "apply": + return notImplemented("apply", "MinecraftServer manifest apply", stderr) + case "-h", "--help", "help": + fmt.Fprint(stdout, usage) + return 0 + default: + fmt.Fprintf(stderr, "felis: unknown command %q\n\n%s", cmd, usage) + return 2 + } +} + +// notImplemented reports a subcommand that is wired into the CLI surface but +// whose implementation lands in a later phase. It fails loudly rather than +// pretending to do work. +func notImplemented(name, desc string, stderr io.Writer) int { + fmt.Fprintf(stderr, "felis %s: not implemented yet — %s\n", name, desc) + return 3 +} diff --git a/cmd/felis/run_test.go b/cmd/felis/run_test.go new file mode 100644 index 0000000..64df26b --- /dev/null +++ b/cmd/felis/run_test.go @@ -0,0 +1,86 @@ +package main + +import ( + "bytes" + "strings" + "testing" +) + +func TestRunNoArgsPrintsUsage(t *testing.T) { + var out, errBuf bytes.Buffer + if code := run(nil, &out, &errBuf); code != 2 { + t.Errorf("exit code = %d, want 2", code) + } + if !strings.Contains(errBuf.String(), "Usage:") { + t.Errorf("expected usage on stderr, got %q", errBuf.String()) + } +} + +func TestRunHelp(t *testing.T) { + var out, errBuf bytes.Buffer + if code := run([]string{"help"}, &out, &errBuf); code != 0 { + t.Errorf("exit code = %d, want 0", code) + } + if !strings.Contains(out.String(), "felis") { + t.Errorf("expected usage on stdout, got %q", out.String()) + } +} + +func TestRunUnknownCommand(t *testing.T) { + var out, errBuf bytes.Buffer + if code := run([]string{"frobnicate"}, &out, &errBuf); code != 2 { + t.Errorf("exit code = %d, want 2", code) + } + if !strings.Contains(errBuf.String(), "unknown command") { + t.Errorf("expected unknown-command error, got %q", errBuf.String()) + } +} + +func TestRunNotImplementedSubcommands(t *testing.T) { + for _, cmd := range []string{"apply"} { + var out, errBuf bytes.Buffer + if code := run([]string{cmd}, &out, &errBuf); code != 3 { + t.Errorf("%s exit code = %d, want 3", cmd, code) + } + if !strings.Contains(errBuf.String(), "not implemented yet") { + t.Errorf("%s: expected not-implemented notice, got %q", cmd, errBuf.String()) + } + } +} + +func TestRunReaperValidatesConfigBeforeDialing(t *testing.T) { + var out, errBuf bytes.Buffer + // Like api, reaper must fail fast (exit 1) at config load, before any + // database or cluster contact. + code := run([]string{"reaper", "-config", "this-file-does-not-exist.toml"}, &out, &errBuf) + if code != 1 { + t.Errorf("exit code = %d, want 1", code) + } + if !strings.Contains(errBuf.String(), "felis reaper:") { + t.Errorf("expected reaper error on stderr, got %q", errBuf.String()) + } +} + +func TestRunAPIValidatesConfigBeforeDialing(t *testing.T) { + var out, errBuf bytes.Buffer + // A non-existent config must fail fast (exit 1) at config load, before any + // database or cluster contact. + code := run([]string{"api", "-config", "this-file-does-not-exist.toml"}, &out, &errBuf) + if code != 1 { + t.Errorf("exit code = %d, want 1", code) + } + if !strings.Contains(errBuf.String(), "felis api:") { + t.Errorf("expected api error on stderr, got %q", errBuf.String()) + } +} + +func TestRunMigrateRequiresUpVerb(t *testing.T) { + var out, errBuf bytes.Buffer + // "migrate" with no verb should fail fast on usage, not touch a database. + if code := run([]string{"migrate"}, &out, &errBuf); code != 2 { + t.Errorf("exit code = %d, want 2", code) + } + if !strings.Contains(errBuf.String(), "felis migrate up") { + t.Errorf("expected migrate usage, got %q", errBuf.String()) + } +} diff --git a/internal/platform/bundle.go b/internal/platform/bundle.go new file mode 100644 index 0000000..8e3ab13 --- /dev/null +++ b/internal/platform/bundle.go @@ -0,0 +1,152 @@ +package platform + +import ( + "bytes" + "fmt" + + "felis.lolicon.best/internal/build" + "felis.lolicon.best/internal/restore" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "sigs.k8s.io/yaml" +) + +// Object is the common interface of every rendered install object. runtime.Object +// supplies the GVK (so the YAML carries apiVersion/kind via the embedded +// TypeMeta) and metav1.Object supplies name/namespace. Every type Objects emits +// satisfies it, which lets the §22 tests iterate the bundle uniformly. +type Object interface { + runtime.Object + metav1.Object +} + +// Objects assembles the complete security-fence install bundle for p, in a +// deterministic order: the namespaces, the control-plane identities (SAs + +// namespaced Roles + RoleBindings — felis-api and felis-operator always, plus the +// destructive felis-reaper identity only when retention is enabled, gated with its +// CronJob), the two weak Job SAs (build/restore, which have NO Role anywhere), the +// build-namespace egress NetworkPolicy, and the minecraft-namespace ingress +// NetworkPolicies. +// +// Scope: this is the authorization + network fence (spec §21, §22) plus the +// running control-plane workloads it fences — the felis-api / felis-operator +// Deployments and the in-cluster registry (Deployment + Service + PVC), which +// make the SAs and NetworkPolicy peers refer to something real (see workloads.go). +// The reaper CronJob is also part of Workloads, rendered only when the retention +// storage topology is supplied (WorldsHostPath + BackupPVC + ArchiveLocalPath — +// workloads.go documents the gate and the shape-asserted hostPath caveat). Still +// deliberately NOT rendered: a felis-api Service (its exposure is an out-of-band +// deployment choice and nothing in-tree dials it). The per-server StatefulSet is +// never a static manifest — the operator renders it at reconcile time +// (internal/operator). +func Objects(p Params) []Object { + p = p.withDefaults() + var objs []Object + + // Namespaces first, each carrying the immutable name label the NetworkPolicy + // namespaceSelectors match on. (K8s ≥1.21 adds this label automatically, but + // rendering it makes the bundle self-contained and the selectors provable.) + for _, ns := range distinctNamespaces(p) { + objs = append(objs, namespaceObject(ns)) + } + + // Control-plane RBAC: SAs, then Roles, then RoleBindings. + rbac := ControlPlaneRBAC(p) + for _, sa := range rbac.ServiceAccounts { + objs = append(objs, sa) + } + for _, r := range rbac.Roles { + objs = append(objs, r) + } + for _, rb := range rbac.RoleBindings { + objs = append(objs, rb) + } + + // Weak Job SAs. They come from the build/restore packages (single source of + // truth for AutomountServiceAccountToken=false), which set ObjectMeta but not + // TypeMeta — stamp it so the YAML header is present. The restore Job runs in + // the minecraft namespace (felis-api creates it there); the build Job in the + // build namespace. + objs = append(objs, + withSATypeMeta(build.BuildServiceAccount(p.BuildNamespace, SABuild)), + withSATypeMeta(restore.RestoreServiceAccount(p.MinecraftNamespace, SARestore)), + ) + + // Build-namespace egress lock (reused from internal/build; TypeMeta stamped). + buildNP := build.BuildNetworkPolicy(build.NetPolParams{ + Namespace: p.BuildNamespace, + RegistryNamespace: p.RegistryNamespace, + RegistryPort: p.RegistryPort, + PackageSourceCIDRs: p.PackageSourceCIDRs, + }) + buildNP.TypeMeta = metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"} + objs = append(objs, buildNP) + + // Minecraft-namespace ingress fence (default-deny + RCON + game). + for _, np := range MinecraftNetworkPolicies(p) { + objs = append(objs, np) + } + + // The running control-plane the fence protects: felis-api/operator Deployments + // (which bind the SAs to workloads and stamp the RCON-peer labels) and the + // in-cluster registry (Deployment + Service + PVC) the build egress policy + // targets. Rendered last so the identities/policies they reference appear first. + objs = append(objs, Workloads(p)...) + + return objs +} + +// RenderYAML marshals Objects(p) into a single multi-document YAML stream (the +// `---`-separated form kubectl apply consumes). It is the verifiable source of +// truth a Helm chart would otherwise only re-encode. +func RenderYAML(p Params) ([]byte, error) { + var buf bytes.Buffer + for i, obj := range Objects(p) { + if i > 0 { + buf.WriteString("---\n") + } + b, err := yaml.Marshal(obj) + if err != nil { + return nil, fmt.Errorf("marshal %T %s/%s: %w", obj, obj.GetNamespace(), obj.GetName(), err) + } + buf.Write(b) + } + return buf.Bytes(), nil +} + +// distinctNamespaces lists the namespaces the bundle installs into, de-duplicated +// and in a stable order (control, minecraft, build, then registry if it is a +// distinct namespace). +func distinctNamespaces(p Params) []string { + order := []string{p.ControlNamespace, p.MinecraftNamespace, p.BuildNamespace, p.RegistryNamespace} + seen := make(map[string]bool, len(order)) + out := make([]string, 0, len(order)) + for _, ns := range order { + if ns == "" || seen[ns] { + continue + } + seen[ns] = true + out = append(out, ns) + } + return out +} + +// namespaceObject renders a Namespace carrying the immutable name label the +// NetworkPolicy namespaceSelectors key on. +func namespaceObject(name string) *corev1.Namespace { + return &corev1.Namespace{ + TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Namespace"}, + ObjectMeta: metav1.ObjectMeta{ + Name: name, + Labels: map[string]string{"kubernetes.io/metadata.name": name}, + }, + } +} + +// withSATypeMeta stamps the apiVersion/kind on a ServiceAccount built by a package +// that only set its ObjectMeta. +func withSATypeMeta(sa *corev1.ServiceAccount) *corev1.ServiceAccount { + sa.TypeMeta = metav1.TypeMeta{APIVersion: "v1", Kind: "ServiceAccount"} + return sa +} diff --git a/internal/platform/bundle_test.go b/internal/platform/bundle_test.go new file mode 100644 index 0000000..83af86c --- /dev/null +++ b/internal/platform/bundle_test.go @@ -0,0 +1,141 @@ +package platform + +import ( + "strings" + "testing" + + corev1 "k8s.io/api/core/v1" + "sigs.k8s.io/yaml" +) + +// TestObjects_EveryDocHasTypeMeta enforces that every rendered object carries an +// apiVersion and a kind. The build/restore packages build their SAs/NetworkPolicy +// without TypeMeta, so this guards the stamping in bundle.go specifically. +func TestObjects_EveryDocHasTypeMeta(t *testing.T) { + for _, obj := range Objects(testParams()) { + gvk := obj.GetObjectKind().GroupVersionKind() + if gvk.Kind == "" || gvk.Version == "" { + t.Errorf("%T %s/%s has empty TypeMeta (kind=%q version=%q)", + obj, obj.GetNamespace(), obj.GetName(), gvk.Kind, gvk.Version) + } + } +} + +// TestObjects_NamespacesLabeled checks the three namespaces are rendered with the +// immutable name label the NetworkPolicy namespaceSelectors key on. +func TestObjects_NamespacesLabeled(t *testing.T) { + want := map[string]bool{"felis": false, "minecraft": false, "felis-build": false} + for _, obj := range Objects(testParams()) { + ns, ok := obj.(*corev1.Namespace) + if !ok { + continue + } + if _, expected := want[ns.Name]; expected { + want[ns.Name] = true + } + if got := ns.Labels["kubernetes.io/metadata.name"]; got != ns.Name { + t.Errorf("namespace %q metadata.name label = %q, want %q", ns.Name, got, ns.Name) + } + } + for name, found := range want { + if !found { + t.Errorf("namespace %q not rendered", name) + } + } +} + +// TestWeakJobSAs_Isolated proves the build/restore SAs are present, disable token +// auto-mounting, and — the key isolation invariant — are referenced by NO +// RoleBinding anywhere. Their powerlessness is the absence of any binding. +func TestWeakJobSAs_Isolated(t *testing.T) { + objs := Objects(testParams()) + + var build, restore *corev1.ServiceAccount + for _, obj := range objs { + sa, ok := obj.(*corev1.ServiceAccount) + if !ok { + continue + } + switch sa.Name { + case SABuild: + build = sa + case SARestore: + restore = sa + } + } + if build == nil { + t.Fatal("build SA not rendered") + } + if restore == nil { + t.Fatal("restore SA not rendered") + } + for _, sa := range []*corev1.ServiceAccount{build, restore} { + if sa.AutomountServiceAccountToken == nil || *sa.AutomountServiceAccountToken { + t.Errorf("%s must set AutomountServiceAccountToken=false", sa.Name) + } + } + + // No RoleBinding may name the weak SAs as a subject. + for _, rb := range ControlPlaneRBAC(testParams()).RoleBindings { + for _, s := range rb.Subjects { + if s.Name == SABuild || s.Name == SARestore { + t.Errorf("binding %q must NOT grant any Role to weak SA %q", rb.Name, s.Name) + } + } + } +} + +// TestRenderYAML_ParsesAndIsFenced renders the full bundle and asserts: every +// document parses with an apiVersion+kind, the expected kinds are present, and the +// stream contains no cluster-scoped RBAC (the ClusterRole/ClusterRoleBinding red +// line, checked on the literal output the way CI would). +func TestRenderYAML_ParsesAndIsFenced(t *testing.T) { + out, err := RenderYAML(testParams()) + if err != nil { + t.Fatalf("RenderYAML: %v", err) + } + text := string(out) + + if strings.Contains(text, "ClusterRole") { + t.Error("rendered bundle must not contain ClusterRole or ClusterRoleBinding") + } + + kinds := map[string]bool{} + for _, doc := range strings.Split(text, "\n---\n") { + doc = strings.TrimSpace(doc) + if doc == "" { + continue + } + var m map[string]interface{} + if err := yaml.Unmarshal([]byte(doc), &m); err != nil { + t.Fatalf("doc does not parse: %v\n---\n%s", err, doc) + } + kind, _ := m["kind"].(string) + apiVersion, _ := m["apiVersion"].(string) + if kind == "" || apiVersion == "" { + t.Errorf("doc missing apiVersion/kind: %s", doc) + } + kinds[kind] = true + } + for _, want := range []string{"Namespace", "ServiceAccount", "Role", "RoleBinding", "NetworkPolicy"} { + if !kinds[want] { + t.Errorf("rendered bundle is missing a %s", want) + } + } +} + +// TestRenderYAML_Deterministic guards that the render is stable (no map-ordering +// nondeterminism leaking into the manifest), so a regenerated bundle diffs cleanly. +func TestRenderYAML_Deterministic(t *testing.T) { + a, err := RenderYAML(testParams()) + if err != nil { + t.Fatal(err) + } + b, err := RenderYAML(testParams()) + if err != nil { + t.Fatal(err) + } + if string(a) != string(b) { + t.Error("RenderYAML must be deterministic across calls") + } +} diff --git a/internal/platform/doc.go b/internal/platform/doc.go new file mode 100644 index 0000000..fb5d0c5 --- /dev/null +++ b/internal/platform/doc.go @@ -0,0 +1,58 @@ +// Package platform renders the cluster-install objects that bound what every +// Felis identity may do (spec §21, §22): the RBAC for the control-plane +// identities (felis-api and felis-operator always; felis-reaper only when the +// retention reaper is enabled, gated with its CronJob), the weak service +// accounts for the build/restore Jobs, and the minecraft-namespace +// NetworkPolicies that fence server pods. Like internal/build and +// internal/restore, the objects are pure Go-typed values so the security-critical +// shape is unit-tested here, because no cluster runs in this environment. +// +// # Why the RBAC is derived from code, not from the spec prose +// +// The allowlist is built from the actual K8s API call sites in the control +// plane, not transcribed from §21's wording, because the prose is incomplete and +// in places wrong. Concretely, the rendered Roles encode these realities the +// prose misses: +// +// - felis-api creates the restore Job in the minecraft namespace (see +// internal/restore) and the build Job in the build namespace (see +// internal/build), so it needs batch/jobs verbs in BOTH namespaces — not +// just minecraftservers + secrets. +// - The RCON port (25575) is reached by BOTH felis-api (console writes, see +// internal/api.console) AND felis-operator (the readiness prober, see +// internal/operator.prober). The RCON NetworkPolicy peer is therefore +// {api, operator}, not api alone. +// - felis-api ALSO reads pod logs for the read-side console (spec §8 读=pods/log +// follow, see internal/api.logstream): it gets pods:list (to find a server's +// running pod by label) + pods/log:get (to follow it), and deliberately NOT +// pods:get — the streamer never reads a pod's full object. This is the only +// pods grant in the bundle, and it belongs to the api, not the operator. +// - felis-operator uses the manager's CACHED client (informers), so even a +// single Get on a type requires list+watch on it; felis-api and felis-reaper +// use direct clients, so they need only the verbs they literally call. +// - The reaper is a distinct, destructive identity: it deletes +// PersistentVolumeClaims and patches minecraftservers, powers that belong to +// neither the api nor the operator. It gets its own SA and Role. +// - The operator never emits Events, sets finalizers, deletes workloads, or +// touches pods/PVCs, so none of those appear in its Role — and +// minecraftservers/status carries only `update` (it calls Status().Update), +// never get/patch. +// +// # Honesty +// +// The typed render and the §22 invariant tests are Oracle-verifiable here. What +// is NOT verifiable in this toolchain, and is therefore code-complete-but- +// unverified (the same bucket as the K8s E2E): +// +// - that the RBAC actually binds and authorizes on a live apiserver; +// - that the NetworkPolicies are enforced by the cluster CNI (a CNI without +// NetworkPolicy support silently ignores them); +// - the egress_mode=loadbalancer source-IP caveat (spec §20): with MetalLB L2, +// a per-server Service must set externalTrafficPolicy=Local or the client IP +// is SNAT'd and the 25565 ipBlock allowlist will not match the Velocity host. +// +// A Helm-chart wrapper around these objects is intentionally NOT produced: it +// could not be validated here (no helm/kubeconform), and the typed generator +// (RenderYAML, surfaced by `felis manifests`) is the verifiable source of truth a +// chart would only re-encode. +package platform diff --git a/internal/platform/identities.go b/internal/platform/identities.go new file mode 100644 index 0000000..5a32f89 --- /dev/null +++ b/internal/platform/identities.go @@ -0,0 +1,178 @@ +package platform + +// Identity constants and the Params that parameterise the install bundle. +// +// The label and SA-name constants are PINNED here because three independent +// things must agree on them and there is no cluster to catch a disagreement at +// runtime: (1) the control-plane Deployments that carry the pod labels and run +// as these SAs, (2) the NetworkPolicy peers that select the RCON callers by those +// labels, and (3) the RoleBindings whose subjects name those SAs. Changing a +// value here is a deliberate, test-guarded act. + +// Recommended-label keys (the app.kubernetes.io/* set) applied to control-plane +// objects. We select on Component + PartOf in the RCON NetworkPolicy, so these +// must stay stable: a NetworkPolicy peer match is a plain string compare, and a +// renamed value silently stops matching the pods it is meant to admit. +const ( + LabelName = "app.kubernetes.io/name" + LabelComponent = "app.kubernetes.io/component" + LabelPartOf = "app.kubernetes.io/part-of" + + appName = "felis" + controlPlanePartOf = "felis-control-plane" + + // Component values distinguish the three control-plane workloads. The RCON + // policy admits only {api, operator}; reaper never opens an RCON connection, + // so it is deliberately excluded. + ComponentAPI = "api" + ComponentOperator = "operator" + ComponentReaper = "reaper" + + // ComponentRegistry labels the in-cluster image registry. It is deliberately + // NOT part-of=felis-control-plane: the registry is a supporting workload, not + // a control-plane identity, so the RCON NetworkPolicy peer (which requires + // part-of=felis-control-plane) can never select it. + ComponentRegistry = "registry" +) + +// Service-account names. The control-plane SAs (api/operator/reaper) are bound to +// the namespaced Roles in this package; the weak Job SAs (build/restore) have NO +// Role anywhere — their isolation is the absence of any binding (spec §16, §21). +const ( + SAAPI = "felis-api" + SAOperator = "felis-operator" + SAReaper = "felis-reaper" + SABuild = "felis-build" + SARestore = "felis-restore" +) + +// Default namespaces. They match the defaults used elsewhere in the tree +// (config.defaultNamespace = "minecraft", build/restore package defaults) so an +// unconfigured deployment is internally consistent. +const ( + DefaultControlNamespace = "felis" + DefaultMinecraftNamespace = "minecraft" + DefaultBuildNamespace = "felis-build" + + defaultRegistryPort int32 = 5000 + + // defaultRegistryImage is the upstream CNCF Distribution registry. It is an + // official, stable image and the only registry implementation the build/restore + // subsystems are exercised against (registry..svc:5000). + defaultRegistryImage = "registry:2" +) + +// Params parameterises the install bundle. Namespaces and the registry location +// have safe defaults; VelocityCIDRs has none — see the field comment. +type Params struct { + // ControlNamespace is where felis-api/operator/reaper run. Their SAs live here + // and the RoleBindings' subjects reference them here, even though the Roles + // they bind to live in the minecraft (and build) namespaces. + ControlNamespace string + // MinecraftNamespace is where MinecraftServer workloads, their RCON Secrets, + // and their world PVCs live. All three identities' minecraft-scoped Roles, and + // every server NetworkPolicy, are installed here. + MinecraftNamespace string + // BuildNamespace is where image-build Jobs run under the weak felis-build SA, + // with the egress-locked NetworkPolicy. + BuildNamespace string + // VelocityCIDRs are the off-cluster Velocity proxy source addresses permitted + // to reach server game ports (25565). Velocity runs on a separate macvlan host + // (spec §20), NOT a Kubernetes node, so this is an ipBlock allowlist and can + // never be a podSelector. It has NO default: an empty list renders a + // fail-closed game policy that admits no one (never an accidental allow-all), + // and the `felis manifests` generator refuses to emit a bundle without it. + VelocityCIDRs []string + // RegistryNamespace / RegistryPort locate the in-cluster image registry the + // build egress policy may reach (spec §16). RegistryNamespace defaults to the + // control namespace (registry co-located with the control plane). + RegistryNamespace string + RegistryPort int32 + // PackageSourceCIDRs is the explicit package-mirror egress allowlist for build + // Pods (spec §16). Empty means no internet egress at all — the locked-down + // default the build subsystem already enforces. + PackageSourceCIDRs []string + // FelisImage is the container image the felis-api and felis-operator + // Deployments run (the multi-call `felis` binary). It has NO default and no + // safe guess: `felis manifests` REQUIRES --felis-image and refuses to render + // without it, the same fail-loud contract as --velocity-cidr. The api pod also + // passes this value through as FELIS_IMAGE so the restore executor launches its + // `felis restore` Job using the very same image. + FelisImage string + // RegistryImage is the in-cluster registry image. Defaults to registry:2. + RegistryImage string + // BackupPVC is the name of the backup PersistentVolumeClaim the felis-api pod + // advertises to its restore executor via FELIS_BACKUP_PVC. It is OPTIONAL: with + // no backup PVC the restore endpoint degrades to 503 (cmd/felis/api.go), so the + // env var is rendered only when this is set. It must name the same PVC that the + // felis.toml archive.local_path is the mount path for, but that agreement lives + // in the out-of-band config Secret and cannot be enforced by the manifest. The + // reaper CronJob (when rendered) mounts this same PVC read-write to write + // archives into it — see WorldsHostPath / ArchiveLocalPath. + BackupPVC string + // WorldsHostPath is the node directory under which each server's world PVC is + // visible as / — the on-disk root the reaper CronJob mounts + // (read-only) at /worlds to archive idle worlds before reclaiming them (spec §18, + // §19 tarLocal-on-local-path starter). It has NO default and is the master switch + // for retention: empty ⇒ the reaper CronJob is NOT rendered (fail-safe — no + // CronJob is far safer than one that deletes PVCs while reading worlds from the + // wrong place). A hostPath ties the reaper to a single node, which is exactly the + // §19 starter topology (the operator provisions per-server ReadWriteOnce world + // PVCs, so a shared RWX worlds mount would contradict it); multi-node retention is + // a later storage evolution. Setting it REQUIRES BackupPVC and ArchiveLocalPath + // too — `felis manifests` enforces the trio (fail-loud). + // + // SHAPE-ASSERTED, runtime-unverified, and ARRANGEMENT-DEPENDENT: the reaper's + // resolver looks for /. Stock local-path-provisioner lays volumes out + // under PV-name paths (…/pvc-__/), NOT /, so this mount + // only finds worlds if the operator/storage is deliberately arranged to expose + // them as /. The rendered CronJob is the correct K8s object; whether + // the tar finds a world on a given cluster is not provable without one. + WorldsHostPath string + // ArchiveLocalPath is the path the backup PVC is mounted at inside the reaper + // CronJob's pod, and MUST equal felis.toml's [archive] local_path. tarLocal writes + // archive refs as absolute paths under [archive] local_path (internal/backup), and + // the restore Job mounts the backup PVC at that same path so the stored ref + // resolves (cmd/felis/api.go restoreConfig). `felis manifests` cannot read the + // out-of-band config Secret, so this path is supplied explicitly and documented as + // must-match. It is reaper-only and has no default; empty (with WorldsHostPath set) + // is rejected fail-loud by the generator. + ArchiveLocalPath string +} + +// withDefaults returns a copy of p with zero namespace/registry fields filled. +// VelocityCIDRs and PackageSourceCIDRs are intentionally left as-is: their empty +// states are meaningful (fail-closed game policy, no-internet build policy). +func (p Params) withDefaults() Params { + if p.ControlNamespace == "" { + p.ControlNamespace = DefaultControlNamespace + } + if p.MinecraftNamespace == "" { + p.MinecraftNamespace = DefaultMinecraftNamespace + } + if p.BuildNamespace == "" { + p.BuildNamespace = DefaultBuildNamespace + } + if p.RegistryNamespace == "" { + p.RegistryNamespace = p.ControlNamespace + } + if p.RegistryPort == 0 { + p.RegistryPort = defaultRegistryPort + } + if p.RegistryImage == "" { + p.RegistryImage = defaultRegistryImage + } + return p +} + +// controlPlanePodLabels is the recommended-label set stamped on a control-plane +// workload of the given component. The NetworkPolicy RCON peer selects on the +// PartOf + Component subset, so the labels here and the selector there are one +// source of truth. +func controlPlanePodLabels(component string) map[string]string { + return map[string]string{ + LabelName: appName, + LabelComponent: component, + LabelPartOf: controlPlanePartOf, + } +} diff --git a/internal/platform/netpol.go b/internal/platform/netpol.go new file mode 100644 index 0000000..7c65ff1 --- /dev/null +++ b/internal/platform/netpol.go @@ -0,0 +1,140 @@ +package platform + +import ( + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/operator" + corev1 "k8s.io/api/core/v1" + networkingv1 "k8s.io/api/networking/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/intstr" +) + +// gamePort is the Minecraft TCP port, sourced from the operator so the policy and +// the StatefulSet container port are one source of truth. +const gamePort = operator.GamePort + +// rconPort is the RCON port the allow-rcon policy opens, sourced from the operator +// so the policy and the server container's default RCON port are one source of +// truth. +// +// LIMITATION (honestly labeled, not verifiable without a cluster): RCON is +// per-server overridable via spec.rcon.port (internal/operator.rconPort), but this +// is one namespace-wide policy that can open only a single port. It opens the +// default. A server that overrides spec.rcon.port to a non-default value would have +// its RCON port denied by this fence, so the operator's readiness prober could not +// reach it. The supported deployment keeps the default RCON port; a per-server-port +// deployment would need per-server NetworkPolicies, deferred until a concrete need +// exists. +const rconPort = operator.DefaultRconPort + +// serverPodSelector matches every operator-managed Minecraft server pod by the +// exact labels the operator stamps (internal/operator.labelsFor). Sourcing the +// values from the operator package means the fence can never silently stop +// matching the pods it protects. +func serverPodSelector() metav1.LabelSelector { + return metav1.LabelSelector{MatchLabels: map[string]string{ + v1alpha1.LabelManagedBy: operator.ManagedByValue, + v1alpha1.LabelComponent: operator.ComponentValue, + }} +} + +// MinecraftNetworkPolicies renders the ingress fence for the minecraft namespace +// (spec §20, §21): a default-deny baseline, RCON (25575) reachable only from the +// {api, operator} control-plane pods, and the game port (25565) reachable only +// from the off-cluster Velocity proxy host(s). NetworkPolicies are additive, so +// the union admits exactly those two paths to server pods and denies all else. +func MinecraftNetworkPolicies(p Params) []*networkingv1.NetworkPolicy { + p = p.withDefaults() + return []*networkingv1.NetworkPolicy{ + defaultDenyIngress(p), + allowRConFromControlPlane(p), + allowGameFromVelocity(p), + } +} + +// defaultDenyIngress selects every pod in the namespace and permits no ingress — +// the baseline that makes the two allow policies a strict allowlist. +func defaultDenyIngress(p Params) *networkingv1.NetworkPolicy { + return netpol("felis-default-deny-ingress", p.MinecraftNamespace, + metav1.LabelSelector{}, // empty selector = all pods in the namespace + nil, // nil ingress rules = deny all ingress + ) +} + +// allowRConFromControlPlane opens 25575 on server pods to the felis-api and +// felis-operator pods only. Both dial RCON: felis-api for console writes +// (internal/api.console) and felis-operator for the readiness prober +// (internal/operator.prober). felis-reaper never opens RCON, so it is excluded. +// +// The single peer combines a namespaceSelector AND a podSelector, which K8s reads +// as an intersection: pods matching the podSelector that also live in the control +// namespace. Splitting them into two peers would be a union (allow ALL pods in +// the control ns OR api/operator pods anywhere) — the wrong, wider semantics. +func allowRConFromControlPlane(p Params) *networkingv1.NetworkPolicy { + tcp := corev1.ProtocolTCP + port := intstr.FromInt32(rconPort) + np := netpol("felis-allow-rcon-from-control-plane", p.MinecraftNamespace, + serverPodSelector(), + []networkingv1.NetworkPolicyIngressRule{{ + From: []networkingv1.NetworkPolicyPeer{{ + NamespaceSelector: &metav1.LabelSelector{ + MatchLabels: map[string]string{"kubernetes.io/metadata.name": p.ControlNamespace}, + }, + PodSelector: &metav1.LabelSelector{ + MatchLabels: map[string]string{LabelPartOf: controlPlanePartOf}, + MatchExpressions: []metav1.LabelSelectorRequirement{{ + Key: LabelComponent, + Operator: metav1.LabelSelectorOpIn, + Values: []string{ComponentAPI, ComponentOperator}, + }}, + }, + }}, + Ports: []networkingv1.NetworkPolicyPort{{Protocol: &tcp, Port: &port}}, + }}, + ) + return np +} + +// allowGameFromVelocity opens 25565 on server pods to the off-cluster Velocity +// proxy host(s) by ipBlock. Velocity is NOT a K8s pod (spec §20: it runs on a +// separate macvlan host), so the peer can only be an ipBlock — never a +// podSelector. +// +// Fail-closed: with no VelocityCIDRs the policy carries NO ingress rule (deny all +// game ingress), never an empty-From rule, which K8s would read as allow-all. The +// `felis manifests` generator additionally refuses an empty velocity list, so the +// rendered bundle is always either correctly scoped or absent — never accidentally +// open. +func allowGameFromVelocity(p Params) *networkingv1.NetworkPolicy { + tcp := corev1.ProtocolTCP + port := intstr.FromInt32(gamePort) + + var ingress []networkingv1.NetworkPolicyIngressRule + if len(p.VelocityCIDRs) > 0 { + peers := make([]networkingv1.NetworkPolicyPeer, 0, len(p.VelocityCIDRs)) + for _, cidr := range p.VelocityCIDRs { + peers = append(peers, networkingv1.NetworkPolicyPeer{ + IPBlock: &networkingv1.IPBlock{CIDR: cidr}, + }) + } + ingress = []networkingv1.NetworkPolicyIngressRule{{ + From: peers, + Ports: []networkingv1.NetworkPolicyPort{{Protocol: &tcp, Port: &port}}, + }} + } + return netpol("felis-allow-game-from-velocity", p.MinecraftNamespace, serverPodSelector(), ingress) +} + +// netpol assembles an ingress-only NetworkPolicy. A nil/empty ingress slice with +// PolicyTypeIngress is the canonical "deny all ingress" shape. +func netpol(name, ns string, sel metav1.LabelSelector, ingress []networkingv1.NetworkPolicyIngressRule) *networkingv1.NetworkPolicy { + return &networkingv1.NetworkPolicy{ + TypeMeta: metav1.TypeMeta{APIVersion: "networking.k8s.io/v1", Kind: "NetworkPolicy"}, + ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns}, + Spec: networkingv1.NetworkPolicySpec{ + PodSelector: sel, + PolicyTypes: []networkingv1.PolicyType{networkingv1.PolicyTypeIngress}, + Ingress: ingress, + }, + } +} diff --git a/internal/platform/netpol_test.go b/internal/platform/netpol_test.go new file mode 100644 index 0000000..b17ffd1 --- /dev/null +++ b/internal/platform/netpol_test.go @@ -0,0 +1,192 @@ +package platform + +import ( + "testing" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/operator" + networkingv1 "k8s.io/api/networking/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +func npByName(t *testing.T, nps []*networkingv1.NetworkPolicy, name string) *networkingv1.NetworkPolicy { + t.Helper() + for _, np := range nps { + if np.Name == name { + return np + } + } + t.Fatalf("network policy %q not found", name) + return nil +} + +// TestServerSelector_MatchesOperatorLabels proves the fence selects exactly the +// pods the operator labels. If the operator ever renamed its label values the +// policy would silently stop protecting the pods — this couples the two. +func TestServerSelector_MatchesOperatorLabels(t *testing.T) { + sel := serverPodSelector() + want := map[string]string{ + v1alpha1.LabelManagedBy: operator.ManagedByValue, + v1alpha1.LabelComponent: operator.ComponentValue, + } + if len(sel.MatchLabels) != len(want) { + t.Fatalf("selector has %d labels, want %d", len(sel.MatchLabels), len(want)) + } + for k, v := range want { + if sel.MatchLabels[k] != v { + t.Errorf("selector[%q] = %q, want %q", k, sel.MatchLabels[k], v) + } + } + // Guard against the values being empty (a typo making the selector match-all-ish). + if operator.ManagedByValue == "" || operator.ComponentValue == "" { + t.Fatal("operator label values must be non-empty") + } +} + +// TestDefaultDeny enforces the baseline: select all pods, declare Ingress policy, +// admit nothing. +func TestDefaultDeny(t *testing.T) { + nps := MinecraftNetworkPolicies(testParams()) + dd := npByName(t, nps, "felis-default-deny-ingress") + + if len(dd.Spec.PodSelector.MatchLabels) != 0 || len(dd.Spec.PodSelector.MatchExpressions) != 0 { + t.Error("default-deny must select ALL pods (empty podSelector)") + } + if !hasPolicyType(dd, networkingv1.PolicyTypeIngress) { + t.Error("default-deny must declare the Ingress policy type") + } + if len(dd.Spec.Ingress) != 0 { + t.Error("default-deny must have NO ingress rules (deny all)") + } +} + +// TestAllowRcon enforces the AND-semantics control-plane peer on port 25575. +func TestAllowRcon(t *testing.T) { + nps := MinecraftNetworkPolicies(testParams()) + rcon := npByName(t, nps, "felis-allow-rcon-from-control-plane") + + // Selects server pods, not all pods. + if rcon.Spec.PodSelector.MatchLabels[v1alpha1.LabelComponent] != operator.ComponentValue { + t.Error("rcon policy must select server pods") + } + if len(rcon.Spec.Ingress) != 1 { + t.Fatalf("rcon policy must have exactly 1 ingress rule, got %d", len(rcon.Spec.Ingress)) + } + rule := rcon.Spec.Ingress[0] + + // Exactly one peer, combining BOTH selectors (intersection = AND), never two + // peers (which would be a union/OR — far wider). + if len(rule.From) != 1 { + t.Fatalf("rcon peer count = %d, want 1 (AND-combined); 2 would be OR semantics", len(rule.From)) + } + peer := rule.From[0] + if peer.NamespaceSelector == nil || peer.PodSelector == nil { + t.Fatal("rcon peer must set BOTH namespaceSelector AND podSelector") + } + if got := peer.NamespaceSelector.MatchLabels["kubernetes.io/metadata.name"]; got != "felis" { + t.Errorf("rcon namespaceSelector = %q, want control ns felis", got) + } + if peer.PodSelector.MatchLabels[LabelPartOf] != controlPlanePartOf { + t.Error("rcon podSelector must require part-of=felis-control-plane") + } + + // Component must be In {api, operator} — and NOT include reaper. + var compReq *metav1.LabelSelectorRequirement + for i := range peer.PodSelector.MatchExpressions { + if peer.PodSelector.MatchExpressions[i].Key == LabelComponent { + compReq = &peer.PodSelector.MatchExpressions[i] + } + } + if compReq == nil { + t.Fatal("rcon podSelector must constrain the component label") + } + if compReq.Operator != metav1.LabelSelectorOpIn { + t.Errorf("component requirement operator = %q, want In", compReq.Operator) + } + if !contains(compReq.Values, ComponentAPI) || !contains(compReq.Values, ComponentOperator) { + t.Errorf("component values = %v, want both api and operator", compReq.Values) + } + if contains(compReq.Values, ComponentReaper) { + t.Error("reaper must NOT be allowed to reach RCON") + } + + // Port = the operator's default RCON port (single source of truth, so the + // prober can reach a default-port server through the fence). Pinned to the + // concrete 25575 too, guarding an accidental change to the operator default. + if operator.DefaultRconPort != 25575 { + t.Errorf("operator.DefaultRconPort = %d, want 25575 (conventional RCON port)", operator.DefaultRconPort) + } + assertSinglePort(t, rule.Ports, int(operator.DefaultRconPort)) +} + +// TestAllowGame_WithCIDRs renders the ipBlock allow path on port 25565. +func TestAllowGame_WithCIDRs(t *testing.T) { + p := testParams() + p.VelocityCIDRs = []string{"10.0.0.5/32", "10.0.0.6/32"} + game := npByName(t, MinecraftNetworkPolicies(p), "felis-allow-game-from-velocity") + + if len(game.Spec.Ingress) != 1 { + t.Fatalf("game policy must have 1 ingress rule, got %d", len(game.Spec.Ingress)) + } + rule := game.Spec.Ingress[0] + if len(rule.From) != 2 { + t.Fatalf("game peers = %d, want 2 ipBlocks", len(rule.From)) + } + for i, peer := range rule.From { + if peer.IPBlock == nil { + t.Errorf("game peer %d must be an ipBlock (Velocity is off-cluster, not a pod)", i) + } + if peer.PodSelector != nil || peer.NamespaceSelector != nil { + t.Errorf("game peer %d must NOT use pod/namespace selectors", i) + } + } + if rule.From[0].IPBlock.CIDR != "10.0.0.5/32" { + t.Errorf("game cidr[0] = %q", rule.From[0].IPBlock.CIDR) + } + if game.Spec.PodSelector.MatchLabels[v1alpha1.LabelComponent] != operator.ComponentValue { + t.Error("game policy must select server pods") + } + assertSinglePort(t, rule.Ports, int(operator.GamePort)) +} + +// TestAllowGame_FailsClosed is the footgun guard: no VelocityCIDRs ⇒ a policy that +// admits NOBODY (no ingress rule), never an empty-From rule (which K8s reads as +// allow-all). +func TestAllowGame_FailsClosed(t *testing.T) { + p := testParams() + p.VelocityCIDRs = nil + game := npByName(t, MinecraftNetworkPolicies(p), "felis-allow-game-from-velocity") + + if len(game.Spec.Ingress) != 0 { + t.Fatalf("game policy with no CIDRs must have ZERO ingress rules (fail-closed), got %d", len(game.Spec.Ingress)) + } + // Still a valid, selecting policy (so it actively denies, paired with default-deny). + if game.Spec.PodSelector.MatchLabels[v1alpha1.LabelComponent] != operator.ComponentValue { + t.Error("game policy must still select server pods even when fail-closed") + } + if !hasPolicyType(game, networkingv1.PolicyTypeIngress) { + t.Error("game policy must declare Ingress policy type") + } +} + +func hasPolicyType(np *networkingv1.NetworkPolicy, pt networkingv1.PolicyType) bool { + for _, t := range np.Spec.PolicyTypes { + if t == pt { + return true + } + } + return false +} + +func assertSinglePort(t *testing.T, ports []networkingv1.NetworkPolicyPort, want int) { + t.Helper() + if len(ports) != 1 { + t.Fatalf("expected exactly 1 port, got %d", len(ports)) + } + if ports[0].Port == nil || ports[0].Port.IntValue() != want { + t.Errorf("port = %v, want %d", ports[0].Port, want) + } + if ports[0].Protocol == nil || *ports[0].Protocol != "TCP" { + t.Errorf("protocol = %v, want TCP", ports[0].Protocol) + } +} diff --git a/internal/platform/rbac.go b/internal/platform/rbac.go new file mode 100644 index 0000000..b3cbc38 --- /dev/null +++ b/internal/platform/rbac.go @@ -0,0 +1,204 @@ +package platform + +import ( + "felis.lolicon.best/internal/apis/felis/v1alpha1" + corev1 "k8s.io/api/core/v1" + rbacv1 "k8s.io/api/rbac/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +// API groups used by the rules. The felis group is sourced from v1alpha1 so the +// CRD's identity and its RBAC can never drift apart. +const ( + groupCore = "" // core/v1: secrets, services, persistentvolumeclaims + groupApps = "apps" + groupBatch = "batch" +) + +var groupFelis = v1alpha1.GroupName // "felis.lolicon.best" + +// RBAC is the control-plane authorization bundle: one SA per identity and the +// namespaced Roles + RoleBindings that grant each exactly the verbs its code path +// exercises. There is deliberately no ClusterRole or ClusterRoleBinding anywhere. +type RBAC struct { + ServiceAccounts []*corev1.ServiceAccount + Roles []*rbacv1.Role + RoleBindings []*rbacv1.RoleBinding +} + +// ControlPlaneRBAC assembles the full RBAC bundle for p. +// +// The reaper identity (felis-reaper SA + Role + RoleBinding) is rendered ONLY when +// the retention reaper CronJob is — both gate on reaperEnabled(p), the same storage +// trio (workloads.go). This coupling is deliberate least-privilege: the reaper's +// Role is the one and only place persistentvolumeclaims:delete appears in the whole +// bundle (world reclamation) — neither felis-api nor felis-operator can delete a +// PVC. Leaving that destructive grant standing in a deployment that never runs the +// reaper would widen the blast radius of a control-plane compromise for no benefit +// (a control-namespace foothold could mount felis-reaper and destroy world PVCs), +// since nothing would consume it. So the destructive identity exists exactly as +// long as its consumer does, and the manifests command's fail-loud trio check +// guarantees the CronJob and this RBAC are always rendered together or not at all. +func ControlPlaneRBAC(p Params) RBAC { + p = p.withDefaults() + rbac := RBAC{ + ServiceAccounts: []*corev1.ServiceAccount{ + controlPlaneServiceAccount(p.ControlNamespace, SAAPI, ComponentAPI), + controlPlaneServiceAccount(p.ControlNamespace, SAOperator, ComponentOperator), + }, + Roles: []*rbacv1.Role{ + APIMinecraftRole(p), + APIBuildRole(p), + OperatorRole(p), + }, + // Each binding lives in the Role's namespace and names the subject SA in the + // control namespace (a RoleBinding may reference an SA from another namespace; + // its roleRef must be a Role in the binding's own namespace). + RoleBindings: []*rbacv1.RoleBinding{ + bindRole(p.MinecraftNamespace, "felis-api", p.ControlNamespace, SAAPI, ComponentAPI), + bindRole(p.BuildNamespace, "felis-api-builds", p.ControlNamespace, SAAPI, ComponentAPI), + bindRole(p.MinecraftNamespace, "felis-operator", p.ControlNamespace, SAOperator, ComponentOperator), + }, + } + // The destructive fourth power is conditional on its consumer (see the doc above). + if reaperEnabled(p) { + rbac.ServiceAccounts = append(rbac.ServiceAccounts, + controlPlaneServiceAccount(p.ControlNamespace, SAReaper, ComponentReaper)) + rbac.Roles = append(rbac.Roles, ReaperRole(p)) + rbac.RoleBindings = append(rbac.RoleBindings, + bindRole(p.MinecraftNamespace, "felis-reaper", p.ControlNamespace, SAReaper, ComponentReaper)) + } + return rbac +} + +// APIMinecraftRole grants felis-api exactly what it does in the minecraft +// namespace: drive MinecraftServer specs (internal/api.k8scluster — get/list/ +// create/patch, never status), read RCON passwords for console writes +// (internal/api.console — secrets:get), create the restore Job +// (internal/restore — jobs:create), and stream the live console for the read +// side (internal/api.logstream — pods:list to find the server's running pod, +// then pods/log:get to follow it; spec §8 读=pods/log follow). felis-api uses a +// DIRECT client, so it needs no list/watch beyond the explicit List calls. +// +// The read-side grant is deliberately minimal: pods:list + pods/log:get, NOT +// pods:get — the streamer lists pods by the server label then reads the chosen +// pod's log subresource, never Gets a pod object. Keeping pods:get out is the +// least-privilege line the rbac test asserts (a pod's full object can carry more +// than its logs). +func APIMinecraftRole(p Params) *rbacv1.Role { + p = p.withDefaults() + return role(p.MinecraftNamespace, "felis-api", ComponentAPI, []rbacv1.PolicyRule{ + rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "list", "create", "patch"}), + rule([]string{groupCore}, []string{"secrets"}, []string{"get"}), + rule([]string{groupBatch}, []string{"jobs"}, []string{"create"}), + // Read-side console (spec §8 读=pods/log follow): list pods to find the + // server's running pod, then read its log subresource. Two separate rules so + // the verbs stay tight — list on pods, get on pods/log, and nothing else. + rule([]string{groupCore}, []string{"pods"}, []string{"list"}), + rule([]string{groupCore}, []string{"pods/log"}, []string{"get"}), + }) +} + +// APIBuildRole grants felis-api the build-Job lifecycle in the build namespace +// (internal/build.k8sjobs — Create/Get/Delete) plus the read-side build-log +// stream (spec §16, §416 日志流复用 §8): list build Pods to find the build Job's +// Pod by build-id label, then read its log subresource. This is a SEPARATE +// namespace from the api's minecraft powers, so it is a separate Role + +// RoleBinding; the api SA reaches across both from the control namespace. The log +// grant mirrors felis-api's minecraft-ns console read (pods:list + pods/log:get, +// no pods:get) — read-only and least-privilege; it does NOT touch the build SA +// token or any secret. +func APIBuildRole(p Params) *rbacv1.Role { + p = p.withDefaults() + return role(p.BuildNamespace, "felis-api-builds", ComponentAPI, []rbacv1.PolicyRule{ + rule([]string{groupBatch}, []string{"jobs"}, []string{"create", "get", "delete"}), + // Read-side build logs (spec §16): list build Pods to find the build Job's + // Pod, then read its log subresource — and nothing wider. No pods:get (the + // streamer lists then reads pods/log, never Gets a Pod object, whose full + // spec carries more than its logs). + rule([]string{groupCore}, []string{"pods"}, []string{"list"}), + rule([]string{groupCore}, []string{"pods/log"}, []string{"get"}), + }) +} + +// OperatorRole grants felis-operator what the reconciler exercises through the +// manager's CACHED client (internal/operator.reconciler). Because reads go +// through informers, every watched type needs list+watch even for a single Get; +// the manager's cache is namespace-scoped (see cmd/felis/operator.go), so a +// namespaced Role is sufficient. The operator owns StatefulSets and Services +// (Get/Create/Update — never patch or delete), writes only minecraftservers +// status (Status().Update — `update` only), and reads RCON Secrets. It never +// touches pods, PVCs, Events, or finalizers, so none appear here. +func OperatorRole(p Params) *rbacv1.Role { + p = p.withDefaults() + return role(p.MinecraftNamespace, "felis-operator", ComponentOperator, []rbacv1.PolicyRule{ + rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "list", "watch"}), + rule([]string{groupFelis}, []string{"minecraftservers/status"}, []string{"update"}), + rule([]string{groupApps}, []string{"statefulsets"}, []string{"get", "list", "watch", "create", "update"}), + rule([]string{groupCore}, []string{"services"}, []string{"get", "list", "watch", "create", "update"}), + rule([]string{groupCore}, []string{"secrets"}, []string{"get", "list", "watch"}), + }) +} + +// ReaperRole grants felis-reaper its two destructive, disjoint powers +// (internal/reaper.k8scluster): patch a MinecraftServer to Stop it and delete its +// world PVC. Candidate servers come from the Postgres store, not a cluster List, +// so no list/watch is needed; the reaper uses a direct client. It can read+patch +// minecraftservers but cannot create them, and holds no power over StatefulSets, +// Services, or Secrets — those belong to the operator and api. +// +// Note no identity anywhere holds minecraftservers:delete. That is intentional, not +// a missing grant: reaping releases a server by flipping desiredState=Stopped and +// reclaiming the world PVC (k8scluster.go does "nothing else"), leaving the CR in +// place so a former owner can re-claim it within the retention window (spec §466). +// The MinecraftServer CR is the lifecycle source of truth and is retained, never +// hard-deleted, so the delete verb is deliberately absent from every Role. +func ReaperRole(p Params) *rbacv1.Role { + p = p.withDefaults() + return role(p.MinecraftNamespace, "felis-reaper", ComponentReaper, []rbacv1.PolicyRule{ + rule([]string{groupFelis}, []string{"minecraftservers"}, []string{"get", "patch"}), + rule([]string{groupCore}, []string{"persistentvolumeclaims"}, []string{"delete"}), + }) +} + +// controlPlaneServiceAccount renders a control-plane SA. Unlike the weak +// build/restore SAs, these identities legitimately call the K8s API, so the token +// mounts (via their Deployment) — AutomountServiceAccountToken is left nil +// (cluster default = mount) rather than false. +func controlPlaneServiceAccount(ns, name, component string) *corev1.ServiceAccount { + return &corev1.ServiceAccount{ + TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "ServiceAccount"}, + ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: controlPlanePodLabels(component)}, + } +} + +func role(ns, name, component string, rules []rbacv1.PolicyRule) *rbacv1.Role { + return &rbacv1.Role{ + TypeMeta: metav1.TypeMeta{APIVersion: "rbac.authorization.k8s.io/v1", Kind: "Role"}, + ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: controlPlanePodLabels(component)}, + Rules: rules, + } +} + +// bindRole binds the Role named roleName (in roleNS) to the ServiceAccount saName +// in saNS. The RoleBinding lives in roleNS; the subject SA may live elsewhere. +func bindRole(roleNS, roleName, saNS, saName, component string) *rbacv1.RoleBinding { + return &rbacv1.RoleBinding{ + TypeMeta: metav1.TypeMeta{APIVersion: "rbac.authorization.k8s.io/v1", Kind: "RoleBinding"}, + ObjectMeta: metav1.ObjectMeta{Name: roleName, Namespace: roleNS, Labels: controlPlanePodLabels(component)}, + Subjects: []rbacv1.Subject{{ + Kind: rbacv1.ServiceAccountKind, + Name: saName, + Namespace: saNS, + }}, + RoleRef: rbacv1.RoleRef{ + APIGroup: rbacv1.GroupName, + Kind: "Role", + Name: roleName, + }, + } +} + +func rule(apiGroups, resources, verbs []string) rbacv1.PolicyRule { + return rbacv1.PolicyRule{APIGroups: apiGroups, Resources: resources, Verbs: verbs} +} diff --git a/internal/platform/rbac_test.go b/internal/platform/rbac_test.go new file mode 100644 index 0000000..fb75275 --- /dev/null +++ b/internal/platform/rbac_test.go @@ -0,0 +1,354 @@ +package platform + +import ( + "strings" + "testing" + + rbacv1 "k8s.io/api/rbac/v1" +) + +// testParams is a fully-specified Params used across the platform tests. It sets +// VelocityCIDRs so the game policy renders its allow path; tests that need the +// fail-closed path clear it explicitly. +func testParams() Params { + return Params{ + ControlNamespace: "felis", + MinecraftNamespace: "minecraft", + BuildNamespace: "felis-build", + FelisImage: "registry.felis.svc:5000/felis:test", + VelocityCIDRs: []string{"10.0.0.5/32"}, + } +} + +func roleByName(t *testing.T, roles []*rbacv1.Role, name string) *rbacv1.Role { + t.Helper() + for _, r := range roles { + if r.Name == name { + return r + } + } + t.Fatalf("role %q not found in bundle", name) + return nil +} + +// hasRule reports whether role grants verb on (apiGroup, resource). +func hasRule(role *rbacv1.Role, apiGroup, resource, verb string) bool { + for _, r := range role.Rules { + if !contains(r.APIGroups, apiGroup) || !contains(r.Resources, resource) { + continue + } + if contains(r.Verbs, verb) { + return true + } + } + return false +} + +// grantsResource reports whether role has ANY rule touching (apiGroup, resource). +func grantsResource(role *rbacv1.Role, apiGroup, resource string) bool { + for _, r := range role.Rules { + if contains(r.APIGroups, apiGroup) && contains(r.Resources, resource) { + return true + } + } + return false +} + +func contains(haystack []string, needle string) bool { + for _, s := range haystack { + if s == needle { + return true + } + } + return false +} + +// TestAPIRole_CreatesJobsInBothNamespaces is the #1 regression guard: felis-api +// must be able to create the build Job (build ns) AND the restore Job (minecraft +// ns). An earlier reading of §21 omitted batch/jobs entirely; this asserts both. +func TestAPIRole_CreatesJobsInBothNamespaces(t *testing.T) { + rbac := ControlPlaneRBAC(testParams()) + + mc := roleByName(t, rbac.Roles, "felis-api") + if mc.Namespace != "minecraft" { + t.Errorf("felis-api minecraft Role namespace = %q, want minecraft", mc.Namespace) + } + if !hasRule(mc, "batch", "jobs", "create") { + t.Error("felis-api (minecraft) must have batch/jobs:create for the restore Job") + } + + build := roleByName(t, rbac.Roles, "felis-api-builds") + if build.Namespace != "felis-build" { + t.Errorf("felis-api-builds Role namespace = %q, want felis-build", build.Namespace) + } + for _, v := range []string{"create", "get", "delete"} { + if !hasRule(build, "batch", "jobs", v) { + t.Errorf("felis-api-builds must have batch/jobs:%s for the build-Job lifecycle", v) + } + } + // Read-side build logs (spec §16, §416 日志流复用 §8): list build Pods to find the + // build Job's pod, then read its log subresource — and nothing wider. This + // mirrors the minecraft console-read least-privilege line: pods:list + pods/log, + // never pods:get (a pod's full object can carry more than its logs). + if !hasRule(build, groupCore, "pods", "list") { + t.Error("felis-api-builds must list pods (pods:list) to find the build Job's pod for the §16 build-log read") + } + if !hasRule(build, groupCore, "pods/log", "get") { + t.Error("felis-api-builds must read pod logs (pods/log:get) for the §16 build-log stream") + } + if hasRule(build, groupCore, "pods", "get") { + t.Error("felis-api-builds must NOT have pods:get (the build-log streamer lists then reads pods/log, never Gets a pod)") + } +} + +// TestAPIRole_MinecraftPowersExact pins felis-api's minecraft verbs and, crucially, +// what it must NOT have: no status writes, no delete. +func TestAPIRole_MinecraftPowersExact(t *testing.T) { + mc := roleByName(t, ControlPlaneRBAC(testParams()).Roles, "felis-api") + for _, v := range []string{"get", "list", "create", "patch"} { + if !hasRule(mc, groupFelis, "minecraftservers", v) { + t.Errorf("felis-api must have minecraftservers:%s", v) + } + } + if hasRule(mc, groupFelis, "minecraftservers", "delete") { + t.Error("felis-api must NOT delete minecraftservers (lifecycle is the reaper/operator's)") + } + if grantsResource(mc, groupFelis, "minecraftservers/status") { + t.Error("felis-api must NOT touch minecraftservers/status (status is the operator's alone)") + } + if !hasRule(mc, groupCore, "secrets", "get") { + t.Error("felis-api must read RCON secrets (secrets:get) for console writes") + } + // Read-side console (spec §8 读=pods/log follow): list pods to find the + // running pod, then read its log subresource — and nothing wider. + if !hasRule(mc, groupCore, "pods", "list") { + t.Error("felis-api must list pods (pods:list) to find a server's running pod for the console read") + } + if !hasRule(mc, groupCore, "pods/log", "get") { + t.Error("felis-api must read pod logs (pods/log:get) for the console read (spec §8 读=pods/log follow)") + } + // Least-privilege line: the streamer lists pods then reads pods/log, it never + // Gets a pod object — so pods:get must be ABSENT (a pod's full object can carry + // more than its logs). + if hasRule(mc, groupCore, "pods", "get") { + t.Error("felis-api must NOT have pods:get (the console streamer lists then reads pods/log, never Gets a pod)") + } +} + +// TestOperatorRole_ScopeExact pins the operator's cached-client verb set and its +// hard exclusions: no PVC, no pods, no events, no statefulset patch/delete, and +// status carrying only `update`. +func TestOperatorRole_ScopeExact(t *testing.T) { + op := roleByName(t, ControlPlaneRBAC(testParams()).Roles, "felis-operator") + + // Watched types need list+watch because reads go through informers. + for _, v := range []string{"get", "list", "watch"} { + if !hasRule(op, groupFelis, "minecraftservers", v) { + t.Errorf("operator must have minecraftservers:%s (cached client)", v) + } + } + for _, v := range []string{"get", "list", "watch", "create", "update"} { + if !hasRule(op, "apps", "statefulsets", v) { + t.Errorf("operator must have statefulsets:%s", v) + } + } + // Owns workloads by create/update only — never patch or delete. + for _, v := range []string{"patch", "delete"} { + if hasRule(op, "apps", "statefulsets", v) { + t.Errorf("operator must NOT have statefulsets:%s (create/update only)", v) + } + } + // Status is update-only. + if !hasRule(op, groupFelis, "minecraftservers/status", "update") { + t.Error("operator must have minecraftservers/status:update") + } + for _, v := range []string{"get", "patch"} { + if hasRule(op, groupFelis, "minecraftservers/status", v) { + t.Errorf("operator status rule must be update-only, found %s", v) + } + } + // Hard exclusions. + for _, res := range []string{"persistentvolumeclaims", "pods", "events"} { + if grantsResource(op, groupCore, res) { + t.Errorf("operator must NOT touch core/%s", res) + } + } +} + +// TestReaperRole_ScopeExact pins the reaper's two destructive powers and confirms +// it holds none of the operator's/api's resources. It uses reaperParams because the +// reaper identity is gated on the retention trio (see TestReaperRBAC_GatedOnRetention). +func TestReaperRole_ScopeExact(t *testing.T) { + rp := roleByName(t, ControlPlaneRBAC(reaperParams()).Roles, "felis-reaper") + + if !hasRule(rp, groupCore, "persistentvolumeclaims", "delete") { + t.Error("reaper must delete PVCs (world reclamation)") + } + if !hasRule(rp, groupFelis, "minecraftservers", "patch") || !hasRule(rp, groupFelis, "minecraftservers", "get") { + t.Error("reaper must get+patch minecraftservers (to Stop them)") + } + if hasRule(rp, groupFelis, "minecraftservers", "create") { + t.Error("reaper must NOT create minecraftservers") + } + for _, res := range []struct{ group, name string }{ + {"apps", "statefulsets"}, {groupCore, "services"}, {groupCore, "secrets"}, + } { + if grantsResource(rp, res.group, res.name) { + t.Errorf("reaper must NOT touch %s/%s (operator/api territory)", res.group, res.name) + } + } + // The reaper never lists from the cluster (candidates come from Postgres). + for _, v := range []string{"list", "watch"} { + if hasRule(rp, groupFelis, "minecraftservers", v) { + t.Errorf("reaper must NOT %s minecraftservers (candidates come from the store)", v) + } + } +} + +// TestReaperRBAC_GatedOnRetention locks the reaper identity to its consumer: the +// felis-reaper SA, Role, and RoleBinding (which carry the bundle's ONLY +// persistentvolumeclaims:delete grant) render iff the retention trio is supplied — +// the same gate as the CronJob. A standing, unconsumed pvc:delete grant would widen +// the blast radius of a control-plane compromise, so it must not exist without the +// reaper that needs it. +func TestReaperRBAC_GatedOnRetention(t *testing.T) { + hasSA := func(rbac RBAC, name string) bool { + for _, sa := range rbac.ServiceAccounts { + if sa.Name == name { + return true + } + } + return false + } + reaperRoleBindings := func(rbac RBAC) (roles, bindings int) { + for _, r := range rbac.Roles { + if r.Name == "felis-reaper" { + roles++ + } + } + for _, rb := range rbac.RoleBindings { + if rb.Name == "felis-reaper" { + bindings++ + } + } + return + } + + // No retention storage ⇒ no reaper identity anywhere in the bundle. + off := ControlPlaneRBAC(testParams()) + if hasSA(off, SAReaper) { + t.Error("felis-reaper SA must NOT render without the retention trio") + } + if roles, bindings := reaperRoleBindings(off); roles != 0 || bindings != 0 { + t.Errorf("reaper Role/RoleBinding present without retention (roles=%d bindings=%d)", roles, bindings) + } + // And the bundle then holds NO pvc:delete grant at all — the worst-case primitive + // is simply absent, not merely unbound. + for _, role := range off.Roles { + if hasRule(role, groupCore, "persistentvolumeclaims", "delete") { + t.Errorf("%s grants pvc:delete with no reaper deployed — the destructive grant must be gated", role.Name) + } + } + + // Full trio ⇒ the SA, its Role, and the binding all render together. + on := ControlPlaneRBAC(reaperParams()) + if !hasSA(on, SAReaper) { + t.Error("felis-reaper SA must render with the retention trio") + } + if roles, bindings := reaperRoleBindings(on); roles != 1 || bindings != 1 { + t.Errorf("want exactly 1 reaper Role + 1 binding with retention, got roles=%d bindings=%d", roles, bindings) + } +} + +// TestNoIdentityDeletesMinecraftServers locks the lifecycle invariant: the +// MinecraftServer CR is the retained source of truth (released by Stop + PVC +// reclaim, never hard-deleted, so a former owner can re-claim within the retention +// window — spec §466). No control-plane identity may hold minecraftservers:delete; +// if a future change adds it, this fails loudly so the decision is deliberate. +func TestNoIdentityDeletesMinecraftServers(t *testing.T) { + for _, role := range ControlPlaneRBAC(testParams()).Roles { + if hasRule(role, groupFelis, "minecraftservers", "delete") { + t.Errorf("%s grants minecraftservers:delete — the CR is retained, never hard-deleted", role.Name) + } + } +} + +// TestNoClusterScopedRBAC enforces the §22 red line: nothing in the bundle is a +// ClusterRole or ClusterRoleBinding, and no Role uses wildcards, escalation verbs, +// or grants power over RBAC resources themselves. +func TestNoClusterScopedRBAC(t *testing.T) { + rbac := ControlPlaneRBAC(testParams()) + + dangerousVerbs := map[string]bool{"escalate": true, "bind": true, "impersonate": true} + rbacResources := map[string]bool{ + "roles": true, "rolebindings": true, "clusterroles": true, "clusterrolebindings": true, + } + + for _, role := range rbac.Roles { + if role.Kind != "Role" { + t.Errorf("%s: Kind = %q, want Role (no ClusterRole)", role.Name, role.Kind) + } + if role.Namespace == "" { + t.Errorf("%s: Role must be namespaced", role.Name) + } + for _, r := range role.Rules { + for _, g := range r.APIGroups { + if g == "*" { + t.Errorf("%s: wildcard apiGroup", role.Name) + } + } + for _, res := range r.Resources { + if res == "*" { + t.Errorf("%s: wildcard resource", role.Name) + } + if rbacResources[strings.ToLower(res)] { + t.Errorf("%s: must not grant power over RBAC resource %q", role.Name, res) + } + } + for _, v := range r.Verbs { + if v == "*" { + t.Errorf("%s: wildcard verb", role.Name) + } + if dangerousVerbs[strings.ToLower(v)] { + t.Errorf("%s: dangerous verb %q", role.Name, v) + } + } + } + } +} + +// TestRoleBindings_RefLocalRoleAndControlPlaneSA enforces that every binding +// references a namespaced Role (never a ClusterRole) and a ServiceAccount subject +// living in the control namespace. +func TestRoleBindings_RefLocalRoleAndControlPlaneSA(t *testing.T) { + p := testParams() + rbac := ControlPlaneRBAC(p) + roleNames := map[string]bool{} + for _, r := range rbac.Roles { + roleNames[r.Namespace+"/"+r.Name] = true + } + + for _, rb := range rbac.RoleBindings { + if rb.RoleRef.Kind != "Role" { + t.Errorf("%s: RoleRef.Kind = %q, want Role", rb.Name, rb.RoleRef.Kind) + } + if rb.RoleRef.APIGroup != rbacv1.GroupName { + t.Errorf("%s: RoleRef.APIGroup = %q, want %q", rb.Name, rb.RoleRef.APIGroup, rbacv1.GroupName) + } + // The referenced Role must exist in the binding's own namespace. + if !roleNames[rb.Namespace+"/"+rb.RoleRef.Name] { + t.Errorf("%s: references Role %q absent from namespace %q", rb.Name, rb.RoleRef.Name, rb.Namespace) + } + if len(rb.Subjects) == 0 { + t.Fatalf("%s: no subjects", rb.Name) + } + for _, s := range rb.Subjects { + if s.Kind != rbacv1.ServiceAccountKind { + t.Errorf("%s: subject Kind = %q, want ServiceAccount", rb.Name, s.Kind) + } + if s.Namespace != p.ControlNamespace { + t.Errorf("%s: subject SA namespace = %q, want control ns %q", rb.Name, s.Namespace, p.ControlNamespace) + } + } + } +} diff --git a/internal/platform/workloads.go b/internal/platform/workloads.go new file mode 100644 index 0000000..92f24c1 --- /dev/null +++ b/internal/platform/workloads.go @@ -0,0 +1,527 @@ +package platform + +import ( + "fmt" + + appsv1 "k8s.io/api/apps/v1" + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/intstr" +) + +// This file renders the running control-plane workloads — the felis-api and +// felis-operator Deployments, the in-cluster image registry (Deployment + +// Service + PVC), and (when configured) the reaper CronJob. They are what make +// the RBAC and NetworkPolicy fence MEAN something: each Deployment runs as its +// matching control-plane SA (so the namespaced Roles actually bind to a workload) +// and stamps the recommended labels the RCON NetworkPolicy peer selects on (so +// the api console / operator prober can reach RCON through the fence). +// workloads_test.go evaluates those correspondences with the SAME selector +// machinery K8s uses, because no cluster runs here. +// +// The reaper CronJob (spec §18 three-clock retention) renders ONLY when the +// storage topology is supplied — WorldsHostPath + BackupPVC + ArchiveLocalPath, +// gated by reaperEnabled. It is opt-in-when-configured rather than always-on +// because archiving idle worlds means mounting where the worlds physically live, +// and the spec keeps that open (§18/§19: tarLocal-on-local-path is the starter, +// Longhorn/snapshot the documented evolution, and they do not share a mount +// model). The starter model — a node-local hostPath worlds-root mounted +// read-only — is the only one coherent with the operator's per-server +// ReadWriteOnce world PVCs (a shared RWX worlds mount would contradict them), so +// that is what renders; when the trio is absent no CronJob is emitted, which is +// the fail-safe choice for a workload that deletes PVCs. SHAPE-ASSERTED and +// runtime-unverified: the rendered CronJob is the correct K8s object, but whether +// the tar finds a world under / on a given cluster depends on +// how that node's storage is arranged (stock local-path-provisioner uses +// PV-name paths, not /) and is not provable without a cluster — see the +// WorldsHostPath field doc. No nodeSelector is set: the single-node starter pins +// the worlds to one node implicitly; a multi-node deployment MUST add one (or the +// CronJob could schedule on a node where the hostPath is empty) — a hazard left on +// record here until multi-node retention is built. +// +// Deliberately NOT rendered: +// - A Service for felis-api. Its external face (8080) is exposed out-of-band +// (Ingress/LoadBalancer is a deployment choice) and its internal face's only +// consumer is the Velocity plugin; nothing in-tree dials a felis-api Service +// name, so rendering one would be a speculative selector. The registry Service +// IS rendered because registry..svc:5000 is a pinned consumer hardcoded +// across the build subsystem and config. + +const ( + // configSecretName / serviceTokenSecretName are referenced BY NAME and NEVER + // rendered into the bundle: felis.toml carries the database URL (a credential) + // and the service token is a credential, so writing either into a checked-in + // manifest is a hard red line. The deployment provisions both Secrets + // out-of-band before applying these workloads. + configSecretName = "felis-config" + configSecretKey = "felis.toml" + configMountPath = "/etc/felis" + configFilePath = "/etc/felis/felis.toml" + serviceTokenSecretName = "felis-service-token" + serviceTokenSecretKey = "token" + + // Ports, single-sourced with the entrypoints (cmd/felis). The api external + // port must match server.listen in felis.toml (default 0.0.0.0:8080); that + // agreement lives in the out-of-band config Secret and cannot be enforced here. + apiExternalPort int32 = 8080 + apiInternalPort int32 = 8081 + operatorMetricsPort int32 = 8080 + + registryName = "registry" + registryDataPath = "/var/lib/registry" + registryStorageSize = "10Gi" + + configVolume = "config" + tmpVolume = "tmp" + registryVolume = "data" + worldsVolume = "worlds" + backupVolume = "backup" + + // worldsMountPath is where the reaper CronJob mounts the worlds-root (read-only). + // It is the default of `felis reaper --worlds-root`; the resolver then reads each + // world at /. Single-sourced with cmd/felis/reaper.go. + worldsMountPath = "/worlds" + + // reaperSchedule is the daily retention cadence (spec §18: a daily batch). 04:00 + // is an off-peak window; the reaper itself is idempotent and run-once, so the + // exact minute is not load-bearing. ConcurrencyPolicy=Forbid keeps a slow run + // from overlapping the next day's. + reaperSchedule = "0 4 * * *" + + // reaperStartingDeadlineSeconds bounds how late a missed run may still start (a + // controller outage at 04:00 shouldn't silently skip retention) without letting + // a long backlog pile up. reaperActiveDeadlineSeconds caps a single run so a + // wedged archive can't hold the Forbid lock forever. + reaperStartingDeadlineSeconds int64 = 300 + reaperActiveDeadlineSeconds int64 = 3600 + reaperBackoffLimit int32 = 2 + reaperHistoryLimit int32 = 3 + + // nonRootUID matches the tree's non-root identity convention + // (internal/restore.defaultRunAsID). + nonRootUID int64 = 1000 +) + +// Workloads renders the running control-plane: the felis-api Deployment, the +// felis-operator Deployment, and the in-cluster registry (Deployment + Service + +// PVC), plus the reaper CronJob when reaperEnabled(p). FelisImage is required — +// `felis manifests` enforces it (fail-loud), so a rendered bundle always names a +// concrete image. +func Workloads(p Params) []Object { + p = p.withDefaults() + objs := []Object{ + APIDeployment(p), + OperatorDeployment(p), + registryDeployment(p), + registryService(p), + registryPVC(p), + } + if reaperEnabled(p) { + objs = append(objs, reaperCronJob(p)) + } + return objs +} + +// reaperEnabled reports whether the retention CronJob should render. It needs all +// three storage coordinates: WorldsHostPath (where worlds live, mounted to read +// them), BackupPVC (where archives are written) and ArchiveLocalPath (the mount +// path that must equal [archive] local_path so absolute archive refs resolve). +// Any missing ⇒ no CronJob (fail-safe). `felis manifests` enforces the trio +// together so a partial configuration fails loudly rather than silently dropping +// retention here. +func reaperEnabled(p Params) bool { + return p.WorldsHostPath != "" && p.BackupPVC != "" && p.ArchiveLocalPath != "" +} + +// APIDeployment renders the felis-api Deployment (spec §7). It runs as the +// felis-api SA (so APIMinecraftRole/APIBuildRole bind to a real workload) and +// carries controlPlanePodLabels(api), which the allow-rcon NetworkPolicy peer +// selects — that correspondence lets the console reach server RCON through the +// fence and is asserted in workloads_test.go. +// +// felis.toml is mounted read-only from a Secret (it carries the database URL, a +// credential, so it must never be a ConfigMap); FELIS_SERVICE_TOKEN comes from a +// second Secret by reference. FELIS_IMAGE is the felis image itself, so the +// restore executor launches `felis restore` with the same image. FELIS_BACKUP_PVC +// is rendered only when a backup PVC is named — otherwise the restore endpoint +// degrades to 503 rather than enqueuing a Job that cannot mount its backup. +func APIDeployment(p Params) *appsv1.Deployment { + p = p.withDefaults() + + env := []corev1.EnvVar{ + { + Name: "FELIS_SERVICE_TOKEN", + ValueFrom: &corev1.EnvVarSource{ + SecretKeyRef: &corev1.SecretKeySelector{ + LocalObjectReference: corev1.LocalObjectReference{Name: serviceTokenSecretName}, + Key: serviceTokenSecretKey, + }, + }, + }, + {Name: "FELIS_IMAGE", Value: p.FelisImage}, + } + if p.BackupPVC != "" { + env = append(env, corev1.EnvVar{Name: "FELIS_BACKUP_PVC", Value: p.BackupPVC}) + } + + container := corev1.Container{ + Name: ComponentAPI, + Image: p.FelisImage, + Command: []string{"felis", "api"}, + Args: []string{ + "--config", configFilePath, + "--internal-addr", fmt.Sprintf(":%d", apiInternalPort), + }, + Env: env, + Ports: []corev1.ContainerPort{ + {Name: "external", ContainerPort: apiExternalPort, Protocol: corev1.ProtocolTCP}, + {Name: "internal", ContainerPort: apiInternalPort, Protocol: corev1.ProtocolTCP}, + }, + VolumeMounts: []corev1.VolumeMount{ + {Name: configVolume, MountPath: configMountPath, ReadOnly: true}, + {Name: tmpVolume, MountPath: "/tmp"}, + }, + Resources: controlPlaneResources(), + SecurityContext: hardenedContainerSecurityContext(), + } + + volumes := []corev1.Volume{ + { + Name: configVolume, + VolumeSource: corev1.VolumeSource{ + Secret: &corev1.SecretVolumeSource{SecretName: configSecretName}, + }, + }, + {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, + } + + return controlPlaneDeployment(p, SAAPI, container, volumes) +} + +// OperatorDeployment renders the felis-operator Deployment (spec §5). It runs as +// the felis-operator SA and carries controlPlanePodLabels(operator), the second +// pod the allow-rcon peer admits (the readiness prober dials RCON). It takes NO +// config Secret: the operator reads everything from flags + the in-cluster API, +// so it never holds the database URL — a deliberately smaller attack surface than +// the api. It watches the minecraft namespace (--namespace) while running in the +// control namespace, exactly the split cmd/felis/operator.go documents. +func OperatorDeployment(p Params) *appsv1.Deployment { + p = p.withDefaults() + + container := corev1.Container{ + Name: ComponentOperator, + Image: p.FelisImage, + Command: []string{"felis", "operator"}, + Args: []string{ + "--namespace", p.MinecraftNamespace, + "--metrics-bind-address", fmt.Sprintf(":%d", operatorMetricsPort), + }, + Ports: []corev1.ContainerPort{ + {Name: "metrics", ContainerPort: operatorMetricsPort, Protocol: corev1.ProtocolTCP}, + }, + VolumeMounts: []corev1.VolumeMount{ + {Name: tmpVolume, MountPath: "/tmp"}, + }, + Resources: controlPlaneResources(), + SecurityContext: hardenedContainerSecurityContext(), + } + + volumes := []corev1.Volume{ + {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, + } + + return controlPlaneDeployment(p, SAOperator, container, volumes) +} + +// reaperCronJob renders the world-retention CronJob (spec §18). It runs the +// `felis reaper` run-once entrypoint on a daily cadence — the CronJob, not the +// process, owns scheduling, so the reaper stays idempotent and restart-safe. +// +// Identity & reach. It runs as SAReaper (felis-reaper), the only SA holding +// minecraftservers:[get,patch] + persistentvolumeclaims:delete in the minecraft +// namespace (rbac.go ReaperRole). Unlike the build/restore/registry pods — which +// set AutomountServiceAccountToken=false because they never call the K8s API — the +// reaper LEGITIMATELY patches MinecraftServers (flip desiredState) and deletes +// world PVCs, so its SA token is left to auto-mount (nil). Its pod carries +// controlPlanePodLabels(ComponentReaper); reaper is deliberately excluded from the +// RCON NetworkPolicy peer (component ∉ {api,operator}), asserted in +// workloads_test.go — the reaper never opens an RCON connection. +// +// Mounts (the storage crux). felis.toml is mounted read-only from the config +// Secret (it carries the DB URL). The worlds-root is a node-local hostPath mounted +// READ-ONLY at /worlds: the reaper only reads worlds to tar them; deleting a world +// is a K8s API call (DeletePVC), never an rm, so the mount never needs write. The +// backup PVC is mounted READ-WRITE at p.ArchiveLocalPath — which MUST equal +// felis.toml [archive] local_path, because tarLocal writes archive refs as absolute +// paths under it and the restore Job later mounts the same PVC at the same path to +// resolve them (see the ArchiveLocalPath field doc). A /tmp emptyDir absorbs writes +// under the read-only root filesystem. hostPath type Directory fails the pod loud +// if the worlds-root is absent, rather than silently creating an empty dir and +// archiving nothing. +// +// Pre-conditions are the caller's: reaperCronJob assumes reaperEnabled(p) — it +// dereferences WorldsHostPath / BackupPVC / ArchiveLocalPath without re-checking. +func reaperCronJob(p Params) *batchv1.CronJob { + p = p.withDefaults() + labels := controlPlanePodLabels(ComponentReaper) + hostPathDir := corev1.HostPathDirectory + + container := corev1.Container{ + Name: ComponentReaper, + Image: p.FelisImage, + Command: []string{"felis", "reaper"}, + Args: []string{ + "--config", configFilePath, + "--worlds-root", worldsMountPath, + }, + VolumeMounts: []corev1.VolumeMount{ + {Name: configVolume, MountPath: configMountPath, ReadOnly: true}, + {Name: worldsVolume, MountPath: worldsMountPath, ReadOnly: true}, + {Name: backupVolume, MountPath: p.ArchiveLocalPath}, + {Name: tmpVolume, MountPath: "/tmp"}, + }, + Resources: controlPlaneResources(), + SecurityContext: hardenedContainerSecurityContext(), + } + + volumes := []corev1.Volume{ + { + Name: configVolume, + VolumeSource: corev1.VolumeSource{ + Secret: &corev1.SecretVolumeSource{SecretName: configSecretName}, + }, + }, + { + Name: worldsVolume, + VolumeSource: corev1.VolumeSource{ + HostPath: &corev1.HostPathVolumeSource{Path: p.WorldsHostPath, Type: &hostPathDir}, + }, + }, + { + Name: backupVolume, + VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: p.BackupPVC}, + }, + }, + {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, + } + + return &batchv1.CronJob{ + TypeMeta: metav1.TypeMeta{APIVersion: "batch/v1", Kind: "CronJob"}, + ObjectMeta: metav1.ObjectMeta{Name: SAReaper, Namespace: p.ControlNamespace, Labels: labels}, + Spec: batchv1.CronJobSpec{ + Schedule: reaperSchedule, + ConcurrencyPolicy: batchv1.ForbidConcurrent, + StartingDeadlineSeconds: int64Ptr(reaperStartingDeadlineSeconds), + SuccessfulJobsHistoryLimit: int32Ptr(reaperHistoryLimit), + FailedJobsHistoryLimit: int32Ptr(reaperHistoryLimit), + JobTemplate: batchv1.JobTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: labels}, + Spec: batchv1.JobSpec{ + BackoffLimit: int32Ptr(reaperBackoffLimit), + ActiveDeadlineSeconds: int64Ptr(reaperActiveDeadlineSeconds), + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: labels}, + Spec: corev1.PodSpec{ + ServiceAccountName: SAReaper, + RestartPolicy: corev1.RestartPolicyNever, + SecurityContext: hardenedPodSecurityContext(), + Containers: []corev1.Container{container}, + Volumes: volumes, + }, + }, + }, + }, + }, + } +} + +// controlPlaneDeployment assembles a single-replica control-plane Deployment. The +// Deployment, its selector, and the pod template all carry +// controlPlanePodLabels(component) so the three agree (a selector mismatch would +// leave pods unmanaged). The component is taken from the container name, which is +// the component value for both control-plane workloads. +// +// Strategy is Recreate, not RollingUpdate: neither the operator nor the api wires +// leader election, so a RollingUpdate's maxSurge overlap would briefly run two +// instances — two reconcilers fighting, or two processes binding the same ports. +// Recreate guarantees the old pod is gone before the new one starts. +func controlPlaneDeployment(p Params, sa string, container corev1.Container, volumes []corev1.Volume) *appsv1.Deployment { + labels := controlPlanePodLabels(container.Name) + return &appsv1.Deployment{ + TypeMeta: metav1.TypeMeta{APIVersion: "apps/v1", Kind: "Deployment"}, + ObjectMeta: metav1.ObjectMeta{Name: sa, Namespace: p.ControlNamespace, Labels: labels}, + Spec: appsv1.DeploymentSpec{ + Replicas: int32Ptr(1), + Strategy: appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}, + Selector: &metav1.LabelSelector{MatchLabels: labels}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: labels}, + Spec: corev1.PodSpec{ + ServiceAccountName: sa, + SecurityContext: hardenedPodSecurityContext(), + Containers: []corev1.Container{container}, + Volumes: volumes, + }, + }, + }, + } +} + +// registryDeployment renders the in-cluster Distribution registry (spec §16). +// Build Jobs push to registry..svc:, the destination the build +// egress NetworkPolicy opens — so this Deployment+Service+PVC make that policy +// target real. The registry never calls the K8s API, so its token auto-mount is +// disabled (matching the weak build/restore SA hygiene), and REGISTRY_HTTP_ADDR +// pins its listen port to the Service port instead of trusting the image default. +func registryDeployment(p Params) *appsv1.Deployment { + p = p.withDefaults() + labels := registryLabels() + + container := corev1.Container{ + Name: registryName, + Image: p.RegistryImage, + Env: []corev1.EnvVar{ + {Name: "REGISTRY_HTTP_ADDR", Value: fmt.Sprintf(":%d", p.RegistryPort)}, + }, + Ports: []corev1.ContainerPort{ + {Name: registryName, ContainerPort: p.RegistryPort, Protocol: corev1.ProtocolTCP}, + }, + VolumeMounts: []corev1.VolumeMount{ + {Name: registryVolume, MountPath: registryDataPath}, + {Name: tmpVolume, MountPath: "/tmp"}, + }, + Resources: controlPlaneResources(), + SecurityContext: hardenedContainerSecurityContext(), + } + + return &appsv1.Deployment{ + TypeMeta: metav1.TypeMeta{APIVersion: "apps/v1", Kind: "Deployment"}, + ObjectMeta: metav1.ObjectMeta{Name: registryName, Namespace: p.RegistryNamespace, Labels: labels}, + Spec: appsv1.DeploymentSpec{ + Replicas: int32Ptr(1), + Strategy: appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}, + Selector: &metav1.LabelSelector{MatchLabels: labels}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: labels}, + Spec: corev1.PodSpec{ + AutomountServiceAccountToken: boolPtr(false), + SecurityContext: hardenedPodSecurityContext(), + Containers: []corev1.Container{container}, + Volumes: []corev1.Volume{ + { + Name: registryVolume, + VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: registryName}, + }, + }, + {Name: tmpVolume, VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{}}}, + }, + }, + }, + }, + } +} + +// registryService renders the ClusterIP Service that gives the registry its +// pinned DNS name registry..svc: — hardcoded across the build +// subsystem and config. Its selector matches the registry pod labels; because +// those labels are NOT part-of=felis-control-plane, the registry is invisible to +// the RCON NetworkPolicy peer. +func registryService(p Params) *corev1.Service { + p = p.withDefaults() + labels := registryLabels() + return &corev1.Service{ + TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Service"}, + ObjectMeta: metav1.ObjectMeta{Name: registryName, Namespace: p.RegistryNamespace, Labels: labels}, + Spec: corev1.ServiceSpec{ + Type: corev1.ServiceTypeClusterIP, + Selector: labels, + Ports: []corev1.ServicePort{{ + Name: registryName, + Port: p.RegistryPort, + TargetPort: intstr.FromInt32(p.RegistryPort), + Protocol: corev1.ProtocolTCP, + }}, + }, + } +} + +// registryPVC renders the registry's data volume (RWO). No storageClassName is +// set, so it binds the cluster's default class — pinning one would be a guess. +func registryPVC(p Params) *corev1.PersistentVolumeClaim { + p = p.withDefaults() + return &corev1.PersistentVolumeClaim{ + TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "PersistentVolumeClaim"}, + ObjectMeta: metav1.ObjectMeta{Name: registryName, Namespace: p.RegistryNamespace, Labels: registryLabels()}, + Spec: corev1.PersistentVolumeClaimSpec{ + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + Resources: corev1.VolumeResourceRequirements{ + Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse(registryStorageSize)}, + }, + }, + } +} + +// registryLabels are the registry's recommended labels. Note the absence of +// part-of=felis-control-plane: that is what keeps the registry out of the RCON +// NetworkPolicy peer's reach (asserted in workloads_test.go). +func registryLabels() map[string]string { + return map[string]string{ + LabelName: appName, + LabelComponent: ComponentRegistry, + } +} + +// controlPlaneResources are conservative starting requests/limits. They bound +// resource exhaustion (a security-hygiene baseline) without claiming to be tuned +// for production load — that is a deployment concern. +func controlPlaneResources() corev1.ResourceRequirements { + return corev1.ResourceRequirements{ + Requests: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("50m"), + corev1.ResourceMemory: resource.MustParse("64Mi"), + }, + Limits: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse("500m"), + corev1.ResourceMemory: resource.MustParse("256Mi"), + }, + } +} + +// hardenedPodSecurityContext is the pod-level hardening shared by every workload +// here: run as a fixed non-root uid/gid with a matching fsGroup (so the registry +// can write its group-owned PVC) and the RuntimeDefault seccomp profile. +// +// SHAPE-ASSERTED, runtime-unverified: this asserts the images can run as +// nonRootUID. The felis image is built to; registry:2 (CNCF Distribution) can, +// with the data PVC fsGroup-owned. Without a cluster the actual start-up is not +// proven here. +func hardenedPodSecurityContext() *corev1.PodSecurityContext { + return &corev1.PodSecurityContext{ + RunAsNonRoot: boolPtr(true), + RunAsUser: int64Ptr(nonRootUID), + RunAsGroup: int64Ptr(nonRootUID), + FSGroup: int64Ptr(nonRootUID), + SeccompProfile: &corev1.SeccompProfile{Type: corev1.SeccompProfileTypeRuntimeDefault}, + } +} + +// hardenedContainerSecurityContext mirrors the build/restore Job containers: no +// privilege, no escalation, read-only root filesystem (all writes go to the +// mounted volumes — the config/data mounts and the /tmp emptyDir), drop ALL +// capabilities. readOnlyRootFilesystem is shape-asserted, not runtime-proven. +func hardenedContainerSecurityContext() *corev1.SecurityContext { + return &corev1.SecurityContext{ + Privileged: boolPtr(false), + AllowPrivilegeEscalation: boolPtr(false), + ReadOnlyRootFilesystem: boolPtr(true), + Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}}, + } +} + +func boolPtr(b bool) *bool { return &b } +func int32Ptr(i int32) *int32 { return &i } +func int64Ptr(i int64) *int64 { return &i } diff --git a/internal/platform/workloads_test.go b/internal/platform/workloads_test.go new file mode 100644 index 0000000..15af9d3 --- /dev/null +++ b/internal/platform/workloads_test.go @@ -0,0 +1,555 @@ +package platform + +import ( + "testing" + + appsv1 "k8s.io/api/apps/v1" + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/labels" +) + +// podSpec returns the single container and the pod template of a Deployment, +// failing if the shape is not the expected single-container pod. +func podSpec(t *testing.T, d *appsv1.Deployment) (corev1.PodSpec, corev1.Container) { + t.Helper() + ps := d.Spec.Template.Spec + if len(ps.Containers) != 1 { + t.Fatalf("%s: want exactly 1 container, got %d", d.Name, len(ps.Containers)) + } + return ps, ps.Containers[0] +} + +// rconPeerSelector returns the podSelector of the allow-rcon NetworkPolicy peer, +// compiled into the same labels.Selector K8s evaluates at runtime. This is the +// real gate: a pod reaches server RCON iff its labels Match this selector. +func rconPeerSelector(t *testing.T, p Params) labels.Selector { + t.Helper() + rcon := npByName(t, MinecraftNetworkPolicies(p), "felis-allow-rcon-from-control-plane") + if len(rcon.Spec.Ingress) != 1 || len(rcon.Spec.Ingress[0].From) != 1 { + t.Fatalf("rcon policy shape changed; want 1 ingress / 1 peer") + } + sel, err := metav1.LabelSelectorAsSelector(rcon.Spec.Ingress[0].From[0].PodSelector) + if err != nil { + t.Fatalf("compiling rcon podSelector: %v", err) + } + return sel +} + +// mapSelectorMatches evaluates a Service-style equality selector (a plain label +// map: every entry must be present and equal) against a pod's labels. +func mapSelectorMatches(selector, podLabels map[string]string) bool { + if len(selector) == 0 { + return false // an empty Service selector selects nothing useful here + } + for k, v := range selector { + if podLabels[k] != v { + return false + } + } + return true +} + +// TestControlPlaneDeployments_RunAsMatchingSA is the SA↔workload binding: the +// namespaced Roles only mean something if a workload actually runs as each SA. +// The Deployment is also NAMED for its SA (the tree convention), so a rename +// can't silently detach the labels from the identity. +func TestControlPlaneDeployments_RunAsMatchingSA(t *testing.T) { + p := testParams() + cases := []struct { + name string + dep *appsv1.Deployment + sa string + }{ + {"api", APIDeployment(p), SAAPI}, + {"operator", OperatorDeployment(p), SAOperator}, + } + for _, c := range cases { + ps, _ := podSpec(t, c.dep) + if ps.ServiceAccountName != c.sa { + t.Errorf("%s pod serviceAccountName = %q, want %q", c.name, ps.ServiceAccountName, c.sa) + } + if c.dep.Name != c.sa { + t.Errorf("%s Deployment name = %q, want %q (named for its SA)", c.name, c.dep.Name, c.sa) + } + if c.dep.Namespace != p.ControlNamespace { + t.Errorf("%s Deployment namespace = %q, want control ns %q", c.name, c.dep.Namespace, p.ControlNamespace) + } + // Selector, template labels, and object labels must agree (a mismatch + // orphans the pods). + if !labels.Equals(c.dep.Spec.Selector.MatchLabels, c.dep.Spec.Template.Labels) { + t.Errorf("%s selector %v != template labels %v", c.name, c.dep.Spec.Selector.MatchLabels, c.dep.Spec.Template.Labels) + } + } +} + +// TestControlPlanePods_SatisfyRConPeer is the second correspondence: the api and +// operator pods (which legitimately open RCON — console writes, readiness probes) +// carry labels that SATISFY the allow-rcon peer, while the registry and the reaper +// do NOT. Evaluated with the live selector, so it proves the labels and the policy +// agree rather than re-asserting the policy's shape. +func TestControlPlanePods_SatisfyRConPeer(t *testing.T) { + p := testParams() + sel := rconPeerSelector(t, p) + + apiPod := APIDeployment(p).Spec.Template.Labels + opPod := OperatorDeployment(p).Spec.Template.Labels + regPod := registryDeployment(p).Spec.Template.Labels + reaperPod := controlPlanePodLabels(ComponentReaper) + + if !sel.Matches(labels.Set(apiPod)) { + t.Errorf("api pod labels %v must satisfy the rcon peer", apiPod) + } + if !sel.Matches(labels.Set(opPod)) { + t.Errorf("operator pod labels %v must satisfy the rcon peer", opPod) + } + if sel.Matches(labels.Set(regPod)) { + t.Errorf("registry pod labels %v must NOT satisfy the rcon peer (no part-of=control-plane)", regPod) + } + if sel.Matches(labels.Set(reaperPod)) { + t.Errorf("reaper pod labels %v must NOT satisfy the rcon peer (component not in {api,operator})", reaperPod) + } +} + +// TestControlPlanePods_Hardened asserts the pod/container SecurityContext on every +// workload here mirrors the build/restore Job hardening: non-root, no privilege, +// no escalation, read-only root fs, all caps dropped. (Shape-asserted: no cluster +// proves the images actually start under these constraints.) +func TestControlPlanePods_Hardened(t *testing.T) { + p := testParams() + for _, d := range []*appsv1.Deployment{APIDeployment(p), OperatorDeployment(p), registryDeployment(p)} { + ps, c := podSpec(t, d) + + if ps.SecurityContext == nil || ps.SecurityContext.RunAsNonRoot == nil || !*ps.SecurityContext.RunAsNonRoot { + t.Errorf("%s: pod must set runAsNonRoot=true", d.Name) + } + if ps.SecurityContext == nil || ps.SecurityContext.RunAsUser == nil || *ps.SecurityContext.RunAsUser != nonRootUID { + t.Errorf("%s: pod runAsUser must be %d", d.Name, nonRootUID) + } + sc := c.SecurityContext + if sc == nil { + t.Fatalf("%s: container has no SecurityContext", d.Name) + } + if sc.Privileged == nil || *sc.Privileged { + t.Errorf("%s: container must not be privileged", d.Name) + } + if sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation { + t.Errorf("%s: container must set allowPrivilegeEscalation=false", d.Name) + } + if sc.ReadOnlyRootFilesystem == nil || !*sc.ReadOnlyRootFilesystem { + t.Errorf("%s: container must set readOnlyRootFilesystem=true", d.Name) + } + if sc.Capabilities == nil || len(sc.Capabilities.Drop) == 0 || sc.Capabilities.Drop[0] != "ALL" { + t.Errorf("%s: container must drop ALL capabilities", d.Name) + } + } +} + +// TestAPIDeployment_Wiring pins the api entrypoint, the credential plumbing, and +// the FELIS_IMAGE passthrough. +func TestAPIDeployment_Wiring(t *testing.T) { + p := testParams() + d := APIDeployment(p) + ps, c := podSpec(t, d) + + if got := append(append([]string{}, c.Command...), c.Args...); !containsSeq(got, []string{"felis", "api"}) { + t.Errorf("api command/args = %v, want it to start `felis api`", got) + } + if !contains(c.Args, "--config") || !contains(c.Args, configFilePath) { + t.Errorf("api args must mount config at %s, got %v", configFilePath, c.Args) + } + if !contains(c.Args, "--internal-addr") { + t.Errorf("api args must set --internal-addr, got %v", c.Args) + } + if c.Image != p.FelisImage { + t.Errorf("api image = %q, want FelisImage %q", c.Image, p.FelisImage) + } + + // FELIS_IMAGE passthrough (used to launch the restore Job with the same image). + if v := envValue(c.Env, "FELIS_IMAGE"); v != p.FelisImage { + t.Errorf("FELIS_IMAGE = %q, want %q", v, p.FelisImage) + } + // FELIS_SERVICE_TOKEN must come from a Secret, never a literal value. + tok := envVar(c.Env, "FELIS_SERVICE_TOKEN") + if tok == nil || tok.ValueFrom == nil || tok.ValueFrom.SecretKeyRef == nil { + t.Fatal("FELIS_SERVICE_TOKEN must be sourced from a secretKeyRef") + } + if tok.Value != "" { + t.Error("FELIS_SERVICE_TOKEN must not carry a literal value") + } + + // felis.toml carries the DB URL, so its volume must be a Secret (NOT a + // ConfigMap), mounted read-only. + cfgVol := volumeByName(ps.Volumes, configVolume) + if cfgVol == nil || cfgVol.Secret == nil { + t.Fatal("config volume must be sourced from a Secret") + } + if cfgVol.ConfigMap != nil { + t.Error("config volume must NOT be a ConfigMap (felis.toml holds the DB credential)") + } + if cfgVol.Secret.SecretName != configSecretName { + t.Errorf("config Secret name = %q, want %q", cfgVol.Secret.SecretName, configSecretName) + } + if m := mountByName(c.VolumeMounts, configVolume); m == nil || !m.ReadOnly { + t.Error("config volume must be mounted read-only") + } + + // No backup PVC in testParams ⇒ no FELIS_BACKUP_PVC env (restore degrades to 503). + if envVar(c.Env, "FELIS_BACKUP_PVC") != nil { + t.Error("FELIS_BACKUP_PVC must be absent when no backup PVC is configured") + } +} + +// TestAPIDeployment_BackupPVC proves the FELIS_BACKUP_PVC env appears only when a +// backup PVC is named. +func TestAPIDeployment_BackupPVC(t *testing.T) { + p := testParams() + p.BackupPVC = "felis-backups" + _, c := podSpec(t, APIDeployment(p)) + if v := envValue(c.Env, "FELIS_BACKUP_PVC"); v != "felis-backups" { + t.Errorf("FELIS_BACKUP_PVC = %q, want %q", v, "felis-backups") + } +} + +// TestOperatorDeployment_Wiring pins the operator entrypoint, its namespace split, +// and its deliberately smaller surface (NO config Secret — it holds no DB URL). +func TestOperatorDeployment_Wiring(t *testing.T) { + p := testParams() + d := OperatorDeployment(p) + ps, c := podSpec(t, d) + + if got := append(append([]string{}, c.Command...), c.Args...); !containsSeq(got, []string{"felis", "operator"}) { + t.Errorf("operator command/args = %v, want it to start `felis operator`", got) + } + if !contains(c.Args, "--namespace") || !contains(c.Args, p.MinecraftNamespace) { + t.Errorf("operator must watch --namespace %s, got %v", p.MinecraftNamespace, c.Args) + } + if c.Image != p.FelisImage { + t.Errorf("operator image = %q, want FelisImage %q", c.Image, p.FelisImage) + } + // No config Secret volume: the operator reads config from flags + the API only. + for _, v := range ps.Volumes { + if v.Secret != nil { + t.Errorf("operator must mount NO Secret volume, found %q", v.Name) + } + } + // And it must hold no credential env at all. + if len(c.Env) != 0 { + t.Errorf("operator must carry no env (flags-only), got %v", c.Env) + } +} + +// TestRegistry_DeploymentServicePVC pins the in-cluster registry: its pinned +// listen port, token-mount hygiene, the Service that gives it its DNS name, and +// the backing PVC — the trio the build egress policy targets. +func TestRegistry_DeploymentServicePVC(t *testing.T) { + // testParams leaves RegistryNamespace/RegistryPort zero; the renderers fill them + // via withDefaults, so compare against the defaulted Params. + p := testParams().withDefaults() + dep := registryDeployment(p) + svc := registryService(p) + pvc := registryPVC(p) + + ps, c := podSpec(t, dep) + if c.Image != defaultRegistryImage { + t.Errorf("registry image = %q, want default %q", c.Image, defaultRegistryImage) + } + // REGISTRY_HTTP_ADDR pins the listen port to the Service port rather than + // trusting the image default. + if v := envValue(c.Env, "REGISTRY_HTTP_ADDR"); v != ":5000" { + t.Errorf("REGISTRY_HTTP_ADDR = %q, want :5000", v) + } + // Registry never calls the K8s API ⇒ no auto-mounted token. + if ps.AutomountServiceAccountToken == nil || *ps.AutomountServiceAccountToken { + t.Error("registry pod must set automountServiceAccountToken=false") + } + // Data is on the PVC named "registry". + dataVol := volumeByName(ps.Volumes, registryVolume) + if dataVol == nil || dataVol.PersistentVolumeClaim == nil || dataVol.PersistentVolumeClaim.ClaimName != registryName { + t.Errorf("registry data volume must be PVC %q", registryName) + } + + // Service: gives the pinned DNS name registry..svc:5000. + if svc.Name != registryName || svc.Namespace != p.RegistryNamespace { + t.Errorf("registry Service = %s/%s, want %s/%s", svc.Namespace, svc.Name, p.RegistryNamespace, registryName) + } + if len(svc.Spec.Ports) != 1 || svc.Spec.Ports[0].Port != p.RegistryPort { + t.Errorf("registry Service port = %v, want %d", svc.Spec.Ports, p.RegistryPort) + } + // The Service selector must select the registry pods... + if !mapSelectorMatches(svc.Spec.Selector, dep.Spec.Template.Labels) { + t.Errorf("registry Service selector %v does not select registry pod labels %v", svc.Spec.Selector, dep.Spec.Template.Labels) + } + // ...and must NOT select the api pods (distinct component). + if mapSelectorMatches(svc.Spec.Selector, APIDeployment(p).Spec.Template.Labels) { + t.Error("registry Service selector must not select the api pod") + } + + // PVC: RWO with a concrete request, no pinned storage class. + if pvc.Name != registryName || pvc.Namespace != p.RegistryNamespace { + t.Errorf("registry PVC = %s/%s, want %s/%s", pvc.Namespace, pvc.Name, p.RegistryNamespace, registryName) + } + if !contains(accessModeStrings(pvc.Spec.AccessModes), string(corev1.ReadWriteOnce)) { + t.Errorf("registry PVC access modes = %v, want ReadWriteOnce", pvc.Spec.AccessModes) + } + if pvc.Spec.Resources.Requests.Storage().IsZero() { + t.Error("registry PVC must request a non-zero storage size") + } +} + +// TestWorkloads_BundleContents sanity-checks the slice Workloads returns: the two +// control-plane Deployments + the registry Deployment/Service/PVC, every one with +// TypeMeta (so its YAML header renders). +func TestWorkloads_BundleContents(t *testing.T) { + objs := Workloads(testParams()) + if len(objs) != 5 { + t.Fatalf("Workloads returned %d objects, want 5", len(objs)) + } + for _, o := range objs { + gvk := o.GetObjectKind().GroupVersionKind() + if gvk.Kind == "" || gvk.Version == "" { + t.Errorf("%T missing TypeMeta (kind=%q version=%q)", o, gvk.Kind, gvk.Version) + } + } +} + +// cronPodSpec returns the single container and pod template of a CronJob's Job +// template, failing if the shape is not a single-container pod (the reaper's shape). +func cronPodSpec(t *testing.T, cj *batchv1.CronJob) (corev1.PodSpec, corev1.Container) { + t.Helper() + ps := cj.Spec.JobTemplate.Spec.Template.Spec + if len(ps.Containers) != 1 { + t.Fatalf("%s: want exactly 1 container, got %d", cj.Name, len(ps.Containers)) + } + return ps, ps.Containers[0] +} + +// findCronJob returns the first CronJob in objs, or nil — used to assert the +// reaper's presence/absence in the rendered Workloads slice. +func findCronJob(objs []Object) *batchv1.CronJob { + for _, o := range objs { + if cj, ok := o.(*batchv1.CronJob); ok { + return cj + } + } + return nil +} + +// reaperParams is testParams with the retention storage trio supplied, so the +// reaper CronJob renders. The paths are illustrative (no cluster runs here). +func reaperParams() Params { + p := testParams() + p.WorldsHostPath = "/var/lib/felis/worlds" + p.BackupPVC = "felis-backups" + p.ArchiveLocalPath = "/backups" + return p +} + +// TestReaperCronJob_Gating proves the reaper renders iff all three storage +// coordinates are present: an incomplete configuration must produce NO CronJob +// (the partial-flag mistake is rejected at the CLI; here the renderer fails safe). +func TestReaperCronJob_Gating(t *testing.T) { + cases := []struct { + name string + mutate func(p *Params) + want bool + }{ + {"none", func(p *Params) {}, false}, + {"worlds only", func(p *Params) { p.WorldsHostPath = "/w" }, false}, + {"worlds+backup", func(p *Params) { p.WorldsHostPath = "/w"; p.BackupPVC = "b" }, false}, + {"worlds+archive", func(p *Params) { p.WorldsHostPath = "/w"; p.ArchiveLocalPath = "/a" }, false}, + {"backup+archive (no worlds)", func(p *Params) { p.BackupPVC = "b"; p.ArchiveLocalPath = "/a" }, false}, + {"all three", func(p *Params) { p.WorldsHostPath = "/w"; p.BackupPVC = "b"; p.ArchiveLocalPath = "/a" }, true}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + p := testParams() + c.mutate(&p) + if got := reaperEnabled(p); got != c.want { + t.Errorf("reaperEnabled = %v, want %v", got, c.want) + } + cj := findCronJob(Workloads(p)) + if c.want && cj == nil { + t.Error("CronJob must be in Workloads when enabled") + } + if !c.want && cj != nil { + t.Error("CronJob must NOT be in Workloads when disabled") + } + }) + } +} + +// TestReaperCronJob_Shape pins the rendered CronJob: its scheduling guards, its +// run-as identity (felis-reaper WITH an auto-mounted token, because it legitimately +// calls the K8s API — unlike the weak Job/registry pods), the hardening, the +// entrypoint, and the three-mount storage crux (config RO, worlds hostPath RO at +// /worlds, backup PVC RW at ArchiveLocalPath). Shape-asserted, runtime-unverified. +func TestReaperCronJob_Shape(t *testing.T) { + p := reaperParams() + cj := reaperCronJob(p) + + if cj.Kind != "CronJob" || cj.APIVersion != "batch/v1" { + t.Errorf("CronJob TypeMeta = %s/%s, want batch/v1 CronJob", cj.APIVersion, cj.Kind) + } + if cj.Name != SAReaper { + t.Errorf("CronJob name = %q, want %q", cj.Name, SAReaper) + } + if cj.Namespace != p.ControlNamespace { + t.Errorf("CronJob namespace = %q, want control ns %q", cj.Namespace, p.ControlNamespace) + } + + spec := cj.Spec + if spec.Schedule == "" { + t.Error("CronJob must set a schedule") + } + if spec.ConcurrencyPolicy != batchv1.ForbidConcurrent { + t.Errorf("concurrencyPolicy = %q, want Forbid (retention runs must not overlap)", spec.ConcurrencyPolicy) + } + if spec.StartingDeadlineSeconds == nil { + t.Error("CronJob must set startingDeadlineSeconds (a missed run should still start, bounded)") + } + if spec.SuccessfulJobsHistoryLimit == nil || spec.FailedJobsHistoryLimit == nil { + t.Error("CronJob must bound job history") + } + js := spec.JobTemplate.Spec + if js.BackoffLimit == nil { + t.Error("Job must set backoffLimit") + } + if js.ActiveDeadlineSeconds == nil { + t.Error("Job must set activeDeadlineSeconds (a wedged run must not hold the Forbid lock forever)") + } + + ps, c := cronPodSpec(t, cj) + if ps.RestartPolicy != corev1.RestartPolicyNever { + t.Errorf("pod restartPolicy = %q, want Never", ps.RestartPolicy) + } + if ps.ServiceAccountName != SAReaper { + t.Errorf("pod serviceAccountName = %q, want %q (it patches MinecraftServers and deletes PVCs)", ps.ServiceAccountName, SAReaper) + } + // The reaper LEGITIMATELY calls the K8s API, so — unlike the build/restore/ + // registry pods — it must NOT disable the SA-token auto-mount. + if ps.AutomountServiceAccountToken != nil { + t.Errorf("reaper pod must auto-mount its SA token (got AutomountServiceAccountToken=%v); it needs the API", *ps.AutomountServiceAccountToken) + } + + // Hardening mirrors the other control-plane pods. + if ps.SecurityContext == nil || ps.SecurityContext.RunAsNonRoot == nil || !*ps.SecurityContext.RunAsNonRoot { + t.Error("reaper pod must set runAsNonRoot=true") + } + if c.SecurityContext == nil || c.SecurityContext.ReadOnlyRootFilesystem == nil || !*c.SecurityContext.ReadOnlyRootFilesystem { + t.Error("reaper container must set readOnlyRootFilesystem=true") + } + if c.SecurityContext == nil || c.SecurityContext.Capabilities == nil || len(c.SecurityContext.Capabilities.Drop) == 0 || c.SecurityContext.Capabilities.Drop[0] != "ALL" { + t.Error("reaper container must drop ALL capabilities") + } + + // Entrypoint: `felis reaper --config --worlds-root /worlds`. + if got := append(append([]string{}, c.Command...), c.Args...); !containsSeq(got, []string{"felis", "reaper"}) { + t.Errorf("reaper command/args = %v, want it to start `felis reaper`", got) + } + if !contains(c.Args, "--config") || !contains(c.Args, configFilePath) { + t.Errorf("reaper must read config at %s, got %v", configFilePath, c.Args) + } + if !contains(c.Args, "--worlds-root") || !contains(c.Args, worldsMountPath) { + t.Errorf("reaper must read worlds at %s, got %v", worldsMountPath, c.Args) + } + if c.Image != p.FelisImage { + t.Errorf("reaper image = %q, want FelisImage %q", c.Image, p.FelisImage) + } + + // config: Secret, mounted read-only (it carries the DB URL). + cfgVol := volumeByName(ps.Volumes, configVolume) + if cfgVol == nil || cfgVol.Secret == nil || cfgVol.Secret.SecretName != configSecretName { + t.Errorf("config volume must be Secret %q", configSecretName) + } + if m := mountByName(c.VolumeMounts, configVolume); m == nil || !m.ReadOnly { + t.Error("config must be mounted read-only") + } + + // worlds: node hostPath at WorldsHostPath, type Directory, mounted READ-ONLY at + // /worlds — the reaper only reads worlds to tar them (deletion is a PVC API call). + wVol := volumeByName(ps.Volumes, worldsVolume) + if wVol == nil || wVol.HostPath == nil || wVol.HostPath.Path != p.WorldsHostPath { + t.Errorf("worlds volume must be hostPath %q, got %+v", p.WorldsHostPath, wVol) + } + if wVol != nil && (wVol.HostPath == nil || wVol.HostPath.Type == nil || *wVol.HostPath.Type != corev1.HostPathDirectory) { + t.Error("worlds hostPath must be type Directory (fail loud if the dir is absent)") + } + if m := mountByName(c.VolumeMounts, worldsVolume); m == nil || m.MountPath != worldsMountPath || !m.ReadOnly { + t.Errorf("worlds must be mounted read-only at %s, got %+v", worldsMountPath, m) + } + + // backup: PVC, mounted READ-WRITE at ArchiveLocalPath (== felis.toml [archive] + // local_path, so tarLocal's absolute archive refs resolve under it). + bVol := volumeByName(ps.Volumes, backupVolume) + if bVol == nil || bVol.PersistentVolumeClaim == nil || bVol.PersistentVolumeClaim.ClaimName != p.BackupPVC { + t.Errorf("backup volume must be PVC %q, got %+v", p.BackupPVC, bVol) + } + if m := mountByName(c.VolumeMounts, backupVolume); m == nil || m.MountPath != p.ArchiveLocalPath || m.ReadOnly { + t.Errorf("backup must be mounted read-write at ArchiveLocalPath %q, got %+v", p.ArchiveLocalPath, m) + } + + // The reaper never opens RCON, so its pod labels must NOT satisfy the RCON peer. + if sel := rconPeerSelector(t, p); sel.Matches(labels.Set(cj.Spec.JobTemplate.Spec.Template.Labels)) { + t.Error("reaper pod labels must NOT satisfy the rcon peer (component not in {api,operator})") + } +} + +// --- small env/volume helpers (test-local) --- + +func envVar(env []corev1.EnvVar, name string) *corev1.EnvVar { + for i := range env { + if env[i].Name == name { + return &env[i] + } + } + return nil +} + +func envValue(env []corev1.EnvVar, name string) string { + if v := envVar(env, name); v != nil { + return v.Value + } + return "" +} + +func volumeByName(vols []corev1.Volume, name string) *corev1.Volume { + for i := range vols { + if vols[i].Name == name { + return &vols[i] + } + } + return nil +} + +func mountByName(mounts []corev1.VolumeMount, name string) *corev1.VolumeMount { + for i := range mounts { + if mounts[i].Name == name { + return &mounts[i] + } + } + return nil +} + +func accessModeStrings(modes []corev1.PersistentVolumeAccessMode) []string { + out := make([]string, len(modes)) + for i, m := range modes { + out[i] = string(m) + } + return out +} + +// containsSeq reports whether sub appears as a contiguous prefix-anchored run at +// the START of seq (command then args), which is what we want for an entrypoint. +func containsSeq(seq, sub []string) bool { + if len(sub) > len(seq) { + return false + } + for i := range sub { + if seq[i] != sub[i] { + return false + } + } + return true +}