package main import ( "flag" "fmt" "io" "log/slog" "os" "felis.lolicon.best/internal/apis/felis/v1alpha1" felismetrics "felis.lolicon.best/internal/metrics" "felis.lolicon.best/internal/operator" "github.com/go-logr/logr" "k8s.io/apimachinery/pkg/runtime" utilruntime "k8s.io/apimachinery/pkg/util/runtime" clientgoscheme "k8s.io/client-go/kubernetes/scheme" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/cache" "sigs.k8s.io/controller-runtime/pkg/healthz" ctrlmetrics "sigs.k8s.io/controller-runtime/pkg/metrics" metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server" ) // cmdOperator runs the MinecraftServer controller-manager (spec §5). It builds // the scheme, wires the Reconciler with the production RCON prober, and blocks // on the manager until the process receives a termination signal. func cmdOperator(args []string, _, stderr io.Writer) int { fs := flag.NewFlagSet("operator", flag.ContinueOnError) fs.SetOutput(stderr) metricsAddr := fs.String("metrics-bind-address", ":8080", "address the metric endpoint binds to") // healthAddr serves the manager's health endpoints (/healthz, /readyz) that the // Deployment's probes dial. Without it the operator pod would carry no probe at // all, and a wedged manager would keep its endpoint forever. It must differ from // metricsAddr: the metrics server owns :8080. healthAddr := fs.String("health-probe-bind-address", ":8081", "address the health probe endpoint binds to") // namespace MUST equal the [k8s] namespace felis-api is configured with, and // the deployment manifests (felis manifests) render both from one value. It // scopes the manager's cache (informers) to a single namespace so the operator // can run under a namespaced Role instead of cluster-admin (spec §21). The // default matches config.defaultNamespace, so an unconfigured deployment // agrees; a mismatch would silently scope the cache to the wrong namespace and // every reconcile would see zero servers — hence the watched namespace is // logged at startup so a divergence surfaces immediately rather than silently. namespace := fs.String("namespace", "minecraft", "namespace to watch; must match felis-api's [k8s] namespace") if err := fs.Parse(args); err != nil { return 2 } scheme := runtime.NewScheme() utilruntime.Must(clientgoscheme.AddToScheme(scheme)) utilruntime.Must(v1alpha1.AddToScheme(scheme)) // controller-runtime logs through its own logr sink; without one, its first // reconcile prints "log.SetLogger(...) was never called" ATTACHED TO A FULL // GOROUTINE STACK — pure noise, not signal. Route it to slog's default handler // so its messages appear as ordinary stderr lines. ctrl.SetLogger(logr.FromSlogHandler(slog.Default().Handler())) mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{ Scheme: scheme, Metrics: metricsserver.Options{BindAddress: *metricsAddr}, HealthProbeBindAddress: *healthAddr, // Scope every informer to the single watched namespace. Without this the // cached client (mgr.GetClient) would LIST/WATCH cluster-wide, which a // namespaced Role cannot grant — the operator would fail closed at runtime // or, worse, demand cluster-admin. With it, the platform.OperatorRole // (get/list/watch in one namespace) is exactly sufficient. Cache: cache.Options{ DefaultNamespaces: map[string]cache.Config{*namespace: {}}, }, }) if err != nil { fmt.Fprintf(stderr, "felis operator: create manager: %v\n", err) return 1 } fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace) // Register the two probe endpoints. controller-runtime only mounts /healthz and // /readyz once at least one check is registered, so a bare listener would 404. // The checks are the canonical always-pass ping: the probes' contract is "the // manager process is up and serving", and a dependency hiccup (e.g. an API blip) // must not restart the operator. if err := mgr.AddHealthzCheck("ping", healthz.Ping); err != nil { fmt.Fprintf(stderr, "felis operator: register healthz check: %v\n", err) return 1 } if err := mgr.AddReadyzCheck("ping", healthz.Ping); err != nil { fmt.Fprintf(stderr, "felis operator: register readyz check: %v\n", err) return 1 } // Publish the named felis_* metrics (spec §23) on the endpoint the manager // already serves (metricsAddr). controller-runtime's metrics server exposes // its global Registry, so registering into it is all that is needed for // /metrics to carry felis_servers_total and friends. Register is idempotent, // so an in-process restart never double-registers fatally. if err := felismetrics.Register(ctrlmetrics.Registry); err != nil { fmt.Fprintf(stderr, "felis operator: register metrics: %v\n", err) return 1 } r := &operator.Reconciler{ Client: mgr.GetClient(), Scheme: mgr.GetScheme(), Prober: operator.RconProber{}, // The operator's own image, for the forwarding-config initContainer it // injects into user servers. The Deployment passes it as FELIS_IMAGE (see // platform.OperatorDeployment); absent, that injection is simply skipped. FelisImage: os.Getenv("FELIS_IMAGE"), // Uncached: the maintenance-lock check lists Jobs only when a server is // about to start, which does not justify a namespace-wide Job informer. Jobs: mgr.GetAPIReader(), } if err := r.SetupWithManager(mgr); err != nil { fmt.Fprintf(stderr, "felis operator: setup controller: %v\n", err) return 1 } // Republish felis_servers_total from a periodic full List of the fleet. A // per-object reconcile can never maintain a fleet-wide gauge correctly, so a // snapshot Runnable owns it; it shares the manager's cached client and stops // with the manager. if err := mgr.Add(&operator.GaugeSyncer{Client: mgr.GetClient()}); err != nil { fmt.Fprintf(stderr, "felis operator: add gauge syncer: %v\n", err) return 1 } if err := mgr.Start(ctrl.SetupSignalHandler()); err != nil { fmt.Fprintf(stderr, "felis operator: manager exited: %v\n", err) return 1 } return 0 }