Files
Felis/cmd/felis/operator.go
T
Lemon-miaow 0c8e29b05a fix(platform): give every control-plane Deployment real probes (#8 follow-up)
The api, operator and registry Deployments shipped with no liveness/readiness
probes at all: a wedged process stayed 'Running' forever, and the operator had
no health listener to probe in the first place. Kaniko build evidence on a
fresh install showed the only cluster-wide red after a disk-pressure pass was
Deployment status that never reflected health.

- felis-api: readiness /readyz (DB + K8s API round-trip) and liveness /healthz
  on the internal face (:8081), the only listener that serves both endpoints;
  liveness deliberately avoids /readyz so a DB blip cannot restart the api.
- felis-operator: new --health-probe-bind-address (:8081) with controller-
  runtime's /healthz + /readyz (registered ping checks; an unregistered handler
  map would 404), plus the matching container port and probes.
- registry: /v2/ probes on the pinned port, so a broken storage backend stops
  reading as 'Running'.

Tests pin paths, ports, and that each probe targets a declared container port.
2026-09-22 22:22:37 +08:00

131 lines
5.8 KiB
Go

package main
import (
"flag"
"fmt"
"io"
"log/slog"
"os"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
felismetrics "felis.lolicon.best/internal/metrics"
"felis.lolicon.best/internal/operator"
"github.com/go-logr/logr"
"k8s.io/apimachinery/pkg/runtime"
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/cache"
"sigs.k8s.io/controller-runtime/pkg/healthz"
ctrlmetrics "sigs.k8s.io/controller-runtime/pkg/metrics"
metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server"
)
// cmdOperator runs the MinecraftServer controller-manager (spec §5). It builds
// the scheme, wires the Reconciler with the production RCON prober, and blocks
// on the manager until the process receives a termination signal.
func cmdOperator(args []string, _, stderr io.Writer) int {
fs := flag.NewFlagSet("operator", flag.ContinueOnError)
fs.SetOutput(stderr)
metricsAddr := fs.String("metrics-bind-address", ":8080", "address the metric endpoint binds to")
// healthAddr serves the manager's health endpoints (/healthz, /readyz) that the
// Deployment's probes dial. Without it the operator pod would carry no probe at
// all, and a wedged manager would keep its endpoint forever. It must differ from
// metricsAddr: the metrics server owns :8080.
healthAddr := fs.String("health-probe-bind-address", ":8081", "address the health probe endpoint binds to")
// namespace MUST equal the [k8s] namespace felis-api is configured with, and
// the deployment manifests (felis manifests) render both from one value. It
// scopes the manager's cache (informers) to a single namespace so the operator
// can run under a namespaced Role instead of cluster-admin (spec §21). The
// default matches config.defaultNamespace, so an unconfigured deployment
// agrees; a mismatch would silently scope the cache to the wrong namespace and
// every reconcile would see zero servers — hence the watched namespace is
// logged at startup so a divergence surfaces immediately rather than silently.
namespace := fs.String("namespace", "minecraft", "namespace to watch; must match felis-api's [k8s] namespace")
if err := fs.Parse(args); err != nil {
return 2
}
scheme := runtime.NewScheme()
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
utilruntime.Must(v1alpha1.AddToScheme(scheme))
// controller-runtime logs through its own logr sink; without one, its first
// reconcile prints "log.SetLogger(...) was never called" ATTACHED TO A FULL
// GOROUTINE STACK — pure noise, not signal. Route it to slog's default handler
// so its messages appear as ordinary stderr lines.
ctrl.SetLogger(logr.FromSlogHandler(slog.Default().Handler()))
mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{
Scheme: scheme,
Metrics: metricsserver.Options{BindAddress: *metricsAddr},
HealthProbeBindAddress: *healthAddr,
// Scope every informer to the single watched namespace. Without this the
// cached client (mgr.GetClient) would LIST/WATCH cluster-wide, which a
// namespaced Role cannot grant — the operator would fail closed at runtime
// or, worse, demand cluster-admin. With it, the platform.OperatorRole
// (get/list/watch in one namespace) is exactly sufficient.
Cache: cache.Options{
DefaultNamespaces: map[string]cache.Config{*namespace: {}},
},
})
if err != nil {
fmt.Fprintf(stderr, "felis operator: create manager: %v\n", err)
return 1
}
fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace)
// Register the two probe endpoints. controller-runtime only mounts /healthz and
// /readyz once at least one check is registered, so a bare listener would 404.
// The checks are the canonical always-pass ping: the probes' contract is "the
// manager process is up and serving", and a dependency hiccup (e.g. an API blip)
// must not restart the operator.
if err := mgr.AddHealthzCheck("ping", healthz.Ping); err != nil {
fmt.Fprintf(stderr, "felis operator: register healthz check: %v\n", err)
return 1
}
if err := mgr.AddReadyzCheck("ping", healthz.Ping); err != nil {
fmt.Fprintf(stderr, "felis operator: register readyz check: %v\n", err)
return 1
}
// Publish the named felis_* metrics (spec §23) on the endpoint the manager
// already serves (metricsAddr). controller-runtime's metrics server exposes
// its global Registry, so registering into it is all that is needed for
// /metrics to carry felis_servers_total and friends. Register is idempotent,
// so an in-process restart never double-registers fatally.
if err := felismetrics.Register(ctrlmetrics.Registry); err != nil {
fmt.Fprintf(stderr, "felis operator: register metrics: %v\n", err)
return 1
}
r := &operator.Reconciler{
Client: mgr.GetClient(),
Scheme: mgr.GetScheme(),
Prober: operator.RconProber{},
// The operator's own image, for the forwarding-config initContainer it
// injects into user servers. The Deployment passes it as FELIS_IMAGE (see
// platform.OperatorDeployment); absent, that injection is simply skipped.
FelisImage: os.Getenv("FELIS_IMAGE"),
}
if err := r.SetupWithManager(mgr); err != nil {
fmt.Fprintf(stderr, "felis operator: setup controller: %v\n", err)
return 1
}
// Republish felis_servers_total from a periodic full List of the fleet. A
// per-object reconcile can never maintain a fleet-wide gauge correctly, so a
// snapshot Runnable owns it; it shares the manager's cached client and stops
// with the manager.
if err := mgr.Add(&operator.GaugeSyncer{Client: mgr.GetClient()}); err != nil {
fmt.Fprintf(stderr, "felis operator: add gauge syncer: %v\n", err)
return 1
}
if err := mgr.Start(ctrl.SetupSignalHandler()); err != nil {
fmt.Fprintf(stderr, "felis operator: manager exited: %v\n", err)
return 1
}
return 0
}