fix(platform): give every control-plane Deployment real probes (#8 follow-up)
The api, operator and registry Deployments shipped with no liveness/readiness probes at all: a wedged process stayed 'Running' forever, and the operator had no health listener to probe in the first place. Kaniko build evidence on a fresh install showed the only cluster-wide red after a disk-pressure pass was Deployment status that never reflected health. - felis-api: readiness /readyz (DB + K8s API round-trip) and liveness /healthz on the internal face (:8081), the only listener that serves both endpoints; liveness deliberately avoids /readyz so a DB blip cannot restart the api. - felis-operator: new --health-probe-bind-address (:8081) with controller- runtime's /healthz + /readyz (registered ping checks; an unregistered handler map would 404), plus the matching container port and probes. - registry: /v2/ probes on the pinned port, so a broken storage backend stops reading as 'Running'. Tests pin paths, ports, and that each probe targets a declared container port.
This commit is contained in:
3 files changed
+162
-4
No files matched your search
+23
-2
@@ -16,6 +16,7 @@ import (
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/cache"
|
||||
"sigs.k8s.io/controller-runtime/pkg/healthz"
|
||||
ctrlmetrics "sigs.k8s.io/controller-runtime/pkg/metrics"
|
||||
metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server"
|
||||
)
|
||||
@@ -27,6 +28,11 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("operator", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
metricsAddr := fs.String("metrics-bind-address", ":8080", "address the metric endpoint binds to")
|
||||
// healthAddr serves the manager's health endpoints (/healthz, /readyz) that the
|
||||
// Deployment's probes dial. Without it the operator pod would carry no probe at
|
||||
// all, and a wedged manager would keep its endpoint forever. It must differ from
|
||||
// metricsAddr: the metrics server owns :8080.
|
||||
healthAddr := fs.String("health-probe-bind-address", ":8081", "address the health probe endpoint binds to")
|
||||
// namespace MUST equal the [k8s] namespace felis-api is configured with, and
|
||||
// the deployment manifests (felis manifests) render both from one value. It
|
||||
// scopes the manager's cache (informers) to a single namespace so the operator
|
||||
@@ -51,8 +57,9 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
ctrl.SetLogger(logr.FromSlogHandler(slog.Default().Handler()))
|
||||
|
||||
mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{
|
||||
Scheme: scheme,
|
||||
Metrics: metricsserver.Options{BindAddress: *metricsAddr},
|
||||
Scheme: scheme,
|
||||
Metrics: metricsserver.Options{BindAddress: *metricsAddr},
|
||||
HealthProbeBindAddress: *healthAddr,
|
||||
// Scope every informer to the single watched namespace. Without this the
|
||||
// cached client (mgr.GetClient) would LIST/WATCH cluster-wide, which a
|
||||
// namespaced Role cannot grant — the operator would fail closed at runtime
|
||||
@@ -68,6 +75,20 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace)
|
||||
|
||||
// Register the two probe endpoints. controller-runtime only mounts /healthz and
|
||||
// /readyz once at least one check is registered, so a bare listener would 404.
|
||||
// The checks are the canonical always-pass ping: the probes' contract is "the
|
||||
// manager process is up and serving", and a dependency hiccup (e.g. an API blip)
|
||||
// must not restart the operator.
|
||||
if err := mgr.AddHealthzCheck("ping", healthz.Ping); err != nil {
|
||||
fmt.Fprintf(stderr, "felis operator: register healthz check: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if err := mgr.AddReadyzCheck("ping", healthz.Ping); err != nil {
|
||||
fmt.Fprintf(stderr, "felis operator: register readyz check: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Publish the named felis_* metrics (spec §23) on the endpoint the manager
|
||||
// already serves (metricsAddr). controller-runtime's metrics server exposes
|
||||
// its global Registry, so registering into it is all that is needed for
|
||||
|
||||
Reference in new issue
Block a user