Files
Felis/internal/metrics/metrics.go
T

236 lines
9.9 KiB
Go

// Package metrics defines and registers the named felis_* Prometheus
// collectors mandated by spec §23.
//
// The collectors are package-level vars so any subsystem (operator, reaper,
// image builder, prober) can record into them without importing back into a
// metrics owner and risking an import cycle. Wiring them into a registry is a
// single Register call that takes a prometheus.Registerer — the same
// inject-the-interface, fake-at-the-edge pattern used elsewhere in Felis
// (ownerStore, PVCResolver): production passes controller-runtime's global
// Registry (served by the operator's :metrics endpoint), tests pass a fresh
// prometheus.NewRegistry() so assertions never collide with global state.
package metrics
import (
"errors"
"github.com/prometheus/client_golang/prometheus"
)
// namespace prefixes every collector, so the exposed names are exactly
// felis_<name> — matching the metric names spec §23 mandates.
const namespace = "felis"
var (
// ServersTotal is the current number of MinecraftServers the operator knows
// about, partitioned by desiredState. It is a gauge, not a monotonic counter:
// the reconcile loop Set()s it to the live fleet size, so it can fall as
// servers are deleted. (The _total suffix follows the name spec §23 fixed.)
ServersTotal = prometheus.NewGaugeVec(prometheus.GaugeOpts{
Namespace: namespace,
Name: "servers_total",
Help: "Current number of Minecraft servers known to the operator, by desired state.",
}, []string{"state"})
// ServerPhase is 1 for each server's current phase and absent for every other
// phase. role is the server's system role (login, lobby), empty for a user
// server, so an alert can single out the login gate; desired is its
// desiredState, so a server that is down on purpose can be told from one that
// failed to come up. SyncServerPhases republishes it from a full List, so a
// deleted server's series goes away instead of freezing at its last phase.
ServerPhase = prometheus.NewGaugeVec(prometheus.GaugeOpts{
Namespace: namespace,
Name: "server_phase",
Help: "1 for each Minecraft server's current phase, by server, system role, phase and desired state.",
}, []string{"server", "role", "phase", "desired"})
// StartDurationSeconds observes the wall-clock time from desiredState=Running
// to a server reporting ready. Buckets are tuned for Minecraft cold starts
// (seconds to a few minutes), not the default sub-second web-latency buckets.
StartDurationSeconds = prometheus.NewHistogram(prometheus.HistogramOpts{
Namespace: namespace,
Name: "start_duration_seconds",
Help: "Time from desiredState=Running to a server reporting ready, in seconds.",
Buckets: []float64{1, 2, 5, 10, 20, 30, 45, 60, 90, 120, 180, 300, 600},
})
// ImageBuildFailuresTotal counts modpack/image build failures (spec §8 lane).
ImageBuildFailuresTotal = prometheus.NewCounter(prometheus.CounterOpts{
Namespace: namespace,
Name: "image_build_failures_total",
Help: "Total number of image build failures.",
})
// ReaperWorldsDeletedTotal counts world volumes deleted by the reaper.
ReaperWorldsDeletedTotal = prometheus.NewCounter(prometheus.CounterOpts{
Namespace: namespace,
Name: "reaper_worlds_deleted_total",
Help: "Total number of world volumes deleted by the reaper.",
})
// OTPLockoutsTotal counts accounts whose email-code door locked after the
// daily wrong-code budget was spent, by code purpose. Outside a person
// fumbling codes, a lockout means someone is guessing at that account.
OTPLockoutsTotal = prometheus.NewCounterVec(prometheus.CounterOpts{
Namespace: namespace,
Name: "auth_otp_lockouts_total",
Help: "Email-code doors locked after too many wrong codes, by purpose.",
}, []string{"purpose"})
// MailTotal counts mail the API tried to send, by kind (otp, notice) and
// result: sent, failed (the relay refused it) or throttled (the
// install-wide mail budget refused it before it reached the relay).
MailTotal = prometheus.NewCounterVec(prometheus.CounterOpts{
Namespace: namespace,
Name: "mail_total",
Help: "Mail the API tried to send, by kind and result (sent, failed, throttled).",
}, []string{"kind", "result"})
// RateLimitedTotal counts requests refused by a volumetric limit, by scope
// (auth_door: one client address calling the public sign-in doors too fast).
RateLimitedTotal = prometheus.NewCounterVec(prometheus.CounterOpts{
Namespace: namespace,
Name: "rate_limited_total",
Help: "Requests refused by a volumetric rate limit, by scope.",
}, []string{"scope"})
// AuthFailuresTotal counts refused sign-in attempts by door (login_email,
// op_login, passkey, passkey_discoverable, setup_redeem, bind_redeem) and
// reason (bad_code, no_account, staff_account, bad_credential, ...). A few a
// day is people mistyping; a steady stream is guessing or enumeration.
AuthFailuresTotal = prometheus.NewCounterVec(prometheus.CounterOpts{
Namespace: namespace,
Name: "auth_failures_total",
Help: "Refused sign-in attempts, by door and reason.",
}, []string{"door", "reason"})
// SessionsRevokedTotal counts sessions ended before expiry, by who ended
// them: logout (the holder signing out), self (the holder ending sessions
// from their session list), security (the other sessions ended when the
// holder removes a passkey or changes email) or admin (the owner revoking a
// user's sessions).
SessionsRevokedTotal = prometheus.NewCounterVec(prometheus.CounterOpts{
Namespace: namespace,
Name: "sessions_revoked_total",
Help: "Sessions revoked before expiry, by who revoked them.",
}, []string{"by"})
// AuditWriteFailuresTotal counts audit rows the API failed to write. The
// action went through; only its record was lost.
AuditWriteFailuresTotal = prometheus.NewCounter(prometheus.CounterOpts{
Namespace: namespace,
Name: "audit_write_failures_total",
Help: "Audit rows the API failed to write.",
})
// BuildInfo is 1 for the process serving it, labelled by component
// ("operator", "api") and version. Both processes register every collector,
// so this is the one series that says which of them a scrape reached: an
// alert on absent(felis_build_info{component="api"}) fires when felis-api is
// down or no longer scraped, where every other felis_* series would still be
// present from the operator.
BuildInfo = prometheus.NewGaugeVec(prometheus.GaugeOpts{
Namespace: namespace,
Name: "build_info",
Help: "1 for the Felis component serving these metrics, by component and version.",
}, []string{"component", "version"})
)
// SetBuildInfo marks this process as component at version on felis_build_info.
func SetBuildInfo(component, version string) {
BuildInfo.WithLabelValues(component, version).Set(1)
}
// OTPPurposes are the email-code doors OTPLockoutsTotal is labelled by.
var OTPPurposes = []string{"onboard_email", "login_email", "op_login", "migrate_confirm"}
// The sign-in alerts watch these counters with increase(). A labelled child
// that does not exist yet has no sample before its first event, so increase()
// would miss exactly the first lockout or throttle; every child the alerts use
// is created at zero up front.
func init() {
for _, kind := range []string{"otp", "notice"} {
for _, result := range []string{"sent", "failed", "throttled"} {
MailTotal.WithLabelValues(kind, result)
}
}
RateLimitedTotal.WithLabelValues("auth_door")
for _, p := range OTPPurposes {
OTPLockoutsTotal.WithLabelValues(p)
}
}
// SyncServerGauge republishes felis_servers_total from a full snapshot of the
// fleet's per-server states. states holds one entry per MinecraftServer the
// operator knows about (its desiredState).
//
// It Resets the GaugeVec before Setting one child per distinct state, so a state
// that has drained to zero reports 0 rather than its stale last value. That is
// the whole reason a periodic full-snapshot is used instead of inc/dec on
// reconcile transitions: a snapshot is self-correcting and cannot drift on a
// missed event. Producing the states slice (a cached List of MinecraftServers)
// is the untestable I/O edge; this Reset+tally+Set logic is pure and unit-tested.
func SyncServerGauge(states []string) {
ServersTotal.Reset()
counts := make(map[string]int, len(states))
for _, s := range states {
counts[s]++
}
for state, n := range counts {
ServersTotal.WithLabelValues(state).Set(float64(n))
}
}
// ServerPhaseSample is one server's felis_server_phase series.
type ServerPhaseSample struct {
Server, Role, Phase, Desired string
}
// SyncServerPhases republishes felis_server_phase from a full snapshot of the
// fleet, Resetting first for the same reason SyncServerGauge does.
func SyncServerPhases(samples []ServerPhaseSample) {
ServerPhase.Reset()
for _, s := range samples {
ServerPhase.WithLabelValues(s.Server, s.Role, s.Phase, s.Desired).Set(1)
}
}
// Collectors returns every felis_* collector in a stable order. Production and
// tests register the same slice, so the test asserting the full set is exposed
// also pins the production surface.
func Collectors() []prometheus.Collector {
return []prometheus.Collector{
ServersTotal,
ServerPhase,
StartDurationSeconds,
ImageBuildFailuresTotal,
ReaperWorldsDeletedTotal,
OTPLockoutsTotal,
MailTotal,
RateLimitedTotal,
AuthFailuresTotal,
SessionsRevokedTotal,
AuditWriteFailuresTotal,
BuildInfo,
}
}
// Register wires every felis_* collector into r.
//
// It is idempotent: re-registering an already-registered collector (a manager
// restart in-process, a test that calls Register twice) is tolerated rather
// than fatal, so a benign double-call never takes the operator down. Any other
// registration error is returned to the caller to surface at startup.
func Register(r prometheus.Registerer) error {
for _, c := range Collectors() {
if err := r.Register(c); err != nil {
var already prometheus.AlreadyRegisteredError
if errors.As(err, &already) {
continue
}
return err
}
}
return nil
}