510 lines
21 KiB
YAML
510 lines
21 KiB
YAML
# promtool unit tests: `promtool test rules felis-alerts_test.yml`
|
|
# Proves every shipped rule actually fires on its target condition (and stays
|
|
# silent before it).
|
|
rule_files:
|
|
- felis-alerts.yaml
|
|
evaluation_interval: 1m
|
|
tests:
|
|
- name: build failure and slow starts
|
|
interval: 1m
|
|
input_series:
|
|
# counter: quiet for 5m, then one failure per step.
|
|
- series: 'felis_image_build_failures_total'
|
|
values: '0x5 1x15'
|
|
# histogram: all observations land in the (300,600] bucket.
|
|
- series: 'felis_start_duration_seconds_bucket{le="120"}'
|
|
values: '0x22'
|
|
- series: 'felis_start_duration_seconds_bucket{le="300"}'
|
|
values: '0x22'
|
|
- series: 'felis_start_duration_seconds_bucket{le="600"}'
|
|
values: '0+10x21'
|
|
- series: 'felis_start_duration_seconds_bucket{le="+Inf"}'
|
|
values: '0+10x21'
|
|
alert_rule_test:
|
|
- eval_time: 2m
|
|
alertname: FelisImageBuildFailures
|
|
exp_alerts: []
|
|
- eval_time: 20m
|
|
alertname: FelisImageBuildFailures
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "modpack/image build failed in the last 6h"
|
|
description: >-
|
|
felis_image_build_failures_total increased. Inspect the failed build Job
|
|
(kubectl logs -n felis-build job/<build-job>); the same error text is on
|
|
GET /api/v1/images/build/{id} and in the submitter's row in the panel.
|
|
- eval_time: 20m
|
|
alertname: FelisSlowServerStarts
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "p90 server start time exceeds 5 minutes"
|
|
description: >-
|
|
Starts regularly take over five minutes (felis_start_duration_seconds,
|
|
observed when readiness is first reached). A start that never completes
|
|
records nothing — cross-check desiredState=Running servers with no ready
|
|
phase (troubleshooting §1).
|
|
- name: node disk and memory thresholds
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'node_filesystem_avail_bytes{device="/dev/vda1",fstype="xfs",instance="node1",job="node-exporter",mountpoint="/"}'
|
|
values: '10x26'
|
|
- series: 'node_filesystem_size_bytes{device="/dev/vda1",fstype="xfs",instance="node1",job="node-exporter",mountpoint="/"}'
|
|
values: '100x26'
|
|
- series: 'kube_node_status_condition{condition="DiskPressure",node="n1",status="true"}'
|
|
values: '0x4 1x22'
|
|
- series: 'node_memory_MemAvailable_bytes{instance="node1",job="node-exporter"}'
|
|
values: '5x26'
|
|
- series: 'node_memory_MemTotal_bytes{instance="node1",job="node-exporter"}'
|
|
values: '100x26'
|
|
alert_rule_test:
|
|
- eval_time: 2m
|
|
alertname: FelisNodeDiskPressure
|
|
exp_alerts: []
|
|
- eval_time: 20m
|
|
alertname: FelisNodeDiskSpaceLow
|
|
exp_alerts:
|
|
- exp_labels:
|
|
device: /dev/vda1
|
|
fstype: xfs
|
|
instance: node1
|
|
job: node-exporter
|
|
mountpoint: /
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "node filesystem / below 15% available"
|
|
description: >-
|
|
Sustained disk pressure evicts game pods and garbage-collects images
|
|
(troubleshooting §13b). Free space before kubelet raises DiskPressure.
|
|
- eval_time: 20m
|
|
alertname: FelisNodeDiskPressure
|
|
exp_alerts:
|
|
- exp_labels:
|
|
condition: DiskPressure
|
|
node: n1
|
|
status: "true"
|
|
severity: critical
|
|
exp_annotations:
|
|
summary: "kubelet reports DiskPressure on n1"
|
|
description: >-
|
|
The eviction chain is in progress: control-plane pods hold
|
|
system-cluster-critical and survive, game pods do not. Free disk now
|
|
(troubleshooting §13b).
|
|
- eval_time: 20m
|
|
alertname: FelisNodeMemoryLow
|
|
exp_alerts:
|
|
- exp_labels:
|
|
instance: node1
|
|
job: node-exporter
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "node memory available below 10% for 15m"
|
|
description: >-
|
|
PostgreSQL, the control plane, the registry and game servers share one
|
|
node; sustained memory pressure risks OOM kills.
|
|
- name: database backup freshness
|
|
interval: 1m
|
|
input_series:
|
|
# The newest bundle was taken at t=0 and none since.
|
|
- series: 'felis_db_backup_last_success_timestamp_seconds{instance="node1",job="node-exporter",label="daily"}'
|
|
values: '0x1630'
|
|
alert_rule_test:
|
|
- eval_time: 25h
|
|
alertname: FelisDBBackupStale
|
|
exp_alerts: []
|
|
- eval_time: 27h
|
|
alertname: FelisDBBackupStale
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: critical
|
|
exp_annotations:
|
|
summary: "no control-plane database backup in over 26h"
|
|
description: >-
|
|
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
|
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
|
with `sudo felis db backup` (troubleshooting §16).
|
|
- eval_time: 27h
|
|
alertname: FelisDBBackupMetricMissing
|
|
exp_alerts: []
|
|
- name: database backup freshness not scraped
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'up{job="node-exporter"}'
|
|
values: '1x200'
|
|
alert_rule_test:
|
|
- eval_time: 1h
|
|
alertname: FelisDBBackupMetricMissing
|
|
exp_alerts: []
|
|
- eval_time: 3h
|
|
alertname: FelisDBBackupMetricMissing
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "database backup freshness is not being scraped"
|
|
description: >-
|
|
No felis_db_backup_last_success_timestamp_seconds series, so
|
|
FelisDBBackupStale cannot fire. Point node-exporter's
|
|
--collector.textfile.directory at the directory of
|
|
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
|
(troubleshooting §16).
|
|
- name: sign-in mail budget and relay
|
|
interval: 1m
|
|
input_series:
|
|
# Created at zero on start; the budget refuses one mail at t=3m.
|
|
- series: 'felis_mail_total{kind="otp",result="throttled",job="felis-api"}'
|
|
values: '0 0 0 1x30'
|
|
- series: 'felis_mail_total{kind="otp",result="failed",job="felis-api"}'
|
|
values: '0x33'
|
|
alert_rule_test:
|
|
- eval_time: 2m
|
|
alertname: FelisMailBudgetExhausted
|
|
exp_alerts: []
|
|
- eval_time: 5m
|
|
alertname: FelisMailBudgetExhausted
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "the install-wide mail budget refused mail"
|
|
description: >-
|
|
felis_mail_total{result="throttled"} increased: [smtp] max_per_hour is
|
|
spent, and every sign-in code is refused with 429 mail_rate_limited until
|
|
it refills. Check felis_rate_limited_total for a flood before raising the
|
|
budget (troubleshooting §17).
|
|
- eval_time: 5m
|
|
alertname: FelisMailDeliveryFailing
|
|
exp_alerts: []
|
|
- name: sign-in flood
|
|
interval: 1m
|
|
input_series:
|
|
# 30 refusals a minute from t=0; a lone refused script at 2/min stays quiet.
|
|
- series: 'felis_rate_limited_total{scope="auth_door",job="felis-api"}'
|
|
values: '0+30x40'
|
|
alert_rule_test:
|
|
- eval_time: 10m
|
|
alertname: FelisSignInFlood
|
|
exp_alerts: []
|
|
- eval_time: 20m
|
|
alertname: FelisSignInFlood
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "sign-in doors refusing over 10 requests a minute"
|
|
description: >-
|
|
The per-address sign-in limit has been refusing callers for 10 minutes.
|
|
A script is hammering the auth doors; if real users report rate_limited
|
|
at once instead, [auth] client_ip_header is missing and everyone shares
|
|
the proxy's address (troubleshooting §17).
|
|
- name: sign-in trickle stays quiet
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_rate_limited_total{scope="auth_door",job="felis-api"}'
|
|
values: '0+2x40'
|
|
alert_rule_test:
|
|
- eval_time: 30m
|
|
alertname: FelisSignInFlood
|
|
exp_alerts: []
|
|
- name: account email-code lock
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_auth_otp_lockouts_total{purpose="login_email",job="felis-api"}'
|
|
values: '0 0 1x90'
|
|
- series: 'felis_auth_otp_lockouts_total{purpose="op_login",job="felis-api"}'
|
|
values: '0x92'
|
|
alert_rule_test:
|
|
- eval_time: 1m
|
|
alertname: FelisOTPAccountLocked
|
|
exp_alerts: []
|
|
- eval_time: 10m
|
|
alertname: FelisOTPAccountLocked
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
purpose: login_email
|
|
exp_annotations:
|
|
summary: "an account's email-code sign-in locked after 10 wrong codes"
|
|
description: >-
|
|
Someone entered 10 wrong codes for one account within 24h (login_email).
|
|
The audit log names the account (action auth.otp.locked); the owner was
|
|
mailed. Unless they fumbled codes, someone is guessing at it
|
|
(troubleshooting §17).
|
|
- eval_time: 90m
|
|
alertname: FelisOTPAccountLocked
|
|
exp_alerts: []
|
|
- name: sign-in failure rate
|
|
interval: 1m
|
|
input_series:
|
|
# Two doors failing at 3/min between them from t=0.
|
|
- series: 'felis_auth_failures_total{door="login_email",reason="bad_code",job="felis-api"}'
|
|
values: '0+2x40'
|
|
- series: 'felis_auth_failures_total{door="op_login",reason="no_account",job="felis-api"}'
|
|
values: '0+1x40'
|
|
alert_rule_test:
|
|
- eval_time: 8m
|
|
alertname: FelisSignInFailures
|
|
exp_alerts: []
|
|
- eval_time: 25m
|
|
alertname: FelisSignInFailures
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "over 30 refused sign-ins in 15 minutes"
|
|
description: >-
|
|
Wrong codes, unknown addresses or bad passkey assertions well above people
|
|
mistyping: someone is guessing or enumerating. `sum by (door, reason)
|
|
(increase(felis_auth_failures_total[15m]))` shows where; the audit rows
|
|
(action auth.<door>.failed) carry each caller's client_ip (troubleshooting §17).
|
|
- name: people mistyping stays quiet
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_auth_failures_total{door="login_email",reason="bad_code",job="felis-api"}'
|
|
values: '0 0 1 1 2 2 3 3 4 4 5x30'
|
|
alert_rule_test:
|
|
- eval_time: 30m
|
|
alertname: FelisSignInFailures
|
|
exp_alerts: []
|
|
- name: audit rows lost
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_audit_write_failures_total{job="felis-api",instance="api-0"}'
|
|
values: '0 0 0 2x20'
|
|
alert_rule_test:
|
|
- eval_time: 2m
|
|
alertname: FelisAuditWriteFailing
|
|
exp_alerts: []
|
|
- eval_time: 5m
|
|
alertname: FelisAuditWriteFailing
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
job: felis-api
|
|
instance: api-0
|
|
exp_annotations:
|
|
summary: "felis-api failed to write audit rows"
|
|
description: >-
|
|
The actions went through but their audit rows were lost. The felis-api log
|
|
names each lost row (`audit: lost ...`); the usual cause is PostgreSQL
|
|
being unreachable or out of disk.
|
|
- name: operator and api presence
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_build_info{component="operator",version="v1",job="felis-operator",instance="op-0"}'
|
|
values: '1x30'
|
|
# felis-api stops being scraped after 5m; the series goes stale 5m later.
|
|
- series: 'felis_build_info{component="api",version="v1",job="felis-api",instance="api-0"}'
|
|
values: '1x5'
|
|
alert_rule_test:
|
|
- eval_time: 25m
|
|
alertname: FelisOperatorDown
|
|
exp_alerts: []
|
|
- eval_time: 15m
|
|
alertname: FelisAPIDown
|
|
exp_alerts: []
|
|
- eval_time: 25m
|
|
alertname: FelisAPIDown
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: critical
|
|
component: api
|
|
exp_annotations:
|
|
summary: "felis-api is down or not scraped"
|
|
description: >-
|
|
No felis_build_info{component="api"} series for 10 minutes. The panel,
|
|
sign-in and the proxy's player lookups all go through felis-api. Check
|
|
`kubectl -n felis get deploy felis-api` and its log; if the pod is
|
|
healthy, the felis-api-internal Service is not being scraped
|
|
(troubleshooting §14).
|
|
- name: operator never scraped
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_build_info{component="api",version="v1",job="felis-api",instance="api-0"}'
|
|
values: '1x30'
|
|
alert_rule_test:
|
|
- eval_time: 5m
|
|
alertname: FelisOperatorDown
|
|
exp_alerts: []
|
|
- eval_time: 15m
|
|
alertname: FelisOperatorDown
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: critical
|
|
component: operator
|
|
exp_annotations:
|
|
summary: "felis-operator is down or not scraped"
|
|
description: >-
|
|
No felis_build_info{component="operator"} series for 10 minutes. Without
|
|
the operator no server starts, stops or recovers. Check
|
|
`kubectl -n felis get deploy felis-operator` and its log; if the pod is
|
|
healthy, the felis-operator-metrics Service is not being scraped
|
|
(troubleshooting §14).
|
|
- name: system and user servers down
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'felis_server_phase{server="login",role="login",phase="Starting",desired="Running"}'
|
|
values: '1x20'
|
|
- series: 'felis_server_phase{server="lobby",role="lobby",phase="Failed",desired="Running"}'
|
|
values: '1x20'
|
|
# Stopped on purpose: not an outage.
|
|
- series: 'felis_server_phase{server="lobby2",role="lobby",phase="Stopped",desired="Stopped"}'
|
|
values: '1x20'
|
|
# A user server carries no role label (the operator publishes role="").
|
|
- series: 'felis_server_phase{server="survival",phase="Failed",desired="Running"}'
|
|
values: '1x20'
|
|
- series: 'felis_server_phase{server="creative",phase="Running",desired="Running"}'
|
|
values: '1x20'
|
|
alert_rule_test:
|
|
- eval_time: 9m
|
|
alertname: FelisLoginGateDown
|
|
exp_alerts: []
|
|
- eval_time: 11m
|
|
alertname: FelisLoginGateDown
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: critical
|
|
server: login
|
|
role: login
|
|
phase: Starting
|
|
desired: Running
|
|
exp_annotations:
|
|
summary: "the login gate login is Starting"
|
|
description: >-
|
|
Every player connection passes through the login server first, so no one
|
|
can join. The MinecraftServer's conditions carry the reason:
|
|
`kubectl -n minecraft describe minecraftserver login`
|
|
(troubleshooting §1, §2).
|
|
- eval_time: 11m
|
|
alertname: FelisSystemServerDown
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
server: lobby
|
|
role: lobby
|
|
phase: Failed
|
|
desired: Running
|
|
exp_annotations:
|
|
summary: "system server lobby (lobby) is Failed"
|
|
description: >-
|
|
Players who sign in are sent to the lobby; while it is down they stay at
|
|
the gate. `kubectl -n minecraft describe minecraftserver lobby`
|
|
shows the reason (troubleshooting §1, §2).
|
|
- eval_time: 3m
|
|
alertname: FelisServerFailed
|
|
exp_alerts: []
|
|
- eval_time: 6m
|
|
alertname: FelisServerFailed
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
server: survival
|
|
phase: Failed
|
|
desired: Running
|
|
exp_annotations:
|
|
summary: "server survival is Failed"
|
|
description: >-
|
|
The operator gave up on this server (a crash loop, an image that will not
|
|
pull, a world volume that will not mount). Its conditions carry the
|
|
reason: `kubectl -n minecraft describe minecraftserver survival`
|
|
(troubleshooting §2).
|
|
- name: operator reconcile errors and a stuck pass
|
|
interval: 1m
|
|
input_series:
|
|
# Two failed reconciles a minute from 6m on.
|
|
- series: 'controller_runtime_reconcile_errors_total{controller="minecraftserver",job="felis-operator"}'
|
|
values: '0x5 0+2x20'
|
|
# One pass that started at 5m and never returns.
|
|
- series: 'workqueue_longest_running_processor_seconds{name="minecraftserver",controller="minecraftserver",job="felis-operator"}'
|
|
values: '0x5 60+60x20'
|
|
alert_rule_test:
|
|
- eval_time: 5m
|
|
alertname: FelisReconcileErrors
|
|
exp_alerts: []
|
|
- eval_time: 25m
|
|
alertname: FelisReconcileErrors
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
exp_annotations:
|
|
summary: "the operator failed over 10 reconciles in 15 minutes"
|
|
description: >-
|
|
Server changes are being retried instead of applied. The felis-operator
|
|
log names each failing server and its error.
|
|
- eval_time: 12m
|
|
alertname: FelisReconcileStuck
|
|
exp_alerts: []
|
|
- eval_time: 20m
|
|
alertname: FelisReconcileStuck
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: critical
|
|
exp_annotations:
|
|
summary: "an operator reconcile has been running for over 5 minutes"
|
|
description: >-
|
|
Each reconcile is bounded at 3 minutes, so this one is ignoring its
|
|
deadline and holding a worker. The liveness probe restarts the operator
|
|
once a pass passes 10 minutes; the log from before the restart shows
|
|
where it hung.
|
|
- name: world job failures and a late reaper
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'kube_job_failed{namespace="minecraft",job_name="backup-survival-abc",condition="true"}'
|
|
values: '0x2 1x10'
|
|
- series: 'kube_job_failed{namespace="minecraft",job_name="backup-survival-abc",condition="false"}'
|
|
values: '1x2 0x10'
|
|
# A build Job in another namespace is FelisImageBuildFailures' business.
|
|
- series: 'kube_job_failed{namespace="felis-build",job_name="build-x",condition="true"}'
|
|
values: '1x12'
|
|
# Evaluation starts at the epoch, so "over a day ago" is a negative timestamp.
|
|
- series: 'kube_cronjob_status_last_successful_time{namespace="minecraft",cronjob="felis-reaper"}'
|
|
values: '-100000x30'
|
|
alert_rule_test:
|
|
- eval_time: 1m
|
|
alertname: FelisWorldJobFailed
|
|
exp_alerts: []
|
|
- eval_time: 5m
|
|
alertname: FelisWorldJobFailed
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
namespace: minecraft
|
|
job_name: backup-survival-abc
|
|
condition: "true"
|
|
exp_annotations:
|
|
summary: "Job backup-survival-abc failed"
|
|
description: >-
|
|
A world backup, restore or reaper run failed; after a failed backup that
|
|
world's newest archive is older than planned.
|
|
`kubectl -n minecraft logs job/backup-survival-abc` has the error
|
|
(troubleshooting §10).
|
|
- eval_time: 5m
|
|
alertname: FelisReaperStale
|
|
exp_alerts: []
|
|
- eval_time: 15m
|
|
alertname: FelisReaperStale
|
|
exp_alerts:
|
|
- exp_labels:
|
|
severity: warning
|
|
namespace: minecraft
|
|
cronjob: felis-reaper
|
|
exp_annotations:
|
|
summary: "the world reaper has not succeeded in over 26h"
|
|
description: >-
|
|
felis-reaper runs daily; idle worlds are neither backed up nor reclaimed
|
|
while it fails. `kubectl -n minecraft get jobs --sort-by=.metadata.creationTimestamp`
|
|
lists its runs, and the newest one's log
|
|
shows why (troubleshooting §10).
|
|
- name: a reaper that ran yesterday stays quiet
|
|
interval: 1m
|
|
input_series:
|
|
- series: 'kube_cronjob_status_last_successful_time{namespace="minecraft",cronjob="felis-reaper"}'
|
|
values: '-50000x30'
|
|
alert_rule_test:
|
|
- eval_time: 25m
|
|
alertname: FelisReaperStale
|
|
exp_alerts: []
|