diff --git a/deploy/alerts/felis-alerts.yaml b/deploy/alerts/felis-alerts.yaml new file mode 100644 index 0000000..c83d8dc --- /dev/null +++ b/deploy/alerts/felis-alerts.yaml @@ -0,0 +1,69 @@ +# Felis alert rules — plain Prometheus format (also the promtool-tested source +# for felis-prometheusrule.yaml). See docs/troubleshooting.md §14 for scraping +# and loading instructions. +# +# felis_* series come from two processes: +# - felis-operator pod :8080/metrics → felis_servers_total, felis_start_duration_seconds +# - felis-api internal :8081/metrics → felis_image_build_failures_total +# node_* / kube_* series come from node-exporter / kube-state-metrics. +groups: + - name: felis.rules + rules: + - alert: FelisImageBuildFailures + expr: increase(felis_image_build_failures_total[6h]) > 0 + for: 5m + labels: + severity: warning + annotations: + summary: "modpack/image build failed in the last 6h" + description: >- + felis_image_build_failures_total increased. Inspect the failed build Job + (kubectl logs -n felis-build job/); the same error text is on + GET /api/v1/images/build/{id} and in the submitter's row in the panel. + - alert: FelisSlowServerStarts + expr: histogram_quantile(0.9, sum by (le) (rate(felis_start_duration_seconds_bucket[30m]))) > 300 + for: 15m + labels: + severity: warning + annotations: + summary: "p90 server start time exceeds 5 minutes" + description: >- + Starts regularly take over five minutes (felis_start_duration_seconds, + observed when readiness is first reached). A start that never completes + records nothing — cross-check desiredState=Running servers with no ready + phase (troubleshooting §1). + - name: felis.node.rules + rules: + - alert: FelisNodeDiskSpaceLow + expr: >- + node_filesystem_avail_bytes{fstype=~"ext4|xfs|btrfs"} + / node_filesystem_size_bytes{fstype=~"ext4|xfs|btrfs"} < 0.15 + for: 15m + labels: + severity: warning + annotations: + summary: "node filesystem {{ $labels.mountpoint }} below 15% available" + description: >- + Sustained disk pressure evicts game pods and garbage-collects images + (troubleshooting §13b). Free space before kubelet raises DiskPressure. + - alert: FelisNodeDiskPressure + expr: kube_node_status_condition{condition="DiskPressure",status="true"} == 1 + for: 5m + labels: + severity: critical + annotations: + summary: "kubelet reports DiskPressure on {{ $labels.node }}" + description: >- + The eviction chain is in progress: control-plane pods hold + system-cluster-critical and survive, game pods do not. Free disk now + (troubleshooting §13b). + - alert: FelisNodeMemoryLow + expr: node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes < 0.10 + for: 15m + labels: + severity: warning + annotations: + summary: "node memory available below 10% for 15m" + description: >- + PostgreSQL, the control plane, the registry and game servers share one + node; sustained memory pressure risks OOM kills. diff --git a/deploy/alerts/felis-alerts_test.yml b/deploy/alerts/felis-alerts_test.yml new file mode 100644 index 0000000..d15e00e --- /dev/null +++ b/deploy/alerts/felis-alerts_test.yml @@ -0,0 +1,107 @@ +# promtool unit tests: `promtool test rules felis-alerts_test.yml` +# Proves every shipped rule actually fires on its target condition (and stays +# silent before it). +rule_files: + - felis-alerts.yaml +evaluation_interval: 1m +tests: + - name: build failure and slow starts + interval: 1m + input_series: + # counter: quiet for 5m, then one failure per step. + - series: 'felis_image_build_failures_total' + values: '0x5 1x15' + # histogram: all observations land in the (300,600] bucket. + - series: 'felis_start_duration_seconds_bucket{le="120"}' + values: '0x22' + - series: 'felis_start_duration_seconds_bucket{le="300"}' + values: '0x22' + - series: 'felis_start_duration_seconds_bucket{le="600"}' + values: '0+10x21' + - series: 'felis_start_duration_seconds_bucket{le="+Inf"}' + values: '0+10x21' + alert_rule_test: + - eval_time: 2m + alertname: FelisImageBuildFailures + exp_alerts: [] + - eval_time: 20m + alertname: FelisImageBuildFailures + exp_alerts: + - exp_labels: + severity: warning + exp_annotations: + summary: "modpack/image build failed in the last 6h" + description: >- + felis_image_build_failures_total increased. Inspect the failed build Job + (kubectl logs -n felis-build job/); the same error text is on + GET /api/v1/images/build/{id} and in the submitter's row in the panel. + - eval_time: 20m + alertname: FelisSlowServerStarts + exp_alerts: + - exp_labels: + severity: warning + exp_annotations: + summary: "p90 server start time exceeds 5 minutes" + description: >- + Starts regularly take over five minutes (felis_start_duration_seconds, + observed when readiness is first reached). A start that never completes + records nothing — cross-check desiredState=Running servers with no ready + phase (troubleshooting §1). + - name: node disk and memory thresholds + interval: 1m + input_series: + - series: 'node_filesystem_avail_bytes{device="/dev/vda1",fstype="xfs",instance="node1",job="node-exporter",mountpoint="/"}' + values: '10x26' + - series: 'node_filesystem_size_bytes{device="/dev/vda1",fstype="xfs",instance="node1",job="node-exporter",mountpoint="/"}' + values: '100x26' + - series: 'kube_node_status_condition{condition="DiskPressure",node="n1",status="true"}' + values: '0x4 1x22' + - series: 'node_memory_MemAvailable_bytes{instance="node1",job="node-exporter"}' + values: '5x26' + - series: 'node_memory_MemTotal_bytes{instance="node1",job="node-exporter"}' + values: '100x26' + alert_rule_test: + - eval_time: 2m + alertname: FelisNodeDiskPressure + exp_alerts: [] + - eval_time: 20m + alertname: FelisNodeDiskSpaceLow + exp_alerts: + - exp_labels: + device: /dev/vda1 + fstype: xfs + instance: node1 + job: node-exporter + mountpoint: / + severity: warning + exp_annotations: + summary: "node filesystem / below 15% available" + description: >- + Sustained disk pressure evicts game pods and garbage-collects images + (troubleshooting §13b). Free space before kubelet raises DiskPressure. + - eval_time: 20m + alertname: FelisNodeDiskPressure + exp_alerts: + - exp_labels: + condition: DiskPressure + node: n1 + status: "true" + severity: critical + exp_annotations: + summary: "kubelet reports DiskPressure on n1" + description: >- + The eviction chain is in progress: control-plane pods hold + system-cluster-critical and survive, game pods do not. Free disk now + (troubleshooting §13b). + - eval_time: 20m + alertname: FelisNodeMemoryLow + exp_alerts: + - exp_labels: + instance: node1 + job: node-exporter + severity: warning + exp_annotations: + summary: "node memory available below 10% for 15m" + description: >- + PostgreSQL, the control plane, the registry and game servers share one + node; sustained memory pressure risks OOM kills. diff --git a/deploy/alerts/felis-prometheusrule.yaml b/deploy/alerts/felis-prometheusrule.yaml new file mode 100644 index 0000000..6967395 --- /dev/null +++ b/deploy/alerts/felis-prometheusrule.yaml @@ -0,0 +1,74 @@ +# prometheus-operator twin of felis-alerts.yaml (kube-prometheus-stack loads +# rules through the PrometheusRule CRD, not rule_files). The plain file is the +# promtool-tested source; keep the groups in sync. +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + name: felis-alerts + namespace: monitoring + labels: + # Change to match your stack's ruleSelector (kube-prometheus-stack's + # default selects on the Helm release name). + release: kube-prometheus-stack +spec: + groups: + - name: felis.rules + rules: + - alert: FelisImageBuildFailures + expr: increase(felis_image_build_failures_total[6h]) > 0 + for: 5m + labels: + severity: warning + annotations: + summary: "modpack/image build failed in the last 6h" + description: >- + felis_image_build_failures_total increased. Inspect the failed build Job + (kubectl logs -n felis-build job/); the same error text is on + GET /api/v1/images/build/{id} and in the submitter's row in the panel. + - alert: FelisSlowServerStarts + expr: histogram_quantile(0.9, sum by (le) (rate(felis_start_duration_seconds_bucket[30m]))) > 300 + for: 15m + labels: + severity: warning + annotations: + summary: "p90 server start time exceeds 5 minutes" + description: >- + Starts regularly take over five minutes (felis_start_duration_seconds, + observed when readiness is first reached). A start that never completes + records nothing — cross-check desiredState=Running servers with no ready + phase (troubleshooting §1). + - name: felis.node.rules + rules: + - alert: FelisNodeDiskSpaceLow + expr: >- + node_filesystem_avail_bytes{fstype=~"ext4|xfs|btrfs"} + / node_filesystem_size_bytes{fstype=~"ext4|xfs|btrfs"} < 0.15 + for: 15m + labels: + severity: warning + annotations: + summary: "node filesystem {{ $labels.mountpoint }} below 15% available" + description: >- + Sustained disk pressure evicts game pods and garbage-collects images + (troubleshooting §13b). Free space before kubelet raises DiskPressure. + - alert: FelisNodeDiskPressure + expr: kube_node_status_condition{condition="DiskPressure",status="true"} == 1 + for: 5m + labels: + severity: critical + annotations: + summary: "kubelet reports DiskPressure on {{ $labels.node }}" + description: >- + The eviction chain is in progress: control-plane pods hold + system-cluster-critical and survive, game pods do not. Free disk now + (troubleshooting §13b). + - alert: FelisNodeMemoryLow + expr: node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes < 0.10 + for: 15m + labels: + severity: warning + annotations: + summary: "node memory available below 10% for 15m" + description: >- + PostgreSQL, the control plane, the registry and game servers share one + node; sustained memory pressure risks OOM kills. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 9bdf334..26e9e80 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -762,6 +762,34 @@ All four mandated metrics have real producers; scrape them when triaging: actually deleted post-backup (§10); a spike here means worlds crossed the 15d idle line — cross-check that join events are flowing (§10 risk vectors). +### Scraping + +The series come from two processes: + +- `felis-operator` pod `:8080/metrics` — `felis_servers_total`, + `felis_start_duration_seconds` (no Service; scrape pod-scoped, e.g. a + PodMonitor targeting port `metrics`). +- `felis-api` internal face `:8081/metrics` (Service `felis-api-internal`) — + `felis_image_build_failures_total`. Unauthenticated like the probes; + ClusterIP-only, and the external face never serves it. +- `felis_reaper_worlds_deleted_total` is produced inside the one-shot reaper + CronJob, which exits long before any scrape interval — without a pushgateway + it has no scrape path. Read the reaper Pod log or the `world_backups` table + for deletions instead. + +### Alert rules + +`deploy/alerts/` ships ready-made rules: build failures, slow starts, node +disk/memory thresholds, and the kubelet `DiskPressure` condition. + +- Plain Prometheus: add `felis-alerts.yaml` to `rule_files`. Check and unit-test + it standalone with `promtool check rules felis-alerts.yaml` and + `promtool test rules felis-alerts_test.yml` (the tests pin exactly when each + alert fires). +- kube-prometheus-stack / prometheus-operator: `kubectl apply -f + felis-prometheusrule.yaml` (adjust its `release:` label to your stack's + ruleSelector). + --- ## 15. Control-plane upgrades, and rolling back a bad one