feat(db): 控制面 PG 定时备份、迁移前快照与原子恢复

This commit is contained in:
Lemon-miaow committed 2026-09-24 15:19:42 +08:00
1 parent abfe60d62d
commit c7db7d4126
31 files changed
+3217 -17

No files matched your search

+27
View File
@@ -5,6 +5,7 @@
# felis_* series come from two processes:
# - felis-operator pod :8080/metrics → felis_servers_total, felis_start_duration_seconds
# - felis-api internal :8081/metrics → felis_image_build_failures_total
# - node-exporter textfile collector → felis_db_backup_* (felis-db-backup.timer)
# node_* / kube_* series come from node-exporter / kube-state-metrics.
groups:
- name: felis.rules
@@ -67,3 +68,29 @@ groups:
description: >-
PostgreSQL, the control plane, the registry and game servers share one
node; sustained memory pressure risks OOM kills.
- name: felis.backup.rules
rules:
- alert: FelisDBBackupStale
expr: time() - max(felis_db_backup_last_success_timestamp_seconds) > 26 * 3600
for: 10m
labels:
severity: critical
annotations:
summary: "no control-plane database backup in over 26h"
description: >-
felis-db-backup.timer runs daily; the newest bundle is more than a day
old. Read `journalctl -u felis-db-backup` on the host, then take one now
with `sudo felis db backup` (troubleshooting §16).
- alert: FelisDBBackupMetricMissing
expr: absent(felis_db_backup_last_success_timestamp_seconds)
for: 2h
labels:
severity: warning
annotations:
summary: "database backup freshness is not being scraped"
description: >-
No felis_db_backup_last_success_timestamp_seconds series, so
FelisDBBackupStale cannot fire. Point node-exporter's
--collector.textfile.directory at the directory of
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
(troubleshooting §16).
+46
View File
@@ -105,3 +105,49 @@ tests:
description: >-
PostgreSQL, the control plane, the registry and game servers share one
node; sustained memory pressure risks OOM kills.
- name: database backup freshness
interval: 1m
input_series:
# The newest bundle was taken at t=0 and none since.
- series: 'felis_db_backup_last_success_timestamp_seconds{instance="node1",job="node-exporter",label="daily"}'
values: '0x1630'
alert_rule_test:
- eval_time: 25h
alertname: FelisDBBackupStale
exp_alerts: []
- eval_time: 27h
alertname: FelisDBBackupStale
exp_alerts:
- exp_labels:
severity: critical
exp_annotations:
summary: "no control-plane database backup in over 26h"
description: >-
felis-db-backup.timer runs daily; the newest bundle is more than a day
old. Read `journalctl -u felis-db-backup` on the host, then take one now
with `sudo felis db backup` (troubleshooting §16).
- eval_time: 27h
alertname: FelisDBBackupMetricMissing
exp_alerts: []
- name: database backup freshness not scraped
interval: 1m
input_series:
- series: 'up{job="node-exporter"}'
values: '1x200'
alert_rule_test:
- eval_time: 1h
alertname: FelisDBBackupMetricMissing
exp_alerts: []
- eval_time: 3h
alertname: FelisDBBackupMetricMissing
exp_alerts:
- exp_labels:
severity: warning
exp_annotations:
summary: "database backup freshness is not being scraped"
description: >-
No felis_db_backup_last_success_timestamp_seconds series, so
FelisDBBackupStale cannot fire. Point node-exporter's
--collector.textfile.directory at the directory of
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
(troubleshooting §16).
+26
View File
@@ -72,3 +72,29 @@ spec:
description: >-
PostgreSQL, the control plane, the registry and game servers share one
node; sustained memory pressure risks OOM kills.
- name: felis.backup.rules
rules:
- alert: FelisDBBackupStale
expr: time() - max(felis_db_backup_last_success_timestamp_seconds) > 26 * 3600
for: 10m
labels:
severity: critical
annotations:
summary: "no control-plane database backup in over 26h"
description: >-
felis-db-backup.timer runs daily; the newest bundle is more than a day
old. Read `journalctl -u felis-db-backup` on the host, then take one now
with `sudo felis db backup` (troubleshooting §16).
- alert: FelisDBBackupMetricMissing
expr: absent(felis_db_backup_last_success_timestamp_seconds)
for: 2h
labels:
severity: warning
annotations:
summary: "database backup freshness is not being scraped"
description: >-
No felis_db_backup_last_success_timestamp_seconds series, so
FelisDBBackupStale cannot fire. Point node-exporter's
--collector.textfile.directory at the directory of
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
(troubleshooting §16).
+67 -1
View File
@@ -140,6 +140,19 @@ FELIS_ARCHIVE_LOCAL_PATH="${FELIS_ARCHIVE_LOCAL_PATH:-/var/lib/felis/archives}"
# from its volumeName. Left unset, no reaper CronJob renders and archives accumulate until
# the backup PVC fills (then backups fail loudly; nothing is deleted).
FELIS_WORLDS_HOST_PATH="${FELIS_WORLDS_HOST_PATH:-}"
# Control-plane database backups (felis db backup): a daily timer bundles pg_dump with the
# /etc/felis state a rebuild needs, and every upgrade that has migrations to apply snapshots
# the database first (felis migrate up). The directory sits outside /var/lib/rancher on
# purpose: reinstalling k3s must not take the database backups with it. Copy it off the
# host for anything beyond "undo a bad upgrade or a mistaken delete" (troubleshooting §16).
FELIS_DB_BACKUP_DIR="${FELIS_DB_BACKUP_DIR:-/var/lib/felis/db-backups}"
FELIS_DB_BACKUP_KEEP="${FELIS_DB_BACKUP_KEEP:-14}"
FELIS_DB_BACKUP_TIME="${FELIS_DB_BACKUP_TIME:-*-*-* 03:30:00}"
# node-exporter textfile collector target; FelisDBBackupStale (deploy/alerts) reads it.
FELIS_DB_BACKUP_METRICS="${FELIS_DB_BACKUP_METRICS:-/var/lib/node_exporter/textfile_collector/felis_db_backup.prom}"
# 0 migrates without the pre-migration snapshot, e.g. against an external database newer
# than this host's pg_dump. The upgrade stops if the snapshot fails and this is not set.
FELIS_PRE_MIGRATE_BACKUP="${FELIS_PRE_MIGRATE_BACKUP:-1}"
INSTALL_MODE="${FELIS_INSTALL_MODE:-}"
# Loopback by default: hasJoined is an unauthenticated endpoint by protocol (Velocity
# sends no token), so a public bind is a free auth relay — anyone can point their own
@@ -237,6 +250,8 @@ HOST_BIN="/usr/local/bin/felis"
# version sits here, and an operator's Go at the conventional path is not ours to swap.
GOROOT_DIR="/opt/felis/go"
NANO_SERVICE="/etc/systemd/system/felis-nano.service"
DB_BACKUP_SERVICE="/etc/systemd/system/felis-db-backup.service"
DB_BACKUP_TIMER="/etc/systemd/system/felis-db-backup.timer"
VELOCITY_DIR="/opt/felis/velocity"
VELOCITY_USER="felis-velocity"
VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service"
@@ -2364,13 +2379,62 @@ ensure_default_config() {
# 8. Migrate + deploy bundle
# ---------------------------------------------------------------------------
run_migrations() {
local backup_flags=(-backup-dir "$FELIS_DB_BACKUP_DIR")
write_felis_toml "${STATE_DIR}/felis.host.toml" "127.0.0.1"
ensure_default_config
if [ "$FELIS_PRE_MIGRATE_BACKUP" = 0 ]; then
warn "FELIS_PRE_MIGRATE_BACKUP=0: pending migrations run without a database snapshot"
backup_flags=(-no-backup)
fi
# Migrations only roll forward. On an existing database with migrations pending, the
# binary bundles the database into FELIS_DB_BACKUP_DIR first and refuses to migrate
# if that fails; a fresh database has nothing to protect and is migrated directly.
log "running database migrations (host binary -> 127.0.0.1)"
"$HOST_BIN" migrate up -config "${STATE_DIR}/felis.host.toml"
"$HOST_BIN" migrate up -config "${STATE_DIR}/felis.host.toml" "${backup_flags[@]}"
ok "migrations applied"
}
# The daily database backup. The first run happens now, so a broken pipeline (pg_dump
# missing, directory unwritable) shows up in this install rather than in the first
# restore someone needs.
install_db_backup_timer() {
install -d -m 0700 "$FELIS_DB_BACKUP_DIR"
cat > "$DB_BACKUP_SERVICE" <<EOF
[Unit]
Description=Felis control-plane database backup (pg_dump + /etc/felis state)
After=postgresql.service k3s.service
Wants=postgresql.service
[Service]
Type=oneshot
ExecStart=${HOST_BIN} db backup -config ${STATE_DIR}/felis.host.toml -dir ${FELIS_DB_BACKUP_DIR} -label daily -keep ${FELIS_DB_BACKUP_KEEP} -metrics-file ${FELIS_DB_BACKUP_METRICS}
Nice=10
IOSchedulingClass=idle
PrivateTmp=yes
NoNewPrivileges=yes
EOF
cat > "$DB_BACKUP_TIMER" <<EOF
[Unit]
Description=Daily Felis control-plane database backup
[Timer]
OnCalendar=${FELIS_DB_BACKUP_TIME}
RandomizedDelaySec=15min
Persistent=true
[Install]
WantedBy=timers.target
EOF
systemctl daemon-reload
systemctl enable --now felis-db-backup.timer
if systemctl start felis-db-backup.service; then
ok "database backups: daily at ${FELIS_DB_BACKUP_TIME}, newest ${FELIS_DB_BACKUP_KEEP} kept in ${FELIS_DB_BACKUP_DIR} (first one taken now)"
else
journalctl -u felis-db-backup.service -n 20 --no-pager >&2 || true
warn "the first database backup failed (log above); fix it before relying on the daily timer: sudo systemctl start felis-db-backup.service"
fi
}
deploy_bundle() {
local had_api=0 had_operator=0
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
@@ -2970,6 +3034,8 @@ main() {
# After deploy_bundle: the proxy dials felis-api's internal ClusterIP, which does not
# exist until the bundle is applied.
install_velocity
# After deploy_bundle: the bundle's MinecraftServer export reads the cluster.
install_db_backup_timer
mark_bootstrap_done
summary
}
+64
View File
@@ -1017,6 +1017,70 @@ fi
rm -f "$fnfile"
# --- database backups: the pre-migration snapshot and the daily timer --------------------
# Migrations only roll forward, so an upgrade must hand `migrate up` the snapshot directory,
# and only an explicit FELIS_PRE_MIGRATE_BACKUP=0 may take that away.
mblock="$(awk '/^run_migrations\(\) \{/,/^}/' "$BS")"
[ -n "$mblock" ] || { echo "FAIL: no run_migrations found in $BS"; exit 1; }
[ "$(printf '%s\n' "$mblock" | wc -l)" -lt 30 ] \
|| { echo "FAIL: the extracted block is not run_migrations -- did its closing brace move?"; exit 1; }
run_migrate() { # FELIS_PRE_MIGRATE_BACKUP
FELIS_PRE_MIGRATE_BACKUP="$1" FELIS_DB_BACKUP_DIR=/var/lib/felis/db-backups STATE_DIR=/etc/felis \
HOST_BIN=fakefelis bash -c '
log() { :; }; ok() { :; }; warn() { printf "WARN: %s\n" "$*"; }
write_felis_toml() { :; }; ensure_default_config() { :; }
fakefelis() { printf "RUN: %s\n" "$*"; }
'"$mblock"'
run_migrations' 2>&1
}
out="$(run_migrate 1)"
expect "an upgrade snapshots into the backup dir" "RUN: migrate up -config /etc/felis/felis.host.toml -backup-dir /var/lib/felis/db-backups" "$out"
out="$(run_migrate 0)"
expect "FELIS_PRE_MIGRATE_BACKUP=0 opts out explicitly" "RUN: migrate up -config /etc/felis/felis.host.toml -no-backup" "$out"
expect "the opt-out is loud" "WARN: FELIS_PRE_MIGRATE_BACKUP=0" "$out"
tblock="$(awk '/^install_db_backup_timer\(\) \{/,/^}/' "$BS")"
[ -n "$tblock" ] || { echo "FAIL: no install_db_backup_timer found in $BS"; exit 1; }
[ "$(printf '%s\n' "$tblock" | wc -l)" -lt 60 ] \
|| { echo "FAIL: the extracted block is not install_db_backup_timer -- did its closing brace move?"; exit 1; }
tdir="$(mktemp -d)"
run_timer() { # exit status of the first backup
FIRST="$1" DB_BACKUP_SERVICE="$tdir/felis-db-backup.service" DB_BACKUP_TIMER="$tdir/felis-db-backup.timer" \
FELIS_DB_BACKUP_DIR="$tdir/db-backups" FELIS_DB_BACKUP_KEEP=7 FELIS_DB_BACKUP_TIME='*-*-* 04:00:00' \
FELIS_DB_BACKUP_METRICS=/var/lib/node_exporter/textfile_collector/felis_db_backup.prom \
HOST_BIN=/usr/local/bin/felis STATE_DIR=/etc/felis bash -c '
ok() { printf "OK: %s\n" "$*"; }; warn() { printf "WARN: %s\n" "$*"; }
systemctl() { printf "SYSTEMCTL: %s\n" "$*"; [ "$1" != start ] || return "$FIRST"; }
journalctl() { printf "JOURNAL: pg_dump: connection refused\n"; }
'"$tblock"'
install_db_backup_timer' 2>&1
}
out="$(run_timer 0)"
unit="$(cat "$tdir/felis-db-backup.service")"
timer="$(cat "$tdir/felis-db-backup.timer")"
expect "the unit runs a daily-labelled backup with the configured retention" \
"ExecStart=/usr/local/bin/felis db backup -config /etc/felis/felis.host.toml -dir $tdir/db-backups -label daily -keep 7 -metrics-file /var/lib/node_exporter/textfile_collector/felis_db_backup.prom" "$unit"
expect "the timer fires at the configured time" "OnCalendar=*-*-* 04:00:00" "$timer"
expect "a missed run (host off at 03:30) catches up at boot" "Persistent=true" "$timer"
expect "the timer is enabled" "SYSTEMCTL: enable --now felis-db-backup.timer" "$out"
expect "the first backup runs during the install" "SYSTEMCTL: start felis-db-backup.service" "$out"
expect "a working first backup is reported" "OK: database backups: daily" "$out"
if [ "$(stat -c %a "$tdir/db-backups" 2>/dev/null || stat -f %Lp "$tdir/db-backups")" = 700 ]; then
echo "PASS the backup directory is private"
else
echo "FAIL the backup directory must be 0700"; fails=$((fails + 1))
fi
out="$(run_timer 1)"
expect "a failed first backup shows its log" "JOURNAL: pg_dump: connection refused" "$out"
expect "a failed first backup is a loud warning" "WARN: the first database backup failed" "$out"
rm -rf "$tdir"
# ---------------------------------------------------------------------------------------
if [ "$fails" -eq 0 ]; then
echo "ALL PASS"