feat(db): 控制面 PG 定时备份、迁移前快照与原子恢复
This commit is contained in:
31 files changed
+3217
-17
No files matched your search
@@ -5,6 +5,7 @@
|
||||
# felis_* series come from two processes:
|
||||
# - felis-operator pod :8080/metrics → felis_servers_total, felis_start_duration_seconds
|
||||
# - felis-api internal :8081/metrics → felis_image_build_failures_total
|
||||
# - node-exporter textfile collector → felis_db_backup_* (felis-db-backup.timer)
|
||||
# node_* / kube_* series come from node-exporter / kube-state-metrics.
|
||||
groups:
|
||||
- name: felis.rules
|
||||
@@ -67,3 +68,29 @@ groups:
|
||||
description: >-
|
||||
PostgreSQL, the control plane, the registry and game servers share one
|
||||
node; sustained memory pressure risks OOM kills.
|
||||
- name: felis.backup.rules
|
||||
rules:
|
||||
- alert: FelisDBBackupStale
|
||||
expr: time() - max(felis_db_backup_last_success_timestamp_seconds) > 26 * 3600
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "no control-plane database backup in over 26h"
|
||||
description: >-
|
||||
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
||||
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
||||
with `sudo felis db backup` (troubleshooting §16).
|
||||
- alert: FelisDBBackupMetricMissing
|
||||
expr: absent(felis_db_backup_last_success_timestamp_seconds)
|
||||
for: 2h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "database backup freshness is not being scraped"
|
||||
description: >-
|
||||
No felis_db_backup_last_success_timestamp_seconds series, so
|
||||
FelisDBBackupStale cannot fire. Point node-exporter's
|
||||
--collector.textfile.directory at the directory of
|
||||
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
||||
(troubleshooting §16).
|
||||
@@ -105,3 +105,49 @@ tests:
|
||||
description: >-
|
||||
PostgreSQL, the control plane, the registry and game servers share one
|
||||
node; sustained memory pressure risks OOM kills.
|
||||
- name: database backup freshness
|
||||
interval: 1m
|
||||
input_series:
|
||||
# The newest bundle was taken at t=0 and none since.
|
||||
- series: 'felis_db_backup_last_success_timestamp_seconds{instance="node1",job="node-exporter",label="daily"}'
|
||||
values: '0x1630'
|
||||
alert_rule_test:
|
||||
- eval_time: 25h
|
||||
alertname: FelisDBBackupStale
|
||||
exp_alerts: []
|
||||
- eval_time: 27h
|
||||
alertname: FelisDBBackupStale
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
severity: critical
|
||||
exp_annotations:
|
||||
summary: "no control-plane database backup in over 26h"
|
||||
description: >-
|
||||
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
||||
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
||||
with `sudo felis db backup` (troubleshooting §16).
|
||||
- eval_time: 27h
|
||||
alertname: FelisDBBackupMetricMissing
|
||||
exp_alerts: []
|
||||
- name: database backup freshness not scraped
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'up{job="node-exporter"}'
|
||||
values: '1x200'
|
||||
alert_rule_test:
|
||||
- eval_time: 1h
|
||||
alertname: FelisDBBackupMetricMissing
|
||||
exp_alerts: []
|
||||
- eval_time: 3h
|
||||
alertname: FelisDBBackupMetricMissing
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: "database backup freshness is not being scraped"
|
||||
description: >-
|
||||
No felis_db_backup_last_success_timestamp_seconds series, so
|
||||
FelisDBBackupStale cannot fire. Point node-exporter's
|
||||
--collector.textfile.directory at the directory of
|
||||
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
||||
(troubleshooting §16).
|
||||
@@ -72,3 +72,29 @@ spec:
|
||||
description: >-
|
||||
PostgreSQL, the control plane, the registry and game servers share one
|
||||
node; sustained memory pressure risks OOM kills.
|
||||
- name: felis.backup.rules
|
||||
rules:
|
||||
- alert: FelisDBBackupStale
|
||||
expr: time() - max(felis_db_backup_last_success_timestamp_seconds) > 26 * 3600
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "no control-plane database backup in over 26h"
|
||||
description: >-
|
||||
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
||||
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
||||
with `sudo felis db backup` (troubleshooting §16).
|
||||
- alert: FelisDBBackupMetricMissing
|
||||
expr: absent(felis_db_backup_last_success_timestamp_seconds)
|
||||
for: 2h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "database backup freshness is not being scraped"
|
||||
description: >-
|
||||
No felis_db_backup_last_success_timestamp_seconds series, so
|
||||
FelisDBBackupStale cannot fire. Point node-exporter's
|
||||
--collector.textfile.directory at the directory of
|
||||
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
||||
(troubleshooting §16).
|
||||
+67
-1
@@ -140,6 +140,19 @@ FELIS_ARCHIVE_LOCAL_PATH="${FELIS_ARCHIVE_LOCAL_PATH:-/var/lib/felis/archives}"
|
||||
# from its volumeName. Left unset, no reaper CronJob renders and archives accumulate until
|
||||
# the backup PVC fills (then backups fail loudly; nothing is deleted).
|
||||
FELIS_WORLDS_HOST_PATH="${FELIS_WORLDS_HOST_PATH:-}"
|
||||
# Control-plane database backups (felis db backup): a daily timer bundles pg_dump with the
|
||||
# /etc/felis state a rebuild needs, and every upgrade that has migrations to apply snapshots
|
||||
# the database first (felis migrate up). The directory sits outside /var/lib/rancher on
|
||||
# purpose: reinstalling k3s must not take the database backups with it. Copy it off the
|
||||
# host for anything beyond "undo a bad upgrade or a mistaken delete" (troubleshooting §16).
|
||||
FELIS_DB_BACKUP_DIR="${FELIS_DB_BACKUP_DIR:-/var/lib/felis/db-backups}"
|
||||
FELIS_DB_BACKUP_KEEP="${FELIS_DB_BACKUP_KEEP:-14}"
|
||||
FELIS_DB_BACKUP_TIME="${FELIS_DB_BACKUP_TIME:-*-*-* 03:30:00}"
|
||||
# node-exporter textfile collector target; FelisDBBackupStale (deploy/alerts) reads it.
|
||||
FELIS_DB_BACKUP_METRICS="${FELIS_DB_BACKUP_METRICS:-/var/lib/node_exporter/textfile_collector/felis_db_backup.prom}"
|
||||
# 0 migrates without the pre-migration snapshot, e.g. against an external database newer
|
||||
# than this host's pg_dump. The upgrade stops if the snapshot fails and this is not set.
|
||||
FELIS_PRE_MIGRATE_BACKUP="${FELIS_PRE_MIGRATE_BACKUP:-1}"
|
||||
INSTALL_MODE="${FELIS_INSTALL_MODE:-}"
|
||||
# Loopback by default: hasJoined is an unauthenticated endpoint by protocol (Velocity
|
||||
# sends no token), so a public bind is a free auth relay — anyone can point their own
|
||||
@@ -237,6 +250,8 @@ HOST_BIN="/usr/local/bin/felis"
|
||||
# version sits here, and an operator's Go at the conventional path is not ours to swap.
|
||||
GOROOT_DIR="/opt/felis/go"
|
||||
NANO_SERVICE="/etc/systemd/system/felis-nano.service"
|
||||
DB_BACKUP_SERVICE="/etc/systemd/system/felis-db-backup.service"
|
||||
DB_BACKUP_TIMER="/etc/systemd/system/felis-db-backup.timer"
|
||||
VELOCITY_DIR="/opt/felis/velocity"
|
||||
VELOCITY_USER="felis-velocity"
|
||||
VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service"
|
||||
@@ -2364,13 +2379,62 @@ ensure_default_config() {
|
||||
# 8. Migrate + deploy bundle
|
||||
# ---------------------------------------------------------------------------
|
||||
run_migrations() {
|
||||
local backup_flags=(-backup-dir "$FELIS_DB_BACKUP_DIR")
|
||||
write_felis_toml "${STATE_DIR}/felis.host.toml" "127.0.0.1"
|
||||
ensure_default_config
|
||||
if [ "$FELIS_PRE_MIGRATE_BACKUP" = 0 ]; then
|
||||
warn "FELIS_PRE_MIGRATE_BACKUP=0: pending migrations run without a database snapshot"
|
||||
backup_flags=(-no-backup)
|
||||
fi
|
||||
# Migrations only roll forward. On an existing database with migrations pending, the
|
||||
# binary bundles the database into FELIS_DB_BACKUP_DIR first and refuses to migrate
|
||||
# if that fails; a fresh database has nothing to protect and is migrated directly.
|
||||
log "running database migrations (host binary -> 127.0.0.1)"
|
||||
"$HOST_BIN" migrate up -config "${STATE_DIR}/felis.host.toml"
|
||||
"$HOST_BIN" migrate up -config "${STATE_DIR}/felis.host.toml" "${backup_flags[@]}"
|
||||
ok "migrations applied"
|
||||
}
|
||||
|
||||
# The daily database backup. The first run happens now, so a broken pipeline (pg_dump
|
||||
# missing, directory unwritable) shows up in this install rather than in the first
|
||||
# restore someone needs.
|
||||
install_db_backup_timer() {
|
||||
install -d -m 0700 "$FELIS_DB_BACKUP_DIR"
|
||||
cat > "$DB_BACKUP_SERVICE" <<EOF
|
||||
[Unit]
|
||||
Description=Felis control-plane database backup (pg_dump + /etc/felis state)
|
||||
After=postgresql.service k3s.service
|
||||
Wants=postgresql.service
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=${HOST_BIN} db backup -config ${STATE_DIR}/felis.host.toml -dir ${FELIS_DB_BACKUP_DIR} -label daily -keep ${FELIS_DB_BACKUP_KEEP} -metrics-file ${FELIS_DB_BACKUP_METRICS}
|
||||
Nice=10
|
||||
IOSchedulingClass=idle
|
||||
PrivateTmp=yes
|
||||
NoNewPrivileges=yes
|
||||
EOF
|
||||
cat > "$DB_BACKUP_TIMER" <<EOF
|
||||
[Unit]
|
||||
Description=Daily Felis control-plane database backup
|
||||
|
||||
[Timer]
|
||||
OnCalendar=${FELIS_DB_BACKUP_TIME}
|
||||
RandomizedDelaySec=15min
|
||||
Persistent=true
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
EOF
|
||||
systemctl daemon-reload
|
||||
systemctl enable --now felis-db-backup.timer
|
||||
if systemctl start felis-db-backup.service; then
|
||||
ok "database backups: daily at ${FELIS_DB_BACKUP_TIME}, newest ${FELIS_DB_BACKUP_KEEP} kept in ${FELIS_DB_BACKUP_DIR} (first one taken now)"
|
||||
else
|
||||
journalctl -u felis-db-backup.service -n 20 --no-pager >&2 || true
|
||||
warn "the first database backup failed (log above); fix it before relying on the daily timer: sudo systemctl start felis-db-backup.service"
|
||||
fi
|
||||
}
|
||||
|
||||
deploy_bundle() {
|
||||
local had_api=0 had_operator=0
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
@@ -2970,6 +3034,8 @@ main() {
|
||||
# After deploy_bundle: the proxy dials felis-api's internal ClusterIP, which does not
|
||||
# exist until the bundle is applied.
|
||||
install_velocity
|
||||
# After deploy_bundle: the bundle's MinecraftServer export reads the cluster.
|
||||
install_db_backup_timer
|
||||
mark_bootstrap_done
|
||||
summary
|
||||
}
|
||||
|
||||
@@ -1017,6 +1017,70 @@ fi
|
||||
|
||||
rm -f "$fnfile"
|
||||
|
||||
# --- database backups: the pre-migration snapshot and the daily timer --------------------
|
||||
# Migrations only roll forward, so an upgrade must hand `migrate up` the snapshot directory,
|
||||
# and only an explicit FELIS_PRE_MIGRATE_BACKUP=0 may take that away.
|
||||
|
||||
mblock="$(awk '/^run_migrations\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$mblock" ] || { echo "FAIL: no run_migrations found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$mblock" | wc -l)" -lt 30 ] \
|
||||
|| { echo "FAIL: the extracted block is not run_migrations -- did its closing brace move?"; exit 1; }
|
||||
|
||||
run_migrate() { # FELIS_PRE_MIGRATE_BACKUP
|
||||
FELIS_PRE_MIGRATE_BACKUP="$1" FELIS_DB_BACKUP_DIR=/var/lib/felis/db-backups STATE_DIR=/etc/felis \
|
||||
HOST_BIN=fakefelis bash -c '
|
||||
log() { :; }; ok() { :; }; warn() { printf "WARN: %s\n" "$*"; }
|
||||
write_felis_toml() { :; }; ensure_default_config() { :; }
|
||||
fakefelis() { printf "RUN: %s\n" "$*"; }
|
||||
'"$mblock"'
|
||||
run_migrations' 2>&1
|
||||
}
|
||||
|
||||
out="$(run_migrate 1)"
|
||||
expect "an upgrade snapshots into the backup dir" "RUN: migrate up -config /etc/felis/felis.host.toml -backup-dir /var/lib/felis/db-backups" "$out"
|
||||
out="$(run_migrate 0)"
|
||||
expect "FELIS_PRE_MIGRATE_BACKUP=0 opts out explicitly" "RUN: migrate up -config /etc/felis/felis.host.toml -no-backup" "$out"
|
||||
expect "the opt-out is loud" "WARN: FELIS_PRE_MIGRATE_BACKUP=0" "$out"
|
||||
|
||||
tblock="$(awk '/^install_db_backup_timer\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$tblock" ] || { echo "FAIL: no install_db_backup_timer found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$tblock" | wc -l)" -lt 60 ] \
|
||||
|| { echo "FAIL: the extracted block is not install_db_backup_timer -- did its closing brace move?"; exit 1; }
|
||||
|
||||
tdir="$(mktemp -d)"
|
||||
run_timer() { # exit status of the first backup
|
||||
FIRST="$1" DB_BACKUP_SERVICE="$tdir/felis-db-backup.service" DB_BACKUP_TIMER="$tdir/felis-db-backup.timer" \
|
||||
FELIS_DB_BACKUP_DIR="$tdir/db-backups" FELIS_DB_BACKUP_KEEP=7 FELIS_DB_BACKUP_TIME='*-*-* 04:00:00' \
|
||||
FELIS_DB_BACKUP_METRICS=/var/lib/node_exporter/textfile_collector/felis_db_backup.prom \
|
||||
HOST_BIN=/usr/local/bin/felis STATE_DIR=/etc/felis bash -c '
|
||||
ok() { printf "OK: %s\n" "$*"; }; warn() { printf "WARN: %s\n" "$*"; }
|
||||
systemctl() { printf "SYSTEMCTL: %s\n" "$*"; [ "$1" != start ] || return "$FIRST"; }
|
||||
journalctl() { printf "JOURNAL: pg_dump: connection refused\n"; }
|
||||
'"$tblock"'
|
||||
install_db_backup_timer' 2>&1
|
||||
}
|
||||
|
||||
out="$(run_timer 0)"
|
||||
unit="$(cat "$tdir/felis-db-backup.service")"
|
||||
timer="$(cat "$tdir/felis-db-backup.timer")"
|
||||
expect "the unit runs a daily-labelled backup with the configured retention" \
|
||||
"ExecStart=/usr/local/bin/felis db backup -config /etc/felis/felis.host.toml -dir $tdir/db-backups -label daily -keep 7 -metrics-file /var/lib/node_exporter/textfile_collector/felis_db_backup.prom" "$unit"
|
||||
expect "the timer fires at the configured time" "OnCalendar=*-*-* 04:00:00" "$timer"
|
||||
expect "a missed run (host off at 03:30) catches up at boot" "Persistent=true" "$timer"
|
||||
expect "the timer is enabled" "SYSTEMCTL: enable --now felis-db-backup.timer" "$out"
|
||||
expect "the first backup runs during the install" "SYSTEMCTL: start felis-db-backup.service" "$out"
|
||||
expect "a working first backup is reported" "OK: database backups: daily" "$out"
|
||||
if [ "$(stat -c %a "$tdir/db-backups" 2>/dev/null || stat -f %Lp "$tdir/db-backups")" = 700 ]; then
|
||||
echo "PASS the backup directory is private"
|
||||
else
|
||||
echo "FAIL the backup directory must be 0700"; fails=$((fails + 1))
|
||||
fi
|
||||
|
||||
out="$(run_timer 1)"
|
||||
expect "a failed first backup shows its log" "JOURNAL: pg_dump: connection refused" "$out"
|
||||
expect "a failed first backup is a loud warning" "WARN: the first database backup failed" "$out"
|
||||
rm -rf "$tdir"
|
||||
|
||||
# ---------------------------------------------------------------------------------------
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS"
|
||||
|
||||
Reference in new issue
Block a user