fix(watchdog): 外部心跳、巡检失败兜底告警、状态文件损坏自动移开

This commit is contained in:
Lemon-miaow committed 2026-09-27 08:40:32 +08:00
1 parent 149ab0be66
commit 82ffa55a8f
11 files changed
+1360 -44

No files matched your search

+96 -6
View File
@@ -266,6 +266,13 @@ FELIS_OFFSITE_BUCKET="${FELIS_OFFSITE_BUCKET:-}"
FELIS_OFFSITE_REGION="${FELIS_OFFSITE_REGION:-}"
FELIS_OFFSITE_PREFIX="${FELIS_OFFSITE_PREFIX:-}"
FELIS_OFFSITE_DB_KEEP="${FELIS_OFFSITE_DB_KEEP:-}"
# The watchdog's heartbeat (troubleshooting §14): the ping URL of a check at a monitoring
# service such as Healthchecks.io, a dead man's switch. Every watchdog run pings it, and
# the service mails its own users when the pings stop or report failure: the host down,
# the watchdog broken, alerts that reach no one. Nothing on this host can report its own
# death. The URL is kept in /etc/felis/watchdog-heartbeat-url, mode 0600, as its path is
# the key that pings the check; off removes it, and a re-run without it keeps it.
FELIS_WATCHDOG_HEARTBEAT_URL="${FELIS_WATCHDOG_HEARTBEAT_URL:-}"
INSTALL_MODE="${FELIS_INSTALL_MODE:-}"
# strict stops the install on any preflight problem (preflight below); warn reports them
# and goes on, for a host the checks misjudge.
@@ -416,6 +423,8 @@ NANO_SERVICE="/etc/systemd/system/felis-nano.service"
DB_BACKUP_SERVICE="/etc/systemd/system/felis-db-backup.service"
DB_BACKUP_TIMER="/etc/systemd/system/felis-db-backup.timer"
WATCHDOG_SERVICE="/etc/systemd/system/felis-watchdog.service"
# OnFailure= of felis-watchdog.service: reports a run that failed (felis watchdog -unit-failed).
WATCHDOG_FAILED_SERVICE="/etc/systemd/system/felis-watchdog-failed.service"
WATCHDOG_TIMER="/etc/systemd/system/felis-watchdog.timer"
UPDATE_CHECK_SERVICE="/etc/systemd/system/felis-update-check.service"
UPDATE_CHECK_TIMER="/etc/systemd/system/felis-update-check.timer"
@@ -436,6 +445,8 @@ BUILD_TOOLS_STATUS="/var/lib/felis/build-tools/status.json"
# restarts the control plane and the system servers on purpose. cleanup removes it; the
# time in it is the backstop for an installer killed before its EXIT trap runs.
WATCHDOG_QUIET_FILE="/run/felis/watchdog-quiet-until"
# FELIS_WATCHDOG_HEARTBEAT_URL, where felis watchdog reads it by default.
WATCHDOG_HEARTBEAT_FILE="${STATE_DIR}/watchdog-heartbeat-url"
VELOCITY_DIR="/opt/felis/velocity"
VELOCITY_USER="felis-velocity"
VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service"
@@ -926,6 +937,7 @@ validate_settings() {
validate_listen FELIS_NANO_LISTEN "$FELIS_NANO_LISTEN"
validate_cidr FELIS_NANO_PROXY_CIDR "$FELIS_NANO_PROXY_CIDR"
validate_offsite_settings
validate_heartbeat_url
case "$FELIS_PREFLIGHT" in
strict|warn) ;;
*) die "FELIS_PREFLIGHT must be strict or warn (got '${FELIS_PREFLIGHT}')" ;;
@@ -1028,6 +1040,18 @@ validate_offsite_settings() {
fi
}
# validate_heartbeat_url checks FELIS_WATCHDOG_HEARTBEAT_URL before anything is
# installed: a check's http(s) ping URL, or off. The messages leave the URL out: its path
# is the key that pings the check.
validate_heartbeat_url() {
case "$FELIS_WATCHDOG_HEARTBEAT_URL" in
"" | off) ;;
*[[:space:]]* | *\"* | *\'* | *\\*) die "FELIS_WATCHDOG_HEARTBEAT_URL must not contain spaces, quotes or backslashes" ;;
http://[!/]* | https://[!/]*) ;;
*) die "FELIS_WATCHDOG_HEARTBEAT_URL must be the http:// or https:// ping URL of a monitoring service's check, or off" ;;
esac
}
# ---------------------------------------------------------------------------
# 0. Privilege & host facts
# ---------------------------------------------------------------------------
@@ -4685,11 +4709,6 @@ summary_offsite() {
fi
}
# The platform watchdog: every two minutes it checks the control plane, the login gate,
# the fleet, PostgreSQL, the game proxy, the database backups and the host's disks and
# memory, and mails the owners (their verified addresses, over the [smtp] relay) what
# has stayed wrong long enough to matter. It runs on the host so a k3s that is down is
# still reported. The first run happens now, so a broken unit shows up in this install.
# The daily version check. Felis applies no update on its own; `felis update --record`
# compares what this host runs with the newest upstream releases and stores the result,
# which the panel's Updates page shows with the command that applies each update. It runs
@@ -4730,16 +4749,29 @@ EOF
ok "version check: daily; the panel's Updates page shows what has a newer release (journalctl -u felis-update-check)"
}
# The platform watchdog: every two minutes it checks the control plane, the login gate,
# the fleet, PostgreSQL, the game proxy, the database backups and the host's disks and
# memory, and mails the owners (their verified addresses, over the [smtp] relay) what
# has stayed wrong long enough to matter. It runs on the host so a k3s that is down is
# still reported. The first run happens now, so a broken unit shows up in this install.
# A run that fails starts felis-watchdog-failed.service (OnFailure=), which mails the
# failure once five runs in a row failed, through the relay the last good run cached, and
# pings the heartbeat's failure endpoint. The heartbeat (FELIS_WATCHDOG_HEARTBEAT_URL) is
# what notices a host that is down or a watchdog that no longer runs at all. The units
# name no heartbeat flag: felis watchdog reads WATCHDOG_HEARTBEAT_FILE by default, so an
# older binary put back under these units still runs.
install_watchdog_timer() {
local disks="/,/var/lib/rancher/k3s,/var/lib/felis" path
for path in "$FELIS_WORLDS_HOST_PATH" "$FELIS_ARCHIVE_LOCAL_PATH" "$FELIS_DB_BACKUP_DIR"; do
if [ -n "$path" ]; then disks="${disks},${path}"; fi
done
write_heartbeat_url
install -d -m 0700 "$(dirname "$WATCHDOG_STATE")"
cat > "$WATCHDOG_SERVICE" <<EOF
[Unit]
Description=Felis platform watchdog (health checks, owner alert mail)
After=network-online.target k3s.service
OnFailure=felis-watchdog-failed.service
[Service]
Type=oneshot
@@ -4749,6 +4781,19 @@ Nice=5
PrivateTmp=yes
NoNewPrivileges=yes
ProtectSystem=full
EOF
cat > "$WATCHDOG_FAILED_SERVICE" <<EOF
[Unit]
Description=Felis platform watchdog failure report (owner alert mail, heartbeat failure ping)
[Service]
Type=oneshot
ExecStart=${HOST_BIN} watchdog -unit-failed -config ${STATE_DIR}/felis.host.toml -state ${WATCHDOG_STATE} -quiet-file ${WATCHDOG_QUIET_FILE}
TimeoutStartSec=2min
Nice=5
PrivateTmp=yes
NoNewPrivileges=yes
ProtectSystem=full
EOF
cat > "$WATCHDOG_TIMER" <<EOF
[Unit]
@@ -4764,14 +4809,58 @@ WantedBy=timers.target
EOF
systemctl daemon-reload
systemctl enable --now felis-watchdog.timer
local reach="mails the owners' verified addresses"
if [ -f "$WATCHDOG_HEARTBEAT_FILE" ]; then
reach="${reach} and pings the heartbeat at $(heartbeat_host)"
fi
if systemctl start felis-watchdog.service; then
ok "watchdog: checks every 2 minutes and mails the owners' verified addresses (journalctl -u felis-watchdog)"
ok "watchdog: checks every 2 minutes and ${reach} (journalctl -u felis-watchdog)"
else
journalctl -u felis-watchdog.service -n 20 --no-pager >&2 || true
warn "the first watchdog run failed (log above); nothing will be mailed until it runs: sudo systemctl start felis-watchdog.service"
fi
}
# write_heartbeat_url keeps FELIS_WATCHDOG_HEARTBEAT_URL in WATCHDOG_HEARTBEAT_FILE, mode
# 0600 and replaced whole; off removes the file, and no value keeps it as it is.
write_heartbeat_url() {
local tmp
case "$FELIS_WATCHDOG_HEARTBEAT_URL" in
"") return 0 ;;
off)
rm -f -- "$WATCHDOG_HEARTBEAT_FILE"
return 0
;;
esac
# mktemp creates the file 0600, before the key is in it.
tmp="$(mktemp "${WATCHDOG_HEARTBEAT_FILE}.XXXXXX")"
printf '%s\n' "$FELIS_WATCHDOG_HEARTBEAT_URL" > "$tmp"
mv -f -- "$tmp" "$WATCHDOG_HEARTBEAT_FILE"
}
# heartbeat_host is the heartbeat URL as the install shows it: its scheme and host.
heartbeat_host() {
local url rest host
url="$(head -n 1 "$WATCHDOG_HEARTBEAT_FILE")"
rest="${url#*://}"
host="${rest%%/*}"
host="${host%%\?*}"
host="${host##*@}"
printf '%s://%s/...' "${url%%://*}" "$host"
}
# summary_heartbeat closes the install on the heartbeat: without one, nothing off this
# machine notices it going down.
summary_heartbeat() {
if [ -f "$WATCHDOG_HEARTBEAT_FILE" ]; then
log "Heartbeat: every watchdog run pings $(heartbeat_host); that service mails you when the pings stop."
return 0
fi
warn "NO HEARTBEAT: nothing off this machine notices it going down or its watchdog stopping."
warn "Create a check at a monitoring service (Healthchecks.io or alike; period 2 min, grace 10 min),"
warn "then re-run with FELIS_WATCHDOG_HEARTBEAT_URL=<its ping URL> (docs/troubleshooting.md §14)."
}
# The build lane's tools: kaniko and trivy (pinned by digest in internal/build/tools.go)
# and Trivy's vulnerability and Java DBs, copied into the registry's mirror/ where build
# Jobs pull them; the build namespace has no internet egress. The timer refreshes the DBs
@@ -5374,6 +5463,7 @@ summary() {
fi
echo
summary_offsite
summary_heartbeat
echo
}