diff --git a/deploy/bootstrap.sh b/deploy/bootstrap.sh index fc36b05..824c4b3 100644 --- a/deploy/bootstrap.sh +++ b/deploy/bootstrap.sh @@ -104,6 +104,13 @@ # registry, uploads and world-archive PVCs request on first install # (defaults: 10Gi, 5Gi, 10Gi). An existing claim keeps its size; on # k3s local-path the number is not enforced, see troubleshooting §9 +# FELIS_MANAGE_TIME_SYNC 0 leaves the host's clock alone; by default the installer +# turns NTP on (installing chrony when nothing can) and waits for it +# to synchronize (default: 1) +# FELIS_MANAGE_JOURNAL 0 leaves journald alone; by default the installer makes the +# system journal persistent so logs survive a reboot (default: 1) +# FELIS_JOURNAL_MAX_USE the persistent journal's size cap, journald's SystemMaxUse +# written K|M|G (default: 1G) # PKG_LOCK_TIMEOUT seconds to wait for package-manager locks (default: 900) # APT_LOCK_TIMEOUT legacy alias for PKG_LOCK_TIMEOUT set -Eeuo pipefail @@ -245,6 +252,9 @@ CLOUDFLARED_BIN=/usr/local/bin/cloudflared # pins both. An installed k3s moves only under FELIS_UPGRADE_DEPS=1. FELIS_K3S_VERSION="${FELIS_K3S_VERSION:-v1.36.4+k3s1}" FELIS_UPGRADE_DEPS="${FELIS_UPGRADE_DEPS:-0}" +FELIS_MANAGE_TIME_SYNC="${FELIS_MANAGE_TIME_SYNC:-1}" +FELIS_MANAGE_JOURNAL="${FELIS_MANAGE_JOURNAL:-1}" +FELIS_JOURNAL_MAX_USE="${FELIS_JOURNAL_MAX_USE:-1G}" # The in-cluster registry's image, by digest. It must equal platform.defaultRegistryImage # (internal/platform/identities.go, TestBootstrapPinsTheRegistryImage): the renderer puts # that ref in the Deployment, and this script caches and pins the same ref in containerd. @@ -381,6 +391,17 @@ K3S_BIN="${K3S_BIN_DIR}/k3s" # (not just the literal path) so bootstrap_test.sh can point the writer at a # scratch file. K3S_REGISTRIES_FILE="/etc/rancher/k3s/registries.yaml" +# The installer's k3s settings (write_k3s_config), a drop-in k3s reads after any +# config.yaml the operator keeps. The unit, the admin kubeconfig and the kubelet's +# client certificate are variables for the same reason as the file above. +K3S_CONFIG_DROPIN="/etc/rancher/k3s/config.yaml.d/50-felis.yaml" +K3S_UNIT_FILE="/etc/systemd/system/k3s.service" +K3S_KUBECONFIG="/etc/rancher/k3s/k3s.yaml" +K3S_KUBELET_CERT="/var/lib/rancher/k3s/agent/client-kubelet.crt" +# ensure_persistent_journal's drop-in, and the directory journald creates once it +# stores the journal persistently. +JOURNALD_DROPIN="/etc/systemd/journald.conf.d/50-felis.conf" +JOURNAL_DIR="/var/log/journal" APT_LOCK_FILES=( /var/lib/dpkg/lock-frontend /var/lib/dpkg/lock @@ -500,6 +521,10 @@ restore_previous_host_binary() { # exit-status remember_temp() { TEMP_PATHS+=("$1"); } remember_container() { DOCKER_CONTAINERS+=("$1"); } +# Gives a file the SELinux label its path calls for, on hosts that have SELinux. It +# repairs files earlier installers wrote under /tmp and moved into place, which kept +# user_tmp_t (a confined daemon is then denied them). +restore_label() { if command -v restorecon >/dev/null 2>&1; then restorecon "$1" || true; fi; } trap 'on_error "$LINENO" "$?"' ERR trap cleanup EXIT @@ -768,6 +793,20 @@ validate_settings() { 0|1) ;; *) die "FELIS_UPGRADE_DEPS must be 0 or 1 (got '${FELIS_UPGRADE_DEPS}')" ;; esac + case "$FELIS_MANAGE_TIME_SYNC" in + 0|1) ;; + *) die "FELIS_MANAGE_TIME_SYNC must be 0 or 1 (got '${FELIS_MANAGE_TIME_SYNC}')" ;; + esac + case "$FELIS_MANAGE_JOURNAL" in + 0|1) ;; + *) die "FELIS_MANAGE_JOURNAL must be 0 or 1 (got '${FELIS_MANAGE_JOURNAL}')" ;; + esac + # Stripping the unit off a size with none leaves it whole, which the first arm catches. + local journal_n="${FELIS_JOURNAL_MAX_USE%[KMG]}" + case "$journal_n" in + "$FELIS_JOURNAL_MAX_USE"|""|0*|*[!0-9]*) + die "FELIS_JOURNAL_MAX_USE must be written K, M or G (got '${FELIS_JOURNAL_MAX_USE}')" ;; + esac } # version_newer reports whether version $1 sorts after $2 (a leading v is ignored). @@ -895,6 +934,20 @@ detect_node_ip() { fi } +# warn_dynamic_node_ip warns when NODE_IP is a DHCP lease. The address is written into +# the database connection string, pg_hba, the network policies, the panel certificate +# and the default nip.io domain, and nothing re-addresses a live install, so a lease +# that later comes back different takes the whole platform down (the watchdog then +# reports host-address). `ip -o addr` marks a leased address "dynamic". +warn_dynamic_node_ip() { + if ip -4 -o addr show 2>/dev/null | awk -v ip="$NODE_IP" ' + { split($4, a, "/"); if (a[1] == ip && / dynamic /) found = 1 } + END { exit !found }'; then + warn "${NODE_IP} is a DHCP lease, and the install is bound to this address. Give the host a" + warn "DHCP reservation or a static address before it changes (docs/operations.md §1)." + fi +} + pkg_install() { case "$PKG" in apt) apt_get install -y "$@" ;; @@ -953,6 +1006,88 @@ ensure_swap() { ok "2 GiB swap active" } +# ensure_time_sync turns NTP on. A drifting clock breaks things far from their cause: +# sign-in codes and sessions expire early or late, S3 refuses off-site uploads signed +# more than 15 minutes off, and certificate checks fail. Rocky's minimal image ships +# chronyd disabled (the test host reported NTP=no), so the installer enables whatever +# timedatectl manages and installs chrony only when there is nothing to enable. It +# waits half a minute for the first synchronization and then carries on: the watchdog +# keeps reporting an unsynchronized clock. +ensure_time_sync() { + if [ "$FELIS_MANAGE_TIME_SYNC" = 0 ]; then + log "FELIS_MANAGE_TIME_SYNC=0: leaving time synchronization to the operator" + return 0 + fi + if ! command -v timedatectl >/dev/null 2>&1; then + warn "timedatectl not found; make sure an NTP client keeps this host's clock (docs/operations.md §1)" + return 0 + fi + if [ "$(timedatectl show -p NTP --value 2>/dev/null)" != yes ]; then + # set-ntp fails with "NTP not supported" when no NTP unit is installed at all + # (Debian's minimal image splits systemd-timesyncd into its own package). + if ! timedatectl set-ntp true 2>/dev/null; then + log "no NTP client to enable; installing chrony" + pkg_install chrony + if ! timedatectl set-ntp true; then + warn "could not turn NTP on; set up time synchronization by hand (docs/troubleshooting.md §13c)" + return 0 + fi + fi + log "turned NTP time synchronization on" + fi + local _ + for _ in $(seq 1 15); do + if [ "$(timedatectl show -p NTPSynchronized --value 2>/dev/null)" = yes ]; then + ok "system clock synchronized by NTP" + return 0 + fi + sleep 2 + done + warn "the system clock is not synchronized yet; check 'timedatectl' (docs/troubleshooting.md §13c)" +} + +# ensure_persistent_journal keeps the system journal across reboots. Rocky's journald +# stores it under /run unless /var/log/journal exists, and its minimal image does not +# create that directory, so a reboot (the moment an operator most needs to know what +# came before it) erased every log. journald creates JOURNAL_DIR itself once it runs +# with Storage=persistent, so the directory is the proof the drop-in took: journald is +# restarted when the drop-in changed, or when it is current and the directory is still +# missing. +ensure_persistent_journal() { + if [ "$FELIS_MANAGE_JOURNAL" = 0 ]; then + log "FELIS_MANAGE_JOURNAL=0: leaving journald as it is" + return 0 + fi + local file="$JOURNALD_DROPIN" tmp + mkdir -p "$(dirname "$file")" + # Made beside its destination so the file is born with that directory's SELinux + # label. A file made under /tmp keeps user_tmp_t through the mv, and journald was + # denied it on the test host ("Failed to open configuration file ... Permission + # denied"). journald reads only *.conf, so the temporary name is never loaded. + tmp="$(mktemp "${file}.XXXXXX")" + remember_temp "$tmp" + printf '[Journal]\nStorage=persistent\nSystemMaxUse=%s\n' "$FELIS_JOURNAL_MAX_USE" > "$tmp" + if [ -f "$file" ] && cmp -s "$tmp" "$file"; then + rm -f "$tmp" + if [ -d "$JOURNAL_DIR" ]; then + ok "system journal already persistent (capped at ${FELIS_JOURNAL_MAX_USE})" + return 0 + fi + log "journald has not taken up ${file}; restarting it" + else + chmod 0644 "$tmp" + mv "$tmp" "$file" + fi + restore_label "$file" + systemctl restart systemd-journald + journalctl --flush >/dev/null 2>&1 || true + if [ -d "$JOURNAL_DIR" ]; then + ok "system journal is persistent under ${JOURNAL_DIR} (capped at ${FELIS_JOURNAL_MAX_USE})" + else + warn "journald did not create ${JOURNAL_DIR}, so logs still end at a reboot; see: journalctl -u systemd-journald" + fi +} + # --------------------------------------------------------------------------- # 2. Base packages # --------------------------------------------------------------------------- @@ -1142,7 +1277,11 @@ configure_k3s_firewall() { install_k3s() { configure_k3s_firewall + # Before the installer runs: a fresh k3s reads the drop-in on its first start. + K3S_RESTART_NEEDED=0 + write_k3s_config + local installer_ran=0 if [ -x "$K3S_BIN" ]; then local current current="$("$K3S_BIN" --version 2>/dev/null | awk 'NR == 1 { print $3 }')" @@ -1153,18 +1292,112 @@ install_k3s() { elif k3s_upgrade_allowed "$current" "$FELIS_K3S_VERSION"; then log "upgrading k3s ${current} to ${FELIS_K3S_VERSION}; running pods keep running while it restarts" run_k3s_installer + installer_ran=1 fi else log "installing k3s ${FELIS_K3S_VERSION} into ${K3S_BIN_DIR} (no traefik/servicelb/metrics-server)" run_k3s_installer + installer_ran=1 fi [ -x "$K3S_BIN" ] || die "k3s installation completed but ${K3S_BIN} is missing" + strip_k3s_kubeconfig_mode_flag systemctl enable --now k3s - export KUBECONFIG=/etc/rancher/k3s/k3s.yaml + # The installer restarts k3s itself; otherwise a changed drop-in or unit takes a + # restart to load. Pods keep running across it (k3s leaves the containers be). + if [ "$K3S_RESTART_NEEDED" = 1 ] && [ "$installer_ran" = 0 ]; then + log "restarting k3s to load its new settings (${K3S_CONFIG_DROPIN})" + systemctl restart k3s + fi + export KUBECONFIG="$K3S_KUBECONFIG" log "waiting for the node to become Ready" wait_for_node_ready + # k3s applies write-kubeconfig-mode as it writes the file; this covers a k3s that + # has not rewritten it since the mode changed. + chmod 0600 "$K3S_KUBECONFIG" +} + +# k3s_node_name prints the name this node must keep. Every local-path volume (worlds, +# registry, uploads, backups) is bound to its node by name, and k3s takes the name from +# the hostname on every start, so a renamed host came back as a second, empty node with +# every volume stuck Pending on the old one. Precedence: the name already pinned; else +# the name the node registered under, which its kubelet client certificate carries as +# system:node: and which is readable with k3s stopped; else, where k3s has never +# run, the lowercased hostname k3s itself would pick. It prints nothing when k3s has run +# but neither source can be read: pinning a guess there would rename the node. +k3s_node_name() { + local name="" + if [ -f "$K3S_CONFIG_DROPIN" ]; then + name="$(awk -F'"' '/^node-name:/ { print $2; exit }' "$K3S_CONFIG_DROPIN")" + fi + if [ -z "$name" ] && [ -f "$K3S_KUBELET_CERT" ]; then + # OpenSSL 3 prints "CN=system:node:x", 1.1 "CN = system:node:x", older "/CN=...". + # An unreadable certificate leaves the name empty (and warned about), not a failed run. + name="$(openssl x509 -in "$K3S_KUBELET_CERT" -noout -subject 2>/dev/null | + sed -n 's/.*CN *= *system:node:\([^,/]*\).*/\1/p' || true)" + elif [ -z "$name" ] && [ ! -x "$K3S_BIN" ]; then + name="$(uname -n | tr '[:upper:]' '[:lower:]')" + fi + printf '%s' "$name" +} + +# write_k3s_config writes the installer's k3s settings to K3S_CONFIG_DROPIN and sets +# K3S_RESTART_NEEDED when they changed, since k3s reads the file only as it starts. +# write-kubeconfig-mode keeps the admin kubeconfig root-only: it is cluster-admin, and +# installs before this passed 644, which let every local account read it, felis-velocity +# (the account the internet-facing proxy runs as) included. +write_k3s_config() { + local file="$K3S_CONFIG_DROPIN" name tmp + name="$(k3s_node_name)" + mkdir -p "$(dirname "$file")" + # Beside its destination for the directory's SELinux label (ensure_persistent_journal + # has the story); k3s loads only *.yaml and *.yml from the directory. + tmp="$(mktemp "${file}.XXXXXX")" + remember_temp "$tmp" + { + echo "# Written by the Felis installer (deploy/bootstrap.sh); a rerun rewrites it." + echo 'write-kubeconfig-mode: "0600"' + if [ -n "$name" ]; then + printf 'node-name: "%s"\n' "$name" + fi + } > "$tmp" + if [ -z "$name" ]; then + warn "could not read this node's k3s name, so it is not pinned; a hostname change would orphan every volume" + fi + if [ -f "$file" ] && cmp -s "$tmp" "$file"; then + rm -f "$tmp" + restore_label "$file" + ok "k3s settings already current${name:+ (node name ${name})}" + return 0 + fi + chmod 0600 "$tmp" + mv "$tmp" "$file" + K3S_RESTART_NEEDED=1 + log "wrote ${file}${name:+ (node name pinned to ${name})}" +} + +# A command-line flag outranks every config file, and k3s's installer writes +# INSTALL_K3S_EXEC into the unit's ExecStart one quoted word per line, so installs from +# before the drop-in keep "'--write-kubeconfig-mode' \" followed by "'644' \" there. +# This drops both lines (or the single --write-kubeconfig-mode= spelling). An +# upgrade through run_k3s_installer rewrites the unit without them anyway. +strip_k3s_kubeconfig_mode_flag() { + local unit="$K3S_UNIT_FILE" tmp + [ -f "$unit" ] && grep -q -- '--write-kubeconfig-mode' "$unit" || return 0 + tmp="$(mktemp)" + remember_temp "$tmp" + awk ' + skip { skip = 0; next } + index($0, "--write-kubeconfig-mode") { if (index($0, "=") == 0) skip = 1; next } + { print } + ' "$unit" > "$tmp" + # Rewritten in place, so the unit keeps its owner, mode and SELinux label. + cat "$tmp" > "$unit" + rm -f "$tmp" + systemctl daemon-reload + K3S_RESTART_NEEDED=1 + log "dropped --write-kubeconfig-mode from ${unit}; the admin kubeconfig becomes root-only" } # The script from the release's own tag rather than get.k3s.io, which serves whatever @@ -1174,7 +1407,7 @@ run_k3s_installer() { curl -sfL --retry 5 --retry-delay 2 "https://raw.githubusercontent.com/k3s-io/k3s/${FELIS_K3S_VERSION}/install.sh" | \ INSTALL_K3S_VERSION="$FELIS_K3S_VERSION" \ INSTALL_K3S_BIN_DIR="$K3S_BIN_DIR" \ - INSTALL_K3S_EXEC="--disable traefik --disable servicelb --disable metrics-server --write-kubeconfig-mode 644" \ + INSTALL_K3S_EXEC="--disable traefik --disable servicelb --disable metrics-server" \ sh - } @@ -3321,7 +3554,7 @@ After=network-online.target k3s.service postgresql.service [Service] Type=oneshot -ExecStart=${HOST_BIN} watchdog -config ${STATE_DIR}/felis.host.toml -state ${WATCHDOG_STATE} -quiet-file ${WATCHDOG_QUIET_FILE} -backup-dir ${FELIS_DB_BACKUP_DIR} -proxy-addr 127.0.0.1:${FELIS_GAME_PORT} -disk-paths ${disks} +ExecStart=${HOST_BIN} watchdog -config ${STATE_DIR}/felis.host.toml -state ${WATCHDOG_STATE} -quiet-file ${WATCHDOG_QUIET_FILE} -backup-dir ${FELIS_DB_BACKUP_DIR} -proxy-addr 127.0.0.1:${FELIS_GAME_PORT} -disk-paths ${disks}${NODE_IP:+ -node-ip ${NODE_IP}} TimeoutStartSec=3min Nice=5 PrivateTmp=yes @@ -4204,8 +4437,11 @@ main() { quiet_watchdog pause_package_background_timers detect_node_ip + warn_dynamic_node_ip ensure_swap install_base + ensure_time_sync + ensure_persistent_journal # Right after install_base because it is the first point curl exists, and well before # docker and k3s: a missing FELIS_GITHUB_TOKEN or an unpublished release should cost # the operator seconds, not a k3s install they then have to unwind. This is purely diff --git a/deploy/bootstrap_test.sh b/deploy/bootstrap_test.sh index a19b94b..5723b16 100644 --- a/deploy/bootstrap_test.sh +++ b/deploy/bootstrap_test.sh @@ -850,9 +850,12 @@ run_k3s() { # installed-version pinned-version [FELIS_UPGRADE_DEPS] log() { printf "LOG: %s\n" "$*"; } ok() { printf "OK: %s\n" "$*"; } configure_k3s_firewall() { :; } + write_k3s_config() { :; } + strip_k3s_kubeconfig_mode_flag() { :; } run_k3s_installer() { printf "INSTALLER: %s\n" "$FELIS_K3S_VERSION"; } systemctl() { :; } wait_for_node_ready() { :; } + chmod() { :; } '"$(awk '/^version_newer\(\) \{/,/^}/' "$BS")"' '"$(awk '/^k3s_upgrade_allowed\(\) \{/,/^}/' "$BS")"' '"$(awk '/^install_k3s\(\) \{/,/^}/' "$BS")"' @@ -1566,8 +1569,8 @@ qblock="$(awk '/^quiet_watchdog\(\) \{/,/^}/' "$BS")" [ -n "$qblock" ] || { echo "FAIL: no quiet_watchdog found in $BS"; exit 1; } tdir="$(mktemp -d)" -run_watchdog_timer() { # $1: exit status of the first run, $2: FELIS_WORLDS_HOST_PATH - FIRST="$1" FELIS_WORLDS_HOST_PATH="$2" WATCHDOG_SERVICE="$tdir/felis-watchdog.service" WATCHDOG_TIMER="$tdir/felis-watchdog.timer" \ +run_watchdog_timer() { # $1: exit status of the first run, $2: FELIS_WORLDS_HOST_PATH, $3: NODE_IP + FIRST="$1" FELIS_WORLDS_HOST_PATH="$2" NODE_IP="${3:-}" WATCHDOG_SERVICE="$tdir/felis-watchdog.service" WATCHDOG_TIMER="$tdir/felis-watchdog.timer" \ WATCHDOG_STATE="$tdir/watchdog/state.json" WATCHDOG_QUIET_FILE=/run/felis/watchdog-quiet-until \ FELIS_DB_BACKUP_DIR=/var/lib/felis/db-backups FELIS_ARCHIVE_LOCAL_PATH=/var/lib/felis/archives FELIS_GAME_PORT=25577 \ HOST_BIN=/usr/local/bin/felis STATE_DIR=/etc/felis bash -c ' @@ -1598,6 +1601,14 @@ fi out="$(run_watchdog_timer 0 /srv/worlds)" expect "a custom worlds root is watched for free space" "-disk-paths /,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis,/srv/worlds," "$(cat "$tdir/felis-watchdog.service")" +case "$unit" in + *-node-ip*) echo "FAIL without a node address the watchdog must not check one"; fails=$((fails + 1)) ;; + *) echo "PASS without a node address the watchdog checks none" ;; +esac +out="$(run_watchdog_timer 0 "" 10.211.55.6)" +expect "the watchdog checks the host still holds the install's address" \ + "-disk-paths /,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis,/var/lib/felis/archives,/var/lib/felis/db-backups -node-ip 10.211.55.6 +" "$(cat "$tdir/felis-watchdog.service")" out="$(run_watchdog_timer 1 "")" expect "a failed first watchdog run shows its log" "JOURNAL: parse /etc/felis/felis.host.toml" "$out" @@ -2183,6 +2194,347 @@ expect "the install's apt runs keep needrestart from restarting services" \ "NEEDRESTART_SUSPEND=1 -o DPkg::Lock::Timeout=5 install -y postgresql" "$out" rm -rf "$rrdir" +# --- host hardening: k3s settings, clock, journal, the node address ---------------------- + +kcblock="$(awk '/^k3s_node_name\(\) \{/,/^}/' "$BS") +$(awk '/^write_k3s_config\(\) \{/,/^}/' "$BS")" +case "$kcblock" in + *'k3s_node_name() {'*'write_k3s_config() {'*) ;; + *) echo "FAIL: k3s_node_name / write_k3s_config not found in $BS"; exit 1 ;; +esac +[ "$(printf '%s\n' "$kcblock" | wc -l)" -lt 60 ] \ + || { echo "FAIL: the extracted k3s settings blocks ran past their closing braces"; exit 1; } + +kdir="$(mktemp -d)" +# The kubelet client certificate a k3s agent holds: k3s issues it with exactly this subject. +openssl req -x509 -newkey ec -pkeyopt ec_paramgen_curve:prime256v1 -nodes -days 1 \ + -keyout "$kdir/kubelet.key" -out "$kdir/kubelet.crt" \ + -subj "/O=system:nodes/CN=system:node:localhost.localdomain" >/dev/null 2>&1 +printf '#!/bin/sh\n' > "$kdir/k3s"; chmod +x "$kdir/k3s" + +run_k3s_config() { # $1: kubelet cert path, $2: k3s binary path, $3: hostname, $4: openssl subject stub ("" = real openssl) + CERT="$1" BIN="$2" HOSTNAME_STUB="$3" SUBJECT="${4:-}" DROPIN="$kdir/config.yaml.d/50-felis.yaml" bash -c ' + set -Eeuo pipefail + ok() { echo "OK: $*"; }; log() { echo "LOG: $*"; }; warn() { echo "WARN: $*"; } + remember_temp() { :; } + restore_label() { echo "RELABEL: $*"; } + mktemp() { local p; p="$(command mktemp "$@")"; echo "MKTEMP: $(dirname "$p")" >&2; echo "$p"; } + uname() { echo "$HOSTNAME_STUB"; } + if [ -n "$SUBJECT" ]; then openssl() { echo "$SUBJECT"; }; fi + K3S_CONFIG_DROPIN="$DROPIN" K3S_KUBELET_CERT="$CERT" K3S_BIN="$BIN" + '"$kcblock"' + K3S_RESTART_NEEDED=0 + write_k3s_config + echo "RESTART=$K3S_RESTART_NEEDED"' 2>&1 +} +dropin="$kdir/config.yaml.d/50-felis.yaml" + +out="$(run_k3s_config "$kdir/none.crt" "$kdir/none" Rocky-Box)" +expect "a fresh host pins the hostname k3s would take, lowercased" 'node-name: "rocky-box"' "$(cat "$dropin")" +expect "the admin kubeconfig is root-only" 'write-kubeconfig-mode: "0600"' "$(cat "$dropin")" +expect "a new drop-in asks for a k3s restart" "RESTART=1" "$out" +if [ "$(printf '%s\n' "$out" | sed -n 's/^MKTEMP: //p' | sort -u)" = "$kdir/config.yaml.d" ]; then + echo "PASS the k3s drop-in is made beside itself, never under /tmp" +else + echo "FAIL temporary files for the k3s drop-in were made in: $(printf '%s\n' "$out" | sed -n 's/^MKTEMP: //p')"; fails=$((fails + 1)) +fi +if [ "$(stat -c %a "$dropin" 2>/dev/null || stat -f %Lp "$dropin")" = 600 ]; then + echo "PASS the k3s drop-in is root-only" +else + echo "FAIL the k3s drop-in must be 0600"; fails=$((fails + 1)) +fi + +out="$(run_k3s_config "$kdir/none.crt" "$kdir/none" Rocky-Box)" +expect "an unchanged drop-in is left alone" "RESTART=0" "$out" +expect "an unchanged drop-in says so" 'OK: k3s settings already current (node name rocky-box)' "$out" +expect "an unchanged drop-in is still relabelled (one an earlier installer moved in from /tmp)" "RELABEL: $dropin" "$out" +if [ "$(ls "$kdir/config.yaml.d")" = "50-felis.yaml" ]; then + echo "PASS no temporary file is left beside the k3s drop-in" +else + echo "FAIL config.yaml.d holds: $(ls "$kdir/config.yaml.d")"; fails=$((fails + 1)) +fi + +rm -f "$dropin" +out="$(run_k3s_config "$kdir/kubelet.crt" "$kdir/k3s" renamed-host)" +expect "an installed node keeps the name it registered under, whatever the hostname says now" \ + 'node-name: "localhost.localdomain"' "$(cat "$dropin")" + +out="$(run_k3s_config "$kdir/kubelet.crt" "$kdir/k3s" renamed-host)" +expect "the certificate's name, once pinned, restarts nothing on a rerun" "RESTART=0" "$out" + +printf '# old\nwrite-kubeconfig-mode: "0600"\nnode-name: "first-name"\n' > "$dropin" +out="$(run_k3s_config "$kdir/kubelet.crt" "$kdir/k3s" renamed-host)" +expect "a pinned name outranks the certificate" 'node-name: "first-name"' "$(cat "$dropin")" + +# The three spellings of -subject: OpenSSL 1.1, and the slash form of 1.0 and LibreSSL. +rm -f "$dropin" +run_k3s_config "$kdir/kubelet.crt" "$kdir/k3s" x 'subject=O = system:nodes, CN = system:node:node-a' >/dev/null +expect "OpenSSL 1.1's subject is read" 'node-name: "node-a"' "$(cat "$dropin")" +rm -f "$dropin" +run_k3s_config "$kdir/kubelet.crt" "$kdir/k3s" x 'subject= /O=system:nodes/CN=system:node:node-b' >/dev/null +expect "the slash-form subject is read" 'node-name: "node-b"' "$(cat "$dropin")" + +rm -f "$dropin" +out="$(run_k3s_config "$kdir/none.crt" "$kdir/k3s" renamed-host)" +case "$(cat "$dropin")" in + *node-name*) echo "FAIL an installed k3s whose name cannot be read must not be pinned to the hostname"; fails=$((fails + 1)) ;; + *) echo "PASS an installed k3s whose name cannot be read is not pinned to a guess" ;; +esac +expect "an unpinned name is a warning" "WARN: could not read this node's k3s name" "$out" +rm -f "$dropin" +printf 'half-written\n' > "$kdir/broken.crt" +out="$(run_k3s_config "$kdir/broken.crt" "$kdir/k3s" renamed-host)" +expect "a certificate openssl cannot parse is a warning, not a failed install" "WARN: could not read this node's k3s name" "$out" +expect "and the drop-in is still written" 'write-kubeconfig-mode: "0600"' "$(cat "$dropin")" + +sblock="$(awk '/^strip_k3s_kubeconfig_mode_flag\(\) \{/,/^}/' "$BS")" +[ -n "$sblock" ] || { echo "FAIL: no strip_k3s_kubeconfig_mode_flag found in $BS"; exit 1; } +# ExecStart as k3s's installer writes it (copied off an install made with the old flag). +tab="$(printf '\t')" +cat > "$kdir/k3s.service" <&1 +} +out="$(run_strip)" +want="ExecStartPre=-/sbin/modprobe overlay +ExecStart=/usr/local/bin/k3s \\ + server \\ +${tab}'--disable' \\ +${tab}'metrics-server' \\ +" +if [ "$(cat "$kdir/k3s.service"; echo x)" = "${want} +x" ]; then + echo "PASS the old kubeconfig-mode flag and its value leave the k3s unit, the rest stays" +else + echo "FAIL the stripped unit is:"; cat "$kdir/k3s.service"; fails=$((fails + 1)) +fi +expect "a rewritten unit is reloaded" "SYSTEMCTL: daemon-reload" "$out" +expect "a rewritten unit asks for a k3s restart" "RESTART=1" "$out" +out="$(run_strip)" +expect "a unit without the flag is left alone" "RESTART=0" "$out" +case "$out" in *daemon-reload*) echo "FAIL a unit without the flag must not be reloaded"; fails=$((fails + 1)) ;; esac +printf "ExecStart=/usr/local/bin/k3s \\\\\n server \\\\\n\t'--write-kubeconfig-mode=644' \\\\\n\t'--disable' \\\\\n\t'traefik' \\\\\n" > "$kdir/k3s.service" +run_strip >/dev/null +expect "the one-word spelling goes too, and only that word" "$(printf " server \\\\\n\t'--disable' \\\\\n\t'traefik' \\\\")" "$(cat "$kdir/k3s.service")" +case "$(cat "$kdir/k3s.service")" in *kubeconfig-mode*) echo "FAIL the one-word flag is still in the unit"; fails=$((fails + 1)) ;; esac + +case "$(awk '/^run_k3s_installer\(\) \{/,/^}/' "$BS")" in + *write-kubeconfig-mode*) echo "FAIL the k3s installer must not be told a kubeconfig mode; the drop-in holds it"; fails=$((fails + 1)) ;; + *INSTALL_K3S_EXEC=*) echo "PASS the k3s installer leaves the kubeconfig mode to the drop-in" ;; + *) echo "FAIL: no run_k3s_installer found in $BS"; fails=$((fails + 1)) ;; +esac + +iblock="$(awk '/^install_k3s\(\) \{/,/^}/' "$BS")" +[ -n "$iblock" ] || { echo "FAIL: no install_k3s found in $BS"; exit 1; } +iblock="$iblock +$(awk '/^version_newer\(\) \{/,/^}/' "$BS") +$(awk '/^k3s_upgrade_allowed\(\) \{/,/^}/' "$BS")" +run_install_k3s() { # $1: installed version ("" = none), $2: drop-in changed (0|1), $3: FELIS_UPGRADE_DEPS + INSTALLED="$1" CHANGED="$2" UPGRADE="${3:-0}" KDIR="$kdir" bash -c ' + set -Eeuo pipefail + ok() { echo "OK: $*"; }; log() { echo "LOG: $*"; }; warn() { echo "WARN: $*"; } + die() { echo "DIE: $*"; exit 1; } + FELIS_K3S_VERSION=v1.36.4+k3s1 FELIS_UPGRADE_DEPS="$UPGRADE" K3S_BIN_DIR="$KDIR" K3S_BIN="$KDIR/k3s-under-test" + K3S_CONFIG_DROPIN=/etc/rancher/k3s/config.yaml.d/50-felis.yaml K3S_KUBECONFIG=/etc/rancher/k3s/k3s.yaml + rm -f "$K3S_BIN" + if [ -n "$INSTALLED" ]; then printf "#!/bin/sh\necho \"k3s version %s (abc)\"\n" "$INSTALLED" > "$K3S_BIN"; chmod +x "$K3S_BIN"; fi + configure_k3s_firewall() { :; } + write_k3s_config() { [ "$CHANGED" = 0 ] || K3S_RESTART_NEEDED=1; } + strip_k3s_kubeconfig_mode_flag() { :; } + run_k3s_installer() { echo "INSTALLER"; printf "#!/bin/sh\n" > "$K3S_BIN"; command chmod +x "$K3S_BIN"; } + systemctl() { echo "SYSTEMCTL: $*"; } + wait_for_node_ready() { echo "READY"; } + chmod() { echo "CHMOD: $*"; } + '"$iblock"' + install_k3s' 2>&1 +} +out="$(run_install_k3s v1.36.4+k3s1 1)" +expect "new k3s settings on a running k3s restart it" "SYSTEMCTL: restart k3s" "$out" +expect "the admin kubeconfig is made root-only once the node is up" "READY +CHMOD: 0600 /etc/rancher/k3s/k3s.yaml" "$out" +out="$(run_install_k3s v1.36.4+k3s1 0)" +case "$out" in *"restart k3s"*) echo "FAIL unchanged k3s settings must not restart k3s"; fails=$((fails + 1)) ;; *) echo "PASS unchanged k3s settings restart nothing" ;; esac +out="$(run_install_k3s "" 1)" +expect "a fresh host runs the k3s installer and waits for the node" "INSTALLER +SYSTEMCTL: enable --now k3s" "$out" +expect "a fresh host's kubeconfig is made root-only too" "CHMOD: 0600 /etc/rancher/k3s/k3s.yaml" "$out" +case "$out" in *"restart k3s"*) echo "FAIL the k3s installer already started k3s on the new settings; no second restart"; fails=$((fails + 1)) ;; *) echo "PASS a fresh k3s is not restarted a second time" ;; esac +out="$(run_install_k3s v1.35.2+k3s1 1 1)" +expect "an upgrade runs the k3s installer" "INSTALLER" "$out" +case "$out" in *"restart k3s"*) echo "FAIL the upgrade already restarted k3s on the new settings; no second restart"; fails=$((fails + 1)) ;; *) echo "PASS an upgraded k3s is not restarted a second time" ;; esac +rm -rf "$kdir" + +jblock="$(awk '/^ensure_persistent_journal\(\) \{/,/^}/' "$BS")" +[ -n "$jblock" ] || { echo "FAIL: no ensure_persistent_journal found in $BS"; exit 1; } +jdir="$(mktemp -d)" +jcalls="$(mktemp)" +run_journal() { # $1: FELIS_JOURNAL_MAX_USE, $2: FELIS_MANAGE_JOURNAL, $3: a journald restart creates the journal directory (1|0) + MAXUSE="$1" MANAGE="${2:-1}" MAKES="${3:-1}" DROPIN="$jdir/journald.conf.d/50-felis.conf" JDIR="$jdir/journal" \ + JCALLS="$jcalls" bash -c ' + set -Eeuo pipefail + ok() { echo "OK: $*"; }; log() { echo "LOG: $*"; }; warn() { echo "WARN: $*"; } + remember_temp() { :; } + # Where each temporary file lands, on stderr: stdout is the path the caller captures. + mktemp() { local p; p="$(command mktemp "$@")"; echo "MKTEMP: $(dirname "$p")" >&2; echo "$p"; } + restore_label() { echo "RELABEL: $*"; } + # journald creates the directory as it starts with Storage=persistent. + systemctl() { echo "SYSTEMCTL: $*"; [ "$MAKES" = 0 ] || mkdir -p "$JDIR"; } + journalctl() { echo "JOURNALCTL: $*" >> "$JCALLS"; } + FELIS_JOURNAL_MAX_USE="$MAXUSE" FELIS_MANAGE_JOURNAL="$MANAGE" JOURNALD_DROPIN="$DROPIN" JOURNAL_DIR="$JDIR" + '"$jblock"' + ensure_persistent_journal' 2>&1 +} +out="$(run_journal 1G)" +if [ "$(cat "$jdir/journald.conf.d/50-felis.conf")" = "[Journal] +Storage=persistent +SystemMaxUse=1G" ]; then + echo "PASS the journal is made persistent and capped" +else + echo "FAIL journald drop-in:"; echo "$out"; fails=$((fails + 1)) +fi +expect "a new journald drop-in restarts journald" "SYSTEMCTL: restart systemd-journald" "$out" +jmode="$(stat -c %a "$jdir/journald.conf.d/50-felis.conf" 2>/dev/null || stat -f %Lp "$jdir/journald.conf.d/50-felis.conf")" +if [ "$jmode" = 644 ]; then + echo "PASS the journald drop-in is readable like the rest of /etc/systemd (systemd-analyze cat-config)" +else + echo "FAIL the journald drop-in is mode $jmode, want 644"; fails=$((fails + 1)) +fi +expect "the drop-in gets its directory's SELinux label before journald reads it" "RELABEL: $jdir/journald.conf.d/50-felis.conf +SYSTEMCTL: restart systemd-journald" "$out" +expect "the runtime journal is flushed to disk" "JOURNALCTL: --flush" "$(cat "$jcalls")" +# A file made under /tmp and moved into place keeps user_tmp_t, which journald is denied. +if [ "$(printf '%s\n' "$out" | sed -n 's/^MKTEMP: //p' | sort -u)" = "$jdir/journald.conf.d" ]; then + echo "PASS the journald drop-in is made beside itself, never under /tmp" +else + echo "FAIL temporary files for the journald drop-in were made in: $(printf '%s\n' "$out" | sed -n 's/^MKTEMP: //p')"; fails=$((fails + 1)) +fi +expect "the journal directory journald made is the proof" "OK: system journal is persistent under $jdir/journal" "$out" +if [ "$(ls "$jdir/journald.conf.d")" = "50-felis.conf" ]; then + echo "PASS no temporary file is left beside the journald drop-in" +else + echo "FAIL journald.conf.d holds: $(ls "$jdir/journald.conf.d")"; fails=$((fails + 1)) +fi +out="$(run_journal 1G)" +case "$out" in *restart*) echo "FAIL a current drop-in journald has taken up must not restart it"; fails=$((fails + 1)) ;; *) echo "PASS a current, working drop-in restarts nothing" ;; esac +rmdir "$jdir/journal" +out="$(run_journal 1G)" +expect "a current drop-in journald never took up (no journal directory) restarts it" "LOG: journald has not taken up" "$out" +expect "and that restart happens" "SYSTEMCTL: restart systemd-journald" "$out" +rmdir "$jdir/journal" +out="$(run_journal 1G 1 0)" +expect "a journald that still stores nothing on disk is a warning" "WARN: journald did not create $jdir/journal" "$out" +out="$(run_journal 4G)" +expect "a new cap is written" "SystemMaxUse=4G" "$(cat "$jdir/journald.conf.d/50-felis.conf")" +expect "a new cap restarts journald" "SYSTEMCTL: restart systemd-journald" "$out" +rm -rf "$jdir" +out="$(run_journal 1G 0)" +if [ -e "$jdir/journald.conf.d/50-felis.conf" ]; then + echo "FAIL FELIS_MANAGE_JOURNAL=0 must leave journald alone"; fails=$((fails + 1)) +else + echo "PASS FELIS_MANAGE_JOURNAL=0 leaves journald alone" +fi +rm -f "$jcalls" + +tblock="$(awk '/^ensure_time_sync\(\) \{/,/^}/' "$BS")" +[ -n "$tblock" ] || { echo "FAIL: no ensure_time_sync found in $BS"; exit 1; } +run_time() { # $1: NTP now, $2: set-ntp works before chrony (0|1), $3: synchronizes (0|1), $4: FELIS_MANAGE_TIME_SYNC + NTP="$1" SETWORKS="$2" SYNCS="$3" MANAGE="${4:-1}" bash -c ' + set -Eeuo pipefail + ok() { echo "OK: $*"; }; log() { echo "LOG: $*"; }; warn() { echo "WARN: $*"; } + sleep() { :; } + pkg_install() { echo "PKG: $*"; SETWORKS=1; } + timedatectl() { + case "$*" in + "show -p NTP --value") echo "$NTP" ;; + "show -p NTPSynchronized --value") [ "$SYNCS" = 1 ] && echo yes || echo no ;; + "set-ntp true") echo "TIMEDATECTL: set-ntp true"; [ "$SETWORKS" = 1 ] || { echo "Failed to set ntp: NTP not supported" >&2; return 1; } ;; + *) echo "TIMEDATECTL?: $*" ;; + esac + } + FELIS_MANAGE_TIME_SYNC="$MANAGE" + '"$tblock"' + ensure_time_sync' 2>&1 +} +out="$(run_time no 1 1)" +expect "NTP is turned on where it is off" "TIMEDATECTL: set-ntp true" "$out" +expect "a synchronized clock is reported" "OK: system clock synchronized by NTP" "$out" +case "$out" in *PKG:*) echo "FAIL chrony must not be installed where timedatectl has a client to enable"; fails=$((fails + 1)) ;; esac +case "$out" in *WARN:*) echo "FAIL a synchronized clock must not warn"; fails=$((fails + 1)) ;; esac +out="$(run_time no 0 1)" +expect "with no NTP client to enable, chrony is installed" "PKG: chrony" "$out" +expect "and NTP is turned on after it" "PKG: chrony +TIMEDATECTL: set-ntp true" "$out" +out="$(run_time yes 1 1)" +case "$out" in *set-ntp*) echo "FAIL NTP already on must not be set again"; fails=$((fails + 1)) ;; *) echo "PASS NTP already on is left as it is" ;; esac +out="$(run_time yes 1 0)" +expect "a clock that does not synchronize is a warning, not a stop" "WARN: the system clock is not synchronized yet" "$out" +out="$(run_time no 1 1 0)" +case "$out" in *TIMEDATECTL*) echo "FAIL FELIS_MANAGE_TIME_SYNC=0 must leave the clock alone"; fails=$((fails + 1)) ;; *) echo "PASS FELIS_MANAGE_TIME_SYNC=0 leaves the clock alone" ;; esac + +dblock="$(awk '/^warn_dynamic_node_ip\(\) \{/,/^}/' "$BS")" +[ -n "$dblock" ] || { echo "FAIL: no warn_dynamic_node_ip found in $BS"; exit 1; } +run_dyn() { # $1: NODE_IP, $2: `ip -4 -o addr show` output + NODE_IP="$1" ADDRS="$2" bash -c ' + set -Eeuo pipefail + warn() { echo "WARN: $*"; } + ip() { printf "%s\n" "$ADDRS"; } + '"$dblock"' + warn_dynamic_node_ip; echo done' 2>&1 +} +# `ip -4 -o addr show` lines as iproute2 prints them (the leased one is off the test host). +lo='1: lo inet 127.0.0.1/8 scope host lo\ valid_lft forever preferred_lft forever' +leased='2: enp0s5 inet 10.211.55.6/24 brd 10.211.55.255 scope global dynamic noprefixroute enp0s5\ valid_lft 1459sec preferred_lft 1459sec' +static='2: enp0s5 inet 10.211.55.6/24 brd 10.211.55.255 scope global noprefixroute enp0s5\ valid_lft forever preferred_lft forever' +other='3: wlan0 inet 192.168.1.20/24 brd 192.168.1.255 scope global dynamic wlan0\ valid_lft 3000sec preferred_lft 3000sec' +out="$(run_dyn 10.211.55.6 "$lo +$leased")" +expect "a leased node address is a warning" "WARN: 10.211.55.6 is a DHCP lease" "$out" +out="$(run_dyn 10.211.55.6 "$lo +$static +$other")" +case "$out" in *WARN*) echo "FAIL a static node address must not warn because another interface is leased"; fails=$((fails + 1)) ;; *) echo "PASS a static node address does not warn" ;; esac +out="$(run_dyn 10.211.55.60 "$leased")" +case "$out" in *WARN*) echo "FAIL a different address with the same prefix is not the node's"; fails=$((fails + 1)) ;; *) echo "PASS only the node's own address is judged" ;; esac + +vsnip="$(awk '/local journal_n=/,/^ esac/' "$BS")" +[ -n "$vsnip" ] || { echo "FAIL: no FELIS_JOURNAL_MAX_USE check found in $BS"; exit 1; } +check_max_use() { + FELIS_JOURNAL_MAX_USE="$1" bash -c ' + die() { echo "DIE: $*"; exit 1; } + f() { + '"$vsnip"' + } + f; echo accepted' 2>&1 +} +for v in 1G 512M 900K; do + expect "journal cap $v is accepted" "accepted" "$(check_max_use "$v")" +done +for v in 1g G 01G 1.5G 1GB 2T 1024 ""; do + expect "journal cap '$v' is refused" "DIE: FELIS_JOURNAL_MAX_USE must be written" "$(check_max_use "$v")" +done + +order="$(awk '/^main\(\) \{/,/^}/' "$BS" | grep -nE '^[[:space:]]*(detect_node_ip|warn_dynamic_node_ip|install_base|ensure_time_sync|ensure_persistent_journal|install_k3s)$' | sed 's/^[0-9]*:[[:space:]]*//' | tr '\n' ' ')" +expect "main checks the address, then turns on NTP and the journal once packages install, before k3s" \ + "detect_node_ip warn_dynamic_node_ip install_base ensure_time_sync ensure_persistent_journal install_k3s " "$order" + # --------------------------------------------------------------------------------------- if [ "$fails" -eq 0 ]; then echo "ALL PASS" diff --git a/docs/operations.md b/docs/operations.md index 1729540..f87c561 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -39,6 +39,23 @@ cloudflared is left as it is, see §4): 32-bit hosts are not supported: there is no k3s, JRE or Go build the installer will fetch for them. +Two things the host must keep for as long as the install lives: + +- **Its address.** The install is bound to the IPv4 address it was made on (the + database connection string, `pg_hba.conf`, the network policies, the panel + certificate and the default nip.io domain all carry it). Give the host a static + address or a DHCP reservation before installing; the installer warns when the address + is a lease, and the watchdog reports `host-address` when the host loses it + (troubleshooting §13c). The k3s node name is pinned at install time, so a hostname + change is harmless. +- **A synchronized clock.** The installer turns NTP on (chrony where nothing else can) + and the watchdog reports a clock that stays unsynchronized. Allow outbound UDP 123, + or set `FELIS_MANAGE_TIME_SYNC=0` on a host whose clock is kept another way. + +The installer also makes the system journal persistent (capped at +`FELIS_JOURNAL_MAX_USE`, default 1G; `FELIS_MANAGE_JOURNAL=0` skips it) and writes the +admin kubeconfig `/etc/rancher/k3s/k3s.yaml` root-only: run `sudo k3s kubectl`. + One node is the whole supported shape. A world volume is a ReadWriteOnce claim on the node's local-path storage, so a game server's pod is pinned to the node that first scheduled it and cannot move when that node fails; the operator and felis-api each run diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index d953371..b957a0d 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -8,7 +8,9 @@ see to the code path that emitted it. ## How to read this document -Each entry is **symptom → likely cause → where to look → fix**. Signals are +Each entry is **symptom → likely cause → where to look → fix**. The `kubectl` +commands run as root on the node (`sudo -i`, or `sudo k3s kubectl …`): the admin +kubeconfig `/etc/rancher/k3s/k3s.yaml` is readable by root only (§13c). Signals are graded for how far the in-repo Go test suite proves the behaviour: - **[GO-TESTED]** — a hermetic `*_test.go` exercises this exact path; the @@ -1331,6 +1333,48 @@ the registry), or for a single image deliberate second copy on the node; treat it as the recovery path, not as free space. +## 13c. The host's address, name or clock changed + +An install is bound to the address it was made on. `bootstrap.sh` writes that +address into the database connection string, `pg_hba.conf`, the network +policies, the panel certificate and the default `.nip.io` root domain, and +nothing re-addresses a live install. When the host loses the address (a DHCP +lease that came back different, a moved VM), felis-api cannot reach PostgreSQL +and the panel stops answering on its old name. The watchdog reports it as +`host-address` (critical, after 5 minutes). The installer warns at install time +when the address is a DHCP lease. + +Remedy: give the host its old address back, either as a DHCP reservation on +the router or as a static address (`nmcli con mod ipv4.method manual +ipv4.addresses / ipv4.gateway ipv4.dns && nmcli con up +` on Rocky), then restart felis-api (`sudo k3s kubectl -n felis rollout restart +deploy/felis-api`) and the proxy (`sudo systemctl restart felis-velocity`). Moving an install +to a new address is a reinstall onto a restored backup (docs/operations.md §5). + +The node **name** is pinned. Every local-path volume (worlds, registry, +uploads, backups) is bound to its node by name, and k3s takes the name from the +hostname unless told otherwise, so renaming the host used to bring k3s back as +a second, empty node with every volume Pending on the old one. The installer +pins the name in `/etc/rancher/k3s/config.yaml.d/50-felis.yaml` +(`node-name:`); a hostname change is then harmless. Check the pin with `sudo +k3s kubectl get node -o jsonpath='{.items[0].metadata.annotations.k3s\.io/node-args}'`. +That file also sets `write-kubeconfig-mode: "0600"`: the admin kubeconfig +`/etc/rancher/k3s/k3s.yaml` is cluster-admin and readable by root only, so +use `sudo k3s kubectl` (or `sudo -E kubectl`). + +The **clock** must be kept by NTP. Sign-in codes and sessions expire by it, +S3 refuses off-site uploads signed more than 15 minutes off, and certificate +checks fail on a clock far off. The installer turns NTP on (`timedatectl +set-ntp true`, installing chrony where there is no client to enable) unless +`FELIS_MANAGE_TIME_SYNC=0`. The watchdog reports an unsynchronized clock as +`clock` (warning, after 30 minutes). Check with `timedatectl` (want `System +clock synchronized: yes` and `NTP service: active`) and `chronyc sources`; +a firewall that drops outbound UDP 123 keeps it unsynchronized. + +The system journal is persistent (`/etc/systemd/journald.conf.d/50-felis.conf`, +capped by `FELIS_JOURNAL_MAX_USE`, default 1G), so `journalctl -b -1` shows the +boot before a reboot. + --- ## 14. Health alerts, and metrics for diagnosis (spec §23) @@ -1359,6 +1403,8 @@ Every two minutes the host checks: | Newest control-plane database backup over 26h old, or none (§16) | 10 min | critical | | A watched filesystem below 15% free (below 5%: critical) | 15 min (5 min) | warning | | Host memory available below 10% | 15 min | warning | +| The host no longer holds the address the install was made on (§13c) | 5 min | critical | +| The system clock is not synchronized by NTP (§13c) | 30 min | warning | How it mails: @@ -1388,8 +1434,9 @@ A healthy run logs `every check passed`. Otherwise it logs one line per finding, and the mail's subject once one is sent. A `-dry-run` from a shell uses the command's defaults, and those do not include -the game-proxy check. The unit carries `-proxy-addr 127.0.0.1:` and -the disk list the install chose. `systemctl cat felis-watchdog` shows both. +the game-proxy or the address check. The unit carries `-proxy-addr +127.0.0.1:`, `-node-ip ` and the disk list the +install chose. `systemctl cat felis-watchdog` shows them. ### Metrics