diff --git a/deploy/bootstrap.sh b/deploy/bootstrap.sh index 960f43d..68758f0 100644 --- a/deploy/bootstrap.sh +++ b/deploy/bootstrap.sh @@ -321,6 +321,15 @@ WATCHDOG_QUIET_FILE="/run/felis/watchdog-quiet-until" VELOCITY_DIR="/opt/felis/velocity" VELOCITY_USER="felis-velocity" VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service" +# What the running proxy was last started from (velocity_fingerprint), and the builds the +# login and lobby pods were last started on: a rerun restarts only what actually changed, +# since each of those restarts disconnects every player online. +VELOCITY_FINGERPRINT="${STATE_DIR}/velocity.fingerprint" +SYSTEM_SERVER_IMAGES="${STATE_DIR}/system-server-images" +# The nftables rules that keep PostgreSQL's port to this host and its pods, on a host +# without firewalld (configure_postgres_firewall). +PG_FIREWALL_RULES="${STATE_DIR}/postgres-firewall.nft" +PG_FIREWALL_SERVICE="/etc/systemd/system/felis-postgres-firewall.service" JRE_DIR="/opt/felis/jre" K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}" K3S_BIN="${K3S_BIN_DIR}/k3s" @@ -783,7 +792,18 @@ pkg_refresh_once() { apt) apt_get update -y ;; dnf|yum) : ;; # dnf/yum refresh metadata on demand zypper) wait_for_pkg_locks; zypper --non-interactive refresh ;; - pacman) wait_for_pkg_locks; pacman -Syu --noconfirm ;; + # Arch supports only whole-system upgrades (-Sy alone leaves a partial upgrade), so the + # refresh stays -Syu, with PostgreSQL held back once a cluster exists: a new major + # version cannot open the old data directory, and the upgrade would take the platform's + # database down on an unrelated rerun. check_postgres_major explains the way forward. + pacman) + wait_for_pkg_locks + if [ -f "$(postgres_data_dir)/PG_VERSION" ]; then + pacman -Syu --noconfirm --ignore postgresql + else + pacman -Syu --noconfirm + fi + ;; esac _PKG_REFRESHED=1 } @@ -1791,6 +1811,12 @@ build_game_stack() { --build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \ -t "$FELIS_PAPER_IMAGE" "$GAME_STACK_DIR" + # The builds the system servers run, for restart_existing_system_servers. Docker's layer + # cache gives an unchanged build the same id, so a rerun that rebuilt nothing leaves the + # login and lobby pods (and every player on them) alone. + LIMBO_IMAGE_ID="$(docker image inspect -f '{{.Id}}' "$FELIS_LIMBO_IMAGE")" + LOBBY_IMAGE_ID="$(docker image inspect -f '{{.Id}}' "$FELIS_LOBBY_IMAGE")" + local img for img in "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do log "importing ${img} into k3s containerd" @@ -2187,12 +2213,36 @@ WantedBy=multi-user.target EOF systemctl daemon-reload systemctl enable felis-velocity - # restart, not `enable --now`: on a re-run the old proxy is already up and --now would - # leave it running against the new config. + # A restart disconnects every player on the network, so a rerun that changed nothing the + # proxy runs leaves it up. Otherwise restart, not `enable --now`: on a re-run the old proxy + # is already up and --now would leave it running against the new config. The fingerprint is + # recorded only after the restart, so a run that died between writing and restarting still + # restarts next time. + local fp + fp="$(velocity_fingerprint)" + if systemctl is-active --quiet felis-velocity \ + && [ "$fp" = "$(cat "$VELOCITY_FINGERPRINT" 2>/dev/null || true)" ]; then + ok "felis-velocity unchanged; left running (0.0.0.0:${FELIS_GAME_PORT})" + return 0 + fi systemctl restart felis-velocity + printf '%s\n' "$fp" > "$VELOCITY_FINGERPRINT" ok "felis-velocity.service enabled and started (0.0.0.0:${FELIS_GAME_PORT})" } +# velocity_fingerprint hashes what the proxy process runs: its unit (JVM flags and system +# properties), the JRE, the jars and the files the installer writes for it. The Via config +# and whatever else plugins write at runtime stay out; Via rewrites its config on every load. +velocity_fingerprint() { + local f + for f in "$VELOCITY_SERVICE" "${JRE_DIR}/release" "${VELOCITY_DIR}/velocity.jar" \ + "${VELOCITY_DIR}/velocity.toml" "${VELOCITY_DIR}/forwarding.secret" \ + "${VELOCITY_DIR}/plugins/felis-link/felis-link.properties" "${VELOCITY_DIR}"/plugins/*.jar; do + [ -f "$f" ] || continue + printf '%s %s\n' "$(sha256sum <"$f" | cut -d' ' -f1)" "$f" + done | sha256sum | cut -d' ' -f1 +} + configure_velocity_firewall() { command -v firewall-cmd >/dev/null 2>&1 || return 0 systemctl is-active --quiet firewalld || return 0 @@ -2285,17 +2335,44 @@ install_postgres() { fi init_postgres_data_dir + check_postgres_major systemctl enable --now postgresql ok "postgresql running" } +# check_postgres_major refuses to start a PostgreSQL server whose major version differs from +# the one that created the data directory. The server would not start anyway; this says why +# and what to do, instead of a failed unit in the middle of the install. Debian and Ubuntu +# keep one cluster per version under /var/lib/postgresql/ and upgrade with +# pg_upgradecluster, so they are left to their own tooling. +check_postgres_major() { + local data_dir have want + [ "$PKG" = "apt" ] && return 0 + data_dir="$(postgres_data_dir)" + [ -f "${data_dir}/PG_VERSION" ] || return 0 + have="$(tr -d '[:space:]' < "${data_dir}/PG_VERSION")" + want="$( (postgres --version 2>/dev/null || psql --version 2>/dev/null) | head -n 1 \ + | sed -nE 's/^[^0-9]*([0-9]+)\..*/\1/p')" + [ -n "$want" ] && [ "$have" != "$want" ] || return 0 + die "PostgreSQL ${want} is installed, but ${data_dir} holds a PostgreSQL ${have} cluster. + The server cannot open it. Upgrade the cluster first (pg_upgrade, with the ${have} binaries + still installed), or reinstall PostgreSQL ${have}, then rerun the installer. Take a + database bundle before either: sudo felis db backup" +} + configure_postgres() { - local cfg hba + local cfg hba listen cfg="$(as_postgres psql -tAc 'SHOW config_file;' 2>/dev/null || true)" hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)" [ -n "$cfg" ] && [ -n "$hba" ] || die "could not query postgresql config/hba file paths" + listen="$(as_postgres psql -tAc 'SHOW listen_addresses;' 2>/dev/null || true)" - # Listen on all interfaces (applied on restart). ALTER SYSTEM is idempotent. + # Listen on all interfaces (applied on restart). ALTER SYSTEM is idempotent. Pods reach the + # database at the node IP, and configure_postgres_firewall keeps the port from everyone else. + # + # The connection itself is not encrypted (sslmode=disable in felis.toml). On this + # single-node shape it never leaves the host: pods reach the node IP over their veth pair + # and the host binary uses loopback, so TLS would guard a path no other machine is on. as_postgres psql -v ON_ERROR_STOP=1 -c "ALTER SYSTEM SET listen_addresses = '*';" >/dev/null # Allow the host loopback, the pod CIDR, and the node IP before broader distro defaults. @@ -2317,10 +2394,77 @@ SQL as_postgres createdb -O "$DB_USER" "$DB_NAME" fi - systemctl restart postgresql + # listen_addresses is the only setting here that needs a restart, and a restart cuts + # every connection felis-api holds mid-transaction. pg_hba.conf and the role are live + # after a reload. + if [ "$listen" = "*" ]; then + as_postgres psql -v ON_ERROR_STOP=1 -tAc 'SELECT pg_reload_conf();' >/dev/null + else + systemctl restart postgresql + fi + configure_postgres_firewall ok "postgresql configured (listen=*, role/db '${DB_NAME}', pg_hba opened to pods)" } +# configure_postgres_firewall keeps 5432 to this host and its pods. PostgreSQL listens on +# every address (pods dial the node IP), and pg_hba.conf only refuses a connection after the +# server has spoken to it. firewalld's default zone does not open 5432 (configure_k3s_firewall +# trusts only the pod and service CIDRs), so that host needs nothing more; on any other host a +# small nftables table of our own drops 5432 from everywhere but loopback, the pod CIDR and the +# node's own address, loaded at boot by a oneshot unit ordered before PostgreSQL. +configure_postgres_firewall() { + if command -v firewall-cmd >/dev/null 2>&1 && systemctl is-active --quiet firewalld; then + return 0 + fi + command -v nft >/dev/null 2>&1 || pkg_install nftables + command -v nft >/dev/null 2>&1 || { warn "nft is not available; PostgreSQL's 5432 is reachable from the network (pg_hba still refuses other hosts)"; return 0; } + local node_rule + case "$NODE_IP" in + *:*) node_rule="ip6 saddr ${NODE_IP}" ;; + *) node_rule="ip saddr ${NODE_IP}" ;; + esac + install -d -m 0700 "$STATE_DIR" + cat > "$PG_FIREWALL_RULES" < "$PG_FIREWALL_SERVICE" </dev/null 2>&1 + systemctl restart felis-postgres-firewall.service + ok "5432 accepts loopback, ${POD_CIDR} and ${NODE_IP} only (nftables table felis_postgres)" +} + # --------------------------------------------------------------------------- # 7. Secrets + felis.toml (pod variant reaches Postgres at the node IP; host # variant at 127.0.0.1 for migrations) @@ -3171,18 +3315,32 @@ push_version_tag() { # The login/lobby images use mutable :demo tags. Importing/pushing a replacement # updates containerd, but an existing StatefulSet template is byte-for-byte -# unchanged and Kubernetes will not roll it. Recreate only the two always-on -# system pods so a convergent bootstrap actually starts the images it just built. +# unchanged and Kubernetes will not roll it. Recreate the two always-on system +# pods so a convergent bootstrap actually starts the images it just built — but +# only when the build changed: every player online is on one of these two, and a +# rerun that rebuilt nothing has nothing to start. SYSTEM_SERVER_IMAGES records the +# build each was last started on; it is written after the restarts, so a run that +# died in between restarts them next time. restart_existing_system_servers() { - local name pods - for name in "$LOGIN_SERVER" "$LOBBY_SERVER"; do + local name id pods next="" + while read -r name id; do + [ -n "$name" ] || continue + next="${next}${name} ${id}"$'\n' pods="$(kube -n "$MINECRAFT_NS" get pod \ -l "felis.lolicon.best/server=${name}" -o name 2>/dev/null || true)" [ -n "$pods" ] || continue + if [ -n "$id" ] && grep -qxF "${name} ${id}" "$SYSTEM_SERVER_IMAGES" 2>/dev/null; then + ok "${name} system server already runs this build; left running" + continue + fi log "restarting existing ${name} system server to pick up its imported image" kube -n "$MINECRAFT_NS" delete pod \ -l "felis.lolicon.best/server=${name}" --wait=false - done + done < "$SYSTEM_SERVER_IMAGES" } diagnose_rollout() { diff --git a/deploy/bootstrap_test.sh b/deploy/bootstrap_test.sh index dffe56a..9ef092f 100644 --- a/deploy/bootstrap_test.sh +++ b/deploy/bootstrap_test.sh @@ -1583,6 +1583,204 @@ case "$mainblock" in *) echo "FAIL main must call resolve_felis_image right before build_image"; fails=$((fails + 1)) ;; esac +# --- a rerun restarts only what changed ------------------------------------------------- +# The proxy, the login and lobby pods and PostgreSQL each disconnect every player (or cut +# felis-api's transactions) when restarted, so a rerun that changed none of them must leave +# them running, and one that changed a thing must still restart it. + +vsblock="$(awk '/^install_velocity_service\(\) \{/,/^}/' "$BS"; awk '/^velocity_fingerprint\(\) \{/,/^}/' "$BS")" +[ -n "$vsblock" ] || { echo "FAIL: no install_velocity_service found in $BS"; exit 1; } +vdir="$(mktemp -d)" +mkdir -p "$vdir/v/plugins/felis-link" "$vdir/jre" +printf 'jar\n' > "$vdir/v/velocity.jar" +printf 'plugin\n' > "$vdir/v/plugins/felis-velocity.jar" +printf 'JAVA_VERSION="25"\n' > "$vdir/jre/release" +run_velocity_service() { # is-active(0|1) + ACTIVE="$1" VELOCITY_SERVICE="$vdir/unit" VELOCITY_DIR="$vdir/v" JRE_DIR="$vdir/jre" \ + VELOCITY_FINGERPRINT="$vdir/fp" VELOCITY_USER=felis-velocity FELIS_GAME_PORT=25565 \ + FELIS_LEGACY_FORWARDING_SERVERS= bash -c ' + set -Eeuo pipefail + ok() { printf "OK: %s\n" "$*"; } + felis_internal_ip() { printf "10.43.0.9"; } + systemctl() { + case "$1" in + is-active) [ "$ACTIVE" = 1 ] ;; + *) printf "SYSTEMCTL %s\n" "$*" ;; + esac + } + '"$vsblock"' + install_velocity_service' +} +out="$(run_velocity_service 1)" +expect "a proxy with no recorded start is restarted" "SYSTEMCTL restart felis-velocity" "$out" +[ -s "$vdir/fp" ] && echo "PASS the restart records what the proxy runs" \ + || { echo "FAIL no fingerprint was recorded after the restart"; fails=$((fails + 1)); } +out="$(run_velocity_service 1)" +case "$out" in + *"SYSTEMCTL restart"*) echo "FAIL an unchanged rerun restarted the proxy"; fails=$((fails + 1)) ;; + *"felis-velocity unchanged; left running"*) echo "PASS an unchanged rerun leaves the proxy running" ;; + *) echo "FAIL install_velocity_service died on an unchanged rerun: $out"; fails=$((fails + 1)) ;; +esac +printf 'plugin v2\n' > "$vdir/v/plugins/felis-velocity.jar" +expect "a changed plugin jar restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1)" +expect "a stopped proxy is started whatever the fingerprint" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 0)" +printf 'JAVA_VERSION="25.0.1"\n' > "$vdir/jre/release" +expect "a patched JRE restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1)" +rm -rf "$vdir" + +ssblock="$(awk '/^restart_existing_system_servers\(\) \{/,/^}/' "$BS")" +[ -n "$ssblock" ] || { echo "FAIL: no restart_existing_system_servers found in $BS"; exit 1; } +sdir2="$(mktemp -d)" +run_system_restart() { # limbo-id lobby-id pods(0|1) + LIMBO_IMAGE_ID="$1" LOBBY_IMAGE_ID="$2" PODS="$3" SYSTEM_SERVER_IMAGES="$sdir2/state" \ + LOGIN_SERVER=login LOBBY_SERVER=lobby MINECRAFT_NS=minecraft bash -c ' + set -Eeuo pipefail + ok() { printf "OK: %s\n" "$*"; } + log() { :; } + kube() { + case "$*" in + *"get pod"*) [ "$PODS" = 1 ] && printf "pod/x-0\n" || true ;; + *delete*) printf "KUBE %s\n" "$*" ;; + esac + } + '"$ssblock"' + restart_existing_system_servers' +} +out="$(run_system_restart sha256:aaa sha256:bbb 1)" +expect "an unrecorded login pod is restarted" "delete pod -l felis.lolicon.best/server=login" "$out" +expect "an unrecorded lobby pod is restarted" "delete pod -l felis.lolicon.best/server=lobby" "$out" +out="$(run_system_restart sha256:aaa sha256:bbb 1)" +case "$out" in + *delete*) echo "FAIL an unchanged rebuild restarted a system server"; fails=$((fails + 1)) ;; + *"already runs this build"*) echo "PASS an unchanged rebuild leaves the system servers running" ;; + *) echo "FAIL restart_existing_system_servers died on an unchanged rebuild: $out"; fails=$((fails + 1)) ;; +esac +out="$(run_system_restart sha256:aaa sha256:ccc 1)" +expect "a new lobby build restarts the lobby" "delete pod -l felis.lolicon.best/server=lobby" "$out" +case "$out" in + *"server=login"*) echo "FAIL a new lobby build restarted the login gate too"; fails=$((fails + 1)) ;; + *) echo "PASS a new lobby build leaves the login gate running" ;; +esac +rm -f "$sdir2/state" +out="$(run_system_restart sha256:aaa sha256:ccc 0)" +case "$out" in + *delete*) echo "FAIL a missing pod was deleted"; fails=$((fails + 1)) ;; + *) echo "PASS no pod, nothing to restart" ;; +esac +expect "the builds are recorded even before the pods exist" "lobby sha256:ccc" "$(cat "$sdir2/state")" +rm -rf "$sdir2" + +pgblock="$(awk '/^configure_postgres\(\) \{/,/^}/' "$BS")" +[ -n "$pgblock" ] || { echo "FAIL: no configure_postgres found in $BS"; exit 1; } +pgcalls="$(mktemp)" +run_configure_pg() { # current listen_addresses + : > "$pgcalls" + LISTEN="$1" CALLS="$pgcalls" DB_NAME=felis DB_USER=felis DB_PASSWORD=pw bash -c ' + set -Eeuo pipefail + ok() { :; } + die() { printf "DIE: %s\n" "$*"; exit 1; } + as_postgres() { + case "$*" in + *"SHOW config_file"*) echo /c ;; + *"SHOW hba_file"*) echo /h ;; + *"SHOW listen_addresses"*) echo "$LISTEN" ;; + *pg_reload_conf*) echo RELOAD >> "$CALLS" ;; + *"SELECT 1 FROM pg_database"*) echo 1 ;; + *) cat >/dev/null ;; + esac + } + write_pg_hba_block() { :; } + configure_postgres_firewall() { echo FIREWALL; } + systemctl() { printf "SYSTEMCTL %s\n" "$*"; } + '"$pgblock"' + configure_postgres' < /dev/null + cat "$pgcalls" +} +out="$(run_configure_pg '*')" +case "$out" in + *"SYSTEMCTL restart postgresql"*) echo "FAIL a rerun restarted PostgreSQL under felis-api"; fails=$((fails + 1)) ;; + *RELOAD*) echo "PASS a rerun reloads PostgreSQL instead of restarting it" ;; + *) echo "FAIL configure_postgres neither reloaded nor restarted: $out"; fails=$((fails + 1)) ;; +esac +expect "the firewall step runs" "FIREWALL" "$out" +expect "a first install restarts PostgreSQL to listen on the node" "SYSTEMCTL restart postgresql" "$(run_configure_pg localhost)" +rm -f "$pgcalls" + +pmblock="$(awk '/^check_postgres_major\(\) \{/,/^}/' "$BS")" +[ -n "$pmblock" ] || { echo "FAIL: no check_postgres_major found in $BS"; exit 1; } +pmdir="$(mktemp -d)" +run_pg_major() { # PKG cluster-version server-version + printf '%s\n' "$2" > "$pmdir/PG_VERSION" + PKG="$1" SERVER="$3" DATA="$pmdir" bash -c ' + die() { printf "DIE: %s\n" "$*"; exit 1; } + postgres_data_dir() { printf "%s\n" "$DATA"; } + postgres() { printf "postgres (PostgreSQL) %s\n" "$SERVER"; } + '"$pmblock"' + check_postgres_major && echo STARTS' +} +expect "a new major version over an old cluster is refused" "DIE: PostgreSQL 17 is installed, but $pmdir holds a PostgreSQL 16 cluster" \ + "$(run_pg_major dnf 16 17.2)" +expect "the refusal names the way forward" "pg_upgrade" "$(run_pg_major pacman 16 17.2)" +expect "the same major version starts" "STARTS" "$(run_pg_major dnf 16 16.4)" +expect "Debian-family clusters are left to pg_upgradecluster" "STARTS" "$(run_pg_major apt 15 17.2)" +rm -rf "$pmdir" + +rfblock="$(awk '/^pkg_refresh_once\(\) \{/,/^}/' "$BS")" +pmdir="$(mktemp -d)" +run_refresh() { + PKG=pacman DATA="$pmdir" bash -c ' + wait_for_pkg_locks() { :; } + postgres_data_dir() { printf "%s\n" "$DATA"; } + pacman() { printf "PACMAN %s\n" "$*"; } + '"$rfblock"' + pkg_refresh_once' +} +case "$(run_refresh)" in + *--ignore*) echo "FAIL a host with no cluster yet held PostgreSQL back"; fails=$((fails + 1)) ;; + *"PACMAN -Syu"*) echo "PASS a fresh Arch host upgrades normally" ;; + *) echo "FAIL pkg_refresh_once did not refresh pacman"; fails=$((fails + 1)) ;; +esac +printf '16\n' > "$pmdir/PG_VERSION" +expect "an Arch rerun holds PostgreSQL at the cluster's version" "PACMAN -Syu --noconfirm --ignore postgresql" "$(run_refresh)" +rm -rf "$pmdir" + +# The rules heredoc closes nft blocks with a bare "}", so stop at the function's own brace. +fwblock="$(awk '/^configure_postgres_firewall\(\) \{/ { f = 1 } + f { print; if ($0 ~ /</dev/null +expect "a v6 node address gets a v6 rule" "tcp dport 5432 ip6 saddr 2001:db8::7 accept" "$(cat "$fwdir/pg.nft")" +rm -rf "$fwdir" + # --------------------------------------------------------------------------------------- if [ "$fails" -eq 0 ]; then diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index ab665f1..4c715b8 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -1356,6 +1356,28 @@ picks up the new release on its next start. A release that changes the game pod in any other way still restarts running servers once, as a server edit does. +**What a rerun restarts.** Each of these drops every connected player (or cuts +felis-api's open transactions), so the installer restarts one only when what it +runs changed: + +| Component | Restarted when | +|---|---| +| `felis-velocity` (the proxy) | its unit, the JRE, `velocity.jar`, `velocity.toml`, the forwarding secret, the felis-link settings or a plugin jar changed, or it was not running. The fingerprint lives in `/etc/felis/velocity.fingerprint`; delete it to force a restart. | +| login and lobby pods | the rebuilt limbo or lobby image has a new image ID (`/etc/felis/system-server-images`). Each restarts on its own. | +| PostgreSQL | first install only (`listen_addresses` needs a restart). A rerun reloads the configuration, which keeps connections open. | +| felis-api, felis-operator | the image tag changed (an upgrade), or a same-version rerun rebuilt it. | + +**PostgreSQL across reruns.** On hosts without firewalld the installer loads an +nftables table, `inet felis_postgres`, from `felis-postgres-firewall.service`: +port 5432 accepts loopback, the pod network and the node's own address and drops +everything else (`nft list table inet felis_postgres`). firewalld hosts already +keep 5432 closed to the network. The installer also refuses to start a +PostgreSQL whose major version differs from the cluster in the data directory, +and prints the `pg_upgrade` steps; distributions that move the server package to +a new major (Arch, Fedora) would otherwise leave the database unable to start. +On Arch the installer's `pacman -Syu` holds `postgresql` back once a cluster +exists, so the database is upgraded only when you run `pg_upgrade` yourself. + `rollout undo` reverts the image only. The upgrade's database migrations stay applied; when they are the problem, restore the `pre-migrate` bundle the upgrade took (§16, "Roll back an upgrade that broke the database"). diff --git a/plugins/velocity/build.gradle b/plugins/velocity/build.gradle index faff81a..81fa67d 100644 --- a/plugins/velocity/build.gradle +++ b/plugins/velocity/build.gradle @@ -64,3 +64,11 @@ sourceSets { tasks.withType(JavaCompile).configureEach { options.encoding = 'UTF-8' } + +// Byte-identical jars from identical sources. deploy/bootstrap.sh restarts the proxy only when +// a file it runs changed (velocity_fingerprint), and build-time timestamps in the jar would +// make every rebuild look like a change and disconnect every player on each rerun. +tasks.withType(AbstractArchiveTask).configureEach { + preserveFileTimestamps = false + reproducibleFileOrder = true +}