fix(bootstrap): 重跑只重启有变化的 Velocity、系统服与 PostgreSQL,无 firewalld 时用 nftables 收口 5432,PG 大版本不符拒绝启动
This commit is contained in:
4 files changed
+396
-10
No files matched your search
+168
-10
@@ -321,6 +321,15 @@ WATCHDOG_QUIET_FILE="/run/felis/watchdog-quiet-until"
|
|||||||
VELOCITY_DIR="/opt/felis/velocity"
|
VELOCITY_DIR="/opt/felis/velocity"
|
||||||
VELOCITY_USER="felis-velocity"
|
VELOCITY_USER="felis-velocity"
|
||||||
VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service"
|
VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service"
|
||||||
|
# What the running proxy was last started from (velocity_fingerprint), and the builds the
|
||||||
|
# login and lobby pods were last started on: a rerun restarts only what actually changed,
|
||||||
|
# since each of those restarts disconnects every player online.
|
||||||
|
VELOCITY_FINGERPRINT="${STATE_DIR}/velocity.fingerprint"
|
||||||
|
SYSTEM_SERVER_IMAGES="${STATE_DIR}/system-server-images"
|
||||||
|
# The nftables rules that keep PostgreSQL's port to this host and its pods, on a host
|
||||||
|
# without firewalld (configure_postgres_firewall).
|
||||||
|
PG_FIREWALL_RULES="${STATE_DIR}/postgres-firewall.nft"
|
||||||
|
PG_FIREWALL_SERVICE="/etc/systemd/system/felis-postgres-firewall.service"
|
||||||
JRE_DIR="/opt/felis/jre"
|
JRE_DIR="/opt/felis/jre"
|
||||||
K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}"
|
K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}"
|
||||||
K3S_BIN="${K3S_BIN_DIR}/k3s"
|
K3S_BIN="${K3S_BIN_DIR}/k3s"
|
||||||
@@ -783,7 +792,18 @@ pkg_refresh_once() {
|
|||||||
apt) apt_get update -y ;;
|
apt) apt_get update -y ;;
|
||||||
dnf|yum) : ;; # dnf/yum refresh metadata on demand
|
dnf|yum) : ;; # dnf/yum refresh metadata on demand
|
||||||
zypper) wait_for_pkg_locks; zypper --non-interactive refresh ;;
|
zypper) wait_for_pkg_locks; zypper --non-interactive refresh ;;
|
||||||
pacman) wait_for_pkg_locks; pacman -Syu --noconfirm ;;
|
# Arch supports only whole-system upgrades (-Sy alone leaves a partial upgrade), so the
|
||||||
|
# refresh stays -Syu, with PostgreSQL held back once a cluster exists: a new major
|
||||||
|
# version cannot open the old data directory, and the upgrade would take the platform's
|
||||||
|
# database down on an unrelated rerun. check_postgres_major explains the way forward.
|
||||||
|
pacman)
|
||||||
|
wait_for_pkg_locks
|
||||||
|
if [ -f "$(postgres_data_dir)/PG_VERSION" ]; then
|
||||||
|
pacman -Syu --noconfirm --ignore postgresql
|
||||||
|
else
|
||||||
|
pacman -Syu --noconfirm
|
||||||
|
fi
|
||||||
|
;;
|
||||||
esac
|
esac
|
||||||
_PKG_REFRESHED=1
|
_PKG_REFRESHED=1
|
||||||
}
|
}
|
||||||
@@ -1791,6 +1811,12 @@ build_game_stack() {
|
|||||||
--build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \
|
--build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \
|
||||||
-t "$FELIS_PAPER_IMAGE" "$GAME_STACK_DIR"
|
-t "$FELIS_PAPER_IMAGE" "$GAME_STACK_DIR"
|
||||||
|
|
||||||
|
# The builds the system servers run, for restart_existing_system_servers. Docker's layer
|
||||||
|
# cache gives an unchanged build the same id, so a rerun that rebuilt nothing leaves the
|
||||||
|
# login and lobby pods (and every player on them) alone.
|
||||||
|
LIMBO_IMAGE_ID="$(docker image inspect -f '{{.Id}}' "$FELIS_LIMBO_IMAGE")"
|
||||||
|
LOBBY_IMAGE_ID="$(docker image inspect -f '{{.Id}}' "$FELIS_LOBBY_IMAGE")"
|
||||||
|
|
||||||
local img
|
local img
|
||||||
for img in "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do
|
for img in "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do
|
||||||
log "importing ${img} into k3s containerd"
|
log "importing ${img} into k3s containerd"
|
||||||
@@ -2187,12 +2213,36 @@ WantedBy=multi-user.target
|
|||||||
EOF
|
EOF
|
||||||
systemctl daemon-reload
|
systemctl daemon-reload
|
||||||
systemctl enable felis-velocity
|
systemctl enable felis-velocity
|
||||||
# restart, not `enable --now`: on a re-run the old proxy is already up and --now would
|
# A restart disconnects every player on the network, so a rerun that changed nothing the
|
||||||
# leave it running against the new config.
|
# proxy runs leaves it up. Otherwise restart, not `enable --now`: on a re-run the old proxy
|
||||||
|
# is already up and --now would leave it running against the new config. The fingerprint is
|
||||||
|
# recorded only after the restart, so a run that died between writing and restarting still
|
||||||
|
# restarts next time.
|
||||||
|
local fp
|
||||||
|
fp="$(velocity_fingerprint)"
|
||||||
|
if systemctl is-active --quiet felis-velocity \
|
||||||
|
&& [ "$fp" = "$(cat "$VELOCITY_FINGERPRINT" 2>/dev/null || true)" ]; then
|
||||||
|
ok "felis-velocity unchanged; left running (0.0.0.0:${FELIS_GAME_PORT})"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
systemctl restart felis-velocity
|
systemctl restart felis-velocity
|
||||||
|
printf '%s\n' "$fp" > "$VELOCITY_FINGERPRINT"
|
||||||
ok "felis-velocity.service enabled and started (0.0.0.0:${FELIS_GAME_PORT})"
|
ok "felis-velocity.service enabled and started (0.0.0.0:${FELIS_GAME_PORT})"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# velocity_fingerprint hashes what the proxy process runs: its unit (JVM flags and system
|
||||||
|
# properties), the JRE, the jars and the files the installer writes for it. The Via config
|
||||||
|
# and whatever else plugins write at runtime stay out; Via rewrites its config on every load.
|
||||||
|
velocity_fingerprint() {
|
||||||
|
local f
|
||||||
|
for f in "$VELOCITY_SERVICE" "${JRE_DIR}/release" "${VELOCITY_DIR}/velocity.jar" \
|
||||||
|
"${VELOCITY_DIR}/velocity.toml" "${VELOCITY_DIR}/forwarding.secret" \
|
||||||
|
"${VELOCITY_DIR}/plugins/felis-link/felis-link.properties" "${VELOCITY_DIR}"/plugins/*.jar; do
|
||||||
|
[ -f "$f" ] || continue
|
||||||
|
printf '%s %s\n' "$(sha256sum <"$f" | cut -d' ' -f1)" "$f"
|
||||||
|
done | sha256sum | cut -d' ' -f1
|
||||||
|
}
|
||||||
|
|
||||||
configure_velocity_firewall() {
|
configure_velocity_firewall() {
|
||||||
command -v firewall-cmd >/dev/null 2>&1 || return 0
|
command -v firewall-cmd >/dev/null 2>&1 || return 0
|
||||||
systemctl is-active --quiet firewalld || return 0
|
systemctl is-active --quiet firewalld || return 0
|
||||||
@@ -2285,17 +2335,44 @@ install_postgres() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
init_postgres_data_dir
|
init_postgres_data_dir
|
||||||
|
check_postgres_major
|
||||||
systemctl enable --now postgresql
|
systemctl enable --now postgresql
|
||||||
ok "postgresql running"
|
ok "postgresql running"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# check_postgres_major refuses to start a PostgreSQL server whose major version differs from
|
||||||
|
# the one that created the data directory. The server would not start anyway; this says why
|
||||||
|
# and what to do, instead of a failed unit in the middle of the install. Debian and Ubuntu
|
||||||
|
# keep one cluster per version under /var/lib/postgresql/<major> and upgrade with
|
||||||
|
# pg_upgradecluster, so they are left to their own tooling.
|
||||||
|
check_postgres_major() {
|
||||||
|
local data_dir have want
|
||||||
|
[ "$PKG" = "apt" ] && return 0
|
||||||
|
data_dir="$(postgres_data_dir)"
|
||||||
|
[ -f "${data_dir}/PG_VERSION" ] || return 0
|
||||||
|
have="$(tr -d '[:space:]' < "${data_dir}/PG_VERSION")"
|
||||||
|
want="$( (postgres --version 2>/dev/null || psql --version 2>/dev/null) | head -n 1 \
|
||||||
|
| sed -nE 's/^[^0-9]*([0-9]+)\..*/\1/p')"
|
||||||
|
[ -n "$want" ] && [ "$have" != "$want" ] || return 0
|
||||||
|
die "PostgreSQL ${want} is installed, but ${data_dir} holds a PostgreSQL ${have} cluster.
|
||||||
|
The server cannot open it. Upgrade the cluster first (pg_upgrade, with the ${have} binaries
|
||||||
|
still installed), or reinstall PostgreSQL ${have}, then rerun the installer. Take a
|
||||||
|
database bundle before either: sudo felis db backup"
|
||||||
|
}
|
||||||
|
|
||||||
configure_postgres() {
|
configure_postgres() {
|
||||||
local cfg hba
|
local cfg hba listen
|
||||||
cfg="$(as_postgres psql -tAc 'SHOW config_file;' 2>/dev/null || true)"
|
cfg="$(as_postgres psql -tAc 'SHOW config_file;' 2>/dev/null || true)"
|
||||||
hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
|
hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
|
||||||
[ -n "$cfg" ] && [ -n "$hba" ] || die "could not query postgresql config/hba file paths"
|
[ -n "$cfg" ] && [ -n "$hba" ] || die "could not query postgresql config/hba file paths"
|
||||||
|
listen="$(as_postgres psql -tAc 'SHOW listen_addresses;' 2>/dev/null || true)"
|
||||||
|
|
||||||
# Listen on all interfaces (applied on restart). ALTER SYSTEM is idempotent.
|
# Listen on all interfaces (applied on restart). ALTER SYSTEM is idempotent. Pods reach the
|
||||||
|
# database at the node IP, and configure_postgres_firewall keeps the port from everyone else.
|
||||||
|
#
|
||||||
|
# The connection itself is not encrypted (sslmode=disable in felis.toml). On this
|
||||||
|
# single-node shape it never leaves the host: pods reach the node IP over their veth pair
|
||||||
|
# and the host binary uses loopback, so TLS would guard a path no other machine is on.
|
||||||
as_postgres psql -v ON_ERROR_STOP=1 -c "ALTER SYSTEM SET listen_addresses = '*';" >/dev/null
|
as_postgres psql -v ON_ERROR_STOP=1 -c "ALTER SYSTEM SET listen_addresses = '*';" >/dev/null
|
||||||
|
|
||||||
# Allow the host loopback, the pod CIDR, and the node IP before broader distro defaults.
|
# Allow the host loopback, the pod CIDR, and the node IP before broader distro defaults.
|
||||||
@@ -2317,10 +2394,77 @@ SQL
|
|||||||
as_postgres createdb -O "$DB_USER" "$DB_NAME"
|
as_postgres createdb -O "$DB_USER" "$DB_NAME"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# listen_addresses is the only setting here that needs a restart, and a restart cuts
|
||||||
|
# every connection felis-api holds mid-transaction. pg_hba.conf and the role are live
|
||||||
|
# after a reload.
|
||||||
|
if [ "$listen" = "*" ]; then
|
||||||
|
as_postgres psql -v ON_ERROR_STOP=1 -tAc 'SELECT pg_reload_conf();' >/dev/null
|
||||||
|
else
|
||||||
systemctl restart postgresql
|
systemctl restart postgresql
|
||||||
|
fi
|
||||||
|
configure_postgres_firewall
|
||||||
ok "postgresql configured (listen=*, role/db '${DB_NAME}', pg_hba opened to pods)"
|
ok "postgresql configured (listen=*, role/db '${DB_NAME}', pg_hba opened to pods)"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# configure_postgres_firewall keeps 5432 to this host and its pods. PostgreSQL listens on
|
||||||
|
# every address (pods dial the node IP), and pg_hba.conf only refuses a connection after the
|
||||||
|
# server has spoken to it. firewalld's default zone does not open 5432 (configure_k3s_firewall
|
||||||
|
# trusts only the pod and service CIDRs), so that host needs nothing more; on any other host a
|
||||||
|
# small nftables table of our own drops 5432 from everywhere but loopback, the pod CIDR and the
|
||||||
|
# node's own address, loaded at boot by a oneshot unit ordered before PostgreSQL.
|
||||||
|
configure_postgres_firewall() {
|
||||||
|
if command -v firewall-cmd >/dev/null 2>&1 && systemctl is-active --quiet firewalld; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
command -v nft >/dev/null 2>&1 || pkg_install nftables
|
||||||
|
command -v nft >/dev/null 2>&1 || { warn "nft is not available; PostgreSQL's 5432 is reachable from the network (pg_hba still refuses other hosts)"; return 0; }
|
||||||
|
local node_rule
|
||||||
|
case "$NODE_IP" in
|
||||||
|
*:*) node_rule="ip6 saddr ${NODE_IP}" ;;
|
||||||
|
*) node_rule="ip saddr ${NODE_IP}" ;;
|
||||||
|
esac
|
||||||
|
install -d -m 0700 "$STATE_DIR"
|
||||||
|
cat > "$PG_FIREWALL_RULES" <<EOF
|
||||||
|
# Generated by deploy/bootstrap.sh — do not edit by hand; rerun the installer.
|
||||||
|
# The first two lines make a reload replace the table instead of failing on it.
|
||||||
|
table inet felis_postgres
|
||||||
|
delete table inet felis_postgres
|
||||||
|
table inet felis_postgres {
|
||||||
|
chain input {
|
||||||
|
type filter hook input priority filter; policy accept;
|
||||||
|
tcp dport 5432 iif "lo" accept
|
||||||
|
tcp dport 5432 ip saddr ${POD_CIDR} accept
|
||||||
|
tcp dport 5432 ${node_rule} accept
|
||||||
|
tcp dport 5432 drop
|
||||||
|
}
|
||||||
|
}
|
||||||
|
EOF
|
||||||
|
chmod 0600 "$PG_FIREWALL_RULES"
|
||||||
|
cat > "$PG_FIREWALL_SERVICE" <<EOF
|
||||||
|
[Unit]
|
||||||
|
Description=Felis: keep PostgreSQL's port to this host and its pods
|
||||||
|
DefaultDependencies=no
|
||||||
|
After=nftables.service
|
||||||
|
# nftables.service starts from a flushed ruleset; a restart of it reloads this table too.
|
||||||
|
PartOf=nftables.service
|
||||||
|
Before=network-pre.target postgresql.service
|
||||||
|
Wants=network-pre.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
RemainAfterExit=yes
|
||||||
|
ExecStart=$(command -v nft) -f ${PG_FIREWALL_RULES}
|
||||||
|
ExecStop=$(command -v nft) delete table inet felis_postgres
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
|
EOF
|
||||||
|
systemctl daemon-reload
|
||||||
|
systemctl enable felis-postgres-firewall.service >/dev/null 2>&1
|
||||||
|
systemctl restart felis-postgres-firewall.service
|
||||||
|
ok "5432 accepts loopback, ${POD_CIDR} and ${NODE_IP} only (nftables table felis_postgres)"
|
||||||
|
}
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# 7. Secrets + felis.toml (pod variant reaches Postgres at the node IP; host
|
# 7. Secrets + felis.toml (pod variant reaches Postgres at the node IP; host
|
||||||
# variant at 127.0.0.1 for migrations)
|
# variant at 127.0.0.1 for migrations)
|
||||||
@@ -3171,18 +3315,32 @@ push_version_tag() {
|
|||||||
|
|
||||||
# The login/lobby images use mutable :demo tags. Importing/pushing a replacement
|
# The login/lobby images use mutable :demo tags. Importing/pushing a replacement
|
||||||
# updates containerd, but an existing StatefulSet template is byte-for-byte
|
# updates containerd, but an existing StatefulSet template is byte-for-byte
|
||||||
# unchanged and Kubernetes will not roll it. Recreate only the two always-on
|
# unchanged and Kubernetes will not roll it. Recreate the two always-on system
|
||||||
# system pods so a convergent bootstrap actually starts the images it just built.
|
# pods so a convergent bootstrap actually starts the images it just built — but
|
||||||
|
# only when the build changed: every player online is on one of these two, and a
|
||||||
|
# rerun that rebuilt nothing has nothing to start. SYSTEM_SERVER_IMAGES records the
|
||||||
|
# build each was last started on; it is written after the restarts, so a run that
|
||||||
|
# died in between restarts them next time.
|
||||||
restart_existing_system_servers() {
|
restart_existing_system_servers() {
|
||||||
local name pods
|
local name id pods next=""
|
||||||
for name in "$LOGIN_SERVER" "$LOBBY_SERVER"; do
|
while read -r name id; do
|
||||||
|
[ -n "$name" ] || continue
|
||||||
|
next="${next}${name} ${id}"$'\n'
|
||||||
pods="$(kube -n "$MINECRAFT_NS" get pod \
|
pods="$(kube -n "$MINECRAFT_NS" get pod \
|
||||||
-l "felis.lolicon.best/server=${name}" -o name 2>/dev/null || true)"
|
-l "felis.lolicon.best/server=${name}" -o name 2>/dev/null || true)"
|
||||||
[ -n "$pods" ] || continue
|
[ -n "$pods" ] || continue
|
||||||
|
if [ -n "$id" ] && grep -qxF "${name} ${id}" "$SYSTEM_SERVER_IMAGES" 2>/dev/null; then
|
||||||
|
ok "${name} system server already runs this build; left running"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
log "restarting existing ${name} system server to pick up its imported image"
|
log "restarting existing ${name} system server to pick up its imported image"
|
||||||
kube -n "$MINECRAFT_NS" delete pod \
|
kube -n "$MINECRAFT_NS" delete pod \
|
||||||
-l "felis.lolicon.best/server=${name}" --wait=false
|
-l "felis.lolicon.best/server=${name}" --wait=false
|
||||||
done
|
done <<EOF
|
||||||
|
${LOGIN_SERVER} ${LIMBO_IMAGE_ID:-}
|
||||||
|
${LOBBY_SERVER} ${LOBBY_IMAGE_ID:-}
|
||||||
|
EOF
|
||||||
|
printf '%s' "$next" > "$SYSTEM_SERVER_IMAGES"
|
||||||
}
|
}
|
||||||
|
|
||||||
diagnose_rollout() {
|
diagnose_rollout() {
|
||||||
|
|||||||
@@ -1583,6 +1583,204 @@ case "$mainblock" in
|
|||||||
*) echo "FAIL main must call resolve_felis_image right before build_image"; fails=$((fails + 1)) ;;
|
*) echo "FAIL main must call resolve_felis_image right before build_image"; fails=$((fails + 1)) ;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
|
# --- a rerun restarts only what changed -------------------------------------------------
|
||||||
|
# The proxy, the login and lobby pods and PostgreSQL each disconnect every player (or cut
|
||||||
|
# felis-api's transactions) when restarted, so a rerun that changed none of them must leave
|
||||||
|
# them running, and one that changed a thing must still restart it.
|
||||||
|
|
||||||
|
vsblock="$(awk '/^install_velocity_service\(\) \{/,/^}/' "$BS"; awk '/^velocity_fingerprint\(\) \{/,/^}/' "$BS")"
|
||||||
|
[ -n "$vsblock" ] || { echo "FAIL: no install_velocity_service found in $BS"; exit 1; }
|
||||||
|
vdir="$(mktemp -d)"
|
||||||
|
mkdir -p "$vdir/v/plugins/felis-link" "$vdir/jre"
|
||||||
|
printf 'jar\n' > "$vdir/v/velocity.jar"
|
||||||
|
printf 'plugin\n' > "$vdir/v/plugins/felis-velocity.jar"
|
||||||
|
printf 'JAVA_VERSION="25"\n' > "$vdir/jre/release"
|
||||||
|
run_velocity_service() { # is-active(0|1)
|
||||||
|
ACTIVE="$1" VELOCITY_SERVICE="$vdir/unit" VELOCITY_DIR="$vdir/v" JRE_DIR="$vdir/jre" \
|
||||||
|
VELOCITY_FINGERPRINT="$vdir/fp" VELOCITY_USER=felis-velocity FELIS_GAME_PORT=25565 \
|
||||||
|
FELIS_LEGACY_FORWARDING_SERVERS= bash -c '
|
||||||
|
set -Eeuo pipefail
|
||||||
|
ok() { printf "OK: %s\n" "$*"; }
|
||||||
|
felis_internal_ip() { printf "10.43.0.9"; }
|
||||||
|
systemctl() {
|
||||||
|
case "$1" in
|
||||||
|
is-active) [ "$ACTIVE" = 1 ] ;;
|
||||||
|
*) printf "SYSTEMCTL %s\n" "$*" ;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
'"$vsblock"'
|
||||||
|
install_velocity_service'
|
||||||
|
}
|
||||||
|
out="$(run_velocity_service 1)"
|
||||||
|
expect "a proxy with no recorded start is restarted" "SYSTEMCTL restart felis-velocity" "$out"
|
||||||
|
[ -s "$vdir/fp" ] && echo "PASS the restart records what the proxy runs" \
|
||||||
|
|| { echo "FAIL no fingerprint was recorded after the restart"; fails=$((fails + 1)); }
|
||||||
|
out="$(run_velocity_service 1)"
|
||||||
|
case "$out" in
|
||||||
|
*"SYSTEMCTL restart"*) echo "FAIL an unchanged rerun restarted the proxy"; fails=$((fails + 1)) ;;
|
||||||
|
*"felis-velocity unchanged; left running"*) echo "PASS an unchanged rerun leaves the proxy running" ;;
|
||||||
|
*) echo "FAIL install_velocity_service died on an unchanged rerun: $out"; fails=$((fails + 1)) ;;
|
||||||
|
esac
|
||||||
|
printf 'plugin v2\n' > "$vdir/v/plugins/felis-velocity.jar"
|
||||||
|
expect "a changed plugin jar restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1)"
|
||||||
|
expect "a stopped proxy is started whatever the fingerprint" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 0)"
|
||||||
|
printf 'JAVA_VERSION="25.0.1"\n' > "$vdir/jre/release"
|
||||||
|
expect "a patched JRE restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1)"
|
||||||
|
rm -rf "$vdir"
|
||||||
|
|
||||||
|
ssblock="$(awk '/^restart_existing_system_servers\(\) \{/,/^}/' "$BS")"
|
||||||
|
[ -n "$ssblock" ] || { echo "FAIL: no restart_existing_system_servers found in $BS"; exit 1; }
|
||||||
|
sdir2="$(mktemp -d)"
|
||||||
|
run_system_restart() { # limbo-id lobby-id pods(0|1)
|
||||||
|
LIMBO_IMAGE_ID="$1" LOBBY_IMAGE_ID="$2" PODS="$3" SYSTEM_SERVER_IMAGES="$sdir2/state" \
|
||||||
|
LOGIN_SERVER=login LOBBY_SERVER=lobby MINECRAFT_NS=minecraft bash -c '
|
||||||
|
set -Eeuo pipefail
|
||||||
|
ok() { printf "OK: %s\n" "$*"; }
|
||||||
|
log() { :; }
|
||||||
|
kube() {
|
||||||
|
case "$*" in
|
||||||
|
*"get pod"*) [ "$PODS" = 1 ] && printf "pod/x-0\n" || true ;;
|
||||||
|
*delete*) printf "KUBE %s\n" "$*" ;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
'"$ssblock"'
|
||||||
|
restart_existing_system_servers'
|
||||||
|
}
|
||||||
|
out="$(run_system_restart sha256:aaa sha256:bbb 1)"
|
||||||
|
expect "an unrecorded login pod is restarted" "delete pod -l felis.lolicon.best/server=login" "$out"
|
||||||
|
expect "an unrecorded lobby pod is restarted" "delete pod -l felis.lolicon.best/server=lobby" "$out"
|
||||||
|
out="$(run_system_restart sha256:aaa sha256:bbb 1)"
|
||||||
|
case "$out" in
|
||||||
|
*delete*) echo "FAIL an unchanged rebuild restarted a system server"; fails=$((fails + 1)) ;;
|
||||||
|
*"already runs this build"*) echo "PASS an unchanged rebuild leaves the system servers running" ;;
|
||||||
|
*) echo "FAIL restart_existing_system_servers died on an unchanged rebuild: $out"; fails=$((fails + 1)) ;;
|
||||||
|
esac
|
||||||
|
out="$(run_system_restart sha256:aaa sha256:ccc 1)"
|
||||||
|
expect "a new lobby build restarts the lobby" "delete pod -l felis.lolicon.best/server=lobby" "$out"
|
||||||
|
case "$out" in
|
||||||
|
*"server=login"*) echo "FAIL a new lobby build restarted the login gate too"; fails=$((fails + 1)) ;;
|
||||||
|
*) echo "PASS a new lobby build leaves the login gate running" ;;
|
||||||
|
esac
|
||||||
|
rm -f "$sdir2/state"
|
||||||
|
out="$(run_system_restart sha256:aaa sha256:ccc 0)"
|
||||||
|
case "$out" in
|
||||||
|
*delete*) echo "FAIL a missing pod was deleted"; fails=$((fails + 1)) ;;
|
||||||
|
*) echo "PASS no pod, nothing to restart" ;;
|
||||||
|
esac
|
||||||
|
expect "the builds are recorded even before the pods exist" "lobby sha256:ccc" "$(cat "$sdir2/state")"
|
||||||
|
rm -rf "$sdir2"
|
||||||
|
|
||||||
|
pgblock="$(awk '/^configure_postgres\(\) \{/,/^}/' "$BS")"
|
||||||
|
[ -n "$pgblock" ] || { echo "FAIL: no configure_postgres found in $BS"; exit 1; }
|
||||||
|
pgcalls="$(mktemp)"
|
||||||
|
run_configure_pg() { # current listen_addresses
|
||||||
|
: > "$pgcalls"
|
||||||
|
LISTEN="$1" CALLS="$pgcalls" DB_NAME=felis DB_USER=felis DB_PASSWORD=pw bash -c '
|
||||||
|
set -Eeuo pipefail
|
||||||
|
ok() { :; }
|
||||||
|
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||||
|
as_postgres() {
|
||||||
|
case "$*" in
|
||||||
|
*"SHOW config_file"*) echo /c ;;
|
||||||
|
*"SHOW hba_file"*) echo /h ;;
|
||||||
|
*"SHOW listen_addresses"*) echo "$LISTEN" ;;
|
||||||
|
*pg_reload_conf*) echo RELOAD >> "$CALLS" ;;
|
||||||
|
*"SELECT 1 FROM pg_database"*) echo 1 ;;
|
||||||
|
*) cat >/dev/null ;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
write_pg_hba_block() { :; }
|
||||||
|
configure_postgres_firewall() { echo FIREWALL; }
|
||||||
|
systemctl() { printf "SYSTEMCTL %s\n" "$*"; }
|
||||||
|
'"$pgblock"'
|
||||||
|
configure_postgres' < /dev/null
|
||||||
|
cat "$pgcalls"
|
||||||
|
}
|
||||||
|
out="$(run_configure_pg '*')"
|
||||||
|
case "$out" in
|
||||||
|
*"SYSTEMCTL restart postgresql"*) echo "FAIL a rerun restarted PostgreSQL under felis-api"; fails=$((fails + 1)) ;;
|
||||||
|
*RELOAD*) echo "PASS a rerun reloads PostgreSQL instead of restarting it" ;;
|
||||||
|
*) echo "FAIL configure_postgres neither reloaded nor restarted: $out"; fails=$((fails + 1)) ;;
|
||||||
|
esac
|
||||||
|
expect "the firewall step runs" "FIREWALL" "$out"
|
||||||
|
expect "a first install restarts PostgreSQL to listen on the node" "SYSTEMCTL restart postgresql" "$(run_configure_pg localhost)"
|
||||||
|
rm -f "$pgcalls"
|
||||||
|
|
||||||
|
pmblock="$(awk '/^check_postgres_major\(\) \{/,/^}/' "$BS")"
|
||||||
|
[ -n "$pmblock" ] || { echo "FAIL: no check_postgres_major found in $BS"; exit 1; }
|
||||||
|
pmdir="$(mktemp -d)"
|
||||||
|
run_pg_major() { # PKG cluster-version server-version
|
||||||
|
printf '%s\n' "$2" > "$pmdir/PG_VERSION"
|
||||||
|
PKG="$1" SERVER="$3" DATA="$pmdir" bash -c '
|
||||||
|
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||||
|
postgres_data_dir() { printf "%s\n" "$DATA"; }
|
||||||
|
postgres() { printf "postgres (PostgreSQL) %s\n" "$SERVER"; }
|
||||||
|
'"$pmblock"'
|
||||||
|
check_postgres_major && echo STARTS'
|
||||||
|
}
|
||||||
|
expect "a new major version over an old cluster is refused" "DIE: PostgreSQL 17 is installed, but $pmdir holds a PostgreSQL 16 cluster" \
|
||||||
|
"$(run_pg_major dnf 16 17.2)"
|
||||||
|
expect "the refusal names the way forward" "pg_upgrade" "$(run_pg_major pacman 16 17.2)"
|
||||||
|
expect "the same major version starts" "STARTS" "$(run_pg_major dnf 16 16.4)"
|
||||||
|
expect "Debian-family clusters are left to pg_upgradecluster" "STARTS" "$(run_pg_major apt 15 17.2)"
|
||||||
|
rm -rf "$pmdir"
|
||||||
|
|
||||||
|
rfblock="$(awk '/^pkg_refresh_once\(\) \{/,/^}/' "$BS")"
|
||||||
|
pmdir="$(mktemp -d)"
|
||||||
|
run_refresh() {
|
||||||
|
PKG=pacman DATA="$pmdir" bash -c '
|
||||||
|
wait_for_pkg_locks() { :; }
|
||||||
|
postgres_data_dir() { printf "%s\n" "$DATA"; }
|
||||||
|
pacman() { printf "PACMAN %s\n" "$*"; }
|
||||||
|
'"$rfblock"'
|
||||||
|
pkg_refresh_once'
|
||||||
|
}
|
||||||
|
case "$(run_refresh)" in
|
||||||
|
*--ignore*) echo "FAIL a host with no cluster yet held PostgreSQL back"; fails=$((fails + 1)) ;;
|
||||||
|
*"PACMAN -Syu"*) echo "PASS a fresh Arch host upgrades normally" ;;
|
||||||
|
*) echo "FAIL pkg_refresh_once did not refresh pacman"; fails=$((fails + 1)) ;;
|
||||||
|
esac
|
||||||
|
printf '16\n' > "$pmdir/PG_VERSION"
|
||||||
|
expect "an Arch rerun holds PostgreSQL at the cluster's version" "PACMAN -Syu --noconfirm --ignore postgresql" "$(run_refresh)"
|
||||||
|
rm -rf "$pmdir"
|
||||||
|
|
||||||
|
# The rules heredoc closes nft blocks with a bare "}", so stop at the function's own brace.
|
||||||
|
fwblock="$(awk '/^configure_postgres_firewall\(\) \{/ { f = 1 }
|
||||||
|
f { print; if ($0 ~ /<<EOF$/) h = 1; else if ($0 == "EOF") h = 0; else if (!h && $0 == "}") exit }' "$BS")"
|
||||||
|
[ -n "$fwblock" ] || { echo "FAIL: no configure_postgres_firewall found in $BS"; exit 1; }
|
||||||
|
fwdir="$(mktemp -d)"
|
||||||
|
run_pg_firewall() { # firewalld-active(0|1) node-ip
|
||||||
|
FWD="$1" NODE_IP="$2" POD_CIDR=10.42.0.0/16 STATE_DIR="$fwdir" \
|
||||||
|
PG_FIREWALL_RULES="$fwdir/pg.nft" PG_FIREWALL_SERVICE="$fwdir/pg.service" bash -c '
|
||||||
|
set -Eeuo pipefail
|
||||||
|
ok() { :; }
|
||||||
|
warn() { printf "WARN: %s\n" "$*"; }
|
||||||
|
pkg_install() { :; }
|
||||||
|
firewall-cmd() { :; }
|
||||||
|
nft() { :; }
|
||||||
|
systemctl() {
|
||||||
|
case "$1" in
|
||||||
|
is-active) [ "$FWD" = 1 ] ;;
|
||||||
|
*) printf "SYSTEMCTL %s\n" "$*" ;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
'"$fwblock"'
|
||||||
|
configure_postgres_firewall'
|
||||||
|
}
|
||||||
|
out="$(run_pg_firewall 1 203.0.113.7)"
|
||||||
|
[ -e "$fwdir/pg.nft" ] && { echo "FAIL a firewalld host got a second firewall"; fails=$((fails + 1)); } \
|
||||||
|
|| echo "PASS firewalld already closes 5432"
|
||||||
|
out="$(run_pg_firewall 0 203.0.113.7)"
|
||||||
|
rules="$(cat "$fwdir/pg.nft")"
|
||||||
|
expect "the pods may reach the database" "tcp dport 5432 ip saddr 10.42.0.0/16 accept" "$rules"
|
||||||
|
expect "the node itself may reach the database" "tcp dport 5432 ip saddr 203.0.113.7 accept" "$rules"
|
||||||
|
expect "everyone else is dropped" "tcp dport 5432 drop" "$rules"
|
||||||
|
expect "the rules load at boot" "SYSTEMCTL restart felis-postgres-firewall.service" "$out"
|
||||||
|
expect "the unit orders itself before PostgreSQL" "Before=network-pre.target postgresql.service" "$(cat "$fwdir/pg.service")"
|
||||||
|
run_pg_firewall 0 2001:db8::7 >/dev/null
|
||||||
|
expect "a v6 node address gets a v6 rule" "tcp dport 5432 ip6 saddr 2001:db8::7 accept" "$(cat "$fwdir/pg.nft")"
|
||||||
|
rm -rf "$fwdir"
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------------------
|
||||||
if [ "$fails" -eq 0 ]; then
|
if [ "$fails" -eq 0 ]; then
|
||||||
|
|||||||
@@ -1356,6 +1356,28 @@ picks up the new release on its next start. A release that changes the game
|
|||||||
pod in any other way still restarts running servers once, as a server edit
|
pod in any other way still restarts running servers once, as a server edit
|
||||||
does.
|
does.
|
||||||
|
|
||||||
|
**What a rerun restarts.** Each of these drops every connected player (or cuts
|
||||||
|
felis-api's open transactions), so the installer restarts one only when what it
|
||||||
|
runs changed:
|
||||||
|
|
||||||
|
| Component | Restarted when |
|
||||||
|
|---|---|
|
||||||
|
| `felis-velocity` (the proxy) | its unit, the JRE, `velocity.jar`, `velocity.toml`, the forwarding secret, the felis-link settings or a plugin jar changed, or it was not running. The fingerprint lives in `/etc/felis/velocity.fingerprint`; delete it to force a restart. |
|
||||||
|
| login and lobby pods | the rebuilt limbo or lobby image has a new image ID (`/etc/felis/system-server-images`). Each restarts on its own. |
|
||||||
|
| PostgreSQL | first install only (`listen_addresses` needs a restart). A rerun reloads the configuration, which keeps connections open. |
|
||||||
|
| felis-api, felis-operator | the image tag changed (an upgrade), or a same-version rerun rebuilt it. |
|
||||||
|
|
||||||
|
**PostgreSQL across reruns.** On hosts without firewalld the installer loads an
|
||||||
|
nftables table, `inet felis_postgres`, from `felis-postgres-firewall.service`:
|
||||||
|
port 5432 accepts loopback, the pod network and the node's own address and drops
|
||||||
|
everything else (`nft list table inet felis_postgres`). firewalld hosts already
|
||||||
|
keep 5432 closed to the network. The installer also refuses to start a
|
||||||
|
PostgreSQL whose major version differs from the cluster in the data directory,
|
||||||
|
and prints the `pg_upgrade` steps; distributions that move the server package to
|
||||||
|
a new major (Arch, Fedora) would otherwise leave the database unable to start.
|
||||||
|
On Arch the installer's `pacman -Syu` holds `postgresql` back once a cluster
|
||||||
|
exists, so the database is upgraded only when you run `pg_upgrade` yourself.
|
||||||
|
|
||||||
`rollout undo` reverts the image only. The upgrade's database migrations stay
|
`rollout undo` reverts the image only. The upgrade's database migrations stay
|
||||||
applied; when they are the problem, restore the `pre-migrate` bundle the upgrade
|
applied; when they are the problem, restore the `pre-migrate` bundle the upgrade
|
||||||
took (§16, "Roll back an upgrade that broke the database").
|
took (§16, "Roll back an upgrade that broke the database").
|
||||||
|
|||||||
@@ -64,3 +64,11 @@ sourceSets {
|
|||||||
tasks.withType(JavaCompile).configureEach {
|
tasks.withType(JavaCompile).configureEach {
|
||||||
options.encoding = 'UTF-8'
|
options.encoding = 'UTF-8'
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Byte-identical jars from identical sources. deploy/bootstrap.sh restarts the proxy only when
|
||||||
|
// a file it runs changed (velocity_fingerprint), and build-time timestamps in the jar would
|
||||||
|
// make every rebuild look like a change and disconnect every player on each rerun.
|
||||||
|
tasks.withType(AbstractArchiveTask).configureEach {
|
||||||
|
preserveFileTimestamps = false
|
||||||
|
reproducibleFileOrder = true
|
||||||
|
}
|
||||||
Reference in new issue
Block a user