diff --git a/deploy/uninstall.sh b/deploy/uninstall.sh index d54b30f..20bcfb5 100644 --- a/deploy/uninstall.sh +++ b/deploy/uninstall.sh @@ -6,17 +6,20 @@ # curl -fsSL /deploy/uninstall.sh | sudo bash -s -- --yes # # Options: -# --purge also drop the felis database and role, and delete /etc/felis and -# /var/lib/felis (the database bundles, and anything an earlier keep-data -# run set aside). Asks for the word "purge" unless --yes is given. +# --purge also delete /etc/felis and /var/lib/felis (the database's cluster, the +# database bundles, and anything an earlier keep-data run set aside), and +# drop the felis database and role from a host PostgreSQL an earlier +# release installed. Asks for the word "purge" unless --yes is given. # --keep-k3s leave k3s installed and remove only Felis's namespaces and CRD. # --remove-k3s run k3s's own uninstaller even when other workloads live in the cluster. # --no-backup skip the final database bundle keep-data mode takes first. # --yes do not ask. # # Keep-data mode (the default) first takes a database bundle (`felis db backup -label -# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the -# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive, +# manual`) and stops if that fails. It leaves /etc/felis (the secrets, felis.toml, +# offsite.env) and /var/lib/felis in place, and with it the felis database: felis-postgres +# keeps its cluster in /var/lib/felis/postgres, stopped cleanly before k3s goes, and a +# host PostgreSQL an earlier release installed keeps its copy. The world, archive, # registry and upload volumes live under k3s's storage directory, which k3s's uninstaller # deletes, so they are moved to /var/lib/felis/retained/k3s-storage- first; with # --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go. @@ -41,6 +44,12 @@ CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}" VELOCITY_USER="felis-velocity" DB_NAME="felis" DB_USER="felis" +CONTROL_NS="felis" +# The database bootstrap runs in k3s, its cluster on a hostPath under DATA_DIR, and the +# marker bootstrap leaves once it moved a host PostgreSQL's felis database into it. +PG_DEPLOYMENT="felis-postgres" +PG_DATA_DIR="${DATA_DIR}/postgres" +PG_MOVED_MARKER="${DATA_DIR}/postgres-moved" POD_CIDR="10.42.0.0/16" SERVICE_CIDR="10.43.0.0/16" FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}" @@ -131,11 +140,12 @@ print_plan() { absent) ;; esac if [ "$PURGE" = 1 ]; then - log " PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles" - log " and anything set aside before), every world and archive, the Felis images and Docker's build cache" + log " PURGE: ${STATE_DIR} (secrets), ${DATA_DIR} (the ${DB_NAME} database's cluster, the database" + log " bundles and anything set aside before), the ${DB_NAME} database and role in a host PostgreSQL," + log " every world and archive, the Felis images and Docker's build cache" else [ "$BACKUP" = 1 ] && log " after a final database bundle into ${DATA_DIR}/db-backups" - log " kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/" + log " kept: ${STATE_DIR}, ${DATA_DIR} (the ${DB_NAME} database in ${PG_DATA_DIR}); the volumes move to ${RETAIN_DIR}/" fi } @@ -273,8 +283,18 @@ remove_from_cluster() { ok "Felis removed from the cluster; k3s stays" } +# stop_database_pod shuts felis-postgres down cleanly: k3s-killall.sh SIGKILLs every +# container, and the cluster it leaves in PG_DATA_DIR is what a reinstall starts from. +stop_database_pod() { + kube -n "$CONTROL_NS" scale deployment "$PG_DEPLOYMENT" --replicas=0 >/dev/null 2>&1 || return 0 + kube -n "$CONTROL_NS" wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres \ + --timeout=120s >/dev/null 2>&1 \ + || warn "${PG_DEPLOYMENT} did not stop within 2 minutes; its cluster recovers from its WAL on the next start" +} + remove_k3s() { local stamp + [ "$PURGE" = 1 ] || stop_database_pod if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then # k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a # world while it moves. @@ -325,7 +345,8 @@ remove_host_files() { as_postgres() { (cd / && runuser -u postgres -- "$@"); } -# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else. +# remove_hba_block drops the block bootstrap heads pg_hba.conf with (the rules of an +# install on the host server, or the lockout the move into k3s left), and nothing else. remove_hba_block() { # file local tmp tmp="$(mktemp)" @@ -374,23 +395,46 @@ SQL die "the ${DB_USER} role still holds $(printf '%s' "$held" | paste -sd ';' - | sed 's/;/; /g'), so DROP ROLE would fail halfway through the purge; nothing was removed. Hand them to postgres first (sudo -u postgres psql -c 'ALTER DATABASE OWNER TO postgres', or REASSIGN OWNED BY ${DB_USER} TO postgres; DROP OWNED BY ${DB_USER}; inside that database), or rerun without --purge" } +# purge_database drops the felis database and role from a host PostgreSQL an earlier +# release installed; the cluster felis-postgres runs goes with DATA_DIR (purge_state). That +# host server either still serves the platform, or the move into felis-postgres stopped it +# with the pre-move copy left in it for a rollback: a purge takes that copy too and leaves +# the server stopped. The units and k3s are gone by now, so a failure here is a warning +# with the commands to finish by hand, and the purge goes on. purge_database() { [ "$PURGE" = 1 ] || return 0 + local hba started=0 if ! systemctl is-active --quiet postgresql 2>/dev/null; then - warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it" - return 0 + [ -e "$PG_MOVED_MARKER" ] && systemctl cat postgresql >/dev/null 2>&1 || return 0 + if ! systemctl start postgresql >/dev/null 2>&1; then + warn "could not start the host PostgreSQL, so its copy of the ${DB_NAME} database from before the move into k3s stays in it; drop it once it runs: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}" + return 0 + fi + started=1 fi - local hba hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)" - as_postgres psql -v ON_ERROR_STOP=1 -q < pg_backend_pid(); DROP DATABASE IF EXISTS ${DB_NAME}; DROP ROLE IF EXISTS ${DB_USER}; ALTER SYSTEM RESET listen_addresses; SQL - [ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba" - systemctl restart postgresql - ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again" + then + [ "$started" = 0 ] || systemctl stop postgresql >/dev/null 2>&1 || true + warn "could not drop the ${DB_NAME} database and role from the host PostgreSQL; drop them by hand: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}" + return 0 + fi + if [ -n "$hba" ] && [ -f "$hba" ]; then + remove_hba_block "$hba" + rm -f "${hba}.pre-pg-move" + fi + if [ "$started" = 1 ]; then + systemctl stop postgresql + ok "the host PostgreSQL's copy of '${DB_NAME}' from before the move into k3s dropped; the server stays stopped" + else + systemctl restart postgresql + ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again" + fi } purge_images() { @@ -419,6 +463,10 @@ purge_state() { && cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)" if [ -n "$cred" ]; then rm -f "$cred"; fi rm -f "$TUNNEL_CONFIG" + # The file-context rule bootstrap gave the database's cluster directory. + if command -v semanage >/dev/null 2>&1; then + semanage fcontext -d "${PG_DATA_DIR}(/.*)?" >/dev/null 2>&1 || true + fi rm -rf "$STATE_DIR" "$DATA_DIR" ok "${STATE_DIR} and ${DATA_DIR} removed" } @@ -453,7 +501,7 @@ main() { if [ "$PURGE" = 1 ]; then ok "Felis is gone from this host" else - ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}" + ok "Felis is removed; the data stays in ${STATE_DIR} and ${DATA_DIR}, the ${DB_NAME} database in ${PG_DATA_DIR}" log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\"" fi } diff --git a/deploy/uninstall_test.sh b/deploy/uninstall_test.sh index 5747034..a202236 100644 --- a/deploy/uninstall_test.sh +++ b/deploy/uninstall_test.sh @@ -44,6 +44,8 @@ fresh_host() { printf 'x\n' > "$root/h/etc/$f" done printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat" + mkdir -p "$root/h/data/postgres/18/docker" + printf '18\n' > "$root/h/data/postgres/18/docker/PG_VERSION" for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b" chmod +x "$root/h/bin/$b" @@ -72,11 +74,19 @@ run_uninstall() { . "$0" calls="$ROOT/calls" id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; } + # PG_HOST: the host PostgreSQL is active (the default), stopped, or not installed (none); + # PG_START=fail for one that will not start. systemctl() { echo "SYSTEMCTL $*" >> "$calls" - case "$*" in "is-active --quiet firewalld") return 1 ;; esac - return 0 + case "$*" in + "is-active --quiet firewalld") return 1 ;; + "is-active --quiet postgresql") [ "${PG_HOST:-active}" = active ] ;; + "cat postgresql") [ "${PG_HOST:-active}" != none ] ;; + "start postgresql") [ "${PG_START:-}" != fail ] ;; + *) return 0 ;; + esac } + semanage() { echo "SEMANAGE $*" >> "$calls"; } kube() { echo "KUBE $*" >> "$calls" case "$*" in @@ -98,7 +108,7 @@ run_uninstall() { echo "PSQL-CHECK" >> "$calls" [ "${PG_CHECK:-}" = fail ] && { echo "psql: error: connection refused" >&2; return 2; } [ -z "${PG_HELD:-}" ] || printf "%s\n" "$PG_HELD" ;; - *) echo "PSQL $* $sql" >> "$calls" ;; + *) echo "PSQL $* $sql" >> "$calls"; [ "${PG_DROP:-}" != fail ] ;; esac ;; esac } @@ -135,6 +145,15 @@ refute "keep-data leaves the database alone" "DROP DATABASE" "$calls" expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls" expect "the velocity user is removed" "USERDEL felis-velocity" "$calls" expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out" +[ -f "$root/h/data/postgres/18/docker/PG_VERSION" ] && echo "PASS the database's cluster stays for the reinstall" \ + || { echo "FAIL keep-data removed the database's cluster"; fails=$((fails + 1)); } +stop_at="$(grep -n 'KUBE -n felis scale deployment felis-postgres --replicas=0' "$root/calls" | head -n 1 | cut -d: -f1)" +kill_at="$(grep -n 'RUN k3s-killall.sh' "$root/calls" | head -n 1 | cut -d: -f1)" +[ -n "$stop_at" ] && [ -n "$kill_at" ] && [ "$stop_at" -lt "$kill_at" ] \ + && echo "PASS the database shuts down cleanly before k3s-killall.sh kills what is left" \ + || { echo "FAIL felis-postgres was not stopped before k3s-killall.sh: $calls"; fails=$((fails + 1)); } +expect "and the uninstall waits for it to stop" "KUBE -n felis wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres" "$calls" +refute "keep-data keeps the cluster directory's SELinux rule" "SEMANAGE" "$calls" # --- a failed final bundle stops everything ------------------------------------------------ fresh_host @@ -189,6 +208,42 @@ case "$hba" in *) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;; esac expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls" +expect "purge drops the cluster directory's SELinux rule" "SEMANAGE fcontext -d $root/h/data/postgres(/.*)?" "$calls" +refute "purge leaves k3s's pods to k3s-uninstall.sh" "scale deployment felis-postgres" "$calls" + +# --- purge after the move into felis-postgres --------------------------------------------- +# The move stopped the host server with the felis database left in it for a rollback. +moved_host() { fresh_host; printf 'moved\n' > "$root/h/data/postgres-moved"; printf 'saved\n' > "$root/h/hba.conf.pre-pg-move"; } +moved_host +out="$(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes)" +calls="$(cat "$root/calls")" +expect "purge starts the stopped host server to drop the pre-move copy" "SYSTEMCTL start postgresql" "$calls" +expect "and drops it" "DROP DATABASE IF EXISTS felis;" "$calls" +expect "the server stays stopped afterwards" "SYSTEMCTL stop postgresql" "$calls" +refute "and is not restarted" "SYSTEMCTL restart postgresql" "$calls" +refute "the lockout leaves pg_hba.conf" "FELIS MANAGED" "$(cat "$root/h/hba.conf")" +[ ! -e "$root/h/hba.conf.pre-pg-move" ] && echo "PASS the pg_hba.conf the move saved goes with the copy" \ + || { echo "FAIL purge left pg_hba.conf.pre-pg-move"; fails=$((fails + 1)); } +moved_host +out="$(PG_HOST=stopped PG_START=fail run_uninstall "default felis minecraft" --purge --yes)" +calls="$(cat "$root/calls")" +expect "a host server that will not start is named" "could not start the host PostgreSQL" "$out" +refute "so nothing is dropped" "DROP DATABASE" "$calls" +[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS and the purge goes on" \ + || { echo "FAIL the purge stopped at the host server"; fails=$((fails + 1)); } +moved_host +out="$(PG_HOST=stopped PG_DROP=fail run_uninstall "default felis minecraft" --purge --yes)" +calls="$(cat "$root/calls")" +expect "a drop that fails says how to finish by hand" "sudo -u postgres dropdb felis" "$out" +expect "and stops the server it started" "SYSTEMCTL stop postgresql" "$calls" +[ ! -e "$root/h/data" ] && echo "PASS a failed drop does not stop the purge halfway" \ + || { echo "FAIL the purge stopped at a failed drop"; fails=$((fails + 1)); } +fresh_host +(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes >/dev/null) # a subshell: sh keeps a prefix assignment to a function +refute "a stopped host server the move never touched is left alone" "SYSTEMCTL start postgresql" "$(cat "$root/calls")" +moved_host +(PG_HOST=none run_uninstall "default felis minecraft" --purge --yes >/dev/null) # a subshell: sh keeps a prefix assignment to a function +refute "a host server removed since the move is not started" "SYSTEMCTL start postgresql" "$(cat "$root/calls")" # --- a purge DROP ROLE would refuse ------------------------------------------------------- # The VM drill: the PG contract tests' felis_pgint was owned by felis, the purge removed the @@ -253,6 +308,13 @@ for u in $(sed -n 's|^[A-Z_]*="/etc/systemd/system/\([^"]*\)"$|\1|p' "$(dirname expect "the uninstaller removes $u" " $u" " $(printf '%s' "$units" | tr '\n' ' ')" done +# The database's cluster and the move's marker are where bootstrap put them, or keep-data +# and purge act on a directory that is not there. +paths="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; printf "%s %s\n" "$PG_DATA_DIR" "$PG_MOVED_MARKER"' "$US")" +bspaths="$(sed -n 's/^PG_DATA_DIR="\(.*\)"$/\1/p; s/^PG_MOVED_MARKER="\(.*\)"$/\1/p' "$(dirname "$US")/bootstrap.sh" | paste -sd ' ' -)" +[ -n "$bspaths" ] && [ "$paths" = "$bspaths" ] && echo "PASS the database paths agree with bootstrap" \ + || { echo "FAIL uninstall's database paths <$paths> differ from bootstrap's <$bspaths>"; fails=$((fails + 1)); } + if [ "$fails" -eq 0 ]; then echo "ALL PASS" else