Unverified Commit 96b4f31b authored by Lemon-miaow's avatar Lemon-miaow
Browse files

feat(uninstall): 按 k3s 内的 felis-postgres 保留或清除数据库并清理迁移留下的宿主副本

parent 40d6e4d0
Loading
Loading
Loading
Loading
+62 −14
Changes for deploy/uninstall.sh: 62 added lines, 14 removed lines.
Original line number Diff line number Diff line
@@ -6,17 +6,20 @@
#   curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes
#
# Options:
#   --purge       also drop the felis database and role, and delete /etc/felis and
#                 /var/lib/felis (the database bundles, and anything an earlier keep-data
#                 run set aside). Asks for the word "purge" unless --yes is given.
#   --purge       also delete /etc/felis and /var/lib/felis (the database's cluster, the
#                 database bundles, and anything an earlier keep-data run set aside), and
#                 drop the felis database and role from a host PostgreSQL an earlier
#                 release installed. Asks for the word "purge" unless --yes is given.
#   --keep-k3s    leave k3s installed and remove only Felis's namespaces and CRD.
#   --remove-k3s  run k3s's own uninstaller even when other workloads live in the cluster.
#   --no-backup   skip the final database bundle keep-data mode takes first.
#   --yes         do not ask.
#
# Keep-data mode (the default) first takes a database bundle (`felis db backup -label
# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the
# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive,
# manual`) and stops if that fails. It leaves /etc/felis (the secrets, felis.toml,
# offsite.env) and /var/lib/felis in place, and with it the felis database: felis-postgres
# keeps its cluster in /var/lib/felis/postgres, stopped cleanly before k3s goes, and a
# host PostgreSQL an earlier release installed keeps its copy. The world, archive,
# registry and upload volumes live under k3s's storage directory, which k3s's uninstaller
# deletes, so they are moved to /var/lib/felis/retained/k3s-storage-<UTC stamp> first; with
# --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go.
@@ -41,6 +44,12 @@ CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}"
VELOCITY_USER="felis-velocity"
DB_NAME="felis"
DB_USER="felis"
CONTROL_NS="felis"
# The database bootstrap runs in k3s, its cluster on a hostPath under DATA_DIR, and the
# marker bootstrap leaves once it moved a host PostgreSQL's felis database into it.
PG_DEPLOYMENT="felis-postgres"
PG_DATA_DIR="${DATA_DIR}/postgres"
PG_MOVED_MARKER="${DATA_DIR}/postgres-moved"
POD_CIDR="10.42.0.0/16"
SERVICE_CIDR="10.43.0.0/16"
FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}"
@@ -131,11 +140,12 @@ print_plan() {
    absent) ;;
  esac
  if [ "$PURGE" = 1 ]; then
    log "  PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles"
    log "  and anything set aside before), every world and archive, the Felis images and Docker's build cache"
    log "  PURGE: ${STATE_DIR} (secrets), ${DATA_DIR} (the ${DB_NAME} database's cluster, the database"
    log "  bundles and anything set aside before), the ${DB_NAME} database and role in a host PostgreSQL,"
    log "  every world and archive, the Felis images and Docker's build cache"
  else
    [ "$BACKUP" = 1 ] && log "  after a final database bundle into ${DATA_DIR}/db-backups"
    log "  kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/"
    log "  kept: ${STATE_DIR}, ${DATA_DIR} (the ${DB_NAME} database in ${PG_DATA_DIR}); the volumes move to ${RETAIN_DIR}/"
  fi
}

@@ -273,8 +283,18 @@ remove_from_cluster() {
  ok "Felis removed from the cluster; k3s stays"
}

# stop_database_pod shuts felis-postgres down cleanly: k3s-killall.sh SIGKILLs every
# container, and the cluster it leaves in PG_DATA_DIR is what a reinstall starts from.
stop_database_pod() {
  kube -n "$CONTROL_NS" scale deployment "$PG_DEPLOYMENT" --replicas=0 >/dev/null 2>&1 || return 0
  kube -n "$CONTROL_NS" wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres \
    --timeout=120s >/dev/null 2>&1 \
    || warn "${PG_DEPLOYMENT} did not stop within 2 minutes; its cluster recovers from its WAL on the next start"
}

remove_k3s() {
  local stamp
  [ "$PURGE" = 1 ] || stop_database_pod
  if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then
    # k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a
    # world while it moves.
@@ -325,7 +345,8 @@ remove_host_files() {

as_postgres() { (cd / && runuser -u postgres -- "$@"); }

# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else.
# remove_hba_block drops the block bootstrap heads pg_hba.conf with (the rules of an
# install on the host server, or the lockout the move into k3s left), and nothing else.
remove_hba_block() { # file
  local tmp
  tmp="$(mktemp)"
@@ -374,23 +395,46 @@ SQL
  die "the ${DB_USER} role still holds $(printf '%s' "$held" | paste -sd ';' - | sed 's/;/; /g'), so DROP ROLE would fail halfway through the purge; nothing was removed. Hand them to postgres first (sudo -u postgres psql -c 'ALTER DATABASE <name> OWNER TO postgres', or REASSIGN OWNED BY ${DB_USER} TO postgres; DROP OWNED BY ${DB_USER}; inside that database), or rerun without --purge"
}

# purge_database drops the felis database and role from a host PostgreSQL an earlier
# release installed; the cluster felis-postgres runs goes with DATA_DIR (purge_state). That
# host server either still serves the platform, or the move into felis-postgres stopped it
# with the pre-move copy left in it for a rollback: a purge takes that copy too and leaves
# the server stopped. The units and k3s are gone by now, so a failure here is a warning
# with the commands to finish by hand, and the purge goes on.
purge_database() {
  [ "$PURGE" = 1 ] || return 0
  local hba started=0
  if ! systemctl is-active --quiet postgresql 2>/dev/null; then
    warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it"
    [ -e "$PG_MOVED_MARKER" ] && systemctl cat postgresql >/dev/null 2>&1 || return 0
    if ! systemctl start postgresql >/dev/null 2>&1; then
      warn "could not start the host PostgreSQL, so its copy of the ${DB_NAME} database from before the move into k3s stays in it; drop it once it runs: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}"
      return 0
    fi
  local hba
    started=1
  fi
  hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
  as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
  if ! as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = '${DB_NAME}' AND pid <> pg_backend_pid();
DROP DATABASE IF EXISTS ${DB_NAME};
DROP ROLE IF EXISTS ${DB_USER};
ALTER SYSTEM RESET listen_addresses;
SQL
  [ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba"
  then
    [ "$started" = 0 ] || systemctl stop postgresql >/dev/null 2>&1 || true
    warn "could not drop the ${DB_NAME} database and role from the host PostgreSQL; drop them by hand: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}"
    return 0
  fi
  if [ -n "$hba" ] && [ -f "$hba" ]; then
    remove_hba_block "$hba"
    rm -f "${hba}.pre-pg-move"
  fi
  if [ "$started" = 1 ]; then
    systemctl stop postgresql
    ok "the host PostgreSQL's copy of '${DB_NAME}' from before the move into k3s dropped; the server stays stopped"
  else
    systemctl restart postgresql
    ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again"
  fi
}

purge_images() {
@@ -419,6 +463,10 @@ purge_state() {
    && cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)"
  if [ -n "$cred" ]; then rm -f "$cred"; fi
  rm -f "$TUNNEL_CONFIG"
  # The file-context rule bootstrap gave the database's cluster directory.
  if command -v semanage >/dev/null 2>&1; then
    semanage fcontext -d "${PG_DATA_DIR}(/.*)?" >/dev/null 2>&1 || true
  fi
  rm -rf "$STATE_DIR" "$DATA_DIR"
  ok "${STATE_DIR} and ${DATA_DIR} removed"
}
@@ -453,7 +501,7 @@ main() {
  if [ "$PURGE" = 1 ]; then
    ok "Felis is gone from this host"
  else
    ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}"
    ok "Felis is removed; the data stays in ${STATE_DIR} and ${DATA_DIR}, the ${DB_NAME} database in ${PG_DATA_DIR}"
    log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\""
  fi
}
+65 −3
Changes for deploy/uninstall_test.sh: 65 added lines, 3 removed lines.
Original line number Diff line number Diff line
@@ -44,6 +44,8 @@ fresh_host() {
    printf 'x\n' > "$root/h/etc/$f"
  done
  printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat"
  mkdir -p "$root/h/data/postgres/18/docker"
  printf '18\n' > "$root/h/data/postgres/18/docker/PG_VERSION"
  for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do
    printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b"
    chmod +x "$root/h/bin/$b"
@@ -72,11 +74,19 @@ run_uninstall() {
    . "$0"
    calls="$ROOT/calls"
    id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; }
    # PG_HOST: the host PostgreSQL is active (the default), stopped, or not installed (none);
    # PG_START=fail for one that will not start.
    systemctl() {
      echo "SYSTEMCTL $*" >> "$calls"
      case "$*" in "is-active --quiet firewalld") return 1 ;; esac
      return 0
      case "$*" in
        "is-active --quiet firewalld") return 1 ;;
        "is-active --quiet postgresql") [ "${PG_HOST:-active}" = active ] ;;
        "cat postgresql") [ "${PG_HOST:-active}" != none ] ;;
        "start postgresql") [ "${PG_START:-}" != fail ] ;;
        *) return 0 ;;
      esac
    }
    semanage() { echo "SEMANAGE $*" >> "$calls"; }
    kube() {
      echo "KUBE $*" >> "$calls"
      case "$*" in
@@ -98,7 +108,7 @@ run_uninstall() {
              echo "PSQL-CHECK" >> "$calls"
              [ "${PG_CHECK:-}" = fail ] && { echo "psql: error: connection refused" >&2; return 2; }
              [ -z "${PG_HELD:-}" ] || printf "%s\n" "$PG_HELD" ;;
            *) echo "PSQL $* $sql" >> "$calls" ;;
            *) echo "PSQL $* $sql" >> "$calls"; [ "${PG_DROP:-}" != fail ] ;;
          esac ;;
      esac
    }
@@ -135,6 +145,15 @@ refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls"
expect "the velocity user is removed" "USERDEL felis-velocity" "$calls"
expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out"
[ -f "$root/h/data/postgres/18/docker/PG_VERSION" ] && echo "PASS the database's cluster stays for the reinstall" \
  || { echo "FAIL keep-data removed the database's cluster"; fails=$((fails + 1)); }
stop_at="$(grep -n 'KUBE -n felis scale deployment felis-postgres --replicas=0' "$root/calls" | head -n 1 | cut -d: -f1)"
kill_at="$(grep -n 'RUN k3s-killall.sh' "$root/calls" | head -n 1 | cut -d: -f1)"
[ -n "$stop_at" ] && [ -n "$kill_at" ] && [ "$stop_at" -lt "$kill_at" ] \
  && echo "PASS the database shuts down cleanly before k3s-killall.sh kills what is left" \
  || { echo "FAIL felis-postgres was not stopped before k3s-killall.sh: $calls"; fails=$((fails + 1)); }
expect "and the uninstall waits for it to stop" "KUBE -n felis wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres" "$calls"
refute "keep-data keeps the cluster directory's SELinux rule" "SEMANAGE" "$calls"

# --- a failed final bundle stops everything ------------------------------------------------
fresh_host
@@ -189,6 +208,42 @@ case "$hba" in
  *) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;;
esac
expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls"
expect "purge drops the cluster directory's SELinux rule" "SEMANAGE fcontext -d $root/h/data/postgres(/.*)?" "$calls"
refute "purge leaves k3s's pods to k3s-uninstall.sh" "scale deployment felis-postgres" "$calls"

# --- purge after the move into felis-postgres ---------------------------------------------
# The move stopped the host server with the felis database left in it for a rollback.
moved_host() { fresh_host; printf 'moved\n' > "$root/h/data/postgres-moved"; printf 'saved\n' > "$root/h/hba.conf.pre-pg-move"; }
moved_host
out="$(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
expect "purge starts the stopped host server to drop the pre-move copy" "SYSTEMCTL start postgresql" "$calls"
expect "and drops it" "DROP DATABASE IF EXISTS felis;" "$calls"
expect "the server stays stopped afterwards" "SYSTEMCTL stop postgresql" "$calls"
refute "and is not restarted" "SYSTEMCTL restart postgresql" "$calls"
refute "the lockout leaves pg_hba.conf" "FELIS MANAGED" "$(cat "$root/h/hba.conf")"
[ ! -e "$root/h/hba.conf.pre-pg-move" ] && echo "PASS the pg_hba.conf the move saved goes with the copy" \
  || { echo "FAIL purge left pg_hba.conf.pre-pg-move"; fails=$((fails + 1)); }
moved_host
out="$(PG_HOST=stopped PG_START=fail run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
expect "a host server that will not start is named" "could not start the host PostgreSQL" "$out"
refute "so nothing is dropped" "DROP DATABASE" "$calls"
[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS and the purge goes on" \
  || { echo "FAIL the purge stopped at the host server"; fails=$((fails + 1)); }
moved_host
out="$(PG_HOST=stopped PG_DROP=fail run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
expect "a drop that fails says how to finish by hand" "sudo -u postgres dropdb felis" "$out"
expect "and stops the server it started" "SYSTEMCTL stop postgresql" "$calls"
[ ! -e "$root/h/data" ] && echo "PASS a failed drop does not stop the purge halfway" \
  || { echo "FAIL the purge stopped at a failed drop"; fails=$((fails + 1)); }
fresh_host
(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes >/dev/null)  # a subshell: sh keeps a prefix assignment to a function
refute "a stopped host server the move never touched is left alone" "SYSTEMCTL start postgresql" "$(cat "$root/calls")"
moved_host
(PG_HOST=none run_uninstall "default felis minecraft" --purge --yes >/dev/null)  # a subshell: sh keeps a prefix assignment to a function
refute "a host server removed since the move is not started" "SYSTEMCTL start postgresql" "$(cat "$root/calls")"

# --- a purge DROP ROLE would refuse -------------------------------------------------------
# The VM drill: the PG contract tests' felis_pgint was owned by felis, the purge removed the
@@ -253,6 +308,13 @@ for u in $(sed -n 's|^[A-Z_]*="/etc/systemd/system/\([^"]*\)"$|\1|p' "$(dirname
  expect "the uninstaller removes $u" " $u" " $(printf '%s' "$units" | tr '\n' ' ')"
done

# The database's cluster and the move's marker are where bootstrap put them, or keep-data
# and purge act on a directory that is not there.
paths="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; printf "%s %s\n" "$PG_DATA_DIR" "$PG_MOVED_MARKER"' "$US")"
bspaths="$(sed -n 's/^PG_DATA_DIR="\(.*\)"$/\1/p; s/^PG_MOVED_MARKER="\(.*\)"$/\1/p' "$(dirname "$US")/bootstrap.sh" | paste -sd ' ' -)"
[ -n "$bspaths" ] && [ "$paths" = "$bspaths" ] && echo "PASS the database paths agree with bootstrap" \
  || { echo "FAIL uninstall's database paths <$paths> differ from bootstrap's <$bspaths>"; fails=$((fails + 1)); }

if [ "$fails" -eq 0 ]; then
  echo "ALL PASS"
else