diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a8c0d9f..4335d81 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -135,6 +135,7 @@ jobs: ./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh') - run: sh deploy/bootstrap_test.sh + - run: sh deploy/uninstall_test.sh panel: runs-on: ubuntu-latest diff --git a/deploy/uninstall.sh b/deploy/uninstall.sh new file mode 100644 index 0000000..80c4bf1 --- /dev/null +++ b/deploy/uninstall.sh @@ -0,0 +1,428 @@ +#!/usr/bin/env bash +# Removes what deploy/bootstrap.sh installed on this host. +# +# sudo bash deploy/uninstall.sh # remove Felis, keep the data +# sudo bash deploy/uninstall.sh --purge # remove the data too +# curl -fsSL /deploy/uninstall.sh | sudo bash -s -- --yes +# +# Options: +# --purge also drop the felis database and role, and delete /etc/felis and +# /var/lib/felis (the database bundles, and anything an earlier keep-data +# run set aside). Asks for the word "purge" unless --yes is given. +# --keep-k3s leave k3s installed and remove only Felis's namespaces and CRD. +# --remove-k3s run k3s's own uninstaller even when other workloads live in the cluster. +# --no-backup skip the final database bundle keep-data mode takes first. +# --yes do not ask. +# +# Keep-data mode (the default) first takes a database bundle (`felis db backup -label +# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the +# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive, +# registry and upload volumes live under k3s's storage directory, which k3s's uninstaller +# deletes, so they are moved to /var/lib/felis/retained/k3s-storage- first; with +# --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go. +# docs/operations.md walks through reinstalling on top of what is left. +# +# Either mode leaves packages alone (Docker, PostgreSQL, git and the rest), and the swap +# file a low-memory host got: other software may use them. docs/operations.md lists the +# package commands for a bare host. +set -Eeuo pipefail + +STATE_DIR="${STATE_DIR:-/etc/felis}" +DATA_DIR="${DATA_DIR:-/var/lib/felis}" +RETAIN_DIR="${DATA_DIR}/retained" +HOST_BIN="${HOST_BIN:-/usr/local/bin/felis}" +OPT_DIR="${OPT_DIR:-/opt/felis}" +UNIT_DIR="${UNIT_DIR:-/etc/systemd/system}" +K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}" +K3S_STORAGE="${K3S_STORAGE:-/var/lib/rancher/k3s/storage}" +K3S_REGISTRIES="${K3S_REGISTRIES:-/etc/rancher/k3s/registries.yaml}" +export KUBECONFIG="${KUBECONFIG:-/etc/rancher/k3s/k3s.yaml}" +CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}" +VELOCITY_USER="felis-velocity" +DB_NAME="felis" +DB_USER="felis" +POD_CIDR="10.42.0.0/16" +SERVICE_CIDR="10.43.0.0/16" +FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}" +CONFIRM_TTY="${CONFIRM_TTY:-/dev/tty}" +FELIS_NAMESPACES=(felis minecraft felis-build) +FELIS_CRD="minecraftservers.felis.lolicon.best" +# Every unit the installer and `felis setup` write. Timers first, so none fires into a +# service that is already gone. +FELIS_UNITS=( + felis-db-backup.timer felis-watchdog.timer felis-offsite.timer felis-build-tools.timer + felis-db-backup.service felis-watchdog.service felis-offsite.service felis-build-tools.service + felis-velocity.service felis-nano.service cloudflared-felis.service + felis-postgres-firewall.service +) + +PURGE=0 +K3S_MODE=auto +BACKUP=1 +ASSUME_YES=0 +TUNNEL_CONFIG="" + +log() { printf '\033[1;36m[felis]\033[0m %s\n' "$*"; } +ok() { printf '\033[1;32m[ ok ]\033[0m %s\n' "$*"; } +warn() { printf '\033[1;33m[warn]\033[0m %s\n' "$*" >&2; } +die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; } + +parse_args() { + while [ $# -gt 0 ]; do + case "$1" in + --purge) PURGE=1 ;; + --keep-k3s) K3S_MODE=keep ;; + --remove-k3s) K3S_MODE=remove ;; + --no-backup) BACKUP=0 ;; + --yes|-y) ASSUME_YES=1 ;; + -h|--help) printf 'usage: uninstall.sh [--purge] [--keep-k3s|--remove-k3s] [--no-backup] [--yes]\n'; exit 0 ;; + *) die "unknown option: $1 (see --help)" ;; + esac + shift + done +} + +kube() { "${K3S_BIN_DIR}/k3s" kubectl "$@"; } +k3s_present() { [ -x "${K3S_BIN_DIR}/k3s" ]; } + +# foreign_namespaces prints the namespaces that are neither k3s's own nor Felis's, one per +# line. A cluster with none of them exists for Felis alone, and removing k3s takes nothing +# else with it. +foreign_namespaces() { + kube get namespaces -o 'jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}' \ + | awk '$0 != "" && $0 != "default" && $0 !~ /^kube-/ && $0 != "felis" && $0 != "minecraft" && $0 != "felis-build"' +} + +# decide_k3s turns K3S_MODE=auto into keep or remove. +decide_k3s() { + k3s_present || { K3S_MODE=absent; return 0; } + [ "$K3S_MODE" = auto ] || return 0 + local others + if ! others="$(foreign_namespaces)"; then + die "k3s does not answer, so this cannot tell whether it runs anything besides Felis; start it (systemctl start k3s) or pass --keep-k3s or --remove-k3s" + fi + if [ -z "$others" ]; then + K3S_MODE=remove + else + K3S_MODE=keep + log "k3s also runs namespaces Felis did not create ($(printf '%s' "$others" | tr '\n' ' ')); leaving k3s installed" + fi +} + +confirm() { + [ "$ASSUME_YES" = 1 ] && return 0 + local want="yes" answer="" + [ "$PURGE" = 1 ] && want="purge" + # A piped script has no stdin to read from; the terminal is asked directly. + { exec 3<"$CONFIRM_TTY" 4>>"$CONFIRM_TTY"; } 2>/dev/null || die "no terminal to confirm on; re-run with --yes" + printf 'Type "%s" to continue: ' "$want" >&4 + read -r answer <&3 || true + exec 3<&- 4>&- + [ "$answer" = "$want" ] || die "not confirmed; nothing was changed" +} + +print_plan() { + log "this will remove from $(uname -n):" + log " the felis-* systemd units, cloudflared-felis.service, the ${VELOCITY_USER} user," + log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld openings" + case "$K3S_MODE" in + remove) log " k3s, with everything in it (${K3S_BIN_DIR}/k3s-uninstall.sh)" ;; + keep) log " Felis's namespaces (${FELIS_NAMESPACES[*]}) and the ${FELIS_CRD} CRD; k3s stays" ;; + absent) ;; + esac + if [ "$PURGE" = 1 ]; then + log " PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles" + log " and anything set aside before), every world and archive, the Felis images and Docker's build cache" + else + [ "$BACKUP" = 1 ] && log " after a final database bundle into ${DATA_DIR}/db-backups" + log " kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/" + fi +} + +final_backup() { + [ "$PURGE" = 0 ] && [ "$BACKUP" = 1 ] || return 0 + [ -x "$HOST_BIN" ] && [ -r "${STATE_DIR}/felis.host.toml" ] || { + warn "no ${HOST_BIN} or ${STATE_DIR}/felis.host.toml; skipping the final database bundle" + return 0 + } + log "taking a final database bundle" + "$HOST_BIN" db backup -config "${STATE_DIR}/felis.host.toml" -label manual \ + || die "the final database bundle failed, so nothing was removed. Fix the database (sudo felis db check), or pass --no-backup to go on without one" + ok "database bundle written to ${DATA_DIR}/db-backups" +} + +# game_port reads the proxy's port from velocity.toml before /opt/felis goes. +game_port() { + local toml="${OPT_DIR}/velocity/velocity.toml" port="" + [ -r "$toml" ] && port="$(sed -n 's/^bind *= *"[^"]*:\([0-9][0-9]*\)".*/\1/p' "$toml" | head -n 1)" + printf '%s\n' "${FELIS_GAME_PORT:-${port:-25565}}" +} + +# tunnel_config reads the cloudflared config felis setup pointed its unit at (by default +# /etc/felis/cloudflared.yml) before the unit goes. +tunnel_config() { + local unit="${UNIT_DIR}/cloudflared-felis.service" path="" + [ -r "$unit" ] && path="$(sed -n 's/^ExecStart=.* --config \([^ ]*\) tunnel run$/\1/p' "$unit" | head -n 1)" + printf '%s\n' "${path:-${STATE_DIR}/cloudflared.yml}" +} + +# nano_port reads felis-nano's port from its unit before the unit goes. +nano_port() { + local unit="${UNIT_DIR}/felis-nano.service" + [ -r "$unit" ] || return 0 + sed -n 's/^ExecStart=.* -listen [^ ]*:\([0-9][0-9]*\).*$/\1/p' "$unit" | head -n 1 +} + +remove_units() { + local unit removed=0 + for unit in "${FELIS_UNITS[@]}"; do + [ -f "${UNIT_DIR}/${unit}" ] || continue + systemctl disable --now "$unit" >/dev/null 2>&1 || systemctl stop "$unit" >/dev/null 2>&1 || true + rm -f "${UNIT_DIR}/${unit}" + removed=$((removed + 1)) + done + systemctl daemon-reload + systemctl reset-failed >/dev/null 2>&1 || true + ok "${removed} systemd unit(s) removed" +} + +remove_nft_tables() { + command -v nft >/dev/null 2>&1 || return 0 + local t + for t in felis_postgres felis_edge; do + if nft list table inet "$t" >/dev/null 2>&1; then + nft delete table inet "$t" + ok "nftables table inet ${t} removed" + fi + done +} + +# remove_firewalld_rules takes back what configure_k3s_firewall, configure_velocity_firewall +# and the nano setup opened. The k3s ones stay when k3s does. +remove_firewalld_rules() { # game-port nano-port + command -v firewall-cmd >/dev/null 2>&1 || return 0 + systemctl is-active --quiet firewalld || return 0 + local game="$1" nano="$2" port rule changed=0 + local ports=("${game}/tcp" "${FELIS_PANEL_NODEPORT}/tcp") + [ -n "$nano" ] && ports+=("${nano}/tcp") + [ "$K3S_MODE" = remove ] && ports+=("6443/tcp") + for port in "${ports[@]}"; do + if firewall-cmd --permanent --query-port="$port" >/dev/null 2>&1; then + firewall-cmd --permanent --remove-port="$port" >/dev/null + changed=1 + fi + done + if [ -n "$nano" ]; then + while IFS= read -r rule; do + case "$rule" in + *"port=\"${nano}\""*) firewall-cmd --permanent --remove-rich-rule="$rule" >/dev/null; changed=1 ;; + esac + done < <(firewall-cmd --permanent --list-rich-rules 2>/dev/null) + fi + if [ "$K3S_MODE" = remove ]; then + local cidr + for cidr in "$POD_CIDR" "$SERVICE_CIDR"; do + if firewall-cmd --permanent --zone=trusted --query-source="$cidr" >/dev/null 2>&1; then + firewall-cmd --permanent --zone=trusted --remove-source="$cidr" >/dev/null + changed=1 + fi + done + fi + if [ "$changed" = 1 ]; then + firewall-cmd --reload >/dev/null + ok "firewalld openings removed" + fi +} + +# retain_volumes_in_cluster keeps every volume Felis's claims are bound to when the +# namespaces go: local-path deletes a Delete-policy volume's directory with its claim. +retain_volumes_in_cluster() { + local pv + while IFS= read -r pv; do + [ -n "$pv" ] || continue + kube patch pv "$pv" -p '{"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' >/dev/null + done < <(kube get pv -o 'jsonpath={range .items[*]}{.metadata.name} {.spec.claimRef.namespace}{"\n"}{end}' \ + | awk '$2 == "felis" || $2 == "minecraft" || $2 == "felis-build" { print $1 }') + ok "Felis's volumes set to Retain; their directories stay under ${K3S_STORAGE}" +} + +remove_from_cluster() { + [ "$PURGE" = 1 ] || retain_volumes_in_cluster + log "deleting Felis's namespaces and CRD" + # The operator is part of what goes, so nothing would clear a MinecraftServer finalizer + # and the minecraft namespace would stay Terminating. Drop them first. + local s + while IFS= read -r s; do + [ -n "$s" ] || continue + kube -n minecraft patch minecraftserver "$s" --type=merge -p '{"metadata":{"finalizers":null}}' >/dev/null 2>&1 || true + done < <(kube -n minecraft get minecraftservers -o 'jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}' 2>/dev/null || true) + kube delete namespace "${FELIS_NAMESPACES[@]}" --ignore-not-found --wait=true --timeout=300s >/dev/null \ + || warn "a namespace is still terminating; check: k3s kubectl get namespaces" + kube delete crd "$FELIS_CRD" --ignore-not-found >/dev/null || true + if [ "$PURGE" = 1 ]; then + local pv + while IFS= read -r pv; do + [ -n "$pv" ] && kube delete pv "$pv" --ignore-not-found >/dev/null + done < <(kube get pv -o 'jsonpath={range .items[*]}{.metadata.name} {.spec.claimRef.namespace}{"\n"}{end}' \ + | awk '$2 == "felis" || $2 == "minecraft" || $2 == "felis-build" { print $1 }') + fi + if [ -f "$K3S_REGISTRIES" ] && grep -q 'registry\.felis\.svc' "$K3S_REGISTRIES"; then + rm -f "$K3S_REGISTRIES" + systemctl restart k3s + fi + ok "Felis removed from the cluster; k3s stays" +} + +remove_k3s() { + local stamp + if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then + # k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a + # world while it moves. + "${K3S_BIN_DIR}/k3s-killall.sh" >/dev/null 2>&1 || systemctl stop k3s + stamp="$(date -u +%Y%m%dT%H%M%SZ)" + install -d -m 0700 "$RETAIN_DIR" + mv "$K3S_STORAGE" "${RETAIN_DIR}/k3s-storage-${stamp}" + ok "volumes moved to ${RETAIN_DIR}/k3s-storage-${stamp}" + fi + if [ -x "${K3S_BIN_DIR}/k3s-uninstall.sh" ]; then + log "running k3s-uninstall.sh" + "${K3S_BIN_DIR}/k3s-uninstall.sh" >/dev/null 2>&1 || warn "k3s-uninstall.sh reported an error; check /var/lib/rancher and /etc/rancher" + ok "k3s removed" + else + warn "k3s is at ${K3S_BIN_DIR}/k3s but ${K3S_BIN_DIR}/k3s-uninstall.sh is missing; remove k3s by hand" + fi +} + +# remove_cloudflared_binary deletes the binary the installer put in /usr/local/bin, unless +# a unit other than Felis's still runs it. +remove_cloudflared_binary() { + [ -x "$CLOUDFLARED_BIN" ] || return 0 + local others + others="$(grep -ls "$CLOUDFLARED_BIN" "${UNIT_DIR}"/*.service /lib/systemd/system/*.service /usr/lib/systemd/system/*.service 2>/dev/null || true)" + if [ -n "$others" ]; then + log "leaving ${CLOUDFLARED_BIN}: $(printf '%s' "$others" | tr '\n' ' ')uses it" + return 0 + fi + rm -f "$CLOUDFLARED_BIN" + ok "${CLOUDFLARED_BIN} removed" +} + +remove_host_files() { + if id "$VELOCITY_USER" >/dev/null 2>&1; then + userdel "$VELOCITY_USER" >/dev/null 2>&1 || warn "could not remove the ${VELOCITY_USER} user" + fi + rm -rf "$OPT_DIR" + rm -f "$HOST_BIN" "${HOST_BIN}.new" + remove_cloudflared_binary + if [ "$PURGE" = 0 ]; then + # What describes the removed install goes; what a reinstall reuses stays. Without + # bootstrap.done the next run takes the first-install path. + rm -f "${STATE_DIR}/bootstrap.done" "${STATE_DIR}/system-server-images" \ + "${STATE_DIR}/velocity.fingerprint" "${STATE_DIR}/previous-felis-image" + fi + ok "${OPT_DIR} and ${HOST_BIN} removed" +} + +as_postgres() { (cd / && runuser -u postgres -- "$@"); } + +# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else. +remove_hba_block() { # file + local tmp + tmp="$(mktemp)" + awk ' + $0 == "# BEGIN FELIS MANAGED HBA" { skip = 1; next } + $0 == "# END FELIS MANAGED HBA" { skip = 0; blank = 1; next } + blank && $0 == "" { blank = 0; next } + { blank = 0 } + !skip { print } + ' "$1" > "$tmp" + cat "$tmp" > "$1" + rm -f "$tmp" +} + +purge_database() { + [ "$PURGE" = 1 ] || return 0 + if ! systemctl is-active --quiet postgresql 2>/dev/null; then + warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it" + return 0 + fi + local hba + hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)" + as_postgres psql -v ON_ERROR_STOP=1 -q < pg_backend_pid(); +DROP DATABASE IF EXISTS ${DB_NAME}; +DROP ROLE IF EXISTS ${DB_USER}; +ALTER SYSTEM RESET listen_addresses; +SQL + [ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba" + systemctl restart postgresql + ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again" +} + +purge_images() { + [ "$PURGE" = 1 ] || return 0 + command -v docker >/dev/null 2>&1 || return 0 + # The installer stops Docker after its builds; start it just long enough to clean up. + local was_active=1 refs + systemctl is-active --quiet docker || { was_active=0; systemctl start docker >/dev/null 2>&1 || return 0; } + refs="$(docker image ls --format '{{.Repository}}:{{.Tag}}' | grep -E '^(registry\.felis\.svc:5000/felis/|felis/)' || true)" + if [ -n "$refs" ]; then + # shellcheck disable=SC2086 # one ref per word + docker image rm -f $refs >/dev/null 2>&1 || true + fi + docker builder prune -af >/dev/null 2>&1 || true + [ "$was_active" = 1 ] || systemctl stop docker docker.socket >/dev/null 2>&1 || true + ok "Felis images and Docker's build cache removed" +} + +purge_state() { + [ "$PURGE" = 1 ] || return 0 + # felis setup's tunnel credentials sit beside cloudflared's login (cert.pem, which stays: + # it is the Cloudflare account's, not Felis's). The tunnel itself lives on in the account + # until it is deleted there (docs/operations.md). + local cred="" + [ -r "$TUNNEL_CONFIG" ] \ + && cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)" + if [ -n "$cred" ]; then rm -f "$cred"; fi + rm -f "$TUNNEL_CONFIG" + rm -rf "$STATE_DIR" "$DATA_DIR" + ok "${STATE_DIR} and ${DATA_DIR} removed" +} + +main() { + parse_args "$@" + [ "$(id -u)" = 0 ] || die "run as root: sudo bash $0" + [ -e "$STATE_DIR" ] || [ -e "$HOST_BIN" ] || [ -e "$OPT_DIR" ] \ + || die "no Felis install here (${STATE_DIR}, ${HOST_BIN} and ${OPT_DIR} are all absent)" + decide_k3s + print_plan + confirm + final_backup + + local game nano + game="$(game_port)" + nano="$(nano_port)" + TUNNEL_CONFIG="$(tunnel_config)" + remove_units + case "$K3S_MODE" in + remove) remove_k3s ;; + keep) remove_from_cluster ;; + esac + remove_nft_tables + remove_firewalld_rules "$game" "$nano" + remove_host_files + purge_database + purge_images + purge_state + + if [ "$PURGE" = 1 ]; then + ok "Felis is gone from this host" + else + ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}" + log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\"" + fi +} + +if [ "${FELIS_UNINSTALL_SOURCED:-0}" != 1 ]; then + main "$@" +fi diff --git a/deploy/uninstall_test.sh b/deploy/uninstall_test.sh new file mode 100644 index 0000000..a90a20e --- /dev/null +++ b/deploy/uninstall_test.sh @@ -0,0 +1,214 @@ +#!/bin/sh +# Checks for deploy/uninstall.sh. Run it as: sh deploy/uninstall_test.sh +# +# The script is sourced with FELIS_UNINSTALL_SOURCED=1, every host path pointed into a +# scratch directory, and the commands that would change the host (systemctl, k3s, nft, +# firewall-cmd, runuser, docker, userdel) replaced by stubs that log their arguments. +set -u + +US="${1:-$(dirname "$0")/uninstall.sh}" +[ -f "$US" ] || { echo "no such script: $US"; exit 1; } +fails=0 + +expect() { # label needle haystack + case "$3" in + *"$2"*) echo "PASS $1" ;; + *) echo "FAIL $1: expected <$2> in:"; echo "$3"; fails=$((fails + 1)) ;; + esac +} +refute() { # label needle haystack + case "$3" in + *"$2"*) echo "FAIL $1: did not expect <$2> in:"; echo "$3"; fails=$((fails + 1)) ;; + *) echo "PASS $1" ;; + esac +} + +root="$(mktemp -d)" +trap 'rm -rf "$root"' EXIT + +# fresh_host lays out what an install leaves: units, state, /opt/felis, the host binary, a +# k3s with its uninstaller and a volume, a tunnel config and its credentials. +fresh_host() { + rm -rf "$root/h" + mkdir -p "$root/h/units" "$root/h/etc" "$root/h/data/db-backups" "$root/h/opt/velocity" \ + "$root/h/bin" "$root/h/storage/pvc-1_minecraft_world-a-0" "$root/h/cf" + for u in felis-db-backup.timer felis-db-backup.service felis-velocity.service felis-postgres-firewall.service; do + printf '[Unit]\n' > "$root/h/units/$u" + done + printf '[Service]\nExecStart=/usr/local/bin/cloudflared --config %s tunnel run\n' "$root/h/etc/cloudflared.yml" \ + > "$root/h/units/cloudflared-felis.service" + printf 'tunnel: abc\ncredentials-file: %s\n' "$root/h/cf/abc.json" > "$root/h/etc/cloudflared.yml" + printf '{}\n' > "$root/h/cf/abc.json" + printf 'bind = "0.0.0.0:25577"\n' > "$root/h/opt/velocity/velocity.toml" + for f in secrets.env felis.host.toml bootstrap.done system-server-images velocity.fingerprint; do + printf 'x\n' > "$root/h/etc/$f" + done + printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat" + for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do + printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b" + chmod +x "$root/h/bin/$b" + done + cat > "$root/h/hba.conf" <<'EOF' +# BEGIN FELIS MANAGED HBA +# Felis rules must precede distro defaults such as 127.0.0.1 ident. +host felis felis 127.0.0.1/32 scram-sha-256 +# END FELIS MANAGED HBA + +local all all peer +host all all 127.0.0.1/32 ident +EOF + : > "$root/calls" +} + +# run_uninstall : runs main with the stubs; is what +# `kubectl get namespaces` answers, or "down" for a k3s that does not answer. +run_uninstall() { + ns="$1"; shift + NS="$ns" ROOT="$root" STATE_DIR="$root/h/etc" DATA_DIR="$root/h/data" HOST_BIN="$root/h/bin/felis" \ + OPT_DIR="$root/h/opt" UNIT_DIR="$root/h/units" K3S_BIN_DIR="$root/h/bin" \ + K3S_STORAGE="$root/h/storage" K3S_REGISTRIES="$root/h/registries.yaml" \ + CLOUDFLARED_BIN="$root/h/no-cloudflared" FELIS_UNINSTALL_SOURCED=1 bash -c ' + set -Eeuo pipefail + . "$0" + calls="$ROOT/calls" + id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; } + systemctl() { + echo "SYSTEMCTL $*" >> "$calls" + case "$*" in "is-active --quiet firewalld") return 1 ;; esac + return 0 + } + kube() { + echo "KUBE $*" >> "$calls" + case "$*" in + "get namespaces"*) [ "$NS" = down ] && return 1; printf "%s\n" $NS ;; + "get pv"*) printf "pvc-1 minecraft\npvc-9 other\n" ;; + esac + } + nft() { echo "NFT $*" >> "$calls"; return 1; } + runuser() { + shift 3 + case "$*" in + *"SHOW hba_file"*) echo "$ROOT/h/hba.conf" ;; + *) echo "PSQL $* $(cat)" >> "$calls" ;; + esac + } + docker() { echo "DOCKER $*" >> "$calls"; } + userdel() { echo "USERDEL $*" >> "$calls"; } + uname() { echo testhost; } + main "$@"' "$US" "$@" 2>&1 +} + +# --- keep-data, a cluster that runs only Felis ------------------------------------------- +fresh_host +out="$(run_uninstall "default kube-system felis minecraft felis-build" --yes)" +calls="$(cat "$root/calls")" +expect "keep-data takes a final bundle first" "RUN felis db backup -config $root/h/etc/felis.host.toml -label manual" "$calls" +expect "a Felis-only cluster is removed with k3s's uninstaller" "RUN k3s-uninstall.sh" "$calls" +expect "the pods are stopped before the volumes move" "RUN k3s-killall.sh" "$calls" +kept="$(ls "$root/h/data/retained" 2>/dev/null)" +expect "the volumes are set aside before k3s deletes them" "k3s-storage-" "$kept" +[ -f "$root/h/data/retained/$kept/pvc-1_minecraft_world-a-0/level.dat" ] \ + && echo "PASS a world survives the uninstall" \ + || { echo "FAIL the world did not survive: $(ls -R "$root/h/data")"; fails=$((fails + 1)); } +[ -f "$root/h/etc/secrets.env" ] && [ -f "$root/h/etc/felis.host.toml" ] \ + && echo "PASS the secrets and felis.toml stay" \ + || { echo "FAIL keep-data removed the secrets"; fails=$((fails + 1)); } +[ ! -e "$root/h/etc/bootstrap.done" ] && [ ! -e "$root/h/etc/velocity.fingerprint" ] \ + && echo "PASS the markers of the removed install go, so a reinstall starts fresh" \ + || { echo "FAIL bootstrap.done or the proxy fingerprint was left"; fails=$((fails + 1)); } +refute "keep-data leaves the database alone" "DROP DATABASE" "$calls" +[ ! -e "$root/h/opt" ] && [ ! -e "$root/h/bin/felis" ] \ + && echo "PASS /opt/felis and the host binary are removed" \ + || { echo "FAIL /opt/felis or the host binary is still there"; fails=$((fails + 1)); } +[ -z "$(ls "$root/h/units")" ] && echo "PASS every Felis unit file is removed" \ + || { echo "FAIL units left: $(ls "$root/h/units")"; fails=$((fails + 1)); } +expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls" +expect "the velocity user is removed" "USERDEL felis-velocity" "$calls" +expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out" + +# --- a failed final bundle stops everything ------------------------------------------------ +fresh_host +printf '#!/bin/sh\necho "RUN felis $*" >> "%s"\nexit 1\n' "$root/calls" > "$root/h/bin/felis" +out="$(run_uninstall "default felis minecraft" --yes)" +expect "a failed bundle is fatal" "the final database bundle failed, so nothing was removed" "$out" +[ -d "$root/h/opt" ] && [ -f "$root/h/units/felis-velocity.service" ] \ + && echo "PASS nothing is removed when the bundle fails" \ + || { echo "FAIL the uninstall went on after the bundle failed"; fails=$((fails + 1)); } +run_uninstall "default felis minecraft" --yes --no-backup >/dev/null +[ ! -d "$root/h/opt" ] && echo "PASS --no-backup goes on without one" \ + || { echo "FAIL --no-backup did not remove anything"; fails=$((fails + 1)); } + +# --- a shared cluster -------------------------------------------------------------------- +fresh_host +out="$(run_uninstall "default kube-system felis minecraft felis-build shop" --yes)" +calls="$(cat "$root/calls")" +refute "a cluster that runs something else keeps k3s" "RUN k3s-uninstall.sh" "$calls" +expect "the reason names the other namespace" "shop" "$out" +expect "Felis's volumes are retained before their claims go" 'KUBE patch pv pvc-1 -p {"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' "$calls" +refute "a volume of another namespace is not touched" "patch pv pvc-9" "$calls" +expect "Felis's namespaces are deleted" "KUBE delete namespace felis minecraft felis-build" "$calls" +expect "the CRD is deleted" "KUBE delete crd minecraftservers.felis.lolicon.best" "$calls" + +out="$(fresh_host; run_uninstall down --yes)" +expect "a k3s that does not answer stops the run" "k3s does not answer" "$out" +fresh_host +run_uninstall down --yes --keep-k3s >/dev/null +calls="$(cat "$root/calls")" +refute "--keep-k3s never runs k3s's uninstaller" "RUN k3s-uninstall.sh" "$calls" + +# --- purge ------------------------------------------------------------------------------- +fresh_host +out="$(run_uninstall "default felis minecraft" --purge --yes)" +calls="$(cat "$root/calls")" +refute "purge takes no bundle" "db backup" "$calls" +expect "purge drops the database" "DROP DATABASE IF EXISTS felis;" "$calls" +expect "purge drops the role" "DROP ROLE IF EXISTS felis;" "$calls" +expect "purge puts listen_addresses back" "ALTER SYSTEM RESET listen_addresses;" "$calls" +[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS purge removes /etc/felis and /var/lib/felis" \ + || { echo "FAIL purge left state behind"; fails=$((fails + 1)); } +[ ! -e "$root/h/cf/abc.json" ] && echo "PASS purge removes the tunnel's credentials file" \ + || { echo "FAIL the tunnel credentials are still there"; fails=$((fails + 1)); } +[ ! -e "$root/h/data/retained" ] && echo "PASS purge does not set the volumes aside" \ + || { echo "FAIL purge kept the volumes"; fails=$((fails + 1)); } +hba="$(cat "$root/h/hba.conf")" +refute "the Felis block leaves pg_hba.conf" "FELIS MANAGED" "$hba" +expect "the distro's own rules stay" "host all all 127.0.0.1/32 ident" "$hba" +case "$hba" in + "local all all peer"*) echo "PASS no blank line is left where the block was" ;; + *) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;; +esac +expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls" + +# --- the pieces read before they are removed --------------------------------------------- +fresh_host +lib() { STATE_DIR="$root/h/etc" OPT_DIR="$root/h/opt" UNIT_DIR="$root/h/units" FELIS_UNINSTALL_SOURCED=1 \ + bash -c '. "$0"; '"$1" "$US"; } +expect "the game port comes from velocity.toml" "25577" "$(lib game_port)" +expect "FELIS_GAME_PORT overrides it" "25599" "$(FELIS_GAME_PORT=25599 lib game_port)" +rm "$root/h/opt/velocity/velocity.toml" +expect "without velocity.toml the port is the default" "25565" "$(lib game_port)" +printf '[Service]\nExecStart=/usr/local/bin/felis nano -listen 10.0.0.5:8082 -config x\n' > "$root/h/units/felis-nano.service" +expect "the nano port comes from its unit" "8082" "$(lib nano_port)" +expect "the tunnel config comes from the cloudflared unit" "$root/h/etc/cloudflared.yml" "$(lib tunnel_config)" +rm "$root/h/units/cloudflared-felis.service" +expect "without the unit the tunnel config is the default path" "$root/h/etc/cloudflared.yml" "$(lib tunnel_config)" + +out="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; parse_args --bogus' "$US" 2>&1)" +expect "an unknown option is refused" "unknown option: --bogus" "$out" +printf 'nope\n' > "$root/tty" +out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)" +expect "any answer but the word stops it" "not confirmed; nothing was changed" "$out" +printf 'yes\n' > "$root/tty" +out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)" +expect "yes goes on" "WENT ON" "$out" +out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; PURGE=1; confirm; echo WENT ON' "$US" 2>&1)" +expect "a purge wants the word purge" "not confirmed" "$out" +out="$(CONFIRM_TTY="$root/no-tty/x" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)" +expect "with no terminal it asks for --yes" "no terminal to confirm on; re-run with --yes" "$out" + +if [ "$fails" -eq 0 ]; then + echo "ALL PASS" +else + echo "$fails FAILED" +fi +exit "$fails" diff --git a/docs/operations.md b/docs/operations.md new file mode 100644 index 0000000..f8530ee --- /dev/null +++ b/docs/operations.md @@ -0,0 +1,250 @@ +# Felis Operations Guide + +What a Felis host needs, how big it should be, how to take Felis off it again, and where +the disaster-recovery procedures live. Fault-finding is in +[troubleshooting.md](troubleshooting.md); this document refers to its sections as §N. + +Evidence tags follow troubleshooting.md: **[VM-VERIFIED]** was run on a real host, +**[GO-TESTED]** / **[SH-TESTED]** is covered by `go test` or the shell tests under +`deploy/`, **[CODE-ONLY]** is what the code does and has not been run end to end. + +## 1. Supported hosts + +`deploy/bootstrap.sh` provisions a single node. It needs systemd, root, and one of the +package managers below; everything else (Docker, k3s, PostgreSQL, the JRE, cloudflared) +it installs. + +| OS family | Package manager | Architectures | Status | +|---|---|---|---| +| CentOS Stream 9 (firewalld active, PostgreSQL 13) | dnf | aarch64 | **[VM-VERIFIED]** install, same-version rerun, upgrade, uninstall and reinstall | +| Ubuntu 24.04 LTS | apt | x86_64 | [CODE-ONLY] | +| RHEL / Rocky / Alma 9, Fedora | dnf | x86_64, aarch64 | [CODE-ONLY] same code path as CentOS Stream | +| Debian 12, other Ubuntu releases | apt | x86_64, aarch64 | [CODE-ONLY] | +| openSUSE Leap / Tumbleweed | zypper | x86_64, aarch64 | [CODE-ONLY] | +| Arch Linux | pacman | x86_64, aarch64 | [CODE-ONLY] | + +Pinned component versions (a fresh install gets exactly these; an installed k3s or +cloudflared is left as it is, see §4): + +| Component | Version | Where it is pinned | +|---|---|---| +| k3s | v1.36.4+k3s1 | `FELIS_K3S_VERSION` in `bootstrap.sh` | +| cloudflared | 2026.9.1 | `FELIS_CLOUDFLARED_VERSION`, sha256 per architecture | +| Temurin JRE (Velocity) | 25, patch build pinned | `FELIS_JRE_VERSION`, sha256 per architecture | +| Go (nano builds) | 1.26.8 | `GO_PINNED_VERSION`, sha256 per architecture | +| Minecraft / Limbo / Paper / Velocity / LuckPerms | `deploy/game-stack.lock` | §15b | +| PostgreSQL | the distribution's package | 13 and 18 are exercised by the `pgint` CI job | + +32-bit hosts are not supported: there is no k3s, JRE or Go build the installer will fetch +for them. + +## 2. Sizing + +### What the platform itself uses + +Measured on the verification host (4 vCPU, 5.5 GB RAM, 6 GB swap, CentOS Stream 9 +aarch64) with the control plane, the login and lobby system servers and one idle Paper +server running **[VM-VERIFIED]**: + +| Process | Resident memory | +|---|---| +| k3s (server, kubelet, containerd) | ~1.1 GB | +| Velocity (`-Xms512M -Xmx1G`, heap pre-touched) | ~0.73 GB | +| lobby (Paper, pod limit 1 GiB) | ~0.7–0.85 GB | +| login (Limbo, pod limit 512 MiB) | ~0.16 GB | +| felis-api, felis-operator, registry gate | ~50 MB each | +| PostgreSQL | ~30 MB plus page cache | +| **Total in use** | **~3.4 GB** | + +Every game server adds the memory its owner gave it: the pod's limit equals its request, +and the JVM heap is derived from it (§1a). Quotas cap it per user (panel → 管理 → 配额). + +The installer's own peak is the image builds (Docker plus a Gradle container); it stops +Docker afterwards so that memory goes back to the servers. On a host under 2 GB of RAM +without swap it adds a 2 GiB `/swapfile`. + +### Recommendations + +| Concurrent players | Game servers running | CPU | RAM | `FELIS_VELOCITY_XMX` | +|---|---|---|---|---| +| up to 20 | 1–2 small | 2 vCPU | 4 GB + 2 GB swap | 1G (default) | +| up to 100 | 3–5 | 4 vCPU | 8–16 GB | 1G | +| up to 300 | 5–10 | 8 vCPU | 16–32 GB | 2G | +| 300+ | more | 8+ vCPU | 32 GB+ | 3G–4G | + +The player-count rows are planning figures, not measurements: a Minecraft server's cost +depends mostly on what its players do (view distance, redstone, mods). Size RAM as the +platform's ~3.5 GB plus the sum of the servers you expect to run at once, then add a +quarter for the page cache and PostgreSQL. Velocity itself needs little per player; raise +its heap when `journalctl -u felis-velocity` shows long GC pauses or `OutOfMemoryError`. + +`FELIS_VELOCITY_XMX` (default `1G`, at least `256M`, written `M` or `G`) is read on +every installer run. The initial heap stays at 512M, or equals the maximum when that is +lower. Changing it rewrites the unit, and the rerun restarts the proxy, which disconnects +everyone online; do it in a quiet hour **[VM-VERIFIED]**: + +``` +curl -fsSL /deploy/bootstrap.sh | sudo FELIS_VELOCITY_XMX=2G bash +``` + +### Disk + +| What | Where | Size | +|---|---|---| +| Worlds | one volume per server under `/var/lib/rancher/k3s/storage` | what the world grows to | +| World archives | the `felis-backups` volume (`FELIS_BACKUP_STORAGE`, default 10Gi requested) | about one compressed world per backup kept | +| In-cluster registry | the `registry` volume (default 10Gi requested) | 2–3 GB for the stock images; grows with custom builds, pruned daily (§9) | +| k3s's containerd images | `/var/lib/rancher/k3s/agent/containerd` | 6–9 GB | +| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store) | 5–10 GB after repeated upgrades | +| Toolchains and sources | `/opt/felis` | ~2.5 GB | +| Database bundles | `/var/lib/felis/db-backups` | a few MB each, 14 daily kept | + +k3s's local-path volumes do not enforce the requested sizes (§9), so every volume shares +the root filesystem. Give the host at least **40 GB**, and 60 GB or more once worlds and +custom images accumulate. The watchdog mails the owners when a watched filesystem passes +its threshold, and §13b covers a full disk. `docker builder prune -af` (with Docker +started) reclaims the build cache when space is short; the next upgrade rebuilds it. + +## 3. Uninstall + +`deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove +and asks before it starts (`--yes` skips the question) **[SH-TESTED] +[VM-VERIFIED]**: + +``` +curl -fsSL /deploy/uninstall.sh | sudo bash -s -- --yes # keep the data +curl -fsSL /deploy/uninstall.sh | sudo bash -s -- --purge # remove the data too +``` + +With a private repository, fetch it the way the README fetches `bootstrap.sh`. + +Both modes remove the `felis-*` systemd units and `cloudflared-felis.service`, the +Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the installer's cloudflared binary +(unless another unit runs it), the `felis_postgres` and `felis_edge` nftables tables and +the firewalld ports the installer opened. k3s goes with k3s's own `k3s-uninstall.sh` when +the cluster holds nothing but Felis's namespaces; when it runs anything else only +`felis`, `minecraft`, `felis-build` and the MinecraftServer CRD are deleted. +`--keep-k3s` and `--remove-k3s` override that choice. + +| | keep data (default) | `--purge` | +|---|---|---| +| Final database bundle | taken first (`felis db backup -label manual`); a failure stops the uninstall before anything is removed. `--no-backup` skips it | none | +| `felis` database and role | kept | dropped; `listen_addresses` and `pg_hba.conf` go back to how they were | +| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file | +| `/var/lib/felis` (database bundles) | kept | deleted | +| Worlds, archives, registry, uploads | moved to `/var/lib/felis/retained/k3s-storage-/` (with `--keep-k3s`: their volumes switch to `Retain` and stay in place) | deleted | +| Felis images, Docker build cache | kept | deleted | + +Neither mode removes packages (Docker, PostgreSQL, git, nftables) or the swap file: other +software may use them. On a host that should end up bare: + +``` +sudo swapoff /swapfile && sudo rm /swapfile && sudo sed -i '\|^/swapfile |d' /etc/fstab +sudo dnf remove docker-ce docker-ce-cli containerd.io postgresql-server # or apt/zypper/pacman +``` + +The Cloudflare side outlives the host. After an uninstall that is final, delete the +tunnel (Zero Trust → Networks → Tunnels, or `cloudflared tunnel delete `), its +DNS records for the panel hostnames, and the Access application. + +### Reinstall on top of kept data + +A keep-data uninstall leaves everything a reinstall needs. The installer reuses +`/etc/felis/secrets.env`, so the database password and the forwarding and session +secrets are unchanged, and it migrates the kept database instead of creating one +**[VM-VERIFIED]**. + +Each step below was run on the reference VM after a keep-data uninstall, and the +restored worlds matched their kept `level.dat` checksums **[VM-VERIFIED]**. `kept` names +the directory the uninstall moved the volumes to: + +``` +kept="$(ls -d /var/lib/felis/retained/k3s-storage-* | tail -n 1)" +store=/var/lib/rancher/k3s/storage +``` + +1. Install as usual (`curl ... | sudo bash`). Name the same root domain if it was not + the `.nip.io` default: `felis.host.toml` is kept, and the installer reads the + domain from it. +2. Run `sudo felis setup`. It recreates the login and lobby servers; the Owner already + exists, so it opens on the status screen and you can quit there. +3. Put the image registry and the uploads back. They hold every custom server image + and uploaded file; without the registry, a restored server fails to pull its image. + + ``` + sudo k3s kubectl -n felis scale deploy/registry deploy/felis-api --replicas=0 + sudo k3s kubectl -n felis wait --for=delete pod -l app.kubernetes.io/component=registry --timeout=120s + sudo k3s kubectl -n felis wait --for=delete pod -l app.kubernetes.io/component=api --timeout=120s + sudo rsync -a --delete "$kept"/pvc-*_felis_registry/ "$(ls -d $store/pvc-*_felis_registry)"/ + sudo rsync -a --delete "$kept"/pvc-*_felis_felis-uploads/ "$(ls -d $store/pvc-*_felis_felis-uploads)"/ + sudo k3s kubectl -n felis scale deploy/registry deploy/felis-api --replicas=1 + ``` + + Then run the installer once more. It pushes this release's images over the older + copies the kept registry carried. +4. Bring the game servers back. The final bundle holds every MinecraftServer as it was; + the selector skips login and lobby, which step 2 created for this release: + + ``` + b="$(ls -t /var/lib/felis/db-backups/felis-db-*-manual.tar | head -n 1)" + tar -xOf "$b" k8s/minecraftservers.json \ + | sudo k3s kubectl apply -l '!felis.lolicon.best/system-role' -f - + ``` + +5. Put each world back. A server's volume exists once it has started once, so start it + from the panel, stop it again, and copy the kept world over the new one: + + ``` + s= + sudo rsync -a --delete "$kept"/pvc-*_minecraft_world-$s-0/ "$(ls -d $store/pvc-*_minecraft_world-$s-0)"/ + ``` + + Then start it. The lobby works the same way: stop it with + `sudo k3s kubectl -n minecraft patch minecraftserver lobby --type=merge -p '{"spec":{"desiredState":"Stopped"}}'`, + copy `world-lobby-0`, and patch it back to `Running`. +6. Bring the archives back so the panel's restore points work again. The archive volume + appears with the first backup, so back up any server from the panel first, then: + + ``` + sudo rsync -a "$kept"/pvc-*_minecraft_felis-backups/ "$(ls -d $store/pvc-*_minecraft_felis-backups)"/ + ``` + + With an off-site bucket configured, `sudo felis offsite fetch-worlds` fetches them + instead (troubleshooting §16). +7. Delete `/var/lib/felis/retained/` once every server is back. + +## 4. Upgrading the pieces around Felis + +A rerun of the installer upgrades Felis itself (§15). The components it installs keep +the version they were installed with unless noted: + +| Component | How a rerun treats it | Upgrade | +|---|---|---| +| Velocity, Limbo, Paper, LuckPerms | follow `deploy/game-stack.lock` | rerun after a release that moves the lock (§15b) | +| Temurin JRE | moves to the pinned patch build | rerun | +| k3s | left alone | by hand, one minor version at a time: `curl -sfL https://get.k3s.io \| INSTALL_K3S_VERSION= sh -` | +| cloudflared | left alone | replace `/usr/local/bin/cloudflared` with the release binary, then `systemctl restart cloudflared-felis` | +| PostgreSQL | the distribution's package | the package manager; a major version needs `pg_upgrade` first (the installer refuses to start a newer server on an older cluster) | +| Docker, git, nftables | distribution packages | the package manager | + +`sudo felis update` reports Felis, Velocity, k3s and cloudflared against their newest +releases. + +## 5. Disaster recovery + +The procedures are in §16: what a database bundle holds, restoring one on the same host, +rolling back an upgrade, and rebuilding on a new host from the off-site copy. For a +production install: + +- **Configure the off-site copy** (`FELIS_OFFSITE_*`, §16 "Keep a copy somewhere + else"). Without it the world archives sit on the same disk as the worlds, and the + database bundles on the same disk as the database; losing the disk loses both. The + installer ends with `NO OFF-SITE COPY` until it is set. +- **Keep the off-site encryption key off the host**, in a password manager. The bucket + holds only sealed objects. +- **Keep one database bundle off the host** as well when there is no bucket. It contains + `secrets.env`, which a rebuild needs to read the rest. +- **Rehearse the rebuild** once on a spare VM: §16 "Rebuild on a new host", steps 1–5, + then log in and restore one world. `felis offsite status` and `felis db check` exit + non-zero when the copy or the newest bundle is stale; wire them into your monitoring, + or rely on the watchdog's mail.