feat(deploy): 新增 uninstall.sh(默认保留数据,--purge 全删)与 docs/operations.md 支持矩阵、容量与重装恢复手册

This commit is contained in:
Lemon-miaow committed 2026-09-25 01:20:52 +08:00
1 parent 989be55b4a
commit 6f7b8d7b30
4 files changed
+893

No files matched your search

+1
View File
@@ -135,6 +135,7 @@ jobs:
./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh') ./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh')
- run: sh deploy/bootstrap_test.sh - run: sh deploy/bootstrap_test.sh
- run: sh deploy/uninstall_test.sh
panel: panel:
runs-on: ubuntu-latest runs-on: ubuntu-latest
+428
View File
@@ -0,0 +1,428 @@
#!/usr/bin/env bash
# Removes what deploy/bootstrap.sh installed on this host.
#
# sudo bash deploy/uninstall.sh # remove Felis, keep the data
# sudo bash deploy/uninstall.sh --purge # remove the data too
# curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes
#
# Options:
# --purge also drop the felis database and role, and delete /etc/felis and
# /var/lib/felis (the database bundles, and anything an earlier keep-data
# run set aside). Asks for the word "purge" unless --yes is given.
# --keep-k3s leave k3s installed and remove only Felis's namespaces and CRD.
# --remove-k3s run k3s's own uninstaller even when other workloads live in the cluster.
# --no-backup skip the final database bundle keep-data mode takes first.
# --yes do not ask.
#
# Keep-data mode (the default) first takes a database bundle (`felis db backup -label
# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the
# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive,
# registry and upload volumes live under k3s's storage directory, which k3s's uninstaller
# deletes, so they are moved to /var/lib/felis/retained/k3s-storage-<UTC stamp> first; with
# --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go.
# docs/operations.md walks through reinstalling on top of what is left.
#
# Either mode leaves packages alone (Docker, PostgreSQL, git and the rest), and the swap
# file a low-memory host got: other software may use them. docs/operations.md lists the
# package commands for a bare host.
set -Eeuo pipefail
STATE_DIR="${STATE_DIR:-/etc/felis}"
DATA_DIR="${DATA_DIR:-/var/lib/felis}"
RETAIN_DIR="${DATA_DIR}/retained"
HOST_BIN="${HOST_BIN:-/usr/local/bin/felis}"
OPT_DIR="${OPT_DIR:-/opt/felis}"
UNIT_DIR="${UNIT_DIR:-/etc/systemd/system}"
K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}"
K3S_STORAGE="${K3S_STORAGE:-/var/lib/rancher/k3s/storage}"
K3S_REGISTRIES="${K3S_REGISTRIES:-/etc/rancher/k3s/registries.yaml}"
export KUBECONFIG="${KUBECONFIG:-/etc/rancher/k3s/k3s.yaml}"
CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}"
VELOCITY_USER="felis-velocity"
DB_NAME="felis"
DB_USER="felis"
POD_CIDR="10.42.0.0/16"
SERVICE_CIDR="10.43.0.0/16"
FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}"
CONFIRM_TTY="${CONFIRM_TTY:-/dev/tty}"
FELIS_NAMESPACES=(felis minecraft felis-build)
FELIS_CRD="minecraftservers.felis.lolicon.best"
# Every unit the installer and `felis setup` write. Timers first, so none fires into a
# service that is already gone.
FELIS_UNITS=(
felis-db-backup.timer felis-watchdog.timer felis-offsite.timer felis-build-tools.timer
felis-db-backup.service felis-watchdog.service felis-offsite.service felis-build-tools.service
felis-velocity.service felis-nano.service cloudflared-felis.service
felis-postgres-firewall.service
)
PURGE=0
K3S_MODE=auto
BACKUP=1
ASSUME_YES=0
TUNNEL_CONFIG=""
log() { printf '\033[1;36m[felis]\033[0m %s\n' "$*"; }
ok() { printf '\033[1;32m[ ok ]\033[0m %s\n' "$*"; }
warn() { printf '\033[1;33m[warn]\033[0m %s\n' "$*" >&2; }
die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; }
parse_args() {
while [ $# -gt 0 ]; do
case "$1" in
--purge) PURGE=1 ;;
--keep-k3s) K3S_MODE=keep ;;
--remove-k3s) K3S_MODE=remove ;;
--no-backup) BACKUP=0 ;;
--yes|-y) ASSUME_YES=1 ;;
-h|--help) printf 'usage: uninstall.sh [--purge] [--keep-k3s|--remove-k3s] [--no-backup] [--yes]\n'; exit 0 ;;
*) die "unknown option: $1 (see --help)" ;;
esac
shift
done
}
kube() { "${K3S_BIN_DIR}/k3s" kubectl "$@"; }
k3s_present() { [ -x "${K3S_BIN_DIR}/k3s" ]; }
# foreign_namespaces prints the namespaces that are neither k3s's own nor Felis's, one per
# line. A cluster with none of them exists for Felis alone, and removing k3s takes nothing
# else with it.
foreign_namespaces() {
kube get namespaces -o 'jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}' \
| awk '$0 != "" && $0 != "default" && $0 !~ /^kube-/ && $0 != "felis" && $0 != "minecraft" && $0 != "felis-build"'
}
# decide_k3s turns K3S_MODE=auto into keep or remove.
decide_k3s() {
k3s_present || { K3S_MODE=absent; return 0; }
[ "$K3S_MODE" = auto ] || return 0
local others
if ! others="$(foreign_namespaces)"; then
die "k3s does not answer, so this cannot tell whether it runs anything besides Felis; start it (systemctl start k3s) or pass --keep-k3s or --remove-k3s"
fi
if [ -z "$others" ]; then
K3S_MODE=remove
else
K3S_MODE=keep
log "k3s also runs namespaces Felis did not create ($(printf '%s' "$others" | tr '\n' ' ')); leaving k3s installed"
fi
}
confirm() {
[ "$ASSUME_YES" = 1 ] && return 0
local want="yes" answer=""
[ "$PURGE" = 1 ] && want="purge"
# A piped script has no stdin to read from; the terminal is asked directly.
{ exec 3<"$CONFIRM_TTY" 4>>"$CONFIRM_TTY"; } 2>/dev/null || die "no terminal to confirm on; re-run with --yes"
printf 'Type "%s" to continue: ' "$want" >&4
read -r answer <&3 || true
exec 3<&- 4>&-
[ "$answer" = "$want" ] || die "not confirmed; nothing was changed"
}
print_plan() {
log "this will remove from $(uname -n):"
log " the felis-* systemd units, cloudflared-felis.service, the ${VELOCITY_USER} user,"
log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld openings"
case "$K3S_MODE" in
remove) log " k3s, with everything in it (${K3S_BIN_DIR}/k3s-uninstall.sh)" ;;
keep) log " Felis's namespaces (${FELIS_NAMESPACES[*]}) and the ${FELIS_CRD} CRD; k3s stays" ;;
absent) ;;
esac
if [ "$PURGE" = 1 ]; then
log " PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles"
log " and anything set aside before), every world and archive, the Felis images and Docker's build cache"
else
[ "$BACKUP" = 1 ] && log " after a final database bundle into ${DATA_DIR}/db-backups"
log " kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/"
fi
}
final_backup() {
[ "$PURGE" = 0 ] && [ "$BACKUP" = 1 ] || return 0
[ -x "$HOST_BIN" ] && [ -r "${STATE_DIR}/felis.host.toml" ] || {
warn "no ${HOST_BIN} or ${STATE_DIR}/felis.host.toml; skipping the final database bundle"
return 0
}
log "taking a final database bundle"
"$HOST_BIN" db backup -config "${STATE_DIR}/felis.host.toml" -label manual \
|| die "the final database bundle failed, so nothing was removed. Fix the database (sudo felis db check), or pass --no-backup to go on without one"
ok "database bundle written to ${DATA_DIR}/db-backups"
}
# game_port reads the proxy's port from velocity.toml before /opt/felis goes.
game_port() {
local toml="${OPT_DIR}/velocity/velocity.toml" port=""
[ -r "$toml" ] && port="$(sed -n 's/^bind *= *"[^"]*:\([0-9][0-9]*\)".*/\1/p' "$toml" | head -n 1)"
printf '%s\n' "${FELIS_GAME_PORT:-${port:-25565}}"
}
# tunnel_config reads the cloudflared config felis setup pointed its unit at (by default
# /etc/felis/cloudflared.yml) before the unit goes.
tunnel_config() {
local unit="${UNIT_DIR}/cloudflared-felis.service" path=""
[ -r "$unit" ] && path="$(sed -n 's/^ExecStart=.* --config \([^ ]*\) tunnel run$/\1/p' "$unit" | head -n 1)"
printf '%s\n' "${path:-${STATE_DIR}/cloudflared.yml}"
}
# nano_port reads felis-nano's port from its unit before the unit goes.
nano_port() {
local unit="${UNIT_DIR}/felis-nano.service"
[ -r "$unit" ] || return 0
sed -n 's/^ExecStart=.* -listen [^ ]*:\([0-9][0-9]*\).*$/\1/p' "$unit" | head -n 1
}
remove_units() {
local unit removed=0
for unit in "${FELIS_UNITS[@]}"; do
[ -f "${UNIT_DIR}/${unit}" ] || continue
systemctl disable --now "$unit" >/dev/null 2>&1 || systemctl stop "$unit" >/dev/null 2>&1 || true
rm -f "${UNIT_DIR}/${unit}"
removed=$((removed + 1))
done
systemctl daemon-reload
systemctl reset-failed >/dev/null 2>&1 || true
ok "${removed} systemd unit(s) removed"
}
remove_nft_tables() {
command -v nft >/dev/null 2>&1 || return 0
local t
for t in felis_postgres felis_edge; do
if nft list table inet "$t" >/dev/null 2>&1; then
nft delete table inet "$t"
ok "nftables table inet ${t} removed"
fi
done
}
# remove_firewalld_rules takes back what configure_k3s_firewall, configure_velocity_firewall
# and the nano setup opened. The k3s ones stay when k3s does.
remove_firewalld_rules() { # game-port nano-port
command -v firewall-cmd >/dev/null 2>&1 || return 0
systemctl is-active --quiet firewalld || return 0
local game="$1" nano="$2" port rule changed=0
local ports=("${game}/tcp" "${FELIS_PANEL_NODEPORT}/tcp")
[ -n "$nano" ] && ports+=("${nano}/tcp")
[ "$K3S_MODE" = remove ] && ports+=("6443/tcp")
for port in "${ports[@]}"; do
if firewall-cmd --permanent --query-port="$port" >/dev/null 2>&1; then
firewall-cmd --permanent --remove-port="$port" >/dev/null
changed=1
fi
done
if [ -n "$nano" ]; then
while IFS= read -r rule; do
case "$rule" in
*"port=\"${nano}\""*) firewall-cmd --permanent --remove-rich-rule="$rule" >/dev/null; changed=1 ;;
esac
done < <(firewall-cmd --permanent --list-rich-rules 2>/dev/null)
fi
if [ "$K3S_MODE" = remove ]; then
local cidr
for cidr in "$POD_CIDR" "$SERVICE_CIDR"; do
if firewall-cmd --permanent --zone=trusted --query-source="$cidr" >/dev/null 2>&1; then
firewall-cmd --permanent --zone=trusted --remove-source="$cidr" >/dev/null
changed=1
fi
done
fi
if [ "$changed" = 1 ]; then
firewall-cmd --reload >/dev/null
ok "firewalld openings removed"
fi
}
# retain_volumes_in_cluster keeps every volume Felis's claims are bound to when the
# namespaces go: local-path deletes a Delete-policy volume's directory with its claim.
retain_volumes_in_cluster() {
local pv
while IFS= read -r pv; do
[ -n "$pv" ] || continue
kube patch pv "$pv" -p '{"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' >/dev/null
done < <(kube get pv -o 'jsonpath={range .items[*]}{.metadata.name} {.spec.claimRef.namespace}{"\n"}{end}' \
| awk '$2 == "felis" || $2 == "minecraft" || $2 == "felis-build" { print $1 }')
ok "Felis's volumes set to Retain; their directories stay under ${K3S_STORAGE}"
}
remove_from_cluster() {
[ "$PURGE" = 1 ] || retain_volumes_in_cluster
log "deleting Felis's namespaces and CRD"
# The operator is part of what goes, so nothing would clear a MinecraftServer finalizer
# and the minecraft namespace would stay Terminating. Drop them first.
local s
while IFS= read -r s; do
[ -n "$s" ] || continue
kube -n minecraft patch minecraftserver "$s" --type=merge -p '{"metadata":{"finalizers":null}}' >/dev/null 2>&1 || true
done < <(kube -n minecraft get minecraftservers -o 'jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}' 2>/dev/null || true)
kube delete namespace "${FELIS_NAMESPACES[@]}" --ignore-not-found --wait=true --timeout=300s >/dev/null \
|| warn "a namespace is still terminating; check: k3s kubectl get namespaces"
kube delete crd "$FELIS_CRD" --ignore-not-found >/dev/null || true
if [ "$PURGE" = 1 ]; then
local pv
while IFS= read -r pv; do
[ -n "$pv" ] && kube delete pv "$pv" --ignore-not-found >/dev/null
done < <(kube get pv -o 'jsonpath={range .items[*]}{.metadata.name} {.spec.claimRef.namespace}{"\n"}{end}' \
| awk '$2 == "felis" || $2 == "minecraft" || $2 == "felis-build" { print $1 }')
fi
if [ -f "$K3S_REGISTRIES" ] && grep -q 'registry\.felis\.svc' "$K3S_REGISTRIES"; then
rm -f "$K3S_REGISTRIES"
systemctl restart k3s
fi
ok "Felis removed from the cluster; k3s stays"
}
remove_k3s() {
local stamp
if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then
# k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a
# world while it moves.
"${K3S_BIN_DIR}/k3s-killall.sh" >/dev/null 2>&1 || systemctl stop k3s
stamp="$(date -u +%Y%m%dT%H%M%SZ)"
install -d -m 0700 "$RETAIN_DIR"
mv "$K3S_STORAGE" "${RETAIN_DIR}/k3s-storage-${stamp}"
ok "volumes moved to ${RETAIN_DIR}/k3s-storage-${stamp}"
fi
if [ -x "${K3S_BIN_DIR}/k3s-uninstall.sh" ]; then
log "running k3s-uninstall.sh"
"${K3S_BIN_DIR}/k3s-uninstall.sh" >/dev/null 2>&1 || warn "k3s-uninstall.sh reported an error; check /var/lib/rancher and /etc/rancher"
ok "k3s removed"
else
warn "k3s is at ${K3S_BIN_DIR}/k3s but ${K3S_BIN_DIR}/k3s-uninstall.sh is missing; remove k3s by hand"
fi
}
# remove_cloudflared_binary deletes the binary the installer put in /usr/local/bin, unless
# a unit other than Felis's still runs it.
remove_cloudflared_binary() {
[ -x "$CLOUDFLARED_BIN" ] || return 0
local others
others="$(grep -ls "$CLOUDFLARED_BIN" "${UNIT_DIR}"/*.service /lib/systemd/system/*.service /usr/lib/systemd/system/*.service 2>/dev/null || true)"
if [ -n "$others" ]; then
log "leaving ${CLOUDFLARED_BIN}: $(printf '%s' "$others" | tr '\n' ' ')uses it"
return 0
fi
rm -f "$CLOUDFLARED_BIN"
ok "${CLOUDFLARED_BIN} removed"
}
remove_host_files() {
if id "$VELOCITY_USER" >/dev/null 2>&1; then
userdel "$VELOCITY_USER" >/dev/null 2>&1 || warn "could not remove the ${VELOCITY_USER} user"
fi
rm -rf "$OPT_DIR"
rm -f "$HOST_BIN" "${HOST_BIN}.new"
remove_cloudflared_binary
if [ "$PURGE" = 0 ]; then
# What describes the removed install goes; what a reinstall reuses stays. Without
# bootstrap.done the next run takes the first-install path.
rm -f "${STATE_DIR}/bootstrap.done" "${STATE_DIR}/system-server-images" \
"${STATE_DIR}/velocity.fingerprint" "${STATE_DIR}/previous-felis-image"
fi
ok "${OPT_DIR} and ${HOST_BIN} removed"
}
as_postgres() { (cd / && runuser -u postgres -- "$@"); }
# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else.
remove_hba_block() { # file
local tmp
tmp="$(mktemp)"
awk '
$0 == "# BEGIN FELIS MANAGED HBA" { skip = 1; next }
$0 == "# END FELIS MANAGED HBA" { skip = 0; blank = 1; next }
blank && $0 == "" { blank = 0; next }
{ blank = 0 }
!skip { print }
' "$1" > "$tmp"
cat "$tmp" > "$1"
rm -f "$tmp"
}
purge_database() {
[ "$PURGE" = 1 ] || return 0
if ! systemctl is-active --quiet postgresql 2>/dev/null; then
warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it"
return 0
fi
local hba
hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = '${DB_NAME}' AND pid <> pg_backend_pid();
DROP DATABASE IF EXISTS ${DB_NAME};
DROP ROLE IF EXISTS ${DB_USER};
ALTER SYSTEM RESET listen_addresses;
SQL
[ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba"
systemctl restart postgresql
ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again"
}
purge_images() {
[ "$PURGE" = 1 ] || return 0
command -v docker >/dev/null 2>&1 || return 0
# The installer stops Docker after its builds; start it just long enough to clean up.
local was_active=1 refs
systemctl is-active --quiet docker || { was_active=0; systemctl start docker >/dev/null 2>&1 || return 0; }
refs="$(docker image ls --format '{{.Repository}}:{{.Tag}}' | grep -E '^(registry\.felis\.svc:5000/felis/|felis/)' || true)"
if [ -n "$refs" ]; then
# shellcheck disable=SC2086 # one ref per word
docker image rm -f $refs >/dev/null 2>&1 || true
fi
docker builder prune -af >/dev/null 2>&1 || true
[ "$was_active" = 1 ] || systemctl stop docker docker.socket >/dev/null 2>&1 || true
ok "Felis images and Docker's build cache removed"
}
purge_state() {
[ "$PURGE" = 1 ] || return 0
# felis setup's tunnel credentials sit beside cloudflared's login (cert.pem, which stays:
# it is the Cloudflare account's, not Felis's). The tunnel itself lives on in the account
# until it is deleted there (docs/operations.md).
local cred=""
[ -r "$TUNNEL_CONFIG" ] \
&& cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)"
if [ -n "$cred" ]; then rm -f "$cred"; fi
rm -f "$TUNNEL_CONFIG"
rm -rf "$STATE_DIR" "$DATA_DIR"
ok "${STATE_DIR} and ${DATA_DIR} removed"
}
main() {
parse_args "$@"
[ "$(id -u)" = 0 ] || die "run as root: sudo bash $0"
[ -e "$STATE_DIR" ] || [ -e "$HOST_BIN" ] || [ -e "$OPT_DIR" ] \
|| die "no Felis install here (${STATE_DIR}, ${HOST_BIN} and ${OPT_DIR} are all absent)"
decide_k3s
print_plan
confirm
final_backup
local game nano
game="$(game_port)"
nano="$(nano_port)"
TUNNEL_CONFIG="$(tunnel_config)"
remove_units
case "$K3S_MODE" in
remove) remove_k3s ;;
keep) remove_from_cluster ;;
esac
remove_nft_tables
remove_firewalld_rules "$game" "$nano"
remove_host_files
purge_database
purge_images
purge_state
if [ "$PURGE" = 1 ]; then
ok "Felis is gone from this host"
else
ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}"
log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\""
fi
}
if [ "${FELIS_UNINSTALL_SOURCED:-0}" != 1 ]; then
main "$@"
fi
+214
View File
@@ -0,0 +1,214 @@
#!/bin/sh
# Checks for deploy/uninstall.sh. Run it as: sh deploy/uninstall_test.sh
#
# The script is sourced with FELIS_UNINSTALL_SOURCED=1, every host path pointed into a
# scratch directory, and the commands that would change the host (systemctl, k3s, nft,
# firewall-cmd, runuser, docker, userdel) replaced by stubs that log their arguments.
set -u
US="${1:-$(dirname "$0")/uninstall.sh}"
[ -f "$US" ] || { echo "no such script: $US"; exit 1; }
fails=0
expect() { # label needle haystack
case "$3" in
*"$2"*) echo "PASS $1" ;;
*) echo "FAIL $1: expected <$2> in:"; echo "$3"; fails=$((fails + 1)) ;;
esac
}
refute() { # label needle haystack
case "$3" in
*"$2"*) echo "FAIL $1: did not expect <$2> in:"; echo "$3"; fails=$((fails + 1)) ;;
*) echo "PASS $1" ;;
esac
}
root="$(mktemp -d)"
trap 'rm -rf "$root"' EXIT
# fresh_host lays out what an install leaves: units, state, /opt/felis, the host binary, a
# k3s with its uninstaller and a volume, a tunnel config and its credentials.
fresh_host() {
rm -rf "$root/h"
mkdir -p "$root/h/units" "$root/h/etc" "$root/h/data/db-backups" "$root/h/opt/velocity" \
"$root/h/bin" "$root/h/storage/pvc-1_minecraft_world-a-0" "$root/h/cf"
for u in felis-db-backup.timer felis-db-backup.service felis-velocity.service felis-postgres-firewall.service; do
printf '[Unit]\n' > "$root/h/units/$u"
done
printf '[Service]\nExecStart=/usr/local/bin/cloudflared --config %s tunnel run\n' "$root/h/etc/cloudflared.yml" \
> "$root/h/units/cloudflared-felis.service"
printf 'tunnel: abc\ncredentials-file: %s\n' "$root/h/cf/abc.json" > "$root/h/etc/cloudflared.yml"
printf '{}\n' > "$root/h/cf/abc.json"
printf 'bind = "0.0.0.0:25577"\n' > "$root/h/opt/velocity/velocity.toml"
for f in secrets.env felis.host.toml bootstrap.done system-server-images velocity.fingerprint; do
printf 'x\n' > "$root/h/etc/$f"
done
printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat"
for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do
printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b"
chmod +x "$root/h/bin/$b"
done
cat > "$root/h/hba.conf" <<'EOF'
# BEGIN FELIS MANAGED HBA
# Felis rules must precede distro defaults such as 127.0.0.1 ident.
host felis felis 127.0.0.1/32 scram-sha-256
# END FELIS MANAGED HBA
local all all peer
host all all 127.0.0.1/32 ident
EOF
: > "$root/calls"
}
# run_uninstall <namespaces> <args...>: runs main with the stubs; <namespaces> is what
# `kubectl get namespaces` answers, or "down" for a k3s that does not answer.
run_uninstall() {
ns="$1"; shift
NS="$ns" ROOT="$root" STATE_DIR="$root/h/etc" DATA_DIR="$root/h/data" HOST_BIN="$root/h/bin/felis" \
OPT_DIR="$root/h/opt" UNIT_DIR="$root/h/units" K3S_BIN_DIR="$root/h/bin" \
K3S_STORAGE="$root/h/storage" K3S_REGISTRIES="$root/h/registries.yaml" \
CLOUDFLARED_BIN="$root/h/no-cloudflared" FELIS_UNINSTALL_SOURCED=1 bash -c '
set -Eeuo pipefail
. "$0"
calls="$ROOT/calls"
id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; }
systemctl() {
echo "SYSTEMCTL $*" >> "$calls"
case "$*" in "is-active --quiet firewalld") return 1 ;; esac
return 0
}
kube() {
echo "KUBE $*" >> "$calls"
case "$*" in
"get namespaces"*) [ "$NS" = down ] && return 1; printf "%s\n" $NS ;;
"get pv"*) printf "pvc-1 minecraft\npvc-9 other\n" ;;
esac
}
nft() { echo "NFT $*" >> "$calls"; return 1; }
runuser() {
shift 3
case "$*" in
*"SHOW hba_file"*) echo "$ROOT/h/hba.conf" ;;
*) echo "PSQL $* $(cat)" >> "$calls" ;;
esac
}
docker() { echo "DOCKER $*" >> "$calls"; }
userdel() { echo "USERDEL $*" >> "$calls"; }
uname() { echo testhost; }
main "$@"' "$US" "$@" 2>&1
}
# --- keep-data, a cluster that runs only Felis -------------------------------------------
fresh_host
out="$(run_uninstall "default kube-system felis minecraft felis-build" --yes)"
calls="$(cat "$root/calls")"
expect "keep-data takes a final bundle first" "RUN felis db backup -config $root/h/etc/felis.host.toml -label manual" "$calls"
expect "a Felis-only cluster is removed with k3s's uninstaller" "RUN k3s-uninstall.sh" "$calls"
expect "the pods are stopped before the volumes move" "RUN k3s-killall.sh" "$calls"
kept="$(ls "$root/h/data/retained" 2>/dev/null)"
expect "the volumes are set aside before k3s deletes them" "k3s-storage-" "$kept"
[ -f "$root/h/data/retained/$kept/pvc-1_minecraft_world-a-0/level.dat" ] \
&& echo "PASS a world survives the uninstall" \
|| { echo "FAIL the world did not survive: $(ls -R "$root/h/data")"; fails=$((fails + 1)); }
[ -f "$root/h/etc/secrets.env" ] && [ -f "$root/h/etc/felis.host.toml" ] \
&& echo "PASS the secrets and felis.toml stay" \
|| { echo "FAIL keep-data removed the secrets"; fails=$((fails + 1)); }
[ ! -e "$root/h/etc/bootstrap.done" ] && [ ! -e "$root/h/etc/velocity.fingerprint" ] \
&& echo "PASS the markers of the removed install go, so a reinstall starts fresh" \
|| { echo "FAIL bootstrap.done or the proxy fingerprint was left"; fails=$((fails + 1)); }
refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
[ ! -e "$root/h/opt" ] && [ ! -e "$root/h/bin/felis" ] \
&& echo "PASS /opt/felis and the host binary are removed" \
|| { echo "FAIL /opt/felis or the host binary is still there"; fails=$((fails + 1)); }
[ -z "$(ls "$root/h/units")" ] && echo "PASS every Felis unit file is removed" \
|| { echo "FAIL units left: $(ls "$root/h/units")"; fails=$((fails + 1)); }
expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls"
expect "the velocity user is removed" "USERDEL felis-velocity" "$calls"
expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out"
# --- a failed final bundle stops everything ------------------------------------------------
fresh_host
printf '#!/bin/sh\necho "RUN felis $*" >> "%s"\nexit 1\n' "$root/calls" > "$root/h/bin/felis"
out="$(run_uninstall "default felis minecraft" --yes)"
expect "a failed bundle is fatal" "the final database bundle failed, so nothing was removed" "$out"
[ -d "$root/h/opt" ] && [ -f "$root/h/units/felis-velocity.service" ] \
&& echo "PASS nothing is removed when the bundle fails" \
|| { echo "FAIL the uninstall went on after the bundle failed"; fails=$((fails + 1)); }
run_uninstall "default felis minecraft" --yes --no-backup >/dev/null
[ ! -d "$root/h/opt" ] && echo "PASS --no-backup goes on without one" \
|| { echo "FAIL --no-backup did not remove anything"; fails=$((fails + 1)); }
# --- a shared cluster --------------------------------------------------------------------
fresh_host
out="$(run_uninstall "default kube-system felis minecraft felis-build shop" --yes)"
calls="$(cat "$root/calls")"
refute "a cluster that runs something else keeps k3s" "RUN k3s-uninstall.sh" "$calls"
expect "the reason names the other namespace" "shop" "$out"
expect "Felis's volumes are retained before their claims go" 'KUBE patch pv pvc-1 -p {"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' "$calls"
refute "a volume of another namespace is not touched" "patch pv pvc-9" "$calls"
expect "Felis's namespaces are deleted" "KUBE delete namespace felis minecraft felis-build" "$calls"
expect "the CRD is deleted" "KUBE delete crd minecraftservers.felis.lolicon.best" "$calls"
out="$(fresh_host; run_uninstall down --yes)"
expect "a k3s that does not answer stops the run" "k3s does not answer" "$out"
fresh_host
run_uninstall down --yes --keep-k3s >/dev/null
calls="$(cat "$root/calls")"
refute "--keep-k3s never runs k3s's uninstaller" "RUN k3s-uninstall.sh" "$calls"
# --- purge -------------------------------------------------------------------------------
fresh_host
out="$(run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
refute "purge takes no bundle" "db backup" "$calls"
expect "purge drops the database" "DROP DATABASE IF EXISTS felis;" "$calls"
expect "purge drops the role" "DROP ROLE IF EXISTS felis;" "$calls"
expect "purge puts listen_addresses back" "ALTER SYSTEM RESET listen_addresses;" "$calls"
[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS purge removes /etc/felis and /var/lib/felis" \
|| { echo "FAIL purge left state behind"; fails=$((fails + 1)); }
[ ! -e "$root/h/cf/abc.json" ] && echo "PASS purge removes the tunnel's credentials file" \
|| { echo "FAIL the tunnel credentials are still there"; fails=$((fails + 1)); }
[ ! -e "$root/h/data/retained" ] && echo "PASS purge does not set the volumes aside" \
|| { echo "FAIL purge kept the volumes"; fails=$((fails + 1)); }
hba="$(cat "$root/h/hba.conf")"
refute "the Felis block leaves pg_hba.conf" "FELIS MANAGED" "$hba"
expect "the distro's own rules stay" "host all all 127.0.0.1/32 ident" "$hba"
case "$hba" in
"local all all peer"*) echo "PASS no blank line is left where the block was" ;;
*) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;;
esac
expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls"
# --- the pieces read before they are removed ---------------------------------------------
fresh_host
lib() { STATE_DIR="$root/h/etc" OPT_DIR="$root/h/opt" UNIT_DIR="$root/h/units" FELIS_UNINSTALL_SOURCED=1 \
bash -c '. "$0"; '"$1" "$US"; }
expect "the game port comes from velocity.toml" "25577" "$(lib game_port)"
expect "FELIS_GAME_PORT overrides it" "25599" "$(FELIS_GAME_PORT=25599 lib game_port)"
rm "$root/h/opt/velocity/velocity.toml"
expect "without velocity.toml the port is the default" "25565" "$(lib game_port)"
printf '[Service]\nExecStart=/usr/local/bin/felis nano -listen 10.0.0.5:8082 -config x\n' > "$root/h/units/felis-nano.service"
expect "the nano port comes from its unit" "8082" "$(lib nano_port)"
expect "the tunnel config comes from the cloudflared unit" "$root/h/etc/cloudflared.yml" "$(lib tunnel_config)"
rm "$root/h/units/cloudflared-felis.service"
expect "without the unit the tunnel config is the default path" "$root/h/etc/cloudflared.yml" "$(lib tunnel_config)"
out="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; parse_args --bogus' "$US" 2>&1)"
expect "an unknown option is refused" "unknown option: --bogus" "$out"
printf 'nope\n' > "$root/tty"
out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)"
expect "any answer but the word stops it" "not confirmed; nothing was changed" "$out"
printf 'yes\n' > "$root/tty"
out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)"
expect "yes goes on" "WENT ON" "$out"
out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; PURGE=1; confirm; echo WENT ON' "$US" 2>&1)"
expect "a purge wants the word purge" "not confirmed" "$out"
out="$(CONFIRM_TTY="$root/no-tty/x" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)"
expect "with no terminal it asks for --yes" "no terminal to confirm on; re-run with --yes" "$out"
if [ "$fails" -eq 0 ]; then
echo "ALL PASS"
else
echo "$fails FAILED"
fi
exit "$fails"
+250
View File
@@ -0,0 +1,250 @@
# Felis Operations Guide
What a Felis host needs, how big it should be, how to take Felis off it again, and where
the disaster-recovery procedures live. Fault-finding is in
[troubleshooting.md](troubleshooting.md); this document refers to its sections as §N.
Evidence tags follow troubleshooting.md: **[VM-VERIFIED]** was run on a real host,
**[GO-TESTED]** / **[SH-TESTED]** is covered by `go test` or the shell tests under
`deploy/`, **[CODE-ONLY]** is what the code does and has not been run end to end.
## 1. Supported hosts
`deploy/bootstrap.sh` provisions a single node. It needs systemd, root, and one of the
package managers below; everything else (Docker, k3s, PostgreSQL, the JRE, cloudflared)
it installs.
| OS family | Package manager | Architectures | Status |
|---|---|---|---|
| CentOS Stream 9 (firewalld active, PostgreSQL 13) | dnf | aarch64 | **[VM-VERIFIED]** install, same-version rerun, upgrade, uninstall and reinstall |
| Ubuntu 24.04 LTS | apt | x86_64 | [CODE-ONLY] |
| RHEL / Rocky / Alma 9, Fedora | dnf | x86_64, aarch64 | [CODE-ONLY] same code path as CentOS Stream |
| Debian 12, other Ubuntu releases | apt | x86_64, aarch64 | [CODE-ONLY] |
| openSUSE Leap / Tumbleweed | zypper | x86_64, aarch64 | [CODE-ONLY] |
| Arch Linux | pacman | x86_64, aarch64 | [CODE-ONLY] |
Pinned component versions (a fresh install gets exactly these; an installed k3s or
cloudflared is left as it is, see §4):
| Component | Version | Where it is pinned |
|---|---|---|
| k3s | v1.36.4+k3s1 | `FELIS_K3S_VERSION` in `bootstrap.sh` |
| cloudflared | 2026.9.1 | `FELIS_CLOUDFLARED_VERSION`, sha256 per architecture |
| Temurin JRE (Velocity) | 25, patch build pinned | `FELIS_JRE_VERSION`, sha256 per architecture |
| Go (nano builds) | 1.26.8 | `GO_PINNED_VERSION`, sha256 per architecture |
| Minecraft / Limbo / Paper / Velocity / LuckPerms | `deploy/game-stack.lock` | §15b |
| PostgreSQL | the distribution's package | 13 and 18 are exercised by the `pgint` CI job |
32-bit hosts are not supported: there is no k3s, JRE or Go build the installer will fetch
for them.
## 2. Sizing
### What the platform itself uses
Measured on the verification host (4 vCPU, 5.5 GB RAM, 6 GB swap, CentOS Stream 9
aarch64) with the control plane, the login and lobby system servers and one idle Paper
server running **[VM-VERIFIED]**:
| Process | Resident memory |
|---|---|
| k3s (server, kubelet, containerd) | ~1.1 GB |
| Velocity (`-Xms512M -Xmx1G`, heap pre-touched) | ~0.73 GB |
| lobby (Paper, pod limit 1 GiB) | ~0.7–0.85 GB |
| login (Limbo, pod limit 512 MiB) | ~0.16 GB |
| felis-api, felis-operator, registry gate | ~50 MB each |
| PostgreSQL | ~30 MB plus page cache |
| **Total in use** | **~3.4 GB** |
Every game server adds the memory its owner gave it: the pod's limit equals its request,
and the JVM heap is derived from it (§1a). Quotas cap it per user (panel → 管理 → 配额).
The installer's own peak is the image builds (Docker plus a Gradle container); it stops
Docker afterwards so that memory goes back to the servers. On a host under 2 GB of RAM
without swap it adds a 2 GiB `/swapfile`.
### Recommendations
| Concurrent players | Game servers running | CPU | RAM | `FELIS_VELOCITY_XMX` |
|---|---|---|---|---|
| up to 20 | 1–2 small | 2 vCPU | 4 GB + 2 GB swap | 1G (default) |
| up to 100 | 3–5 | 4 vCPU | 8–16 GB | 1G |
| up to 300 | 5–10 | 8 vCPU | 16–32 GB | 2G |
| 300+ | more | 8+ vCPU | 32 GB+ | 3G–4G |
The player-count rows are planning figures, not measurements: a Minecraft server's cost
depends mostly on what its players do (view distance, redstone, mods). Size RAM as the
platform's ~3.5 GB plus the sum of the servers you expect to run at once, then add a
quarter for the page cache and PostgreSQL. Velocity itself needs little per player; raise
its heap when `journalctl -u felis-velocity` shows long GC pauses or `OutOfMemoryError`.
`FELIS_VELOCITY_XMX` (default `1G`, at least `256M`, written `<n>M` or `<n>G`) is read on
every installer run. The initial heap stays at 512M, or equals the maximum when that is
lower. Changing it rewrites the unit, and the rerun restarts the proxy, which disconnects
everyone online; do it in a quiet hour **[VM-VERIFIED]**:
```
curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo FELIS_VELOCITY_XMX=2G bash
```
### Disk
| What | Where | Size |
|---|---|---|
| Worlds | one volume per server under `/var/lib/rancher/k3s/storage` | what the world grows to |
| World archives | the `felis-backups` volume (`FELIS_BACKUP_STORAGE`, default 10Gi requested) | about one compressed world per backup kept |
| In-cluster registry | the `registry` volume (default 10Gi requested) | 2–3 GB for the stock images; grows with custom builds, pruned daily (§9) |
| k3s's containerd images | `/var/lib/rancher/k3s/agent/containerd` | 6–9 GB |
| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store) | 5–10 GB after repeated upgrades |
| Toolchains and sources | `/opt/felis` | ~2.5 GB |
| Database bundles | `/var/lib/felis/db-backups` | a few MB each, 14 daily kept |
k3s's local-path volumes do not enforce the requested sizes (§9), so every volume shares
the root filesystem. Give the host at least **40 GB**, and 60 GB or more once worlds and
custom images accumulate. The watchdog mails the owners when a watched filesystem passes
its threshold, and §13b covers a full disk. `docker builder prune -af` (with Docker
started) reclaims the build cache when space is short; the next upgrade rebuilds it.
## 3. Uninstall
`deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove
and asks before it starts (`--yes` skips the question) **[SH-TESTED]
[VM-VERIFIED]**:
```
curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes # keep the data
curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --purge # remove the data too
```
With a private repository, fetch it the way the README fetches `bootstrap.sh`.
Both modes remove the `felis-*` systemd units and `cloudflared-felis.service`, the
Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the installer's cloudflared binary
(unless another unit runs it), the `felis_postgres` and `felis_edge` nftables tables and
the firewalld ports the installer opened. k3s goes with k3s's own `k3s-uninstall.sh` when
the cluster holds nothing but Felis's namespaces; when it runs anything else only
`felis`, `minecraft`, `felis-build` and the MinecraftServer CRD are deleted.
`--keep-k3s` and `--remove-k3s` override that choice.
| | keep data (default) | `--purge` |
|---|---|---|
| Final database bundle | taken first (`felis db backup -label manual`); a failure stops the uninstall before anything is removed. `--no-backup` skips it | none |
| `felis` database and role | kept | dropped; `listen_addresses` and `pg_hba.conf` go back to how they were |
| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file |
| `/var/lib/felis` (database bundles) | kept | deleted |
| Worlds, archives, registry, uploads | moved to `/var/lib/felis/retained/k3s-storage-<stamp>/` (with `--keep-k3s`: their volumes switch to `Retain` and stay in place) | deleted |
| Felis images, Docker build cache | kept | deleted |
Neither mode removes packages (Docker, PostgreSQL, git, nftables) or the swap file: other
software may use them. On a host that should end up bare:
```
sudo swapoff /swapfile && sudo rm /swapfile && sudo sed -i '\|^/swapfile |d' /etc/fstab
sudo dnf remove docker-ce docker-ce-cli containerd.io postgresql-server # or apt/zypper/pacman
```
The Cloudflare side outlives the host. After an uninstall that is final, delete the
tunnel (Zero Trust → Networks → Tunnels, or `cloudflared tunnel delete <name>`), its
DNS records for the panel hostnames, and the Access application.
### Reinstall on top of kept data
A keep-data uninstall leaves everything a reinstall needs. The installer reuses
`/etc/felis/secrets.env`, so the database password and the forwarding and session
secrets are unchanged, and it migrates the kept database instead of creating one
**[VM-VERIFIED]**.
Each step below was run on the reference VM after a keep-data uninstall, and the
restored worlds matched their kept `level.dat` checksums **[VM-VERIFIED]**. `kept` names
the directory the uninstall moved the volumes to:
```
kept="$(ls -d /var/lib/felis/retained/k3s-storage-* | tail -n 1)"
store=/var/lib/rancher/k3s/storage
```
1. Install as usual (`curl ... | sudo bash`). Name the same root domain if it was not
the `<ip>.nip.io` default: `felis.host.toml` is kept, and the installer reads the
domain from it.
2. Run `sudo felis setup`. It recreates the login and lobby servers; the Owner already
exists, so it opens on the status screen and you can quit there.
3. Put the image registry and the uploads back. They hold every custom server image
and uploaded file; without the registry, a restored server fails to pull its image.
```
sudo k3s kubectl -n felis scale deploy/registry deploy/felis-api --replicas=0
sudo k3s kubectl -n felis wait --for=delete pod -l app.kubernetes.io/component=registry --timeout=120s
sudo k3s kubectl -n felis wait --for=delete pod -l app.kubernetes.io/component=api --timeout=120s
sudo rsync -a --delete "$kept"/pvc-*_felis_registry/ "$(ls -d $store/pvc-*_felis_registry)"/
sudo rsync -a --delete "$kept"/pvc-*_felis_felis-uploads/ "$(ls -d $store/pvc-*_felis_felis-uploads)"/
sudo k3s kubectl -n felis scale deploy/registry deploy/felis-api --replicas=1
```
Then run the installer once more. It pushes this release's images over the older
copies the kept registry carried.
4. Bring the game servers back. The final bundle holds every MinecraftServer as it was;
the selector skips login and lobby, which step 2 created for this release:
```
b="$(ls -t /var/lib/felis/db-backups/felis-db-*-manual.tar | head -n 1)"
tar -xOf "$b" k8s/minecraftservers.json \
| sudo k3s kubectl apply -l '!felis.lolicon.best/system-role' -f -
```
5. Put each world back. A server's volume exists once it has started once, so start it
from the panel, stop it again, and copy the kept world over the new one:
```
s=<server>
sudo rsync -a --delete "$kept"/pvc-*_minecraft_world-$s-0/ "$(ls -d $store/pvc-*_minecraft_world-$s-0)"/
```
Then start it. The lobby works the same way: stop it with
`sudo k3s kubectl -n minecraft patch minecraftserver lobby --type=merge -p '{"spec":{"desiredState":"Stopped"}}'`,
copy `world-lobby-0`, and patch it back to `Running`.
6. Bring the archives back so the panel's restore points work again. The archive volume
appears with the first backup, so back up any server from the panel first, then:
```
sudo rsync -a "$kept"/pvc-*_minecraft_felis-backups/ "$(ls -d $store/pvc-*_minecraft_felis-backups)"/
```
With an off-site bucket configured, `sudo felis offsite fetch-worlds` fetches them
instead (troubleshooting §16).
7. Delete `/var/lib/felis/retained/` once every server is back.
## 4. Upgrading the pieces around Felis
A rerun of the installer upgrades Felis itself (§15). The components it installs keep
the version they were installed with unless noted:
| Component | How a rerun treats it | Upgrade |
|---|---|---|
| Velocity, Limbo, Paper, LuckPerms | follow `deploy/game-stack.lock` | rerun after a release that moves the lock (§15b) |
| Temurin JRE | moves to the pinned patch build | rerun |
| k3s | left alone | by hand, one minor version at a time: `curl -sfL https://get.k3s.io \| INSTALL_K3S_VERSION=<version> sh -` |
| cloudflared | left alone | replace `/usr/local/bin/cloudflared` with the release binary, then `systemctl restart cloudflared-felis` |
| PostgreSQL | the distribution's package | the package manager; a major version needs `pg_upgrade` first (the installer refuses to start a newer server on an older cluster) |
| Docker, git, nftables | distribution packages | the package manager |
`sudo felis update` reports Felis, Velocity, k3s and cloudflared against their newest
releases.
## 5. Disaster recovery
The procedures are in §16: what a database bundle holds, restoring one on the same host,
rolling back an upgrade, and rebuilding on a new host from the off-site copy. For a
production install:
- **Configure the off-site copy** (`FELIS_OFFSITE_*`, §16 "Keep a copy somewhere
else"). Without it the world archives sit on the same disk as the worlds, and the
database bundles on the same disk as the database; losing the disk loses both. The
installer ends with `NO OFF-SITE COPY` until it is set.
- **Keep the off-site encryption key off the host**, in a password manager. The bucket
holds only sealed objects.
- **Keep one database bundle off the host** as well when there is no bucket. It contains
`secrets.env`, which a rebuild needs to read the rest.
- **Rehearse the rebuild** once on a spare VM: §16 "Rebuild on a new host", steps 1–5,
then log in and restore one world. `felis offsite status` and `felis db check` exit
non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
or rely on the watchdog's mail.