#!/usr/bin/env bash # # Felis one-line bootstrap installer. # # curl -fsSL /deploy/bootstrap.sh | sudo bash # # Brings a fresh single-node Linux host from nothing to a running Felis control # plane: it installs whatever is missing (picking apt/dnf/yum/zypper/pacman by OS), provisions a # swap file on tiny hosts, then configures Docker, k3s and PostgreSQL, builds and # imports the felis image, runs database migrations and applies the rendered # install bundle (CRD + namespaces + RBAC + NetworkPolicies + control-plane # Deployments + in-cluster registry). # # The recommended entrypoint is now `sudo felis setup`, which wraps this # bootstrap in a TUI and then continues to the Owner/edge setup. This script # remains usable directly for raw host provisioning. # # At the start it asks what to install: # [1] Felis — the full control plane described above. # [2] Felis-nano — ONLY the Yggdrasil hasJoined multiplexer (`felis nano`) as a # systemd service: no k3s, no Postgres, no bundle. For a # third-party server operator who just wants multi-Yggdrasil # auth federation. Preselect non-interactively with # FELIS_INSTALL_MODE=nano. # # The script is idempotent: re-running it converges rather than duplicating, and # generated secrets are persisted to /etc/felis/secrets.env so reruns reuse them. # # Tunables override the demo defaults. sudo resets the environment, so a variable exported # before `curl ... | sudo bash` never arrives. Name it on the sudo line, or keep it with -E: # curl -fsSL /deploy/bootstrap.sh | sudo FELIS_INSTALL_MODE=nano FELIS_NANO_LISTEN=10.0.0.5:8081 bash # export FELIS_INSTALL_MODE=nano; curl -fsSL /deploy/bootstrap.sh | sudo -E bash # FELIS_INSTALL_MODE full|nano — skip the prompt (default: ask on a tty, else full; nano # instead on a host that runs felis-nano and no full install) # FELIS_NANO_LISTEN listen addr for `felis nano` (default: the address an installed # felis-nano already uses, else 127.0.0.1:8081 — loopback only; set a # private-network IP to serve an off-host proxy) # FELIS_NANO_PROXY_CIDR the proxy allowed to reach a non-loopback nano bind, as an address # with a prefix length (for example 10.0.0.7/32). firewalld opens the # port to that source only; unset, it opens nothing # FELIS_LEGACY_FORWARDING_SERVERS comma-separated backends that always receive their # identity through the handshake address instead of modern forwarding # (default: legacy18). A floor: any server whose MinecraftServer CR is # labelled felis.lolicon.best/forwarding=legacy joins it while the # proxy runs (docs/operations.md), so a new 1.8 backend needs a label, # not a re-run. Changing the floor itself means re-running this script. # FELIS_VELOCITY_XMX maximum heap of the Velocity proxy, as M or G (default: 1G; # at least 256M). docs/operations.md sizes it by player count. # FELIS_VELOCITY_FORK_JAR path to a Felis-Legacy Velocity fork build to install as the # proxy instead of the stock download (default: unset, stock). # FELIS_VELOCITY_FORK_JAR_SHA256 expected sha256 of that jar. REQUIRED whenever the jar # above is set; the install refuses on a mismatch. # FELIS_GAME_STACK pinned|latest — which Limbo, Paper, LuckPerms and Velocity builds to # install (default: pinned, the builds deploy/game-stack.lock names, # each checked against its sha256). latest resolves upstream's newest # builds on every run; the Minecraft version then follows Limbo's CI. # FELIS_JRE_VERSION Temurin feature version for the proxy (default: 25, whose build and # digests are pinned; a rerun moves an installer-managed JRE to the # pinned build). Another feature version is checked against the # digest Adoptium's API publishes for it. # FELIS_GO_VERSION Go toolchain used to build the nano binary (default: 1.26.8) # FELIS_GO_SHA256 sha256 of that version's linux tarball for this host's architecture. # REQUIRED for a non-default FELIS_GO_VERSION; the default's is pinned. # FELIS_K3S_VERSION k3s release a fresh install gets (default: v1.36.4+k3s1). An # installed k3s is left alone unless FELIS_UPGRADE_DEPS=1. # FELIS_CLOUDFLARED_VERSION / FELIS_CLOUDFLARED_SHA256 cloudflared release installed # when none is present (default: 2026.9.1, digests pinned); the # sha256 is REQUIRED for any other version # FELIS_UPGRADE_DEPS 1 moves an installed k3s and cloudflared to the versions above # (k3s one minor version at a time; neither is ever downgraded) and # restarts cloudflared-felis onto the new binary (default: 0) # FELIS_REPO_URL git URL to build from (raw script mode only) # FELIS_VERSION_BOOTSTRAP release|dev — which version to install (default: release). # release DOWNLOADS the prebuilt felis binary published for the newest # tag (panel included — it is go:embed'ed into that same binary) and # builds only a thin image around it; dev clones and compiles. If the # asset is missing or this architecture has none, release warns and falls # back to compiling the SAME tag. The downloaded binary must match # the release's SHA256SUMS before it is run; a release without one # is compiled from source too. The game stack is always built here. # FELIS_GITHUB_TOKEN GitHub token; REQUIRED while the repo is private # FELIS_REF branch/tag/sha — pins the build, overrides the channel, and forces a # source build (naming a ref asks for that tree, not a published asset) # FELIS_IMAGE control-plane image ref (default: # registry.felis.svc:5000/felis/felis:, so each # release has its own tag and `kubectl rollout undo` returns to the # previous one; :demo when the version is unknown — never :latest; # anything not under the registry is used as-is but is NOT # mirrored into it, so it has no pull source after an image GC) # FELIS_ROOT_DOMAIN deployment root domain (default: .nip.io; first install # only -- a rerun keeps the installed one, and sudo felis domain set # moves it) # FELIS_PANEL_NODEPORT local HTTPS panel/API NodePort (default: 30443) # FELIS_EGRESS_MODE loadbalancer|nodeport (default: nodeport — no MetalLB on a demo box) # FELIS_BACKUP_PVC world-archive PVC the installer renders and felis-api hands to its # backup/restore Jobs (default: felis-backups; empty string disables # backups — the endpoints answer 503) # FELIS_ARCHIVE_LOCAL_PATH path that PVC is mounted at inside those Jobs; written into # felis.toml [archive] local_path (default: /var/lib/felis/archives) # FELIS_WORLDS_HOST_PATH node directory holding the world volumes (on the k3s this # installer provisions: /var/lib/rancher/k3s/storage). Setting it # lets the daily reaper also archive and then delete worlds idle # for 15 days (without it the reaper only deletes backups past # their expiry and leaves every world); the reaper reads that # root as root, so it keeps k3s's own 0700 root:root # (default: unset = worlds are never reaped) # FELIS_REGISTRY_STORAGE / FELIS_UPLOADS_STORAGE / FELIS_BACKUP_STORAGE capacity the # registry, uploads and world-archive PVCs request on first install # (defaults: 10Gi, 5Gi, 10Gi). An existing claim keeps its size; on # k3s local-path the number is not enforced, see troubleshooting §9 # FELIS_MANAGE_TIME_SYNC 0 leaves the host's clock alone; by default the installer # turns NTP on (installing chrony when nothing can) and waits for it # to synchronize (default: 1) # FELIS_MANAGE_JOURNAL 0 leaves journald alone; by default the installer makes the # system journal persistent so logs survive a reboot (default: 1) # FELIS_JOURNAL_MAX_USE the persistent journal's size cap, journald's SystemMaxUse # written K|M|G (default: 1G) # PKG_LOCK_TIMEOUT seconds to wait for package-manager locks (default: 900) # APT_LOCK_TIMEOUT legacy alias for PKG_LOCK_TIMEOUT set -Eeuo pipefail # --------------------------------------------------------------------------- # Configuration & constants # --------------------------------------------------------------------------- FELIS_REPO_URL="${FELIS_REPO_URL:-https://github.com/FelisMC/Felis.git}" # Which version to install. "release" builds the newest published GitHub release; # "dev" builds the tip of main. Release is the default because an installer that # tracks a moving branch by default hands every new host a different, untested # commit — the version an operator reports in a bug is then meaningless. FELIS_VERSION_BOOTSTRAP="${FELIS_VERSION_BOOTSTRAP:-release}" # Empty by default and resolved from the channel below. Setting it explicitly pins the # build to that ref and skips resolution entirely: naming a ref IS asking for a # development build, so it takes the dev-FORM stamp regardless of the channel. Skipping # resolution also skips the tag lookup, so the base stays v0.0.0 and the stamp is # v0.0.0+g rather than +g — deliberate, so pinning a ref costs no # network call the operator did not ask for. FELIS_REF="${FELIS_REF:-}" # Recorded HERE because resolve_install_ref overwrites FELIS_REF on both channels — after it # runs, "did the operator pin a ref?" is unanswerable. Naming a ref asks for THAT tree to be # built, so it takes the source path even on the release channel. FELIS_REF_PINNED="" if [ -n "$FELIS_REF" ]; then FELIS_REF_PINNED=1; fi # Set once a prebuilt felis binary is installed at HOST_BIN, by either the TUI hand-off or a # release download. It is what the image build, the CRD apply and the game stack key off: # all three only need "is there a binary and no checkout", never "which route got us here". HAVE_PREBUILT_BINARY="" # The image the control plane ran before this run moved it (deploy_bundle), for the rollback # hint in summary. PREVIOUS_FELIS_IMAGE="" # Optional GitHub credential, needed while this repository is private: GitHub answers # 404 (not 403) for a repo the caller cannot see, so without it both the release lookup # and the clone fail as "not found". Exported because git's credential helper below runs # as a child process and reads it from the environment — which is also why it is never # interpolated into the clone URL. A token in the URL is written verbatim into # .git/config and survives the install; an environment variable does not. FELIS_GITHUB_TOKEN="${FELIS_GITHUB_TOKEN:-}" export FELIS_GITHUB_TOKEN # Set by resolve_install_ref/stamp_version and linked into the binary as main.version. FELIS_VERSION="" FELIS_VERSION_BASE="" # Where the installer parks every image it builds, so kubelet can re-pull one the # image GC has collected (the disk-pressure drill's dead end: ImagePullBackOff with # nothing to pull from). The node pulls through the loopback hostPort the registry # Deployment binds (configure_registry_mirror below); pushes go through # REGISTRY_PUSH_HOST — docker treats 127.0.0.1 as insecure by default, so the # daemon needs no insecure-registries entry for it. REGISTRY_URL="registry.felis.svc:5000" REGISTRY_PUSH_HOST="127.0.0.1:${REGISTRY_URL##*:}" # Empty unless the operator names one: resolve_felis_image derives the tag from the version # this run installs, which is only known once the binary or the checkout is. FELIS_IMAGE="${FELIS_IMAGE:-}" FELIS_EGRESS_MODE="${FELIS_EGRESS_MODE:-nodeport}" FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}" # World-archive storage. The installer renders this PVC (minecraft namespace) and felis-api # advertises it to its backup/restore Jobs; emptying it disables backups (503). The archive # path is written into felis.toml so the Jobs' mount and [archive] local_path agree by # construction — a mismatch would leave tarLocal's absolute archive refs unresolvable. FELIS_BACKUP_PVC="${FELIS_BACKUP_PVC:-felis-backups}" FELIS_ARCHIVE_LOCAL_PATH="${FELIS_ARCHIVE_LOCAL_PATH:-/var/lib/felis/archives}" # PVC capacities; empty keeps `felis manifests`' defaults. FELIS_REGISTRY_STORAGE="${FELIS_REGISTRY_STORAGE:-}" FELIS_UPLOADS_STORAGE="${FELIS_UPLOADS_STORAGE:-}" FELIS_BACKUP_STORAGE="${FELIS_BACKUP_STORAGE:-}" # Retention is opt-in because it DELETES worlds (after a verified archive): point this at the # node directory the world volumes live under. On the k3s this installer provisions that is # /var/lib/rancher/k3s/storage — the reaper resolves each PVC's local-path directory exactly # from its volumeName. Left unset, the reaper CronJob renders retention-only: backups past # their expiry are still deleted daily, and no world is ever archived or deleted. FELIS_WORLDS_HOST_PATH="${FELIS_WORLDS_HOST_PATH:-}" # k3s's local-path provisioner root. It appears with the first volume the provisioner # creates, which on a fresh install is after the reaper's PV has been applied. K3S_STORAGE_ROOT="/var/lib/rancher/k3s/storage" # Control-plane database backups (felis db backup): a daily timer bundles pg_dump with the # /etc/felis state a rebuild needs, and every upgrade that has migrations to apply snapshots # the database first (felis migrate up). The directory sits outside /var/lib/rancher on # purpose: reinstalling k3s must not take the database backups with it. Copy it off the # host for anything beyond "undo a bad upgrade or a mistaken delete" (troubleshooting §16). FELIS_DB_BACKUP_DIR="${FELIS_DB_BACKUP_DIR:-/var/lib/felis/db-backups}" FELIS_DB_BACKUP_KEEP="${FELIS_DB_BACKUP_KEEP:-14}" FELIS_DB_BACKUP_TIME="${FELIS_DB_BACKUP_TIME:-*-*-* 03:30:00}" # node-exporter textfile collector target; FelisDBBackupStale (deploy/alerts) reads it. FELIS_DB_BACKUP_METRICS="${FELIS_DB_BACKUP_METRICS:-/var/lib/node_exporter/textfile_collector/felis_db_backup.prom}" # 0 migrates without the pre-migration snapshot, e.g. against an external database newer # than this host's pg_dump. The upgrade stops if the snapshot fails and this is not set. FELIS_PRE_MIGRATE_BACKUP="${FELIS_PRE_MIGRATE_BACKUP:-1}" # The off-site copy (felis offsite, troubleshooting §16). Every hour the host encrypts each # world archive and the newest database bundles and copies them to an S3-compatible bucket, # and the reaper deletes an idle world only once its archive is there. Without a bucket # every backup lives on this one machine, and losing its disk loses them all. Set the # bucket and its endpoint to turn it on; FELIS_OFFSITE_ACCESS_KEY / FELIS_OFFSITE_SECRET_KEY # (and optionally a FELIS_OFFSITE_KEY from `felis offsite keygen`) go into # /etc/felis/offsite.env, mode 0600, with the encryption key generated when there is none. # A re-run without these keeps the [offsite] felis.host.toml already has. FELIS_OFFSITE_ENDPOINT="${FELIS_OFFSITE_ENDPOINT:-}" FELIS_OFFSITE_BUCKET="${FELIS_OFFSITE_BUCKET:-}" FELIS_OFFSITE_REGION="${FELIS_OFFSITE_REGION:-}" FELIS_OFFSITE_PREFIX="${FELIS_OFFSITE_PREFIX:-}" FELIS_OFFSITE_DB_KEEP="${FELIS_OFFSITE_DB_KEEP:-}" INSTALL_MODE="${FELIS_INSTALL_MODE:-}" # Loopback by default: hasJoined is an unauthenticated endpoint by protocol (Velocity # sends no token), so a public bind is a free auth relay — anyone can point their own # proxy at it and spend YOUR egress IP on Mojang, until Mojang rate-limits you and your # own players stop getting in. Same-host Velocity reaches 127.0.0.1 fine; a proxy on # another machine must opt in explicitly with FELIS_NANO_LISTEN=:8081. # Left empty here: resolve_nano_listen applies that default only after an existing unit's # address has had its say. FELIS_NANO_LISTEN="${FELIS_NANO_LISTEN:-}" FELIS_NANO_PROXY_CIDR="${FELIS_NANO_PROXY_CIDR:-}" # Backends that take their forwarded identity through the handshake address instead of # proxy-wide modern forwarding. See write_velocity_service for why a protocol-47 backend # needs this. Overridable because adding a second 1.8 backend otherwise means editing this # script; it is still a restart-time list, not one that follows the CRs. FELIS_LEGACY_FORWARDING_SERVERS="${FELIS_LEGACY_FORWARDING_SERVERS:-legacy18}" FELIS_VELOCITY_XMX="${FELIS_VELOCITY_XMX:-1G}" # The Go tarball is unpacked and run as root, so the default version is pinned by the sha256 # go.dev/dl publishes for each architecture install_go_toolchain handles. Move all three # together; any other FELIS_GO_VERSION has to bring its own FELIS_GO_SHA256. GO_PINNED_VERSION="1.26.8" GO_PINNED_SHA256_AMD64="d0f743b33e8d8945e6b1f432edd15785c70507121d6e2a723b21285eddf8b57b" GO_PINNED_SHA256_ARM64="211ffced9dcb9633a55eac6364816ec0ddd951389a740e88fa8b3337971bdda0" FELIS_GO_VERSION="${FELIS_GO_VERSION:-$GO_PINNED_VERSION}" FELIS_GO_SHA256="${FELIS_GO_SHA256:-}" # cloudflared runs as root on the edge, so it gets the same treatment: a pinned release and # the sha256 GitHub lists for each asset. A different FELIS_CLOUDFLARED_VERSION has to bring # its own FELIS_CLOUDFLARED_SHA256. An installed binary is replaced only under # FELIS_UPGRADE_DEPS=1; `felis update --cloudflared` reports when that would change it. CLOUDFLARED_PINNED_VERSION="2026.9.1" CLOUDFLARED_PINNED_SHA256_AMD64="03f1f25d1cc93b9ad6c60569d44060bc4f17ed97075760ed8cfca4b12dcd68cc" CLOUDFLARED_PINNED_SHA256_ARM64="3d97437c71848bd8df68041e12436b484a661d95073ea1937f01a845ce88faa3" CLOUDFLARED_PINNED_SHA256_ARM="093ffa3638ab2b636de63c43a8c68f96a69cf71f9699dd8277a91b160b0f4fc0" FELIS_CLOUDFLARED_VERSION="${FELIS_CLOUDFLARED_VERSION:-$CLOUDFLARED_PINNED_VERSION}" FELIS_CLOUDFLARED_SHA256="${FELIS_CLOUDFLARED_SHA256:-}" CLOUDFLARED_BIN=/usr/local/bin/cloudflared # The k3s release a fresh install gets, and the tag its install script is read from. The # script checks the k3s binary against that release's sha256sum file, so pinning the tag # pins both. An installed k3s moves only under FELIS_UPGRADE_DEPS=1. FELIS_K3S_VERSION="${FELIS_K3S_VERSION:-v1.36.4+k3s1}" FELIS_UPGRADE_DEPS="${FELIS_UPGRADE_DEPS:-0}" FELIS_MANAGE_TIME_SYNC="${FELIS_MANAGE_TIME_SYNC:-1}" FELIS_MANAGE_JOURNAL="${FELIS_MANAGE_JOURNAL:-1}" FELIS_JOURNAL_MAX_USE="${FELIS_JOURNAL_MAX_USE:-1G}" # The in-cluster registry's image, by digest. It must equal platform.defaultRegistryImage # (internal/platform/identities.go, TestBootstrapPinsTheRegistryImage): the renderer puts # that ref in the Deployment, and this script caches and pins the same ref in containerd. REGISTRY_IMAGE="docker.io/library/registry:2.8.3@sha256:a3d8aaa63ed8681a604f1dea0aa03f100d5895b6a58ace528858a7b332415373" PKG_LOCK_TIMEOUT="${PKG_LOCK_TIMEOUT:-${APT_LOCK_TIMEOUT:-900}}" APT_LOCK_TIMEOUT="${APT_LOCK_TIMEOUT:-$PKG_LOCK_TIMEOUT}" # --- the game stack: proxy on the host, the two always-on backends in k3s --- FELIS_LIMBO_IMAGE="${FELIS_LIMBO_IMAGE:-${REGISTRY_URL}/felis/limbo:demo}" FELIS_LOBBY_IMAGE="${FELIS_LOBBY_IMAGE:-${REGISTRY_URL}/felis/lobby:demo}" # Plain Paper base recommended for a user's own server (deploy/paper). Not a system # server — forwarding is applied by the operator's init-forwarding initContainer, so it # needs no secret. Seeded recommended in 0019_recommended_paper.sql. FELIS_PAPER_IMAGE="${FELIS_PAPER_IMAGE:-${REGISTRY_URL}/felis/paper:demo}" # The Velocity build comes from deploy/game-stack.lock (VELOCITY_VERSION and its jar digest). # Setting FELIS_VELOCITY_VERSION to another version, or FELIS_GAME_STACK=latest, installs # the newest BUILD of that minor instead. The minor itself is never discovered: PaperMC's # Fill v3 groups velocity builds by version group, and "newest across all groups" can mean # an unreleased 4.x SNAPSHOT that needs another runtime. Crossing a major is a deliberate # code change. FELIS_VELOCITY_VERSION="${FELIS_VELOCITY_VERSION:-}" VELOCITY_LATEST_MINOR="3.5.1" # pinned installs the builds deploy/game-stack.lock names (resolve_game_jars); latest asks # upstream for its newest ones, which is how that lock file gets refreshed. FELIS_GAME_STACK="${FELIS_GAME_STACK:-pinned}" # Path to a Felis-Legacy Velocity fork build, installed as the proxy in place of the # stock download. Unset — the default — changes nothing. # # Stock Velocity will not offer the login-plugin-message exchange below 1.13, so a 1.8 # client reaching a modern-forwarding backend today is a side effect of Via replacing # the channel initializers before that check runs. It works, and nobody designed it. # The fork registers the login packets on 1.7.2 and drops the gate, which makes the # same outcome deliberate. # # Opt-in because it is unmeasured where it counts: FL-008's probe runs offline-mode # against a stub, and this jar would carry every real Mojang session on the server. FELIS_VELOCITY_FORK_JAR="${FELIS_VELOCITY_FORK_JAR:-}" # Expected sha256 of that jar, REQUIRED whenever it is set. Case and internal spaces are # ignored, so whatever sha256sum, Get-FileHash or certutil printed can be pasted as-is. # No digest is hardcoded here: # the build lives in Felis-Legacy and has never been reproduced on a second machine, so # any constant this script carried would pin one machine's output rather than the fork. # # So this is not a supply-chain signature and does not pretend to be one — an operator # who can write the jar can write this value too. What it does buy: a path is not an # identity, and every re-run of this script re-checks it. A truncated copy, a stale build # left at the same path, or the two-patch jar where the three-patch one was meant all # change the digest and stop the install. Naming the digest once is what turns "whatever # is at that path today" into one specific build. FELIS_VELOCITY_FORK_JAR_SHA256="${FELIS_VELOCITY_FORK_JAR_SHA256:-}" # Temurin 25: Velocity 3.5 needs 21+, and 25 is also what a future Velocity 4 requires, # so the runtime does not have to move again when the pin does. Distro JDK packaging is # a lottery across four package managers — a tarball is one code path everywhere (same # reasoning as install_go_toolchain). FELIS_JRE_VERSION="${FELIS_JRE_VERSION:-25}" # The JRE the proxy runs on is unpacked as root and carries every player session, so the # default feature version is pinned to one build and the sha256 Adoptium publishes for each # architecture. Move the three together (the Adoptium API lists them: # /v3/assets/latest/25/hotspot?image_type=jre&os=linux). JRE_PINNED_FEATURE="25" JRE_PINNED_RELEASE="25.0.4.1+1" JRE_PINNED_SHA256_X64="1731a34baadec5479258ea0202e4d5d865d2efeee60cb0c7d7eb056fe96ca219" JRE_PINNED_SHA256_AARCH64="34828cbb93ed31c281c84ecb31ddab655d11a802f263c1fc019d42e9e0230fed" # The port the proxy listens on: the ONLY Minecraft port players ever touch. Backends # are ClusterIP-only, verify the modern-forwarding HMAC, and use NetworkPolicy to limit # non-node ingress to the declared proxy CIDRs. FELIS_GAME_PORT="${FELIS_GAME_PORT:-25565}" # Velocity server names — must match internal/naming (SystemLoginServer/SystemLobbyServer); # felis-api hands the proxy backends under exactly these names. LOGIN_SERVER="login" LOBBY_SERVER="lobby" CONTROL_NS="felis" MINECRAFT_NS="minecraft" BUILD_NS="felis-build" POD_CIDR="10.42.0.0/16" # k3s default cluster CIDR SERVICE_CIDR="10.43.0.0/16" # k3s default service CIDR DB_NAME="felis" DB_USER="felis" STATE_DIR="/etc/felis" SECRETS_ENV="${STATE_DIR}/secrets.env" BOOTSTRAP_DONE="${STATE_DIR}/bootstrap.done" PANEL_TLS_CERT="${STATE_DIR}/panel-tls.crt" PANEL_TLS_KEY="${STATE_DIR}/panel-tls.key" SRC_DIR="/opt/felis/src" HOST_BIN="/usr/local/bin/felis" # The binary this run replaced (keep_previous_host_binary), and whether the new one has # been put to use: once migrations start, or the nano service restarts onto it, the old # one no longer matches what is running and a failed run keeps the new one. HOST_BIN_PREV="" HOST_BIN_KEPT=0 HOST_BIN_IN_USE=0 # Felis's own build toolchain, not /usr/local/go: install_go_toolchain replaces whatever # version sits here, and an operator's Go at the conventional path is not ours to swap. GOROOT_DIR="/opt/felis/go" NANO_SERVICE="/etc/systemd/system/felis-nano.service" DB_BACKUP_SERVICE="/etc/systemd/system/felis-db-backup.service" DB_BACKUP_TIMER="/etc/systemd/system/felis-db-backup.timer" WATCHDOG_SERVICE="/etc/systemd/system/felis-watchdog.service" WATCHDOG_TIMER="/etc/systemd/system/felis-watchdog.timer" UPDATE_CHECK_SERVICE="/etc/systemd/system/felis-update-check.service" UPDATE_CHECK_TIMER="/etc/systemd/system/felis-update-check.timer" WATCHDOG_STATE="/var/lib/felis/watchdog/state.json" OFFSITE_ENV="${STATE_DIR}/offsite.env" OFFSITE_SERVICE="/etc/systemd/system/felis-offsite.service" OFFSITE_TIMER="/etc/systemd/system/felis-offsite.timer" BUILD_TOOLS_SERVICE="/etc/systemd/system/felis-build-tools.service" BUILD_TOOLS_TIMER="/etc/systemd/system/felis-build-tools.timer" BUILD_TOOLS_STATUS="/var/lib/felis/build-tools/status.json" # While this marker holds a future Unix time, felis watchdog mails nothing: an install # restarts the control plane and the system servers on purpose. cleanup removes it; the # time in it is the backstop for an installer killed before its EXIT trap runs. WATCHDOG_QUIET_FILE="/run/felis/watchdog-quiet-until" VELOCITY_DIR="/opt/felis/velocity" VELOCITY_USER="felis-velocity" VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service" # What the running proxy was last started from (velocity_fingerprint), and the builds the # login and lobby pods were last started on: a rerun restarts only what actually changed, # since each of those restarts disconnects every player online. VELOCITY_FINGERPRINT="${STATE_DIR}/velocity.fingerprint" SYSTEM_SERVER_IMAGES="${STATE_DIR}/system-server-images" # The nftables rules that keep PostgreSQL's port to this host and its pods, on a host # without firewalld (configure_postgres_firewall). PG_FIREWALL_RULES="${STATE_DIR}/postgres-firewall.nft" PG_FIREWALL_SERVICE="/etc/systemd/system/felis-postgres-firewall.service" JRE_DIR="/opt/felis/jre" # The gradle image the lobby and limbo Dockerfiles build their plugins in, digest # included; build_velocity_plugin runs the same one. PLUGIN_BUILD_IMAGE="gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01" K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}" K3S_BIN="${K3S_BIN_DIR}/k3s" # k3s's containerd mirror config, written by configure_registry_mirror. A variable # (not just the literal path) so bootstrap_test.sh can point the writer at a # scratch file. K3S_REGISTRIES_FILE="/etc/rancher/k3s/registries.yaml" # The installer's k3s settings (write_k3s_config), a drop-in k3s reads after any # config.yaml the operator keeps. The unit, the admin kubeconfig and the kubelet's # client certificate are variables for the same reason as the file above. K3S_CONFIG_DROPIN="/etc/rancher/k3s/config.yaml.d/50-felis.yaml" K3S_UNIT_FILE="/etc/systemd/system/k3s.service" K3S_KUBECONFIG="/etc/rancher/k3s/k3s.yaml" K3S_KUBELET_CERT="/var/lib/rancher/k3s/agent/client-kubelet.crt" # ensure_persistent_journal's drop-in, and the directory journald creates once it # stores the journal persistently. JOURNALD_DROPIN="/etc/systemd/journald.conf.d/50-felis.conf" JOURNAL_DIR="/var/log/journal" APT_LOCK_FILES=( /var/lib/dpkg/lock-frontend /var/lib/dpkg/lock /var/cache/apt/archives/lock /var/lib/apt/lists/lock ) APT_BACKGROUND_TIMERS=( apt-daily.timer apt-daily-upgrade.timer ) APT_BACKGROUND_SERVICES=( apt-daily.service apt-daily-upgrade.service unattended-upgrades.service ) DNF_BACKGROUND_TIMERS=( dnf-makecache.timer dnf-automatic.timer ) DNF_BACKGROUND_SERVICES=( dnf-makecache.service dnf-automatic.service ) YUM_BACKGROUND_TIMERS=( yum-cron.timer ) YUM_BACKGROUND_SERVICES=( yum-cron.service ) ZYPPER_BACKGROUND_TIMERS=( packagekit-background.timer ) ZYPPER_BACKGROUND_SERVICES=( packagekit.service ) # --------------------------------------------------------------------------- # Clean PATH (sudo may strip /usr/local/bin) # --------------------------------------------------------------------------- PATH="/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" export PATH # BuildKit attaches a provenance attestation to every build by default, and it records the # build's start time. On Docker's containerd image store the image id is the digest of the # index that carries it, so an unchanged rebuild would get a new id each run: the login and # lobby pods would restart on every rerun, and each run would push another versioned tag. # Without it the id is the manifest digest, which an all-cached build reproduces. export BUILDX_NO_DEFAULT_ATTESTATIONS=1 # --------------------------------------------------------------------------- # Logging # --------------------------------------------------------------------------- log() { printf '\033[1;36m[felis]\033[0m %s\n' "$*"; } ok() { printf '\033[1;32m[ ok ]\033[0m %s\n' "$*"; } warn() { printf '\033[1;33m[warn]\033[0m %s\n' "$*" >&2; } die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; } TEMP_PATHS=() DOCKER_CONTAINERS=() REGISTRY_DOCKER_CONFIG="" PKG_TIMERS_TO_RESTORE=() on_error() { local line="$1" code="$2" warn "bootstrap failed near line ${line} (exit ${code})" } cleanup() { local status=$? id path unit restore_previous_host_binary "$status" for unit in "${PKG_TIMERS_TO_RESTORE[@]-}"; do [ -n "$unit" ] || continue systemctl start "$unit" >/dev/null 2>&1 || true done if command -v docker >/dev/null 2>&1; then for id in "${DOCKER_CONTAINERS[@]-}"; do [ -n "$id" ] && docker rm "$id" >/dev/null 2>&1 || true done fi for path in "${TEMP_PATHS[@]-}"; do [ -n "$path" ] && rm -rf -- "$path" || true done # A failed install leaves something broken the owners should hear about, so the # watchdog speaks again the moment the installer exits, however it exits. rm -f -- "$WATCHDOG_QUIET_FILE" 2>/dev/null || true } # keep_previous_host_binary copies the felis binary this run is about to replace, once. A # run that fails before the new binary is in use puts it back (restore_previous_host_binary): # until then the old binary, the old cluster and the unmigrated database still agree, while # a new binary left behind runs the host timers against a schema it was not built for, and # `felis setup` with it would migrate the database under the old control plane. keep_previous_host_binary() { [ "$HOST_BIN_KEPT" = 0 ] || return 0 HOST_BIN_KEPT=1 [ -x "$HOST_BIN" ] || return 0 HOST_BIN_PREV="${HOST_BIN}.prev" rm -f "$HOST_BIN_PREV" cp "$HOST_BIN" "$HOST_BIN_PREV" } restore_previous_host_binary() { # exit-status [ -n "$HOST_BIN_PREV" ] && [ -f "$HOST_BIN_PREV" ] || return 0 if [ "$1" -ne 0 ] && [ "$HOST_BIN_IN_USE" != 1 ]; then # install(1) onto a fresh file, as everywhere else HOST_BIN is written (SELinux label). rm -f "$HOST_BIN" if install -m 0755 "$HOST_BIN_PREV" "$HOST_BIN"; then command -v restorecon >/dev/null 2>&1 && restorecon "$HOST_BIN" >/dev/null 2>&1 || true warn "restored the previous felis binary at ${HOST_BIN}; the database was not migrated, so rerunning the installer picks up where this run stopped" else warn "could not restore the previous felis binary; it is at ${HOST_BIN_PREV}" return 0 fi fi rm -f -- "$HOST_BIN_PREV" } remember_temp() { TEMP_PATHS+=("$1"); } remember_container() { DOCKER_CONTAINERS+=("$1"); } # Gives a file the SELinux label its path calls for, on hosts that have SELinux. It # repairs files earlier installers wrote under /tmp and moved into place, which kept # user_tmp_t (a confined daemon is then denied them). restore_label() { if command -v restorecon >/dev/null 2>&1; then restorecon "$1" || true; fi; } trap 'on_error "$LINENO" "$?"' ERR trap cleanup EXIT k3s_cmd() { [ -x "$K3S_BIN" ] || die "k3s binary not found at ${K3S_BIN}"; "$K3S_BIN" "$@"; } kube() { k3s_cmd kubectl "$@"; } # Create or update a single-key Secret without putting the value in kubectl's argv. # The temporary file is mode 0600 and is also registered with the EXIT cleanup path. apply_literal_secret() { local namespace="$1" name="$2" key="$3" value="$4" tmp tmp="$(umask 077; mktemp)" remember_temp "$tmp" printf '%s' "$value" > "$tmp" kube -n "$namespace" create secret generic "$name" \ --from-file="${key}=${tmp}" \ --dry-run=client -o yaml | kube apply -f - rm -f "$tmp" } # The control plane mounts felis-config from its own namespace; the workload # namespace's backup/restore/fileedit Jobs and the reaper mount a local copy (a # secretKeyRef is namespace-local). The installer owns the rendered config, so both # copies are (re)applied on every run — unlike the create-if-absent credential # replicas `felis setup` makes, because a stale config copy keeps an old database URL # or archive policy after an upgrade or a credential rotation. apply_felis_config_secrets() { kube -n "$CONTROL_NS" create secret generic felis-config \ --from-file=felis.toml="${STATE_DIR}/felis.pod.toml" \ --dry-run=client -o yaml | kube apply -f - kube -n "$MINECRAFT_NS" create secret generic felis-config \ --from-file=felis.toml="${STATE_DIR}/felis.pod.toml" \ --dry-run=client -o yaml | kube apply -f - } # The registry gate reads one token file per principal from felis-registry-auth # (registry namespace = control namespace); a build Job's push container reads the # build principal's username/password from felis-registry-push in the build # namespace. Values go through 0600 temp files, never kubectl's argv. apply_registry_secrets() { local dir dir="$(umask 077; mktemp -d)" remember_temp "$dir" printf '%s' "$REGISTRY_PLATFORM_TOKEN" > "${dir}/platform" printf '%s' "$REGISTRY_BUILD_TOKEN" > "${dir}/build" printf '%s' "$REGISTRY_PRUNE_TOKEN" > "${dir}/prune" printf '%s' build > "${dir}/username" kube -n "$CONTROL_NS" create secret generic felis-registry-auth \ --from-file=platform="${dir}/platform" \ --from-file=build="${dir}/build" \ --from-file=prune="${dir}/prune" \ --dry-run=client -o yaml | kube apply -f - kube -n "$BUILD_NS" create secret generic felis-registry-push \ --from-file=username="${dir}/username" \ --from-file=password="${dir}/build" \ --dry-run=client -o yaml | kube apply -f - rm -rf -- "$dir" } # node_global_cidrs prints one host-length CIDR per global address on this node. # Game server egress already excludes every private range; this adds the node's # public addresses, which would otherwise let a server dial the panel NodePort, # SSH, or anything else the host serves on them. node_global_cidrs() { command -v ip >/dev/null 2>&1 || return 0 ip -o addr show scope global 2>/dev/null | awk ' $3 == "inet" { split($4, a, "/"); print a[1] "/32" } $3 == "inet6" { split($4, a, "/"); print a[1] "/128" } ' | sort -u } as_postgres() { if command -v runuser >/dev/null 2>&1; then runuser -u postgres -- "$@" else sudo -u postgres "$@" fi } bootstrap_from_tui() { [ "${FELIS_BOOTSTRAP_FROM_TUI:-}" = "1" ] } pause_package_background_timers() { command -v systemctl >/dev/null 2>&1 || return 0 local active=0 timers=() services=() unit case "${PKG:-}" in apt) timers=("${APT_BACKGROUND_TIMERS[@]}"); services=("${APT_BACKGROUND_SERVICES[@]}") ;; dnf) timers=("${DNF_BACKGROUND_TIMERS[@]}"); services=("${DNF_BACKGROUND_SERVICES[@]}") ;; yum) timers=("${YUM_BACKGROUND_TIMERS[@]}"); services=("${YUM_BACKGROUND_SERVICES[@]}") ;; zypper) timers=("${ZYPPER_BACKGROUND_TIMERS[@]}"); services=("${ZYPPER_BACKGROUND_SERVICES[@]}") ;; *) return 0 ;; esac for unit in "${timers[@]}"; do if systemctl is-active --quiet "$unit"; then PKG_TIMERS_TO_RESTORE+=("$unit") active=1 fi done if [ "$active" -eq 1 ]; then log "pausing package-manager timers during bootstrap: ${PKG_TIMERS_TO_RESTORE[*]}" systemctl stop "${PKG_TIMERS_TO_RESTORE[@]}" || warn "could not stop package-manager timers; package operations may need to wait" fi for unit in "${services[@]}"; do if systemctl is-active --quiet "$unit"; then log "stopping package-manager background service during bootstrap: ${unit}" systemctl stop "$unit" || warn "could not stop ${unit}; package operations may need to wait" fi done } pkg_lock_files() { case "${PKG:-}" in apt) printf '%s\n' "${APT_LOCK_FILES[@]}" ;; dnf|yum) printf '%s\n' \ /var/lib/rpm/.rpm.lock \ /var/lib/dnf/rpmdb_lock.pid \ /var/cache/dnf/metadata_lock.pid \ /run/dnf.pid \ /var/run/dnf.pid ;; zypper) printf '%s\n' \ /var/lib/rpm/.rpm.lock \ /run/zypp.pid \ /var/run/zypp.pid ;; pacman) printf '%s\n' /var/lib/pacman/db.lck ;; esac } pkg_lock_process_names() { case "${PKG:-}" in dnf) printf '%s\n' dnf dnf5 rpm ;; yum) printf '%s\n' yum rpm ;; zypper) printf '%s\n' zypper rpm ;; pacman) printf '%s\n' pacman ;; esac } pkg_busy_pids() { local file file_count name { if command -v fuser >/dev/null 2>&1; then local files=() file_count=0 while IFS= read -r file; do if [ -e "$file" ]; then files+=("$file") file_count=$((file_count + 1)) fi done < <(pkg_lock_files) [ "$file_count" -eq 0 ] || fuser "${files[@]}" 2>/dev/null | tr ' ' '\n' fi if command -v pgrep >/dev/null 2>&1; then while IFS= read -r name; do [ -n "$name" ] && pgrep -x "$name" 2>/dev/null || true done < <(pkg_lock_process_names) fi } | awk 'NF && !seen[$1]++' } pkg_lock_busy() { [ -n "$(pkg_busy_pids)" ] } pkg_lock_holders() { local pids pids="$(pkg_busy_pids | paste -sd, - || true)" [ -n "$pids" ] || return 0 ps -o pid=,comm= -p "$pids" 2>/dev/null | awk '{$1=$1; print}' | paste -sd ';' - } wait_for_pkg_locks() { local deadline holders next_notice deadline=$((SECONDS + PKG_LOCK_TIMEOUT)) next_notice=0 while pkg_lock_busy; do if [ "$SECONDS" -ge "$next_notice" ]; then holders="$(pkg_lock_holders)" if [ -n "$holders" ]; then log "waiting for ${PKG} package locks to clear (timeout ${PKG_LOCK_TIMEOUT}s; holders: ${holders})" else log "waiting for ${PKG} package locks to clear (timeout ${PKG_LOCK_TIMEOUT}s)" fi next_notice=$((SECONDS + 30)) fi [ "$SECONDS" -lt "$deadline" ] || die "${PKG} package manager is still busy after ${PKG_LOCK_TIMEOUT}s; wait for the current package operation to finish, then retry" sleep 5 done } # NEEDRESTART_SUSPEND keeps Ubuntu's needrestart hook from restarting services after # the install's own apt runs. It would restart felis-velocity whenever a rerun # happens to install or upgrade a package (the running JVM maps files the rerun # replaces), dropping every player on a rerun that changed nothing the proxy runs. # The installer restarts what it changes itself. apt_get() { wait_for_pkg_locks NEEDRESTART_SUSPEND=1 DEBIAN_FRONTEND=noninteractive apt-get \ -o DPkg::Lock::Timeout="$PKG_LOCK_TIMEOUT" \ "$@" } validate_timeout() { local name="$1" value="$2" case "$value" in ''|*[!0-9]*) die "${name} must be a non-negative integer (seconds), got: ${value}" ;; esac } validate_nodeport() { local name="$1" value="$2" case "$value" in ''|*[!0-9]*) die "${name} must be a Kubernetes NodePort integer, got: ${value}" ;; esac if [ "$value" -lt 30000 ] || [ "$value" -gt 32767 ]; then die "${name} must be in Kubernetes NodePort range 30000-32767, got: ${value}" fi } # A value without a usable port would reach the firewall and the summary as-is: 8081 opens # port 8081 while nano binds nothing, 127.0.0.1 prints http://127.0.0.1:127.0.0.1/..., and # the unit crash-loops either way. validate_listen() { local name="$1" value="$2" port="${2##*:}" host="${2%:*}" case "$value" in *:*) ;; *) port="" ;; esac case "$port" in ''|*[!0-9]*) die "${name} must be host:port (for example 127.0.0.1:8081), got: ${value}" ;; esac [ "$port" -ge 1 ] && [ "$port" -le 65535 ] || die "${name} port must be 1-65535, got: ${value}" # Go takes a colon in the host only inside brackets; ::1:8081 would crash-loop the unit. case "$host" in *:*) case "$host" in "["*"]") ;; *) die "${name} needs an IPv6 host in brackets (for example [::1]:8081), got: ${value}" ;; esac ;; esac } # The value lands inside a firewalld rich rule, so anything but address characters and one # prefix length is refused here rather than handed to firewall-cmd. validate_cidr() { case "$2" in "") return 0 ;; *[!0-9A-Fa-f.:/]*|*/*/*|*/|/*) ;; */[0-9]*) return 0 ;; esac die "$1 must be an address with a prefix length (for example 10.0.0.7/32 or fd00::7/128), got: $2" } validate_settings() { validate_timeout PKG_LOCK_TIMEOUT "$PKG_LOCK_TIMEOUT" validate_timeout APT_LOCK_TIMEOUT "$APT_LOCK_TIMEOUT" validate_nodeport FELIS_PANEL_NODEPORT "$FELIS_PANEL_NODEPORT" validate_listen FELIS_NANO_LISTEN "$FELIS_NANO_LISTEN" validate_cidr FELIS_NANO_PROXY_CIDR "$FELIS_NANO_PROXY_CIDR" validate_offsite_settings case "$FELIS_GAME_STACK" in pinned|latest) ;; *) die "FELIS_GAME_STACK must be pinned or latest (got '${FELIS_GAME_STACK}')" ;; esac [ "$(heap_megabytes "$FELIS_VELOCITY_XMX")" -ge 256 ] \ || die "FELIS_VELOCITY_XMX must be a heap size of at least 256M, written M or G (got '${FELIS_VELOCITY_XMX}')" case "$FELIS_UPGRADE_DEPS" in 0|1) ;; *) die "FELIS_UPGRADE_DEPS must be 0 or 1 (got '${FELIS_UPGRADE_DEPS}')" ;; esac case "$FELIS_MANAGE_TIME_SYNC" in 0|1) ;; *) die "FELIS_MANAGE_TIME_SYNC must be 0 or 1 (got '${FELIS_MANAGE_TIME_SYNC}')" ;; esac case "$FELIS_MANAGE_JOURNAL" in 0|1) ;; *) die "FELIS_MANAGE_JOURNAL must be 0 or 1 (got '${FELIS_MANAGE_JOURNAL}')" ;; esac # Stripping the unit off a size with none leaves it whole, which the first arm catches. local journal_n="${FELIS_JOURNAL_MAX_USE%[KMG]}" case "$journal_n" in "$FELIS_JOURNAL_MAX_USE"|""|0*|*[!0-9]*) die "FELIS_JOURNAL_MAX_USE must be written K, M or G (got '${FELIS_JOURNAL_MAX_USE}')" ;; esac } # version_newer reports whether version $1 sorts after $2 (a leading v is ignored). version_newer() { local a="${1#v}" b="${2#v}" [ "$a" != "$b" ] && [ "$(printf '%s\n%s\n' "$a" "$b" | sort -V | tail -n 1)" = "$a" ] } # heap_megabytes prints a JVM heap size written M or G in megabytes, or 0 for any # other spelling. heap_megabytes() { local n="${1%?}" case "$n" in ""|*[!0-9]*|0*) echo 0; return ;; esac # Six digits of gigabytes is far past any host; the cap keeps the arithmetic in range. [ "${#n}" -le 6 ] || { echo 0; return; } case "$1" in *[Mm]) echo "$n" ;; *[Gg]) echo "$((n * 1024))" ;; *) echo 0 ;; esac } # validate_offsite_settings checks the FELIS_OFFSITE_* inputs before anything is # installed. They are written into felis.toml as TOML strings and the secrets into a # single-quoted env file, so a quote, a backslash or a line break is refused outright. validate_offsite_settings() { local name value for name in FELIS_OFFSITE_ENDPOINT FELIS_OFFSITE_BUCKET FELIS_OFFSITE_REGION FELIS_OFFSITE_PREFIX \ FELIS_OFFSITE_ACCESS_KEY FELIS_OFFSITE_SECRET_KEY FELIS_OFFSITE_KEY; do value="${!name:-}" case "$value" in *\"* | *\'* | *\\* | *$'\n'* | *$'\r'*) die "${name} must not contain quotes, backslashes or line breaks" ;; esac done if [ -z "$FELIS_OFFSITE_BUCKET" ]; then if [ -n "$FELIS_OFFSITE_ENDPOINT$FELIS_OFFSITE_REGION$FELIS_OFFSITE_PREFIX$FELIS_OFFSITE_DB_KEEP" ]; then die "FELIS_OFFSITE_* is set without FELIS_OFFSITE_BUCKET; name the bucket too" fi return 0 fi [ -n "$FELIS_OFFSITE_ENDPOINT" ] || die "FELIS_OFFSITE_BUCKET needs FELIS_OFFSITE_ENDPOINT (https://host[:port] of the S3-compatible store)" case "$FELIS_OFFSITE_BUCKET" in */* | *' '*) die "FELIS_OFFSITE_BUCKET is a bucket name; put a path inside it in FELIS_OFFSITE_PREFIX" ;; esac if [ -n "$FELIS_OFFSITE_DB_KEEP" ] && ! [[ "$FELIS_OFFSITE_DB_KEEP" =~ ^[1-9][0-9]*$ ]]; then die "FELIS_OFFSITE_DB_KEEP must be a positive number of bundles, got: ${FELIS_OFFSITE_DB_KEEP}" fi } # --------------------------------------------------------------------------- # 0. Privilege & host facts # --------------------------------------------------------------------------- if [ "$(id -u)" -ne 0 ]; then if [ -r "$0" ]; then log "re-executing under sudo" exec sudo -E bash "$0" "$@" fi die "must run as root (for a piped installer, use: curl -fsSL | sudo bash)" fi detect_os() { [ -r /etc/os-release ] || die "cannot read /etc/os-release; unsupported host" # shellcheck disable=SC1091 . /etc/os-release OS_ID="${ID:-unknown}" OS_VERSION="${VERSION_ID:-unknown}" OS_ID_LIKE="${ID_LIKE:-}" OS_CODENAME="${VERSION_CODENAME:-${UBUNTU_CODENAME:-}}" if command -v apt-get >/dev/null 2>&1; then PKG="apt" elif command -v dnf >/dev/null 2>&1; then PKG="dnf" elif command -v yum >/dev/null 2>&1; then PKG="yum" elif command -v zypper >/dev/null 2>&1; then PKG="zypper" elif command -v pacman >/dev/null 2>&1; then PKG="pacman" else die "no supported package manager (apt/dnf/yum/zypper/pacman) found on ${OS_ID} ${OS_VERSION}" fi log "host: ${PRETTY_NAME:-$OS_ID $OS_VERSION} (package manager: ${PKG})" } # persisted_root_domain echoes the root_domain an earlier run wrote, or nothing. The # generated toml is the only durable record of it: nothing else on the host stores the # domain, and it is written on every successful install. persisted_root_domain() { local f="${STATE_DIR}/felis.host.toml" [ -r "$f" ] || return 0 awk -F'"' '/^[[:space:]]*root_domain[[:space:]]*=/ { print $2; exit }' "$f" } detect_node_ip() { NODE_IP="$(ip -4 route get 1.1.1.1 2>/dev/null | awk '{for(i=1;i<=NF;i++) if($i=="src"){print $(i+1); exit}}')" [ -n "${NODE_IP:-}" ] || NODE_IP="$(hostname -I 2>/dev/null | awk '{print $1}')" [ -n "${NODE_IP:-}" ] || die "could not determine this host's primary IPv4 address" # Precedence: an explicit FELIS_ROOT_DOMAIN, then whatever the last run persisted, then # the nip.io default. The middle step is what makes a re-run idempotent. Without it this # installer re-derived the domain from scratch every time and defaulted to nip.io, so # re-running it on a live install -- the only way to move felis-api to a newer release, # and what `felis update` points operators at -- rewrote root_domain, panel_hostname and # admin_hostname to nip.io names while the write-once panel certificate kept the old # ones. Secrets never had this problem: load_or_make_secrets has always sourced # secrets.env before generating anything. # # A different FELIS_ROOT_DOMAIN on an installed host is refused. The name is on more # surfaces than this script rewrites -- the panel certificate, the felis-api-tls and # felis-config Secrets, the proxy's felis-link.properties, the login gate's CR env, the # Cloudflare tunnel -- and a half-moved install serves a certificate for the old names. # `felis domain set` moves all of them and `felis domain check` proves each one. local persisted persisted="$(persisted_root_domain)" if [ -n "${FELIS_ROOT_DOMAIN:-}" ] && [ -n "$persisted" ] && [ "$FELIS_ROOT_DOMAIN" != "$persisted" ]; then die "FELIS_ROOT_DOMAIN (${FELIS_ROOT_DOMAIN}) differs from the installed ${persisted}. The installer keeps the installed domain; to move the install, rerun it without FELIS_ROOT_DOMAIN, then run: sudo felis domain set ${FELIS_ROOT_DOMAIN} (docs/operations.md, Changing the root domain)" fi FELIS_ROOT_DOMAIN="${FELIS_ROOT_DOMAIN:-${persisted:-${NODE_IP}.nip.io}}" if [ -n "$persisted" ] && [ "$FELIS_ROOT_DOMAIN" = "$persisted" ]; then log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN} (reusing the installed domain)" else log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN}" fi } # warn_dynamic_node_ip warns when NODE_IP is a DHCP lease. The address is written into # the database connection string, pg_hba, the network policies, the panel certificate # and the default nip.io domain, and nothing re-addresses a live install, so a lease # that later comes back different takes the whole platform down (the watchdog then # reports host-address). `ip -o addr` marks a leased address "dynamic". warn_dynamic_node_ip() { if ip -4 -o addr show 2>/dev/null | awk -v ip="$NODE_IP" ' { split($4, a, "/"); if (a[1] == ip && / dynamic /) found = 1 } END { exit !found }'; then warn "${NODE_IP} is a DHCP lease, and the install is bound to this address. Give the host a" warn "DHCP reservation or a static address before it changes (docs/operations.md §1)." fi } pkg_install() { case "$PKG" in apt) apt_get install -y "$@" ;; dnf) wait_for_pkg_locks; dnf install -y "$@" ;; yum) wait_for_pkg_locks; yum install -y "$@" ;; zypper) wait_for_pkg_locks; zypper --non-interactive install -y "$@" ;; pacman) wait_for_pkg_locks; pacman -S --noconfirm --needed "$@" ;; esac } pkg_refresh_once() { [ -n "${_PKG_REFRESHED:-}" ] && return 0 case "$PKG" in apt) apt_get update -y ;; dnf|yum) : ;; # dnf/yum refresh metadata on demand zypper) wait_for_pkg_locks; zypper --non-interactive refresh ;; # Arch supports only whole-system upgrades (-Sy alone leaves a partial upgrade), so the # refresh stays -Syu, with PostgreSQL held back once a cluster exists: a new major # version cannot open the old data directory, and the upgrade would take the platform's # database down on an unrelated rerun. check_postgres_major explains the way forward. pacman) wait_for_pkg_locks if [ -f "$(postgres_data_dir)/PG_VERSION" ]; then pacman -Syu --noconfirm --ignore postgresql else pacman -Syu --noconfirm fi ;; esac _PKG_REFRESHED=1 } # --------------------------------------------------------------------------- # 1. Swap — k3s + Postgres + a Go build will OOM on a <2 GiB box without it # --------------------------------------------------------------------------- ensure_swap() { local mem_kb swap_kb mem_kb="$(awk '/^MemTotal:/{print $2}' /proc/meminfo)" swap_kb="$(awk '/^SwapTotal:/{print $2}' /proc/meminfo)" if [ "${swap_kb:-0}" -gt 0 ]; then ok "swap already present ($((swap_kb/1024)) MiB)" return 0 fi if [ "${mem_kb:-0}" -ge 2097152 ]; then ok "RAM $((mem_kb/1024)) MiB is sufficient; skipping swap" return 0 fi log "low RAM ($((mem_kb/1024)) MiB) and no swap — creating a 2 GiB swap file" if ! fallocate -l 2G /swapfile 2>/dev/null; then dd if=/dev/zero of=/swapfile bs=1M count=2048 status=none fi chmod 600 /swapfile mkswap /swapfile >/dev/null swapon /swapfile grep -q '^/swapfile ' /etc/fstab || echo '/swapfile none swap sw 0 0' >> /etc/fstab ok "2 GiB swap active" } # ensure_time_sync turns NTP on. A drifting clock breaks things far from their cause: # sign-in codes and sessions expire early or late, S3 refuses off-site uploads signed # more than 15 minutes off, and certificate checks fail. Rocky's minimal image ships # chronyd disabled (the test host reported NTP=no), so the installer enables whatever # timedatectl manages and installs chrony only when there is nothing to enable. It # waits half a minute for the first synchronization and then carries on: the watchdog # keeps reporting an unsynchronized clock. ensure_time_sync() { if [ "$FELIS_MANAGE_TIME_SYNC" = 0 ]; then log "FELIS_MANAGE_TIME_SYNC=0: leaving time synchronization to the operator" return 0 fi if ! command -v timedatectl >/dev/null 2>&1; then warn "timedatectl not found; make sure an NTP client keeps this host's clock (docs/operations.md §1)" return 0 fi if [ "$(timedatectl show -p NTP --value 2>/dev/null)" != yes ]; then # set-ntp fails with "NTP not supported" when no NTP unit is installed at all # (Debian's minimal image splits systemd-timesyncd into its own package). if ! timedatectl set-ntp true 2>/dev/null; then log "no NTP client to enable; installing chrony" pkg_install chrony if ! timedatectl set-ntp true; then warn "could not turn NTP on; set up time synchronization by hand (docs/troubleshooting.md §13c)" return 0 fi fi log "turned NTP time synchronization on" fi local _ for _ in $(seq 1 15); do if [ "$(timedatectl show -p NTPSynchronized --value 2>/dev/null)" = yes ]; then ok "system clock synchronized by NTP" return 0 fi sleep 2 done warn "the system clock is not synchronized yet; check 'timedatectl' (docs/troubleshooting.md §13c)" } # ensure_persistent_journal keeps the system journal across reboots. Rocky's journald # stores it under /run unless /var/log/journal exists, and its minimal image does not # create that directory, so a reboot (the moment an operator most needs to know what # came before it) erased every log. journald creates JOURNAL_DIR itself once it runs # with Storage=persistent, so the directory is the proof the drop-in took: journald is # restarted when the drop-in changed, or when it is current and the directory is still # missing. ensure_persistent_journal() { if [ "$FELIS_MANAGE_JOURNAL" = 0 ]; then log "FELIS_MANAGE_JOURNAL=0: leaving journald as it is" return 0 fi local file="$JOURNALD_DROPIN" tmp mkdir -p "$(dirname "$file")" # Made beside its destination so the file is born with that directory's SELinux # label. A file made under /tmp keeps user_tmp_t through the mv, and journald was # denied it on the test host ("Failed to open configuration file ... Permission # denied"). journald reads only *.conf, so the temporary name is never loaded. tmp="$(mktemp "${file}.XXXXXX")" remember_temp "$tmp" printf '[Journal]\nStorage=persistent\nSystemMaxUse=%s\n' "$FELIS_JOURNAL_MAX_USE" > "$tmp" if [ -f "$file" ] && cmp -s "$tmp" "$file"; then rm -f "$tmp" if [ -d "$JOURNAL_DIR" ]; then ok "system journal already persistent (capped at ${FELIS_JOURNAL_MAX_USE})" return 0 fi log "journald has not taken up ${file}; restarting it" else chmod 0644 "$tmp" mv "$tmp" "$file" fi restore_label "$file" systemctl restart systemd-journald journalctl --flush >/dev/null 2>&1 || true if [ -d "$JOURNAL_DIR" ]; then ok "system journal is persistent under ${JOURNAL_DIR} (capped at ${FELIS_JOURNAL_MAX_USE})" else warn "journald did not create ${JOURNAL_DIR}, so logs still end at a reboot; see: journalctl -u systemd-journald" fi } # --------------------------------------------------------------------------- # 2. Base packages # --------------------------------------------------------------------------- install_base() { local packages=(ca-certificates openssl) pkg_refresh_once command -v curl >/dev/null 2>&1 || packages+=(curl) command -v tar >/dev/null 2>&1 || packages+=(tar) if ! bootstrap_from_tui && ! command -v git >/dev/null 2>&1; then packages+=(git) fi pkg_install "${packages[@]}" ok "base tools present" } install_cloudflared() { local current="" path if path="$(command -v cloudflared 2>/dev/null)"; then current="$(cloudflared --version 2>/dev/null | awk '{ for (i = 1; i < NF; i++) if ($i == "version") { print $(i + 1); exit } }')" if [ "$current" = "$FELIS_CLOUDFLARED_VERSION" ]; then ok "cloudflared ${current} already installed" return 0 fi if [ "$FELIS_UPGRADE_DEPS" != 1 ]; then ok "cloudflared ${current:-(version unreadable)} already installed; this release pins ${FELIS_CLOUDFLARED_VERSION} (FELIS_UPGRADE_DEPS=1 moves it)" return 0 fi if [ "$path" != "$CLOUDFLARED_BIN" ]; then warn "cloudflared at ${path} was not installed by Felis; upgrade it the way it was installed" return 0 fi if [ -n "$current" ] && version_newer "$current" "$FELIS_CLOUDFLARED_VERSION"; then ok "cloudflared ${current} is newer than the pinned ${FELIS_CLOUDFLARED_VERSION}; left as it is" return 0 fi fi local machine arch url tmp want have machine="$(uname -m)" case "$machine" in x86_64|amd64) arch="amd64"; want="$CLOUDFLARED_PINNED_SHA256_AMD64" ;; aarch64|arm64) arch="arm64"; want="$CLOUDFLARED_PINNED_SHA256_ARM64" ;; armv7l|armv6l) arch="arm"; want="$CLOUDFLARED_PINNED_SHA256_ARM" ;; *) die "unsupported architecture for cloudflared: ${machine}" ;; esac [ "$FELIS_CLOUDFLARED_VERSION" = "$CLOUDFLARED_PINNED_VERSION" ] || want="$FELIS_CLOUDFLARED_SHA256" [ -n "$want" ] || die "no pinned sha256 for cloudflared ${FELIS_CLOUDFLARED_VERSION}; set FELIS_CLOUDFLARED_SHA256 to the digest of cloudflared-linux-${arch} on that GitHub release" url="https://github.com/cloudflare/cloudflared/releases/download/${FELIS_CLOUDFLARED_VERSION}/cloudflared-linux-${arch}" tmp="$(mktemp)" remember_temp "$tmp" log "installing cloudflared ${FELIS_CLOUDFLARED_VERSION} (${arch})" curl -fsSL --retry 5 --retry-delay 2 "$url" -o "$tmp" have="$(sha256sum <"$tmp" | cut -d' ' -f1)" if [ "$have" != "$(printf '%s' "$want" | tr 'A-Z' 'a-z')" ]; then rm -f "$tmp" die "cloudflared-linux-${arch} ${FELIS_CLOUDFLARED_VERSION} hashes to ${have}, expected ${want}; refusing to install it" fi install -m 0755 "$tmp" "$CLOUDFLARED_BIN" rm -f "$tmp" ok "cloudflared installed ($(cloudflared --version | head -n 1))" # The running tunnel keeps the old binary mapped until it restarts. if [ -n "$current" ] && systemctl is-active --quiet cloudflared-felis 2>/dev/null; then systemctl restart cloudflared-felis ok "cloudflared-felis restarted onto ${FELIS_CLOUDFLARED_VERSION}" fi } # --------------------------------------------------------------------------- # 3. Docker (used only to build & export the felis image; k3s uses containerd) # --------------------------------------------------------------------------- docker_apt_repo_os() { case "$OS_ID" in debian|ubuntu) printf '%s\n' "$OS_ID" ;; *) die "Docker apt repository is not configured for ${OS_ID} ${OS_VERSION}" ;; esac } install_docker_apt() { local arch keyring repo_os repo_os="$(docker_apt_repo_os)" [ -n "$OS_CODENAME" ] || die "cannot determine apt codename for ${OS_ID} ${OS_VERSION}" arch="$(dpkg --print-architecture)" keyring="/etc/apt/keyrings/docker.asc" log "installing docker apt repository" install -m 0755 -d /etc/apt/keyrings curl -fsSL "https://download.docker.com/linux/${repo_os}/gpg" -o "$keyring" \ || die "failed to download Docker GPG key for ${repo_os}" chmod a+r "$keyring" cat > /etc/apt/sources.list.d/docker.list </dev/null 2>&1; then ok "docker already installed" else case "$PKG" in apt) install_docker_apt ;; dnf|yum) install_docker_rpm ;; zypper) install_docker_zypper ;; pacman) install_docker_pacman ;; *) die "Docker installation is not supported with package manager: ${PKG}" ;; esac fi systemctl enable --now docker ok "docker running" } # --------------------------------------------------------------------------- # 4. k3s — single node, trimmed for RAM. NetworkPolicy stays ENABLED on purpose: # Felis's minecraft fence (default-deny + allow-rcon/allow-game) is a core # security claim, so we must NOT pass --disable-network-policy. # On SUSE-family hosts firewalld ships active by default; open the required # rules rather than disabling the firewall. # --------------------------------------------------------------------------- configure_k3s_firewall() { command -v firewall-cmd >/dev/null 2>&1 || return 0 systemctl is-active --quiet firewalld || return 0 log "configuring firewalld for k3s" firewall-cmd --permanent --add-port=6443/tcp firewall-cmd --permanent --add-port="${FELIS_PANEL_NODEPORT}/tcp" firewall-cmd --permanent --zone=trusted --add-source="$POD_CIDR" firewall-cmd --permanent --zone=trusted --add-source="$SERVICE_CIDR" firewall-cmd --reload } install_k3s() { configure_k3s_firewall # Before the installer runs: a fresh k3s reads the drop-in on its first start. K3S_RESTART_NEEDED=0 write_k3s_config local installer_ran=0 if [ -x "$K3S_BIN" ]; then local current current="$("$K3S_BIN" --version 2>/dev/null | awk 'NR == 1 { print $3 }')" if [ "$current" = "$FELIS_K3S_VERSION" ]; then ok "k3s ${current} already installed at ${K3S_BIN}" elif [ "$FELIS_UPGRADE_DEPS" != 1 ]; then ok "k3s ${current:-(version unreadable)} already installed at ${K3S_BIN}; this release pins ${FELIS_K3S_VERSION} (FELIS_UPGRADE_DEPS=1 moves it)" elif k3s_upgrade_allowed "$current" "$FELIS_K3S_VERSION"; then log "upgrading k3s ${current} to ${FELIS_K3S_VERSION}; running pods keep running while it restarts" run_k3s_installer installer_ran=1 fi else log "installing k3s ${FELIS_K3S_VERSION} into ${K3S_BIN_DIR} (no traefik/servicelb/metrics-server)" run_k3s_installer installer_ran=1 fi [ -x "$K3S_BIN" ] || die "k3s installation completed but ${K3S_BIN} is missing" strip_k3s_kubeconfig_mode_flag systemctl enable --now k3s # The installer restarts k3s itself; otherwise a changed drop-in or unit takes a # restart to load. Pods keep running across it (k3s leaves the containers be). if [ "$K3S_RESTART_NEEDED" = 1 ] && [ "$installer_ran" = 0 ]; then log "restarting k3s to load its new settings (${K3S_CONFIG_DROPIN})" systemctl restart k3s fi export KUBECONFIG="$K3S_KUBECONFIG" log "waiting for the node to become Ready" wait_for_node_ready # k3s applies write-kubeconfig-mode as it writes the file; this covers a k3s that # has not rewritten it since the mode changed. chmod 0600 "$K3S_KUBECONFIG" } # k3s_node_name prints the name this node must keep. Every local-path volume (worlds, # registry, uploads, backups) is bound to its node by name, and k3s takes the name from # the hostname on every start, so a renamed host came back as a second, empty node with # every volume stuck Pending on the old one. Precedence: the name already pinned; else # the name the node registered under, which its kubelet client certificate carries as # system:node: and which is readable with k3s stopped; else, where k3s has never # run, the lowercased hostname k3s itself would pick. It prints nothing when k3s has run # but neither source can be read: pinning a guess there would rename the node. k3s_node_name() { local name="" if [ -f "$K3S_CONFIG_DROPIN" ]; then name="$(awk -F'"' '/^node-name:/ { print $2; exit }' "$K3S_CONFIG_DROPIN")" fi if [ -z "$name" ] && [ -f "$K3S_KUBELET_CERT" ]; then # OpenSSL 3 prints "CN=system:node:x", 1.1 "CN = system:node:x", older "/CN=...". # An unreadable certificate leaves the name empty (and warned about), not a failed run. name="$(openssl x509 -in "$K3S_KUBELET_CERT" -noout -subject 2>/dev/null | sed -n 's/.*CN *= *system:node:\([^,/]*\).*/\1/p' || true)" elif [ -z "$name" ] && [ ! -x "$K3S_BIN" ]; then name="$(uname -n | tr '[:upper:]' '[:lower:]')" fi printf '%s' "$name" } # write_k3s_config writes the installer's k3s settings to K3S_CONFIG_DROPIN and sets # K3S_RESTART_NEEDED when they changed, since k3s reads the file only as it starts. # write-kubeconfig-mode keeps the admin kubeconfig root-only: it is cluster-admin, and # installs before this passed 644, which let every local account read it, felis-velocity # (the account the internet-facing proxy runs as) included. write_k3s_config() { local file="$K3S_CONFIG_DROPIN" name tmp name="$(k3s_node_name)" mkdir -p "$(dirname "$file")" # Beside its destination for the directory's SELinux label (ensure_persistent_journal # has the story); k3s loads only *.yaml and *.yml from the directory. tmp="$(mktemp "${file}.XXXXXX")" remember_temp "$tmp" { echo "# Written by the Felis installer (deploy/bootstrap.sh); a rerun rewrites it." echo 'write-kubeconfig-mode: "0600"' if [ -n "$name" ]; then printf 'node-name: "%s"\n' "$name" fi } > "$tmp" if [ -z "$name" ]; then warn "could not read this node's k3s name, so it is not pinned; a hostname change would orphan every volume" fi if [ -f "$file" ] && cmp -s "$tmp" "$file"; then rm -f "$tmp" restore_label "$file" ok "k3s settings already current${name:+ (node name ${name})}" return 0 fi chmod 0600 "$tmp" mv "$tmp" "$file" K3S_RESTART_NEEDED=1 log "wrote ${file}${name:+ (node name pinned to ${name})}" } # A command-line flag outranks every config file, and k3s's installer writes # INSTALL_K3S_EXEC into the unit's ExecStart one quoted word per line, so installs from # before the drop-in keep "'--write-kubeconfig-mode' \" followed by "'644' \" there. # This drops both lines (or the single --write-kubeconfig-mode= spelling). An # upgrade through run_k3s_installer rewrites the unit without them anyway. strip_k3s_kubeconfig_mode_flag() { local unit="$K3S_UNIT_FILE" tmp [ -f "$unit" ] && grep -q -- '--write-kubeconfig-mode' "$unit" || return 0 tmp="$(mktemp)" remember_temp "$tmp" awk ' skip { skip = 0; next } index($0, "--write-kubeconfig-mode") { if (index($0, "=") == 0) skip = 1; next } { print } ' "$unit" > "$tmp" # Rewritten in place, so the unit keeps its owner, mode and SELinux label. cat "$tmp" > "$unit" rm -f "$tmp" systemctl daemon-reload K3S_RESTART_NEEDED=1 log "dropped --write-kubeconfig-mode from ${unit}; the admin kubeconfig becomes root-only" } # The script from the release's own tag rather than get.k3s.io, which serves whatever # master holds today. '+' is literal in a URL path, so the tag needs no escaping. On an # installed k3s the same script replaces the binary in place and restarts the service. run_k3s_installer() { curl -sfL --retry 5 --retry-delay 2 "https://raw.githubusercontent.com/k3s-io/k3s/${FELIS_K3S_VERSION}/install.sh" | \ INSTALL_K3S_VERSION="$FELIS_K3S_VERSION" \ INSTALL_K3S_BIN_DIR="$K3S_BIN_DIR" \ INSTALL_K3S_EXEC="--disable traefik --disable servicelb --disable metrics-server" \ sh - } # k3s_upgrade_allowed decides whether an installed k3s ($1) may move to $2. Kubernetes # supports upgrading one minor version at a time, so a larger jump stops the install # before anything changed; a newer installed k3s is left as it is. k3s_upgrade_allowed() { local current="$1" want="$2" cur_major cur_minor want_major want_minor rest IFS=. read -r cur_major cur_minor rest <<<"${current#v}" IFS=. read -r want_major want_minor rest <<<"${want#v}" case "${cur_major}${cur_minor}${want_major}${want_minor}" in ""|*[!0-9]*) die "cannot compare the installed k3s '${current}' with ${want}; upgrade it by hand (docs/operations.md §4)" ;; esac if version_newer "$current" "$want"; then ok "k3s ${current} is newer than the pinned ${want}; left as it is" return 1 fi if [ "$cur_major" != "$want_major" ] || [ "$((want_minor - cur_minor))" -gt 1 ]; then die "k3s ${current} -> ${want} skips a minor version, and Kubernetes upgrades one minor at a time. Rerun with FELIS_K3S_VERSION set to the newest v${cur_major}.$((cur_minor + 1)).x+k3sN release first (https://github.com/k3s-io/k3s/releases)" fi return 0 } # Waits for the (single) node to report Ready. Shared by the k3s install and the # registry-mirror restart below: both restart the agent, and a bootstrap that # proceeds early fails later with a misleading "not found"/timeout instead. wait_for_node_ready() { local _ for _ in $(seq 1 60); do if kube get nodes 2>/dev/null | grep -q ' Ready '; then ok "k3s node Ready" return 0 fi sleep 5 done kube get nodes || true die "k3s node did not become Ready in time" } # --------------------------------------------------------------------------- # 4b. The in-cluster registry: the node-side pull path, the registry's own # image, and the hosting of every image this installer builds. # # Kubelet's image GC collects an unused image under disk pressure (drilled: # the game images were collected and ImagePullBackOff had nothing to pull # from). The fix is a pull source that is always there — the registry the # bundle already renders. Two node-level facts make that work: # * kubelet cannot reach the registry Service VIP (the live stack answered # "Empty reply"), so containerd is told to go through the loopback # hostPort the registry Deployment binds (the Deployment renders it) — # that is configure_registry_mirror below; # * the registry's own image (REGISTRY_IMAGE) must already be in containerd # before the registry Deployment can start at all — # import_registry_image below caches it. # After deploy_bundle, push_images_to_registry mirrors the built images into # the registry, so containerd's imported copies are a first-boot cache # rather than the only copy. # --------------------------------------------------------------------------- # The node's containerd cannot dial the registry Service VIP, so pulls arrive # over the loopback hostPort the registry Deployment binds. k3s reads this file # when the agent starts and regenerates containerd's certs.d from it — no # restart, no effect — so a CONTENT change restarts k3s; an identical file # (every re-run) restarts nothing. K3S_REGISTRIES_FILE is a variable so # bootstrap_test.sh can point the function at a scratch file. configure_registry_mirror() { local file="$K3S_REGISTRIES_FILE" tmp tmp="$(mktemp)" remember_temp "$tmp" cat > "$tmp" < http://${REGISTRY_PUSH_HOST})" return 0 fi mkdir -p "$(dirname "$file")" mv "$tmp" "$file" log "restarting k3s to load the registry mirror (${REGISTRY_URL} -> http://${REGISTRY_PUSH_HOST})" systemctl restart k3s export KUBECONFIG=/etc/rancher/k3s/k3s.yaml wait_for_node_ready } # The registry Deployment runs REGISTRY_IMAGE (platform.defaultRegistryImage; the # renderer's default — this script never passes --registry-image). On a box # that cannot reach Docker Hub the Deployment can never start without a local # copy, so the installer caches one whenever it can. Best-effort by design: if # the pull fails the registry rollout still fails loudly at deploy_bundle, with # the regular diagnostics — but for every box that CAN pull, the image is # fetched exactly once, here, instead of at first pod start. # # crictl pulls through CRI, the same call kubelet makes, so containerd records the # digest ref the Deployment names and kubelet finds it. A docker save/import round # trip rewrites the manifest and would leave a copy the digest ref never matches. import_registry_image() { local images # Read the list whole before matching; see postgres_installed for the SIGPIPE. images="$(k3s_cmd ctr images ls -q 2>/dev/null || true)" if grep -qxF "$(registry_image_containerd_ref)" <<<"$images"; then ok "registry image ${REGISTRY_IMAGE} already in k3s containerd" return 0 fi log "pulling the registry's own image (${REGISTRY_IMAGE}) into k3s containerd" if k3s_cmd crictl pull "$REGISTRY_IMAGE" >/dev/null; then ok "registry image ${REGISTRY_IMAGE} pulled" else warn "could not pull ${REGISTRY_IMAGE}: the in-cluster registry will start only if the node can pull it from Docker Hub; on an air-gapped box import it by hand (docs/troubleshooting.md §8e)" fi } # registry_image_containerd_ref prints the name containerd lists REGISTRY_IMAGE under # once CRI has pulled it: the repository and the digest, with the tag dropped. registry_image_containerd_ref() { printf '%s@%s\n' "${REGISTRY_IMAGE%%:*}" "${REGISTRY_IMAGE#*@}" } # The registry pod runs REGISTRY_IMAGE and, as its registry-gate sidecar, the felis # image — neither of which can be pulled from the registry they make up. A kubelet # image GC that collected either would leave the registry, and every pull through # it, dead until someone re-imported by hand. containerd reports an image labelled # io.cri-containerd.pinned=pinned as pinned over CRI, and kubelet's image GC never # removes a pinned image. Older felis/felis tags are unpinned first, so upgrades # do not pile up pinned images forever. pin_registry_images() { local ref registry_ref registry_ref="$(registry_image_containerd_ref)" while read -r ref; do case "$ref" in "$FELIS_IMAGE"|"$registry_ref") ;; */felis/felis:*|docker.io/library/registry[:@]*) k3s_cmd ctr images label "$ref" io.cri-containerd.pinned= >/dev/null 2>&1 || true ;; esac done < <(k3s_cmd ctr images ls -q 2>/dev/null || true) for ref in "$FELIS_IMAGE" "$registry_ref"; do if k3s_cmd ctr images label "$ref" io.cri-containerd.pinned=pinned >/dev/null 2>&1; then ok "pinned ${ref} in containerd (exempt from kubelet image GC)" else warn "could not pin ${ref} in containerd: if the kubelet's image GC collects it, the registry pod cannot restart until it is re-imported (docs/troubleshooting.md §8e)" fi done } # pin_user_server_images fixes every user server still naming a tag in the platform # registry (felis/paper:demo) to the digest that tag names now. It has to run before # build_game_stack and push_images_to_registry put new builds under those tags: a # server left on the bare tag would boot the new build on its next wake and open its # world with a newer Minecraft version, and chunk upgrades cannot be undone. felis-api # pins every server it creates; this catches the ones created before it did. A fresh # install has no CRD, so nothing to pin; a running server restarts once onto the # build it already runs. pin_user_server_images() { kube get crd minecraftservers.felis.lolicon.best >/dev/null 2>&1 || return 0 # The registry answers the lookups, and a k3s restart above may have left its pod # still starting. A registry that never comes up fails the lookups below, loudly. kube -n "$CONTROL_NS" rollout status deployment/registry --timeout=180s >/dev/null 2>&1 || true if "$HOST_BIN" pin-images --namespace "$MINECRAFT_NS" \ --registry "$REGISTRY_URL" --endpoint "$REGISTRY_PUSH_HOST"; then ok "user servers pinned to the builds they run" else warn "could not pin every user server listed above to its current build: each one still names a tag this run is about to point at a new build, so its next start may open its world with a newer Minecraft version. Pin it before starting it again: sudo felis pin-images (docs/troubleshooting.md §15b)" fi } # --------------------------------------------------------------------------- # 5. Source/binary + image build + containerd import # --------------------------------------------------------------------------- # repo_slug prints the "owner/name" of FELIS_REPO_URL, for the REST API. repo_slug() { printf '%s\n' "$FELIS_REPO_URL" | sed -e 's#^.*github\.com[:/]##' -e 's#\.git$##' } # github_api GETs a REST path and prints the body. # # The token goes in through `curl --config -` rather than `-H "Authorization: ..."` # because argv is world-readable via /proc while this runs. Same reason git_auth uses a # credential helper instead of a URL: a bootstrap that leaks its own credential to any # local user has not really installed anything privately. github_api() { local url="https://api.github.com/$1" local ua="felis-bootstrap (+${FELIS_REPO_URL})" if [ -n "$FELIS_GITHUB_TOKEN" ]; then printf 'header = "Authorization: Bearer %s"\n' "$FELIS_GITHUB_TOKEN" \ | curl -fsSL --retry 5 --retry-delay 2 --config - \ -A "$ua" -H "Accept: application/vnd.github+json" "$url" else curl -fsSL --retry 5 --retry-delay 2 \ -A "$ua" -H "Accept: application/vnd.github+json" "$url" fi } # github_latest_tag prints the tag of the newest published stable release, or fails. # Same endpoint internal/updater/github.go polls, so `felis update` and the installer # can never disagree about what "latest" means. github_latest_tag() { local json tag # Fetch first, filter second — the SIGPIPE reason documented on resolve_latest_game_jars. json="$(github_api "repos/$(repo_slug)/releases/latest")" || return 1 tag="$(printf '%s' "$json" | grep -o '"tag_name"[[:space:]]*:[[:space:]]*"[^"]*"' || true)" tag="${tag%%$'\n'*}" tag="${tag#*:}" # "v1.2.3" tag="${tag#*\"}" # v1.2.3" tag="${tag%\"}" # v1.2.3 [ -n "$tag" ] || return 1 printf '%s\n' "$tag" } # felis_asset_arch prints the release-asset suffix for this host, or fails. Unlike the # cloudflared/JRE/Go mappers just below, this one does NOT die on an unmapped architecture: # those have no local alternative, whereas a missing prebuilt binary only means we compile # the same tag on this host — slower, byte-for-byte the same result. felis_asset_arch() { case "$(uname -m)" in x86_64|amd64) printf 'amd64\n' ;; aarch64|arm64) printf 'arm64\n' ;; *) return 1 ;; esac } # github_asset_id prints the numeric id of the asset named $2 on release tag $1. # # Two details here are load-bearing and both are wrong in the obvious version. The newline # flatten comes FIRST: api.github.com pretty-prints, so "name" and "url" land on different # lines and splitting on "{" alone matches nothing at all. And the id is read out of the # asset's own url, never from a bare "id" key — the payload carries a release id, an author # id, one id per asset AND one per nested uploader, so matching "id" silently resolves the # wrong file. Anchoring on the CLOSING quote keeps felis-linux-amd64 from also matching a # sibling like felis-linux-amd64.sha256. # # The segment this greps ends at the nested "uploader" object, which is fine because "url" # precedes it. Anything published after uploader (size, digest, browser_download_url) is NOT # reachable this way — fetch releases/assets/ with the JSON Accept if that is ever needed. github_asset_id() { local tag="$1" name="$2" json id # Fetch first, filter second — the SIGPIPE reason documented on resolve_latest_game_jars. json="$(github_api "repos/$(repo_slug)/releases/tags/${tag}")" || return 1 id="$(printf '%s' "$json" | tr -d '\n' | tr '{' '\n' \ | grep "\"name\":[[:space:]]*\"${name}\"" \ | grep -o 'releases/assets/[0-9]\{1,\}' | head -1)" || true id="${id##*/}" [ -n "$id" ] || return 1 printf '%s\n' "$id" } # download_release_asset fetches ONE asset of a published release into $3. # # It RETURNS non-zero rather than dying: every caller falls back to building the same tag # from source, so a tag whose release workflow has not finished uploading yet — a real window, # since that job runs vet, tests and a full image build first — still installs. # # This cannot ride the github_api helper above. That helper sends # Accept: application/vnd.github+json, and with it this endpoint returns the asset's METADATA # as JSON under HTTP 200: a "successful" download of a text blob that passes every check and # only surfaces much later, inside docker build, as an exec format error. # # Plain -L, never --location-trusted. GitHub 302s to a DIFFERENT host whose signed URL carries # its own credentials in the query string; curl drops Authorization across that hop, and the # drop is REQUIRED — forwarding the token makes the storage backend reject the request with # 400 while leaking the credential to a third host for nothing. # # -f is not cosmetic: without it curl writes GitHub's error JSON into the output file and # still exits 0. download_release_asset() { local tag="$1" name="$2" dest="$3" id ua url rc=0 ua="felis-bootstrap (+${FELIS_REPO_URL})" id="$(github_asset_id "$tag" "$name")" || return 1 url="https://api.github.com/repos/$(repo_slug)/releases/assets/${id}" log "downloading ${name} from release ${tag}" if [ -n "$FELIS_GITHUB_TOKEN" ]; then printf 'header = "Authorization: Bearer %s"\n' "$FELIS_GITHUB_TOKEN" \ | curl -fsSL --retry 5 --retry-delay 2 --config - \ -A "$ua" -H "Accept: application/octet-stream" -o "$dest" "$url" || rc=$? else curl -fsSL --retry 5 --retry-delay 2 \ -A "$ua" -H "Accept: application/octet-stream" -o "$dest" "$url" || rc=$? fi # curl -f leaves a PARTIAL file behind when a transfer dies mid-stream, so a failed # download must not hand the caller something it could mistake for a complete one. if [ "$rc" -ne 0 ] || [ ! -s "$dest" ]; then rm -f "$dest" return 1 fi } # verify_release_checksum checks the downloaded asset $2 (at $3) against the SHA256SUMS # file release.yml publishes next to it on tag $1. It returns non-zero, with a warning, # when the release has no SHA256SUMS (a tag cut before release.yml wrote one, or a release # still uploading), when the file does not list the asset, or when the hash differs. verify_release_checksum() { local tag="$1" name="$2" file="$3" sums want have sums="$(mktemp)" remember_temp "$sums" if ! download_release_asset "$tag" SHA256SUMS "$sums"; then rm -f "$sums" warn "release ${tag} publishes no SHA256SUMS, so ${name} cannot be verified; building ${tag} from source on this host instead" return 1 fi # sha256sum's text-mode line is " ", binary mode " *". want="$(awk -v n="$name" '$2 == n || $2 == "*" n { print $1; exit }' "$sums")" rm -f "$sums" if [ -z "$want" ]; then warn "release ${tag}'s SHA256SUMS does not list ${name}; building ${tag} from source on this host instead" return 1 fi # Hash stdin, never the path: the same sha256sum escaping install_via_plugins avoids. have="$(sha256sum <"$file" | cut -d' ' -f1)" if [ "$have" != "$want" ]; then warn "downloaded ${name} hashes to ${have}, but release ${tag}'s SHA256SUMS says ${want}; discarding it and building ${tag} from source on this host instead" return 1 fi ok "${name} matches release ${tag}'s SHA256SUMS" } # use_release_binary reports whether this run should install a prebuilt binary instead of # compiling one. Only the plain release channel qualifies: FELIS_SKIP_FETCH means "build # exactly what I staged" and a pinned FELIS_REF means "build that tree", both of which are # explicit requests for a source build, and the dev channel has no release to download. use_release_binary() { [ -z "${FELIS_SKIP_FETCH:-}" ] || return 1 [ -z "$FELIS_REF_PINNED" ] || return 1 [ "$FELIS_VERSION_BOOTSTRAP" = "release" ] } # download_release_binary installs the prebuilt felis binary for $FELIS_REF onto the host. # It returns non-zero to ask the caller to build that same tag from source instead; it never # dies, because no reachable failure here is worth aborting an install over. # # One asset covers the panel too: internal/panel/panel.go go:embeds internal/panel/static, and # release.yml builds through the repo Dockerfile so that tree holds the real npm output rather # than the tracked placeholder. There is nothing else to fetch. download_release_binary() { local arch asset tmp got if ! arch="$(felis_asset_arch)"; then warn "no prebuilt felis binary for architecture $(uname -m); building ${FELIS_REF} from source on this host instead" return 1 fi asset="felis-linux-${arch}" # Convergence check, and the cheapest one available: no API call, no download, and it asks # the exact question that matters. Reruns are the common case for this installer. if [ -x "$HOST_BIN" ] && [ "$("$HOST_BIN" version 2>/dev/null | head -n 1)" = "felis ${FELIS_REF}" ]; then HAVE_PREBUILT_BINARY=1 ok "host binary is already ${FELIS_REF}; skipping the download" return 0 fi # Staged next to HOST_BIN rather than in TMPDIR, because the validation below EXECUTES it # and /tmp is mounted noexec on CIS-hardened images. There the exec dies 126, the check # reads it as a bad asset, and every such host silently falls back to the full on-host # compile this path exists to avoid. /usr/local/bin has to be exec for felis to run at # all, so validating there tests the binary instead of the mount — and it keeps the # private-repo artifact out of a world-readable 1777 directory on the way in. mkdir -p "$(dirname "$HOST_BIN")" tmp="$(mktemp "$(dirname "$HOST_BIN")/.felis-download.XXXXXX")" remember_temp "$tmp" if ! download_release_asset "$FELIS_REF" "$asset" "$tmp"; then rm -f "$tmp" # Deliberately NOT "set FELIS_VERSION_BOOTSTRAP=dev": that channel builds main, a # different commit than the tag the operator asked for. Falling back keeps the tag and # changes only how it is obtained, so no operator action is needed at all. warn "release ${FELIS_REF} publishes no usable ${asset}; building ${FELIS_REF} from source on this host instead" return 1 fi # The hash comes BEFORE the exec below: until the file matches the release's SHA256SUMS it # is unverified bytes, and running it as root to ask its version would hand root to # whoever could swap the asset or sit on the download path. Missing or unmatched both # fall back to the source build of the same tag, which trusts only the git fetch. if ! verify_release_checksum "$FELIS_REF" "$asset" "$tmp"; then rm -f "$tmp" return 1 fi chmod 0755 "$tmp" # Run it once. This single exec subsumes three checks that would otherwise each need their # own code: a truncated download segfaults (curl reports success for an EOF-delimited body, # so nothing earlier catches it), a wrong-architecture asset fails to exec, and an unstamped # or mis-stamped build reports the wrong string here — rather than silently disabling # `felis update` for the entire life of the install. got="$("$tmp" version 2>/dev/null | head -n 1 || true)" if [ "$got" != "felis ${FELIS_REF}" ]; then rm -f "$tmp" warn "downloaded ${asset} reports '${got:-nothing}' rather than 'felis ${FELIS_REF}'; building ${FELIS_REF} from source on this host instead" return 1 fi # install(1) onto a freshly created destination, matching install_embedded_binary. A rename # would carry the source SELinux label instead of type-transitioning to bin_t — see # build_nano_binary for the 203/EXEC this shape avoids. Same-directory staging does not # change that: install(1) still creates the destination and copies. keep_previous_host_binary rm -f "$HOST_BIN" install -m 0755 "$tmp" "$HOST_BIN" rm -f "$tmp" command -v restorecon >/dev/null 2>&1 && restorecon "$HOST_BIN" >/dev/null 2>&1 || true HAVE_PREBUILT_BINARY=1 ok "installed ${asset} ${FELIS_REF} at ${HOST_BIN}" } # git_auth runs git with the token supplied by an inline credential helper. The helper # is a shell snippet that READS the exported variable when git asks; the token is never # in argv and never reaches .git/config, so it does not outlive the process. # # The EMPTY credential.helper in front is load-bearing, not a typo. credential.helper is a # MULTI-valued config: a bare `-c credential.helper=...` APPENDS to whatever the host has # configured, it does not replace it. On a host with `credential.helper=store` git then runs # both — ours answers the prompt, and store writes the token to ~/.git-credentials on the # approve that follows a successful auth, so it outlives the install after all. An empty # value is git's documented list reset. Verified on Fedora: with store configured the single # -c form persists the token to disk, the reset form writes nothing. # # GIT_TERMINAL_PROMPT=0 on both arms: GitHub answers a private repo with no or a bad token # by asking for credentials, and git would put that prompt on /dev/tty, where a piped # install sits waiting instead of failing with the FELIS_GITHUB_TOKEN hint. git_auth() { if [ -n "$FELIS_GITHUB_TOKEN" ]; then GIT_TERMINAL_PROMPT=0 git -c 'credential.helper=' \ -c 'credential.helper=!f() { printf "username=x-access-token\npassword=%s\n" "$FELIS_GITHUB_TOKEN"; }; f' "$@" else GIT_TERMINAL_PROMPT=0 git "$@" fi } # resolve_install_ref decides WHICH commit to build and, on the release channel, what to # stamp it as. Runs before fetch_source because it chooses that fetch's ref. resolve_install_ref() { # Idempotent: main calls it early to fail fast on a missing token, and fetch_source # calls it again because the nano path reaches fetch_source by its own route. [ -n "${REF_RESOLVED:-}" ] && return 0 REF_RESOLVED=1 if [ -n "$FELIS_REF" ]; then ok "building the pinned ref ${FELIS_REF}" return 0 fi case "$FELIS_VERSION_BOOTSTRAP" in release) log "resolving the newest published Felis release" FELIS_REF="$(github_latest_tag)" || die "could not resolve the newest Felis release. If the repository is private, set FELIS_GITHUB_TOKEN to a token with read access to it. If no release has been published yet, set FELIS_VERSION_BOOTSTRAP=dev to build main instead." # A release IS its tag, so the stamp is final here and stamp_version leaves it be. FELIS_VERSION="$FELIS_REF" ok "release channel: ${FELIS_REF}" ;; dev) FELIS_REF="main" # Best effort: the tag only names what this build is AHEAD of, and a repo with no # release yet is exactly when someone reaches for the dev channel. v0.0.0 keeps the # stamp parseable so `felis update` still compares rather than refusing. FELIS_VERSION_BASE="$(github_latest_tag 2>/dev/null || true)" [ -n "$FELIS_VERSION_BASE" ] || FELIS_VERSION_BASE="v0.0.0" ok "dev channel: main (ahead of ${FELIS_VERSION_BASE})" ;; *) die "FELIS_VERSION_BOOTSTRAP must be 'release' or 'dev', got '${FELIS_VERSION_BOOTSTRAP}'" ;; esac } # stamp_version finalises the dev/pinned stamp once a checkout exists to read the SHA # from. Release builds are already stamped and return untouched. # # The form is "+g", NOT `git describe`'s "--g", and the difference # is load-bearing. updates.Parse reads a "-" tail as a PRERELEASE, which sorts BELOW the # plain tag: a dev build 14 commits past v1.2.3 would compare as older than v1.2.3, and # `felis update` would propose "upgrading" onto the release it already contains. After # "+" the tail is build metadata, ignored for ordering, so the build reads as current # against v1.2.3 and as behind against v1.3.0 — both correct. # # `git describe` is also unreliable here: the primary clone is --depth 1 and carries no # tags, so describe falls back to a bare SHA, which updates.Parse rejects outright (it # fails closed on a non-numeric core). fetch_source does retry with a full clone when the # shallow one fails, which WOULD carry tags — that is exactly the point: the stamp must not # depend on which arm happened to win. rev-parse needs no history at all. stamp_version() { local sha [ -n "$FELIS_VERSION" ] && return 0 sha="$(git -C "$SRC_DIR" rev-parse --short HEAD 2>/dev/null || true)" [ -n "$sha" ] || sha="unknown" [ -n "$FELIS_VERSION_BASE" ] || FELIS_VERSION_BASE="v0.0.0" FELIS_VERSION="${FELIS_VERSION_BASE}+g${sha}" ok "build stamp: ${FELIS_VERSION}" } fetch_source() { if [ -n "${FELIS_SKIP_FETCH:-}" ]; then [ -d "$SRC_DIR" ] || die "FELIS_SKIP_FETCH set but ${SRC_DIR} does not exist" ok "skipping fetch; using pre-staged source at ${SRC_DIR}" # Whatever is staged is what gets built, so the channel has no say here and no # release lookup is made. stamp_version reads the staged checkout's own SHA. stamp_version return 0 fi resolve_install_ref if [ -d "${SRC_DIR}/.git" ]; then log "updating source in ${SRC_DIR}" git_auth -C "$SRC_DIR" fetch --depth 1 origin "$FELIS_REF" \ || die "could not fetch ${FELIS_REF} from ${FELIS_REPO_URL}; if the repository is private, set FELIS_GITHUB_TOKEN to a token with read access to it" git -C "$SRC_DIR" checkout -f FETCH_HEAD else log "cloning ${FELIS_REPO_URL} (${FELIS_REF})" mkdir -p "$(dirname "$SRC_DIR")" # The fallback checks the ref out explicitly. It used to be a bare full clone, which # silently landed on the default branch: harmless when FELIS_REF was always "main", # but the release channel now asks for a tag, and a build stamped v1.2.3 that # actually contains main is worse than a failed install. git_auth clone --depth 1 --branch "$FELIS_REF" "$FELIS_REPO_URL" "$SRC_DIR" 2>/dev/null \ || { git_auth clone "$FELIS_REPO_URL" "$SRC_DIR" \ && git_auth -C "$SRC_DIR" checkout -f "$FELIS_REF"; } \ || die "could not check out ${FELIS_REF} from ${FELIS_REPO_URL}; if the repository is private, set FELIS_GITHUB_TOKEN to a token with read access to it" fi stamp_version ok "source ready at ${SRC_DIR}" } install_embedded_binary() { local src src="${FELIS_BOOTSTRAP_BINARY:-}" [ -n "$src" ] || die "FELIS_BOOTSTRAP_BINARY is not set; cannot install the embedded setup binary" [ -x "$src" ] || die "FELIS_BOOTSTRAP_BINARY is not executable: ${src}" mkdir -p "$(dirname "$HOST_BIN")" if [ "$(readlink -f "$src")" != "$(readlink -f "$HOST_BIN" 2>/dev/null || true)" ]; then log "installing current felis binary onto the host (${HOST_BIN})" keep_previous_host_binary install -m 0755 "$src" "$HOST_BIN" else ok "host binary already installed at ${HOST_BIN}" fi HAVE_PREBUILT_BINARY=1 } build_image_from_binary() { local tmp tmp="$(mktemp -d)" remember_temp "$tmp" cp "$HOST_BIN" "${tmp}/felis" # static-debian12, matching the repo Dockerfile's final stage. Every binary that can reach # HOST_BIN traces back to that Dockerfile's CGO_ENABLED=0 build: the downloaded CI asset, # the binary the TUI is already running, and the one build_image_from_source docker-cp's # out of the image it just built. None of them link glibc, so the larger base-debian12 # bought nothing and only widened the runtime surface. It mattered little while this was # the rare fallback; now that the release channel downloads a binary and wraps it here, # this is the image most installs actually run, and it should be the one CI publishes. cat > "${tmp}/Dockerfile" <<'EOF' FROM gcr.io/distroless/static-debian12:nonroot ENV PATH=/usr/local/bin:/usr/bin:/bin COPY felis /usr/local/bin/felis USER 65532:65532 ENTRYPOINT ["/usr/local/bin/felis"] EOF chmod 0755 "${tmp}/felis" log "building ${FELIS_IMAGE} from the current felis binary" docker build -t "$FELIS_IMAGE" "$tmp" rm -rf "$tmp" } verify_image_starts() { log "verifying ${FELIS_IMAGE} starts" docker run --rm --user 1000:1000 --entrypoint /usr/local/bin/felis "$FELIS_IMAGE" help >/dev/null } build_image_from_source() { log "building ${FELIS_IMAGE} ${FELIS_VERSION:+(${FELIS_VERSION}) }(this compiles the Go binary; first run is slow)" # Without the stamp main.version stays "dev", and `felis update` refuses to compare a # "dev" build against upstream rather than treating it as 0.0.0. So an unstamped image # is not a cosmetic problem: it silently disables update reporting for the install. docker build -t "$FELIS_IMAGE" \ --build-arg FELIS_VERSION="${FELIS_VERSION:-dev}" "$SRC_DIR" log "extracting the felis binary onto the host (${HOST_BIN})" local cid cid="$(docker create "$FELIS_IMAGE")" remember_container "$cid" keep_previous_host_binary docker cp "${cid}:/usr/local/bin/felis" "$HOST_BIN" docker rm "$cid" >/dev/null chmod 0755 "$HOST_BIN" } # resolve_felis_image names the control-plane image after the release it carries # (registry.felis.svc:5000/felis/felis:v1.2.3) unless FELIS_IMAGE was given. One tag per # release is what makes `kubectl rollout undo` a rollback: the previous ReplicaSet names the # previous tag, and the registry keeps the newest five of them (its pruner, §9). Under one # mutable tag the undo re-created the pods on the image the upgrade had just written over it. # # The version is the release tag on the download path, the stamp on a source build # (v1.2.3+gabc1234 becomes the tag v1.2.3-gabc1234: '+' is not allowed in a tag), and the # binary's own report on the setup-console path, which skips both. A rerun of the same version # reuses its tag; deploy_bundle restarts the pods onto the rebuilt image then. resolve_felis_image() { local v [ -z "$FELIS_IMAGE" ] || return 0 v="$FELIS_VERSION" if [ -z "$v" ] && [ -n "$HAVE_PREBUILT_BINARY" ]; then v="$("$HOST_BIN" version 2>/dev/null | head -n 1 || true)" v="${v#felis }" fi FELIS_IMAGE="${REGISTRY_URL}/felis/felis:$(image_tag_for_version "$v")" ok "control-plane image: ${FELIS_IMAGE}" } # image_tag_for_version turns a felis version into an image tag: every character a tag may not # hold becomes '-'. An unknown version ("dev" is what an unstamped binary reports) is :demo. image_tag_for_version() { local v v="$(printf '%s' "$1" | tr -c 'A-Za-z0-9_.-' '-')" case "$v" in ""|dev|[!A-Za-z0-9_]*) printf 'demo' ;; *) printf '%s' "${v:0:128}" ;; esac } build_image() { systemctl start docker # Keyed on the binary, not on the route that produced it: the TUI hand-off and a release # download both land on exactly the same state (a felis binary at HOST_BIN, no checkout), # and a release download that fell back to source has cleared this so the source build runs. if [ -n "$HAVE_PREBUILT_BINARY" ]; then build_image_from_binary else build_image_from_source fi verify_image_starts log "importing ${FELIS_IMAGE} into k3s containerd" remove_k3s_image "$FELIS_IMAGE" docker save "$FELIS_IMAGE" | k3s_cmd ctr images import - # Reclaim the ~150 MiB the docker daemon holds; reruns restart it on demand. systemctl stop docker docker.socket 2>/dev/null || true ok "image built, binary on host, image imported" } remove_k3s_image() { local image="$1" k3s_cmd ctr images rm "$image" >/dev/null 2>&1 || true case "$image" in */*) ;; *) k3s_cmd ctr images rm "docker.io/library/${image}" >/dev/null 2>&1 || true ;; esac } # --------------------------------------------------------------------------- # 5b. The game stack: the login limbo + lobby images, and the Velocity proxy. # # Without this, `felis setup` asks the Owner to bind by joining Minecraft while # no Minecraft server exists — the whole point of installing is a joinable, # Mojang-authenticating server, so the installer produces one. # # The trust chain, end to end: # player --(Mojang auth)--> Velocity --(modern forwarding, HMAC)--> login limbo # Velocity is the ONLY thing that talks to Mojang; the backends run offline-mode and # trust the forwarded profile, which is exactly why the forwarding secret is the # identity boundary. NetworkPolicy narrows reachability but cannot block the node # hosting a pod, so it is defense in depth rather than a substitute for the HMAC. # --------------------------------------------------------------------------- # game_stack_source sets GAME_STACK_DIR to a docker build context holding # deploy/{limbo,lobby} and plugins/. The TUI path pipes this script in over stdin and # has no checkout on disk, so there the sources come out of the felis binary itself. # # Keyed on HAVE_PREBUILT_BINARY, the same flag build_image uses, because the question is # identical: a prebuilt binary means fetch_source never ran, so SRC_DIR is whatever an # EARLIER install left behind. Trusting it there would build felis-paper and the Velocity # plugin from the old commit while the control plane is the freshly downloaded release — # a silent version skew across the plugin/API boundary. The embedded tar always matches # the binary it came out of, so it is the correct source on every prebuilt path. game_stack_source() { if [ -z "$HAVE_PREBUILT_BINARY" ] && [ -f "${SRC_DIR}/deploy/limbo/Dockerfile" ]; then GAME_STACK_DIR="$SRC_DIR" ok "game-stack sources: ${SRC_DIR}" return 0 fi GAME_STACK_DIR="$(mktemp -d)" remember_temp "$GAME_STACK_DIR" log "unpacking the embedded game-stack sources (no checkout on this host)" "$HOST_BIN" bootstrap-assets game-stack | tar -x -C "$GAME_STACK_DIR" \ || die "could not unpack the embedded game-stack sources" [ -f "${GAME_STACK_DIR}/deploy/limbo/Dockerfile" ] \ || die "embedded game-stack tar is missing deploy/limbo/Dockerfile" ok "game-stack sources unpacked to ${GAME_STACK_DIR}" } # meta_get prints a small metadata document. curl's --retry covers transient HTTP # statuses and timeouts; a TLS handshake cut mid-way (exit 35, seen against Fill over # a flaky IPv6 path) is outside its retry set, so the outer loop retries every failure # twice more. --retry-all-errors would say the same but needs curl 7.71+. meta_get() { local url="$1" out attempt for attempt in 1 2 3; do if out="$(curl -fsSL --retry 5 --retry-delay 2 \ -A "felis-bootstrap (+https://github.com/FelisMC/Felis)" "$url")"; then printf '%s' "$out" return 0 fi if [ "$attempt" -lt 3 ]; then warn "fetching ${url} failed (attempt ${attempt}/3); retrying" sleep 5 fi done return 1 } # resolve_game_jars sets the artifacts the three game images and the proxy are built from: # the builds deploy/game-stack.lock names (FELIS_GAME_STACK=pinned, the default), or # upstream's newest ones (latest). Pinned is what makes an install reproducible: every host # installing one release gets the same login gate, lobby and proxy, each download is # checked against the lock's sha256, and a rerun's docker builds hit their cache, so # restart_existing_system_servers leaves the login and lobby pods running. resolve_game_jars() { if [ "$FELIS_GAME_STACK" = "latest" ]; then resolve_latest_game_jars return 0 fi load_game_stack_lock "${GAME_STACK_DIR}/deploy/game-stack.lock" ok "Limbo ${LIMBO_VERSION} + Paper on Minecraft ${MC_VERSION}, LuckPerms and Velocity ${VELOCITY_VERSION}: the builds game-stack.lock pins" } GAME_STACK_LOCK_KEYS="MC_VERSION LIMBO_VERSION LIMBO_JAR_URL LIMBO_JAR_SHA256 LIMBO_SCHEM_URL LIMBO_SCHEM_SHA256 PAPER_JAR_URL PAPER_JAR_SHA256 LUCKPERMS_JAR_URL LUCKPERMS_JAR_SHA256 VELOCITY_VERSION VELOCITY_JAR_URL VELOCITY_JAR_SHA256" # load_game_stack_lock reads the lock's KEY=value lines into the globals of the same names. # It never sources the file: only the keys above are accepted, every one has to be set, the # values are limited to URL and version characters (they reach docker build-args), and each # *_SHA256 has to be a sha256. load_game_stack_lock() { local file="$1" line key value [ -f "$file" ] || die "no game-stack lock at ${file}; FELIS_GAME_STACK=latest resolves upstream's newest builds instead" for key in $GAME_STACK_LOCK_KEYS; do printf -v "$key" '%s' ""; done while IFS= read -r line || [ -n "$line" ]; do case "$line" in ''|'#'*) continue ;; esac key="${line%%=*}" value="${line#*=}" [ "$key" != "$line" ] || die "${file}: not a KEY=value line: ${line}" case " ${GAME_STACK_LOCK_KEYS} " in *" ${key} "*) ;; *) die "${file}: unknown key ${key}" ;; esac case "$value" in ''|*[!A-Za-z0-9._:/+%-]*) die "${file}: ${key} has an unexpected value: ${value}" ;; esac printf -v "$key" '%s' "$value" done < "$file" for key in $GAME_STACK_LOCK_KEYS; do [ -n "${!key}" ] || die "${file} does not set ${key}" case "$key" in *_SHA256) [[ "${!key}" =~ ^[0-9a-f]{64}$ ]] || die "${file}: ${key} is not a lowercase sha256" ;; esac done } # url_sha256 prints the sha256 of what $1 serves. url_sha256() { local sum sum="$(curl -fsSL --retry 5 --retry-delay 2 -A "felis-bootstrap (+https://github.com/FelisMC/Felis)" "$1" | sha256sum)" || return 1 printf '%s\n' "${sum%% *}" } # resolve_latest_game_jars pins Limbo and Paper to the SAME Minecraft version. LOOHP/Limbo # speaks exactly one protocol per build, so the login gate dictates the version and Paper # follows — a client that can pass the gate must also be able to reach the lobby. # MC_VERSION is read off Limbo's CI artifact name (Limbo--.jar), which # is the only place the pairing is published. # # Limbo's CI and LuckPerms publish no digest, so latest hashes their downloads as it finds # them: the image builds still check that they receive those bytes, but nothing vouches for # the bytes themselves. That is what the lock file adds. resolve_latest_game_jars() { local ci="https://ci.loohpjames.com/job/Limbo/lastSuccessfulBuild" meta build file base rest paper log "resolving the newest LOOHP/Limbo CI build" # Fetch first, filter second: `curl | grep | head` dies of SIGPIPE under `set -o pipefail` # the moment head closes the pipe early. Same shape everywhere below. meta="$(meta_get "${ci}/api/json")" \ || die "could not read the LOOHP/Limbo CI build metadata" file="$(printf '%s' "$meta" | grep -o 'Limbo-[0-9A-Za-z._-]*\.jar' || true)" file="${file%%$'\n'*}" [ -n "$file" ] || die "no Limbo jar in the LOOHP/Limbo CI artifact list" # The numbered build, so the jar hashed below is the jar the image build downloads even # if CI finishes another build in between. build="$(printf '%s' "$meta" | grep -o '"number":[0-9]*' || true)" build="${build%%$'\n'*}" build="${build#*:}" [ -n "$build" ] || die "no build number in the LOOHP/Limbo CI metadata" ci="https://ci.loohpjames.com/job/Limbo/${build}" base="${file%.jar}" # Limbo-2026.0.2-ALPHA-26.2 MC_VERSION="${base##*-}" # 26.2 rest="${base%-*}" # Limbo-2026.0.2-ALPHA LIMBO_VERSION="${rest#Limbo-}" [ -n "$MC_VERSION" ] && [ -n "$LIMBO_VERSION" ] || die "cannot parse Limbo artifact name: ${file}" LIMBO_JAR_URL="${ci}/artifact/target/${file}" LIMBO_SCHEM_URL="${ci}/artifact/spawn.schem" # PaperMC Fill v3. The old api.papermc.io v2 has returned HTTP 410 since 2026-07-01 and # is never coming back; Fill wants a descriptive User-Agent. log "resolving the newest Paper ${MC_VERSION} build" paper="$(papermc_latest_jar paper "$MC_VERSION")" \ || die "could not resolve a Paper build for Minecraft ${MC_VERSION} (the login gate pins this protocol; the build likely exists — Fill upstream is down, flapping, or no longer content-addressed)" PAPER_JAR_URL="${paper% *}" PAPER_JAR_SHA256="${paper##* }" # LuckPerms is not version-matched to MC_VERSION the way Paper is: it ships one # current Bukkit build that supports the whole supported Minecraft range, so there is # no per-version endpoint to ask. log "resolving the newest LuckPerms build" LUCKPERMS_JAR_URL="$(luckperms_latest_jar)" \ || die "could not resolve a LuckPerms build (metadata.luckperms.net is down or flapping); the lobby needs it for the panel's permission controls" log "hashing the Limbo and LuckPerms downloads (their upstreams publish no digest)" LIMBO_JAR_SHA256="$(url_sha256 "$LIMBO_JAR_URL")" || die "could not download ${LIMBO_JAR_URL}" LIMBO_SCHEM_SHA256="$(url_sha256 "$LIMBO_SCHEM_URL")" || die "could not download ${LIMBO_SCHEM_URL}" LUCKPERMS_JAR_SHA256="$(url_sha256 "$LUCKPERMS_JAR_URL")" || die "could not download ${LUCKPERMS_JAR_URL}" VELOCITY_VERSION="$VELOCITY_LATEST_MINOR" VELOCITY_JAR_URL="" VELOCITY_JAR_SHA256="" ok "Limbo ${LIMBO_VERSION} (CI build ${build}) + Paper, both on Minecraft ${MC_VERSION}; LuckPerms resolved" warn "FELIS_GAME_STACK=latest: these are upstream's builds as of now, not the ones this release pins" } # luckperms_latest_jar prints the download URL of the current LuckPerms Bukkit build. # Bukkit, not bukkit-legacy: legacy targets Minecraft 1.8-1.12, and Paper 26.2 is far # past that. The same fetch-then-grep shape (and --retry rationale) as # papermc_latest_jar; the metadata endpoint hands back every platform's URL at once, so # the grep has to pin the /bukkit/ path segment or it would just as happily return the # Fabric or Velocity jar, neither of which Paper can load. luckperms_latest_jar() { local json url json="$(meta_get "https://metadata.luckperms.net/data/all")" || return 1 url="$(printf '%s' "$json" \ | grep -o 'https://download\.luckperms\.net/[0-9]\{1,\}/bukkit/loader/[^"]*\.jar' || true)" url="${url%%$'\n'*}" [ -n "$url" ] || return 1 printf '%s\n' "$url" } # papermc_latest_jar prints " " for the newest build of . # --retry rides out Fill's transient gateway errors (502/503/504 are in curl's retry # set): a single blip must not abort the whole bootstrap claiming the build is missing. # Plain --retry only, deliberately: --retry-connrefused needs curl 7.52+, which the yum # (el7) path does not have, and it would only add ECONNREFUSED to an already-covered set. # The digest is not fished out of the JSON separately: Fill's download URLs are # content-addressed (/v1/objects//.jar), so the path segment names the # bytes the URL serves and both halves come from the same grep of the same response. A # URL without that shape fails the resolve rather than waving the download through # unchecked. papermc_latest_jar() { local project="$1" version="$2" json urls url sha json="$(meta_get "https://fill.papermc.io/v3/projects/${project}/versions/${version}/builds/latest")" || return 1 urls="$(printf '%s' "$json" | grep -o 'https://fill-data\.papermc\.io/[^"]*\.jar' || true)" url="${urls%%$'\n'*}" [ -n "$url" ] || return 1 sha="${url#*/objects/}" sha="${sha%%/*}" case "$sha" in *[!0-9a-f]*|"") return 1 ;; esac [ "${#sha}" -eq 64 ] || return 1 printf '%s %s\n' "$url" "$sha" } build_game_stack() { systemctl start docker game_stack_source resolve_game_jars log "building ${FELIS_LIMBO_IMAGE} (LOOHP/Limbo ${LIMBO_VERSION}, Minecraft ${MC_VERSION})" docker build -f "${GAME_STACK_DIR}/deploy/limbo/Dockerfile" \ --build-arg LIMBO_JAR_URL="$LIMBO_JAR_URL" \ --build-arg LIMBO_JAR_SHA256="$LIMBO_JAR_SHA256" \ --build-arg LIMBO_SCHEM_URL="$LIMBO_SCHEM_URL" \ --build-arg LIMBO_SCHEM_SHA256="$LIMBO_SCHEM_SHA256" \ --build-arg LIMBO_VERSION="$LIMBO_VERSION" \ -t "$FELIS_LIMBO_IMAGE" "$GAME_STACK_DIR" log "building ${FELIS_LOBBY_IMAGE} (Paper ${MC_VERSION} + felis-paper /menu + LuckPerms)" docker build -f "${GAME_STACK_DIR}/deploy/lobby/Dockerfile" \ --build-arg PAPER_JAR_URL="$PAPER_JAR_URL" \ --build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \ --build-arg LUCKPERMS_JAR_URL="$LUCKPERMS_JAR_URL" \ --build-arg LUCKPERMS_JAR_SHA256="$LUCKPERMS_JAR_SHA256" \ -t "$FELIS_LOBBY_IMAGE" "$GAME_STACK_DIR" # Plain Paper, same MC_VERSION and PAPER_JAR_URL (no new dependency). Forwarding is the # operator initContainer's job, so this image carries no /menu plugin and no secret gate. log "building ${FELIS_PAPER_IMAGE} (plain Paper ${MC_VERSION}, forwarding via the operator initContainer)" docker build -f "${GAME_STACK_DIR}/deploy/paper/Dockerfile" \ --build-arg PAPER_JAR_URL="$PAPER_JAR_URL" \ --build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \ -t "$FELIS_PAPER_IMAGE" "$GAME_STACK_DIR" # The builds the system servers run, for restart_existing_system_servers. Docker's layer # cache gives an unchanged build the same id, so a rerun that rebuilt nothing leaves the # login and lobby pods (and every player on them) alone. LIMBO_IMAGE_ID="$(docker image inspect -f '{{.Id}}' "$FELIS_LIMBO_IMAGE")" LOBBY_IMAGE_ID="$(docker image inspect -f '{{.Id}}' "$FELIS_LOBBY_IMAGE")" local img for img in "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do log "importing ${img} into k3s containerd" remove_k3s_image "$img" docker save "$img" | k3s_cmd ctr images import - done build_velocity_plugin systemctl stop docker docker.socket 2>/dev/null || true ok "login + lobby images imported; felis-velocity.jar staged" } ensure_velocity_directory() { local path="$1" mode="$2" owner="$3" group="$4" if [ -L "$path" ]; then rm -f -- "$path" elif [ -e "$path" ] && [ ! -d "$path" ]; then die "velocity path exists but is not a directory: ${path}" fi install -d -o "$owner" -g "$group" -m "$mode" "$path" } # Velocity manages its own working directory: on startup it migrates velocity.toml to # the running config-version, extracts localizations, and creates plugins/bStats — all # fatal or noisy if it cannot write. So the service owns the tree and velocity.toml. The # immutable artifacts (velocity.jar, felis-velocity.jar) and the forwarding secret stay # root-owned and read-only to the proxy; ProtectSystem=strict confines writes to VELOCITY_DIR. prepare_velocity_layout() { id -u "$VELOCITY_USER" >/dev/null 2>&1 \ || useradd --system --home-dir "$VELOCITY_DIR" --shell /usr/sbin/nologin "$VELOCITY_USER" ensure_velocity_directory "$VELOCITY_DIR" 0750 "$VELOCITY_USER" "$VELOCITY_USER" ensure_velocity_directory "${VELOCITY_DIR}/plugins" 0750 "$VELOCITY_USER" "$VELOCITY_USER" ensure_velocity_directory "${VELOCITY_DIR}/plugins/felis-link" 0750 "$VELOCITY_USER" "$VELOCITY_USER" ensure_velocity_directory "${VELOCITY_DIR}/logs" 0750 "$VELOCITY_USER" "$VELOCITY_USER" ensure_velocity_directory "${VELOCITY_DIR}/crash-reports" 0750 "$VELOCITY_USER" "$VELOCITY_USER" } atomic_install_file() { local source="$1" target="$2" mode="$3" owner="$4" group="$5" local parent base staged parent="$(dirname "$target")" base="$(basename "$target")" [ -d "$parent" ] && [ ! -L "$parent" ] \ || die "refusing to install through a non-directory/symlink parent: ${parent}" if [ -d "$target" ] && [ ! -L "$target" ]; then die "refusing to replace directory with file: ${target}" fi staged="$(mktemp "${parent}/.${base}.XXXXXX")" remember_temp "$staged" install -o "$owner" -g "$group" -m "$mode" "$source" "$staged" mv -fT "$staged" "$target" } # build_velocity_plugin compiles plugins/velocity in the same gradle image the two # Dockerfiles use, and drops the jar where Velocity will look for it. Docker is the # toolchain here on purpose: the host needs no JDK and no gradle, only a JRE. Gradle # checks every dependency against plugins/velocity/gradle/verification-metadata.xml. build_velocity_plugin() { log "building felis-velocity.jar (gradle in a container; the host gets no JDK)" prepare_velocity_layout # :z relabels the bind mount for SELinux (Fedora/EL enforce it; elsewhere it is a no-op). docker run --rm \ -v "${GAME_STACK_DIR}:/src:z" \ -w /src/plugins/velocity \ "$PLUGIN_BUILD_IMAGE" gradle --no-daemon clean build \ || die "felis-velocity plugin build failed" local -a jars=( "${GAME_STACK_DIR}"/plugins/velocity/build/libs/felis-velocity-*.jar ) [ "${#jars[@]}" -eq 1 ] && [ -f "${jars[0]}" ] \ || die "felis-velocity build must produce exactly one plugin jar" atomic_install_file "${jars[0]}" "${VELOCITY_DIR}/plugins/felis-velocity.jar" 0644 root root } # install_via_plugins stages ViaVersion + ViaBackwards + ViaRewind so players on clients # older than the proxy can still join. # # Modern forwarding nominally refuses anything below 1.13: HandshakeSessionHandler#handleLogin # reads the handshake protocol version and disconnects with # velocity.error.modern-forwarding-needs-new-client. That gate stops firing once Via is # present — it logs "Replacing channel initializers" during startup, so the version reaching # the check is plausibly already the rewritten one. The 1.13 floor is a property of the # UNASSISTED proxy pipeline, not of the forwarding protocol, so nothing here changes # player-info-forwarding-mode and no backend is patched or downgraded. # # Measured end to end rather than assumed (Felis-Legacy FL-007, cell modern-via121): a # protocol-47 client joined a stock Paper 1.21.11 backend through a modern-forwarding proxy. # The proof is the join itself, not the log line — that backend ran velocity.enabled=true with # a shared secret, and Paper in that state rejects any login not carrying forwarding data # signed with a matching HMAC. Only protocol 47 was measured; the rest of Via's 1.7-1.12 range # is its own documented support. # # Pinned by hash and not by "latest" on purpose. These three jars sit in front of every packet # on the proxy, and they are the exact bytes FL-007 measured — a moving tag would quietly make # this an unmeasured configuration. Bumping a version means bumping its checksum here. # # The three versions are a set, not three independent pins. ViaRewind is the component that # carries 1.8/1.7 support, and 4.1.2 against ViaVersion/ViaBackwards 5.11.0 fails to load # Protocol1_9To1_8 — the single protocol every 1.8 client needs — with "Invalid version: 1" # at proxy startup. 4.1.3 is the release that adds 5.11.0 compatibility; a two-arm run of the # same proxy image logs that error three times on 4.1.2 and not at all on 4.1.3. Read the # ViaRewind release notes before moving ViaVersion or ViaBackwards. install_via_plugins() { prepare_velocity_layout local name version want target url tmp have while read -r name version want; do [ -n "$name" ] || continue target="${VELOCITY_DIR}/plugins/${name}.jar" # Hash stdin, never the path: sha256sum escapes its output line when the filename # carries a backslash or a newline, and a leading "\" on the digest silently fails # every comparison below. have="" [ -f "$target" ] && have="$(sha256sum <"$target" | cut -d' ' -f1)" if [ "$have" = "$want" ]; then ok "${name} ${version} already staged" continue fi url="https://github.com/ViaVersion/${name}/releases/download/${version}/${name}-${version}.jar" log "downloading ${name} ${version}" tmp="$(mktemp "${VELOCITY_DIR}/.${name}.jar.XXXXXX")" remember_temp "$tmp" curl -fsSL "$url" -o "$tmp" || die "failed to download ${name} ${version}: ${url}" have="$(sha256sum <"$tmp" | cut -d' ' -f1)" [ "$have" = "$want" ] \ || die "${name} ${version} checksum mismatch: got ${have}, expected ${want}" atomic_install_file "$tmp" "$target" 0644 root root done <<'EOF' ViaVersion 5.11.0 18d19e90fc9467d68128c076630ae8700449c901402a3ef421837ce006bc8cae ViaBackwards 5.11.0 b21983d561e3f92df257683f0133ab6c68ec68175e8acfd82c6231723bf83587 ViaRewind 4.1.3 2d5970d22b4711c9ab2800932326c7b08acdace25ed7c6bbb8f6ea81054962b4 EOF pin_via_block_connections ok "Via staged; clients from 1.8 up can join under modern forwarding" } # pin_via_block_connections turns ViaVersion's serverside block-connection tracking off. # # ConnectionData.init() only builds its block-connection provider when Via's lowest supported # protocol is below 1.13. Under modern forwarding the Velocity injector reports 393 (1.13), so # init() returns early, blockConnectionProvider stays null, and the first 1.12.2->1.13 chunk # rewrite dereferences it. A 1.8 client on a protocol-47 backend takes an NPE on the first chunk # it is sent and never finishes joining. Every call site in protocols/v1_12_2to1_13 is behind # isServersideBlockConnections(), so switching the option off skips all of them. The cost is # cosmetic and pre-1.13 only: fences and glass panes stop being drawn connected. # # ViaVersion's default is true, so a fresh install ships that NPE unless it is corrected here. # Seeding a file with this one key is enough: Config#loadConfig parses the bundled # assets/viaversion/config.yml as the base map and merges the on-disk file over it, so every # other option still comes from the shipped default and stays current across version bumps. # # Do not read the absence of "Loading block connection mappings" from the log as proof this # worked. init() gates on the protocol version as well, and under modern forwarding that half # fails on its own — the line is missing either way. The config value is the only evidence. pin_via_block_connections() { local dir="${VELOCITY_DIR}/plugins/viaversion" config tmp config="${dir}/config.yml" ensure_velocity_directory "$dir" 0750 "$VELOCITY_USER" "$VELOCITY_USER" tmp="$(mktemp "${VELOCITY_DIR}/.viaversion-config.XXXXXX")" remember_temp "$tmp" if [ ! -f "$config" ]; then printf 'serverside-blockconnections: false\n' > "$tmp" elif grep -qE '^serverside-blockconnections:' "$config"; then sed -E 's/^serverside-blockconnections:.*/serverside-blockconnections: false/' "$config" > "$tmp" else { cat "$config"; printf 'serverside-blockconnections: false\n'; } > "$tmp" fi # Via rewrites this file itself on every load, so the proxy user has to own it. atomic_install_file "$tmp" "$config" 0640 "$VELOCITY_USER" "$VELOCITY_USER" } install_jre() { local arch want url release json case "$(uname -m)" in x86_64|amd64) arch="x64"; want="$JRE_PINNED_SHA256_X64" ;; aarch64|arm64) arch="aarch64"; want="$JRE_PINNED_SHA256_AARCH64" ;; *) if [ -x "${JRE_DIR}/bin/java" ]; then ok "JRE already installed at ${JRE_DIR}" return 0 fi die "no Temurin JRE build for architecture $(uname -m); pre-stage one at ${JRE_DIR}" ;; esac if [ "$FELIS_JRE_VERSION" = "$JRE_PINNED_FEATURE" ]; then release="$JRE_PINNED_RELEASE" url="https://github.com/adoptium/temurin${JRE_PINNED_FEATURE}-binaries/releases/download/jdk-${release/+/%2B}/OpenJDK${JRE_PINNED_FEATURE}U-jre_${arch}_linux_hotspot_${release/+/_}.tar.gz" else # Another feature version is installed once and then left alone; its digest is the # one Adoptium's API publishes next to the link. if [ -x "${JRE_DIR}/bin/java" ]; then ok "JRE already installed at ${JRE_DIR}" return 0 fi json="$(meta_get "https://api.adoptium.net/v3/assets/latest/${FELIS_JRE_VERSION}/hotspot?architecture=${arch}&image_type=jre&os=linux&vendor=eclipse")" \ || die "could not ask the Adoptium API for a Temurin ${FELIS_JRE_VERSION} JRE" url="$(printf '%s' "$json" | grep -o '"link": *"[^"]*\.tar\.gz"' || true)" url="${url%%$'\n'*}" url="${url%\"}" url="${url##*\"}" want="$(printf '%s' "$json" | grep -o '"checksum": *"[0-9a-f]\{64\}"' || true)" want="${want%%$'\n'*}" want="${want%\"}" want="${want##*\"}" release="$(printf '%s' "$json" | grep -o '"release_name": *"jdk-[^"]*"' || true)" release="${release%%$'\n'*}" release="${release%\"}" release="${release##*\"jdk-}" [ -n "$url" ] && [ -n "$want" ] && [ -n "$release" ] \ || die "the Adoptium API lists no Temurin ${FELIS_JRE_VERSION} JRE for linux/${arch}" fi # A rerun moves the installer's own JRE to the pinned build, which is how a JRE security # release reaches the proxy: bump the pin, rerun, and install_velocity_service restarts # the proxy because the JRE's release file changed. A JRE someone else put here (another # vendor, or pre-staged for an architecture Temurin does not build) is left alone. if [ -x "${JRE_DIR}/bin/java" ]; then if grep -qxF "IMPLEMENTOR_VERSION=\"Temurin-${release}\"" "${JRE_DIR}/release" 2>/dev/null; then ok "Temurin ${release} JRE already installed at ${JRE_DIR}" return 0 fi if ! grep -qxF 'IMPLEMENTOR="Eclipse Adoptium"' "${JRE_DIR}/release" 2>/dev/null; then ok "JRE at ${JRE_DIR} is not a Temurin build this installer put there; left as is" return 0 fi log "moving the proxy's JRE to Temurin ${release}" fi log "installing Temurin ${release} JRE (${arch}) to ${JRE_DIR}" local tmp have tmp="$(mktemp -d)" remember_temp "$tmp" curl -fsSL --retry 5 --retry-delay 2 "$url" -o "${tmp}/jre.tar.gz" \ || die "failed to download the Temurin JRE: ${url}" have="$(sha256sum <"${tmp}/jre.tar.gz" | cut -d' ' -f1)" [ "$have" = "$want" ] \ || die "Temurin ${release} JRE (${arch}) hashes to ${have}, expected ${want}; refusing to install it" # Unpacked beside the live one and swapped in with two renames, so a failed unpack leaves # the proxy's runtime untouched. The running proxy keeps the files it has open. rm -rf "${JRE_DIR}.new" "${JRE_DIR}.old" mkdir -p "${JRE_DIR}.new" # The tarball has a single versioned top-level directory (jdk-25+36-jre/); strip it so # the path in the systemd unit never carries a build number. tar -C "${JRE_DIR}.new" --strip-components=1 -xzf "${tmp}/jre.tar.gz" || die "failed to unpack the JRE" [ -x "${JRE_DIR}.new/bin/java" ] || die "unpacked JRE has no bin/java" [ ! -e "$JRE_DIR" ] || mv "$JRE_DIR" "${JRE_DIR}.old" mv "${JRE_DIR}.new" "$JRE_DIR" rm -rf "${JRE_DIR}.old" ok "JRE at ${JRE_DIR}/bin/java (Temurin ${release})" } install_velocity() { install_jre local url tmp have want resolved prepare_velocity_layout if [ -n "$FELIS_VELOCITY_FORK_JAR" ]; then [ -f "$FELIS_VELOCITY_FORK_JAR" ] \ || die "FELIS_VELOCITY_FORK_JAR is not a readable file: ${FELIS_VELOCITY_FORK_JAR}" # Hash stdin, never the path — same reason as install_via_plugins: sha256sum escapes its # output line for a filename carrying a backslash or a newline, and the leading "\" that # adds would fail every comparison below. have="$(sha256sum <"$FELIS_VELOCITY_FORK_JAR" | cut -d' ' -f1)" # Refuse rather than warn. This jar is the proxy every player connects through, and a # warning in an install log is not a gate. The digest is printed so the first run after # a deliberate rebuild is one copy-paste, not an investigation. [ -n "$FELIS_VELOCITY_FORK_JAR_SHA256" ] || die \ "FELIS_VELOCITY_FORK_JAR_SHA256 is required whenever FELIS_VELOCITY_FORK_JAR is set. The jar at that path hashes to ${have}. Check that against the build you meant to install, then re-run with FELIS_VELOCITY_FORK_JAR_SHA256=${have}" # Normalise the operator's digest before comparing. sha256sum prints lowercase, but the # build host is often Windows, where Get-FileHash prints uppercase and certutil has # shipped both with and without spaces between the bytes. All three name the same jar, # so comparing raw would refuse two of the three spellings and word it as tampering. want="$(printf '%s' "$FELIS_VELOCITY_FORK_JAR_SHA256" | tr -d '[:space:]' | tr 'A-Z' 'a-z')" [ "$have" = "$want" ] || die \ "FELIS_VELOCITY_FORK_JAR checksum mismatch: got ${have}, expected ${want}" log "installing the Felis-Legacy Velocity fork from ${FELIS_VELOCITY_FORK_JAR} (sha256 ${have})" atomic_install_file "$FELIS_VELOCITY_FORK_JAR" "${VELOCITY_DIR}/velocity.jar" 0644 root root else local version="${FELIS_VELOCITY_VERSION:-${VELOCITY_VERSION:-$VELOCITY_LATEST_MINOR}}" if [ "$FELIS_GAME_STACK" = "pinned" ] && [ "$version" = "${VELOCITY_VERSION:-}" ]; then url="$VELOCITY_JAR_URL" want="$VELOCITY_JAR_SHA256" else log "resolving the newest Velocity ${version} build" resolved="$(papermc_latest_jar velocity "$version")" \ || die "no Velocity build for ${version} (override with FELIS_VELOCITY_VERSION)" url="${resolved% *}" want="${resolved##* }" fi stage_velocity_jar "$url" "$want" "$version" fi install_via_plugins write_velocity_config install_velocity_service configure_velocity_firewall } # stage_velocity_jar installs the stock proxy from $1 unless velocity.jar already hashes to # $2, so a rerun of the same build neither downloads nor touches the file the proxy runs. stage_velocity_jar() { local url="$1" want="$2" version="$3" have="" tmp [ -f "${VELOCITY_DIR}/velocity.jar" ] && have="$(sha256sum <"${VELOCITY_DIR}/velocity.jar" | cut -d' ' -f1)" if [ "$have" = "$want" ]; then ok "Velocity ${version} already staged" return 0 fi log "downloading Velocity ${version}" tmp="$(mktemp "${VELOCITY_DIR}/.velocity.jar.XXXXXX")" remember_temp "$tmp" curl -fsSL --retry 5 --retry-delay 2 "$url" -o "$tmp" || die "failed to download Velocity: ${url}" # The same gate the Via plugins and the fork jar pass: this jar is the proxy every # player connects through, and the lock file (or Fill's content-addressed URL) names its # digest — a truncated or tampered download becomes a refusal here, not a proxy that # won't boot. have="$(sha256sum <"$tmp" | cut -d' ' -f1)" [ "$have" = "$want" ] \ || die "Velocity ${version} checksum mismatch: got ${have}, expected ${want}" atomic_install_file "$tmp" "${VELOCITY_DIR}/velocity.jar" 0644 root root } # felis_internal_ip echoes the felis-api-internal Service ClusterIP. Cluster DNS does not # resolve from the host, but a Service ClusterIP DOES route from the node (kube-proxy programs # the host netns) — the same trick the on-node break-glass console uses. The internal face is # deliberately ClusterIP-only: it is service-token authenticated and must never be published on # a node's external IP. Both the felis-link plugin config and the Velocity sessionserver # override (install_velocity_service) point at it, so the lookup lives here once. felis_internal_ip() { local ip ip="$(kube -n "$CONTROL_NS" get svc felis-api-internal -o jsonpath='{.spec.clusterIP}')" \ || die "could not resolve the felis-api-internal ClusterIP" [ -n "$ip" ] || die "felis-api-internal has no ClusterIP" printf '%s' "$ip" } write_velocity_config() { local api_ip tmp panel_host admin_host api_ip="$(felis_internal_ip)" panel_host="$(auth_hostname panel_hostname "console.${FELIS_ROOT_DOMAIN}")" admin_host="$(auth_hostname admin_hostname "op.console.${FELIS_ROOT_DOMAIN}")" prepare_velocity_layout tmp="$(mktemp -d)" remember_temp "$tmp" # The forwarding key. Velocity refuses to start on an empty one ("The forwarding-secret # file must not be empty."), which is the failure mode we want if this ever goes wrong. (umask 077; printf '%s' "$FORWARDING_SECRET" > "${tmp}/forwarding.secret") cat > "${tmp}/velocity.toml" < "${tmp}/felis-link.properties" <backend pipeline to protocol 47; only the handshake field survives Via. The Felis # fork reads this list from -Dfelis.legacy-forwarding.servers and forwards those servers legacy; # every other backend keeps modern+secret untouched. # # This value is the floor of the list. The felis-velocity plugin adds every server whose # MinecraftServer CR is labelled felis.lolicon.best/forwarding=legacy by rewriting the same # property on each server-list refresh (LegacyForwarding.java), and drops it again when the # label goes; the floor always stays in. A fork carrying patch 0004 re-reads the property on # every backend connection, so a label applies from the next connection. A fork with 0003 # alone reads it once, after the plugin's first refresh, so a label applies at the next proxy # restart. Stock Velocity ignores it, and the plugin logs a warning for a labelled server. # # The -D below is double-quoted in ExecStart on purpose. The fork trims each element, so it # accepts "legacy18, legacy112", but systemd splits ExecStart on whitespace before java ever # sees it -- unquoted, that spelling would hand java a stray "legacy112" argument and the unit # would not start. Quoting keeps the whole property one argv item. local legacy_forwarding_servers="${FELIS_LEGACY_FORWARDING_SERVERS}" # -Xms stays at 512M so a small proxy does not reserve its whole ceiling up front, unless # the ceiling itself is lower (the JVM refuses an initial heap above the maximum). local xmx="$FELIS_VELOCITY_XMX" xms="512M" [ "$(heap_megabytes "$xmx")" -ge 512 ] || xms="$xmx" cat > "$VELOCITY_SERVICE" </dev/null || true)" ]; then ok "felis-velocity unchanged; left running (0.0.0.0:${FELIS_GAME_PORT})" return 0 fi systemctl restart felis-velocity printf '%s\n' "$fp" > "$VELOCITY_FINGERPRINT" ok "felis-velocity.service enabled and started (0.0.0.0:${FELIS_GAME_PORT})" } # velocity_fingerprint hashes what the proxy process runs: its unit (JVM flags and system # properties), the JRE, the jars and the files the installer writes for it. The Via config # and whatever else plugins write at runtime stay out; Via rewrites its config on every load. velocity_fingerprint() { local f for f in "$VELOCITY_SERVICE" "${JRE_DIR}/release" "${VELOCITY_DIR}/velocity.jar" \ "${VELOCITY_DIR}/velocity.toml" "${VELOCITY_DIR}/forwarding.secret" \ "${VELOCITY_DIR}/plugins/felis-link/felis-link.properties" "${VELOCITY_DIR}"/plugins/*.jar; do [ -f "$f" ] || continue printf '%s %s\n' "$(sha256sum <"$f" | cut -d' ' -f1)" "$f" done | sha256sum | cut -d' ' -f1 } configure_velocity_firewall() { command -v firewall-cmd >/dev/null 2>&1 || return 0 systemctl is-active --quiet firewalld || return 0 log "opening firewalld port ${FELIS_GAME_PORT}/tcp for the Minecraft proxy" firewall-cmd --permanent --add-port="${FELIS_GAME_PORT}/tcp" firewall-cmd --reload } # --------------------------------------------------------------------------- # 6. PostgreSQL on the host. felis-api pods reach it at :5432; # migrations run from the host binary against 127.0.0.1. # --------------------------------------------------------------------------- write_pg_hba_block() { local hba="$1" tmp tmp_new node_cidr node_cidr="${NODE_IP}/32" tmp="$(mktemp)" tmp_new="${tmp}.new" remember_temp "$tmp" remember_temp "$tmp_new" awk \ -v db="$DB_NAME" \ -v user="$DB_USER" \ -v pod="$POD_CIDR" \ -v node="$node_cidr" ' $0 == "# BEGIN FELIS MANAGED HBA" { skip = 1; next } $0 == "# END FELIS MANAGED HBA" { skip = 0; next } skip { next } # Clean up rules appended by older bootstrap versions. $1 == "host" && $2 == db && $3 == user && $5 == "scram-sha-256" && ($4 == "127.0.0.1/32" || $4 == pod || $4 == node) { next } { print } ' "$hba" > "$tmp" { printf "# BEGIN FELIS MANAGED HBA\n" printf "# Felis rules must precede distro defaults such as 127.0.0.1 ident.\n" printf "host %s %s 127.0.0.1/32 scram-sha-256\n" "$DB_NAME" "$DB_USER" printf "host %s %s %s scram-sha-256\n" "$DB_NAME" "$DB_USER" "$POD_CIDR" printf "host %s %s %s scram-sha-256\n" "$DB_NAME" "$DB_USER" "$node_cidr" printf "# END FELIS MANAGED HBA\n" printf "\n" cat "$tmp" } > "$tmp_new" cat "$tmp_new" > "$hba" rm -f "$tmp" "$tmp_new" } postgres_data_dir() { case "$PKG" in pacman) printf '%s\n' /var/lib/postgres/data ;; *) printf '%s\n' /var/lib/pgsql/data ;; esac } init_postgres_data_dir() { local data_dir data_dir="$(postgres_data_dir)" [ "$PKG" = "apt" ] && return 0 [ -f "${data_dir}/PG_VERSION" ] && return 0 log "initialising postgresql data directory at ${data_dir}" if command -v postgresql-setup >/dev/null 2>&1; then postgresql-setup --initdb || /usr/bin/postgresql-setup initdb elif command -v initdb >/dev/null 2>&1; then install -d -o postgres -g postgres -m 0700 "$data_dir" as_postgres initdb -D "$data_dir" else die "cannot initialise postgresql data directory: postgresql-setup/initdb not found" fi } # postgres_installed reports whether a PostgreSQL client and server unit are present. # The unit list is read into a variable before matching: `systemctl list-unit-files | # grep -q` dies of SIGPIPE under pipefail once grep stops reading a list longer than # one write, and the install then took the "not installed" branch on a host that has it. postgres_installed() { local units command -v psql >/dev/null 2>&1 || return 1 units="$(systemctl list-unit-files 2>/dev/null)" || return 1 grep -q '^postgresql' <<<"$units" } install_postgres() { if postgres_installed; then ok "postgresql already installed" else log "installing postgresql" case "$PKG" in apt) pkg_install postgresql ;; dnf) pkg_install postgresql-server postgresql ;; yum) pkg_install postgresql-server postgresql ;; zypper) pkg_install postgresql-server postgresql ;; pacman) pkg_install postgresql ;; esac fi init_postgres_data_dir check_postgres_major systemctl enable --now postgresql ok "postgresql running" } # check_postgres_major refuses to start a PostgreSQL server whose major version differs from # the one that created the data directory. The server would not start anyway; this says why # and what to do, instead of a failed unit in the middle of the install. Debian and Ubuntu # keep one cluster per version under /var/lib/postgresql/ and upgrade with # pg_upgradecluster, so they are left to their own tooling. check_postgres_major() { local data_dir have want [ "$PKG" = "apt" ] && return 0 data_dir="$(postgres_data_dir)" [ -f "${data_dir}/PG_VERSION" ] || return 0 have="$(tr -d '[:space:]' < "${data_dir}/PG_VERSION")" want="$( (postgres --version 2>/dev/null || psql --version 2>/dev/null) | head -n 1 \ | sed -nE 's/^[^0-9]*([0-9]+)\..*/\1/p')" [ -n "$want" ] && [ "$have" != "$want" ] || return 0 die "PostgreSQL ${want} is installed, but ${data_dir} holds a PostgreSQL ${have} cluster. The server cannot open it. Upgrade the cluster first (pg_upgrade, with the ${have} binaries still installed), or reinstall PostgreSQL ${have}, then rerun the installer. Take a database bundle before either: sudo felis db backup" } configure_postgres() { local cfg hba listen cfg="$(as_postgres psql -tAc 'SHOW config_file;' 2>/dev/null || true)" hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)" [ -n "$cfg" ] && [ -n "$hba" ] || die "could not query postgresql config/hba file paths" listen="$(as_postgres psql -tAc 'SHOW listen_addresses;' 2>/dev/null || true)" # Listen on all interfaces (applied on restart). ALTER SYSTEM is idempotent. Pods reach the # database at the node IP, and configure_postgres_firewall keeps the port from everyone else. # # The connection itself is not encrypted (sslmode=disable in felis.toml). On this # single-node shape it never leaves the host: pods reach the node IP over their veth pair # and the host binary uses loopback, so TLS would guard a path no other machine is on. as_postgres psql -v ON_ERROR_STOP=1 -c "ALTER SYSTEM SET listen_addresses = '*';" >/dev/null # Allow the host loopback, the pod CIDR, and the node IP before broader distro defaults. write_pg_hba_block "$hba" # Role + database (idempotent), and (re)set the password to our generated one. as_postgres psql -v ON_ERROR_STOP=1 </dev/null SET password_encryption = 'scram-sha-256'; DO \$\$ BEGIN IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = '${DB_USER}') THEN CREATE ROLE ${DB_USER} LOGIN PASSWORD '${DB_PASSWORD}'; END IF; END \$\$; ALTER ROLE ${DB_USER} WITH LOGIN PASSWORD '${DB_PASSWORD}'; SQL if ! as_postgres psql -tAc "SELECT 1 FROM pg_database WHERE datname='${DB_NAME}'" | grep -q 1; then as_postgres createdb -O "$DB_USER" "$DB_NAME" fi # listen_addresses is the only setting here that needs a restart, and a restart cuts # every connection felis-api holds mid-transaction. pg_hba.conf and the role are live # after a reload. if [ "$listen" = "*" ]; then as_postgres psql -v ON_ERROR_STOP=1 -tAc 'SELECT pg_reload_conf();' >/dev/null else systemctl restart postgresql fi configure_postgres_firewall ok "postgresql configured (listen=*, role/db '${DB_NAME}', pg_hba opened to pods)" } # configure_postgres_firewall keeps 5432 to this host and its pods. PostgreSQL listens on # every address (pods dial the node IP), and pg_hba.conf only refuses a connection after the # server has spoken to it. firewalld's default zone does not open 5432 (configure_k3s_firewall # trusts only the pod and service CIDRs), so that host needs nothing more; on any other host a # small nftables table of our own drops 5432 from everywhere but loopback, the pod CIDR and the # node's own address, loaded at boot by a oneshot unit ordered before PostgreSQL. configure_postgres_firewall() { if command -v firewall-cmd >/dev/null 2>&1 && systemctl is-active --quiet firewalld; then return 0 fi command -v nft >/dev/null 2>&1 || pkg_install nftables command -v nft >/dev/null 2>&1 || { warn "nft is not available; PostgreSQL's 5432 is reachable from the network (pg_hba still refuses other hosts)"; return 0; } local node_rule case "$NODE_IP" in *:*) node_rule="ip6 saddr ${NODE_IP}" ;; *) node_rule="ip saddr ${NODE_IP}" ;; esac install -d -m 0700 "$STATE_DIR" cat > "$PG_FIREWALL_RULES" < "$PG_FIREWALL_SERVICE" </dev/null 2>&1 systemctl restart felis-postgres-firewall.service ok "5432 accepts loopback, ${POD_CIDR} and ${NODE_IP} only (nftables table felis_postgres)" } # --------------------------------------------------------------------------- # 7. Secrets + felis.toml (pod variant reaches Postgres at the node IP; host # variant at 127.0.0.1 for migrations) # --------------------------------------------------------------------------- load_or_make_secrets() { mkdir -p "$STATE_DIR" chmod 0700 "$STATE_DIR" if [ -f "$SECRETS_ENV" ]; then # shellcheck disable=SC1090 . "$SECRETS_ENV" ok "reusing persisted secrets from ${SECRETS_ENV}" fi DB_PASSWORD="${DB_PASSWORD:-$(openssl rand -hex 24)}" # One felis-api internal token per caller (naming.CallerTokens), so each is scoped # to its own routes and a leak is contained to that caller: SERVICE_TOKEN is the # proxy's (felis-link.properties), LIMBO_TOKEN the login gate's, BUILD_TOKEN what a # build Job fetches its context with, OPS_TOKEN what `felis backup-now` presents. # `felis rotate-token ` rewrites the matching line here. SERVICE_TOKEN="${SERVICE_TOKEN:-$(openssl rand -hex 32)}" LIMBO_TOKEN="${LIMBO_TOKEN:-$(openssl rand -hex 32)}" BUILD_TOKEN="${BUILD_TOKEN:-$(openssl rand -hex 32)}" OPS_TOKEN="${OPS_TOKEN:-$(openssl rand -hex 32)}" SESSION_SECRET="${SESSION_SECRET:-$(openssl rand -hex 32)}" # The Velocity modern-forwarding key. It is what makes a backend's UUID trustworthy: # the proxy does the Mojang handshake and HMACs the resulting profile with this key, # and a backend that cannot verify it would fall back to an offline UUID derived from # the username — i.e. anyone could join as anyone, the Owner included. Same value on # the proxy (forwarding.secret) and in every backend pod (felis-forwarding-secret). FORWARDING_SECRET="${FORWARDING_SECRET:-$(openssl rand -hex 32)}" # Registry write credentials, one per principal the registry gate knows # (internal/registrygate): platform pushes the installer's own images and the # Trivy DB mirrors, build is what a build Job's push container presents and may # never write under felis/ or mirror/, prune is felis-api deleting manifests # nothing references (internal/registryprune). Reads stay anonymous. REGISTRY_PLATFORM_TOKEN="${REGISTRY_PLATFORM_TOKEN:-$(openssl rand -hex 32)}" REGISTRY_BUILD_TOKEN="${REGISTRY_BUILD_TOKEN:-$(openssl rand -hex 32)}" REGISTRY_PRUNE_TOKEN="${REGISTRY_PRUNE_TOKEN:-$(openssl rand -hex 32)}" ( umask 077 cat > "$SECRETS_ENV" < "$conf" </dev/null 2>&1 chmod 0600 "$PANEL_TLS_KEY" chmod 0644 "$PANEL_TLS_CERT" ok "panel TLS certificate ready (${PANEL_TLS_CERT})" } # persisted_smtp_block echoes the [smtp] section an earlier run left behind, or # nothing. Unlike every other value in the generated toml, [smtp] is not derived # from this script's inputs -- `felis setup`'s SMTP screen writes it, after # proving the relay works. A wholesale `cat >` therefore erases it on every # re-run, and since re-running the installer is the documented way to update # felis-api, an operator who updates loses mail: OTP delivery silently reverts # to the no-Mailer path and every code is logged instead of sent. Same defect # family as the root_domain loss fixed in ecbeb20 -- generated file, hand-set # value, no carry-forward. # # The carry is the section's header and key lines only. Printing every line up to # the next section header hoarded the generated [[auth_source]] comment block # that sits below [smtp] into this carry: each re-run then re-emitted the hoard # plus a fresh template copy, growing both config files by one comment block per # run (audit #50). The extraction is idempotent, which is also why re-reading the # freshly rewritten host file on the pod pass is safe. The pod toml remains the # fallback for a host file with no [smtp] section at all. persisted_smtp_block() { local f out for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do [ -r "$f" ] || continue out="$(awk ' /^[[:space:]]*\[/ { if (insmtp) exit insmtp = ($0 ~ /^[[:space:]]*\[smtp\][[:space:]]*$/) if (insmtp) print next } insmtp && /^[[:space:]]*("[A-Za-z_][A-Za-z0-9_]*"|[A-Za-z_][A-Za-z0-9_]*)[[:space:]]*=/ { print } ' "$f")" [ -n "$out" ] || continue printf '%s' "$out" return 0 done } # persisted_auth_lines echoes the key lines of the [auth] section an earlier run left # behind, or nothing. Two writers own keys there that nothing in this script's inputs # derives: the Cloudflare edge setup (access_jwt_aud, client_ip_header -- the header the # sign-in rate limit keys on) and an operator who serves the panel or the admin console # on a name other than console. / op.console.. A wholesale rewrite dropped # all of them on every re-run: behind Cloudflare the rate limit fell back to the tunnel's # address, one bucket for everyone. Header-and-keys only, like persisted_smtp_block, and # the same first-readable-file rule. persisted_auth_lines() { local f out for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do [ -r "$f" ] || continue out="$(awk ' /^[[:space:]]*\[/ { if (inauth) exit inauth = ($0 ~ /^[[:space:]]*\[auth\][[:space:]]*$/) next } inauth && /^[[:space:]]*[A-Za-z_][A-Za-z0-9_]*[[:space:]]*=/ { print } ' "$f")" [ -n "$out" ] || continue printf '%s\n' "$out" return 0 done } # auth_hostname echoes the installed value of an [auth] hostname key, or the default # derived from the root domain: `auth_hostname panel_hostname "console.${FELIS_ROOT_DOMAIN}"`. # The lines are read into a variable before matching, as in auth_lines. auth_hostname() { local lines v lines="$(persisted_auth_lines)" v="$(awk -F'"' -v k="$1" '$1 ~ "^[[:space:]]*" k "[[:space:]]*=[[:space:]]*$" { print $2; exit }' <<<"$lines")" printf '%s' "${v:-$2}" } # auth_lines is the body of the [auth] section this run writes: the carried keys, after # the two hostnames derived from the root domain when the carry lacks them. A first # install gets exactly the two derived lines; a re-run reproduces the carried section. # The keys are matched in a here-string, never `printf | grep -q`: grep exits at the # match, printf dies of SIGPIPE on the lines after it, pipefail fails the test, and a # second admin_hostname then breaks the file (see postgres_installed). auth_lines() { local carried carried="$(persisted_auth_lines)" grep -Eq '^[[:space:]]*admin_hostname[[:space:]]*=' <<<"$carried" || printf 'admin_hostname = "op.console.%s"\n' "$FELIS_ROOT_DOMAIN" grep -Eq '^[[:space:]]*panel_hostname[[:space:]]*=' <<<"$carried" || printf 'panel_hostname = "console.%s"\n' "$FELIS_ROOT_DOMAIN" if [ -n "$carried" ]; then printf '%s\n' "$carried"; fi } # persisted_auth_source_blocks echoes the [[auth_source]] tables an earlier run left # behind, or the LittleSkin default when there is no earlier felis.toml at all. The # list is the operator's: it is the only way to add or drop a Yggdrasil root on a full # install, and nothing in this script's inputs derives it. Without the carry-forward a # re-run would put LittleSkin back after the operator removed it and silently drop any # root they added. An earlier file with no tables stays that way — that is a Mojang-only # server, not a missing value. Same first-readable-file rule as persisted_smtp_block. persisted_auth_source_blocks() { local f for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do [ -r "$f" ] || continue # Every [[auth_source]] table, up to (not including) the next other section header. # TOML also accepts [[ auth_source ]] and a quoted key; a header this does not # recognise would silently drop that table. awk '/^[[:space:]]*\[/ { f = /^[[:space:]]*\[\[[[:space:]]*["\047]?auth_source["\047]?[[:space:]]*\]\]/ } f { print }' "$f" return 0 done printf '%s\n' '[[auth_source]]' 'tag = "littleskin"' 'prefix = "LS"' \ 'url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined"' } # persisted_archive_block echoes the operator-owned [archive] keys an earlier run # left behind — the retention window, the pre-reap warn offsets, the local cap, # the on-demand backup retention/count/cooldown — so a re-run does not silently # revert them to the built-ins (felis reaper, felis backup and felis api read # these from the config Secret; defaults: 90d retention, 3d/1d warnings, no cap, # manual backups kept 30d, 5 per server, one per 10m). store and local_path are # NOT carried: this script owns them (FELIS_ARCHIVE_LOCAL_PATH must equal the # mount). Same first-readable-file rule as persisted_smtp_block; warn_before must # be a single-line TOML array (the shape every writer here emits). persisted_archive_block() { local f out for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do [ -r "$f" ] || continue out="$(awk ' /^[[:space:]]*\[/ { sect = $0; next } sect ~ /^[[:space:]]*\[archive\][[:space:]]*$/ && /^[[:space:]]*(retention|warn_before|max_local_bytes|manual_retention|manual_keep|manual_cooldown)[[:space:]]*=/ { print } ' "$f")" [ -n "$out" ] || continue printf '%s\n' "$out" return 0 done } # persisted_registry_block echoes the operator-owned [registry] keys an earlier run # left behind — the build-lane executor mirrors, the resource caps, the uploads # backend and its [registry.s3] subtable — so §15's upgrade path (re-run the # installer) does not silently revert them. Nothing in this script's inputs # derives these: they are hand-written per docs/troubleshooting.md §8e or stamped # by the storage wizard. url and build_namespace are NOT carried: this script # owns them (they must match REGISTRY_URL / BUILD_NS). Same first-readable-file # rule as persisted_smtp_block. persisted_registry_block() { local f out for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do [ -r "$f" ] || continue out="$(awk ' /^[[:space:]]*\[/ { sect = $0; next } sect ~ /^[[:space:]]*\[registry\][[:space:]]*$/ && /^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|max_concurrent_builds|scan_fail_on|scan_fail_unfixed|scan_accept|user_uploads_context|user_uploads_max_bytes|context_max_bytes)[[:space:]]*=/ { print } sect ~ /^[[:space:]]*\[registry\.s3\][[:space:]]*$/ && /^[[:space:]]*[A-Za-z_]+[[:space:]]*=/ { if (!s3hdr) { printf "[registry.s3]\n"; s3hdr = 1 } print } ' "$f")" [ -n "$out" ] || continue printf '%s\n' "$out" return 0 done } # persisted_offsite_block echoes the [offsite] section an earlier run (or the operator) # left behind: after the first install the bucket is configured by editing felis.host.toml # or by re-running with FELIS_OFFSITE_*, and a plain re-run must keep it. Deleting the # section and re-running is how the off-site copy is turned off. Same first-readable-file # rule as persisted_smtp_block. persisted_offsite_block() { local f out for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do [ -r "$f" ] || continue out="$(awk ' /^[[:space:]]*\[/ { if (inoff) exit inoff = ($0 ~ /^[[:space:]]*\[offsite\][[:space:]]*$/) if (inoff) print next } inoff && /^[[:space:]]*(endpoint|region|bucket|prefix|access_key_ref|secret_key_ref|key_ref|db_keep)[[:space:]]*=/ { print } ' "$f")" [ -n "$out" ] || continue printf '%s\n' "$out" return 0 done } # offsite_block is the [offsite] section this run writes: from FELIS_OFFSITE_* when the # bucket is given, else the one an earlier run left. offsite_block() { if [ -z "$FELIS_OFFSITE_BUCKET" ]; then persisted_offsite_block return 0 fi printf '[offsite]\n' printf 'endpoint = "%s"\n' "$FELIS_OFFSITE_ENDPOINT" printf 'bucket = "%s"\n' "$FELIS_OFFSITE_BUCKET" if [ -n "$FELIS_OFFSITE_REGION" ]; then printf 'region = "%s"\n' "$FELIS_OFFSITE_REGION"; fi if [ -n "$FELIS_OFFSITE_PREFIX" ]; then printf 'prefix = "%s"\n' "$FELIS_OFFSITE_PREFIX"; fi if [ -n "$FELIS_OFFSITE_DB_KEEP" ]; then printf 'db_keep = %s\n' "$FELIS_OFFSITE_DB_KEEP"; fi } # offsite_enabled: the [offsite] section this run writes names a bucket. offsite_enabled() { offsite_block | grep -Eq '^[[:space:]]*bucket[[:space:]]*=[[:space:]]*"[^"]+"' } write_felis_toml() { local target="$1" db_host="$2" smtp_block auth_body auth_source_blocks registry_block archive_block offsite_section smtp_block="$(persisted_smtp_block)" if [ -n "$smtp_block" ]; then log "carrying forward the configured [smtp] relay" smtp_block="${smtp_block}"$'\n' # keep a blank line before the next section fi if [ -n "$(persisted_auth_lines)" ]; then log "carrying forward the configured [auth] keys" fi auth_body="$(auth_lines)" auth_source_blocks="$(persisted_auth_source_blocks)" registry_block="$(persisted_registry_block)" if [ -n "$registry_block" ]; then log "carrying forward the configured [registry] overrides" registry_block="${registry_block}"$'\n' # keep a blank line before the next section fi archive_block="$(persisted_archive_block)" if [ -n "$archive_block" ]; then log "carrying forward the configured [archive] overrides" archive_block="${archive_block}"$'\n' # keep a blank line before the next section fi # The pod copy needs it as much as the host one: the reaper reads [offsite] from the # config Secret to know it must wait for each archive's off-site copy. offsite_section="$(offsite_block)" if [ -n "$offsite_section" ]; then offsite_section="${offsite_section}"$'\n\n' # keep a blank line before the next section fi cat > "$target" < [server] listen = "0.0.0.0:8080" root_domain = "${FELIS_ROOT_DOMAIN}" [database] url = "postgres://${DB_USER}:${DB_PASSWORD}@${db_host}:5432/${DB_NAME}?sslmode=disable" [k8s] namespace = "${MINECRAFT_NS}" egress_mode = "${FELIS_EGRESS_MODE}" [velocity] # The two always-on system servers that felis setup provisions. They are built and imported # into k3s by build_game_stack below, so setup never has to be told "build these first". login_image = "${FELIS_LIMBO_IMAGE}" lobby_image = "${FELIS_LOBBY_IMAGE}" # The public port players connect on; the panel shows it in server addresses. game_port = ${FELIS_GAME_PORT} [registry] url = "${REGISTRY_URL}" build_namespace = "${BUILD_NS}" ${registry_block} [archive] store = "tarLocal" local_path = "${FELIS_ARCHIVE_LOCAL_PATH}" ${archive_block} ${offsite_section}[auth] ${auth_body} ${smtp_block} # Third-party Yggdrasil sources federated by the hasJoined multiplexer. Mojang is # always the code-owned identity anchor (premium-first), prepended in Go; sources here # append as namespace-rewritten guests. A fresh install federates LittleSkin. Edit the # list in ${STATE_DIR}/felis.host.toml and rerun the installer; re-runs keep it as it # is, and with no [[auth_source]] at all the server is Mojang-only. # Order is trust: the first source that answers 200 wins, so list the most trusted roots # first, and remove a compromised root rather than just moving it down. # A tag is permanent: it is hashed into every player UUID of its source, so changing it # (even its case) gives all of them new UUIDs and orphans their data, links and bans. ${auth_source_blocks} EOF } ensure_default_config() { local target="${STATE_DIR}/felis.toml" backup if [ -L "$target" ] && [ "$(readlink "$target")" = "${STATE_DIR}/felis.host.toml" ]; then ok "default host config already points at ${STATE_DIR}/felis.host.toml" return 0 fi if [ -e "$target" ] || [ -L "$target" ]; then if bootstrap_from_tui; then backup="${target}.bak.$(date -u +%Y%m%d%H%M%S).$$" warn "replacing existing ${target}; backup saved at ${backup}" mv "$target" "$backup" ln -s "${STATE_DIR}/felis.host.toml" "$target" ok "default host config: ${target} -> ${STATE_DIR}/felis.host.toml" return 0 fi warn "leaving existing ${target}; setup can use -config ${STATE_DIR}/felis.host.toml if needed" return 0 fi ln -s "${STATE_DIR}/felis.host.toml" "$target" ok "default host config: ${target} -> ${STATE_DIR}/felis.host.toml" } # --------------------------------------------------------------------------- # 8. Migrate + deploy bundle # --------------------------------------------------------------------------- run_migrations() { local backup_flags=(-backup-dir "$FELIS_DB_BACKUP_DIR") write_felis_toml "${STATE_DIR}/felis.host.toml" "127.0.0.1" ensure_default_config if [ "$FELIS_PRE_MIGRATE_BACKUP" = 0 ]; then warn "FELIS_PRE_MIGRATE_BACKUP=0: pending migrations run without a database snapshot" backup_flags=(-no-backup) fi # Migrations only roll forward. On an existing database with migrations pending, the # binary bundles the database into FELIS_DB_BACKUP_DIR first and refuses to migrate # if that fails; a fresh database has nothing to protect and is migrated directly. log "running database migrations (host binary -> 127.0.0.1)" # From here the database may move forward, and the binary that moved it stays. HOST_BIN_IN_USE=1 "$HOST_BIN" migrate up -config "${STATE_DIR}/felis.host.toml" "${backup_flags[@]}" ok "migrations applied" } # Silence felis watchdog's mail for the rest of this install (see WATCHDOG_QUIET_FILE). # Two hours covers a slow source build; cleanup lifts it as soon as the installer exits. quiet_watchdog() { install -d -m 0755 "$(dirname "$WATCHDOG_QUIET_FILE")" printf '%s\n' "$(( $(date +%s) + 7200 ))" > "$WATCHDOG_QUIET_FILE" } # The hourly off-site copy. The bucket is checked now, in the install, so wrong # credentials or an unreachable endpoint show up here; the first copy itself runs in the # background, since a host with many archives can take a long while to upload them. # With no bucket configured the timer is removed (the operator deleted [offsite]) and the # install says loudly that every backup is on this machine only. install_offsite_timer() { if [ "${OFFSITE_ENABLED:-0}" != 1 ]; then if [ -e "$OFFSITE_TIMER" ] || [ -e "$OFFSITE_SERVICE" ]; then systemctl disable --now felis-offsite.timer >/dev/null 2>&1 || true rm -f "$OFFSITE_TIMER" "$OFFSITE_SERVICE" systemctl daemon-reload warn "no [offsite] bucket is configured any more; the off-site copy timer was removed" fi return 0 fi cat > "$OFFSITE_SERVICE" < "$OFFSITE_TIMER" </dev/null; then systemctl start --no-block felis-offsite.service ok "off-site copy: hourly to the [offsite] bucket, first copy started (sudo felis offsite status; journalctl -u felis-offsite)" else warn "the [offsite] bucket did not answer (error above); nothing is copied off this machine until it does: fix ${OFFSITE_ENV} or [offsite] in ${STATE_DIR}/felis.host.toml, then sudo systemctl start felis-offsite.service" fi } # summary_offsite is the installer's last word on where the backups live. summary_offsite() { if [ "${OFFSITE_ENABLED:-0}" != 1 ]; then warn "NO OFF-SITE COPY: every world archive and database backup is on this machine only." warn "Losing its disk loses them all. Set FELIS_OFFSITE_BUCKET, FELIS_OFFSITE_ENDPOINT," warn "FELIS_OFFSITE_ACCESS_KEY and FELIS_OFFSITE_SECRET_KEY and re-run (docs/troubleshooting.md §16)." return 0 fi if [ "${OFFSITE_KEY_NEW:-0}" = 1 ]; then echo warn "================================================================================" warn "The off-site copies are encrypted with this key. Store it NOW somewhere other than" warn "this machine (a password manager): without it nothing in the bucket can be read." warn "" warn " FELIS_OFFSITE_KEY=${FELIS_OFFSITE_KEY}" warn "" warn "It is also in ${OFFSITE_ENV}, which is lost with this machine." warn "================================================================================" else log "Off-site copy: the encryption key is in ${OFFSITE_ENV}; keep a copy of it off this machine." fi } # The platform watchdog: every two minutes it checks the control plane, the login gate, # the fleet, PostgreSQL, the game proxy, the database backups and the host's disks and # memory, and mails the owners (their verified addresses, over the [smtp] relay) what # has stayed wrong long enough to matter. It runs on the host so a k3s that is down is # still reported. The first run happens now, so a broken unit shows up in this install. # The daily version check. Felis applies no update on its own; `felis update --record` # compares what this host runs with the newest upstream releases and stores the result, # which the panel's Updates page shows with the command that applies each update. It runs # on the host because that is where the installed versions are readable. The first check # runs in the background: it waits on the release feeds, and nothing in the install # depends on it. install_update_check_timer() { cat > "$UPDATE_CHECK_SERVICE" < "$UPDATE_CHECK_TIMER" < "$WATCHDOG_SERVICE" < "$WATCHDOG_TIMER" <&2 || true warn "the first watchdog run failed (log above); nothing will be mailed until it runs: sudo systemctl start felis-watchdog.service" fi } # The build lane's tools: kaniko and trivy (pinned by digest in internal/build/tools.go) # and Trivy's vulnerability and Java DBs, copied into the registry's mirror/ where build # Jobs pull them; the build namespace has no internet egress. The timer refreshes the DBs # twice a day (upstream publishes every six hours) and the watchdog warns when three days # pass without a clean run. The first copy starts now in the background: the Java DB # alone is several hundred MB. install_build_tools_timer() { install -d -m 0755 "$(dirname "$BUILD_TOOLS_STATUS")" cat > "$BUILD_TOOLS_SERVICE" < "$BUILD_TOOLS_TIMER" < "$DB_BACKUP_SERVICE" < "$DB_BACKUP_TIMER" <&2 || true warn "the first database backup failed (log above); fix it before relying on the daily timer: sudo systemctl start felis-db-backup.service" fi } # configure_offsite writes OFFSITE_ENV, the secrets behind [offsite]: the bucket's access # keys and the key every off-site object is sealed with. Values in this run's environment # replace the file's (rotating the bucket credentials is a re-run); the encryption key is # generated when neither has one. A different key than the file's is refused: every object # already in the bucket is sealed with the old one, and swapping it would make them # unreadable without a word. configure_offsite() { OFFSITE_ENABLED=0 OFFSITE_KEY_NEW=0 offsite_enabled || return 0 OFFSITE_ENABLED=1 local env_ak="${FELIS_OFFSITE_ACCESS_KEY:-}" env_sk="${FELIS_OFFSITE_SECRET_KEY:-}" env_key="${FELIS_OFFSITE_KEY:-}" local file_key="" FELIS_OFFSITE_ACCESS_KEY="" FELIS_OFFSITE_SECRET_KEY="" FELIS_OFFSITE_KEY="" if [ -f "$OFFSITE_ENV" ]; then # shellcheck disable=SC1090 . "$OFFSITE_ENV" file_key="$FELIS_OFFSITE_KEY" fi FELIS_OFFSITE_ACCESS_KEY="${env_ak:-$FELIS_OFFSITE_ACCESS_KEY}" FELIS_OFFSITE_SECRET_KEY="${env_sk:-$FELIS_OFFSITE_SECRET_KEY}" if [ -n "$env_key" ] && [ -n "$file_key" ] && [ "$env_key" != "$file_key" ]; then die "FELIS_OFFSITE_KEY differs from the key in ${OFFSITE_ENV}; the objects already in the bucket are sealed with that one. Unset FELIS_OFFSITE_KEY to keep it (docs/troubleshooting.md §16)" fi FELIS_OFFSITE_KEY="${env_key:-$file_key}" if [ -z "$FELIS_OFFSITE_ACCESS_KEY" ] || [ -z "$FELIS_OFFSITE_SECRET_KEY" ]; then die "[offsite] names a bucket but there are no credentials for it: set FELIS_OFFSITE_ACCESS_KEY and FELIS_OFFSITE_SECRET_KEY (they are kept in ${OFFSITE_ENV})" fi if [ -z "$FELIS_OFFSITE_KEY" ]; then FELIS_OFFSITE_KEY="$(openssl rand -base64 32)" OFFSITE_KEY_NEW=1 fi ( umask 077 cat > "$OFFSITE_ENV" </dev/null 2>&1 && getfacl -cpn "$dir" 2>/dev/null | grep -q '^user:1000:'; then if setfacl -x u:1000 "$dir"; then log "revoked the old uid-1000 traverse grant on ${dir}" else warn "could not revoke the old uid-1000 traverse grant on ${dir}; remove it with: setfacl -x u:1000 ${dir}" fi fi done if [ -d "$K3S_STORAGE_ROOT" ] && [ -n "$(find "$K3S_STORAGE_ROOT" -maxdepth 0 -perm -o=x)" ]; then if chmod o-rwx "$K3S_STORAGE_ROOT"; then log "revoked the old world-traversable mode on ${K3S_STORAGE_ROOT} (back to k3s's 0700)" else warn "could not restore ${K3S_STORAGE_ROOT} to 0700; any local account can reach the world volumes below it: chmod o-rwx ${K3S_STORAGE_ROOT}" fi fi } deploy_bundle() { local prev_api prev_operator prev_gate export KUBECONFIG=/etc/rancher/k3s/k3s.yaml write_felis_toml "${STATE_DIR}/felis.pod.toml" "${NODE_IP}" prev_api="$(deployment_image felis-api api)" prev_operator="$(deployment_image felis-operator operator)" prev_gate="$(deployment_image registry registry-gate)" if [ -n "$prev_api" ] && [ "$prev_api" != "$FELIS_IMAGE" ]; then PREVIOUS_FELIS_IMAGE="$prev_api" printf '%s\n' "$prev_api" > "${STATE_DIR}/previous-felis-image" fi # Always the embedded copy. It is byte-identical to deploy/crd/ (bootstrap_asset.go embeds # that very file), it needs no checkout — which the release-download path does not have — # and one source beats a branch whose two arms have to be kept in agreement by hand. log "applying MinecraftServer CRD" "$HOST_BIN" bootstrap-assets crd | kube apply -f - log "ensuring namespaces" local ns for ns in "$CONTROL_NS" "$MINECRAFT_NS" "$BUILD_NS"; do kube create namespace "$ns" --dry-run=client -o yaml | kube apply -f - done log "provisioning felis-config + internal caller tokens + felis-forwarding-secret + registry credentials + panel TLS secrets (out-of-band, never in the bundle)" apply_felis_config_secrets # felis-api mounts all four caller tokens from the control namespace. The login # gate's and the build Job's are also applied into the namespace their pods run in # (a secretKeyRef is namespace-local); applying them here rather than leaving it to # `felis setup` means an upgrade has them in place before the new operator points # the login pod at felis-limbo-token. The proxy's and the ops token stay here only. apply_literal_secret "$CONTROL_NS" felis-service-token token "$SERVICE_TOKEN" apply_literal_secret "$CONTROL_NS" felis-limbo-token token "$LIMBO_TOKEN" apply_literal_secret "$MINECRAFT_NS" felis-limbo-token token "$LIMBO_TOKEN" apply_literal_secret "$CONTROL_NS" felis-build-token token "$BUILD_TOKEN" apply_literal_secret "$BUILD_NS" felis-build-token token "$BUILD_TOKEN" apply_literal_secret "$CONTROL_NS" felis-ops-token token "$OPS_TOKEN" # The forwarding key every backend verifies the proxy's handshake with. `felis setup` # replicates it into the minecraft namespace (ensureSecretReplica) before it creates # the pods that mount it; the operator injects it into EVERY backend, because Velocity's # forwarding mode is one proxy-wide setting — a backend that does not speak it is not # "less secure", it is unjoinable. apply_literal_secret "$CONTROL_NS" felis-forwarding-secret secret "$FORWARDING_SECRET" apply_registry_secrets kube -n "$CONTROL_NS" create secret tls felis-api-tls \ --cert="$PANEL_TLS_CERT" \ --key="$PANEL_TLS_KEY" \ --dry-run=client -o yaml | kube apply -f - revoke_worlds_root_grant log "rendering + applying the control-plane bundle" local -a manifest_args=( --felis-image "$FELIS_IMAGE" --panel-node-port "$FELIS_PANEL_NODEPORT" --velocity-cidr "${NODE_IP}/32" ) local cidr while read -r cidr; do [ -n "$cidr" ] && manifest_args+=(--server-egress-deny-cidr "$cidr") done < <(node_global_cidrs) # Backups are on by default (the renderer's own default names felis-backups); an emptied # FELIS_BACKUP_PVC asks for the no-backup shape explicitly, and a custom name must be # passed through or the api would advertise a PVC the bundle never created. if [ -n "$FELIS_BACKUP_PVC" ]; then manifest_args+=(--backup-pvc "$FELIS_BACKUP_PVC") else manifest_args+=(--backup-pvc=) fi # With an archive store the reaper always renders, since backups past their expiry have # to leave it; it reaps idle worlds only when the operator names where the worlds live. # The archive path must equal the [archive] local_path written above. if [ -n "$FELIS_BACKUP_PVC" ]; then manifest_args+=(--archive-local-path "$FELIS_ARCHIVE_LOCAL_PATH") if [ -z "$FELIS_WORLDS_HOST_PATH" ]; then log "idle-world retention is off: the daily reaper deletes expired backups and keeps every world (set FELIS_WORLDS_HOST_PATH=${K3S_STORAGE_ROOT} to reap worlds idle for 15 days)" fi fi if [ -n "$FELIS_WORLDS_HOST_PATH" ]; then log "retention enabled: the daily reaper will read worlds from ${FELIS_WORLDS_HOST_PATH}" # The reaper reads this root as root with DAC_OVERRIDE (platform.reaperPodSecurityContext) # through a static hostPath PV, so the host directory keeps k3s's own 0700 root:root and # needs no extra grant. It must exist, though: the PV declares type Directory. if [ ! -d "$FELIS_WORLDS_HOST_PATH" ]; then if [ "$FELIS_WORLDS_HOST_PATH" = "$K3S_STORAGE_ROOT" ]; then # The provisioner would create it moments later with this same 0700 root:root; # creating it now keeps a fresh install from warning about its own default. install -d -m 0700 -o root -g root "$FELIS_WORLDS_HOST_PATH" else warn "worlds root ${FELIS_WORLDS_HOST_PATH} does not exist yet; the reaper CronJob cannot start until it does (hostPath type Directory)" fi fi manifest_args+=(--worlds-host-path "$FELIS_WORLDS_HOST_PATH") fi local size size="$(pvc_size "$CONTROL_NS" registry "$FELIS_REGISTRY_STORAGE" FELIS_REGISTRY_STORAGE)" if [ -n "$size" ]; then manifest_args+=(--registry-storage "$size"); fi size="$(pvc_size "$CONTROL_NS" felis-uploads "$FELIS_UPLOADS_STORAGE" FELIS_UPLOADS_STORAGE)" if [ -n "$size" ]; then manifest_args+=(--uploads-storage "$size"); fi if [ -n "$FELIS_BACKUP_PVC" ]; then size="$(pvc_size "$MINECRAFT_NS" "$FELIS_BACKUP_PVC" "$FELIS_BACKUP_STORAGE" FELIS_BACKUP_STORAGE)" if [ -n "$size" ]; then manifest_args+=(--backup-storage "$size"); fi fi "$HOST_BIN" manifests "${manifest_args[@]}" | kube apply -f - restart_existing_control_plane "$prev_api" "$prev_operator" "$prev_gate" log "waiting for control-plane rollouts" local d for d in $(kube -n "$CONTROL_NS" get deploy -o name); do if ! kube -n "$CONTROL_NS" rollout status "$d" --timeout=180s; then diagnose_rollout "$d" die "control-plane rollout did not complete: ${d}" fi done # Before per-caller tokens the proxy's token was replicated into the workload # namespaces for the login gate and the build Jobs. The new operator and api no # longer reference those copies; leaving them would keep the proxy's credential # readable from namespaces that have no business with it. kube -n "$MINECRAFT_NS" delete secret felis-service-token --ignore-not-found kube -n "$BUILD_NS" delete secret felis-service-token --ignore-not-found } # pvc_size prints the size to render the claim # with: its current request when it exists, else the wanted size (empty = the renderer's # default). A claim's request can only grow, and only on a storage class that allows # expansion (k3s local-path does not), so re-applying a different size would fail the # whole apply; a mismatch is reported and left to the operator. pvc_size() { local ns="$1" claim="$2" want="$3" env="$4" have have="$(kube -n "$ns" get pvc "$claim" -o jsonpath='{.spec.resources.requests.storage}' 2>/dev/null || true)" if [ -z "$have" ]; then printf '%s' "$want" return 0 fi if [ -n "$want" ] && [ "$want" != "$have" ]; then warn "PVC ${ns}/${claim} already requests ${have}; keeping it (${env}=${want} applies to a new claim; grow this one with kubectl patch where its storage class allows expansion)" fi printf '%s' "$have" } # deployment_image prints the image that container of a # control-plane Deployment runs now, or nothing when the Deployment does not exist yet. deployment_image() { kube -n "$CONTROL_NS" get deployment "$1" \ -o "jsonpath={.spec.template.spec.containers[?(@.name==\"$2\")].image}" 2>/dev/null || true } # restart_existing_control_plane restarts the Deployments the bundle apply left as they were: # those that already ran FELIS_IMAGE, whose tag now names a rebuilt image (a rerun of the same # version, or a FELIS_IMAGE the operator reuses). A Deployment whose image changed is rolling # from the apply already, and must not be restarted on top: the restart is a second template # change, so `rollout undo` would step back to the new image instead of the previous release. # # The registry pod runs the same binary in its registry-gate and registry-gc containers, so it # follows the same rule; the rollout wait below covers it before anything is pushed. restart_existing_control_plane() { local prev_api="$1" prev_operator="$2" prev_gate="${3:-}" # `if`, not `[ test ] && cmd`: as the LAST command of the function the and-list returns 1 # when the test is false, which becomes the function's exit status and kills the whole # install under `set -Eeuo pipefail` — right after the bundle is applied and before the # rollout wait. if [ "$prev_api" = "$FELIS_IMAGE" ]; then log "restarting felis-api onto the rebuilt ${FELIS_IMAGE}" kube -n "$CONTROL_NS" rollout restart deployment/felis-api fi if [ "$prev_operator" = "$FELIS_IMAGE" ]; then log "restarting felis-operator onto the rebuilt ${FELIS_IMAGE}" kube -n "$CONTROL_NS" rollout restart deployment/felis-operator fi if [ "$prev_gate" = "$FELIS_IMAGE" ]; then log "restarting the registry's gate onto the rebuilt ${FELIS_IMAGE}" kube -n "$CONTROL_NS" rollout restart deployment/registry fi } # push_image_to_registry re-tags a locally built image for the node's # loopback push endpoint and uploads it. The registry keys a repository by the # path AFTER the host, so pushing 127.0.0.1:5000/felis/felis:demo lands exactly # where a later kubelet pull of registry.felis.svc:5000/felis/felis:demo (the # mirror rewrites the host) will look. A ref not under REGISTRY_URL is not # mirrored — warn, don't fail: the install is still self-consistent, that image # just has no pull source once GC collects its containerd copy. push_image_to_registry() { local ref="$1" push_ref case "$ref" in "${REGISTRY_URL}/"*) push_ref="${REGISTRY_PUSH_HOST}/${ref#"${REGISTRY_URL}/"}" ;; *) warn "not mirroring ${ref} into the internal registry: it is not under ${REGISTRY_URL}; once the image GC collects that tag, nothing can re-pull it" return 0 ;; esac log "mirroring ${ref} into the internal registry" docker tag "$ref" "$push_ref" || die "could not tag ${ref} as ${push_ref} — is docker healthy?" local attempt=1 until docker --config "$REGISTRY_DOCKER_CONFIG" push "$push_ref"; do # The gate answers writes 503 while the registry-gc sidecar sweeps; wait # that out, and fail at once on anything else. if [ "$attempt" -ge 40 ] || ! registry_read_only; then die "could not mirror ${ref} into the internal registry — check the registry Deployment/pod (the registry, registry-gate and registry-gc containers) and its PVC" fi warn "the registry is read-only for garbage collection; retrying the push of ${push_ref} in 30s (${attempt}/40)" attempt=$((attempt + 1)) sleep 30 done docker rmi "$push_ref" >/dev/null 2>&1 || true } # registry_read_only asks the gate, over its pod-loopback maintenance listener, # whether a garbage-collection window is open. registry_read_only() { kubectl -n "$CONTROL_NS" exec deploy/registry -c registry-gc -- \ wget -q -O /dev/null "http://127.0.0.1:$((${REGISTRY_URL##*:} + 2))/readonly" >/dev/null 2>&1 } # registry_docker_login logs a throwaway docker config into the registry gate as # the platform principal: writes are refused anonymously, and this identity is # the only one allowed under felis/. The config lives in a 0700 temp dir that the # EXIT trap removes, so the token never lands in root's ~/.docker. registry_docker_login() { REGISTRY_DOCKER_CONFIG="$(umask 077; mktemp -d)" remember_temp "$REGISTRY_DOCKER_CONFIG" printf '%s' "$REGISTRY_PLATFORM_TOKEN" | docker --config "$REGISTRY_DOCKER_CONFIG" \ login --username platform --password-stdin "$REGISTRY_PUSH_HOST" >/dev/null \ || die "could not log in to the internal registry at ${REGISTRY_PUSH_HOST} as platform — check the registry-gate container's log and the felis-registry-auth Secret" } # Every image this installer builds is hosted in the registry, so the copies it # imported into containerd are a first-boot cache, not the only copy: kubelet # re-pulls from the registry after any image GC. Runs AFTER deploy_bundle — the # registry it pushes into does not exist before that. # # Docker is started once for the whole batch and stopped once at the end. A # start/stop pair per image trips systemd's start rate limit — observed live on # a re-run: three fast pushes, then "Start request repeated too quickly / # start-limit-hit" and the fourth image never got mirrored. docker.service is # socket-triggered, so each cycle counts twice against the burst limit. push_images_to_registry() { local img systemctl start docker registry_docker_login for img in "$FELIS_IMAGE" "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do [ -n "$img" ] || continue push_image_to_registry "$img" done for img in "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do [ -n "$img" ] || continue push_version_tag "$img" done systemctl stop docker docker.socket 2>/dev/null || true } # push_version_tag mirrors a game image a second time under a tag no later run # rewrites: -, e.g. # felis/paper:26.2-3f9c0a1b2c4d. The :demo tag moves with every run, and servers are # pinned to the digest it named when they were created, so this is the readable name # for each build: an admin can whitelist it to create servers on that exact # Minecraft version long after :demo has moved on. push_version_tag() { local ref="$1" id versioned [ -n "${MC_VERSION:-}" ] || return 0 id="$(docker image inspect -f '{{.Id}}' "$ref" 2>/dev/null)" || return 0 id="${id#sha256:}" versioned="${ref%:*}:${MC_VERSION}-${id:0:12}" docker tag "$ref" "$versioned" || die "could not tag ${ref} as ${versioned} — is docker healthy?" push_image_to_registry "$versioned" docker rmi "$versioned" >/dev/null 2>&1 || true } # The login/lobby images are built under mutable :demo tags, and an existing # StatefulSet whose template still names that tag will not roll onto a new build by # itself. When a build changed, each system server is pinned to the digest its tag # names now (felis pin-images --system, after push_images_to_registry): the new ref # changes the template, the operator rolls the pod onto it, and the build each one # runs is written in its spec. A pin that fails (registry down) falls back to # recreating the pod, which picks the build up only while the spec names the bare # tag. Only a changed build does either: every player online is on one of these # two, and a rerun that rebuilt nothing has nothing to start. SYSTEM_SERVER_IMAGES # records the build each was last moved to; it is written after the loop, so a run # that died in between moves them next time. restart_existing_system_servers() { local name id pods next="" while read -r name id; do [ -n "$name" ] || continue next="${next}${name} ${id}"$'\n' if [ -n "$id" ] && grep -qxF "${name} ${id}" "$SYSTEM_SERVER_IMAGES" 2>/dev/null; then ok "${name} system server already runs this build; left running" continue fi if "$HOST_BIN" pin-images --system "$name" --namespace "$MINECRAFT_NS" \ --registry "$REGISTRY_URL" --endpoint "$REGISTRY_PUSH_HOST"; then continue fi pods="$(kube -n "$MINECRAFT_NS" get pod \ -l "felis.lolicon.best/server=${name}" -o name 2>/dev/null || true)" [ -n "$pods" ] || continue warn "could not pin the ${name} system server to its new build (above); restarting its pod, which starts the new build only if its spec.image still names the bare tag. Rerun the installer once the registry answers." kube -n "$MINECRAFT_NS" delete pod \ -l "felis.lolicon.best/server=${name}" --wait=false done < "$SYSTEM_SERVER_IMAGES" } diagnose_rollout() { local deploy="$1" name selector pod name="${deploy##*/}" warn "rollout not complete: ${deploy}" kube -n "$CONTROL_NS" describe "$deploy" || true case "$name" in felis-api) selector='app.kubernetes.io/name=felis,app.kubernetes.io/component=api' ;; felis-operator) selector='app.kubernetes.io/name=felis,app.kubernetes.io/component=operator' ;; registry) selector='app.kubernetes.io/name=felis,app.kubernetes.io/component=registry' ;; *) selector='' ;; esac [ -n "$selector" ] || return 0 kube -n "$CONTROL_NS" get pods -l "$selector" -o wide || true for pod in $(kube -n "$CONTROL_NS" get pods -l "$selector" -o name 2>/dev/null); do warn "pod detail: ${pod}" kube -n "$CONTROL_NS" describe "$pod" || true warn "recent logs: ${pod}" kube -n "$CONTROL_NS" logs "$pod" --all-containers --tail=120 || true kube -n "$CONTROL_NS" logs "$pod" --all-containers --previous --tail=120 || true done } mark_bootstrap_done() { date -u +%Y-%m-%dT%H:%M:%SZ > "$BOOTSTRAP_DONE" chmod 0644 "$BOOTSTRAP_DONE" } # --------------------------------------------------------------------------- # 9. Summary # --------------------------------------------------------------------------- summary() { export KUBECONFIG=/etc/rancher/k3s/k3s.yaml echo ok "Felis control plane deployed." echo kube -n "$CONTROL_NS" get pods -o wide || true echo systemctl --no-pager --full status felis-velocity 2>/dev/null | head -n 4 || true echo log "Player panel: https://$(auth_hostname panel_hostname "console.${FELIS_ROOT_DOMAIN}") — served on 443 once your edge/Cloudflare Tunnel routes it here." log "Operator console (Op/Admin/Owner): https://$(auth_hostname admin_hostname "op.console.${FELIS_ROOT_DOMAIN}") — the Owner runs 'felis setup' and onboards here." log "Before the edge is ready: direct + self-signed at https://${NODE_IP}:${FELIS_PANEL_NODEPORT} (browser will warn on first visit)." log "Minecraft address: ${NODE_IP}:${FELIS_GAME_PORT} (point mc.${FELIS_ROOT_DOMAIN} here)" log "The proxy authenticates against Mojang and forwards the verified profile to the" log "login gate; the backends are reachable in-cluster only. Follow it with:" log " sudo journalctl -u felis-velocity -f" if [ "${FELIS_BOOTSTRAP_FROM_TUI:-}" = "1" ]; then log "Returning to the setup console to create the Owner account and verify panel access." else log "Next: run 'sudo felis setup' on this host to create the Owner account." fi log "setup provisions the login/lobby servers, then asks the Owner to bind by joining" log "the proxy in Minecraft — that is what makes the Owner's admin identity a real" log "Mojang account rather than a password." log "Use 'sudo felis breakGlass' only for emergency local Owner recovery/reset." if [ -n "${PREVIOUS_FELIS_IMAGE:-}" ]; then echo log "The control plane moved from ${PREVIOUS_FELIS_IMAGE} to ${FELIS_IMAGE}" log "(also recorded in ${STATE_DIR}/previous-felis-image). To go back to it:" log " kubectl -n ${CONTROL_NS} rollout undo deployment/felis-api deployment/felis-operator" log "The database stays migrated; docs/troubleshooting.md §16 has the full rollback." fi echo summary_offsite echo } # --------------------------------------------------------------------------- # 10. Felis-nano install path — the auth multiplexer only: felis binary + a # minimal [[auth_source]] config + a systemd unit running `felis nano`. # No k3s, no Postgres, no control-plane bundle. Chosen at the top-of-run # prompt (or FELIS_INSTALL_MODE=nano). # --------------------------------------------------------------------------- prompt_install_mode() { # felis setup carries on to the Owner and edge setup, which needs the control plane, so # a nano install under it could only end in a setup error. if bootstrap_from_tui; then [ "$INSTALL_MODE" != nano ] || die "felis setup installs the full control plane; for Felis-nano run deploy/bootstrap.sh with FELIS_INSTALL_MODE=nano" INSTALL_MODE="full" log "install mode: full (felis setup)" return 0 fi case "$INSTALL_MODE" in full|nano) log "install mode: ${INSTALL_MODE} (from FELIS_INSTALL_MODE)"; return 0 ;; "") ;; *) die "FELIS_INSTALL_MODE must be 'full' or 'nano', got: ${INSTALL_MODE}" ;; esac # A felis-nano unit with no full install beside it makes this re-run a nano update; # defaulting to full there would put k3s and Postgres on a host that asked for neither. local def=full n=1 reply if [ -e "$NANO_SERVICE" ] && [ ! -e "$BOOTSTRAP_DONE" ]; then def=nano n=2; fi # No override: ask on the controlling terminal. Under `curl | sudo bash` stdin # is the script, so we must read /dev/tty, not stdin. No tty (CI/cloud-init) → # take the default. Open it to find out: /dev/tty is mode 0666 on every Linux host, so # `-r` passes even when there is no controlling terminal and only the open fails. if ! (: /dev/null; then INSTALL_MODE="$def" log "no terminal for a prompt; defaulting to a ${def} install (set FELIS_INSTALL_MODE=full or nano to override)" return 0 fi printf '\n' printf 'What do you want to install on this host?\n' printf ' [1] Felis — full control plane (k3s + Postgres + panel; orchestrates Minecraft servers)\n' printf ' [2] Felis-nano — auth multiplexer only (federates Mojang + third-party Yggdrasil; no k3s/DB)\n' while :; do printf 'Choose [1/2] (default %s): ' "$n" IFS= read -r reply &1 | grep -qw nano || die "built binary has no 'nano' subcommand" # Stage-then-install, never `go build -o ${HOST_BIN}` directly: the Go linker renames # its output out of $TMPDIR, and a same-filesystem rename CARRIES THE SOURCE SELinux # label — the binary lands in /usr/local/bin still labelled user_tmp_t. Root (being # unconfined) can still run it, so it looks fine by hand, but the DynamicUser service # cannot exec it and felis-nano dies with 203/EXEC. Creating the file fresh at the # destination lets the policy's type transition label it bin_t; restorecon is the belt. mkdir -p "$(dirname "$HOST_BIN")" keep_previous_host_binary rm -f "$HOST_BIN" install -m 0755 "$staged" "$HOST_BIN" rm -f "$staged" command -v restorecon >/dev/null 2>&1 && restorecon "$HOST_BIN" >/dev/null 2>&1 || true ok "felis binary on host at ${HOST_BIN}" } acquire_nano_binary() { # The release channel takes the same prebuilt binary the control plane does. This is the # biggest win on this path: a host that only wants the auth multiplexer stops needing a Go # toolchain and a checkout at all. if use_release_binary; then resolve_install_ref if download_release_binary; then return 0 fi fi # Raw curl|bash with no usable release: build it straight from source with a pinned Go # toolchain — nano needs one static binary, not an image, so dragging in docker # (as the full control-plane path does) buys nothing and costs a daemon that must # start. It does not start on EL10: the docker-ce el10 rpms install but dockerd # fails, which used to kill the whole nano install at `systemctl enable --now docker`. fetch_source install_go_toolchain build_nano_binary } write_nano_config() { local target="${STATE_DIR}/felis.toml" # The unit is a DynamicUser, so it can read felis.toml only if it can search this # directory. The mode is explicit because a hardened root umask (027) would leave it 0750, # which is what older installers did to nano-only hosts. Only the full install keeps its # own 0700: that directory holds secrets, and install_nano_service reports the lockout # rather than this widening it. if [ ! -d "$STATE_DIR" ]; then mkdir -p "$STATE_DIR" chmod 0755 "$STATE_DIR" elif [ ! -e "$SECRETS_ENV" ] && [ ! -e "$BOOTSTRAP_DONE" ]; then chmod 0755 "$STATE_DIR" fi if [ -e "$target" ]; then ok "config already present at ${target}; leaving it (edit it to add [[auth_source]] roots)" return 0 fi cat > "$target" <<'EOF' # Felis-nano — Yggdrasil hasJoined multiplexer. # Mojang is always the first (identity) source, added in code — do NOT list it here. # Add each third-party Yggdrasil root below (priority = order). url is the FULL # hasJoined endpoint. After editing: sudo systemctl restart felis-nano # # Order is trust: the first source that answers 200 wins, so list the most trusted roots # first, and remove a compromised root rather than just moving it down. # tag is permanent: it is hashed into every player UUID of its source, so changing it # (even its case) gives all of them new UUIDs and orphans their data, links and bans. # # prefix is required, 1-4 letters/digits, unique per source. A player of this source # whose name belongs to a Mojang account joins as PREFIX_name (LS_steve) instead — # otherwise the proxy, which keys its player list on the NAME, refuses to have both # online at once ("You are already connected to this proxy!"). Everyone else keeps # their own name. # # [[auth_source]] # tag = "littleskin" # prefix = "LS" # url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined" EOF chmod 0644 "$target" ok "wrote nano config template ${target} (edit it to add your Yggdrasil sources)" } # resolve_nano_listen settles FELIS_NANO_LISTEN: the operator's value, else the address the # installed felis-nano unit listens on, else loopback. Re-running this script is how a nano # host updates, and without the middle step that re-run moved an off-host proxy's endpoint # back to 127.0.0.1, so every login through it failed. resolve_nano_listen() { if [ -z "$FELIS_NANO_LISTEN" ] && [ -r "$NANO_SERVICE" ]; then FELIS_NANO_LISTEN="$(sed -n 's/^ExecStart=.* -listen \([^ ]*\).*$/\1/p' "$NANO_SERVICE")" fi FELIS_NANO_LISTEN="${FELIS_NANO_LISTEN:-127.0.0.1:8081}" } nano_listen_is_loopback() { case "${FELIS_NANO_LISTEN%:*}" in 127.*|localhost|"[::1]") return 0 ;; *) return 1 ;; esac } configure_nano_firewall() { # A loopback bind is unreachable from off-host by construction, so opening the port # would advertise a hole nothing answers on. Only punch it for a routable bind. if nano_listen_is_loopback; then ok "nano listens on ${FELIS_NANO_LISTEN} (loopback); no firewall port opened" return 0 fi command -v firewall-cmd >/dev/null 2>&1 || return 0 systemctl is-active --quiet firewalld || return 0 local port="${FELIS_NANO_LISTEN##*:}" family=ipv4 # hasJoined takes no token, so the port is opened to the proxy alone. Earlier installers # opened it to every source, and a re-run must not leave that behind. A rule for a previous # FELIS_NANO_PROXY_CIDR is not tracked; it stays until removed by hand. if firewall-cmd --permanent --query-port="${port}/tcp" >/dev/null 2>&1; then log "closing firewalld port ${port}/tcp, which an earlier install opened to every source" firewall-cmd --permanent --remove-port="${port}/tcp" fi if [ -n "$FELIS_NANO_PROXY_CIDR" ]; then case "$FELIS_NANO_PROXY_CIDR" in *:*) family=ipv6 ;; esac log "opening firewalld port ${port}/tcp to ${FELIS_NANO_PROXY_CIDR} only" firewall-cmd --permanent --add-rich-rule="rule family=\"${family}\" source address=\"${FELIS_NANO_PROXY_CIDR}\" port port=\"${port}\" protocol=\"tcp\" accept" else warn "no FELIS_NANO_PROXY_CIDR, so firewalld keeps ${port}/tcp closed; the summary shows how to admit your proxy" fi firewall-cmd --reload } install_nano_service() { # DynamicUser: no static account, ephemeral UID. nano writes nothing (logs go to # journald) and only reads the world-readable felis.toml, so strict sandboxing fits. cat > "$NANO_SERVICE" </dev/null | head -n 6 || true echo log "hasJoined endpoint: http://${host}:${port}/session/minecraft/hasJoined" log "Point Velocity at it — add to the proxy JVM startup flags (between java and -jar):" log " -Dmojang.sessionserver=http://${host}:${port}/session/minecraft/hasJoined" log " (the FULL endpoint URL, path included — Velocity's default for this property is" log " the full https://sessionserver.mojang.com/session/minecraft/hasJoined)" if nano_listen_is_loopback; then log "Bound to loopback: reachable from Velocity on THIS host, and from nowhere else." log "Proxy on another machine? Re-run with the address on the sudo line (sudo drops" log "exported variables):" log " curl -fsSL /deploy/bootstrap.sh | sudo FELIS_NANO_LISTEN=:${port} FELIS_NANO_PROXY_CIDR=/32 bash" log "firewalld then admits ${port}/tcp ONLY from that proxy — hasJoined takes no auth token," log "so an internet-facing one is a free auth relay burning your Mojang egress IP." elif [ -n "$FELIS_NANO_PROXY_CIDR" ]; then log "Bound to ${FELIS_NANO_LISTEN}. firewalld, where it runs, admits ${port}/tcp only from" log "${FELIS_NANO_PROXY_CIDR}; any other firewall in front of this host must do the same." else log "WARNING: bound to ${FELIS_NANO_LISTEN} with no FELIS_NANO_PROXY_CIDR. hasJoined takes no auth" log "token, so admit ${port}/tcp from your proxy alone, or anyone can relay their logins" log "through you. firewalld, where it runs, keeps the port closed until you add:" log " firewall-cmd --permanent --add-rich-rule='rule family=\"ipv4\" source address=\"/32\" port port=\"${port}\" protocol=\"tcp\" accept' && firewall-cmd --reload" fi log "Then edit ${STATE_DIR}/felis.toml to add your [[auth_source]] roots and run:" log " sudo systemctl restart felis-nano" log "Follow live login traffic with: sudo journalctl -u felis-nano -f" echo } main_nano() { detect_node_ip install_base acquire_nano_binary write_nano_config install_nano_service configure_nano_firewall summary_nano } main() { resolve_nano_listen validate_settings detect_os prompt_install_mode if [ "$INSTALL_MODE" = "nano" ]; then main_nano return fi quiet_watchdog pause_package_background_timers detect_node_ip warn_dynamic_node_ip ensure_swap install_base ensure_time_sync ensure_persistent_journal # Right after install_base because it is the first point curl exists, and well before # docker and k3s: a missing FELIS_GITHUB_TOKEN or an unpublished release should cost # the operator seconds, not a k3s install they then have to unwind. This is purely # fail-fast — fetch_source resolves again itself — so it must skip on exactly the paths # that never consume the result, or it invents a network dependency and a version they # do not have: the TUI rebuilds the binary it is already running, and FELIS_SKIP_FETCH # builds whatever is staged, which stamp_version reads the SHA off. Resolving anyway # would set FELIS_VERSION to the newest tag and stamp a staged tree as that release. bootstrap_from_tui || [ -n "${FELIS_SKIP_FETCH:-}" ] || resolve_install_ref install_cloudflared load_or_make_secrets configure_offsite ensure_panel_tls_cert install_docker install_k3s # The registry mirror must exist before the bundle's pods start pulling (and # before any re-run's rollouts); the registry's own image must be in containerd # before its Deployment can start at all. configure_registry_mirror import_registry_image # Three ways to end up with a felis binary, in preference order. The release download is # the only one that skips compiling: it is the CI artifact for this exact tag, panel # included. Both other arms leave HAVE_PREBUILT_BINARY unset where a source build is what # actually happens, which is what routes build_image below. if bootstrap_from_tui; then install_embedded_binary elif use_release_binary && download_release_binary; then : else fetch_source fi resolve_felis_image build_image # After build_image imported the felis image: the registry pod's gate runs it. pin_registry_images # Before build_game_stack: the builds user servers run must be read off the # registry's tags before new ones replace them. pin_user_server_images build_game_stack install_postgres configure_postgres run_migrations deploy_bundle # AFTER deploy_bundle: the registry the built images are mirrored into is part # of that bundle. push_images_to_registry # After deploy_bundle, like the pushes: it writes into the registry. install_build_tools_timer restart_existing_system_servers # After deploy_bundle: the proxy dials felis-api's internal ClusterIP, which does not # exist until the bundle is applied. install_velocity # After deploy_bundle: the bundle's MinecraftServer export reads the cluster. install_db_backup_timer # After the backup timer: its first bundle is part of the first copy. install_offsite_timer # After install_velocity: the check reads the installed proxy jar's version. install_update_check_timer # Last: its first run should see the platform as this install leaves it. install_watchdog_timer mark_bootstrap_done summary } main "$@"