From ecbeb20761411a205a32ae3a90b6990cf015cbb0 Mon Sep 17 00:00:00 2001 From: Minseong Choi Date: Mon, 20 Jul 2026 20:25:33 +0900 Subject: [PATCH] fix(bootstrap): reuse the installed root domain instead of re-deriving it detect_node_ip recomputed FELIS_ROOT_DOMAIN from scratch on every run and fell back to .nip.io. Nothing read the domain back out of the felis.toml an earlier run wrote, so it survived only as long as the operator kept passing the same environment. That made re-running the installer destructive on any install with a real domain, and re-running it is not optional: it is the only way to move felis-api to a newer release, which is what `felis update` points operators at. A bare re-run rewrote root_domain, panel_hostname and admin_hostname to nip.io names while ensure_panel_tls_cert returned early on the certificate it had already written, leaving the console serving a cert for hostnames it no longer answered to -- with no re-domain flow to recover through. Precedence is now explicit FELIS_ROOT_DOMAIN, then the domain the last run persisted, then the nip.io default. First installs are unaffected. Deliberate re-domains still work, because there is no other route to one, but they now warn that the write-once certificate is not reissued and that the proxy and login config carry the old name too. Secrets were never exposed to this: load_or_make_secrets has always sourced secrets.env before generating anything. The domain was the one piece of install identity with no read-back. The channel is deliberately left alone. FELIS_VERSION_BOOTSTRAP is not persisted either, but defaulting a re-run to the release channel installs a working build rather than breaking one, so cmd/felis/update.go states that instead. Its warning about the domain went with the bug and would now be false. Verified against the shipped function text: the ladder holds for a fresh host, a re-run with and without the variable set, a re-domain, and a felis.toml whose root_domain is missing or empty. Reverting the one line reproduces the nip.io overwrite. --- cmd/felis/update.go | 15 +++++++-------- cmd/felis/update_test.go | 13 ++++++------- deploy/bootstrap.sh | 37 +++++++++++++++++++++++++++++++++++-- 3 files changed, 48 insertions(+), 17 deletions(-) diff --git a/cmd/felis/update.go b/cmd/felis/update.go index d913ece..e5fc10c 100644 --- a/cmd/felis/update.go +++ b/cmd/felis/update.go @@ -262,15 +262,14 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo // rebuilds the image and rolls the deployment from the SAME binary -- a run that looks // like a successful update and leaves the version unchanged. // - // The installer is the only thing that moves felis-api, but it is NOT an updater and - // must not be recommended as one without this warning. detect_node_ip re-derives - // FELIS_ROOT_DOMAIN on every run and defaults it to .nip.io -- nothing reads - // the domain back out of the felis.toml a previous run wrote. A bare re-run therefore - // rewrites root-domain/panel-hostname/admin-hostname to nip.io names while - // ensure_panel_tls_cert, which is write-once, keeps serving the old ones: the console - // stops matching its own certificate. There is no re-domain flow to recover with. + // The installer is the only thing that moves felis-api. It is safe to point at now + // that detect_node_ip reuses the installed root domain, so what is left to warn about + // is the channel: FELIS_VERSION_BOOTSTRAP is not persisted anywhere and defaults to + // release, so a bare re-run on a host tracking main quietly moves it onto releases. + // That is a channel change, not a broken install, which is why it is one clause and + // not a paragraph. if offeredFelisAPI { - b.WriteString("\nfelis-api (panel, plugins) is the exception: setup re-images it from the felis binary\nalready on this host, so it cannot install a NEWER felis-api. Only re-running the\nbootstrap installer does that, and it is a full install run, not an update: give it the\nSAME environment as the original install, FELIS_ROOT_DOMAIN above all. It defaults to\n.nip.io, and a bare re-run re-domains this install while the write-once panel\ncertificate keeps the old hostnames.\n") + b.WriteString("\nfelis-api (panel, plugins) is the exception: setup re-images it from the felis binary\nalready on this host, so it cannot install a NEWER felis-api. Re-run the bootstrap\ninstaller for that -- it keeps this install's root domain. It does default to the\nrelease channel, so pass FELIS_VERSION_BOOTSTRAP=dev if this host tracks main.\n") } return b.String() } diff --git a/cmd/felis/update_test.go b/cmd/felis/update_test.go index b12ccd7..3283e98 100644 --- a/cmd/felis/update_test.go +++ b/cmd/felis/update_test.go @@ -148,13 +148,12 @@ func TestApplyGuidanceScopesTheFelisAPICaveat(t *testing.T) { if !strings.Contains(api, caveat) { t.Fatalf("--panel resolves to felis-api and must carry the caveat:\n%s", api) } - // Naming the installer without naming this is worse than saying nothing: the README - // invocation carries no environment, detect_node_ip then defaults FELIS_ROOT_DOMAIN - // to .nip.io, and the operator re-domains a live install by following our - // own advice. Nothing reads the previous domain back, and the write-once panel cert - // keeps the old hostnames, so there is no recovery path either. - if !strings.Contains(api, "FELIS_ROOT_DOMAIN") { - t.Fatalf("pointing at the installer without the re-domain warning is a footgun:\n%s", api) + // Naming the installer obliges us to name what a bare re-run still changes. The domain + // is handled -- detect_node_ip reuses the installed one -- but the channel is not + // persisted at all and defaults to release, so a host tracking main gets moved onto + // releases by following this advice. + if !strings.Contains(api, "FELIS_VERSION_BOOTSTRAP=dev") { + t.Fatalf("pointing at the installer without the channel caveat misleads a dev host:\n%s", api) } vel := renderApplyGuidance( diff --git a/deploy/bootstrap.sh b/deploy/bootstrap.sh index 628f672..06b9bc7 100644 --- a/deploy/bootstrap.sh +++ b/deploy/bootstrap.sh @@ -432,12 +432,45 @@ detect_os() { log "host: ${PRETTY_NAME:-$OS_ID $OS_VERSION} (package manager: ${PKG})" } +# persisted_root_domain echoes the root_domain an earlier run wrote, or nothing. The +# generated toml is the only durable record of it: nothing else on the host stores the +# domain, and it is written on every successful install. +persisted_root_domain() { + local f="${STATE_DIR}/felis.host.toml" + [ -r "$f" ] || return 0 + awk -F'"' '/^[[:space:]]*root_domain[[:space:]]*=/ { print $2; exit }' "$f" +} + detect_node_ip() { NODE_IP="$(ip -4 route get 1.1.1.1 2>/dev/null | awk '{for(i=1;i<=NF;i++) if($i=="src"){print $(i+1); exit}}')" [ -n "${NODE_IP:-}" ] || NODE_IP="$(hostname -I 2>/dev/null | awk '{print $1}')" [ -n "${NODE_IP:-}" ] || die "could not determine this host's primary IPv4 address" - FELIS_ROOT_DOMAIN="${FELIS_ROOT_DOMAIN:-${NODE_IP}.nip.io}" - log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN}" + + # Precedence: an explicit FELIS_ROOT_DOMAIN, then whatever the last run persisted, then + # the nip.io default. The middle step is what makes a re-run idempotent. Without it this + # installer re-derived the domain from scratch every time and defaulted to nip.io, so + # re-running it on a live install -- the only way to move felis-api to a newer release, + # and what `felis update` points operators at -- rewrote root_domain, panel_hostname and + # admin_hostname to nip.io names. ensure_panel_tls_cert is write-once and kept serving a + # certificate for the OLD hostnames, so the console stopped matching its own cert, with + # no re-domain flow to recover through. Secrets never had this problem: + # load_or_make_secrets has always sourced secrets.env before generating anything. + local persisted + persisted="$(persisted_root_domain)" + if [ -n "${FELIS_ROOT_DOMAIN:-}" ] && [ -n "$persisted" ] && [ "$FELIS_ROOT_DOMAIN" != "$persisted" ]; then + # Deliberate re-domain. Allowed -- there is no other route to it -- but it is not a + # thing this script finishes: the panel certificate, the two secrets, the velocity + # config and the login CR all still carry the old name. + warn "FELIS_ROOT_DOMAIN (${FELIS_ROOT_DOMAIN}) differs from the installed ${persisted}." + warn "This re-domains the install. The write-once panel certificate is NOT reissued and" + warn "will keep the old hostnames; the proxy and login config need the same treatment." + fi + FELIS_ROOT_DOMAIN="${FELIS_ROOT_DOMAIN:-${persisted:-${NODE_IP}.nip.io}}" + if [ -n "$persisted" ] && [ "$FELIS_ROOT_DOMAIN" = "$persisted" ]; then + log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN} (reusing the installed domain)" + else + log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN}" + fi } pkg_install() {