fix(bootstrap): reuse the installed root domain instead of re-deriving it
detect_node_ip recomputed FELIS_ROOT_DOMAIN from scratch on every run and fell back to <node-ip>.nip.io. Nothing read the domain back out of the felis.toml an earlier run wrote, so it survived only as long as the operator kept passing the same environment. That made re-running the installer destructive on any install with a real domain, and re-running it is not optional: it is the only way to move felis-api to a newer release, which is what `felis update` points operators at. A bare re-run rewrote root_domain, panel_hostname and admin_hostname to nip.io names while ensure_panel_tls_cert returned early on the certificate it had already written, leaving the console serving a cert for hostnames it no longer answered to -- with no re-domain flow to recover through. Precedence is now explicit FELIS_ROOT_DOMAIN, then the domain the last run persisted, then the nip.io default. First installs are unaffected. Deliberate re-domains still work, because there is no other route to one, but they now warn that the write-once certificate is not reissued and that the proxy and login config carry the old name too. Secrets were never exposed to this: load_or_make_secrets has always sourced secrets.env before generating anything. The domain was the one piece of install identity with no read-back. The channel is deliberately left alone. FELIS_VERSION_BOOTSTRAP is not persisted either, but defaulting a re-run to the release channel installs a working build rather than breaking one, so cmd/felis/update.go states that instead. Its warning about the domain went with the bug and would now be false. Verified against the shipped function text: the ladder holds for a fresh host, a re-run with and without the variable set, a re-domain, and a felis.toml whose root_domain is missing or empty. Reverting the one line reproduces the nip.io overwrite.
This commit is contained in:
3 files changed
+48
-17
No files matched your search
+7
-8
@@ -262,15 +262,14 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
|||||||
// rebuilds the image and rolls the deployment from the SAME binary -- a run that looks
|
// rebuilds the image and rolls the deployment from the SAME binary -- a run that looks
|
||||||
// like a successful update and leaves the version unchanged.
|
// like a successful update and leaves the version unchanged.
|
||||||
//
|
//
|
||||||
// The installer is the only thing that moves felis-api, but it is NOT an updater and
|
// The installer is the only thing that moves felis-api. It is safe to point at now
|
||||||
// must not be recommended as one without this warning. detect_node_ip re-derives
|
// that detect_node_ip reuses the installed root domain, so what is left to warn about
|
||||||
// FELIS_ROOT_DOMAIN on every run and defaults it to <node-ip>.nip.io -- nothing reads
|
// is the channel: FELIS_VERSION_BOOTSTRAP is not persisted anywhere and defaults to
|
||||||
// the domain back out of the felis.toml a previous run wrote. A bare re-run therefore
|
// release, so a bare re-run on a host tracking main quietly moves it onto releases.
|
||||||
// rewrites root-domain/panel-hostname/admin-hostname to nip.io names while
|
// That is a channel change, not a broken install, which is why it is one clause and
|
||||||
// ensure_panel_tls_cert, which is write-once, keeps serving the old ones: the console
|
// not a paragraph.
|
||||||
// stops matching its own certificate. There is no re-domain flow to recover with.
|
|
||||||
if offeredFelisAPI {
|
if offeredFelisAPI {
|
||||||
b.WriteString("\nfelis-api (panel, plugins) is the exception: setup re-images it from the felis binary\nalready on this host, so it cannot install a NEWER felis-api. Only re-running the\nbootstrap installer does that, and it is a full install run, not an update: give it the\nSAME environment as the original install, FELIS_ROOT_DOMAIN above all. It defaults to\n<node-ip>.nip.io, and a bare re-run re-domains this install while the write-once panel\ncertificate keeps the old hostnames.\n")
|
b.WriteString("\nfelis-api (panel, plugins) is the exception: setup re-images it from the felis binary\nalready on this host, so it cannot install a NEWER felis-api. Re-run the bootstrap\ninstaller for that -- it keeps this install's root domain. It does default to the\nrelease channel, so pass FELIS_VERSION_BOOTSTRAP=dev if this host tracks main.\n")
|
||||||
}
|
}
|
||||||
return b.String()
|
return b.String()
|
||||||
}
|
}
|
||||||
@@ -148,13 +148,12 @@ func TestApplyGuidanceScopesTheFelisAPICaveat(t *testing.T) {
|
|||||||
if !strings.Contains(api, caveat) {
|
if !strings.Contains(api, caveat) {
|
||||||
t.Fatalf("--panel resolves to felis-api and must carry the caveat:\n%s", api)
|
t.Fatalf("--panel resolves to felis-api and must carry the caveat:\n%s", api)
|
||||||
}
|
}
|
||||||
// Naming the installer without naming this is worse than saying nothing: the README
|
// Naming the installer obliges us to name what a bare re-run still changes. The domain
|
||||||
// invocation carries no environment, detect_node_ip then defaults FELIS_ROOT_DOMAIN
|
// is handled -- detect_node_ip reuses the installed one -- but the channel is not
|
||||||
// to <node-ip>.nip.io, and the operator re-domains a live install by following our
|
// persisted at all and defaults to release, so a host tracking main gets moved onto
|
||||||
// own advice. Nothing reads the previous domain back, and the write-once panel cert
|
// releases by following this advice.
|
||||||
// keeps the old hostnames, so there is no recovery path either.
|
if !strings.Contains(api, "FELIS_VERSION_BOOTSTRAP=dev") {
|
||||||
if !strings.Contains(api, "FELIS_ROOT_DOMAIN") {
|
t.Fatalf("pointing at the installer without the channel caveat misleads a dev host:\n%s", api)
|
||||||
t.Fatalf("pointing at the installer without the re-domain warning is a footgun:\n%s", api)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
vel := renderApplyGuidance(
|
vel := renderApplyGuidance(
|
||||||
|
|||||||
+35
-2
@@ -432,12 +432,45 @@ detect_os() {
|
|||||||
log "host: ${PRETTY_NAME:-$OS_ID $OS_VERSION} (package manager: ${PKG})"
|
log "host: ${PRETTY_NAME:-$OS_ID $OS_VERSION} (package manager: ${PKG})"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# persisted_root_domain echoes the root_domain an earlier run wrote, or nothing. The
|
||||||
|
# generated toml is the only durable record of it: nothing else on the host stores the
|
||||||
|
# domain, and it is written on every successful install.
|
||||||
|
persisted_root_domain() {
|
||||||
|
local f="${STATE_DIR}/felis.host.toml"
|
||||||
|
[ -r "$f" ] || return 0
|
||||||
|
awk -F'"' '/^[[:space:]]*root_domain[[:space:]]*=/ { print $2; exit }' "$f"
|
||||||
|
}
|
||||||
|
|
||||||
detect_node_ip() {
|
detect_node_ip() {
|
||||||
NODE_IP="$(ip -4 route get 1.1.1.1 2>/dev/null | awk '{for(i=1;i<=NF;i++) if($i=="src"){print $(i+1); exit}}')"
|
NODE_IP="$(ip -4 route get 1.1.1.1 2>/dev/null | awk '{for(i=1;i<=NF;i++) if($i=="src"){print $(i+1); exit}}')"
|
||||||
[ -n "${NODE_IP:-}" ] || NODE_IP="$(hostname -I 2>/dev/null | awk '{print $1}')"
|
[ -n "${NODE_IP:-}" ] || NODE_IP="$(hostname -I 2>/dev/null | awk '{print $1}')"
|
||||||
[ -n "${NODE_IP:-}" ] || die "could not determine this host's primary IPv4 address"
|
[ -n "${NODE_IP:-}" ] || die "could not determine this host's primary IPv4 address"
|
||||||
FELIS_ROOT_DOMAIN="${FELIS_ROOT_DOMAIN:-${NODE_IP}.nip.io}"
|
|
||||||
log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN}"
|
# Precedence: an explicit FELIS_ROOT_DOMAIN, then whatever the last run persisted, then
|
||||||
|
# the nip.io default. The middle step is what makes a re-run idempotent. Without it this
|
||||||
|
# installer re-derived the domain from scratch every time and defaulted to nip.io, so
|
||||||
|
# re-running it on a live install -- the only way to move felis-api to a newer release,
|
||||||
|
# and what `felis update` points operators at -- rewrote root_domain, panel_hostname and
|
||||||
|
# admin_hostname to nip.io names. ensure_panel_tls_cert is write-once and kept serving a
|
||||||
|
# certificate for the OLD hostnames, so the console stopped matching its own cert, with
|
||||||
|
# no re-domain flow to recover through. Secrets never had this problem:
|
||||||
|
# load_or_make_secrets has always sourced secrets.env before generating anything.
|
||||||
|
local persisted
|
||||||
|
persisted="$(persisted_root_domain)"
|
||||||
|
if [ -n "${FELIS_ROOT_DOMAIN:-}" ] && [ -n "$persisted" ] && [ "$FELIS_ROOT_DOMAIN" != "$persisted" ]; then
|
||||||
|
# Deliberate re-domain. Allowed -- there is no other route to it -- but it is not a
|
||||||
|
# thing this script finishes: the panel certificate, the two secrets, the velocity
|
||||||
|
# config and the login CR all still carry the old name.
|
||||||
|
warn "FELIS_ROOT_DOMAIN (${FELIS_ROOT_DOMAIN}) differs from the installed ${persisted}."
|
||||||
|
warn "This re-domains the install. The write-once panel certificate is NOT reissued and"
|
||||||
|
warn "will keep the old hostnames; the proxy and login config need the same treatment."
|
||||||
|
fi
|
||||||
|
FELIS_ROOT_DOMAIN="${FELIS_ROOT_DOMAIN:-${persisted:-${NODE_IP}.nip.io}}"
|
||||||
|
if [ -n "$persisted" ] && [ "$FELIS_ROOT_DOMAIN" = "$persisted" ]; then
|
||||||
|
log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN} (reusing the installed domain)"
|
||||||
|
else
|
||||||
|
log "node IP: ${NODE_IP} root domain: ${FELIS_ROOT_DOMAIN}"
|
||||||
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
pkg_install() {
|
pkg_install() {
|
||||||
|
|||||||
Reference in new issue
Block a user