diff --git a/deploy/bootstrap.sh b/deploy/bootstrap.sh index 78f3c76..e64875f 100644 --- a/deploy/bootstrap.sh +++ b/deploy/bootstrap.sh @@ -6082,7 +6082,7 @@ main_worker() { # Reuse the release binary path for the local admission checks, without importing game/platform images. bootstrap_from_tui || [ -n "$FELIS_ARTIFACT_DIR" ] || [ -n "${FELIS_SKIP_FETCH:-}" ] || resolve_install_ref acquire_felis_binary - if [ ! -x "$HOST_BIN" ]; then install_go_toolchain; build_nano_binary; fi + if [ -z "$HAVE_PREBUILT_BINARY" ]; then install_go_toolchain; build_nano_binary; fi mkdir -p /etc/rancher/k3s/config.yaml.d (umask 077; printf '%s\n' "$token" > /etc/rancher/k3s/felis-bootstrap-token) unset token @@ -6160,10 +6160,6 @@ main() { # before any re-run's rollouts). configure_registry_mirror acquire_felis_binary - if [ "${DISTRIBUTED:-0}" = 1 ]; then - [ -n "$WORKER_PEERS" ] || WORKER_PEERS="${NODE_EXTERNAL_IP:-$NODE_IP}/32" - "$HOST_BIN" node firewall --controller --controller-ip "${NODE_EXTERNAL_IP:-$NODE_IP}" --peers "$WORKER_PEERS" --pod-cidr "$POD_CIDR" --node-port "$FELIS_PANEL_NODEPORT" --control-namespace "$CONTROL_NS" --namespace "$MINECRAFT_NS" - fi select_release_artifacts resolve_felis_image # The registry's and the database's own images must be in containerd before their @@ -6171,6 +6167,12 @@ main() { import_release_images felis registry postgres import_platform_images build_image + # Source builds install the new HOST_BIN here; an earlier call may execute the + # old release's binary, which has no distributed node commands. + if [ "${DISTRIBUTED:-0}" = 1 ]; then + [ -n "$WORKER_PEERS" ] || WORKER_PEERS="${NODE_EXTERNAL_IP:-$NODE_IP}/32" + "$HOST_BIN" node firewall --controller --controller-ip "${NODE_EXTERNAL_IP:-$NODE_IP}" --peers "$WORKER_PEERS" --pod-cidr "$POD_CIDR" --node-port "$FELIS_PANEL_NODEPORT" --control-namespace "$CONTROL_NS" --namespace "$MINECRAFT_NS" + fi # After build_image imported the felis image: the registry pod's gate runs it. pin_platform_images # Before build_game_stack: the builds user servers run must be read off the diff --git a/deploy/bootstrap_test.sh b/deploy/bootstrap_test.sh index 0c1b7a9..e0ec268 100644 --- a/deploy/bootstrap_test.sh +++ b/deploy/bootstrap_test.sh @@ -4911,9 +4911,15 @@ esac rm -rf "$credir" "$credcalls" # Worker admission reuses the installer but must never enter host control-plane setup. +before "distributed host firewall runs the newly built binary" \ + ' build_image' '"$HOST_BIN" node firewall --controller' "$(bsfn main)" +before "distributed host firewall is installed before platform deployment" \ + '"$HOST_BIN" node firewall --controller' ' deploy_postgres' "$(bsfn main)" expect "distributed admission preserves existing API-server arguments" \ "echo 'kube-apiserver-arg+:'" "$(bsfn write_k3s_config)" worker="$(bsfn main_worker)" +expect "source worker install builds the requested binary even when an older binary exists" \ + 'if [ -z "$HAVE_PREBUILT_BINARY" ]; then install_go_toolchain; build_nano_binary; fi' "$worker" before "worker identity is checked before the agent config is written" \ 'refusing to rename it' 'cat > "$K3S_CONFIG_DROPIN"' "$worker" expect "worker rejects server tokens and verifies the CA-pinned bootstrap shape" \ diff --git a/docs/distributed.md b/docs/distributed.md index a0df0a8..5112a28 100644 --- a/docs/distributed.md +++ b/docs/distributed.md @@ -25,6 +25,14 @@ done 用包含此功能的 Felis 安装器在 A 重跑安装:`FELIS_DISTRIBUTED=1`、`FELIS_NODE_EXTERNAL_IP=`、`FELIS_PEER_CIDRS="$PEERS"`,保留现有安装参数。该步骤会启用 `wireguard-native`、`flannel-external-ip`、NodeRestriction 和独立 agent token,并安装归档服务、最小 RBAC 和宿主机隔离规则。WireGuard 更换需要停服维护窗口。已有 worker 的对等地址列表也必须提前更新。 +仅推送 main 不会自动发布 release。还未使用包含这些变更的发布资产时,在新版源码目录中以 root 执行下面的命令,显式从 main 构建;其他原安装参数继续保留。只更新安装脚本、仍使用默认 release 通道,可能下载到不支持分布式命令的旧二进制。 + +```bash +FELIS_REF=main FELIS_DISTRIBUTED=1 \ + FELIS_NODE_EXTERNAL_IP= \ + FELIS_PEER_CIDRS="$PEERS" bash deploy/bootstrap.sh +``` + 直接生成部署清单时,增加: ```bash diff --git a/docs/operations.md b/docs/operations.md index a9d9a41..20c4774 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -97,14 +97,13 @@ The installer also makes the system journal persistent (capped at `FELIS_JOURNAL_MAX_USE`, default 1G; `FELIS_MANAGE_JOURNAL=0` skips it) and writes the admin kubeconfig `/etc/rancher/k3s/k3s.yaml` root-only: run `sudo k3s kubectl`. -One node is the whole supported shape. A world volume is a ReadWriteOnce claim on the -node's local-path storage, so a game server's pod is pinned to the node that first -scheduled it and cannot move when that node fails; the operator and felis-api each run -as a single replica without leader election, so an upgrade or a node restart pauses -wakes and stops until their pod is back. Joining k3s agents to the cluster is untested -and gains no failover. A multi-node shape would need, at least, storage that can follow a -pod to another node and leader election in felis-operator (controller-runtime's -`LeaderElection`) so a second replica can stand by. +Single-node deployment remains the default. The opt-in [distributed mode](distributed.md) +keeps the sole API and operator on A and runs games on approved k3s agents. A world is +a ReadWriteOnce claim on its node's local-path storage; moving it requires an explicit +stopped migration through A's archive service. There is no automatic failover or +standby controller. An A restart pauses control operations until its workloads return; +a lost worker leaves its worlds on that node. Cross-node networking still requires the +three-machine acceptance described in the distributed runbook. ### Where the binary and the images come from