perf(bootstrap): Velocity 堆从 64M 起步不再预占,Docker 及其 containerd 用完即停且不随开机启动

This commit is contained in:
Lemon-miaow committed 2026-09-29 19:06:37 +08:00
1 parent 451016bebd
commit a25553413c
3 files changed
+120 -22

No files matched your search

+26 -10
View File
@@ -1828,6 +1828,8 @@ install_docker() {
if command -v docker >/dev/null 2>&1; then
ok "docker already installed"
else
local had_containerd=""
command -v containerd >/dev/null 2>&1 && had_containerd=1
case "$PKG" in
apt) install_docker_apt ;;
dnf|yum) install_docker_rpm ;;
@@ -1835,8 +1837,13 @@ install_docker() {
pacman) install_docker_pacman ;;
*) die "Docker installation is not supported with package manager: ${PKG}" ;;
esac
# Docker serves this installer's builds and nothing at runtime, so the one it installed
# does not come up at boot to hold ~200 MiB until the next run; a run that builds starts
# it (ensure_docker). Some packages enable it, and its containerd, as they install.
systemctl disable docker.service docker.socket 2>/dev/null || true
[ -n "$had_containerd" ] || systemctl disable containerd.service 2>/dev/null || true
fi
systemctl enable --now docker
systemctl start docker
ok "docker running"
}
@@ -1873,11 +1880,19 @@ unplanned_build_room() {
die "${problem}; nothing has been built. Rerun the installer once the release's assets download, free space on ${mount}, or set FELIS_PREFLIGHT=warn to build anyway"
}
# stop_docker hands back the ~150 MiB the docker daemon holds once a step is done with it; the
# next step that builds starts it again. A Docker this run never started is left alone.
# stop_docker hands back what Docker holds once a step is done with it, ~150 MiB for the daemon
# and ~45 MiB for its containerd; the next step that builds starts both again. A Docker this run
# never started is left alone, and so is a containerd holding any namespace besides Docker's own
# (moby, moby_history): something else on the host runs on it. k3s's containerd listens on a
# socket of its own and is never the one asked here.
stop_docker() {
[ -n "$DOCKER_INSTALLED" ] || return 0
systemctl stop docker docker.socket 2>/dev/null || true
local ns
ns="$(ctr --address /run/containerd/containerd.sock namespaces ls -q 2>/dev/null)" || return 0
if ! printf '%s\n' "$ns" | grep -qvE '^(moby.*)?$'; then
systemctl stop containerd 2>/dev/null || true
fi
}
# ---------------------------------------------------------------------------
@@ -2929,8 +2944,7 @@ build_image() {
remove_k3s_image "$FELIS_IMAGE"
docker save "$FELIS_IMAGE" | k3s_cmd ctr images import -
# Reclaim the ~150 MiB the docker daemon holds; reruns restart it on demand.
systemctl stop docker docker.socket 2>/dev/null || true
stop_docker
ok "image built, binary on host, image imported"
}
@@ -3851,10 +3865,12 @@ install_velocity_service() {
# sees it -- unquoted, that spelling would hand java a stray "legacy112" argument and the unit
# would not start. Quoting keeps the whole property one argv item.
local legacy_forwarding_servers="${FELIS_LEGACY_FORWARDING_SERVERS}"
# -Xms stays at 512M so a small proxy does not reserve its whole ceiling up front, unless
# the ceiling itself is lower (the JVM refuses an initial heap above the maximum).
local xmx="$FELIS_VELOCITY_XMX" xms="512M"
[ "$(heap_megabytes "$xmx")" -ge 512 ] || xms="$xmx"
# The heap starts small and is not pre-touched. The proxy with its Via plugins holds about 50M
# live; a pre-touched 512M start kept ~0.7 GB resident on an idle network, ~0.25 GB without
# (measured on the verification host). The heap grows toward -Xmx as players arrive, and the
# periodic collection hands the growth back once they have left. FELIS_VELOCITY_XMX is at
# least 256M, so the start never exceeds the ceiling.
local xmx="$FELIS_VELOCITY_XMX" xms="64M"
cat > "$VELOCITY_SERVICE" <<EOF
[Unit]
Description=Felis Velocity proxy (Mojang authentication + modern forwarding)
@@ -3866,7 +3882,7 @@ Type=simple
User=${VELOCITY_USER}
Group=${VELOCITY_USER}
WorkingDirectory=${VELOCITY_DIR}
ExecStart=${JRE_DIR}/bin/java -Xms${xms} -Xmx${xmx} -XX:+UseG1GC -XX:+ParallelRefProcEnabled -XX:+AlwaysPreTouch -Dmojang.sessionserver=http://${api_ip}:8081/session/minecraft/hasJoined "-Dfelis.legacy-forwarding.servers=${legacy_forwarding_servers}" -jar ${VELOCITY_DIR}/velocity.jar
ExecStart=${JRE_DIR}/bin/java -Xms${xms} -Xmx${xmx} -XX:+UseG1GC -XX:+ParallelRefProcEnabled -XX:G1PeriodicGCInterval=60000 -Dmojang.sessionserver=http://${api_ip}:8081/session/minecraft/hasJoined "-Dfelis.legacy-forwarding.servers=${legacy_forwarding_servers}" -jar ${VELOCITY_DIR}/velocity.jar
Restart=on-failure
RestartSec=5
NoNewPrivileges=yes
+83 -3
View File
@@ -951,6 +951,79 @@ case "$out" in
*) echo "PASS a restart of Docker is not checked again" ;;
esac
# --- Docker holds no memory between builds -------------------------------------------------
# Docker serves the installer's builds and nothing at runtime. The one it installed does not
# start at boot, and once a step is done the daemon stops, and its containerd with it unless
# something besides Docker keeps a namespace there.
sdblock="$(awk '/^stop_docker\(\) \{/,/^}/' "$BS")"
[ -n "$sdblock" ] || { echo "FAIL: no stop_docker found in $BS"; exit 1; }
run_sd() { # DOCKER_INSTALLED namespaces-listed|FAIL
DOCKER_INSTALLED="$1" NS="$2" bash -c '
set -Eeuo pipefail
systemctl() { echo "SYSTEMCTL $*"; }
# Only the host containerd answers; k3s runs its own on another socket.
ctr() {
[ "$*" = "--address /run/containerd/containerd.sock namespaces ls -q" ] || { echo "CTR $*"; return 1; }
[ "$NS" != FAIL ] || return 1
printf "%b" "$NS"
}
'"$sdblock"'
stop_docker'
}
expect "a containerd serving Docker alone stops with it" "SYSTEMCTL stop docker docker.socket
SYSTEMCTL stop containerd" "$(run_sd 1 'moby\nmoby_history\n')"
out="$(run_sd 1 'default\nmoby\n')"
case "$out" in
*"stop containerd"*) echo "FAIL a containerd something else uses was stopped"; fails=$((fails + 1)) ;;
*"SYSTEMCTL stop docker docker.socket"*) echo "PASS a containerd with another tenant keeps running" ;;
*) echo "FAIL Docker was not stopped: $out"; fails=$((fails + 1)) ;;
esac
out="$(run_sd 1 FAIL)"
case "$out" in
*"stop containerd"*) echo "FAIL a containerd that could not be asked was stopped"; fails=$((fails + 1)) ;;
*"SYSTEMCTL stop docker docker.socket"*) echo "PASS a containerd that cannot be asked is left alone" ;;
*) echo "FAIL Docker was not stopped: $out"; fails=$((fails + 1)) ;;
esac
out="$(run_sd "" 'moby\n')"
[ -z "$out" ] && echo "PASS a Docker this run never started is left alone" \
|| { echo "FAIL a Docker this run never started was touched: $out"; fails=$((fails + 1)); }
idblock="$(awk '/^install_docker\(\) \{/,/^}/' "$BS")"
[ -n "$idblock" ] || { echo "FAIL: no install_docker found in $BS"; exit 1; }
run_id() { # docker-on-PATH(0|1) containerd-on-PATH(0|1)
fb="$(mktemp -d)"
for tool in docker containerd; do
{ [ "$tool" = docker ] && [ "$1" = 1 ]; } || { [ "$tool" = containerd ] && [ "$2" = 1 ]; } || continue
printf '#!/bin/sh\n' > "$fb/$tool"
chmod +x "$fb/$tool"
done
FB="$fb" bash -c '
set -Eeuo pipefail
PATH="$FB"
ok() { printf "OK: %s\n" "$*"; }; die() { printf "DIE: %s\n" "$*"; exit 1; }
systemctl() { echo "SYSTEMCTL $*"; }
install_docker_apt() { echo "INSTALL apt"; }
PKG=apt
'"$idblock"'
install_docker'
rm -rf "$fb"
}
out="$(run_id 0 0)"
expect "a Docker the installer brings is started without a place at boot" "INSTALL apt
SYSTEMCTL disable docker.service docker.socket
SYSTEMCTL disable containerd.service
SYSTEMCTL start docker" "$out"
case "$out" in *enable*) echo "FAIL the installed Docker was enabled at boot"; fails=$((fails + 1)) ;; *) echo "PASS the installed Docker is not enabled at boot" ;; esac
out="$(run_id 0 1)"
expect "a Docker installed beside an existing containerd leaves that containerd's boot alone" "INSTALL apt
SYSTEMCTL disable docker.service docker.socket
SYSTEMCTL start docker" "$out"
out="$(run_id 1 1)"
expect "a Docker already on the host is only started" "OK: docker already installed
SYSTEMCTL start docker" "$out"
case "$out" in *disable*|*enable*) echo "FAIL the host's own Docker had its boot changed"; fails=$((fails + 1)) ;; *) echo "PASS the host's own Docker keeps its boot setting" ;; esac
# --- a release binary is hashed against SHA256SUMS before anything runs it ---------------
# download_release_binary executes the asset as root to read its version stamp, so the
# checksum has to come first, and every failure has to fall back to the source build.
@@ -1502,6 +1575,7 @@ run_batch() { # PREBUILT_ROLES [ARTIFACT_MODE [ARTIFACT_CACHE]]
DOCKER_INSTALLED=""
systemctl() { printf "SYSTEMCTL %s\n" "$*"; }
ensure_docker() { printf "ENSURE\n"; DOCKER_INSTALLED=1; }
ctr() { return 1; }
push_image_to_registry() { printf "PUSH %s\n" "$1"; }
push_version_tag() { printf "VERSION %s\n" "$1"; }
push_release_image() { printf "RELEASE %s\n" "$1"; }
@@ -2821,7 +2895,12 @@ run_velocity_service() { # is-active(0|1) [heap]
}
out="$(run_velocity_service 1)"
expect "a proxy with no recorded start is restarted" "SYSTEMCTL restart felis-velocity" "$out"
expect "the default heap is 512M..1G" "java -Xms512M -Xmx1G " "$(cat "$vdir/unit")"
expect "the default heap starts at 64M and may grow to 1G" "java -Xms64M -Xmx1G " "$(cat "$vdir/unit")"
case "$(cat "$vdir/unit")" in
*AlwaysPreTouch*) echo "FAIL the proxy pre-touches its heap, holding all of -Xms from the start"; fails=$((fails + 1)) ;;
*"-XX:G1PeriodicGCInterval="*) echo "PASS the proxy neither pre-touches its heap nor keeps growth it no longer uses" ;;
*) echo "FAIL the proxy has no periodic collection to hand back an idle heap"; fails=$((fails + 1)) ;;
esac
[ -s "$vdir/fp" ] && echo "PASS the restart records what the proxy runs" \
|| { echo "FAIL no fingerprint was recorded after the restart"; fails=$((fails + 1)); }
out="$(run_velocity_service 1)"
@@ -2836,9 +2915,9 @@ expect "a stopped proxy is started whatever the fingerprint" "SYSTEMCTL restart
printf 'JAVA_VERSION="25.0.1"\n' > "$vdir/jre/release"
expect "a patched JRE restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1)"
expect "a new heap size restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1 3G)"
expect "the unit carries the new ceiling" "java -Xms512M -Xmx3G " "$(cat "$vdir/unit")"
expect "the unit carries the new ceiling" "java -Xms64M -Xmx3G " "$(cat "$vdir/unit")"
run_velocity_service 1 384M >/dev/null
expect "a ceiling below 512M is also the initial heap" "java -Xms384M -Xmx384M " "$(cat "$vdir/unit")"
expect "a small ceiling keeps the same small start" "java -Xms64M -Xmx384M " "$(cat "$vdir/unit")"
# `felis rotate-token velocity` rewrites service-token, which the plugin re-reads by itself:
# that line alone changing leaves the proxy running, and any other change restarts it.
props="$vdir/v/plugins/felis-link/felis-link.properties"
@@ -4447,6 +4526,7 @@ run_game_stack() { # PREBUILT_ROLES-after-import FELIS_GAME_STACK [ARTIFACT_MODE
game_stack_source() { :; }
resolve_game_jars() { :; }
import_release_images() { echo "IMPORT $*"; PREBUILT_ROLES="$AFTER"; }
ctr() { return 1; }
ensure_docker() { echo ENSURE; DOCKER_INSTALLED=1; }
build_game_image() { echo "BUILD $1"; }
install_velocity_plugin() { echo PLUGIN; }
+11 -9
View File
@@ -222,20 +222,21 @@ server running **[VM-VERIFIED]**:
| Process | Resident memory |
|---|---|
| k3s (server, kubelet, containerd) | ~1.1 GB |
| Velocity (`-Xms512M -Xmx1G`, heap pre-touched) | ~0.73 GB |
| Velocity (`-Xms64M -Xmx1G`, idle; it grows with players) | ~0.25 GB |
| lobby (Paper, pod limit 1 GiB) | ~0.7–0.85 GB |
| login (Limbo, pod limit 512 MiB) | ~0.16 GB |
| felis-api, felis-operator, registry gate | ~50 MB each |
| PostgreSQL (the felis-postgres pod) | ~30 MB plus page cache |
| **Total in use** | **~3.4 GB** |
| **Total in use** | **~2.9 GB** |
Every game server adds the memory its owner gave it: the pod's limit equals its request,
and the JVM heap is derived from it (§1a). Quotas cap it per user (panel → 管理 → 配额).
A release install builds nothing (§1). When the installer builds on the host its peak is
the image builds (Docker plus a Gradle container); it stops Docker afterwards so that memory
goes back to the servers. On a host under 2 GB of RAM without swap it adds a 2 GiB
`/swapfile`.
the image builds (Docker plus a Gradle container). Afterwards it stops Docker, and Docker's
containerd when nothing else uses it, so that memory goes back to the servers; a Docker the
installer put there does not start at boot. On a host under 2 GB of RAM without swap it
adds a 2 GiB `/swapfile`.
### Recommendations
@@ -248,14 +249,15 @@ goes back to the servers. On a host under 2 GB of RAM without swap it adds a 2 G
The player-count rows are planning figures, not measurements: a Minecraft server's cost
depends mostly on what its players do (view distance, redstone, mods). Size RAM as the
platform's ~3.5 GB plus the sum of the servers you expect to run at once, then add a
platform's ~3 GB plus the sum of the servers you expect to run at once, then add a
quarter for the page cache and PostgreSQL. Velocity itself needs little per player; raise
its heap when `journalctl -u felis-velocity` shows long GC pauses or `OutOfMemoryError`.
`FELIS_VELOCITY_XMX` (default `1G`, at least `256M`, written `<n>M` or `<n>G`) is read on
every installer run. The initial heap stays at 512M, or equals the maximum when that is
lower. Changing it rewrites the unit, and the rerun restarts the proxy, which disconnects
everyone online; do it in a quiet hour **[VM-VERIFIED]**:
every installer run. The heap starts at 64M and grows toward the maximum as players arrive;
a periodic collection hands the growth back once they have left. Changing it rewrites the
unit, and the rerun restarts the proxy, which disconnects everyone online; do it in a quiet
hour **[VM-VERIFIED]**:
```
curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo FELIS_VELOCITY_XMX=2G bash