Loading .github/workflows/e2e.yml 0 → 100644 +154 −0 Changes for .github/workflows/e2e.yml: 154 added lines, 0 removed lines. Original line number Diff line number Diff line # deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every # host goes through, run for real instead of by hand on a VM. # # install a full install of this commit, then the same commit again (a rerun must # converge without restarting what did not change) # upgrade the newest published release, then this commit on top of it; skipped until # a release exists # # Each job builds the control plane, the game images and the proxy from scratch, about # half an hour of runner time, so this runs on pushes that touch what gets installed, by # hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium). # deploy/e2e_check.sh holds the assertions. name: e2e on: push: branches: [main] paths: - 'deploy/**' - 'cmd/**' - 'internal/**' - 'panel/**' - 'plugins/**' - 'Dockerfile' - 'go.mod' - 'go.sum' - '.github/workflows/e2e.yml' workflow_dispatch: schedule: - cron: '23 4 * * 1' permissions: contents: read concurrency: group: e2e-${{ github.ref }} cancel-in-progress: true jobs: install: runs-on: ubuntu-24.04 timeout-minutes: 90 steps: - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 # The images, k3s and the JRE need more room than a stock runner leaves free. - name: Free disk space run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL # FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs # have always been done; the .git directory is what stamps the build. Root owns it # so git, run as root by the installer, does not refuse it as dubious. - name: Stage this commit as the installer's source run: | sudo mkdir -p /opt/felis sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src sudo chown -R root:root /opt/felis/src - name: Install run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log - name: Check the install run: sudo bash deploy/e2e_check.sh install - name: Rerun the same commit run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log - name: Check the rerun run: | sudo bash deploy/e2e_check.sh rerun grep -q 'felis-velocity unchanged; left running' rerun.log - name: Diagnostics if: failure() run: | export KUBECONFIG=/etc/rancher/k3s/k3s.yaml k() { sudo -E /usr/local/bin/k3s kubectl "$@"; } k get pods -A -o wide || true k get events -A --sort-by=.lastTimestamp | tail -n 60 || true for p in $(k get pods -A --no-headers 2>/dev/null | awk '$4 != "Running" && $4 != "Completed" {print $1 "/" $2}'); do k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true done sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 if: always() with: name: e2e-install-logs path: '*.log' if-no-files-found: ignore upgrade: runs-on: ubuntu-24.04 timeout-minutes: 120 steps: - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 with: fetch-depth: 0 - name: Free disk space run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL - name: Find the newest release id: release env: GH_TOKEN: ${{ github.token }} run: | tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)" if [ -z "$tag" ]; then echo "::notice::no published release yet; the upgrade path has nothing to start from" fi echo "tag=${tag}" >> "$GITHUB_OUTPUT" # The release's own installer, fetching the release's own binary: what a host that # installed that release is running today. - name: Install the newest release if: steps.release.outputs.tag != '' env: TAG: ${{ steps.release.outputs.tag }} TOKEN: ${{ github.token }} run: | git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log - name: Check the release install if: steps.release.outputs.tag != '' run: sudo bash deploy/e2e_check.sh install - name: Upgrade to this commit if: steps.release.outputs.tag != '' run: | sudo rm -rf /opt/felis/src sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src sudo chown -R root:root /opt/felis/src sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log - name: Check the upgrade if: steps.release.outputs.tag != '' run: sudo bash deploy/e2e_check.sh upgrade - name: Diagnostics if: failure() run: | export KUBECONFIG=/etc/rancher/k3s/k3s.yaml sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 if: always() with: name: e2e-upgrade-logs path: '*.log' if-no-files-found: ignore deploy/e2e_check.sh 0 → 100644 +118 −0 Changes for deploy/e2e_check.sh: 118 added lines, 0 removed lines. Original line number Diff line number Diff line #!/bin/bash # Checks a host that deploy/bootstrap.sh just installed (or re-ran on). The e2e workflow # runs it on a fresh GitHub runner after each of its two installer runs: # # sudo bash deploy/e2e_check.sh install # after the first run # sudo bash deploy/e2e_check.sh rerun # after the same commit ran again # sudo bash deploy/e2e_check.sh upgrade # after this commit ran over a release # # It asks what an operator's first minutes ask: the binary runs, the control plane is # rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft # status ping, and the host timers are there. A rerun must also leave the proxy running # (it restarts only when what it runs changed) and keep every earlier answer. set -euo pipefail phase="${1:?usage: e2e_check.sh install|rerun|upgrade}" export KUBECONFIG=/etc/rancher/k3s/k3s.yaml KUBECTL=(/usr/local/bin/k3s kubectl) PID_FILE=/var/tmp/felis-e2e-velocity.pid fails=0 pass() { printf 'PASS %s\n' "$*"; } fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); } check() { # label command... local label="$1" shift if "$@"; then pass "$label"; else fail "$label"; fi } check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "' for d in felis-api felis-operator registry; do check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s done node_ip="$(ip -4 route get 1.1.1.1 | awk '{for (i = 1; i <= NF; i++) if ($i == "src") { print $(i + 1); exit }}')" check "the panel serves its page on the NodePort" \ sh -c "curl -skf --retry 10 --retry-delay 3 --retry-all-errors https://${node_ip}:30443/ | grep -qi '<html'" # Readiness lives on the internal face only; a Service ClusterIP routes from the node. internal="$("${KUBECTL[@]}" -n felis get svc felis-api-internal -o jsonpath='{.spec.clusterIP}:{.spec.ports[0].port}')" check "felis-api is ready (database and cluster reachable)" \ curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz" for unit in k3s postgresql felis-velocity; do check "${unit} is active" systemctl is-active --quiet "$unit" done for timer in felis-db-backup.timer felis-watchdog.timer; do check "${timer} is scheduled" systemctl is-enabled --quiet "$timer" done # A status ping is the proxy's own answer (ping passthrough is off), so it proves the JRE, # Velocity and its config without a login gate or a Mojang account. ping_proxy() { python3 - <<'EOF' import json, socket, struct, sys def varint(n): out = b"" n &= 0xFFFFFFFF while True: b, n = n & 0x7F, n >> 7 if n: out += bytes([b | 0x80]) else: return out + bytes([b]) def read_varint(s): n = 0 for i in range(5): b = s.recv(1) if not b: raise EOFError("connection closed") n |= (b[0] & 0x7F) << (7 * i) if not b[0] & 0x80: return n raise ValueError("varint too long") host, port = "127.0.0.1", 25565 s = socket.create_connection((host, port), timeout=10) hs = varint(0) + varint(767) + varint(len(host)) + host.encode() + struct.pack(">H", port) + varint(1) s.sendall(varint(len(hs)) + hs + varint(1) + varint(0)) read_varint(s) read_varint(s) size = read_varint(s) data = b"" while len(data) < size: chunk = s.recv(size - len(data)) if not chunk: raise EOFError("short status response") data += chunk print(json.loads(data)["version"]["name"]) EOF } if version="$(ping_proxy 2>&1)"; then pass "the proxy answers a status ping (${version})" else fail "the proxy answers a status ping: ${version}" fi pid="$(systemctl show -p MainPID --value felis-velocity)" case "$phase" in install) printf '%s\n' "$pid" > "$PID_FILE" ;; rerun) if [ "$pid" = "$(cat "$PID_FILE" 2>/dev/null)" ]; then pass "the rerun left the proxy running (pid ${pid})" else fail "the rerun restarted the proxy (pid $(cat "$PID_FILE" 2>/dev/null || echo '?') -> ${pid}) though nothing it runs changed" fi ;; upgrade) ;; # a new release may well change what the proxy runs *) fail "unknown phase ${phase}" ;; esac if [ "$fails" -eq 0 ]; then echo "ALL PASS (${phase})" else echo "${fails} FAILED (${phase})" fi exit "$fails" Loading
.github/workflows/e2e.yml 0 → 100644 +154 −0 Changes for .github/workflows/e2e.yml: 154 added lines, 0 removed lines. Original line number Diff line number Diff line # deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every # host goes through, run for real instead of by hand on a VM. # # install a full install of this commit, then the same commit again (a rerun must # converge without restarting what did not change) # upgrade the newest published release, then this commit on top of it; skipped until # a release exists # # Each job builds the control plane, the game images and the proxy from scratch, about # half an hour of runner time, so this runs on pushes that touch what gets installed, by # hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium). # deploy/e2e_check.sh holds the assertions. name: e2e on: push: branches: [main] paths: - 'deploy/**' - 'cmd/**' - 'internal/**' - 'panel/**' - 'plugins/**' - 'Dockerfile' - 'go.mod' - 'go.sum' - '.github/workflows/e2e.yml' workflow_dispatch: schedule: - cron: '23 4 * * 1' permissions: contents: read concurrency: group: e2e-${{ github.ref }} cancel-in-progress: true jobs: install: runs-on: ubuntu-24.04 timeout-minutes: 90 steps: - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 # The images, k3s and the JRE need more room than a stock runner leaves free. - name: Free disk space run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL # FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs # have always been done; the .git directory is what stamps the build. Root owns it # so git, run as root by the installer, does not refuse it as dubious. - name: Stage this commit as the installer's source run: | sudo mkdir -p /opt/felis sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src sudo chown -R root:root /opt/felis/src - name: Install run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log - name: Check the install run: sudo bash deploy/e2e_check.sh install - name: Rerun the same commit run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log - name: Check the rerun run: | sudo bash deploy/e2e_check.sh rerun grep -q 'felis-velocity unchanged; left running' rerun.log - name: Diagnostics if: failure() run: | export KUBECONFIG=/etc/rancher/k3s/k3s.yaml k() { sudo -E /usr/local/bin/k3s kubectl "$@"; } k get pods -A -o wide || true k get events -A --sort-by=.lastTimestamp | tail -n 60 || true for p in $(k get pods -A --no-headers 2>/dev/null | awk '$4 != "Running" && $4 != "Completed" {print $1 "/" $2}'); do k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true done sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 if: always() with: name: e2e-install-logs path: '*.log' if-no-files-found: ignore upgrade: runs-on: ubuntu-24.04 timeout-minutes: 120 steps: - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 with: fetch-depth: 0 - name: Free disk space run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL - name: Find the newest release id: release env: GH_TOKEN: ${{ github.token }} run: | tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)" if [ -z "$tag" ]; then echo "::notice::no published release yet; the upgrade path has nothing to start from" fi echo "tag=${tag}" >> "$GITHUB_OUTPUT" # The release's own installer, fetching the release's own binary: what a host that # installed that release is running today. - name: Install the newest release if: steps.release.outputs.tag != '' env: TAG: ${{ steps.release.outputs.tag }} TOKEN: ${{ github.token }} run: | git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log - name: Check the release install if: steps.release.outputs.tag != '' run: sudo bash deploy/e2e_check.sh install - name: Upgrade to this commit if: steps.release.outputs.tag != '' run: | sudo rm -rf /opt/felis/src sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src sudo chown -R root:root /opt/felis/src sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log - name: Check the upgrade if: steps.release.outputs.tag != '' run: sudo bash deploy/e2e_check.sh upgrade - name: Diagnostics if: failure() run: | export KUBECONFIG=/etc/rancher/k3s/k3s.yaml sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 if: always() with: name: e2e-upgrade-logs path: '*.log' if-no-files-found: ignore
deploy/e2e_check.sh 0 → 100644 +118 −0 Changes for deploy/e2e_check.sh: 118 added lines, 0 removed lines. Original line number Diff line number Diff line #!/bin/bash # Checks a host that deploy/bootstrap.sh just installed (or re-ran on). The e2e workflow # runs it on a fresh GitHub runner after each of its two installer runs: # # sudo bash deploy/e2e_check.sh install # after the first run # sudo bash deploy/e2e_check.sh rerun # after the same commit ran again # sudo bash deploy/e2e_check.sh upgrade # after this commit ran over a release # # It asks what an operator's first minutes ask: the binary runs, the control plane is # rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft # status ping, and the host timers are there. A rerun must also leave the proxy running # (it restarts only when what it runs changed) and keep every earlier answer. set -euo pipefail phase="${1:?usage: e2e_check.sh install|rerun|upgrade}" export KUBECONFIG=/etc/rancher/k3s/k3s.yaml KUBECTL=(/usr/local/bin/k3s kubectl) PID_FILE=/var/tmp/felis-e2e-velocity.pid fails=0 pass() { printf 'PASS %s\n' "$*"; } fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); } check() { # label command... local label="$1" shift if "$@"; then pass "$label"; else fail "$label"; fi } check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "' for d in felis-api felis-operator registry; do check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s done node_ip="$(ip -4 route get 1.1.1.1 | awk '{for (i = 1; i <= NF; i++) if ($i == "src") { print $(i + 1); exit }}')" check "the panel serves its page on the NodePort" \ sh -c "curl -skf --retry 10 --retry-delay 3 --retry-all-errors https://${node_ip}:30443/ | grep -qi '<html'" # Readiness lives on the internal face only; a Service ClusterIP routes from the node. internal="$("${KUBECTL[@]}" -n felis get svc felis-api-internal -o jsonpath='{.spec.clusterIP}:{.spec.ports[0].port}')" check "felis-api is ready (database and cluster reachable)" \ curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz" for unit in k3s postgresql felis-velocity; do check "${unit} is active" systemctl is-active --quiet "$unit" done for timer in felis-db-backup.timer felis-watchdog.timer; do check "${timer} is scheduled" systemctl is-enabled --quiet "$timer" done # A status ping is the proxy's own answer (ping passthrough is off), so it proves the JRE, # Velocity and its config without a login gate or a Mojang account. ping_proxy() { python3 - <<'EOF' import json, socket, struct, sys def varint(n): out = b"" n &= 0xFFFFFFFF while True: b, n = n & 0x7F, n >> 7 if n: out += bytes([b | 0x80]) else: return out + bytes([b]) def read_varint(s): n = 0 for i in range(5): b = s.recv(1) if not b: raise EOFError("connection closed") n |= (b[0] & 0x7F) << (7 * i) if not b[0] & 0x80: return n raise ValueError("varint too long") host, port = "127.0.0.1", 25565 s = socket.create_connection((host, port), timeout=10) hs = varint(0) + varint(767) + varint(len(host)) + host.encode() + struct.pack(">H", port) + varint(1) s.sendall(varint(len(hs)) + hs + varint(1) + varint(0)) read_varint(s) read_varint(s) size = read_varint(s) data = b"" while len(data) < size: chunk = s.recv(size - len(data)) if not chunk: raise EOFError("short status response") data += chunk print(json.loads(data)["version"]["name"]) EOF } if version="$(ping_proxy 2>&1)"; then pass "the proxy answers a status ping (${version})" else fail "the proxy answers a status ping: ${version}" fi pid="$(systemctl show -p MainPID --value felis-velocity)" case "$phase" in install) printf '%s\n' "$pid" > "$PID_FILE" ;; rerun) if [ "$pid" = "$(cat "$PID_FILE" 2>/dev/null)" ]; then pass "the rerun left the proxy running (pid ${pid})" else fail "the rerun restarted the proxy (pid $(cat "$PID_FILE" 2>/dev/null || echo '?') -> ${pid}) though nothing it runs changed" fi ;; upgrade) ;; # a new release may well change what the proxy runs *) fail "unknown phase ${phase}" ;; esac if [ "$fails" -eq 0 ]; then echo "ALL PASS (${phase})" else echo "${fails} FAILED (${phase})" fi exit "$fails"