diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml new file mode 100644 index 0000000..df45ab9 --- /dev/null +++ b/.github/workflows/e2e.yml @@ -0,0 +1,154 @@ +# deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every +# host goes through, run for real instead of by hand on a VM. +# +# install a full install of this commit, then the same commit again (a rerun must +# converge without restarting what did not change) +# upgrade the newest published release, then this commit on top of it; skipped until +# a release exists +# +# Each job builds the control plane, the game images and the proxy from scratch, about +# half an hour of runner time, so this runs on pushes that touch what gets installed, by +# hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium). +# deploy/e2e_check.sh holds the assertions. +name: e2e + +on: + push: + branches: [main] + paths: + - 'deploy/**' + - 'cmd/**' + - 'internal/**' + - 'panel/**' + - 'plugins/**' + - 'Dockerfile' + - 'go.mod' + - 'go.sum' + - '.github/workflows/e2e.yml' + workflow_dispatch: + schedule: + - cron: '23 4 * * 1' + +permissions: + contents: read + +concurrency: + group: e2e-${{ github.ref }} + cancel-in-progress: true + +jobs: + install: + runs-on: ubuntu-24.04 + timeout-minutes: 90 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + + # The images, k3s and the JRE need more room than a stock runner leaves free. + - name: Free disk space + run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL + + # FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs + # have always been done; the .git directory is what stamps the build. Root owns it + # so git, run as root by the installer, does not refuse it as dubious. + - name: Stage this commit as the installer's source + run: | + sudo mkdir -p /opt/felis + sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src + sudo chown -R root:root /opt/felis/src + + - name: Install + run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log + + - name: Check the install + run: sudo bash deploy/e2e_check.sh install + + - name: Rerun the same commit + run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log + + - name: Check the rerun + run: | + sudo bash deploy/e2e_check.sh rerun + grep -q 'felis-velocity unchanged; left running' rerun.log + + - name: Diagnostics + if: failure() + run: | + export KUBECONFIG=/etc/rancher/k3s/k3s.yaml + k() { sudo -E /usr/local/bin/k3s kubectl "$@"; } + k get pods -A -o wide || true + k get events -A --sort-by=.lastTimestamp | tail -n 60 || true + for p in $(k get pods -A --no-headers 2>/dev/null | awk '$4 != "Running" && $4 != "Completed" {print $1 "/" $2}'); do + k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true + k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true + done + sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true + + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + if: always() + with: + name: e2e-install-logs + path: '*.log' + if-no-files-found: ignore + + upgrade: + runs-on: ubuntu-24.04 + timeout-minutes: 120 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + with: + fetch-depth: 0 + + - name: Free disk space + run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL + + - name: Find the newest release + id: release + env: + GH_TOKEN: ${{ github.token }} + run: | + tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)" + if [ -z "$tag" ]; then + echo "::notice::no published release yet; the upgrade path has nothing to start from" + fi + echo "tag=${tag}" >> "$GITHUB_OUTPUT" + + # The release's own installer, fetching the release's own binary: what a host that + # installed that release is running today. + - name: Install the newest release + if: steps.release.outputs.tag != '' + env: + TAG: ${{ steps.release.outputs.tag }} + TOKEN: ${{ github.token }} + run: | + git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh + sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log + + - name: Check the release install + if: steps.release.outputs.tag != '' + run: sudo bash deploy/e2e_check.sh install + + - name: Upgrade to this commit + if: steps.release.outputs.tag != '' + run: | + sudo rm -rf /opt/felis/src + sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src + sudo chown -R root:root /opt/felis/src + sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log + + - name: Check the upgrade + if: steps.release.outputs.tag != '' + run: sudo bash deploy/e2e_check.sh upgrade + + - name: Diagnostics + if: failure() + run: | + export KUBECONFIG=/etc/rancher/k3s/k3s.yaml + sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true + sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true + + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + if: always() + with: + name: e2e-upgrade-logs + path: '*.log' + if-no-files-found: ignore diff --git a/deploy/e2e_check.sh b/deploy/e2e_check.sh new file mode 100644 index 0000000..f40395a --- /dev/null +++ b/deploy/e2e_check.sh @@ -0,0 +1,118 @@ +#!/bin/bash +# Checks a host that deploy/bootstrap.sh just installed (or re-ran on). The e2e workflow +# runs it on a fresh GitHub runner after each of its two installer runs: +# +# sudo bash deploy/e2e_check.sh install # after the first run +# sudo bash deploy/e2e_check.sh rerun # after the same commit ran again +# sudo bash deploy/e2e_check.sh upgrade # after this commit ran over a release +# +# It asks what an operator's first minutes ask: the binary runs, the control plane is +# rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft +# status ping, and the host timers are there. A rerun must also leave the proxy running +# (it restarts only when what it runs changed) and keep every earlier answer. +set -euo pipefail + +phase="${1:?usage: e2e_check.sh install|rerun|upgrade}" +export KUBECONFIG=/etc/rancher/k3s/k3s.yaml +KUBECTL=(/usr/local/bin/k3s kubectl) +PID_FILE=/var/tmp/felis-e2e-velocity.pid +fails=0 + +pass() { printf 'PASS %s\n' "$*"; } +fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); } +check() { # label command... + local label="$1" + shift + if "$@"; then pass "$label"; else fail "$label"; fi +} + +check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "' + +for d in felis-api felis-operator registry; do + check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s +done + +node_ip="$(ip -4 route get 1.1.1.1 | awk '{for (i = 1; i <= NF; i++) if ($i == "src") { print $(i + 1); exit }}')" +check "the panel serves its page on the NodePort" \ + sh -c "curl -skf --retry 10 --retry-delay 3 --retry-all-errors https://${node_ip}:30443/ | grep -qi '> 7 + if n: + out += bytes([b | 0x80]) + else: + return out + bytes([b]) + +def read_varint(s): + n = 0 + for i in range(5): + b = s.recv(1) + if not b: + raise EOFError("connection closed") + n |= (b[0] & 0x7F) << (7 * i) + if not b[0] & 0x80: + return n + raise ValueError("varint too long") + +host, port = "127.0.0.1", 25565 +s = socket.create_connection((host, port), timeout=10) +hs = varint(0) + varint(767) + varint(len(host)) + host.encode() + struct.pack(">H", port) + varint(1) +s.sendall(varint(len(hs)) + hs + varint(1) + varint(0)) +read_varint(s) +read_varint(s) +size = read_varint(s) +data = b"" +while len(data) < size: + chunk = s.recv(size - len(data)) + if not chunk: + raise EOFError("short status response") + data += chunk +print(json.loads(data)["version"]["name"]) +EOF +} +if version="$(ping_proxy 2>&1)"; then + pass "the proxy answers a status ping (${version})" +else + fail "the proxy answers a status ping: ${version}" +fi + +pid="$(systemctl show -p MainPID --value felis-velocity)" +case "$phase" in + install) printf '%s\n' "$pid" > "$PID_FILE" ;; + rerun) + if [ "$pid" = "$(cat "$PID_FILE" 2>/dev/null)" ]; then + pass "the rerun left the proxy running (pid ${pid})" + else + fail "the rerun restarted the proxy (pid $(cat "$PID_FILE" 2>/dev/null || echo '?') -> ${pid}) though nothing it runs changed" + fi + ;; + upgrade) ;; # a new release may well change what the proxy runs + *) fail "unknown phase ${phase}" ;; +esac + +if [ "$fails" -eq 0 ]; then + echo "ALL PASS (${phase})" +else + echo "${fails} FAILED (${phase})" +fi +exit "$fails"