Unverified Commit cc85aac9 authored by Lemon-miaow's avatar Lemon-miaow
Browse files

ci(e2e): 新增 Ubuntu 24.04 全新安装+重跑 e2e 与 release→当前 commit 升级 e2e

parent 3ed66032
Loading
Loading
Loading
Loading
+154 −0
Changes for .github/workflows/e2e.yml: 154 added lines, 0 removed lines.
Original line number Diff line number Diff line
# deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every
# host goes through, run for real instead of by hand on a VM.
#
#   install  a full install of this commit, then the same commit again (a rerun must
#            converge without restarting what did not change)
#   upgrade  the newest published release, then this commit on top of it; skipped until
#            a release exists
#
# Each job builds the control plane, the game images and the proxy from scratch, about
# half an hour of runner time, so this runs on pushes that touch what gets installed, by
# hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium).
# deploy/e2e_check.sh holds the assertions.
name: e2e

on:
  push:
    branches: [main]
    paths:
      - 'deploy/**'
      - 'cmd/**'
      - 'internal/**'
      - 'panel/**'
      - 'plugins/**'
      - 'Dockerfile'
      - 'go.mod'
      - 'go.sum'
      - '.github/workflows/e2e.yml'
  workflow_dispatch:
  schedule:
    - cron: '23 4 * * 1'

permissions:
  contents: read

concurrency:
  group: e2e-${{ github.ref }}
  cancel-in-progress: true

jobs:
  install:
    runs-on: ubuntu-24.04
    timeout-minutes: 90
    steps:
      - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0

      # The images, k3s and the JRE need more room than a stock runner leaves free.
      - name: Free disk space
        run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL

      # FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs
      # have always been done; the .git directory is what stamps the build. Root owns it
      # so git, run as root by the installer, does not refuse it as dubious.
      - name: Stage this commit as the installer's source
        run: |
          sudo mkdir -p /opt/felis
          sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
          sudo chown -R root:root /opt/felis/src

      - name: Install
        run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log

      - name: Check the install
        run: sudo bash deploy/e2e_check.sh install

      - name: Rerun the same commit
        run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log

      - name: Check the rerun
        run: |
          sudo bash deploy/e2e_check.sh rerun
          grep -q 'felis-velocity unchanged; left running' rerun.log

      - name: Diagnostics
        if: failure()
        run: |
          export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
          k() { sudo -E /usr/local/bin/k3s kubectl "$@"; }
          k get pods -A -o wide || true
          k get events -A --sort-by=.lastTimestamp | tail -n 60 || true
          for p in $(k get pods -A --no-headers 2>/dev/null | awk '$4 != "Running" && $4 != "Completed" {print $1 "/" $2}'); do
            k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true
            k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true
          done
          sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true

      - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
        if: always()
        with:
          name: e2e-install-logs
          path: '*.log'
          if-no-files-found: ignore

  upgrade:
    runs-on: ubuntu-24.04
    timeout-minutes: 120
    steps:
      - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
        with:
          fetch-depth: 0

      - name: Free disk space
        run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL

      - name: Find the newest release
        id: release
        env:
          GH_TOKEN: ${{ github.token }}
        run: |
          tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)"
          if [ -z "$tag" ]; then
            echo "::notice::no published release yet; the upgrade path has nothing to start from"
          fi
          echo "tag=${tag}" >> "$GITHUB_OUTPUT"

      # The release's own installer, fetching the release's own binary: what a host that
      # installed that release is running today.
      - name: Install the newest release
        if: steps.release.outputs.tag != ''
        env:
          TAG: ${{ steps.release.outputs.tag }}
          TOKEN: ${{ github.token }}
        run: |
          git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh
          sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log

      - name: Check the release install
        if: steps.release.outputs.tag != ''
        run: sudo bash deploy/e2e_check.sh install

      - name: Upgrade to this commit
        if: steps.release.outputs.tag != ''
        run: |
          sudo rm -rf /opt/felis/src
          sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
          sudo chown -R root:root /opt/felis/src
          sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log

      - name: Check the upgrade
        if: steps.release.outputs.tag != ''
        run: sudo bash deploy/e2e_check.sh upgrade

      - name: Diagnostics
        if: failure()
        run: |
          export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
          sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
          sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true

      - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
        if: always()
        with:
          name: e2e-upgrade-logs
          path: '*.log'
          if-no-files-found: ignore

deploy/e2e_check.sh

0 → 100644
+118 −0
Changes for deploy/e2e_check.sh: 118 added lines, 0 removed lines.
Original line number Diff line number Diff line
#!/bin/bash
# Checks a host that deploy/bootstrap.sh just installed (or re-ran on). The e2e workflow
# runs it on a fresh GitHub runner after each of its two installer runs:
#
#   sudo bash deploy/e2e_check.sh install   # after the first run
#   sudo bash deploy/e2e_check.sh rerun     # after the same commit ran again
#   sudo bash deploy/e2e_check.sh upgrade   # after this commit ran over a release
#
# It asks what an operator's first minutes ask: the binary runs, the control plane is
# rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft
# status ping, and the host timers are there. A rerun must also leave the proxy running
# (it restarts only when what it runs changed) and keep every earlier answer.
set -euo pipefail

phase="${1:?usage: e2e_check.sh install|rerun|upgrade}"
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
KUBECTL=(/usr/local/bin/k3s kubectl)
PID_FILE=/var/tmp/felis-e2e-velocity.pid
fails=0

pass() { printf 'PASS %s\n' "$*"; }
fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); }
check() { # label command...
  local label="$1"
  shift
  if "$@"; then pass "$label"; else fail "$label"; fi
}

check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "'

for d in felis-api felis-operator registry; do
  check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
done

node_ip="$(ip -4 route get 1.1.1.1 | awk '{for (i = 1; i <= NF; i++) if ($i == "src") { print $(i + 1); exit }}')"
check "the panel serves its page on the NodePort" \
  sh -c "curl -skf --retry 10 --retry-delay 3 --retry-all-errors https://${node_ip}:30443/ | grep -qi '<html'"
# Readiness lives on the internal face only; a Service ClusterIP routes from the node.
internal="$("${KUBECTL[@]}" -n felis get svc felis-api-internal -o jsonpath='{.spec.clusterIP}:{.spec.ports[0].port}')"
check "felis-api is ready (database and cluster reachable)" \
  curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"

for unit in k3s postgresql felis-velocity; do
  check "${unit} is active" systemctl is-active --quiet "$unit"
done
for timer in felis-db-backup.timer felis-watchdog.timer; do
  check "${timer} is scheduled" systemctl is-enabled --quiet "$timer"
done

# A status ping is the proxy's own answer (ping passthrough is off), so it proves the JRE,
# Velocity and its config without a login gate or a Mojang account.
ping_proxy() {
  python3 - <<'EOF'
import json, socket, struct, sys

def varint(n):
    out = b""
    n &= 0xFFFFFFFF
    while True:
        b, n = n & 0x7F, n >> 7
        if n:
            out += bytes([b | 0x80])
        else:
            return out + bytes([b])

def read_varint(s):
    n = 0
    for i in range(5):
        b = s.recv(1)
        if not b:
            raise EOFError("connection closed")
        n |= (b[0] & 0x7F) << (7 * i)
        if not b[0] & 0x80:
            return n
    raise ValueError("varint too long")

host, port = "127.0.0.1", 25565
s = socket.create_connection((host, port), timeout=10)
hs = varint(0) + varint(767) + varint(len(host)) + host.encode() + struct.pack(">H", port) + varint(1)
s.sendall(varint(len(hs)) + hs + varint(1) + varint(0))
read_varint(s)
read_varint(s)
size = read_varint(s)
data = b""
while len(data) < size:
    chunk = s.recv(size - len(data))
    if not chunk:
        raise EOFError("short status response")
    data += chunk
print(json.loads(data)["version"]["name"])
EOF
}
if version="$(ping_proxy 2>&1)"; then
  pass "the proxy answers a status ping (${version})"
else
  fail "the proxy answers a status ping: ${version}"
fi

pid="$(systemctl show -p MainPID --value felis-velocity)"
case "$phase" in
  install) printf '%s\n' "$pid" > "$PID_FILE" ;;
  rerun)
    if [ "$pid" = "$(cat "$PID_FILE" 2>/dev/null)" ]; then
      pass "the rerun left the proxy running (pid ${pid})"
    else
      fail "the rerun restarted the proxy (pid $(cat "$PID_FILE" 2>/dev/null || echo '?') -> ${pid}) though nothing it runs changed"
    fi
    ;;
  upgrade) ;; # a new release may well change what the proxy runs
  *) fail "unknown phase ${phase}" ;;
esac

if [ "$fails" -eq 0 ]; then
  echo "ALL PASS (${phase})"
else
  echo "${fails} FAILED (${phase})"
fi
exit "$fails"