Author SHA1 Message Date
dependabot[bot] abbe40fa8b build(deps): bump actions/setup-java from 4.9.1 to 6.0.1
Bumps [actions/setup-java](https://github.com/actions/setup-java) from 4.9.1 to 6.0.1.
- [Release notes](https://github.com/actions/setup-java/releases)
- [Commits](https://github.com/actions/setup-java/compare/cf277c60eb25467037889841efdb72551f06f6c3...de7274f081f381c8f8158605e0321c36c376e2e6)

---
updated-dependencies:
- dependency-name: actions/setup-java
  dependency-version: 6.0.1
  dependency-type: direct:production
  update-type: version-update:semver-major
...

Signed-off-by: dependabot[bot] <[email protected]>
2026-09-25 15:47:41 +00:00
508 changed files with 4648 additions and 83003 deletions

No files matched your search

+2 -36
View File
@@ -17,15 +17,11 @@ name: ci
#
# release.yml calls this workflow (workflow_call) before it builds anything, so a tag passes
# exactly these gates and there is one list of them.
#
# workflow_dispatch reruns the suite on a commit whose push produced no run, such as one
# pushed while Actions was unavailable.
on:
push:
branches: [main]
pull_request:
workflow_call:
workflow_dispatch:
permissions:
contents: read
@@ -76,9 +72,6 @@ jobs:
# The business stores' SQL against a real PostgreSQL (internal/pgint): the unit suites run
# on fakes, and PGRepo drifted from them three times while those stayed green. 13 is the
# oldest server a supported distribution installs (EL9), 18 the newest (Arch).
# `felis db backup` and `restore` run there too, with the tools inside the service
# container, as production runs them inside felis-postgres: the runner's own client is one
# major version, and pg_dump refuses a newer server.
pgint:
runs-on: ubuntu-latest
strategy:
@@ -109,7 +102,6 @@ jobs:
- run: go test -race -tags pgint -count=1 ./internal/pgint/
env:
FELIS_TEST_PG_URL: postgres://felis:pgint@localhost:5432/felis_pgint?sslmode=disable
FELIS_TEST_PG_EXEC: docker exec -i ${{ job.services.postgres.id }}
shell:
runs-on: ubuntu-latest
@@ -142,34 +134,8 @@ jobs:
tar -xJf shellcheck.tar.xz
./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh')
# An exit status is 8 bits, so `exit "$fails"` reads 256 failures as a pass. The suites
# exit 1 on any failure; this keeps the next one from carrying its count out.
- name: No script exits with its failure count
run: |
if git grep -nE 'exit +"?\$\{?[a-z_]*fail[a-z_]*\}?"?[[:space:]]*$' -- '*.sh'; then exit 1; fi
- run: sh deploy/bootstrap_test.sh
- run: sh deploy/uninstall_test.sh
- run: bash deploy/e2e_release_test.sh
# The shipped alert rules (deploy/alerts): promtool parses them and runs their unit tests,
# which pin when each alert fires and that it stays quiet before. internal/metrics'
# alerts_test.go pins what promtool cannot see from there: the PrometheusRule twin and the
# metric names the rules read.
alerts:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
# A pinned release, like shellcheck's, so a new Prometheus cannot change what fails.
- name: promtool
run: |
curl -fsSL -o prometheus.tar.gz \
https://github.com/prometheus/prometheus/releases/download/v3.15.0/prometheus-3.15.0.linux-amd64.tar.gz
echo "2a542df32eac02ee17b9d844fb2aa1de00dafa5476579ba8a3ba862e9d572ea0 prometheus.tar.gz" | sha256sum -c
tar -xzf prometheus.tar.gz --strip-components=1 prometheus-3.15.0.linux-amd64/promtool
./promtool check rules deploy/alerts/felis-alerts.yaml
./promtool test rules deploy/alerts/felis-alerts_test.yml
panel:
runs-on: ubuntu-latest
@@ -240,7 +206,7 @@ jobs:
# compiled by bootstrap on a live host, and the test mains under plugins/*/test
# were run by hand. JDK 25 is what the plugin build image runs (paper-api 26.x
# needs it); each module's wrapper brings the Gradle the image pins.
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
- uses: actions/setup-java@de7274f081f381c8f8158605e0321c36c376e2e6 # v6.0.1
with:
distribution: temurin
java-version: '25'
@@ -259,7 +225,7 @@ jobs:
# this job nothing ever built them: no install path touches them, and their
# gradlew scripts were committed without the exec bit, so the README's
# one-liners failed on a fresh clone.
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
- uses: actions/setup-java@de7274f081f381c8f8158605e0321c36c376e2e6 # v6.0.1
with:
distribution: temurin
java-version: '17'
+37 -216
View File
@@ -1,26 +1,15 @@
# deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every
# host goes through, run for real instead of by hand on a VM.
#
# artifacts this commit's release assets, built by deploy/build-release-artifacts.sh the
# way release.yml builds a tag's (amd64 only, the runners' architecture)
# install a full install from those assets, the way a release installs, on a host with
# ufw enabled, then the same assets again (a rerun must converge without
# restarting what did not change, importing or uploading an image again, or
# touching Docker)
# readme the README's one-line install as a new host runs it today: this commit's
# installer on its default channel, which installs the newest published
# release's binary, images and plugin from that release's assets; skipped until
# install a full install of this commit, then the same commit again (a rerun must
# converge without restarting what did not change)
# upgrade the newest published release, then this commit on top of it; skipped until
# a release exists
# upgrade the newest published release through its own installer and its own assets,
# rows of every kind seeded into its database, then this commit's assets on
# top of it; skipped until a release exists
# source a full install built on the host from this checkout (FELIS_SKIP_FETCH):
# the fallback a release without assets, or an unsupported one, takes. It builds
# everything with Docker, so it runs by hand and weekly only
#
# This runs on pushes that touch what gets installed, by hand, and weekly (a moving
# upstream: apt mirrors, k3s's install script and release assets, Adoptium).
# deploy/e2e_check.sh holds the assertions, deploy/e2e_seed.sh the upgrade's seed and its check.
# Each job builds the control plane, the game images and the proxy from scratch, about
# half an hour of runner time, so this runs on pushes that touch what gets installed, by
# hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium).
# deploy/e2e_check.sh holds the assertions.
name: e2e
on:
@@ -43,47 +32,14 @@ on:
permissions:
contents: read
# An explicit bash runs with -o pipefail, so `bootstrap.sh | tee install.log` fails the step
# when the installer fails; the default shell reports tee's status.
defaults:
run:
shell: bash
concurrency:
group: e2e-${{ github.ref }}
cancel-in-progress: true
jobs:
artifacts:
runs-on: ubuntu-24.04
timeout-minutes: 60
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
fetch-depth: 0 # the newest tag is the version's base, as on the dev channel
- uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
- name: Free disk space
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
# Stamped the way bootstrap stamps a build of main: "<newest tag>+g<sha>".
- name: Build the release assets
run: |
base="$(git describe --tags --abbrev=0 --match 'v*' 2>/dev/null || echo v0.0.0)"
FELIS_RELEASE_ARCHES=amd64 deploy/build-release-artifacts.sh "${base}+g$(git rev-parse --short HEAD)" dist
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: e2e-assets
path: dist/
if-no-files-found: error
retention-days: 1
install:
needs: artifacts
runs-on: ubuntu-24.04
timeout-minutes: 60
timeout-minutes: 90
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
@@ -91,50 +47,28 @@ jobs:
- name: Free disk space
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: e2e-assets
path: dist
# The runner has Docker preinstalled, so its absence proves nothing: the log shows
# whether bootstrap reached for it. From FELIS_ARTIFACT_DIR every image and the plugin
# come out of the assets, so it must not have. (`! grep` is spelled `if grep ... exit 1`
# below because bash -e ignores a failing `!` command.)
# ufw, enabled on many Ubuntu hosts, drops every inbound packet no rule admits, the
# pods' traffic to the API server among them; the runner ships it disabled. The
# runner's own traffic is outbound, which ufw lets through.
- name: Enable ufw
run: sudo ufw --force enable
# FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs
# have always been done; the .git directory is what stamps the build. Root owns it
# so git, run as root by the installer, does not refuse it as dubious.
- name: Stage this commit as the installer's source
run: |
sudo mkdir -p /opt/felis
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
sudo chown -R root:root /opt/felis/src
- name: Install
run: sudo FELIS_ARTIFACT_DIR="$GITHUB_WORKSPACE/dist" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee install.log
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log
- name: Check the install
run: |
sudo bash deploy/e2e_check.sh install
for tag in felis-k3s-pods felis-k3s-services felis-panel felis-proxy; do
sudo ufw status | grep -qE "# ${tag} *$"
done
if grep -E 'docker already installed|installing docker|docker running' install.log; then exit 1; fi
for role in felis limbo lobby paper; do
grep -q "felis/${role}:[^ ]* is the release's" install.log
done
grep -q "docker.io/library/registry@sha256:[0-9a-f]* is the release's" install.log
grep -q "docker.io/library/postgres@sha256:[0-9a-f]* is the release's" install.log
grep -q "felis-velocity.jar is the release's" install.log
run: sudo bash deploy/e2e_check.sh install
- name: Rerun the same assets
run: sudo FELIS_ARTIFACT_DIR="$GITHUB_WORKSPACE/dist" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee rerun.log
- name: Rerun the same commit
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log
# containerd and the registry already hold every image under the digest the listing
# names, so nothing is imported or uploaded twice.
- name: Check the rerun
run: |
sudo bash deploy/e2e_check.sh rerun
grep -q 'felis-velocity unchanged; left running' rerun.log
if grep -E 'docker already installed|installing docker|docker running' rerun.log; then exit 1; fi
if grep -E 'importing felis-image-|mirroring .* into the internal registry' rerun.log; then exit 1; fi
grep -q 'is already in the internal registry' rerun.log
- name: Diagnostics
if: failure()
@@ -147,8 +81,7 @@ jobs:
k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true
k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true
done
k -n felis logs deploy/felis-postgres --tail=60 || true
sudo journalctl -u k3s -u felis-velocity --no-pager -n 120 || true
sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
if: always()
@@ -157,62 +90,7 @@ jobs:
path: '*.log'
if-no-files-found: ignore
# The README's command on a fresh host: this commit's installer, as main serves it, on its
# default channel with nothing pinned. It resolves the newest release and installs that
# release's binary, images and plugin from its assets, each checked against its
# SHA256SUMS; the private repo's token is the only thing added. FELIS_INSTALL_MODE picks the
# mode the setup console would ask for.
readme:
runs-on: ubuntu-24.04
timeout-minutes: 120
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- name: Free disk space
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
- name: Find the newest release
id: release
env:
GH_TOKEN: ${{ github.token }}
run: bash deploy/e2e_release.sh find
- name: Install as the README does
if: steps.release.outputs.tag != ''
env:
TOKEN: ${{ github.token }}
run: sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee readme.log
# The binary is the release's, so the phase is `release`: its database backup and
# timers are the release's to have or lack.
- name: Check the install
if: steps.release.outputs.tag != ''
env:
TAG: ${{ steps.release.outputs.tag }}
BINARY: ${{ steps.release.outputs.binary }}
SUMS: ${{ steps.release.outputs.sums }}
run: |
sudo bash deploy/e2e_check.sh release
sudo /usr/local/bin/felis version | grep -qx "felis ${TAG}"
bash deploy/e2e_release.sh check-readme readme.log
- name: Diagnostics
if: failure()
run: |
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
sudo -E /usr/local/bin/k3s kubectl -n felis logs deploy/felis-postgres --tail=60 || true
sudo journalctl -u k3s -u felis-velocity -u docker --no-pager -n 120 || true
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
if: always()
with:
name: e2e-readme-logs
path: '*.log'
if-no-files-found: ignore
upgrade:
needs: artifacts
runs-on: ubuntu-24.04
timeout-minutes: 120
steps:
@@ -223,22 +101,19 @@ jobs:
- name: Free disk space
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: e2e-assets
path: dist
- name: Find the newest release
id: release
env:
GH_TOKEN: ${{ github.token }}
run: bash deploy/e2e_release.sh find
run: |
tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)"
if [ -z "$tag" ]; then
echo "::notice::no published release yet; the upgrade path has nothing to start from"
fi
echo "tag=${tag}" >> "$GITHUB_OUTPUT"
# The release's own installer on its default channel, fetching the release's own
# binary: what a host that installed that release is running today. FELIS_REF would
# build the tag from source instead, a path no host takes by default. FELIS_RELEASE pins
# the tag found above; an installer older than FELIS_RELEASE ignores it and resolves
# the newest release, the same tag.
# The release's own installer, fetching the release's own binary: what a host that
# installed that release is running today.
- name: Install the newest release
if: steps.release.outputs.tag != ''
env:
@@ -246,44 +121,29 @@ jobs:
TOKEN: ${{ github.token }}
run: |
git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh
sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_RELEASE="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log
sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log
- name: Check the release install
if: steps.release.outputs.tag != ''
env:
TAG: ${{ steps.release.outputs.tag }}
BINARY: ${{ steps.release.outputs.binary }}
run: |
sudo bash deploy/e2e_check.sh release
bash deploy/e2e_release.sh check-own release.log
# A fresh install's database holds only what its migrations wrote: the seed gives the
# upgrade's move and its pending migrations existing rows to carry.
- name: Seed the release's database
if: steps.release.outputs.tag != ''
run: sudo bash deploy/e2e_seed.sh seed
run: sudo bash deploy/e2e_check.sh release
- name: Upgrade to this commit
if: steps.release.outputs.tag != ''
run: sudo FELIS_ARTIFACT_DIR="$GITHUB_WORKSPACE/dist" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee upgrade.log
run: |
sudo rm -rf /opt/felis/src
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
sudo chown -R root:root /opt/felis/src
sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log
# Both checks run and report: the seeded rows go last, after the restore drill has
# also put them back from a bundle.
- name: Check the upgrade
if: steps.release.outputs.tag != ''
run: |
rc=0
sudo bash deploy/e2e_check.sh upgrade || rc=1
sudo bash deploy/e2e_seed.sh check || rc=1
exit "$rc"
run: sudo bash deploy/e2e_check.sh upgrade
- name: Diagnostics
if: failure()
run: |
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
sudo -E /usr/local/bin/k3s kubectl -n felis logs deploy/felis-postgres --tail=60 || true
# The release ran the database on the host; the upgrade moves it into k3s.
sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
@@ -292,42 +152,3 @@ jobs:
name: e2e-upgrade-logs
path: '*.log'
if-no-files-found: ignore
source:
if: github.event_name != 'push'
runs-on: ubuntu-24.04
timeout-minutes: 90
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- name: Free disk space
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
# FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src; the .git directory is what
# stamps the build. Root owns it so git, run as root by the installer, does not refuse
# it as dubious.
- name: Stage this commit as the installer's source
run: |
sudo mkdir -p /opt/felis
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
sudo chown -R root:root /opt/felis/src
- name: Install
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee source.log
- name: Check the install
run: sudo bash deploy/e2e_check.sh install
- name: Diagnostics
if: failure()
run: |
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
sudo journalctl -u k3s -u felis-velocity -u docker --no-pager -n 120 || true
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
if: always()
with:
name: e2e-source-logs
path: '*.log'
if-no-files-found: ignore
+63 -45
View File
@@ -2,15 +2,14 @@
#
# Two things depend on this job. /repos/{repo}/releases/latest must ANSWER — that endpoint is
# what `felis update` polls (internal/updater/github.go) and what deploy/bootstrap.sh's default
# "release" channel resolves its ref from. And the assets below are what that channel then
# INSTALLS: bootstrap downloads the felis binary, the image tars, their listing and the Velocity
# plugin instead of building anything on the target host, so these are the shipped artifact,
# not a convenience. A host installing a release needs neither Docker, Gradle, Go nor Docker Hub.
# "release" channel resolves its ref from. And the binaries below are what that channel then
# INSTALLS: bootstrap downloads felis-linux-<arch> instead of compiling on the target host, so
# these are the shipped artifact, not a convenience.
#
# deploy/build-release-artifacts.sh builds every asset in one run and documents each name; the
# names are a contract with deploy/bootstrap.sh. They are plain literals on both sides: a YAML
# workflow and a go:embed'ed shell script have no honest way to share a constant, and the
# failure mode is benign — bootstrap warns and falls back to building on the host.
# The asset NAME is a contract with deploy/bootstrap.sh (download_release_binary builds
# "felis-linux-${arch}"). It is deliberately a plain literal on both sides: a GitHub Actions
# YAML and a go:embed'ed shell script have no honest way to share a constant, and the failure
# mode is benign — bootstrap warns and falls back to a source build of the same tag.
#
# The binary is built through the repo Dockerfile rather than a plain `go build`.
# internal/panel/static holds a tracked PLACEHOLDER index.html so the //go:embed
@@ -19,12 +18,13 @@
# build first, and is the same recipe bootstrap uses, so there is one way to build felis
# rather than two that can drift.
#
# SHA256SUMS is a contract with bootstrap too: it refuses any asset whose hash is not listed
# there, BEFORE it runs or imports it.
# SHA256SUMS is a contract with bootstrap too: download_release_binary refuses a binary whose
# hash is not listed there, BEFORE it runs it. A release without the file installs by source
# build instead.
#
# The write token never meets the test suite: `gates` (ci.yml) and `build` run the tests,
# Gradle and the Docker builds (each of which executes third-party code) with a read-only
# token, and `build` hands the assets over as a workflow artifact; `publish` holds contents:write and runs only
# Gradle and the Docker build (each of which executes third-party code) with a read-only
# token, and `build` hands the binaries over as a workflow artifact; `publish` holds contents:write and runs only
# pinned actions and gh. Every action is pinned to a commit SHA (the tag in the trailing
# comment is for humans); .github/dependabot.yml proposes the bumps.
name: release
@@ -39,10 +39,9 @@ permissions:
jobs:
# A tag that ships red is worse than a tag that fails to ship. These are ci.yml's gates,
# called rather than copied: Go (race, vet, staticcheck), govulncheck, the PostgreSQL
# contract suite, shellcheck and the bootstrap tests, promtool on the alert rules, the
# panel, and the Java layer the binary EMBEDS (bootstrap_asset.go ships the plugin
# sources, so a tag whose plugins do not compile turns every install of that release into
# a failed bootstrap).
# contract suite, shellcheck and the bootstrap tests, the panel, and the Java layer the
# binary EMBEDS (bootstrap_asset.go ships the plugin sources, so a tag whose plugins do
# not compile turns every install of that release into a failed bootstrap).
gates:
uses: ./.github/workflows/ci.yml
@@ -53,47 +52,69 @@ jobs:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
# Both architectures, because bootstrap's default release channel DOWNLOADS these
# rather than building on the target host — an arm64 host with no asset falls back to
# a slow on-host build. The felis binary and the plugin jars cross-compile on the
# runner (their build stages are pinned to $BUILDPLATFORM); the game images' runtime
# stages run apt-get for the target platform, which for arm64 takes QEMU.
- uses: docker/setup-qemu-action@99012661954931238ded8c8b007157a8430204e1 # v4.4.0
# rather than compiling on the target host — an arm64 host with no asset silently
# falls back to a slow source build. Neither stage is emulated: the Dockerfile pins
# both build stages to $BUILDPLATFORM and the Go stage cross-compiles via TARGETARCH,
# so the second architecture costs about a minute.
- uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
# Two architectures of five images plus BuildKit's cache outgrow the runner's free disk
# with its preinstalled SDKs in place; none of them is used here.
- name: Free disk space
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
- name: Build the stamped binaries
run: |
docker buildx build --platform linux/amd64,linux/arm64 \
--build-arg FELIS_VERSION="${GITHUB_REF_NAME}" \
--output type=local,dest=out .
mv out/linux_amd64/usr/local/bin/felis ./felis-linux-amd64
mv out/linux_arm64/usr/local/bin/felis ./felis-linux-arm64
chmod +x ./felis-linux-amd64 ./felis-linux-arm64
# The script checks what the old inline steps did: the host binary reports exactly
# "felis <tag>" (unstamped, `felis update` refuses to compare), and the other one is an
# ELF of its architecture, read with file(1) — binfmt would run a misnamed binary happily.
- name: Build the release assets
run: deploy/build-release-artifacts.sh "${GITHUB_REF_NAME}" dist
# The stamp is the whole point and it fails silently: an unstamped binary reports
# "dev", which `felis update` refuses to compare, disabling update reporting for
# every install built from it. Assert it end to end instead of trusting the ARG
# reached the linker.
- name: Verify the version stamp
run: |
got="$(./felis-linux-amd64 version | head -1)"
echo "reported: ${got}"
[ "$got" = "felis ${GITHUB_REF_NAME}" ] \
|| { echo "expected 'felis ${GITHUB_REF_NAME}' — the -X main.version stamp did not reach the binary"; exit 1; }
# The arm64 binary is checked by ELF machine type, NOT by running it. Runners have
# binfmt/QEMU registered, so `./felis-linux-arm64 version` would happily succeed on
# an amd64 binary misnamed arm64 — which is exactly the failure the Dockerfile's
# ${TARGETARCH:-$(go env GOARCH)} fallback produces if buildx did not take. Both
# binaries come out of one RUN with one -ldflags string, so the stamp is checked once.
file ./felis-linux-arm64
file ./felis-linux-arm64 | grep -q 'ARM aarch64' \
|| { echo "felis-linux-arm64 is not an arm64 ELF — TARGETARCH did not reach the go build"; exit 1; }
- name: Checksum the binaries
run: sha256sum felis-linux-amd64 felis-linux-arm64 | tee SHA256SUMS
# A CycloneDX SBOM per binary: the Go modules (and versions) linked into it, read
# from the build info the linker embeds. Written after SHA256SUMS, so beside it in the
# release but outside it: that lists what bootstrap installs. Straight into dist/,
# because the action does not create a missing output directory.
# from the build info the linker embeds.
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0.24.0
with:
file: dist/felis-linux-amd64
file: felis-linux-amd64
format: cyclonedx-json
output-file: dist/felis-linux-amd64.cdx.json
output-file: felis-linux-amd64.cdx.json
upload-artifact: false
upload-release-assets: false
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0.24.0
with:
file: dist/felis-linux-arm64
file: felis-linux-arm64
format: cyclonedx-json
output-file: dist/felis-linux-arm64.cdx.json
output-file: felis-linux-arm64.cdx.json
upload-artifact: false
upload-release-assets: false
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: release-assets
path: dist/
path: |
felis-linux-amd64
felis-linux-arm64
felis-linux-amd64.cdx.json
felis-linux-arm64.cdx.json
SHA256SUMS
if-no-files-found: error
retention-days: 7
@@ -113,7 +134,7 @@ jobs:
- run: sha256sum -c SHA256SUMS
# Signed SLSA provenance: which workflow run, commit and repository produced each
# asset. Check one with `gh attestation verify felis-linux-amd64 --repo FelisMC/Felis`.
# binary. Check one with `gh attestation verify felis-linux-amd64 --repo FelisMC/Felis`.
# GitHub only stores attestations for private repositories on Enterprise Cloud, and a
# failure here would block the release, so a private repository skips the step and
# relies on SHA256SUMS alone.
@@ -122,9 +143,8 @@ jobs:
uses: actions/attest-build-provenance@977bb373ede98d70efdf65b84cb5f73e068dcc2a # v3.0.0
with:
subject-path: |
felis-linux-*
felis-image-*.tar
felis-velocity.jar
felis-linux-amd64
felis-linux-arm64
# --verify-tag refuses to invent a release for a tag that is not pushed.
#
@@ -144,9 +164,7 @@ jobs:
GH_TOKEN: ${{ github.token }}
GH_REPO: ${{ github.repository }}
run: |
# Every file the build handed over: SHA256SUMS names the installable ones, and the
# SBOMs ride along.
assets="$(ls)"
assets="felis-linux-amd64 felis-linux-arm64 felis-linux-amd64.cdx.json felis-linux-arm64.cdx.json SHA256SUMS"
flags=""
case "$GITHUB_REF_NAME" in *-*) flags="--prerelease" ;; esac
if ! gh release view "$GITHUB_REF_NAME" >/dev/null 2>&1; then
+2 -9
View File
@@ -78,15 +78,8 @@ FELIS_TEST_PG_URL='postgres://felis:***@127.0.0.1:5432/felis_pgint?sslmode=disab
```
Run it after touching anything under `internal/api/pgrepo.go`, `internal/submit`,
`internal/build` or `internal/dbbackup` that speaks SQL: the fakes encode the
contract, and this suite exists to catch the drift between the fakes and the real
queries. The `felis db backup` and `restore` tests also run `pg_dump`, `pg_restore`
and `psql`, which must be the server's major version. For a server in a container,
run them in it, as production does in felis-postgres:
```bash
FELIS_TEST_PG_EXEC='docker exec -i <container>' FELIS_TEST_PG_URL=... go test -tags pgint ./internal/pgint/
```
or `internal/build` that speaks SQL: the fakes encode the contract, and this
suite exists to catch the drift between the fakes and the real queries.
Build the CLI:
+2 -10
View File
@@ -1,8 +1,5 @@
# Felis
**此项目仍处于早期开发阶段,您不该在任何生产环境使用该项目。若产生任何问题,贵用户的使用行为与 FelisMC 团队无任何民事刑事法律关系。**
**THIS PROJECT IS STILL WIP, YOU SHOULD DO NOT USE THIS PROJECT IN ANY PRODUCTION USAGE. WE ARE NOT RESPOND FOR ANY LEGAL OR HUMANLY PROBLEM.**
一款 Kubernetes 驱动的 Minecraft 服务器托管平台,一行命令部署,自动管理生命周期与安全。
A Kubernetes-driven Minecraft server hosting platform — one command to deploy, automatic lifecycle, backup, and security.
@@ -22,7 +19,6 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
- **Web 控制面板**:浏览器中查看服务器状态、在线玩家与资源用量,管理备份与恢复。
- **备份与恢复**:一键把整服数据(世界、配置、插件/模组,即整个 /data 卷)打包进集群内的归档库,支持从任意备份点回滚;默认安装就已启用(归档 PVC 与路径由安装器一并生成)。
- **控制面数据库备份**:账号、服务器归属、配额与存档索引所在的数据库每天自动备份,每次升级迁移前先快照,出错可用 `felis db restore` 整库原子回滚;面板「维护与备份」页显示备份是否新鲜(见 [故障排查 §16](docs/troubleshooting.md))。
- **运维自检**:`sudo felis status` 一屏列出节点、控制面、游戏代理、每台服务器、备份与未解决的告警;`sudo felis doctor` 把健康检查全跑一遍,按区域给出问题和下一步去哪看,不发邮件;`sudo felis support-bundle` 打出一个脱敏的诊断包,求助时直接附上(见 [故障排查 §0](docs/troubleshooting.md))。
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。过期备份无论是否开启都会每天清理。
- **多核心支持**:兼容 Paper、Fabric、Forge、NeoForge,经由 Velocity 代理统一入口。
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建;构建产物进入镜像白名单,可直接选用为服务器镜像完成部署。
@@ -31,17 +27,13 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
## 使用方式
在准备好的 Linux 主机上执行(已验证的发行版与架构见 [运维手册 §1](docs/operations.md#1-supported-hosts):CentOS Stream 9 aarch64 实机验证,Ubuntu 24.04 x86_64 每次推送由 CI 跑全新安装、重跑、升级和下面这条命令本身):
在准备好的 Linux 主机上执行(已验证的发行版与架构见 [运维手册 §1](docs/operations.md#1-supported-hosts):CentOS Stream 9 aarch64 实机验证,Ubuntu 24.04 x86_64 每次推送由 CI 跑全新安装、重跑与升级):
```bash
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
```
脚本将自动安装 K3s,在 K3s 内部署 PostgreSQL 与控制平面,并启动设置向导。完成后浏览器访问已配置的域名进入控制面板即可使用。
安装发布版时,二进制、全部镜像与 Velocity 插件都取自该版本在 CI 里预构建好的 release 附件,逐个核对 `SHA256SUMS` 后导入,主机上无需 Docker、Gradle 或 Go,也不从 Docker Hub 拉取;某个附件缺失或校验不符时,只有那一个镜像退回到本机构建,并给出提示(见 [故障排查 §15c](docs/troubleshooting.md))。附件也可以先拷到本机,再用 `FELIS_ARTIFACT_DIR=<绝对路径>` 安装,Felis 自己的二进制、镜像和插件就都取自这个目录;k3s 及其镜像、JRE、cloudflared、Velocity 和 Via 插件照旧从 GitHub 与 PaperMC 下载,RHEL、Fedora、openSUSE Leap 这类开着 SELinux 的主机还要从 rpm.rancher.io 装 k3s-selinux,系统软件包来自发行版的源。所以出网受限的主机要放行这几处的 HTTPS(或设 `https_proxy`),preflight 会在改动主机之前逐个探测,完全断网的主机目前装不了(地址清单见 [运维手册 §1](docs/operations.md#1-supported-hosts))。旧版本装在宿主上的 PostgreSQL 会在重跑时整库迁进 K3s,宿主上的那份停用保留,供回退(见 [运维手册 §4](docs/operations.md#4-upgrading-the-pieces-around-felis))。
动手之前,脚本先检查内存、磁盘、端口、网段冲突、已有的 Kubernetes 和外网连通,把所有问题一次列出并停下,主机上什么都没改(检查项见 [运维手册 §1](docs/operations.md#1-supported-hosts))。
脚本将自动安装 K3s、部署控制平面并启动设置向导。完成后浏览器访问已配置的域名进入控制面板即可使用。
> **本仓库当前为私有**,上面这条会返回 404。请改用带凭据的形式;安装器自身也需要同一个 token
> 去解析并下载 release,所以用 `sudo -E` 把它带进去:
+1 -4
View File
@@ -19,7 +19,6 @@ Table of Contents
- **Web Dashboard**: Monitor server status, online players, and resource usage from your browser, with backup and restore management.
- **Backup & Restore**: One-click snapshots of a server's whole data volume (worlds, config, plugins/mods — the entire /data volume) into the cluster's archive store, with rollback from any backup point — enabled by default (the installer renders the archive PVC and its path).
- **Control-plane database backups**: The database holding accounts, server ownership, quotas and the archive index is backed up daily and snapshotted before every upgrade migrates it; `felis db restore` rolls it back atomically, and the panel's Maintenance & Backups page shows whether the newest backup is fresh (see [troubleshooting §16](docs/troubleshooting.md)).
- **Self-check for operators**: `sudo felis status` shows the node, the control plane, the game proxy, every server, the backups and the open alerts on one screen; `sudo felis doctor` runs every health check once and lists each problem by area with where to look next, mailing nothing; `sudo felis support-bundle` writes one redacted diagnostics archive to attach when asking for help (see [troubleshooting §0](docs/troubleshooting.md)).
- **World Reaper** (opt in): Worlds idle for more than 15 days are automatically backed up and removed to free disk space. Enable it by setting `FELIS_WORLDS_HOST_PATH` at install time (on k3s: `/var/lib/rancher/k3s/storage`); without it, no world is ever deleted.
- **Multi-core Support**: Compatible with Paper, Fabric, Forge, and NeoForge, federated behind a Velocity proxy.
- **Modpack Submission**: Players submit custom modpacks; admin approval triggers an automatic build, and the result is whitelisted as a server image you can select to deploy.
@@ -37,9 +36,7 @@ every push):
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
```
The script installs K3s, deploys PostgreSQL and the control plane inside it, and launches a setup wizard. Once done, open your browser at the configured domain. A PostgreSQL an earlier release installed on the host is moved into K3s on the next rerun, and the host copy is stopped and kept for a rollback (see [Operations §4](docs/operations.md#4-upgrading-the-pieces-around-felis)).
Before it changes anything, the script checks RAM, disk, ports, network-range clashes, any Kubernetes already there and outbound access, lists every problem at once and stops with the host untouched (the checks are in [operations §1](docs/operations.md#1-supported-hosts)).
The script installs K3s, deploys the control plane, and launches a setup wizard. Once done, open your browser at the configured domain.
> **This repository is currently private**, so the command above returns 404. Use the
> credentialed form instead; the installer itself needs the same token to resolve and
+5 -18
View File
@@ -304,19 +304,16 @@ func TestDockerfileBaseImagesArePinnedByDigest(t *testing.T) {
}
}
// The plugin jars are built in four places the installer controls — the lobby and limbo
// image builds, bootstrap's Velocity build and the release build of the same jar — and
// through each module's wrapper by a developer or CI. A tag alone is whatever it points
// at on build day, and two Gradle versions are two chances for a build to pass in one
// place and break in the other, so all of them run one image, pinned by digest, whose
// Gradle is the wrappers' Gradle.
// The plugin jars are built in three places the installer controls — the lobby and limbo
// image builds and bootstrap's Velocity build — and through each module's wrapper by a
// developer or CI. A tag alone is whatever it points at on build day, and two Gradle
// versions are two chances for a build to pass in one place and break in the other, so
// all of them run one image, pinned by digest, whose Gradle is the wrappers' Gradle.
func TestPluginBuildsRunOnePinnedGradle(t *testing.T) {
sources := map[string]string{
"deploy/lobby/Dockerfile": readGameStackFile(t, "deploy/lobby/Dockerfile"),
"deploy/limbo/Dockerfile": readGameStackFile(t, "deploy/limbo/Dockerfile"),
"deploy/bootstrap.sh": BootstrapScript(),
// The release build compiles felis-velocity.jar for hosts that install prebuilt.
"deploy/build-release-artifacts.sh": readRepoFile(t, "deploy/build-release-artifacts.sh"),
}
anyRef := regexp.MustCompile(`gradle:[\w.-]+(@sha256:\w+)?`)
pinned := regexp.MustCompile(`^gradle:(\d+\.\d+(?:\.\d+)?)-jdk\d+@sha256:[0-9a-f]{64}$`)
@@ -490,16 +487,6 @@ func gameStackLock(t *testing.T) map[string]string {
return lock
}
// readRepoFile reads a file of the checkout that the binary does not embed.
func readRepoFile(t *testing.T, name string) string {
t.Helper()
b, err := os.ReadFile(name)
if err != nil {
t.Fatal(err)
}
return string(b)
}
func readGameStackFile(t *testing.T, name string) string {
t.Helper()
b, err := gameStackAssets.ReadFile(name)
+14 -169
View File
@@ -33,7 +33,6 @@ import (
"felis.lolicon.best/internal/restore"
"felis.lolicon.best/internal/retention"
"felis.lolicon.best/internal/submit"
"felis.lolicon.best/internal/worldexport"
"k8s.io/apimachinery/pkg/runtime"
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
"k8s.io/client-go/kubernetes"
@@ -50,22 +49,6 @@ import (
// the code marks Identity (UUIDs trusted verbatim); config can never add another.
const mojangSessionServer = "https://sessionserver.mojang.com/session/minecraft/hasJoined"
// passkeyRelyingParty is the one WebAuthn relying party both web faces share: its id
// is the player console host, derived from server.root_domain the way the panel
// handler derives it when auth.panel_hostname is unset, and its origins are that host
// plus the operator host. An empty id means the install names no panel host at all.
func passkeyRelyingParty(cfg *config.Config) (string, []string) {
rpID := defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname)
if rpID == "" {
return "", nil
}
origins := []string{"https://" + rpID}
if admin := defaultAdminHostname(cfg.Server.RootDomain, cfg.Auth.AdminHostname); admin != "" && admin != rpID {
origins = append(origins, "https://"+admin)
}
return rpID, origins
}
// authSourcesFromConfig builds the multiplexer's priority list from the configured
// [[auth_source]] entries: Mojang leads as the code-owned identity anchor (正版优先, the ONLY
// Identity source — config can only append namespace-rewritten third-party sources, never a
@@ -117,7 +100,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// Before anything serves: an api on a schema it was not built for answers with
// errors, or writes rows the other version cannot read.
drv, err := openPodStore(ctx, cfg.Database.URL, "api", stderr)
drv, err := openStore(ctx, cfg.Database.URL, false)
if err != nil {
fmt.Fprintf(stderr, "felis api: open database: %v\n", err)
return 1
@@ -293,17 +276,6 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
fmt.Fprintln(stderr, "felis api: backup executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — backup endpoint returns 503")
}
// World export: a weak-SA Job mounts the world PVC, or the backup PVC, read-only
// and PUTs the archive to the internal face, which streams it on to the owner's
// browser (internal/worldexport). A backup export mounts the backup PVC, so it
// is wired under the restore gate; otherwise the export routes return 503.
var exporter api.Exporter
if felisImage != "" && backupPVC != "" {
exporter = worldexport.New(clientset, exportConfig(cfg, felisImage, backupPVC))
} else {
fmt.Fprintln(stderr, "felis api: world export disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — export endpoints return 503")
}
// Server file editor: a weak-SA Job mounts ONLY the target world PVC and runs
// `felis files`, printing its result for felis-api to read back through
// pods/log (see internal/fileedit). It needs FELIS_IMAGE but — unlike restore
@@ -312,22 +284,12 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// endpoints honestly return 503. It takes the typed clientset rather than the
// controller-runtime client because the log subresource lives only on the typed
// CoreV1 client, and one client covers its Job create, Pod list, and log read.
//
// Uploads additionally stage their bytes on this pod's disk until the Job
// fetches them from the internal face; whatever a previous process staged is
// orphaned (the index is in memory), so the stage starts empty.
var files api.FileEditor
var fileStage *fileedit.Stage
if felisImage != "" {
files = &fileedit.Editor{
Runner: fileedit.NewK8sRunner(clientset),
Config: fileEditConfig(cfg, felisImage),
}
fileStage = &fileedit.Stage{Dir: fileStagingDir()}
if err := fileStage.Sweep(); err != nil {
fmt.Fprintf(stderr, "felis api: %v — file uploads return 503\n", err)
fileStage = nil
}
} else {
fmt.Fprintln(stderr, "felis api: file editor disabled (needs FELIS_IMAGE) — file endpoints return 503")
}
@@ -375,15 +337,8 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// restore behind each one.
RestoreChains: jobStatus,
Files: files,
FileStage: fileStage,
// The file Job fetches an upload from here; it runs in the minecraft
// namespace, where the internal face is reachable like it is for the login
// gate.
InternalBaseURL: internalAPIBaseURL(),
Exporter: exporter,
Submissions: submissions,
Mailer: mailer,
Schedules: repo,
// The external face authenticates the local session cookie the sign-in doors
// mint, live once `felis breakGlass` flips local_auth_enabled on. Cloudflare
// Access, when the install sits behind it, is enforced at the edge only.
@@ -438,16 +393,23 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// enrolled once asserts on either face — one binding, usable on the player console
// AND the operator console. Both hosts are therefore listed as permitted origins,
// while the RP id stays the panel host so the credential's scope is ONE relying
// party, not two. Without a panel host (neither auth.panel_hostname nor
// server.root_domain) a.Passkey stays nil and the passkey routes honestly return
// 503 (the authenticated enrollment boundary is still enforced by the handlers).
if rpID, origins := passkeyRelyingParty(cfg); rpID == "" {
fmt.Fprintln(stderr, "felis api: passkey verifier disabled (no panel host: set server.root_domain or auth.panel_hostname) — passkey endpoints return 503")
} else if pv, err := passkey.New(rpID, "Felis", origins); err != nil {
// party, not two. Wired only when auth.panel_hostname is configured; otherwise
// a.Passkey stays nil and the passkey routes honestly return 503 (the authenticated
// enrollment boundary is still enforced by the handlers).
if cfg.Auth.PanelHostname != "" {
origins := []string{"https://" + cfg.Auth.PanelHostname}
if admin := defaultAdminHostname(cfg.Server.RootDomain, cfg.Auth.AdminHostname); admin != "" && admin != cfg.Auth.PanelHostname {
origins = append(origins, "https://"+admin)
}
pv, err := passkey.New(cfg.Auth.PanelHostname, "Felis", origins)
if err != nil {
fmt.Fprintf(stderr, "felis api: passkey verifier disabled: %v — passkey endpoints return 503\n", err)
} else {
a.Passkey = pv
}
} else {
fmt.Fprintln(stderr, "felis api: passkey verifier disabled (auth.panel_hostname unset) — passkey endpoints return 503")
}
// Derive the console hostnames when felis.toml leaves them unset, exactly as the
// setup/breakGlass paths do — otherwise the SPA cannot tell which face it is
@@ -478,25 +440,11 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
// reconciles it, but this loop converges builds nobody is polling.
go reconcileBuilds(ctx, builder, stderr)
go settleRestoreChains(ctx, a, stderr)
go runSchedules(ctx, a, stderr)
// A daily restore point of every world played since its last one, taken
// once the server stops ([archive] scheduled_every; 0s turns it off).
if backuper != nil && rcfg.ScheduledEvery > 0 {
go scheduleBackups(ctx, &api.BackupScheduler{API: a, Store: repo, Jobs: jobStatus, Every: rcfg.ScheduledEvery}, stderr)
} else {
fmt.Fprintln(stderr, "felis api: scheduled backups off (needs the backup executor and [archive] scheduled_every above 0s)")
}
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
go pruner.Loop(ctx, registryPruneInterval)
}
go reapRejectedContexts(ctx, submissions, stderr)
if fileStage != nil {
go expireFileSessions(ctx, fileStage, fileSessionSweep, stderr)
}
if exporter != nil {
go expireExports(ctx, a, exportSweep)
}
go retention.Loop(ctx, drv.DB(), retention.Policy{Audit: auditRetention}, retentionInterval, slog.Default())
servers := []*http.Server{internalSrv, externalSrv}
@@ -706,17 +654,6 @@ func backupConfig(cfg *config.Config, image, backupPVC string) backupjob.Config
}
}
// exportConfig builds the world export executor's config. BackupRoot mirrors
// restoreConfig: the stored refs are absolute paths under [archive] local_path.
func exportConfig(cfg *config.Config, image, backupPVC string) worldexport.Config {
return worldexport.Config{
Namespace: cfg.K8s.Namespace,
Image: image,
BackupPVC: backupPVC,
BackupRoot: cfg.Archive.LocalPath,
}
}
// fileEditConfig builds the file editor's config from felis.toml plus the
// deployment-supplied image. It is the shortest of the three: the editor mounts
// only the world PVC, so it needs no archive coordinates at all, and everything
@@ -766,88 +703,6 @@ func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
}
}
// runSchedules runs the servers' scheduled tasks (api.API.RunSchedules). The
// interval is how late a task may start, and how often a restart or backup in
// progress checks whether it can take its next step.
func runSchedules(ctx context.Context, a *api.API, stderr io.Writer) {
t := time.NewTicker(15 * time.Second)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
if err := a.RunSchedules(ctx); err != nil {
fmt.Fprintf(stderr, "felis api: scheduled tasks: %v\n", err)
}
}
}
}
// scheduleBackups starts the scheduled backups (api.BackupScheduler). Each tick
// starts at most one, so the interval also spaces the worlds that stopped at
// the same time: a world that stops waits at most this long for its point to
// start once the Jobs ahead of it are done.
func scheduleBackups(ctx context.Context, s *api.BackupScheduler, stderr io.Writer) {
t := time.NewTicker(2 * time.Minute)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
if err := s.Tick(ctx); err != nil {
fmt.Fprintf(stderr, "felis api: scheduled backups: %v\n", err)
}
}
}
}
// fileSessionSweep is how often expireFileSessions looks for idle upload
// sessions: small beside fileedit.SessionIdle, so an abandoned one gives its
// room back within minutes of going stale.
const fileSessionSweep = 10 * time.Minute
// expireFileSessions drops the file manager's upload sessions left untouched
// for fileedit.SessionIdle. Each reserved room on the staging disk for its whole
// file when it began, so one abandoned would otherwise hold that room until
// felis-api restarts.
func expireFileSessions(ctx context.Context, s *fileedit.Stage, every time.Duration, stderr io.Writer) {
t := time.NewTicker(every)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
if n := s.Expire(); n > 0 {
fmt.Fprintf(stderr, "felis api: dropped %d upload session(s) left idle for %s\n", n, fileedit.SessionIdle)
}
}
}
}
// exportSweep is how often expireExports runs: an export whose Job never
// connected is stopped within a minute of going stale.
const exportSweep = time.Minute
// expireExports runs the export sweep (api.API.ExpireExports) on a ticker. The
// export routes sweep as they are called, and an owner who closed the tab calls
// none; a Job whose Pod never got going would then keep the server from
// starting until the Job's deadline.
func expireExports(ctx context.Context, a interface{ ExpireExports() }, every time.Duration) {
t := time.NewTicker(every)
defer t.Stop()
for {
select {
case <-ctx.Done():
return
case <-t.C:
a.ExpireExports()
}
}
}
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
// submissions rejected more than submit.RejectedContextRetention ago, and the
// chunked uploads left untouched for submit.StalePartRetention. Without it a
@@ -1057,16 +912,6 @@ func uploadPartsDir(contextBase string) string {
return filepath.Join(os.TempDir(), "felis-upload-parts")
}
// fileStagingDir is where file uploads wait for their Job: on the uploads
// volume, whose capacity is its own, or the pod's /tmp when run by hand without
// it — /tmp is the node's disk, which a burst of uploads should not fill.
func fileStagingDir() string {
if fi, err := os.Stat(platform.UploadsLocalPath); err == nil && fi.IsDir() {
return filepath.Join(platform.UploadsLocalPath, ".file-staging")
}
return filepath.Join(os.TempDir(), "felis-file-staging")
}
// contextMaxBytes resolves [registry] context_max_bytes. 0 keeps the submit
// package's own default (1 GiB). The Cloudflare edge refuses a single request
// body over 100 MB, which the panel's chunked upload stays under, so the edge
-108
View File
@@ -1,15 +1,12 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"io"
"net"
"net/http"
"slices"
"sync"
"sync/atomic"
"testing"
"time"
@@ -17,40 +14,8 @@ import (
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/build"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/fileedit"
)
// The passkey relying party follows the panel host the SPA is served on: an install
// that names only its root domain still gets passkeys, on console.<root>, with the
// operator host as the second origin; only an install with no panel host goes without.
func TestPasskeyRelyingParty(t *testing.T) {
for _, tc := range []struct {
name string
root, panel, admin string
wantRP string
wantOrigins []string
}{
{"root domain only", "example.net", "", "", "console.example.net",
[]string{"https://console.example.net", "https://op.console.example.net"}},
{"configured hosts", "example.net", " play.example.net ", "ops.example.net", "play.example.net",
[]string{"https://play.example.net", "https://ops.example.net"}},
{"operator host equal to the panel host", "example.net", "console.example.net", "console.example.net",
"console.example.net", []string{"https://console.example.net"}},
{"panel host without a root domain", "", "console.example.org", "", "console.example.org",
[]string{"https://console.example.org"}},
{"no host at all", "", "", "", "", nil},
} {
t.Run(tc.name, func(t *testing.T) {
cfg := &config.Config{}
cfg.Server.RootDomain, cfg.Auth.PanelHostname, cfg.Auth.AdminHostname = tc.root, tc.panel, tc.admin
rp, origins := passkeyRelyingParty(cfg)
if rp != tc.wantRP || !slices.Equal(origins, tc.wantOrigins) {
t.Fatalf("relying party = %q %q, want %q %q", rp, origins, tc.wantRP, tc.wantOrigins)
}
})
}
}
// TestAuthSourcesFromConfig pins the one place the hasJoined identity anchor is decided:
// Mojang is prepended in code, first, and is the only source whose UUIDs are trusted as-is.
// The empty case matters on its own — both `felis api` and `felis nano` call this with a
@@ -238,76 +203,3 @@ func TestInUseImageRefsCoversEverySource(t *testing.T) {
t.Fatal("a failing whitelist read produced a reference list")
}
}
// TestExpireFileSessions runs the loop against a stage whose clock the test
// holds: the session idle past fileedit.SessionIdle goes, the one touched since
// stays, and the drop is said once.
func TestExpireFileSessions(t *testing.T) {
var mu sync.Mutex
now := time.Date(2026, 9, 28, 10, 0, 0, 0, time.UTC)
advance := func(d time.Duration) { mu.Lock(); now = now.Add(d); mu.Unlock() }
st := &fileedit.Stage{Dir: t.TempDir(), MinFree: 1e-9, Now: func() time.Time {
mu.Lock()
defer mu.Unlock()
return now
}}
idle, err := st.Begin("u1", "survival", "a.jar", 3)
if err != nil {
t.Fatal(err)
}
advance(fileedit.SessionIdle - time.Minute)
fresh, err := st.Begin("u1", "survival", "b.jar", 3)
if err != nil {
t.Fatal(err)
}
advance(2 * time.Minute)
ctx, cancel := context.WithCancel(context.Background())
var out bytes.Buffer
done := make(chan struct{})
go func() { expireFileSessions(ctx, st, time.Millisecond, &out); close(done) }()
for deadline := time.Now().Add(5 * time.Second); ; time.Sleep(time.Millisecond) {
if _, err := st.Status("u1", "survival", idle.ID); errors.Is(err, fileedit.ErrNotStaged) {
break
}
if time.Now().After(deadline) {
cancel()
t.Fatal("the idle session was never dropped")
}
}
// A few more ticks with nothing idle, which must stay quiet.
time.Sleep(20 * time.Millisecond)
cancel()
<-done
if _, err := st.Status("u1", "survival", fresh.ID); err != nil {
t.Fatalf("the session touched since went too: %v", err)
}
if got := out.String(); got != "felis api: dropped 1 upload session(s) left idle for 6h0m0s\n" {
t.Fatalf("said %q", got)
}
}
type sweepCount struct{ n atomic.Int32 }
func (s *sweepCount) ExpireExports() { s.n.Add(1) }
// TestExpireExports: the loop sweeps on each tick, and returns once felis-api
// shuts down.
func TestExpireExports(t *testing.T) {
var s sweepCount
ctx, cancel := context.WithCancel(context.Background())
done := make(chan struct{})
go func() { expireExports(ctx, &s, time.Millisecond); close(done) }()
for deadline := time.Now().Add(5 * time.Second); s.n.Load() < 3; time.Sleep(time.Millisecond) {
if time.Now().After(deadline) {
cancel()
t.Fatalf("swept %d times in 5s at a 1ms tick", s.n.Load())
}
}
cancel()
select {
case <-done:
case <-time.After(5 * time.Second):
t.Fatal("the loop outlived its context")
}
}
+20 -16
View File
@@ -8,9 +8,7 @@ import (
"fmt"
"io"
"os"
"os/signal"
"strings"
"syscall"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/naming"
@@ -18,6 +16,10 @@ import (
apierrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/api/resource"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
)
@@ -109,15 +111,20 @@ func cmdApply(args []string, stdout, stderr io.Writer) int {
}
// ------- K8s client (one context, one client) -------
// The operator runs this on the node, where the kubeconfig is k3s's own file and
// neither $KUBECONFIG nor ~/.kube is set. buildSystemServerClient falls back to that
// file and names what it tried; ctrl.GetConfigOrDie exited 1 there without a word,
// because controller-runtime's logger is never set up in a CLI command.
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
cl, err := buildSystemServerClient()
// SetupSignalHandler must be called exactly once per process —
// controller-runtime panics on a second call. We create ctx and the
// K8s client here and thread both through every downstream call so no
// callee ever needs to call SetupSignalHandler again.
ctx := ctrl.SetupSignalHandler()
scheme := runtime.NewScheme()
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
utilruntime.Must(v1alpha1.AddToScheme(scheme))
cfg := ctrl.GetConfigOrDie()
cl, err := client.New(cfg, client.Options{Scheme: scheme})
if err != nil {
fmt.Fprintf(stderr, "felis apply: %v\n", err)
fmt.Fprintf(stderr, "felis apply: build client: %v\n", err)
return 1
}
@@ -160,10 +167,6 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
if err := naming.ValidateServerName(req.Subdomain); err != nil {
return nil, fmt.Errorf("invalid subdomain: %w", err)
}
displayName, err := naming.CleanDisplayName(req.DisplayName)
if err != nil {
return nil, fmt.Errorf("invalid displayName: %w", err)
}
if strings.TrimSpace(req.Image) == "" {
return nil, fmt.Errorf("image is required")
}
@@ -233,7 +236,7 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
},
Spec: v1alpha1.MinecraftServerSpec{
Subdomain: req.Subdomain,
DisplayName: displayName,
DisplayName: req.DisplayName,
Image: req.Image,
JavaMemory: deriveApplyJavaHeap(memLim),
DesiredState: v1alpha1.DesiredStopped,
@@ -261,7 +264,8 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
// request if any CRD already carries the given spec.subdomain. metadata.name
// uniqueness is enforced by K8s on Create, but spec.subdomain must be checked
// here because two CRDs with different names could otherwise share a subdomain.
// It reuses the caller's context and K8s client.
// It reuses the caller's context and K8s client — it never calls
// SetupSignalHandler or builds its own client.
func checkSubdomainUnique(ctx context.Context, cl client.Client, namespace, subdomain string) error {
var list v1alpha1.MinecraftServerList
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
-36
View File
@@ -1,10 +1,7 @@
package main
import (
"bytes"
"encoding/json"
"os"
"path/filepath"
"strings"
"testing"
@@ -142,7 +139,6 @@ func TestBuildMinecraftServerFromApplyRequest_Valid(t *testing.T) {
req := applyRequest{
Name: "test-server",
Subdomain: "test-server",
DisplayName: " Test Server ",
Image: "registry.felis.svc/paper:1.21",
Memory: "4Gi",
Storage: "20Gi",
@@ -160,9 +156,6 @@ func TestBuildMinecraftServerFromApplyRequest_Valid(t *testing.T) {
if ms.Spec.Subdomain != "test-server" {
t.Errorf("Subdomain = %q", ms.Spec.Subdomain)
}
if ms.Spec.DisplayName != "Test Server" {
t.Errorf("DisplayName = %q, want it trimmed to Test Server", ms.Spec.DisplayName)
}
if ms.Spec.Image != "registry.felis.svc/paper:1.21" {
t.Errorf("Image = %q", ms.Spec.Image)
}
@@ -287,11 +280,6 @@ func TestBuildMinecraftServerFromApplyRequest_Errors(t *testing.T) {
applyRequest{Name: ok, Subdomain: "", Image: "x", Memory: "1Gi", Storage: "1Gi"},
"invalid subdomain",
},
{
"display name with a tab",
applyRequest{Name: ok, Subdomain: ok, DisplayName: "a" + string(rune(0x09)) + "b", Image: "x", Memory: "1Gi", Storage: "1Gi"},
"invalid displayName",
},
{
"empty image",
applyRequest{Name: ok, Subdomain: ok, Image: "", Memory: "1Gi", Storage: "1Gi"},
@@ -382,27 +370,3 @@ func resList(specs ...string) corev1.ResourceList {
}
return rl
}
// TestApplyReportsAMissingKubeconfig pins the node-side failure: with no kubeconfig to
// find, apply says which ones it tried and exits 1. It used to call
// ctrl.GetConfigOrDie, which ended the process with exit 1 and nothing printed.
func TestApplyReportsAMissingKubeconfig(t *testing.T) {
if _, err := os.Stat(hostBootstrapKubeconfigPath); err == nil {
t.Skipf("%s exists on this machine", hostBootstrapKubeconfigPath)
}
dir := t.TempDir()
t.Setenv("KUBECONFIG", filepath.Join(dir, "missing"))
t.Setenv("KUBERNETES_SERVICE_HOST", "")
form := filepath.Join(dir, "server.json")
if err := os.WriteFile(form, []byte(`{"name":"alpha","subdomain":"alpha","image":"registry.felis.svc:5000/felis/paper:demo","memory":"1Gi","storage":"1Gi"}`), 0o600); err != nil {
t.Fatal(err)
}
var out, errw bytes.Buffer
if code := cmdApply([]string{"-f", form}, &out, &errw); code != 1 {
t.Fatalf("exit = %d, want 1; stderr %q", code, errw.String())
}
want := "felis apply: no reachable kubeconfig (tried in-cluster/$KUBECONFIG/~/.kube and " + hostBootstrapKubeconfigPath + "): stat " + hostBootstrapKubeconfigPath + ": no such file or directory\n"
if errw.String() != want || out.Len() != 0 {
t.Fatalf("stdout %q, stderr %q, want stderr %q", out.String(), errw.String(), want)
}
}
+20 -35
View File
@@ -15,6 +15,7 @@ import (
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/reaper"
"felis.lolicon.best/internal/store"
ctrl "sigs.k8s.io/controller-runtime"
)
@@ -37,7 +38,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
server := fs.String("server", "", "server name whose world is being backed up")
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
reason := fs.String("reason", reasonManual, "world_backups reason: manual, pre_restore for the safety snapshot in front of a restore, or scheduled for felis-api's daily restore point")
reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
if err := fs.Parse(args); err != nil {
return 2
@@ -46,8 +47,9 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
fmt.Fprintln(stderr, "felis backup: --server is required")
return 2
}
if _, _, ok := backupPolicy(*reason, reaper.DefaultConfig()); !ok {
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual, %s or %s)\n", *reason, backupjob.ReasonPreRestore, backupjob.ReasonScheduled)
keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
if !ok {
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
return 2
}
@@ -60,14 +62,16 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
return 1
}
// The [archive] parse the reaper uses: it holds each reason's keep and
// retention.
// The [archive] parse the reaper uses; an on-demand backup takes its
// manual_retention and manual_keep.
rcfg, err := reaperConfig(cfg)
if err != nil {
fmt.Fprintf(stderr, "felis backup: %v\n", err)
return 1
}
keep, retention, _ := backupPolicy(*reason, rcfg)
if keep < 0 {
keep = rcfg.ManualKeep
}
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
@@ -99,7 +103,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
len(a.Skipped), strings.Join(a.Skipped[:min(len(a.Skipped), 10)], ", "))
}
drv, err := openPodStore(ctx, cfg.Database.URL, "backup", stderr)
drv, err := store.Open(ctx, cfg.Database.URL)
if err != nil {
fmt.Fprintf(stderr, "felis backup: open database: %v\n", err)
return 1
@@ -113,7 +117,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
BackupRef: string(ref),
SizeBytes: size,
Reason: *reason,
ExpiresAt: time.Now().Add(retention),
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
SHA256: a.SHA256,
SkippedEntries: len(a.Skipped),
@@ -132,7 +136,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
}
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
pruneBackups(ctx, st, archiver, *server, *formerOwner, *reason, keep, *protect, stdout, stderr)
pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
return 0
}
@@ -144,32 +148,13 @@ const (
preRestoreKeep = 3
)
// backupPolicy is how many backups of one reason a server keeps and how long
// each lives: an owner's own backups and the safety snapshots in front of a
// restore by [archive] manual_keep / manual_retention (the snapshots capped at
// preRestoreKeep), felis-api's daily restore points by scheduled_keep /
// scheduled_retention, so neither kind crowds out the other. ok is false for a
// reason this command does not record.
func backupPolicy(reason string, rcfg reaper.Config) (keep int, retention time.Duration, ok bool) {
switch reason {
case reasonManual:
return rcfg.ManualKeep, rcfg.ManualRetention, true
case backupjob.ReasonPreRestore:
return preRestoreKeep, rcfg.ManualRetention, true
case backupjob.ReasonScheduled:
return rcfg.ScheduledKeep, rcfg.ScheduledRetention, true
}
return 0, 0, false
}
// pruneBackups keeps the newest keep backups of this reason that owner holds of
// server and removes the rest, oldest first, so repeated backups of one world
// cannot fill the shared archive store and a new owner's backups never remove a
// previous owner's. protect is never removed: it is the backup a chained restore
// is about to extract. The new backup is already recorded; a removal that fails
// is reported and retried after the next backup.
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, owner, reason string, keep int, protect string, stdout, stderr io.Writer) {
excess, err := st.ExcessBackups(ctx, server, owner, reason, keep, protect)
// pruneBackups keeps server's newest keep backups of this reason and removes the
// rest, oldest first, so repeated backups of one world cannot fill the shared
// archive store. protect is never removed: it is the backup a chained restore is
// about to extract. The new backup is already recorded; a removal that fails is
// reported and retried after the next backup.
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
if err != nil {
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
return
-29
View File
@@ -4,9 +4,6 @@ import (
"bytes"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/reaper"
)
// The reason decides which backups the new one's prune may remove, so an
@@ -20,29 +17,3 @@ func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
t.Fatalf("stderr = %q", stderr.String())
}
}
// Each reason is pruned and expired by its own [archive] keys: a daily
// restore point must never count against, or take the lifetime of, the
// backups an owner asked for.
func TestBackupPolicyPerReason(t *testing.T) {
rcfg := reaper.DefaultConfig()
rcfg.ManualKeep, rcfg.ManualRetention = 5, 30*reaper.Day
rcfg.ScheduledKeep, rcfg.ScheduledRetention = 7, 90*reaper.Day
for _, tc := range []struct {
reason string
keep int
retention time.Duration
}{
{"manual", 5, 30 * reaper.Day},
{"pre_restore", preRestoreKeep, 30 * reaper.Day},
{"scheduled", 7, 90 * reaper.Day},
} {
keep, retention, ok := backupPolicy(tc.reason, rcfg)
if !ok || keep != tc.keep || retention != tc.retention {
t.Errorf("backupPolicy(%q) = (%d, %v, %v); want (%d, %v, true)", tc.reason, keep, retention, ok, tc.keep, tc.retention)
}
}
if _, _, ok := backupPolicy("inactive_15d", rcfg); ok {
t.Error("backupPolicy accepted inactive_15d; the reaper records those itself")
}
}
+2 -453
View File
@@ -4,28 +4,15 @@ import (
"bytes"
"context"
"encoding/json"
"errors"
"flag"
"fmt"
"io"
"net/http"
"os"
"os/signal"
"sort"
"strings"
"syscall"
"time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/operator"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
@@ -35,9 +22,7 @@ import (
// (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console
// cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth)
// while the API is alive, and the API renders the Job and audits the action. This file
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. The
// `felis backup-now` command (cmdBackupNow, below) takes the same route for every
// user server in turn.
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue.
// backupNowOutcome is the durable result of a backup request, re-printed after the TUI
// alt-screen tears down.
@@ -90,7 +75,7 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o
resp, err := hc.Do(req)
if err != nil {
return backupNowOutcome{}, fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
return backupNowOutcome{}, fmt.Errorf("felis-api unreachable (a backup needs it alive): %w", err)
}
defer resp.Body.Close()
@@ -160,439 +145,3 @@ func backupPickable(servers []haltableServer) []haltableServer {
}
return out
}
// cmdBackupNow is `felis backup-now`: the world of every user server (or of the
// named ones) archived now, one at a time, through the internal backup route the
// console's Sync uses. A world lives only in its volume and the off-site copy holds
// only its archives, so this is the lever in front of a planned move to another
// host, a disk swap or anything else that could lose a volume.
func cmdBackupNow(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("backup-now", flag.ContinueOnError)
fs.SetOutput(stderr)
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
yes := fs.Bool("yes", false, "back up; without it the plan is printed and nothing changes")
stop := fs.Bool("stop", false, "stop the running servers first: their players are disconnected and the servers stay stopped")
fs.Usage = func() {
fmt.Fprintln(stderr, "Usage: felis backup-now [-yes] [-stop] [server ...]")
fmt.Fprintln(stderr)
fmt.Fprintln(stderr, "Archives the world of every user server, or of the named ones, one at a time, and waits for each archive.")
fmt.Fprintln(stderr, "A running server is skipped unless -stop is given. Without -yes it prints what it would do.")
fs.PrintDefaults()
}
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if os.Geteuid() != 0 {
fmt.Fprintln(stderr, "felis backup-now: refused — it reads the cluster's ops token, so it must run as root (try: sudo felis backup-now)")
return 1
}
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
return 1
}
cl, err := buildSystemServerClient()
if err != nil {
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
return 1
}
ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer cancel()
ns := cfg.K8s.Namespace
osUser := accountableOSUser()
jobs := api.NewK8sJobStatus(cl, ns)
var baseURL, token string
var repo ownerStore
var drv *store.PostgresDriver
defer func() {
if drv != nil {
_ = drv.Close()
}
}()
hc := &http.Client{Timeout: 10 * time.Second}
r := backupNowRun{
out: stdout,
errw: stderr,
ns: ns,
list: func(ctx context.Context) ([]backupNowWorld, error) { return listBackupNowWorlds(ctx, cl, ns) },
stopped: func(ctx context.Context, name string) (bool, error) {
var ms v1alpha1.MinecraftServer
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &ms); err != nil {
return false, err
}
return backupNowStopped(ctx, cl, &ms)
},
halt: func(ctx context.Context, name string) error {
// The database is opened only once a server is to be stopped: the halt
// is audited like the console's, and a run with nothing running needs
// no more than the API.
if repo == nil {
d, err := store.Open(ctx, cfg.Database.URL)
if err != nil {
return fmt.Errorf("open the database for the audit log: %w", err)
}
drv, repo = d, api.NewPGRepo(d.DB())
}
out, err := performHalt(ctx, cl, repo, ns, name, osUser)
if err != nil {
return err
}
if out.auditErr != nil {
fmt.Fprintf(stderr, "felis backup-now: the audit row for stopping %s was not written: %v\n", name, out.auditErr)
}
return nil
},
request: func(ctx context.Context, name string) error {
if baseURL == "" {
var err error
if baseURL, token, err = resolveInternalAPI(ctx, cl, platform.DefaultControlNamespace); err != nil {
return fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
}
}
_, err := requestBackup(ctx, hc, baseURL, token, name, osUser)
return err
},
jobs: jobs.LatestJobs,
now: time.Now,
sleep: func(ctx context.Context, d time.Duration) { sleepCtx(ctx, d) },
}
return r.run(ctx, fs.Args(), *yes, *stop)
}
// sleepCtx waits d or until ctx ends.
func sleepCtx(ctx context.Context, d time.Duration) {
t := time.NewTimer(d)
defer t.Stop()
select {
case <-ctx.Done():
case <-t.C:
}
}
// backupNowWorld is one user server as backup-now sees it.
type backupNowWorld struct {
name string
phase string // the observed phase, or the desired state before the operator reconciled it
stopped bool // the backup route's stopped gate admits it
hasWorld bool // its world volume exists
}
// listBackupNowWorlds lists the user servers of namespace in the API server's
// order (by name), each with what the backup route checks. System servers are
// left out: they have no row in the servers table, so the route refuses them
// (backupPickable).
func listBackupNowWorlds(ctx context.Context, cl client.Client, namespace string) ([]backupNowWorld, error) {
var list v1alpha1.MinecraftServerList
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
return nil, err
}
var out []backupNowWorld
for i := range list.Items {
ms := &list.Items[i]
if isSystemServer(ms.Name) {
continue
}
stopped, err := backupNowStopped(ctx, cl, ms)
if err != nil {
return nil, err
}
var pvc corev1.PersistentVolumeClaim
err = cl.Get(ctx, types.NamespacedName{Namespace: namespace, Name: naming.WorldPVCName(ms.Name)}, &pvc)
if err != nil && !apierrors.IsNotFound(err) {
return nil, fmt.Errorf("look up the world volume of %s: %w", ms.Name, err)
}
phase := string(ms.Status.Phase)
if phase == "" {
phase = string(ms.Spec.DesiredState)
}
out = append(out, backupNowWorld{name: ms.Name, phase: phase, stopped: stopped, hasWorld: err == nil})
}
return out, nil
}
// backupNowStopped is the backup route's stopped gate (api.enqueueBackup and
// K8sCluster.AcquireMaintenance together): desired Stopped, not ready, phase
// Stopped and no game pod left.
func backupNowStopped(ctx context.Context, cl client.Client, ms *v1alpha1.MinecraftServer) (bool, error) {
if ms.Spec.DesiredState != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
return false, nil
}
var pods corev1.PodList
if err := cl.List(ctx, &pods, client.InNamespace(ms.Namespace), client.MatchingLabels{
v1alpha1.LabelServer: ms.Name, v1alpha1.LabelComponent: operator.ComponentValue,
}); err != nil {
return false, fmt.Errorf("look up the pod of %s: %w", ms.Name, err)
}
return len(pods.Items) == 0, nil
}
// errBackupAPIUnreachable is a backup request that never reached felis-api. It
// ends a backup-now run: every later world would fail the same way, and stopping
// servers for backups that cannot be taken only takes them away from players.
var errBackupAPIUnreachable = errors.New("felis-api unreachable (a backup needs it alive)")
// Polling of backup-now. A graceful stop saves the world first; the backup Job's
// own deadline (backupjob, 30 minutes) ends a Job that hangs, so its wait needs no
// cap of its own.
const (
backupNowPoll = 2 * time.Second
backupNowStopWait = 10 * time.Minute
backupNowJobAppear = time.Minute
)
// backupNowRun is backup-now over seams, so the plan and the run are tested
// without a cluster or felis-api.
type backupNowRun struct {
out, errw io.Writer
ns string // where the servers and their Jobs live, for the kubectl hints
list func(ctx context.Context) ([]backupNowWorld, error)
stopped func(ctx context.Context, name string) (bool, error)
halt func(ctx context.Context, name string) error
request func(ctx context.Context, name string) error
jobs func(ctx context.Context, name string) ([]api.AsyncJob, error)
now func() time.Time
sleep func(ctx context.Context, d time.Duration)
}
// pickBackupNowWorlds narrows worlds to names, in the order given, or keeps them
// all when names is empty.
func pickBackupNowWorlds(worlds []backupNowWorld, names []string) ([]backupNowWorld, error) {
if len(names) == 0 {
return worlds, nil
}
byName := make(map[string]backupNowWorld, len(worlds))
for _, w := range worlds {
byName[w.name] = w
}
var out []backupNowWorld
seen := map[string]bool{}
for _, n := range names {
if seen[n] {
continue
}
seen[n] = true
w, ok := byName[n]
switch {
case ok:
out = append(out, w)
case isSystemServer(n):
return nil, fmt.Errorf("%s is a system server: its world is rebuilt by felis setup and has no backups", n)
default:
return nil, fmt.Errorf("no server named %q", n)
}
}
return out, nil
}
// backupNowAction is what the plan does with a world.
func backupNowAction(w backupNowWorld, stop bool) string {
switch {
case w.stopped && !w.hasWorld:
return "skip: no world volume (never started, nothing to save)"
case w.stopped:
return "back up"
case stop:
return "stop, then back up"
default:
return "skip: running (stop it first, or pass -stop)"
}
}
func (r *backupNowRun) run(ctx context.Context, names []string, yes, stop bool) int {
all, err := r.list(ctx)
if err != nil {
fmt.Fprintf(r.errw, "felis backup-now: list the servers: %v\n", err)
return 1
}
worlds, err := pickBackupNowWorlds(all, names)
if err != nil {
fmt.Fprintf(r.errw, "felis backup-now: %v\n", err)
return 2
}
if len(worlds) == 0 {
fmt.Fprintln(r.out, "felis backup-now: there are no user servers")
return 0
}
// The stopped worlds go first, so a felis-api that cannot take a backup is
// found before any server is stopped for one.
sort.SliceStable(worlds, func(i, j int) bool { return worlds[i].stopped && !worlds[j].stopped })
width := 0
for _, w := range worlds {
width = max(width, len(w.name))
}
fmt.Fprintf(r.out, "felis backup-now: %d server(s), backed up one at a time:\n", len(worlds))
stopping, work := false, 0
for _, w := range worlds {
fmt.Fprintf(r.out, " %-*s %-8s %s\n", width, w.name, w.phase, backupNowAction(w, stop))
stopping = stopping || (!w.stopped && stop)
if (w.stopped && w.hasWorld) || (!w.stopped && stop) {
work++
}
}
if work > 0 {
fmt.Fprintln(r.out, "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.")
}
if stopping {
fmt.Fprintln(r.out, "Stopping disconnects the players on those servers, and they stay stopped afterwards.")
}
if !yes {
if work == 0 {
fmt.Fprintln(r.out, "Nothing to back up.")
} else {
fmt.Fprintln(r.out, "Nothing changed. Run again with -yes to back them up.")
}
return 0
}
var done, failed, running, empty int
var leftStopped []string
for _, w := range worlds {
if err := ctx.Err(); err != nil {
break
}
switch {
case w.stopped && !w.hasWorld:
empty++
continue
case !w.stopped && !stop:
running++
continue
}
if !w.stopped {
halted, err := r.stopWorld(ctx, w.name)
if halted {
leftStopped = append(leftStopped, w.name)
}
if err != nil {
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
failed++
continue
}
}
err := r.backUp(ctx, w.name)
if errors.Is(err, errBackupAPIUnreachable) {
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
fmt.Fprintln(r.out, "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).")
failed++
break
}
if err != nil {
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
failed++
continue
}
done++
}
interrupted := ctx.Err() != nil
if interrupted {
fmt.Fprintln(r.out, "Interrupted: a backup Job already started runs to its end.")
}
parts := []string{fmt.Sprintf("%d backed up", done)}
if failed > 0 {
parts = append(parts, fmt.Sprintf("%d failed", failed))
}
if running > 0 {
parts = append(parts, fmt.Sprintf("%d skipped (running)", running))
}
if empty > 0 {
parts = append(parts, fmt.Sprintf("%d without a world", empty))
}
fmt.Fprintf(r.out, "%s.\n", strings.Join(parts, ", "))
if len(leftStopped) > 0 {
fmt.Fprintf(r.out, "Left stopped: %s. Start them from the panel when you are done.\n", strings.Join(leftStopped, ", "))
}
if done > 0 {
fmt.Fprintln(r.out, "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.")
}
if failed > 0 || running > 0 || interrupted {
return 1
}
return 0
}
// stopWorld stops one server and waits until the backup route would admit it.
// halted is whether the stop was asked for: the server then stays stopped, even
// when it takes longer than the wait.
func (r *backupNowRun) stopWorld(ctx context.Context, name string) (halted bool, err error) {
fmt.Fprintf(r.out, " %s: stopping\n", name)
if err := r.halt(ctx, name); err != nil {
return false, err
}
start := r.now()
for {
ok, err := r.stopped(ctx, name)
if err != nil {
return true, err
}
if ok {
fmt.Fprintf(r.out, " %s: stopped after %s\n", name, r.now().Sub(start).Round(time.Second))
return true, nil
}
if r.now().Sub(start) >= backupNowStopWait {
return true, fmt.Errorf("did not stop within %s (kubectl -n %s describe minecraftserver %s)", backupNowStopWait, r.ns, name)
}
if err := ctx.Err(); err != nil {
return true, err
}
r.sleep(ctx, backupNowPoll)
}
}
// backUp requests one world's backup and waits for its Job to finish. The Job is
// the one of this server that was not there before the request.
func (r *backupNowRun) backUp(ctx context.Context, name string) error {
before, err := r.jobs(ctx, name)
if err != nil {
return fmt.Errorf("list its backup Jobs: %w", err)
}
known := make(map[string]bool, len(before))
for _, j := range before {
known[j.Name] = true
}
if err := r.request(ctx, name); err != nil {
return err
}
fmt.Fprintf(r.out, " %s: backing up\n", name)
start := r.now()
seen := ""
for {
jobs, err := r.jobs(ctx, name)
if err != nil {
return fmt.Errorf("list its backup Jobs: %w", err)
}
var job *api.AsyncJob
for i := range jobs {
if jobs[i].Kind == "backup" && (jobs[i].Name == seen || seen == "" && !known[jobs[i].Name]) {
job = &jobs[i]
break
}
}
switch {
case job == nil && seen != "":
return fmt.Errorf("its backup Job %s was deleted before it finished", seen)
case job == nil && r.now().Sub(start) >= backupNowJobAppear:
return fmt.Errorf("felis-api accepted the backup, but no backup Job appeared within %s (kubectl -n %s get jobs)", backupNowJobAppear, r.ns)
case job != nil && job.State == "succeeded":
fmt.Fprintf(r.out, " %s: archived in %s\n", name, r.now().Sub(start).Round(time.Second))
return nil
case job != nil && job.State == "failed":
msg := job.Message
if msg == "" {
msg = "the backup Job failed"
}
return fmt.Errorf("%s (kubectl -n %s logs job/%s)", msg, r.ns, job.Name)
case job != nil:
seen = job.Name
}
if err := ctx.Err(); err != nil {
return err
}
r.sleep(ctx, backupNowPoll)
}
}
-563
View File
@@ -1,563 +0,0 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"reflect"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/naming"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// bnRig drives backupNowRun against scripted servers and Jobs on a fake clock that
// moves only when the run sleeps.
type bnRig struct {
out, errw bytes.Buffer
clock time.Time
events []string
worlds []backupNowWorld
// stopAfter is how many polls a halted server takes to stop; -1 never does.
stopAfter map[string]int
polls map[string]int
reqErr map[string]error
// states is what the server's new backup Job reports on each poll after the
// request, the last one repeating; "" is no Job.
states map[string][]string
messages map[string]string
jobPolls map[string]int
// stopErr fails a server's stop polls; jobsFail fails the nth (1-based) Job
// list of a server.
stopErr map[string]error
jobsFail map[string]int
jobCalls map[string]int
// ctx is the run's context, and onAct runs after each halt and request.
ctx context.Context
onAct func(event string)
}
func newBNRig(worlds ...backupNowWorld) *bnRig {
return &bnRig{
clock: time.Unix(1_800_000_000, 0),
worlds: worlds,
stopAfter: map[string]int{},
polls: map[string]int{},
reqErr: map[string]error{},
states: map[string][]string{},
messages: map[string]string{},
jobPolls: map[string]int{},
stopErr: map[string]error{},
jobsFail: map[string]int{},
jobCalls: map[string]int{},
ctx: context.Background(),
onAct: func(string) {},
}
}
func (g *bnRig) run(names []string, yes, stop bool) int {
r := backupNowRun{
out: &g.out, errw: &g.errw, ns: "minecraft",
list: func(context.Context) ([]backupNowWorld, error) {
return append([]backupNowWorld(nil), g.worlds...), nil
},
stopped: func(_ context.Context, name string) (bool, error) {
g.polls[name]++
if err := g.stopErr[name]; err != nil {
return false, err
}
n := g.stopAfter[name]
return n >= 0 && g.polls[name] > n, nil
},
halt: func(_ context.Context, name string) error {
g.events = append(g.events, "halt "+name)
g.onAct("halt " + name)
return nil
},
request: func(_ context.Context, name string) error {
g.events = append(g.events, "request "+name)
g.onAct("request " + name)
if err := g.reqErr[name]; err != nil {
return err
}
g.jobPolls[name] = 0
return nil
},
jobs: func(_ context.Context, name string) ([]api.AsyncJob, error) {
// Every server has an older finished backup, and a restore Job that
// shows up with the new backup: neither is the Job to wait for.
g.jobCalls[name]++
if g.jobCalls[name] == g.jobsFail[name] {
return nil, errors.New("the apiserver is gone")
}
out := []api.AsyncJob{{Name: "backup-" + name + "-old", Kind: "backup", State: "succeeded"}}
n, requested := g.jobPolls[name]
if !requested {
return out, nil
}
g.jobPolls[name] = n + 1
states := g.states[name]
if len(states) == 0 {
states = []string{"succeeded"}
}
state := states[min(n, len(states)-1)]
if state == "" {
return out, nil
}
return append([]api.AsyncJob{
{Name: "restore-" + name + "-x", Kind: "restore", State: "succeeded"},
{Name: "backup-" + name + "-new", Kind: "backup", State: state, Message: g.messages[name]},
}, out...), nil
},
now: func() time.Time { return g.clock },
sleep: func(_ context.Context, d time.Duration) { g.clock = g.clock.Add(d) },
}
return r.run(g.ctx, names, yes, stop)
}
func stoppedWorld(name string) backupNowWorld {
return backupNowWorld{name: name, phase: "Stopped", stopped: true, hasWorld: true}
}
func runningWorld(name string) backupNowWorld {
return backupNowWorld{name: name, phase: "Running", hasWorld: true}
}
func emptyWorld(name string) backupNowWorld {
return backupNowWorld{name: name, phase: "Stopped", stopped: true}
}
const (
bnManualKeep = "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.\n"
bnStopping = "Stopping disconnects the players on those servers, and they stay stopped afterwards.\n"
bnOffsite = "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.\n"
)
func TestBackupNowPlanChangesNothing(t *testing.T) {
for _, tc := range []struct {
stop bool
want string
}{
{false, "felis backup-now: 4 server(s), backed up one at a time:\n" +
" alpha Stopped back up\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running skip: running (stop it first, or pass -stop)\n" +
" delta Starting skip: running (stop it first, or pass -stop)\n" +
bnManualKeep +
"Nothing changed. Run again with -yes to back them up.\n"},
{true, "felis backup-now: 4 server(s), backed up one at a time:\n" +
" alpha Stopped back up\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running stop, then back up\n" +
" delta Starting stop, then back up\n" +
bnManualKeep + bnStopping +
"Nothing changed. Run again with -yes to back them up.\n"},
} {
t.Run(fmt.Sprintf("stop=%v", tc.stop), func(t *testing.T) {
delta := runningWorld("delta")
delta.phase = "Starting"
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), delta)
if code := g.run(nil, false, tc.stop); code != 0 {
t.Fatalf("exit = %d, want 0; stderr %q", code, g.errw.String())
}
if g.out.String() != tc.want {
t.Fatalf("plan =\n%s\nwant\n%s", g.out.String(), tc.want)
}
if len(g.events) != 0 || len(g.polls) != 0 {
t.Fatalf("the plan acted: events %v, polls %v", g.events, g.polls)
}
})
}
}
// A plan that saves nothing says so, without the manual_keep warning; a running
// server counts as something to save once -stop is given.
func TestBackupNowPlanWithNothingToSave(t *testing.T) {
for _, tc := range []struct {
stop bool
want string
}{
{false, "felis backup-now: 2 server(s), backed up one at a time:\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running skip: running (stop it first, or pass -stop)\n" +
"Nothing to back up.\n"},
{true, "felis backup-now: 2 server(s), backed up one at a time:\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
" bravo Running stop, then back up\n" +
bnManualKeep + bnStopping +
"Nothing changed. Run again with -yes to back them up.\n"},
} {
g := newBNRig(runningWorld("bravo"), emptyWorld("charlie"))
if code := g.run(nil, false, tc.stop); code != 0 {
t.Fatalf("stop=%v: exit = %d, want 0", tc.stop, code)
}
if g.out.String() != tc.want {
t.Fatalf("stop=%v: plan =\n%s\nwant\n%s", tc.stop, g.out.String(), tc.want)
}
}
}
func TestBackupNowBacksUpEachWorldInTurn(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), stoppedWorld("delta"))
g.states["alpha"] = []string{"", "running", "succeeded"}
g.states["delta"] = []string{"running", "failed"}
g.messages["delta"] = "felis backup: not enough free disk for the archive"
g.stopAfter["bravo"] = 2
g.states["bravo"] = []string{"running", "running", "running", "succeeded"}
code := g.run(nil, true, true)
if code != 1 {
t.Fatalf("exit = %d, want 1 (delta failed)", code)
}
if want := []string{"request alpha", "request delta", "halt bravo", "request bravo"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
_, run, _ := strings.Cut(g.out.String(), bnStopping+"")
want := " alpha: backing up\n" +
" alpha: archived in 4s\n" +
" delta: backing up\n" +
" delta: failed: felis backup: not enough free disk for the archive (kubectl -n minecraft logs job/backup-delta-new)\n" +
" bravo: stopping\n" +
" bravo: stopped after 4s\n" +
" bravo: backing up\n" +
" bravo: archived in 6s\n" +
"2 backed up, 1 failed, 1 without a world.\n" +
"Left stopped: bravo. Start them from the panel when you are done.\n" +
bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
}
func TestBackupNowSkipsRunningServersWithoutStop(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"))
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1 (bravo was not backed up)", code)
}
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
if len(g.polls) != 0 {
t.Fatalf("polled a server it did not stop: %v", g.polls)
}
_, run, _ := strings.Cut(g.out.String(), "Nothing changed")
if run != "" {
t.Fatalf("-yes printed the plan's closing line")
}
_, run, _ = strings.Cut(g.out.String(), bnManualKeep)
want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 skipped (running).\n" + bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
}
func TestBackupNowExitsCleanWhenEverythingIsSaved(t *testing.T) {
// -stop with nothing running stops nothing and says nothing about stopping.
g := newBNRig(stoppedWorld("alpha"), emptyWorld("charlie"))
if code := g.run(nil, true, true); code != 0 {
t.Fatalf("exit = %d, want 0; output\n%s", code, g.out.String())
}
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
if want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 without a world.\n" + bnOffsite; run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
g = newBNRig(emptyWorld("charlie"))
if code := g.run(nil, true, false); code != 0 {
t.Fatalf("exit = %d, want 0 for a server with nothing to save", code)
}
// Nothing to archive: no manual_keep warning, and no off-site hint.
if want := "felis backup-now: 1 server(s), backed up one at a time:\n" +
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
"0 backed up, 1 without a world.\n"; g.out.String() != want {
t.Fatalf("output =\n%s\nwant\n%s", g.out.String(), want)
}
if len(g.events) != 0 {
t.Fatalf("events = %v, want none", g.events)
}
}
func TestBackupNowStopsAtAnUnreachableAPI(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), runningWorld("charlie"))
g.reqErr["alpha"] = fmt.Errorf("%w: dial tcp 10.43.0.9:8081: connect: connection refused", errBackupAPIUnreachable)
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v: nothing after the API proved unreachable, and no server stopped", g.events, want)
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
want := " alpha: failed: felis-api unreachable (a backup needs it alive): dial tcp 10.43.0.9:8081: connect: connection refused\n" +
"Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).\n" +
"0 backed up, 1 failed.\n"
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
// Any other refusal is that world's alone: the run goes on.
g = newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
g.reqErr["alpha"] = errors.New("felis-api: the world is being restored")
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"request alpha", "request bravo"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
}
func TestBackupNowGivesUpOnAServerThatDoesNotStop(t *testing.T) {
g := newBNRig(runningWorld("bravo"), runningWorld("echo"))
g.stopAfter["bravo"] = -1
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"halt bravo", "halt echo", "request echo"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
// One poll at the start and one per 2s sleep up to the 10-minute mark.
if g.polls["bravo"] != 301 {
t.Fatalf("bravo polled %d times, want 301", g.polls["bravo"])
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
want := " bravo: stopping\n" +
" bravo: failed: did not stop within 10m0s (kubectl -n minecraft describe minecraftserver bravo)\n" +
" echo: stopping\n" +
" echo: stopped after 0s\n" +
" echo: backing up\n" +
" echo: archived in 0s\n" +
"1 backed up, 1 failed.\n" +
"Left stopped: bravo, echo. Start them from the panel when you are done.\n" +
bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
}
func TestBackupNowReportsAJobThatNeverRuns(t *testing.T) {
for _, tc := range []struct {
name string
states []string
polls int
want string
}{
{"never appears", []string{""}, 31,
"felis-api accepted the backup, but no backup Job appeared within 1m0s (kubectl -n minecraft get jobs)"},
{"deleted while running", []string{"running", "running", ""}, 3,
"its backup Job backup-alpha-new was deleted before it finished"},
{"fails without a message", []string{"failed"}, 1,
"the backup Job failed (kubectl -n minecraft logs job/backup-alpha-new)"},
} {
t.Run(tc.name, func(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"))
g.states["alpha"] = tc.states
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if !strings.Contains(g.out.String(), " alpha: failed: "+tc.want+"\n") {
t.Fatalf("output =\n%s\nwant the line %q", g.out.String(), tc.want)
}
if g.jobPolls["alpha"] != tc.polls {
t.Fatalf("polled the Jobs %d times after the request, want %d", g.jobPolls["alpha"], tc.polls)
}
})
}
}
func TestBackupNowNamedServers(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), stoppedWorld("delta"))
if code := g.run([]string{"delta", "alpha", "delta"}, true, false); code != 0 {
t.Fatalf("exit = %d, want 0", code)
}
if want := []string{"request delta", "request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
if !strings.HasPrefix(g.out.String(), "felis backup-now: 2 server(s), backed up one at a time:\n delta Stopped back up\n alpha Stopped back up\n") {
t.Fatalf("plan =\n%s", g.out.String())
}
for _, tc := range []struct{ name, want string }{
{"login", "felis backup-now: login is a system server: its world is rebuilt by felis setup and has no backups\n"},
{"lobby", "felis backup-now: lobby is a system server: its world is rebuilt by felis setup and has no backups\n"},
{"nope", "felis backup-now: no server named \"nope\"\n"},
} {
g := newBNRig(stoppedWorld("alpha"))
if code := g.run([]string{"alpha", tc.name}, true, false); code != 2 {
t.Fatalf("%s: exit = %d, want 2", tc.name, code)
}
if g.errw.String() != tc.want {
t.Fatalf("%s: stderr = %q, want %q", tc.name, g.errw.String(), tc.want)
}
if len(g.events) != 0 || g.out.Len() != 0 {
t.Fatalf("%s: acted on a bad name: events %v, output %q", tc.name, g.events, g.out.String())
}
}
g = newBNRig()
if code := g.run(nil, true, false); code != 0 || g.out.String() != "felis backup-now: there are no user servers\n" {
t.Fatalf("empty fleet: exit %d, output %q", code, g.out.String())
}
}
// The world list mirrors the backup route's own gate, so the plan says exactly what
// the route would refuse.
func TestListBackupNowWorlds(t *testing.T) {
withStatus := func(ms *v1alpha1.MinecraftServer, ready bool) *v1alpha1.MinecraftServer {
ms.Status.Ready = ready
return ms
}
pvc := func(server, ns string) *corev1.PersistentVolumeClaim {
return &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: naming.WorldPVCName(server), Namespace: ns}}
}
pod := func(name, ns string, labels map[string]string) *corev1.Pod {
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: labels}}
}
gameLabels := func(server string) map[string]string {
return map[string]string{v1alpha1.LabelServer: server, v1alpha1.LabelComponent: "server"}
}
other := mcServer("zulu", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
other.Namespace = "elsewhere"
hotel := mcServer("hotel", v1alpha1.DesiredStopped, "")
objs := []client.Object{
mcServer("alpha", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("alpha", haltNS),
// A backup Job's pod carries the server label with its own component, and
// a game pod of the same name in another namespace is somebody else's.
pod("backup-alpha-1-x", haltNS, map[string]string{v1alpha1.LabelServer: "alpha", "app.kubernetes.io/component": "world-backup"}),
pod("alpha-0", "elsewhere", gameLabels("alpha")),
withStatus(mcServer("bravo", v1alpha1.DesiredRunning, v1alpha1.PhaseRunning), true), pvc("bravo", haltNS), pod("bravo-0", haltNS, gameLabels("bravo")),
mcServer("charlie", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped),
mcServer("delta", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("delta", haltNS), pod("delta-0", haltNS, gameLabels("delta")),
mcServer("echo", v1alpha1.DesiredStopped, v1alpha1.PhaseStopping), pvc("echo", haltNS),
mcServer("foxtrot", v1alpha1.DesiredRunning, v1alpha1.PhaseStopped), pvc("foxtrot", haltNS),
withStatus(mcServer("golf", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), true), pvc("golf", haltNS),
hotel, pvc("hotel", haltNS),
mcServer("login", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("login", haltNS),
mcServer("lobby", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("lobby", haltNS),
other, pvc("zulu", "elsewhere"),
}
got, err := listBackupNowWorlds(context.Background(), haltClient(t, objs...), haltNS)
if err != nil {
t.Fatal(err)
}
want := []backupNowWorld{
{name: "alpha", phase: "Stopped", stopped: true, hasWorld: true},
{name: "bravo", phase: "Running", hasWorld: true},
{name: "charlie", phase: "Stopped", stopped: true},
{name: "delta", phase: "Stopped", hasWorld: true}, // its pod is still going
{name: "echo", phase: "Stopping", hasWorld: true}, // still stopping
{name: "foxtrot", phase: "Stopped", hasWorld: true}, // asked to start
{name: "golf", phase: "Stopped", hasWorld: true}, // still reports ready
{name: "hotel", phase: "Stopped", hasWorld: true}, // not reconciled yet: the desired state shows
}
if !reflect.DeepEqual(got, want) {
t.Fatalf("worlds =\n%+v\nwant\n%+v", got, want)
}
}
func TestBackupNowStopsWhenInterrupted(t *testing.T) {
t.Run("between worlds", func(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
g.ctx = ctx
g.onAct = func(e string) {
if e == "request alpha" {
cancel()
}
}
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
t.Fatalf("events = %v, want %v", g.events, want)
}
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
want := " alpha: backing up\n alpha: archived in 0s\n" +
"Interrupted: a backup Job already started runs to its end.\n" +
"1 backed up.\n" + bnOffsite
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
})
t.Run("while a Job runs", func(t *testing.T) {
g := newBNRig(stoppedWorld("alpha"))
g.states["alpha"] = []string{"running"}
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
g.ctx = ctx
g.onAct = func(string) { cancel() }
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if g.jobPolls["alpha"] != 1 {
t.Fatalf("polled the Job %d times after the interrupt, want 1", g.jobPolls["alpha"])
}
if !strings.Contains(g.out.String(), " alpha: failed: context canceled\nInterrupted: a backup Job already started runs to its end.\n0 backed up, 1 failed.\n") {
t.Fatalf("output =\n%s", g.out.String())
}
})
t.Run("while a server stops", func(t *testing.T) {
g := newBNRig(runningWorld("bravo"))
g.stopAfter["bravo"] = -1
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
g.ctx = ctx
g.onAct = func(string) { cancel() }
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
if g.polls["bravo"] != 1 {
t.Fatalf("polled bravo %d times after the interrupt, want 1", g.polls["bravo"])
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
want := " bravo: stopping\n bravo: failed: context canceled\n" +
"Interrupted: a backup Job already started runs to its end.\n" +
"0 backed up, 1 failed.\n" +
"Left stopped: bravo. Start them from the panel when you are done.\n"
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
})
}
func TestBackupNowReportsClusterErrors(t *testing.T) {
g := newBNRig(runningWorld("bravo"))
g.stopErr["bravo"] = errors.New("the apiserver is gone")
if code := g.run(nil, true, true); code != 1 {
t.Fatalf("exit = %d, want 1", code)
}
_, run, _ := strings.Cut(g.out.String(), bnStopping)
// The stop was asked for, so the server stays stopped whatever the poll said.
want := " bravo: stopping\n bravo: failed: the apiserver is gone\n0 backed up, 1 failed.\n" +
"Left stopped: bravo. Start them from the panel when you are done.\n"
if run != want {
t.Fatalf("run =\n%s\nwant\n%s", run, want)
}
for _, tc := range []struct {
call int
events []string
}{
{1, nil}, // before the request: nothing is asked for
{2, []string{"request alpha"}}, // the first poll after it
} {
g := newBNRig(stoppedWorld("alpha"))
g.jobsFail["alpha"] = tc.call
if code := g.run(nil, true, false); code != 1 {
t.Fatalf("call %d: exit = %d, want 1", tc.call, code)
}
if !reflect.DeepEqual(g.events, tc.events) {
t.Fatalf("call %d: events = %v, want %v", tc.call, g.events, tc.events)
}
if !strings.Contains(g.out.String(), " alpha: failed: list its backup Jobs: the apiserver is gone\n") {
t.Fatalf("call %d: output =\n%s", tc.call, g.out.String())
}
}
}
-5
View File
@@ -3,7 +3,6 @@ package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"io"
"net/http"
@@ -153,10 +152,6 @@ func TestRequestBackup(t *testing.T) {
if err == nil || !strings.Contains(err.Error(), "unreachable") {
t.Fatalf("err = %v, want an 'unreachable' transport error", err)
}
// backup-now ends its run on this error alone.
if !errors.Is(err, errBackupAPIUnreachable) {
t.Fatalf("err = %v, want it to wrap errBackupAPIUnreachable", err)
}
})
}
+59 -73
View File
@@ -17,7 +17,6 @@ import (
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
tea "github.com/charmbracelet/bubbletea"
@@ -34,15 +33,13 @@ import (
//
// Root is necessary but NOT sufficient for accountability: root is machine
// authority, not a human identity, so the console additionally captures WHO is
// breaking the glass. When a staff account already exists the operator names one
// and types the one-time code the console mails to its verified address
// (breakglass_otp.go); that account is then the accountable actor. When no code can
// be sent or proven, the typed OVERRIDE proceeds as the OS user and the audit row
// says why. When no staff account exists yet it bootstraps the first Owner and
// attributes the act to the OS user. The audit row records which of these
// happened. This attribution is best-effort, not tamper-proof — whoever runs this
// is root and can edit Postgres directly — but it produces an honest trail for an
// honest operator, which is the point.
// breaking the glass. When a staff account already exists it asks the operator to
// authenticate as an existing admin (the verified identity is the accountable
// actor); when none exists yet it bootstraps the first Owner from the typed
// credential and attributes the act to the OS user. The audit row records the
// difference. This attribution is best-effort, not tamper-proof — whoever runs
// this is root and can edit Postgres directly — but it produces an honest trail
// for an honest operator, which is the point.
//
// When a staff account already exists the console opens on a thin top-level menu
// (menuModel) so that operations are peers, not tails of one wizard. Two account
@@ -65,7 +62,7 @@ import (
// suspension for the interactive `cloudflared tunnel login` browser consent.
// breakGlassOverrideToken is the literal an operator must type to proceed when no
// admin could be verified by a mailed code. Requiring an explicit, deliberate word (not a
// admin credential could be verified. Requiring an explicit, deliberate word (not a
// bare Enter) keeps the unverified root override from happening by reflex.
const breakGlassOverrideToken = "OVERRIDE"
@@ -145,7 +142,7 @@ func cmdBreakGlass(args []string, stdout, stderr io.Writer) int {
repo := api.NewPGRepo(drv.DB())
// Decide bootstrap (no admin yet → typed credential mints the first Owner) vs
// recovery (an admin exists → the operator proves one with a mailed code) BEFORE the
// recovery (an admin exists → the operator must authenticate as one) BEFORE the
// alt-screen TUI takes over, so a database fault surfaces as a plain error.
adminExists, err := repo.AdminExists(ctx)
if err != nil {
@@ -153,12 +150,7 @@ func cmdBreakGlass(args []string, stdout, stderr io.Writer) int {
return 1
}
// Recovery mails its code through [smtp]; the relay is opened only if a code is
// asked for.
host, _ := os.Hostname()
recovery := recoveryConfig{open: hostRecoveryMailer(cfg.SMTP, hostSMTPPasswordPath, platform.DefaultControlNamespace), host: host}
res, err := runBreakGlassTUI(ctx, repo, cfg.Database, cfg.Server.RootDomain, cfg.Auth.AdminHostname, cfg.Auth.PanelHostname, cfg.Auth.AccessJWTAud, cfg.K8s.Namespace, accountableOSUser(), adminExists, recovery)
res, err := runBreakGlassTUI(ctx, repo, cfg.Database.URL, cfg.Server.RootDomain, cfg.Auth.AdminHostname, cfg.Auth.PanelHostname, cfg.Auth.AccessJWTAud, cfg.K8s.Namespace, accountableOSUser(), adminExists)
if err != nil {
fmt.Fprintf(stderr, "felis breakGlass: %v\n", err)
return 1
@@ -182,9 +174,6 @@ func cmdBreakGlass(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stdout, "\nfelis breakGlass: Owner account %q provisioned; local session sign-in is ENABLED.\n", res.username)
}
fmt.Fprintf(stdout, "Recorded as %q (mode: %s, os user: %s).\n", res.accountable, res.mode, res.osUser)
if res.mode == "root_override" {
fmt.Fprintln(stdout, "No admin was proven by an email code; the audit row records this run as an unverified root override and why.")
}
if res.setupTokenURL != "" {
fmt.Fprintf(stdout, "One-time setup URL (opens a lockdown session to verify email / enroll passkey):\n\n %s\n\n", res.setupTokenURL)
}
@@ -269,6 +258,29 @@ func newOwnerID() string {
return "usr-" + hex.EncodeToString(b[:])
}
// authenticateAdmin resolves a typed admin username for recovery-mode attribution.
// Password verification is gone (passwordless design); Phase 3 replaces this with
// email-OTP recovery. For now it confirms the named admin exists.
func authenticateAdmin(ctx context.Context, s ownerStore, username string) (matched string, ok bool, err error) {
username = strings.TrimSpace(username)
if username == "" {
return "", false, nil
}
u, err := s.UserByUsername(ctx, username)
if errors.Is(err, api.ErrNotFound) {
return "", false, nil
}
if err != nil {
return "", false, err
}
// Staff means admin OR owner: recovery attribution must accept the Owner (the
// primary break-glass identity), not just plain admins.
if u.Role != "admin" && u.Role != "owner" {
return "", false, nil
}
return u.Username, true, nil
}
// provisionOwner mints or resets the single Owner account direct-to-Postgres,
// passwordless. The account is role=owner with no password — the Owner completes
// passwordless login setup via the web setup-token flow after `felis setup`.
@@ -359,10 +371,6 @@ type breakGlassOp struct {
ownerUsername string
ownerEmail string
attemptedAdmin string // recovery / override: the admin username the operator typed
verifiedBy string // recovery: how the admin was proven (verifiedByEmailOTP)
codeSentTo string // recovery: the address the proving code went to
otpSkipped string // root_override: why no code proved an admin (otpSkip*)
otpSkipDetail string // root_override: what failed, when something did
}
// breakGlassOutcome is what performBreakGlass reports back to the TUI.
@@ -486,7 +494,16 @@ func auditSetupMCBind(ctx context.Context, s ownerStore, osUser, mcUUID, authSou
// does not fail the recovery if this write fails — and intentionally honest: it
// records attribution, it does not prove it (a malicious root can edit the row).
func auditBreakGlass(ctx context.Context, s ownerStore, op breakGlassOp) error {
blob, err := json.Marshal(breakGlassPayload(op, "owner"))
payload := map[string]any{
"mode": op.mode,
"owner": op.ownerUsername,
"os_user": op.osUser,
"verified": op.mode == "recovery",
}
if op.attemptedAdmin != "" {
payload["admin_account"] = op.attemptedAdmin
}
blob, err := json.Marshal(payload)
if err != nil {
return err
}
@@ -498,33 +515,6 @@ func auditBreakGlass(ctx context.Context, s ownerStore, op breakGlassOp) error {
})
}
// breakGlassPayload is the who/how both account audits carry, with the account the
// run wrote under subjectKey. verified is true only for a run a mailed code proved;
// such a run names the address the code went to, and an override names why no code
// proved anyone.
func breakGlassPayload(op breakGlassOp, subjectKey string) map[string]any {
payload := map[string]any{
"mode": op.mode,
subjectKey: op.ownerUsername,
"os_user": op.osUser,
"verified": op.verifiedBy != "",
}
if op.attemptedAdmin != "" {
payload["admin_account"] = op.attemptedAdmin
}
if op.verifiedBy != "" {
payload["verified_by"] = op.verifiedBy
payload["code_sent_to"] = op.codeSentTo
}
if op.otpSkipped != "" {
payload["otp_skipped"] = op.otpSkipped
if op.otpSkipDetail != "" {
payload["otp_skip_detail"] = op.otpSkipDetail
}
}
return payload
}
// performAddOperator mints a NEW Operator account and records a best-effort
// accountability row. It mirrors performBreakGlass — passwordless — with two
// deliberate differences. (1) It provisions insert-only (provisionOperator), so it
@@ -548,7 +538,16 @@ func performAddOperator(ctx context.Context, s ownerStore, op breakGlassOp) (bre
// break_glass.operator_create action, naming the new account under an "operator" key
// rather than "owner".
func auditAddOperator(ctx context.Context, s ownerStore, op breakGlassOp) error {
blob, err := json.Marshal(breakGlassPayload(op, "operator"))
payload := map[string]any{
"mode": op.mode,
"operator": op.ownerUsername,
"os_user": op.osUser,
"verified": op.mode == "recovery",
}
if op.attemptedAdmin != "" {
payload["admin_account"] = op.attemptedAdmin
}
blob, err := json.Marshal(payload)
if err != nil {
return err
}
@@ -627,29 +626,16 @@ const (
cloudflareAPITokenDocsURL = "https://developers.cloudflare.com/fundamentals/api/how-to/account-owned-token-template/"
)
func runBreakGlassTUI(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, recovery recoveryConfig) (breakGlassResult, error) {
return runConsoleTUI(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeBreakGlass, recovery)
func runBreakGlassTUI(ctx context.Context, s ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool) (breakGlassResult, error) {
return runConsoleTUI(ctx, s, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeBreakGlass)
}
// runSetupTUI never reaches recovery: setup with a staff account present lands on
// the status screen, so it has no relay to hand over.
func runSetupTUI(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool) (breakGlassResult, error) {
return runConsoleTUI(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeSetup, recoveryConfig{})
func runSetupTUI(ctx context.Context, s ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool) (breakGlassResult, error) {
return runConsoleTUI(ctx, s, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeSetup)
}
// newConsoleRoot is the console's root model as the host runs it: the summary
// reads this host's alert route.
func newConsoleRoot(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode, recovery recoveryConfig) *rootModel {
rm := newRootModel(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, mode)
rm.recovery = recovery
rm.alertRoute = func(ctx context.Context) alertRoute {
return hostAlertRoute(ctx, hostSetupConfigPath, db.URL, defaultHeartbeatFile)
}
return rm
}
func runConsoleTUI(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode, recovery recoveryConfig) (breakGlassResult, error) {
rm := newConsoleRoot(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, mode, recovery)
func runConsoleTUI(ctx context.Context, s ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode) (breakGlassResult, error) {
rm := newRootModel(ctx, s, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, mode)
final, err := tea.NewProgram(rm, tea.WithAltScreen()).Run()
if err != nil {
return breakGlassResult{}, err
-259
View File
@@ -1,259 +0,0 @@
package main
import (
"context"
"crypto/rand"
"crypto/subtle"
"errors"
"fmt"
"math/big"
"os"
"strings"
"time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/platform"
)
// Recovery mode proves who is breaking the glass (#13). Naming a staff account is
// where it starts: the console then mails a one-time code to that account's verified
// address, and only that code, typed within recoveryCodeTTL, makes the run a
// recovery attributed to the account. Every other ending — no such account, no
// verified address, no relay, a send that fails, a wrong or late code, or the
// operator giving up on the mail — leads to the typed OVERRIDE, which the audit row
// records as an unverified root_override together with the reason (otp_skipped).
// The code goes through the same [smtp] relay as the panel's login codes, so with
// that relay down recovery still works, as an override that says why.
const (
recoveryCodeTTL = 10 * time.Minute
recoveryCodeAttempts = 5
)
// The reasons a run fell back to the override, recorded as otp_skipped.
const (
otpSkipUnknownAdmin = "unknown_admin"
otpSkipNoVerifiedEmail = "no_verified_email"
otpSkipNoRelay = "no_relay"
otpSkipSendFailed = "send_failed"
otpSkipCodeExpired = "code_expired"
otpSkipCodeRejected = "code_rejected"
otpSkipByOperator = "operator_skipped"
)
// verifiedByEmailOTP is the audit's verified_by for a recovery the mailed code proved.
const verifiedByEmailOTP = "email_otp"
// recoveryMailer is the one relay call a recovery code needs; *mail.SMTP has it.
type recoveryMailer interface {
SendNotice(ctx context.Context, email, subject, body string) error
}
// recoveryConfig is what the console needs to mail a recovery code. open resolves
// the relay only when a code is about to go out, so a console used to halt a server
// never touches [smtp] or the cluster; its error says why no relay is available.
// host names this machine in the mail. The zero value has no relay.
type recoveryConfig struct {
open func(ctx context.Context) (recoveryMailer, error)
host string
now func() time.Time
}
func (r recoveryConfig) clock() time.Time {
if r.now != nil {
return r.now()
}
return time.Now()
}
// recoveryCode is one mailed code: its value, when it stops working, and how many
// wrong codes were typed against it.
type recoveryCode struct {
value string
expires time.Time
failures int
}
func newRecoveryCode(now time.Time) (*recoveryCode, error) {
n, err := rand.Int(rand.Reader, big.NewInt(1_000_000))
if err != nil {
return nil, fmt.Errorf("generate recovery code: %w", err)
}
return &recoveryCode{value: fmt.Sprintf("%06d", n.Int64()), expires: now.Add(recoveryCodeTTL)}, nil
}
type codeVerdict int
const (
codeAccepted codeVerdict = iota
codeWrong
codeExpired
codeExhausted
)
// check compares a typed code in constant time. Each wrong code counts; the one
// that reaches recoveryCodeAttempts exhausts the code, which then accepts nothing,
// and neither does an expired one.
func (c *recoveryCode) check(typed string, now time.Time) codeVerdict {
if c.failures >= recoveryCodeAttempts {
return codeExhausted
}
if !now.Before(c.expires) {
return codeExpired
}
if subtle.ConstantTimeCompare([]byte(strings.TrimSpace(typed)), []byte(c.value)) == 1 {
return codeAccepted
}
c.failures++
if c.failures >= recoveryCodeAttempts {
return codeExhausted
}
return codeWrong
}
func (c *recoveryCode) attemptsLeft() int { return recoveryCodeAttempts - c.failures }
// recoveryStart is where naming an admin led: a code on its way to that admin, or
// the reason the run has to fall back to the override.
type recoveryStart struct {
admin *api.StaffUser // the named staff account; nil when none matched
code *recoveryCode // set when the code went out
skip string // otpSkip* when it did not
detail string // what failed, for the override screen and the audit row
}
// resolveAdmin loads the staff account (admin or owner) a typed username names, or
// nil when there is none.
func resolveAdmin(ctx context.Context, s ownerStore, username string) (*api.StaffUser, error) {
username = strings.TrimSpace(username)
if username == "" {
return nil, nil
}
u, err := s.UserByUsername(ctx, username)
if errors.Is(err, api.ErrNotFound) {
return nil, nil
}
if err != nil {
return nil, err
}
// Staff means admin or owner: the Owner is the primary break-glass identity.
if u.Role != "admin" && u.Role != "owner" {
return nil, nil
}
return u, nil
}
// beginRecovery resolves the named admin and mails it a recovery code. Only a
// datastore or entropy fault is an error; every other way the code cannot go out is
// a recoveryStart with skip set.
func beginRecovery(ctx context.Context, s ownerStore, rc recoveryConfig, username, osUser string, op bgOperation) (recoveryStart, error) {
admin, err := resolveAdmin(ctx, s, username)
if err != nil {
return recoveryStart{}, err
}
if admin == nil {
return recoveryStart{skip: otpSkipUnknownAdmin}, nil
}
st := recoveryStart{admin: admin}
// An address nobody ever proved vouches for nobody.
email := strings.TrimSpace(admin.Email)
if email == "" || !admin.EmailVerified {
st.skip = otpSkipNoVerifiedEmail
return st, nil
}
if rc.open == nil {
st.skip, st.detail = otpSkipNoRelay, "this console has no mail relay"
return st, nil
}
relay, err := rc.open(ctx)
if err != nil {
st.skip, st.detail = otpSkipNoRelay, err.Error()
return st, nil
}
code, err := newRecoveryCode(rc.clock())
if err != nil {
return recoveryStart{}, err
}
sendCtx, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
subject, body := recoveryMail(code.value, rc.host, osUser, admin.Username, op)
if err := relay.SendNotice(sendCtx, email, subject, body); err != nil {
st.skip, st.detail = otpSkipSendFailed, err.Error()
return st, nil
}
st.code = code
return st, nil
}
// recoveryMail words the code mail. It says where, by whom and for what the console
// was opened, so an admin who did not ask for it learns that root on that machine is
// in someone else's hands.
func recoveryMail(code, host, osUser, admin string, op bgOperation) (subject, body string) {
what, whatZH := "reset the Owner account", "重置 Owner 账号"
if op == bgAddOperator {
what, whatZH = "add an Operator account", "添加 Operator 账号"
}
if host == "" {
host = "the Felis host"
}
minutes := int(recoveryCodeTTL / time.Minute)
subject = "Felis break-glass recovery code / 紧急恢复验证码"
body = fmt.Sprintf(`Someone with root on %[1]s (OS user %[2]s) opened felis breakGlass and named your staff account %[3]q to %[4]s.
Recovery code: %[6]s
It works for %[7]d minutes.
If this was not you, root on that machine is in someone else's hands: change its credentials and read the audit log for break_glass entries.
有人在 %[1]s 上以 root 身份(系统用户 %[2]s)打开了 felis breakGlass,指名你的管理员账号 %[3]q 来%[5]s。
恢复验证码:%[6]s
%[7]d 分钟内有效。
如果不是你本人,这台机器的 root 已落入他人之手:请更换它的凭据,并查看审计日志中的 break_glass 记录。
`, host, osUser, admin, what, whatZH, code, minutes)
return subject, body
}
// maskEmail keeps the first character of the local part and the domain, enough for
// the operator to recognise the address without putting it on screen whole.
func maskEmail(email string) string {
at := strings.LastIndex(email, "@")
if at <= 0 {
return "***"
}
return email[:1] + strings.Repeat("*", max(at-1, 3)) + email[at:]
}
// hostRecoveryMailer opens the [smtp] relay from the host the way the watchdog does:
// the password is the env var password_ref names when that is set, else the host
// copy at passwordPath, else the felis-smtp Secret, whose absence means a relay
// without AUTH. The cluster is reached only when the host copy is missing, so a
// break-glass on a host whose k3s is down still gets its code.
func hostRecoveryMailer(c config.SMTPConfig, passwordPath, controlNS string) func(context.Context) (recoveryMailer, error) {
return func(ctx context.Context) (recoveryMailer, error) {
if strings.TrimSpace(c.Host) == "" {
return nil, errors.New("[smtp] is not configured in felis.toml")
}
if ref := c.PasswordRef; ref != "" && os.Getenv(ref) != "" {
return smtpRelay(c, os.Getenv(ref)), nil
}
if password, ok, err := readHostCredential(passwordPath); err != nil {
return nil, fmt.Errorf("read the relay password: %w", err)
} else if ok {
return smtpRelay(c, password), nil
}
cl, err := buildSystemServerClient()
if err != nil {
return nil, fmt.Errorf("reach the cluster for the relay password: %w", err)
}
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
defer cancel()
password, err := smtpSecretPassword(ctx, cl, controlNS)
if err != nil {
return nil, fmt.Errorf("read the relay password from %s/%s: %w", controlNS, platform.SMTPSecretName, err)
}
return smtpRelay(c, password), nil
}
}
-498
View File
@@ -1,498 +0,0 @@
package main
import (
"context"
"errors"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/api"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/mail"
"felis.lolicon.best/internal/platform"
tea "github.com/charmbracelet/bubbletea"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime/schema"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
)
// These tests cover the recovery proof (#13): the mailed code's rules, where
// naming an admin leads, what the mail says, how the console walks from a name to
// a proven (or overridden) run, and what the audit row then records. No mail
// leaves the process: the relay is a fake that keeps what it was handed.
type sentMail struct{ to, subject, body string }
type fakeRelay struct {
sent []sentMail
err error
}
func (r *fakeRelay) SendNotice(_ context.Context, to, subject, body string) error {
if r.err != nil {
return r.err
}
r.sent = append(r.sent, sentMail{to, subject, body})
return nil
}
// sentCode pulls the code out of the one mail the relay carried.
func (r *fakeRelay) sentCode(t *testing.T) string {
t.Helper()
if len(r.sent) != 1 {
t.Fatalf("relay carried %d mails, want 1", len(r.sent))
}
_, after, ok := strings.Cut(r.sent[0].body, "Recovery code: ")
if !ok || len(after) < 6 {
t.Fatalf("mail carries no recovery code:\n%s", r.sent[0].body)
}
return after[:6]
}
func relayConfig(r *fakeRelay, now func() time.Time) recoveryConfig {
return recoveryConfig{
open: func(context.Context) (recoveryMailer, error) { return r, nil },
host: "felis-host-1",
now: now,
}
}
func verifiedAdmin(username, email string) *api.StaffUser {
return &api.StaffUser{ID: "usr-" + username, Username: username, Role: "owner", Email: email, EmailVerified: true}
}
func TestRecoveryCodeRules(t *testing.T) {
t0 := time.Date(2026, 9, 26, 8, 0, 0, 0, time.UTC)
t.Run("a fresh code is six digits and lives for the TTL", func(t *testing.T) {
seen := map[string]bool{}
for i := 0; i < 20; i++ {
c, err := newRecoveryCode(t0)
if err != nil {
t.Fatal(err)
}
if !isRecoveryCodeShape(c.value) {
t.Fatalf("code %q is not six digits", c.value)
}
if !c.expires.Equal(t0.Add(recoveryCodeTTL)) {
t.Fatalf("expires = %v, want %v", c.expires, t0.Add(recoveryCodeTTL))
}
seen[c.value] = true
}
if len(seen) < 2 {
t.Error("twenty codes were all the same")
}
})
t.Run("the right code is accepted, surrounding space ignored", func(t *testing.T) {
c := &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
if v := c.check(" 042917 ", t0); v != codeAccepted {
t.Errorf("check = %v, want accepted", v)
}
})
t.Run("wrong codes count down and the last one exhausts it", func(t *testing.T) {
c := &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
for i := 1; i < recoveryCodeAttempts; i++ {
if v := c.check("000000", t0); v != codeWrong {
t.Fatalf("wrong code %d: check = %v, want wrong", i, v)
}
if left := c.attemptsLeft(); left != recoveryCodeAttempts-i {
t.Fatalf("after %d wrong codes attemptsLeft = %d, want %d", i, left, recoveryCodeAttempts-i)
}
}
if v := c.check("000000", t0); v != codeExhausted {
t.Fatalf("wrong code %d: check = %v, want exhausted", recoveryCodeAttempts, v)
}
if v := c.check("042917", t0); v != codeExhausted {
t.Errorf("the right code after exhaustion: check = %v, want exhausted", v)
}
})
t.Run("an expired code accepts nothing", func(t *testing.T) {
c := &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
if v := c.check("042917", t0.Add(recoveryCodeTTL-time.Second)); v != codeAccepted {
t.Fatalf("a second before expiry: check = %v, want accepted", v)
}
c = &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
if v := c.check("042917", t0.Add(recoveryCodeTTL)); v != codeExpired {
t.Errorf("at expiry: check = %v, want expired", v)
}
})
}
func TestBeginRecovery(t *testing.T) {
ctx := context.Background()
t0 := time.Date(2026, 9, 26, 8, 0, 0, 0, time.UTC)
clock := func() time.Time { return t0 }
t.Run("mails a code to the named admin's verified address", func(t *testing.T) {
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": verifiedAdmin("root", "[email protected]")}}
r := &fakeRelay{}
st, err := beginRecovery(ctx, f, relayConfig(r, clock), "root", "alice", bgAddOperator)
if err != nil {
t.Fatal(err)
}
if st.code == nil || st.skip != "" {
t.Fatalf("start = %+v, want a code and no skip", st)
}
if st.admin == nil || st.admin.Username != "root" {
t.Fatalf("start.admin = %+v, want root", st.admin)
}
if got := r.sentCode(t); got != st.code.value {
t.Errorf("mailed code %q, want the code the console checks (%q)", got, st.code.value)
}
m := r.sent[0]
if m.to != "[email protected]" {
t.Errorf("mail went to %q, want [email protected]", m.to)
}
for _, want := range []string{"felis-host-1", "OS user alice", `"root"`, "add an Operator account", "添加 Operator 账号", "10 minutes"} {
if !strings.Contains(m.body, want) {
t.Errorf("mail body lacks %q:\n%s", want, m.body)
}
}
if !st.code.expires.Equal(t0.Add(recoveryCodeTTL)) {
t.Errorf("code expires %v, want %v", st.code.expires, t0.Add(recoveryCodeTTL))
}
})
t.Run("the Owner reset is named as such", func(t *testing.T) {
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": verifiedAdmin("root", "[email protected]")}}
r := &fakeRelay{}
if _, err := beginRecovery(ctx, f, relayConfig(r, clock), "root", "alice", bgProvisionOwner); err != nil {
t.Fatal(err)
}
if body := r.sent[0].body; !strings.Contains(body, "reset the Owner account") || !strings.Contains(body, "重置 Owner 账号") {
t.Errorf("mail body does not name the Owner reset:\n%s", body)
}
})
// Each way the code cannot go out ends in a skip reason and no mail.
unverified := verifiedAdmin("root", "[email protected]")
unverified.EmailVerified = false
noEmail := verifiedAdmin("root", "")
cases := []struct {
name string
user *api.StaffUser
rc func(r *fakeRelay) recoveryConfig
skip string
detail string
wantAdmin bool
relayError error
}{
{name: "unknown admin", user: nil, rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) }, skip: otpSkipUnknownAdmin},
{name: "unverified address", user: unverified, rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) }, skip: otpSkipNoVerifiedEmail, wantAdmin: true},
{name: "no address", user: noEmail, rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) }, skip: otpSkipNoVerifiedEmail, wantAdmin: true},
{name: "no relay wired", user: verifiedAdmin("root", "[email protected]"), rc: func(*fakeRelay) recoveryConfig { return recoveryConfig{} }, skip: otpSkipNoRelay, detail: "this console has no mail relay", wantAdmin: true},
{name: "relay cannot open", user: verifiedAdmin("root", "[email protected]"), rc: func(*fakeRelay) recoveryConfig {
return recoveryConfig{open: func(context.Context) (recoveryMailer, error) {
return nil, errors.New("[smtp] is not configured in felis.toml")
}}
}, skip: otpSkipNoRelay, detail: "[smtp] is not configured in felis.toml", wantAdmin: true},
{name: "send fails", user: verifiedAdmin("root", "[email protected]"), rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) },
skip: otpSkipSendFailed, detail: "554 relay refused", wantAdmin: true, relayError: errors.New("554 relay refused")},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
f := &fakeOwnerStore{users: map[string]*api.StaffUser{}}
if tc.user != nil {
f.users["root"] = tc.user
}
r := &fakeRelay{err: tc.relayError}
st, err := beginRecovery(ctx, f, tc.rc(r), "root", "alice", bgProvisionOwner)
if err != nil {
t.Fatal(err)
}
if st.code != nil || st.skip != tc.skip || st.detail != tc.detail {
t.Errorf("start = {code:%v skip:%q detail:%q}, want no code, skip %q, detail %q", st.code, st.skip, st.detail, tc.skip, tc.detail)
}
if (st.admin != nil) != tc.wantAdmin {
t.Errorf("start.admin = %+v, want present=%v", st.admin, tc.wantAdmin)
}
if len(r.sent) != 0 {
t.Errorf("relay carried %d mails, want none", len(r.sent))
}
})
}
t.Run("a datastore fault is an error", func(t *testing.T) {
f := &fakeOwnerStore{userErr: errors.New("db down")}
if _, err := beginRecovery(ctx, f, relayConfig(&fakeRelay{}, clock), "root", "alice", bgProvisionOwner); err == nil {
t.Fatal("want the store fault")
}
})
}
func TestMaskEmail(t *testing.T) {
for in, want := range map[string]string{
"[email protected]": "a****@example.com",
"[email protected]": "a***@example.com",
"@example.com": "***",
"nonsense": "***",
} {
if got := maskEmail(in); got != want {
t.Errorf("maskEmail(%q) = %q, want %q", in, got, want)
}
}
}
func TestHostRecoveryMailer(t *testing.T) {
ctx := context.Background()
t.Run("no [smtp] host is no relay", func(t *testing.T) {
_, err := hostRecoveryMailer(config.SMTPConfig{}, hostSMTPPasswordPath, "felis")(ctx)
if err == nil || !strings.Contains(err.Error(), "[smtp]") {
t.Fatalf("err = %v, want it to name [smtp]", err)
}
})
t.Run("the password_ref env var supplies the password", func(t *testing.T) {
t.Setenv("FELIS_TEST_RELAY_PW", "from-env")
off := false
c := config.SMTPConfig{Host: "mail.example.com", Port: 2525, From: "[email protected]", Username: "felis", PasswordRef: "FELIS_TEST_RELAY_PW", RequireTLS: &off}
got, err := hostRecoveryMailer(c, hostSMTPPasswordPath, "felis")(ctx)
if err != nil {
t.Fatal(err)
}
relay, ok := got.(*mail.SMTP)
if !ok {
t.Fatalf("relay is %T, want *mail.SMTP", got)
}
if relay.Password != "from-env" || relay.Host != "mail.example.com" || relay.Port != 2525 || relay.From != "[email protected]" || relay.Username != "felis" || relay.RequireTLS {
t.Errorf("relay = %+v, want the [smtp] fields with the env password and TLS as configured", *relay)
}
})
}
func TestSMTPSecretPassword(t *testing.T) {
ctx := context.Background()
scheme := haltScheme(t)
t.Run("reads the password key", func(t *testing.T) {
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(&corev1.Secret{
ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.SMTPSecretName},
Data: map[string][]byte{platform.SMTPSecretPasswordKey: []byte("s3cret")},
}).Build()
if pw, err := smtpSecretPassword(ctx, cl, "felis"); err != nil || pw != "s3cret" {
t.Errorf("smtpSecretPassword = (%q, %v), want (s3cret, nil)", pw, err)
}
})
t.Run("a missing Secret is a relay without AUTH", func(t *testing.T) {
cl := fake.NewClientBuilder().WithScheme(scheme).Build()
if pw, err := smtpSecretPassword(ctx, cl, "felis"); err != nil || pw != "" {
t.Errorf("smtpSecretPassword = (%q, %v), want (\"\", nil)", pw, err)
}
})
t.Run("any other read failure is an error", func(t *testing.T) {
cl := fake.NewClientBuilder().WithScheme(scheme).WithInterceptorFuncs(interceptor.Funcs{
Get: func(context.Context, client.WithWatch, client.ObjectKey, client.Object, ...client.GetOption) error {
return apierrors.NewForbidden(schema.GroupResource{Resource: "secrets"}, platform.SMTPSecretName, errors.New("rbac"))
},
}).Build()
if _, err := smtpSecretPassword(ctx, cl, "felis"); err == nil {
t.Fatal("want the read failure")
}
})
}
// recoveryModel is an Owner-reset console for admin "root" (verified address
// [email protected]), run by OS user alice, with the fake relay behind it.
func recoveryModel(t *testing.T, f *fakeOwnerStore, r *fakeRelay, now func() time.Time) *ownerModel {
t.Helper()
if f.users == nil {
f.users = map[string]*api.StaffUser{"root": verifiedAdmin("root", "[email protected]")}
}
return newOwnerModel(context.Background(), f, "alice", true).withRecovery(relayConfig(r, now))
}
// nameAdmin submits the admin-name form and feeds the result of the send back in.
func nameAdmin(t *testing.T, m *ownerModel, name string) *ownerModel {
t.Helper()
m.authUser = name
_, cmd := m.onFormComplete()
if m.step != owWorking {
t.Fatalf("after naming the admin step = %v, want owWorking", m.step)
}
msg := findMsg[owAuthMsg](t, cmd)
next, _ := m.Update(msg)
return next.(*ownerModel)
}
// findMsg runs a (possibly batched) command and returns the first T it produces.
func findMsg[T any](t *testing.T, cmd tea.Cmd) T {
t.Helper()
var zero T
if cmd == nil {
t.Fatalf("no command, want one producing %T", zero)
}
switch msg := cmd().(type) {
case T:
return msg
case tea.BatchMsg:
for _, c := range msg {
if c == nil {
continue
}
if got, ok := c().(T); ok {
return got
}
}
}
t.Fatalf("command produced no %T", zero)
return zero
}
// typeCode submits the code form with typed.
func typeCode(t *testing.T, m *ownerModel, typed string) {
t.Helper()
if m.step != owCode {
t.Fatalf("step = %v, want owCode", m.step)
}
m.codeInput = typed
m.onFormComplete()
}
// provisionAudit finishes the run as an Owner reset and returns its audit payload.
func provisionAudit(t *testing.T, m *ownerModel, f *fakeOwnerStore) (api.AuditEntry, map[string]any) {
t.Helper()
if m.step != owProvision {
t.Fatalf("step = %v, want owProvision", m.step)
}
m.username = "owner"
msg := m.provisionCmd()().(owProvisionMsg)
if msg.err != nil {
t.Fatalf("provision: %v", msg.err)
}
return auditOf(t, f)
}
func TestRecoveryConsoleFlow(t *testing.T) {
t0 := time.Date(2026, 9, 26, 8, 0, 0, 0, time.UTC)
fixed := func() time.Time { return t0 }
t.Run("the mailed code proves the admin and the audit says so", func(t *testing.T) {
f, r := &fakeOwnerStore{}, &fakeRelay{}
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
typeCode(t, m, r.sentCode(t))
if m.mode != "recovery" || m.accountable != "root" {
t.Fatalf("mode=%q accountable=%q, want recovery attributed to root", m.mode, m.accountable)
}
e, payload := provisionAudit(t, m, f)
if e.Actor != "root" || e.Action != "break_glass.recovery" {
t.Errorf("audit = %+v, want actor=root action=break_glass.recovery", e)
}
if payload["verified"] != true || payload["verified_by"] != verifiedByEmailOTP || payload["code_sent_to"] != "[email protected]" || payload["os_user"] != "alice" {
t.Errorf("payload = %v, want verified by email_otp to [email protected], os_user alice", payload)
}
})
t.Run("a wrong code asks again, and the last wrong one leads to the override", func(t *testing.T) {
f, r := &fakeOwnerStore{}, &fakeRelay{}
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
wrong := "000000"
if r.sentCode(t) == wrong {
wrong = "111111"
}
for i := 1; i < recoveryCodeAttempts; i++ {
typeCode(t, m, wrong)
if m.step != owCode || !strings.Contains(m.codeNote, "wrong") {
t.Fatalf("wrong code %d: step=%v note=%q, want the code form again with a note", i, m.step, m.codeNote)
}
}
typeCode(t, m, wrong)
if m.step != owOverride || m.skip != otpSkipCodeRejected {
t.Fatalf("after %d wrong codes step=%v skip=%q, want the override for code_rejected", recoveryCodeAttempts, m.step, m.skip)
}
m.onFormComplete() // OVERRIDE typed
e, payload := provisionAudit(t, m, f)
if e.Actor != "alice" || e.Action != "break_glass.root_override" {
t.Errorf("audit = %+v, want actor=alice action=break_glass.root_override", e)
}
if payload["verified"] != false || payload["otp_skipped"] != otpSkipCodeRejected || payload["admin_account"] != "root" {
t.Errorf("payload = %v, want unverified, otp_skipped=code_rejected, admin_account=root", payload)
}
})
t.Run("a late code leads to the override", func(t *testing.T) {
now := t0
f, r := &fakeOwnerStore{}, &fakeRelay{}
m := nameAdmin(t, recoveryModel(t, f, r, func() time.Time { return now }), "root")
now = t0.Add(recoveryCodeTTL)
typeCode(t, m, r.sentCode(t))
if m.step != owOverride || m.skip != otpSkipCodeExpired {
t.Fatalf("step=%v skip=%q, want the override for code_expired", m.step, m.skip)
}
})
t.Run("OVERRIDE at the code prompt goes on unverified, saying the operator skipped", func(t *testing.T) {
f, r := &fakeOwnerStore{}, &fakeRelay{}
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
typeCode(t, m, breakGlassOverrideToken)
if m.mode != "root_override" || m.accountable != "alice" {
t.Fatalf("mode=%q accountable=%q, want root_override as alice", m.mode, m.accountable)
}
_, payload := provisionAudit(t, m, f)
if payload["verified"] != false || payload["otp_skipped"] != otpSkipByOperator {
t.Errorf("payload = %v, want unverified, otp_skipped=operator_skipped", payload)
}
})
t.Run("a relay failure leads to the override naming it", func(t *testing.T) {
f, r := &fakeOwnerStore{}, &fakeRelay{err: errors.New("dial tcp 10.0.0.9:587: connect: connection refused")}
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
if m.step != owOverride || m.skip != otpSkipSendFailed {
t.Fatalf("step=%v skip=%q, want the override for send_failed", m.step, m.skip)
}
if reason := m.overrideReason(); !strings.Contains(reason, "connection refused") {
t.Errorf("override reason %q does not name the failure", reason)
}
m.onFormComplete()
_, payload := provisionAudit(t, m, f)
if payload["otp_skipped"] != otpSkipSendFailed || !strings.Contains(payload["otp_skip_detail"].(string), "connection refused") {
t.Errorf("payload = %v, want otp_skipped=send_failed with the relay's error", payload)
}
})
t.Run("esc at the code prompt starts over and forgets the code", func(t *testing.T) {
f, r := &fakeOwnerStore{}, &fakeRelay{}
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
code := r.sentCode(t)
next, _ := m.Update(key(tea.KeyEsc))
m = next.(*ownerModel)
if m.step != owAuth || m.code != nil || m.admin != nil {
t.Fatalf("after esc step=%v code=%v admin=%v, want owAuth with the attempt forgotten", m.step, m.code, m.admin)
}
// The old code cannot be replayed: the next name mails a new one.
m = nameAdmin(t, m, "root")
if len(r.sent) != 2 {
t.Fatalf("relay carried %d mails, want a second one for the new attempt", len(r.sent))
}
_, fresh, _ := strings.Cut(r.sent[1].body, "Recovery code: ")
if m.code.value != fresh[:6] {
t.Errorf("the console checks %q, want the newly mailed %q (old one was %q)", m.code.value, fresh[:6], code)
}
})
}
func TestRootHandsRecoveryToAccountOperations(t *testing.T) {
for _, op := range []bgOperation{bgProvisionOwner, bgAddOperator} {
m := newTestRoot(true, consoleModeBreakGlass, "")
m.recovery = recoveryConfig{host: "felis-host-1"}
m = drive(t, m, menuChoiceMsg{op: op})
om, ok := m.screen.(*ownerModel)
if !ok {
t.Fatalf("op %v: screen = %T, want *ownerModel", op, m.screen)
}
if om.recovery.host != "felis-host-1" {
t.Errorf("op %v: the account screen has no relay config; its codes could never go out", op)
}
}
}
+40 -56
View File
@@ -269,57 +269,71 @@ func TestEnableLocalAuth(t *testing.T) {
}
}
func TestResolveAdmin(t *testing.T) {
func TestAuthenticateAdmin(t *testing.T) {
ctx := context.Background()
// resolveAdmin only finds the staff account a typed name points at; proving the
// operator holds it is the mailed code's job (beginRecovery).
// Password verification is gone (passwordless design): authenticateAdmin now only
// resolves the named admin so recovery can attribute the audit to a real identity.
// The security boundary is the break-glass root gate, not a typed secret.
t.Run("resolves an existing admin", func(t *testing.T) {
t.Run("resolves an existing admin for attribution", func(t *testing.T) {
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": mkAdmin("root")}}
u, err := resolveAdmin(ctx, f, " root ")
matched, ok, err := authenticateAdmin(ctx, f, "root")
if err != nil {
t.Fatalf("resolveAdmin: %v", err)
t.Fatalf("authenticateAdmin: %v", err)
}
if u == nil || u.Username != "root" {
t.Fatalf("resolveAdmin = %+v, want the root admin", u)
if !ok {
t.Fatal("ok = false, want true for an existing admin")
}
if matched != "root" {
t.Errorf("matched = %q, want root", matched)
}
})
t.Run("a non-admin role is no staff account", func(t *testing.T) {
t.Run("a non-admin role can never attribute a break-glass", func(t *testing.T) {
player := mkAdmin("alice")
player.Role = "user" // a player row is not staff
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"alice": player}}
if u, err := resolveAdmin(ctx, f, "alice"); u != nil || err != nil {
t.Errorf("resolveAdmin(player) = (%+v, %v), want (nil, nil)", u, err)
_, ok, err := authenticateAdmin(ctx, f, "alice")
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if ok {
t.Error("ok = true, want false for a non-admin role")
}
})
t.Run("the owner role counts as staff", func(t *testing.T) {
t.Run("the owner role attributes like an admin", func(t *testing.T) {
owner := mkAdmin("root")
owner.Role = "owner" // the platform owner is staff too (migration 0011)
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": owner}}
if u, err := resolveAdmin(ctx, f, "root"); err != nil || u == nil {
t.Fatalf("resolveAdmin(owner) = (%+v, %v), want the owner", u, err)
matched, ok, err := authenticateAdmin(ctx, f, "root")
if err != nil || !ok || matched != "root" {
t.Fatalf("authenticateAdmin(owner) = (%q, %v, %v), want (root, true, nil)", matched, ok, err)
}
})
t.Run("an unknown user is nil, not an error", func(t *testing.T) {
if u, err := resolveAdmin(ctx, &fakeOwnerStore{}, "nobody"); u != nil || err != nil {
t.Errorf("resolveAdmin(unknown) = (%+v, %v), want (nil, nil)", u, err)
t.Run("an unknown user is a non-match, not an error", func(t *testing.T) {
f := &fakeOwnerStore{}
_, ok, err := authenticateAdmin(ctx, f, "nobody")
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if ok {
t.Error("ok = true, want false for an unknown user")
}
})
t.Run("an empty username makes no store call", func(t *testing.T) {
t.Run("an empty username is a non-match with no store call", func(t *testing.T) {
f := &fakeOwnerStore{userErr: errors.New("must not be called")}
if u, err := resolveAdmin(ctx, f, " "); u != nil || err != nil {
t.Errorf("empty username: (%+v, %v), want (nil, nil)", u, err)
if _, ok, err := authenticateAdmin(ctx, f, ""); ok || err != nil {
t.Errorf("empty username: ok=%v err=%v, want false,nil", ok, err)
}
})
t.Run("a datastore fault is surfaced", func(t *testing.T) {
f := &fakeOwnerStore{userErr: errors.New("db down")}
if _, err := resolveAdmin(ctx, f, "root"); err == nil {
if _, _, err := authenticateAdmin(ctx, f, "root"); err == nil {
t.Fatal("want error when the store fails")
}
})
@@ -387,8 +401,6 @@ func TestPerformBreakGlass(t *testing.T) {
osUser: "alice",
ownerUsername: "owner",
attemptedAdmin: "root",
verifiedBy: verifiedByEmailOTP,
codeSentTo: "[email protected]",
}
out, err := performBreakGlass(ctx, f, op)
if err != nil {
@@ -413,23 +425,6 @@ func TestPerformBreakGlass(t *testing.T) {
if payload["admin_account"] != "root" {
t.Errorf("payload.admin_account = %v, want root", payload["admin_account"])
}
if payload["verified_by"] != verifiedByEmailOTP || payload["code_sent_to"] != "[email protected]" {
t.Errorf("payload = %v, want verified_by=email_otp [email protected]", payload)
}
if _, present := payload["otp_skipped"]; present {
t.Error("a proven recovery carries no otp_skipped")
}
})
t.Run("recovery without a proof is recorded unverified", func(t *testing.T) {
f := &fakeOwnerStore{}
op := breakGlassOp{mode: "recovery", accountable: "root", osUser: "alice", ownerUsername: "owner", attemptedAdmin: "root"}
if _, err := performBreakGlass(ctx, f, op); err != nil {
t.Fatalf("performBreakGlass: %v", err)
}
if _, payload := auditOf(t, f); payload["verified"] != false {
t.Errorf("payload.verified = %v, want false: only a mailed code verifies", payload["verified"])
}
})
t.Run("root override records an unverified row attributed to the OS user", func(t *testing.T) {
@@ -440,8 +435,6 @@ func TestPerformBreakGlass(t *testing.T) {
osUser: "alice",
ownerUsername: "owner",
attemptedAdmin: "typo-admin",
otpSkipped: otpSkipSendFailed,
otpSkipDetail: "dial tcp: connection refused",
}
if _, err := performBreakGlass(ctx, f, op); err != nil {
t.Fatalf("performBreakGlass: %v", err)
@@ -457,13 +450,6 @@ func TestPerformBreakGlass(t *testing.T) {
if payload["admin_account"] != "typo-admin" {
t.Errorf("payload.admin_account = %v, want typo-admin", payload["admin_account"])
}
// Why no code proved anyone is part of the record.
if payload["otp_skipped"] != otpSkipSendFailed || payload["otp_skip_detail"] != "dial tcp: connection refused" {
t.Errorf("payload = %v, want otp_skipped=send_failed with its detail", payload)
}
if _, present := payload["verified_by"]; present {
t.Error("an override carries no verified_by")
}
})
t.Run("an audit failure does not fail the recovery", func(t *testing.T) {
@@ -630,8 +616,6 @@ func TestPerformAddOperator(t *testing.T) {
ownerUsername: "ops-jordan",
ownerEmail: "[email protected]",
attemptedAdmin: "root",
verifiedBy: verifiedByEmailOTP,
codeSentTo: "[email protected]",
}
out, err := performAddOperator(ctx, f, op)
if err != nil {
@@ -658,14 +642,14 @@ func TestPerformAddOperator(t *testing.T) {
if _, present := payload["owner"]; present {
t.Error("payload.owner present, want the new account under the operator key")
}
if payload["verified"] != true || payload["admin_account"] != "root" || payload["verified_by"] != verifiedByEmailOTP {
t.Errorf("payload = %v, want verified=true admin_account=root verified_by=email_otp", payload)
if payload["verified"] != true || payload["admin_account"] != "root" {
t.Errorf("payload = %v, want verified=true admin_account=root", payload)
}
})
t.Run("root override records an unverified operator row", func(t *testing.T) {
f := &fakeOwnerStore{}
op := breakGlassOp{mode: "root_override", accountable: "alice", osUser: "alice", ownerUsername: "ops", attemptedAdmin: "typo-admin", otpSkipped: otpSkipUnknownAdmin}
op := breakGlassOp{mode: "root_override", accountable: "alice", osUser: "alice", ownerUsername: "ops", attemptedAdmin: "typo-admin"}
if _, err := performAddOperator(ctx, f, op); err != nil {
t.Fatalf("performAddOperator: %v", err)
}
@@ -673,8 +657,8 @@ func TestPerformAddOperator(t *testing.T) {
t.Fatalf("want 1 insert, got %d", len(f.inserts))
}
_, payload := auditOf(t, f)
if payload["verified"] != false || payload["otp_skipped"] != otpSkipUnknownAdmin {
t.Errorf("payload = %v, want verified=false otp_skipped=unknown_admin", payload)
if payload["verified"] != false {
t.Errorf("payload.verified = %v, want false for root_override", payload["verified"])
}
})
+1 -53
View File
@@ -11,14 +11,12 @@ import (
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/platform"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// cmdConverge is the explicit convergence pass over already-installed system
// servers (#1), plus the idle-stop default for user servers that predate it, and
// with -user-rcon their RCON block (#3). Provisioning is create-if-absent, so a field the desired spec
// servers (#1), plus the idle-stop default for user servers that predate it. Provisioning is create-if-absent, so a field the desired spec
// gained after an install (spec.rcon, spec.startup.healthHTTPPort, a derived env
// key) never reaches the existing CR — and nothing says so. This command fills
// exactly those zero-value fields; see convergeSystemServers for the full contract
@@ -31,7 +29,6 @@ func cmdConverge(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("converge", flag.ContinueOnError)
fs.SetOutput(stderr)
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
userRcon := fs.Bool("user-rcon", false, "also turn RCON on for user servers created before it was the default")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
@@ -62,7 +59,6 @@ func cmdConverge(args []string, stdout, stderr io.Writer) int {
defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
outcomes = append(outcomes, convergeUserServerIdle(context.Background(), cl, cfg.K8s.Namespace)...)
outcomes = append(outcomes, convergeUserServerRcon(context.Background(), cl, cfg.K8s.Namespace, *userRcon)...)
fmt.Fprintln(stdout, "felis converge: filling fields an installed server predates (operator-set values are never overwritten):")
exit := 0
@@ -117,51 +113,3 @@ func convergeUserServerIdle(ctx context.Context, cl client.Client, namespace str
}
return out
}
// convergeUserServerRcon handles user servers created before RCON was part of every
// new server (694e3cb): spec.rcon entirely unset. Such a server has a dead console,
// reports nobody online, and never idles out, because all three ride RCON.
//
// Only with fill does it turn RCON on, with the same block CreateServer writes
// today; without it each such server gets a line saying so. The fill is opt-in
// because the operator gates readiness on the RCON probe: a server whose image does
// not open the listener RCON_PASSWORD asks for would sit in Starting until it is
// marked Failed. Felis's own paper and lobby images open it; an image a user brought
// may not, and only the operator running this can tell. A server that already
// carries any RCON setting (on or off) is left alone and produces no line.
func convergeUserServerRcon(ctx context.Context, cl client.Client, namespace string, fill bool) []systemServerOutcome {
var list v1alpha1.MinecraftServerList
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
return []systemServerOutcome{{name: "user servers", err: fmt.Errorf("list servers: %w", err)}}
}
var out []systemServerOutcome
for i := range list.Items {
ms := &list.Items[i]
if ms.Labels[v1alpha1.LabelSystemRole] != "" || ms.Spec.Rcon != (v1alpha1.RconSpec{}) {
continue
}
if !fill {
out = append(out, systemServerOutcome{name: ms.Name, available: true,
skipped: "no RCON (console, online count and idle stop are off); once its image serves RCON, sudo felis converge -user-rcon turns it on"})
continue
}
changed, err := patchOnConflictRetry(ctx, cl, ms, func() bool {
if ms.Spec.Rcon != (v1alpha1.RconSpec{}) {
return false
}
ms.Spec.Rcon = v1alpha1.RconSpec{
Enabled: true,
SecretRef: v1alpha1.SecretKeyRef{Name: naming.RconSecretName(ms.Name), Key: naming.RconSecretKey},
}
return true
})
if err != nil {
out = append(out, systemServerOutcome{name: ms.Name, err: fmt.Errorf("converge %s: %w", ms.Name, err)})
continue
}
if changed {
out = append(out, systemServerOutcome{name: ms.Name, available: true, updated: true, changes: []string{"spec.rcon"}})
}
}
return out
}
-64
View File
@@ -229,67 +229,3 @@ func TestConvergeUserServerIdle(t *testing.T) {
t.Fatalf("second pass = %+v, want nothing to do", again)
}
}
// TestConvergeUserServerRcon reports a user server with no RCON block at all and
// fills it only when asked, with the block CreateServer writes. RCON turned off on
// purpose, a server with its own secret, and a system server stay as they are and
// produce no line.
func TestConvergeUserServerRcon(t *testing.T) {
scheme := newSystemServerScheme(t)
ctx := context.Background()
mk := func(name string, rcon v1alpha1.RconSpec, role string) *v1alpha1.MinecraftServer {
ms := &v1alpha1.MinecraftServer{}
ms.Name, ms.Namespace = name, "minecraft"
ms.Spec.Rcon = rcon
if role != "" {
ms.Labels = map[string]string{v1alpha1.LabelSystemRole: role}
}
return ms
}
own := v1alpha1.RconSpec{Enabled: true, Port: 25580, SecretRef: v1alpha1.SecretKeyRef{Name: "own", Key: "pw"}}
off := v1alpha1.RconSpec{SecretRef: v1alpha1.SecretKeyRef{Name: "rcon-off", Key: naming.RconSecretKey}}
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
mk("demo", v1alpha1.RconSpec{}, ""),
mk("off", off, ""),
mk("own", own, ""),
mk(naming.SystemLobbyServer, v1alpha1.RconSpec{}, naming.SystemLobbyServer),
).Build()
get := func(name string) v1alpha1.RconSpec {
var ms v1alpha1.MinecraftServer
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
t.Fatalf("get %s: %v", name, err)
}
return ms.Spec.Rcon
}
report := convergeUserServerRcon(ctx, cl, "minecraft", false)
if len(report) != 1 || report[0].name != "demo" || report[0].updated || report[0].err != nil ||
!strings.Contains(report[0].skipped, "-user-rcon") {
t.Fatalf("report = %+v, want one skipped line for demo naming -user-rcon", report)
}
if got := get("demo"); got != (v1alpha1.RconSpec{}) {
t.Fatalf("the report-only pass wrote demo's rcon: %+v", got)
}
filled := convergeUserServerRcon(ctx, cl, "minecraft", true)
if len(filled) != 1 || filled[0].name != "demo" || !filled[0].updated || filled[0].err != nil {
t.Fatalf("fill = %+v, want exactly one update for demo", filled)
}
want := map[string]v1alpha1.RconSpec{
"demo": {Enabled: true, SecretRef: v1alpha1.SecretKeyRef{
Name: naming.RconSecretName("demo"), Key: naming.RconSecretKey}},
"off": off,
"own": own,
naming.SystemLobbyServer: {},
}
for name, rcon := range want {
if got := get(name); got != rcon {
t.Errorf("%s rcon = %+v, want %+v", name, got, rcon)
}
}
for _, fill := range []bool{false, true} {
if again := convergeUserServerRcon(ctx, cl, "minecraft", fill); len(again) != 0 {
t.Fatalf("second pass (fill=%v) = %+v, want nothing to do", fill, again)
}
}
}
+17 -98
View File
@@ -7,7 +7,6 @@ import (
"flag"
"fmt"
"io"
neturl "net/url"
"os"
"os/exec"
"path/filepath"
@@ -16,7 +15,6 @@ import (
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/retention"
)
@@ -36,7 +34,6 @@ var defaultKeep = map[string]int{
dbbackup.LabelDaily: 14,
dbbackup.LabelPreMigrate: 10,
dbbackup.LabelPreRestore: 5,
dbbackup.LabelOffsite: 1,
}
// cmdDB implements `felis db`: logical backups of the control-plane database
@@ -95,49 +92,17 @@ func parseWithArg(fs *flag.FlagSet, args []string) (string, bool) {
}
func dbDatabaseURL(path string) (string, error) {
db, err := dbDatabase(path)
return db.URL, err
}
func dbDatabase(path string) (config.DatabaseConfig, error) {
cfg, err := config.Load(path)
if err != nil {
return config.DatabaseConfig{}, err
return "", err
}
return cfg.Database, nil
}
// dbTools places pg_dump, pg_restore and psql. The installer's database runs in
// k3s and the host has no PostgreSQL client, so when the host config names the
// [database] deployment the tools run in its postgres container over the
// container's socket, as the URL's role on the URL's database. Otherwise they
// come from PATH and connect with the URL.
func dbTools(db config.DatabaseConfig) (dbbackup.Tools, error) {
if db.Deployment == "" {
return dbbackup.Tools{}, nil
}
ns, name, _ := strings.Cut(db.Deployment, "/")
u, err := neturl.Parse(db.URL)
if err != nil || u.User == nil || u.User.Username() == "" || strings.TrimPrefix(u.Path, "/") == "" {
return dbbackup.Tools{}, errors.New("[database] url must name the role and the database to run the tools in the database's pod")
}
return dbbackup.Tools{
Exec: []string{"k3s", "kubectl", "exec", "-i", "-n", ns, "deploy/" + name, "-c", platform.PostgresContainer, "--"},
Conn: fmt.Sprintf("host=%s port=%d dbname=%s user=%s connect_timeout=15",
platform.PostgresSocketDir, platform.PostgresPort,
libpqQuote(strings.TrimPrefix(u.Path, "/")), libpqQuote(u.User.Username())),
}, nil
}
// libpqQuote renders v as a single-quoted libpq connection-string value.
func libpqQuote(v string) string {
return "'" + strings.NewReplacer(`\`, `\\`, `'`, `\'`).Replace(v) + "'"
return cfg.Database.URL, nil
}
func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
label := fs.String("label", dbbackup.LabelManual, "bundle label; daily/pre-migrate/pre-restore bundles are pruned, manual ones never")
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, offsite 1, manual all)")
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, manual all)")
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory to bundle ("" for none)`)
noServers := fs.Bool("no-servers", false, "leave the MinecraftServer objects out of the bundle")
metrics := fs.String("metrics-file", "", "node-exporter textfile to rewrite on success (e.g. /var/lib/node_exporter/textfile_collector/felis_db_backup.prom)")
@@ -148,12 +113,7 @@ func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
fmt.Fprint(stderr, dbUsage)
return 2
}
db, err := dbDatabase(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
return 1
}
tools, err := dbTools(db)
url, err := dbDatabaseURL(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
return 1
@@ -162,7 +122,7 @@ func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
*keep = defaultKeep[*label]
}
o := dbbackup.BackupOptions{
DatabaseURL: db.URL, Tools: tools, Dir: *dir, Label: *label, Keep: *keep,
DatabaseURL: url, Dir: *dir, Label: *label, Keep: *keep,
StateDir: *stateDir, Version: resolvedVersion(), Log: stderr,
MetricsFile: *metrics, Record: true,
}
@@ -172,14 +132,6 @@ func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
defer cancel()
path, err := dbbackup.Backup(ctx, o)
if errors.Is(err, dbbackup.ErrServersMissing) {
// The bundle is on disk and holds the database; the exit status fails
// the timer's run so the gap shows in systemctl and the journal, and
// the panel and the watchdog read it from the record and the manifest.
fmt.Fprintf(stdout, "felis db backup: wrote %s\n", path)
fmt.Fprintf(stderr, "felis db backup: %v\n a restore from %s brings back the database but no servers; check `k3s kubectl get minecraftservers -A`, then run `felis db backup` again\n", err, filepath.Base(path))
return 1
}
if err != nil {
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
return 1
@@ -220,17 +172,12 @@ func dbRestore(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.W
return 1
}
if !*yes {
fmt.Fprintf(stderr, "felis db restore: this replaces every table in the felis database with %s (%s, taken %s, schema %d, holding %s).\n",
filepath.Base(bundle), m.Label, m.CreatedAt.Format(time.RFC3339), m.SchemaVersion, m.Counts.String())
fmt.Fprintf(stderr, "felis db restore: this replaces every table in the felis database with %s (%s, taken %s, schema %d).\n",
filepath.Base(bundle), m.Label, m.CreatedAt.Format(time.RFC3339), m.SchemaVersion)
fmt.Fprintln(stderr, "Scale felis-api and felis-operator to 0 first, then re-run with -yes.")
return 2
}
db, err := dbDatabase(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
return 1
}
tools, err := dbTools(db)
url, err := dbDatabaseURL(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
return 1
@@ -238,7 +185,7 @@ func dbRestore(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.W
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
defer cancel()
_, safety, err := dbbackup.Restore(ctx, dbbackup.RestoreOptions{
DatabaseURL: db.URL, Tools: tools, Bundle: bundle, Dir: *dir, Force: *force, SkipSafetyBackup: *noSafety,
DatabaseURL: url, Bundle: bundle, Dir: *dir, Force: *force, SkipSafetyBackup: *noSafety,
Safety: dbbackup.BackupOptions{Keep: defaultKeep[dbbackup.LabelPreRestore], StateDir: *stateDir,
Version: resolvedVersion(), ExportServers: exportMinecraftServers},
Log: stderr,
@@ -277,9 +224,8 @@ func dbVerify(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
fmt.Fprintf(stderr, "felis db verify: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "%s: ok\n taken %s (%s)\n felis %s\n schema %d\n holds %s\n %s\n",
filepath.Base(bundle), m.CreatedAt.Format(time.RFC3339), m.Label, orUnknown(m.FelisVersion), m.SchemaVersion,
m.Counts.String(), orUnknown(m.PGDumpVersion))
fmt.Fprintf(stdout, "%s: ok\n taken %s (%s)\n felis %s\n schema %d\n %s\n",
filepath.Base(bundle), m.CreatedAt.Format(time.RFC3339), m.Label, orUnknown(m.FelisVersion), m.SchemaVersion, orUnknown(m.PGDumpVersion))
for _, f := range m.Files {
if f.Link != "" {
fmt.Fprintf(stdout, " %-40s -> %s\n", f.Name, f.Link)
@@ -415,11 +361,10 @@ func humanBytes(n int64) string {
return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp])
}
// dbCheck is the freshness probe: exit 1 when the newest daily bundle is
// missing or older than -max-age, for a monitor or the break-glass console to
// act on.
// dbCheck is the freshness probe: exit 1 when the newest bundle is missing or
// older than -max-age, for a monitor or the break-glass console to act on.
func dbCheck(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
maxAge := fs.Duration("max-age", dbbackup.StaleAfter, "oldest acceptable newest daily bundle")
maxAge := fs.Duration("max-age", dbbackup.StaleAfter, "oldest acceptable newest bundle")
if err := fs.Parse(args); err != nil {
return 2
}
@@ -428,40 +373,14 @@ func dbCheck(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wri
fmt.Fprintf(stderr, "felis db check: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "felis db check: ok, newest daily backup %s (%s ago)\n", b.Name, dbbackup.Age(time.Since(b.Created)))
fmt.Fprintf(stdout, "felis db check: ok, newest backup %s (%s ago)\n", b.Name, dbbackup.Age(time.Since(b.Created)))
return 0
}
// serverExportTries and serverExportRetry are how long a backup waits out a
// cluster that is briefly away (an apiserver restart) before its bundle goes
// without the MinecraftServer objects.
const serverExportTries = 3
var serverExportRetry = 10 * time.Second
// exportMinecraftServers is dbbackup's ExportServers on the host: the
// MinecraftServer objects through k3s kubectl, tried serverExportTries times.
func exportMinecraftServers(ctx context.Context) ([]byte, error) {
for try := 1; ; try++ {
out, err := getMinecraftServers(ctx)
if err == nil {
return out, nil
}
if try == serverExportTries {
return nil, fmt.Errorf("%w (tried %d times)", err, try)
}
select {
case <-ctx.Done():
return nil, fmt.Errorf("%w (tried %d times)", err, try)
case <-time.After(serverExportRetry):
}
}
}
// getMinecraftServers reads every MinecraftServer through the host's k3s
// exportMinecraftServers reads every MinecraftServer through the host's k3s
// kubectl and strips what the API server owns, so the result can be fed back
// with `kubectl apply -f` on a rebuilt cluster.
func getMinecraftServers(ctx context.Context) ([]byte, error) {
func exportMinecraftServers(ctx context.Context) ([]byte, error) {
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
// Output, not the CombinedOutput kubectlOutput uses: a deprecation warning
+6 -388
View File
@@ -1,20 +1,15 @@
package main
import (
"archive/tar"
"bytes"
"context"
"encoding/json"
"flag"
"io"
"os"
"path/filepath"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/store"
)
@@ -30,63 +25,13 @@ func TestDBUsage(t *testing.T) {
}
}
func TestDBVerifySaysWhatTheBundleHolds(t *testing.T) {
dir := newPodRig(t)
cfg := podConfig(t, dir)
bundles := filepath.Join(dir, "bundles")
var out, errBuf bytes.Buffer
if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
t.Fatalf("backup: exit %d: %s", code, errBuf.String())
}
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
out.Reset()
if code := run([]string{"db", "verify", "-dir", bundles, filepath.Base(bundle)}, &out, &errBuf); code != 0 {
t.Fatalf("verify: exit %d: %s", code, errBuf.String())
}
for _, want := range []string{filepath.Base(bundle) + ": ok", "schema 3", "holds 4 accounts, 2 servers", "pg_dump (PostgreSQL) 18.6"} {
if !strings.Contains(out.String(), want) {
t.Errorf("verify output lacks %q:\n%s", want, out.String())
}
}
}
// TestDBRestoreNeedsYes: without -yes a restore describes the bundle and stops
// before anything reaches the database, even with -force and
// -no-safety-backup, which would otherwise let the replay run at once.
func TestDBRestoreNeedsYes(t *testing.T) {
dir := newPodRig(t)
cfg := podConfig(t, dir)
bundles := filepath.Join(dir, "bundles")
// A bundle that does not exist fails verification (1) before -yes matters;
// the -yes gate itself is exercised against a real bundle in internal/dbbackup
// and on the VM. Here: the refusal path never reaches the config or database.
var out, errBuf bytes.Buffer
if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
t.Fatalf("backup: exit %d: %s", code, errBuf.String())
}
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
podRuns(t, dir)
out.Reset()
errBuf.Reset()
code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-force", "-no-safety-backup", filepath.Base(bundle)}, &out, &errBuf)
if code != 2 {
t.Fatalf("exit %d, want 2; stderr %q", code, errBuf.String())
}
if want := filepath.Base(bundle) + " (manual, taken "; !strings.Contains(errBuf.String(), want) || !strings.Contains(errBuf.String(), "schema 3, holding 4 accounts, 2 servers).") {
t.Errorf("stderr %q does not describe the bundle", errBuf.String())
}
if !strings.Contains(errBuf.String(), "re-run with -yes") {
t.Errorf("stderr %q does not say how to go on", errBuf.String())
}
if argv, err := os.ReadFile(filepath.Join(dir, "k3s.args")); err == nil {
t.Errorf("a restore without -yes ran in the database pod:\n%s", argv)
}
// A bundle that does not verify is refused before -yes is weighed.
errBuf.Reset()
if code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-yes", "missing.tar"}, &out, &errBuf); code != 1 {
t.Errorf("missing bundle: exit %d, want 1; stderr %q", code, errBuf.String())
}
if _, err := os.Stat(filepath.Join(dir, "k3s.args")); err == nil {
t.Error("a missing bundle reached the database pod")
if code := run([]string{"db", "restore", "-dir", t.TempDir(), "missing.tar"}, &out, &errBuf); code != 1 {
t.Fatalf("exit %d, stderr %q", code, errBuf.String())
}
}
@@ -196,7 +141,7 @@ func TestPreMigrateBackupOnlyGuardsAPopulatedDatabase(t *testing.T) {
{"up to date", map[int]struct{}{1: {}, 2: {}}, false},
{"pending on a populated database", map[int]struct{}{1: {}}, true},
} {
path, err := preMigrateBackup(context.Background(), appliedDriver{done: tc.done}, ms, config.DatabaseConfig{URL: badURL}, t.TempDir(), io.Discard)
path, err := preMigrateBackup(context.Background(), appliedDriver{done: tc.done}, ms, badURL, t.TempDir(), io.Discard)
if attempted := err != nil; attempted != tc.attempt {
t.Errorf("%s: attempted = %v (err %v), want %v", tc.name, attempted, err, tc.attempt)
}
@@ -238,330 +183,3 @@ func TestAuditExportBounds(t *testing.T) {
}
}
}
// podK3s stands in for `k3s kubectl exec ... --`: it logs its argv and runs the
// command after -- from the "container" directory, which is the only place the
// PostgreSQL tools exist, as on an installed host. `kubectl get` lists one
// MinecraftServer, logged to k3s.get, and refuses its first N calls while
// servers_fail holds N.
const podK3s = `#!/bin/sh
if [ "$1" = kubectl ] && [ "$2" = get ]; then
printf '%s\n' "$*" >> "$FAKE_DIR/k3s.get"
n=$(/usr/bin/wc -l < "$FAKE_DIR/k3s.get")
if [ -f "$FAKE_DIR/servers_fail" ] && [ "$n" -le "$(/bin/cat "$FAKE_DIR/servers_fail")" ]; then
echo "The connection to the server 127.0.0.1:6443 was refused - did you specify the right host or port?" >&2
exit 1
fi
echo '{"apiVersion":"v1","kind":"List","items":[{"apiVersion":"felis.lolicon.best/v1alpha1","kind":"MinecraftServer","metadata":{"name":"lobby","namespace":"felis-servers","uid":"u-1"},"spec":{"type":"PAPER"},"status":{"phase":"Running"}}]}'
exit 0
fi
printf '%s\n' "$*" >> "$FAKE_DIR/k3s.args"
while [ $# -gt 0 ] && [ "$1" != "--" ]; do shift; done
shift
tool=$1; shift
exec /usr/bin/env -i FAKE_DIR="$FAKE_DIR" PATH=/usr/bin:/bin "$FAKE_DIR/container/$tool" "$@"
`
var podTools = map[string]string{
"pg_dump": `#!/bin/sh
case "$1" in --version) echo "pg_dump (PostgreSQL) 18.6"; exit 0 ;; esac
printf 'PGDMP-fake-archive'
`,
"pg_restore": `#!/bin/sh
cat > /dev/null
`,
"psql": `#!/bin/sh
for a in "$@"; do case "$a" in *"FROM users"*) echo "4|2"; exit 0 ;; esac; done
for a in "$@"; do [ "$a" = "-c" ] && { echo 3; exit 0; }; done
cat > /dev/null
`,
}
const podPassword = "pw-must-stay-on-the-host"
// podDB is the host config's [database] on an installed host.
var podDB = config.DatabaseConfig{
URL: "postgres://felis:" + podPassword + "@127.0.0.1:15432/felis?sslmode=disable",
Deployment: "felis/felis-postgres",
}
const podExecPrefix = "kubectl exec -i -n felis deploy/felis-postgres -c postgres -- "
// newPodRig puts the fake k3s on PATH, alone, and returns the directory its
// k3s.args log lands in.
func newPodRig(t *testing.T) string {
t.Helper()
dir := t.TempDir()
bin := filepath.Join(dir, "bin")
container := filepath.Join(dir, "container")
for _, d := range []string{bin, container} {
if err := os.Mkdir(d, 0o755); err != nil {
t.Fatal(err)
}
}
writeTestFile(t, filepath.Join(bin, "k3s"), podK3s, 0o755)
for name, body := range podTools {
writeTestFile(t, filepath.Join(container, name), body, 0o755)
}
t.Setenv("PATH", bin)
t.Setenv("FAKE_DIR", dir)
return dir
}
// podRuns returns what the fake k3s ran since the last call, failing on any
// run outside the database container or with the password on its command
// line (visible to every local user in ps).
func podRuns(t *testing.T, dir string) []string {
t.Helper()
log := filepath.Join(dir, "k3s.args")
argv, err := os.ReadFile(log)
if err != nil {
t.Fatalf("nothing ran through k3s: %v", err)
}
os.Remove(log)
var runs []string
for _, line := range strings.Split(strings.TrimSpace(string(argv)), "\n") {
if !strings.HasPrefix(line, podExecPrefix) {
t.Errorf("k3s ran %q, want everything under %q", line, podExecPrefix)
}
if strings.Contains(line, podPassword) {
t.Errorf("the password crossed into the pod on a command line: %q", line)
}
runs = append(runs, strings.TrimPrefix(line, podExecPrefix))
}
return runs
}
// podConfig writes an installed host's felis.toml, [database] pointing at the
// pod, into dir.
func podConfig(t *testing.T, dir string) string {
t.Helper()
toml := strings.Replace(installerTOML("example.com", "127.0.0.1"),
`url = "postgres://felis:[email protected]:5432/felis?sslmode=disable"`,
`url = "`+podDB.URL+`"
deployment = "`+podDB.Deployment+`"`, 1)
cfg := filepath.Join(dir, "felis.toml")
writeTestFile(t, cfg, toml, 0o600)
return cfg
}
func ranIn(runs []string, prefix string) bool {
for _, r := range runs {
if strings.HasPrefix(r, prefix) {
return true
}
}
return false
}
// TestDBBackupAndRestoreRunTheToolsInTheDatabasePod: on an installed host the
// database is a k3s Deployment and no PostgreSQL client exists outside it, so
// `felis db backup` and `restore` must reach the tools through kubectl exec,
// over the pod's socket, and without putting the role's password on a command
// line.
func TestDBBackupAndRestoreRunTheToolsInTheDatabasePod(t *testing.T) {
dir := newPodRig(t)
cfg := podConfig(t, dir)
var out, errBuf bytes.Buffer
bundles := filepath.Join(dir, "bundles")
if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
t.Fatalf("backup: exit %d: %s", code, errBuf.String())
}
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
if _, err := dbbackupVerify(bundle); err != nil {
t.Fatalf("the bundle does not verify: %v", err)
}
runs := podRuns(t, dir)
if !ranIn(runs, "pg_dump --format=custom --no-password --dbname=host=/var/run/postgresql port=5432 dbname='felis' user='felis'") {
t.Errorf("pg_dump did not dump over the pod's socket as felis on felis: %q", runs)
}
out.Reset()
errBuf.Reset()
if code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-yes", "-force", "-no-safety-backup", bundle}, &out, &errBuf); code != 0 {
t.Fatalf("restore: exit %d: %s", code, errBuf.String())
}
runs = podRuns(t, dir)
if !ranIn(runs, "pg_restore --no-owner --no-privileges --file=-") || !ranIn(runs, "psql -X -q -w -v ON_ERROR_STOP=1 -d host=/var/run/postgresql") {
t.Errorf("the replay did not run in the pod: %q", runs)
}
}
// TestPreMigrateBackupRunsInTheDatabasePod: the snapshot in front of an upgrade
// is the one taken most often, by bootstrap on every rerun.
func TestPreMigrateBackupRunsInTheDatabasePod(t *testing.T) {
dir := newPodRig(t)
ms := []store.Migration{{Version: 1}, {Version: 2}}
// The snapshot also bundles /etc/felis, which a test machine may lack; the
// dump runs first either way.
_, err := preMigrateBackup(context.Background(), appliedDriver{done: map[int]struct{}{1: {}}}, ms, podDB, filepath.Join(dir, "bundles"), io.Discard)
if err != nil && !strings.Contains(err.Error(), "read host state") {
t.Fatalf("snapshot: %v", err)
}
if runs := podRuns(t, dir); !ranIn(runs, "pg_dump --format=custom") {
t.Errorf("pg_dump did not run in the pod: %q", runs)
}
}
func TestDBToolsNeedTheRoleAndDatabase(t *testing.T) {
if tools, err := dbTools(config.DatabaseConfig{URL: "postgres://felis:pw@db:5432/felis"}); err != nil || len(tools.Exec) != 0 {
t.Errorf("no deployment: tools %+v err %v, want the PATH tools", tools, err)
}
for _, u := range []string{"postgres://db:5432/felis", "postgres://felis:pw@db:5432/"} {
if _, err := dbTools(config.DatabaseConfig{URL: u, Deployment: "felis/felis-postgres"}); err == nil {
t.Errorf("%s: no error, want a refusal (the pod connection needs the role and the database)", u)
}
}
tools, err := dbTools(config.DatabaseConfig{URL: `postgres://o%27brien@db/my%20db`, Deployment: "felis/felis-postgres"})
if err != nil {
t.Fatal(err)
}
if !strings.Contains(tools.Conn, `dbname='my db' user='o\'brien'`) {
t.Errorf("Conn = %q, want the values quoted for libpq", tools.Conn)
}
}
var dbbackupVerify = dbbackup.Verify
func noServerExportWait(t *testing.T) {
t.Helper()
old := serverExportRetry
serverExportRetry = 0
t.Cleanup(func() { serverExportRetry = old })
}
// serverGets counts the `kubectl get` calls the fake k3s answered or refused.
func serverGets(dir string) int {
b, _ := os.ReadFile(filepath.Join(dir, "k3s.get"))
return strings.Count(string(b), "\n")
}
// bundleServers returns the bundle's k8s/minecraftservers.json, or nil.
func bundleServers(t *testing.T, bundle string) []byte {
t.Helper()
f, err := os.Open(bundle)
if err != nil {
t.Fatal(err)
}
defer f.Close()
tr := tar.NewReader(f)
for {
h, err := tr.Next()
if err == io.EOF {
return nil
}
if err != nil {
t.Fatal(err)
}
if h.Name == "k8s/minecraftservers.json" {
data, err := io.ReadAll(tr)
if err != nil {
t.Fatal(err)
}
return data
}
}
}
// TestDBBackupWithoutServersFails: when the cluster stays away the daily
// bundle is still written, and `felis db backup` exits 1, so the timer's run
// shows failed, saying a restore from the bundle brings back no servers.
func TestDBBackupWithoutServersFails(t *testing.T) {
noServerExportWait(t)
dir := newPodRig(t)
cfg := podConfig(t, dir)
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
var out, errBuf bytes.Buffer
code := run([]string{"db", "backup", "-config", cfg, "-dir", filepath.Join(dir, "bundles"), "-state-dir", "", "-label", "daily"}, &out, &errBuf)
if code != 1 {
t.Fatalf("exit %d, want 1: %s", code, errBuf.String())
}
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
m, err := dbbackupVerify(bundle)
if err != nil {
t.Fatalf("the database must still be bundled: %v", err)
}
if !strings.Contains(m.ServersError, "6443 was refused") || !strings.Contains(m.ServersError, "(tried 3 times)") || bundleServers(t, bundle) != nil {
t.Errorf("manifest servers error = %q", m.ServersError)
}
if n := serverGets(dir); n != serverExportTries {
t.Errorf("export tried %d times, want %d", n, serverExportTries)
}
if msg := errBuf.String(); !strings.Contains(msg, "a restore from "+filepath.Base(bundle)+" brings back the database but no servers") {
t.Errorf("stderr = %q", msg)
}
}
// TestDBBackupRetriesTheServerExport: a cluster back on the last try costs the
// bundle nothing, and what it holds is ready for kubectl apply.
func TestDBBackupRetriesTheServerExport(t *testing.T) {
noServerExportWait(t)
dir := newPodRig(t)
cfg := podConfig(t, dir)
writeTestFile(t, filepath.Join(dir, "servers_fail"), "2", 0o600)
var out, errBuf bytes.Buffer
if code := run([]string{"db", "backup", "-config", cfg, "-dir", filepath.Join(dir, "bundles"), "-state-dir", ""}, &out, &errBuf); code != 0 {
t.Fatalf("exit %d: %s", code, errBuf.String())
}
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
servers := string(bundleServers(t, bundle))
if !strings.Contains(servers, `"name": "lobby"`) || strings.Contains(servers, "status") || strings.Contains(servers, "u-1") {
t.Errorf("k8s/minecraftservers.json = %s", servers)
}
if n := serverGets(dir); n != 3 {
t.Errorf("export tried %d times, want 3", n)
}
}
// TestPreMigrateBackupExportsServers: the snapshot every upgrade takes, often
// the newest bundle, carries the MinecraftServer objects too; a cluster that
// is away does not hold back the migration, whose rollback needs the database
// alone.
func TestPreMigrateBackupExportsServers(t *testing.T) {
noServerExportWait(t)
dir := newPodRig(t)
old := preMigrateStateDir
preMigrateStateDir = ""
t.Cleanup(func() { preMigrateStateDir = old })
ms := []store.Migration{{Version: 1}, {Version: 2}}
bundles := filepath.Join(dir, "bundles")
pending := appliedDriver{done: map[int]struct{}{1: {}}}
path, err := preMigrateBackup(context.Background(), pending, ms, podDB, bundles, io.Discard)
if err != nil {
t.Fatalf("snapshot: %v", err)
}
if !strings.Contains(string(bundleServers(t, path)), `"name": "lobby"`) {
t.Errorf("%s holds no MinecraftServer objects", path)
}
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
path, err = preMigrateBackup(context.Background(), pending, ms, podDB, bundles, io.Discard)
if err != nil || path == "" {
t.Fatalf("snapshot with the cluster away = %q, %v; want the bundle and no error", path, err)
}
if m, err := dbbackupVerify(path); err != nil || m.ServersError == "" || bundleServers(t, path) != nil {
t.Errorf("snapshot with the cluster away: %+v, %v", m, err)
}
}
// TestServerExportStopsWaitingWithTheContext: a backup whose time is up stops
// waiting for the cluster between tries.
func TestServerExportStopsWaitingWithTheContext(t *testing.T) {
dir := newPodRig(t)
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
// Long enough for the first try to run to its refusal: starting the fake
// k3s on a busy machine can take a few hundred ms. Still far below the
// 10s retry wait, so waiting it out would fail the check below.
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
defer cancel()
start := time.Now()
_, err := exportMinecraftServers(ctx)
if err == nil || !strings.Contains(err.Error(), "(tried 1 times)") || serverGets(dir) != 1 {
t.Fatalf("err = %v after %d tries, want the first failure alone", err, serverGets(dir))
}
if took := time.Since(start); took > 5*time.Second {
t.Errorf("took %s, want the context's deadline, not the %s retry wait", took, serverExportRetry)
}
}
-387
View File
@@ -1,387 +0,0 @@
package main
import (
"context"
"errors"
"flag"
"fmt"
"io"
"os"
"os/exec"
"path/filepath"
"slices"
"strings"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/watchdog"
)
// systemdUnitDir is where the installer writes its units.
const systemdUnitDir = "/etc/systemd/system"
// hostCommand runs a host tool (systemctl, journalctl, k3s) and returns its
// stdout. A tool that exits non-zero still returns what it printed:
// `systemctl is-active` prints "inactive" and exits 3.
func hostCommand(ctx context.Context, name string, args ...string) ([]byte, error) {
return exec.CommandContext(ctx, name, args...).Output()
}
// hostServices are the long-running units a full install depends on, checked
// when their unit file is present: k3s runs the cluster, felis-velocity is the
// game proxy, felis-nano the single-binary host that runs without k3s.
var hostServices = []string{"k3s.service", "felis-velocity.service", "felis-nano.service"}
// cmdDoctor runs every check felis watchdog runs, with the settings
// felis-watchdog.service gives it, plus what only the host shows (systemd
// units that failed or stopped, timers that no longer fire, alerts that reach
// no one), and prints them grouped by area with where to look next. It mails
// nothing, pings no heartbeat and leaves the watchdog's state alone: it is
// safe to run at any time, as often as wanted.
func cmdDoctor(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("doctor", flag.ContinueOnError)
fs.SetOutput(stderr)
unitDir := fs.String("systemd-dir", systemdUnitDir, "where the installer's systemd units are")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if os.Geteuid() != 0 {
fmt.Fprintln(stderr, "felis doctor: run as root (sudo felis doctor): the checks read root-only state under /etc/felis and /var/lib/felis")
return 1
}
host, _ := os.Hostname()
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
return runDoctor(ctx, doctorEnv{unitDir: *unitDir, run: hostCommand, now: time.Now(), host: host}, stdout)
}
// doctorEnv is what one doctor run reads the host through.
type doctorEnv struct {
unitDir string
run func(ctx context.Context, name string, args ...string) ([]byte, error)
now time.Time
host string
}
// doctorAreas are the report's headings in order, by the area findingArea
// puts a finding under.
var doctorAreas = []struct{ key, title string }{
{key: "config", title: "configuration"},
{key: "cluster", title: "Kubernetes cluster"},
{key: "postgres", title: "PostgreSQL"},
{key: "proxy", title: "game proxy"},
{key: "db-backup", title: "database backups"},
{key: "offsite", title: "off-site copy"},
{key: "scan-db", title: "build scan database"},
{key: "disk", title: "disk space"},
{key: "memory", title: "memory"},
{key: "k3s-certs", title: "k3s certificates"},
{key: "host-address", title: "node address"},
{key: "clock", title: "clock"},
{key: "systemd", title: "systemd units and timers"},
{key: "alerts", title: "alerting"},
}
func runDoctor(ctx context.Context, env doctorEnv, stdout io.Writer) int {
unit := filepath.Join(env.unitDir, "felis-watchdog.service")
w, found, unitErr := watchdogUnitFlags(unit)
var report watchdog.Report
var notes []string
fmt.Fprintf(stdout, "felis doctor on %s at %s\n", env.host, env.now.UTC().Format("2006-01-02 15:04 UTC"))
switch {
case unitErr != nil:
report.Findings = append(report.Findings, watchdog.Finding{
Key: "watchdog/unit", Severity: watchdog.Critical,
SummaryEN: fmt.Sprintf("cannot read the watchdog's settings: %v", unitErr),
Hint: "rerun the installer (deploy/bootstrap.sh) to rewrite felis-watchdog.service",
})
fmt.Fprintf(stdout, "checks run with the watchdog's defaults (config %s)\n", w.cfgPath)
case !found:
report.Findings = append(report.Findings, watchdog.Finding{
Key: "watchdog/unit", Severity: watchdog.Critical,
SummaryEN: fmt.Sprintf("%s is not installed: nothing checks this host or mails anyone when it breaks", unit),
Hint: "rerun the installer (deploy/bootstrap.sh), which installs felis-watchdog.timer",
})
fmt.Fprintf(stdout, "checks run with the watchdog's defaults (config %s)\n", w.cfgPath)
default:
fmt.Fprintf(stdout, "checks run as %s runs them (config %s)\n", unit, w.cfgPath)
}
fmt.Fprintln(stdout)
skip := map[string]string{}
cfg, cfgErr := config.Load(w.cfgPath)
if cfgErr != nil {
report.Findings = append(report.Findings, watchdog.Finding{
Key: "config", Severity: watchdog.Critical,
SummaryEN: cfgErr.Error(),
Hint: "the installer writes it (deploy/bootstrap.sh); felis watchdog fails on every run until it loads",
})
// The host's units are read without it; whether alerts reach anyone
// is not known without its relay.
for _, a := range doctorAreas {
if a.key != "config" && a.key != "systemd" {
skip[a.key] = "the configuration did not load"
}
}
} else {
_, owners, ownersErr := watchdogProbes(ctx, w, cfg, env.now, &report)
report.Findings = append(report.Findings, alertReachFindings(cfg, owners, ownersErr)...)
if w.proxyAddr == "" {
skip["proxy"] = "no -proxy-addr"
}
if w.backupDir == "" {
skip["db-backup"] = "no -backup-dir"
}
if !cfg.Offsite.Enabled() {
skip["offsite"] = "not configured"
}
if !usesMirroredScanDB(cfg) {
skip["scan-db"] = "builds do not scan against the registry's copy"
}
if len(splitList(w.certDirs)) == 0 {
skip["k3s-certs"] = "no -k3s-cert-dirs"
}
if w.nodeIP == "" {
skip["host-address"] = "no -node-ip"
}
}
report.Findings = append(report.Findings, unitFindings(ctx, env)...)
switch url, err := readHeartbeatURL(w.heartbeatFile); {
case err != nil:
report.Findings = append(report.Findings, watchdog.Finding{
Key: "watchdog/heartbeat", Severity: watchdog.Warning,
SummaryEN: fmt.Sprintf("the heartbeat URL is unusable, so no run pings it: %v", err),
Hint: "rerun the installer with FELIS_WATCHDOG_HEARTBEAT_URL set (docs/troubleshooting.md §14)",
})
case url == "":
notes = append(notes, "no heartbeat URL is set: a host that goes down entirely, or a watchdog that stops running, alerts no one. "+
"Rerun the installer with FELIS_WATCHDOG_HEARTBEAT_URL (docs/troubleshooting.md §14)")
}
if until := watchdog.QuietUntil(w.quietPath); env.now.Before(until) {
notes = append(notes, fmt.Sprintf("the watchdog mails nothing until %s (%s): the installer holds it while it restarts things on purpose, "+
"and a marker an installer killed mid-run left behind holds it until then",
until.UTC().Format("2006-01-02 15:04 UTC"), w.quietPath))
}
return printDoctorReport(stdout, report.Findings, skip, notes)
}
// printDoctorReport prints each area's findings under its heading, an area
// with none as fine or, when skip says why, as not checked, then the notes
// and the count. It returns the exit status: 1 when anything was found.
func printDoctorReport(stdout io.Writer, findings []watchdog.Finding, skip map[string]string, notes []string) int {
byArea := map[string][]watchdog.Finding{}
for _, f := range findings {
a := findingArea(f.Key)
byArea[a] = append(byArea[a], f)
}
var critical, warning int
for _, a := range doctorAreas {
fs := byArea[a.key]
switch {
case len(fs) > 0:
case skip[a.key] != "":
fmt.Fprintf(stdout, "- %s: not checked, %s\n", a.title, skip[a.key])
continue
default:
fmt.Fprintf(stdout, "✓ %s\n", a.title)
continue
}
mark := "!"
if slices.ContainsFunc(fs, func(f watchdog.Finding) bool { return f.Severity == watchdog.Critical }) {
mark = "✗"
}
fmt.Fprintf(stdout, "%s %s\n", mark, a.title)
for _, f := range fs {
if f.Severity == watchdog.Critical {
critical++
} else {
warning++
}
fmt.Fprintf(stdout, " %-8s %s: %s\n", f.Severity, f.Key, f.SummaryEN)
if f.Hint != "" {
fmt.Fprintf(stdout, " → %s\n", f.Hint)
}
}
}
for _, n := range notes {
fmt.Fprintf(stdout, "\nnote: %s\n", n)
}
fmt.Fprintln(stdout)
if critical+warning == 0 {
fmt.Fprintln(stdout, "no problems found")
return 0
}
fmt.Fprintf(stdout, "%d problem(s): %d critical, %d warning(s)\n", critical+warning, critical, warning)
return 1
}
// findingArea is the report heading a finding key goes under.
func findingArea(key string) string {
head, _, _ := strings.Cut(key, "/")
switch head {
case "kube-api", "deployment", "system-server", "server-failed", "job-failed", "reaper-stale", "node":
return "cluster"
case "db-backup", "db-backup-servers":
return "db-backup"
case "unit", "timer":
return "systemd"
case "watchdog":
return "alerts"
}
return head
}
// alertReachFindings is why the watchdog's alerts would reach no one, which
// its own runs only log: no relay, or no owner with a verified address.
func alertReachFindings(cfg *config.Config, owners []string, ownersErr error) []watchdog.Finding {
var out []watchdog.Finding
if cfg.SMTP.Host == "" {
out = append(out, watchdog.Finding{
Key: "alerts/relay", Severity: watchdog.Warning,
SummaryEN: "no [smtp] relay is configured: the watchdog logs its alerts to the journal and mails no one",
Hint: "sudo felis setup, step SMTP",
})
}
if ownersErr == nil && len(owners) == 0 {
out = append(out, watchdog.Finding{
Key: "alerts/recipients", Severity: watchdog.Warning,
SummaryEN: "no owner account has a verified email: the watchdog's alerts reach no one",
Hint: "an owner verifies an address in the panel's account settings",
})
}
return out
}
// unitFindings reports the installer's systemd units that failed, the
// long-running ones that are not running, and timers that no longer fire.
func unitFindings(ctx context.Context, env doctorEnv) []watchdog.Finding {
var out []watchdog.Finding
seen := map[string]bool{}
failed, err := env.run(ctx, "systemctl", "list-units", "--all", "--plain", "--no-legend", "--no-pager", "--state=failed", "felis-*", "k3s.service")
if err != nil && len(failed) == 0 {
return []watchdog.Finding{{
Key: "unit/systemctl", Severity: watchdog.Warning,
SummaryEN: fmt.Sprintf("systemctl list-units failed, so no unit was checked: %v", err),
}}
}
for _, line := range strings.Split(string(failed), "\n") {
fields := strings.Fields(line)
if len(fields) == 0 {
continue
}
name := fields[0]
seen[name] = true
out = append(out, watchdog.Finding{
Key: "unit/" + name, Severity: watchdog.Critical,
SummaryEN: name + " failed",
Hint: fmt.Sprintf("journalctl -u %s -n 100 --no-pager; once fixed, sudo systemctl reset-failed %s (a timer's job clears on its next good run)", name, name),
})
}
for _, name := range hostServices {
if seen[name] {
continue
}
if _, err := os.Stat(filepath.Join(env.unitDir, name)); err != nil {
continue
}
if state := unitActiveState(ctx, env, name); state != "active" {
out = append(out, watchdog.Finding{
Key: "unit/" + name, Severity: watchdog.Critical,
SummaryEN: fmt.Sprintf("%s is %s", name, state),
Hint: fmt.Sprintf("sudo systemctl start %s; journalctl -u %s -n 100 --no-pager", name, name),
})
}
}
timers, _ := filepath.Glob(filepath.Join(env.unitDir, "felis-*.timer"))
for _, path := range timers {
name := filepath.Base(path)
if state := unitActiveState(ctx, env, name); state != "active" {
out = append(out, watchdog.Finding{
Key: "timer/" + name, Severity: watchdog.Warning,
SummaryEN: fmt.Sprintf("%s is %s: the job it starts no longer runs", name, state),
Hint: fmt.Sprintf("sudo systemctl enable --now %s", name),
})
}
}
return out
}
// unitActiveState is what `systemctl is-active` says of unit.
func unitActiveState(ctx context.Context, env doctorEnv, unit string) string {
out, err := env.run(ctx, "systemctl", "is-active", unit)
if state := strings.TrimSpace(string(out)); state != "" {
return state
}
return fmt.Sprintf("in an unknown state (systemctl is-active: %v)", err)
}
// watchdogUnitFlags reads the flags felis-watchdog.service runs felis
// watchdog with. found is false when there is no such unit; w is then the
// watchdog's defaults.
func watchdogUnitFlags(path string) (w watchdogFlags, found bool, err error) {
fs := flag.NewFlagSet("watchdog", flag.ContinueOnError)
fs.SetOutput(io.Discard)
w.register(fs)
raw, err := os.ReadFile(path)
if errors.Is(err, os.ErrNotExist) {
return w, false, nil
}
if err != nil {
return w, false, err
}
for _, line := range strings.Split(string(raw), "\n") {
cmd, ok := strings.CutPrefix(strings.TrimSpace(line), "ExecStart=")
if !ok {
continue
}
fields := execArgs(cmd)
i := slices.Index(fields, "watchdog")
if i < 0 {
continue
}
if err := fs.Parse(fields[i+1:]); err != nil {
return w, true, fmt.Errorf("%s: %w", path, err)
}
return w, true, nil
}
return w, true, fmt.Errorf("%s runs no `felis watchdog`", path)
}
// execArgs splits an ExecStart= command line into its words. A word may be
// quoted with " or ', as systemd allows, which is how an empty value is
// written.
func execArgs(s string) []string {
var out []string
var cur strings.Builder
inWord := false
var quote rune
for _, r := range s {
switch {
case quote != 0:
if r == quote {
quote = 0
} else {
cur.WriteRune(r)
}
case r == '"' || r == '\'':
quote, inWord = r, true
case r == ' ' || r == '\t':
if inWord {
out = append(out, cur.String())
cur.Reset()
inWord = false
}
default:
cur.WriteRune(r)
inWord = true
}
}
if inWord {
out = append(out, cur.String())
}
return out
}
-423
View File
@@ -1,423 +0,0 @@
package main
import (
"bytes"
"context"
"errors"
"os"
"path/filepath"
"regexp"
"strconv"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/watchdog"
)
// bootstrapWatchdogExecStart is the ExecStart= line deploy/bootstrap.sh writes
// into felis-watchdog.service, with its variables filled in as an install
// fills them.
func bootstrapWatchdogExecStart(t *testing.T, vars map[string]string) string {
t.Helper()
raw, err := os.ReadFile("../../deploy/bootstrap.sh")
if err != nil {
t.Fatal(err)
}
var line string
for _, l := range strings.Split(string(raw), "\n") {
if strings.HasPrefix(l, "ExecStart=${HOST_BIN} watchdog -config") {
line = l
}
}
if line == "" {
t.Fatal("deploy/bootstrap.sh writes no `ExecStart=${HOST_BIN} watchdog -config` line")
}
line = strings.ReplaceAll(line, "${NODE_IP:+ -node-ip ${NODE_IP}}", " -node-ip "+vars["NODE_IP"])
line = regexp.MustCompile(`\$\{([A-Za-z_]+)\}`).ReplaceAllStringFunc(line, func(m string) string {
v, ok := vars[m[2:len(m)-1]]
if !ok {
t.Fatalf("bootstrap's watchdog ExecStart= uses %s, which this test does not fill in", m)
}
return v
})
return line
}
// felis doctor reads the watchdog's settings from the unit the installer
// writes, so it checks the paths the timer's runs check.
func TestWatchdogUnitFlagsReadsTheInstallersUnit(t *testing.T) {
dir := t.TempDir()
exec := bootstrapWatchdogExecStart(t, map[string]string{
"HOST_BIN": "/usr/local/bin/felis", "STATE_DIR": "/srv/felis-etc", "WATCHDOG_STATE": "/srv/watchdog/state.json",
"WATCHDOG_QUIET_FILE": "/srv/quiet-until", "FELIS_DB_BACKUP_DIR": "/srv/db-backups", "FELIS_GAME_PORT": "25577",
"disks": "/,/srv/data", "NODE_IP": "10.0.0.5",
})
unit := filepath.Join(dir, "felis-watchdog.service")
writeTestFile(t, unit, "[Unit]\nDescription=Felis watchdog\n\n[Service]\nType=oneshot\n"+exec+"\nTimeoutStartSec=3min\n", 0o644)
w, found, err := watchdogUnitFlags(unit)
if err != nil || !found {
t.Fatalf("found %v, err %v", found, err)
}
got := []string{w.cfgPath, w.statePath, w.quietPath, w.backupDir, w.proxyAddr, w.diskPaths, w.nodeIP, w.heartbeatFile, w.controlNS}
want := []string{"/srv/felis-etc/felis.host.toml", "/srv/watchdog/state.json", "/srv/quiet-until", "/srv/db-backups", "127.0.0.1:25577", "/,/srv/data", "10.0.0.5", defaultHeartbeatFile, "felis"}
if strings.Join(got, "|") != strings.Join(want, "|") {
t.Errorf("read\n %q\nwant\n %q", got, want)
}
w, found, err = watchdogUnitFlags(filepath.Join(dir, "missing.service"))
if err != nil || found || w.cfgPath != "/etc/felis/felis.toml" || w.backupDir != "/var/lib/felis/db-backups" {
t.Errorf("no unit: found %v, err %v, config %q, backups %q; want the watchdog's defaults", found, err, w.cfgPath, w.backupDir)
}
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis version\n", 0o644)
if _, found, err := watchdogUnitFlags(unit); !found || err == nil || !strings.Contains(err.Error(), "runs no `felis watchdog`") {
t.Errorf("a unit that runs something else: found %v, err %v", found, err)
}
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis watchdog -no-such-flag x\n", 0o644)
if _, _, err := watchdogUnitFlags(unit); err == nil {
t.Error("a flag this binary does not know was accepted")
}
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis watchdog -backup-dir \"\"\t-proxy-addr '127.0.0.1:1' -disk-paths \"/a b\"\n", 0o644)
if w, _, err := watchdogUnitFlags(unit); err != nil || w.backupDir != "" || w.proxyAddr != "127.0.0.1:1" || w.diskPaths != "/a b" {
t.Errorf("quoted words: backups %q, proxy %q, disks %q, err %v", w.backupDir, w.proxyAddr, w.diskPaths, err)
}
}
func TestFindingArea(t *testing.T) {
for key, want := range map[string]string{
"kube-api": "cluster",
"deployment/felis-api": "cluster",
"system-server/lobby": "cluster",
"server-failed/survival": "cluster",
"job-failed/reaper-123": "cluster",
"reaper-stale": "cluster",
"node/felis-1/NotReady": "cluster",
"postgres": "postgres",
"proxy": "proxy",
"db-backup": "db-backup",
"db-backup-servers": "db-backup",
"offsite": "offsite",
"scan-db": "scan-db",
"disk//var/lib/felis": "disk",
"memory": "memory",
"k3s-certs": "k3s-certs",
"host-address": "host-address",
"clock": "clock",
"config": "config",
"unit/felis-offsite.service": "systemd",
"timer/felis-db-backup.timer": "systemd",
"watchdog/unit": "alerts",
"watchdog/heartbeat": "alerts",
"alerts/relay": "alerts",
"alerts/recipients": "alerts",
"something-a-later-release-reported": "something-a-later-release-reported",
} {
if got := findingArea(key); got != want {
t.Errorf("findingArea(%q) = %q, want %q", key, got, want)
}
}
}
func TestAlertReachFindings(t *testing.T) {
relay := &config.Config{SMTP: config.SMTPConfig{Host: "smtp.example.com"}}
keys := func(fs []watchdog.Finding) string {
var k []string
for _, f := range fs {
k = append(k, f.Key)
}
return strings.Join(k, ",")
}
for _, tc := range []struct {
what string
cfg *config.Config
owners []string
ownersErr error
want string
}{
{"a relay and an owner", relay, []string{"[email protected]"}, nil, ""},
{"no relay", &config.Config{}, []string{"[email protected]"}, nil, "alerts/relay"},
{"no owner with an address", relay, nil, nil, "alerts/recipients"},
{"PostgreSQL down: its own finding says so", relay, nil, errors.New("refused"), ""},
{"neither", &config.Config{}, nil, nil, "alerts/relay,alerts/recipients"},
} {
if got := keys(alertReachFindings(tc.cfg, tc.owners, tc.ownersErr)); got != tc.want {
t.Errorf("%s: %q, want %q", tc.what, got, tc.want)
}
}
}
// fakeSystemctl answers list-units with failed and is-active from states;
// a unit missing from states is "inactive", as systemctl says, and one whose
// state is "" gets no answer.
func fakeSystemctl(failed string, states map[string]string) func(ctx context.Context, name string, args ...string) ([]byte, error) {
return func(_ context.Context, name string, args ...string) ([]byte, error) {
if name != "systemctl" || len(args) == 0 {
return nil, errors.New("unexpected command " + name)
}
switch args[0] {
case "list-units":
return []byte(failed), nil
case "is-active":
if s, ok := states[args[1]]; ok && s == "" {
return nil, errors.New("signal: killed")
} else if ok {
return []byte(s + "\n"), nil
}
return []byte("inactive\n"), errors.New("exit status 3")
}
return nil, errors.New("unexpected systemctl " + args[0])
}
}
func TestUnitFindings(t *testing.T) {
dir := t.TempDir()
for _, f := range []string{"k3s.service", "felis-velocity.service", "felis-db-backup.timer", "felis-offsite.timer", "felis-offsite.service"} {
writeTestFile(t, filepath.Join(dir, f), "[Unit]\n", 0o644)
}
env := doctorEnv{unitDir: dir, run: fakeSystemctl(
"felis-offsite.service loaded failed failed Felis off-site copy\nfelis-velocity.service loaded failed failed Velocity\n",
map[string]string{"k3s.service": "active", "felis-db-backup.timer": "active", "felis-offsite.timer": "inactive"},
)}
var got []string
for _, f := range unitFindings(context.Background(), env) {
got = append(got, string(f.Severity)+" "+f.Key+": "+f.SummaryEN)
}
want := []string{
"critical unit/felis-offsite.service: felis-offsite.service failed",
"critical unit/felis-velocity.service: felis-velocity.service failed",
"warning timer/felis-offsite.timer: felis-offsite.timer is inactive: the job it starts no longer runs",
}
if strings.Join(got, "\n") != strings.Join(want, "\n") {
t.Errorf("findings:\n%s\nwant (felis-nano.service has no unit file here, and a failed unit is reported once):\n%s", strings.Join(got, "\n"), strings.Join(want, "\n"))
}
env.run = fakeSystemctl("", map[string]string{"k3s.service": "activating", "felis-velocity.service": "active", "felis-db-backup.timer": "active", "felis-offsite.timer": "active"})
got = nil
for _, f := range unitFindings(context.Background(), env) {
got = append(got, f.Key+": "+f.SummaryEN)
}
if strings.Join(got, "\n") != "unit/k3s.service: k3s.service is activating" {
t.Errorf("findings %q, want only k3s.service, which is not active", got)
}
env.run = fakeSystemctl("", map[string]string{"k3s.service": "active", "felis-velocity.service": "active", "felis-db-backup.timer": "", "felis-offsite.timer": "active"})
got = nil
for _, f := range unitFindings(context.Background(), env) {
got = append(got, f.Key+": "+f.SummaryEN)
}
if want := "timer/felis-db-backup.timer: felis-db-backup.timer is in an unknown state (systemctl is-active: signal: killed): the job it starts no longer runs"; strings.Join(got, "\n") != want {
t.Errorf("findings %q, want %q", got, want)
}
env.run = func(context.Context, string, ...string) ([]byte, error) { return nil, errors.New("no systemctl") }
if fs := unitFindings(context.Background(), env); len(fs) != 1 || fs[0].Key != "unit/systemctl" || fs[0].Severity != watchdog.Warning {
t.Errorf("without systemctl: %+v, want the one unit/systemctl warning", fs)
}
}
// quoteArgs writes args as an ExecStart= line does, each word in quotes so an
// empty one survives.
func quoteArgs(args []string) string {
q := make([]string, len(args))
for i, a := range args {
q[i] = `"` + a + `"`
}
return strings.Join(q, " ")
}
// doctorHost is a host with the installer's watchdog unit, whose API server
// and PostgreSQL are down.
func doctorHost(t *testing.T, cfg string, extraFlags string) (env doctorEnv, h *watchdogHost) {
t.Helper()
h = newWatchdogHost(t, cfg, nil)
unitDir := t.TempDir()
writeTestFile(t, filepath.Join(unitDir, "felis-watchdog.service"),
"[Service]\nType=oneshot\nExecStart=/usr/local/bin/felis watchdog "+quoteArgs(h.args)+
// Off this machine's disk, whose free space is not the test's.
` -disk-paths "/nonexistent-felis-doctor-test"`+extraFlags+"\n", 0o644)
writeTestFile(t, filepath.Join(unitDir, "felis-velocity.service"), "[Unit]\n", 0o644)
writeTestFile(t, filepath.Join(unitDir, "felis-offsite.timer"), "[Unit]\n", 0o644)
return doctorEnv{
unitDir: unitDir, now: time.Now(), host: "felis-test",
run: fakeSystemctl("felis-db-backup.service loaded failed failed Felis database backup\n",
map[string]string{"felis-velocity.service": "active", "felis-offsite.timer": "active"}),
}, h
}
func TestDoctorReportsByArea(t *testing.T) {
env, _ := doctorHost(t, testWatchdogConfig, "")
var out bytes.Buffer
code := runDoctor(context.Background(), env, &out)
got := out.String()
if code != 1 {
t.Errorf("exit %d, want 1 with problems found", code)
}
for _, want := range []string{
"felis doctor on felis-test at ",
"checks run as " + filepath.Join(env.unitDir, "felis-watchdog.service") + " runs them (config ",
"✓ configuration\n",
"✗ Kubernetes cluster\n critical kube-api: ",
"✗ PostgreSQL\n critical postgres: ",
"- game proxy: not checked, no -proxy-addr\n",
"- database backups: not checked, no -backup-dir\n",
"- off-site copy: not checked, not configured\n",
"- build scan database: not checked, builds do not scan against the registry's copy\n",
"- k3s certificates: not checked, no -k3s-cert-dirs\n",
"- node address: not checked, no -node-ip\n",
"✗ systemd units and timers\n critical unit/felis-db-backup.service: felis-db-backup.service failed\n" +
" → journalctl -u felis-db-backup.service -n 100 --no-pager; once fixed, sudo systemctl reset-failed felis-db-backup.service",
"! alerting\n warning alerts/relay: no [smtp] relay is configured",
"\n4 problem(s): 3 critical, 1 warning(s)\n",
} {
if !strings.Contains(got, want) {
t.Errorf("report lacks %q:\n%s", want, got)
}
}
if strings.Contains(got, "alerts/recipients") {
t.Errorf("reported no recipients although PostgreSQL, which names them, is down:\n%s", got)
}
if strings.Contains(got, "note: no heartbeat URL") {
t.Errorf("the host has a heartbeat URL:\n%s", got)
}
}
// A doctor run is only a look: whatever is due to be mailed stays due, no
// heartbeat is pinged, and the watchdog's state is left as it was.
func TestDoctorMailsPingsAndSavesNothing(t *testing.T) {
cfg := testWatchdogConfig + "[smtp]\nhost = \"smtp.example.com\"\nport = 587\nfrom = \"[email protected]\"\n"
env, h := doctorHost(t, cfg, "")
if err := watchdog.SaveState(h.statePath, duePostgres("cached-pw")); err != nil {
t.Fatal(err)
}
before, err := os.ReadFile(h.statePath)
if err != nil {
t.Fatal(err)
}
watchdogSender = h.rec.sender
defer func() { watchdogSender = smtpSender }()
var out bytes.Buffer
runDoctor(context.Background(), env, &out)
if !strings.Contains(out.String(), "critical postgres: ") {
t.Fatalf("the due PostgreSQL alert was not seen:\n%s", out.String())
}
if len(h.rec.sent) != 0 || len(h.rec.relays) != 0 {
t.Errorf("mailed %v", h.rec.sent)
}
if n := len(h.pings.pings); n != 0 {
t.Errorf("pinged the heartbeat %d times", n)
}
after, err := os.ReadFile(h.statePath)
if err != nil || !bytes.Equal(before, after) {
t.Errorf("the watchdog state changed (err %v)", err)
}
if _, err := os.Stat(h.fallbackPath); !errors.Is(err, os.ErrNotExist) {
t.Errorf("a fallback state was written: %v", err)
}
}
func TestDoctorNotes(t *testing.T) {
env, h := doctorHost(t, testWatchdogConfig, "")
if err := os.Remove(filepath.Join(h.dir, "watchdog-heartbeat-url")); err != nil {
t.Fatal(err)
}
until := env.now.Add(10 * time.Minute).Unix()
writeTestFile(t, filepath.Join(h.dir, "quiet"), strconv.FormatInt(until, 10)+"\n", 0o644)
var out bytes.Buffer
runDoctor(context.Background(), env, &out)
for _, want := range []string{
"\nnote: no heartbeat URL is set: ",
"\nnote: the watchdog mails nothing until " + time.Unix(until, 0).UTC().Format("2006-01-02 15:04 UTC") + " (" + filepath.Join(h.dir, "quiet") + ")",
} {
if !strings.Contains(out.String(), want) {
t.Errorf("report lacks %q:\n%s", want, out.String())
}
}
writeTestFile(t, filepath.Join(h.dir, "watchdog-heartbeat-url"), "not a url\n", 0o600)
out.Reset()
runDoctor(context.Background(), env, &out)
if !strings.Contains(out.String(), "warning watchdog/heartbeat: the heartbeat URL is unusable") {
t.Errorf("an unusable heartbeat URL is not reported:\n%s", out.String())
}
}
// Without the watchdog's unit the doctor still checks, with the watchdog's
// defaults, and says the host has no watchdog; a configuration that does not
// load leaves the checks that need it unchecked and the host's own checks on.
func TestDoctorWithoutUnitOrConfig(t *testing.T) {
env, _ := doctorHost(t, testWatchdogConfig, "")
if err := os.Remove(filepath.Join(env.unitDir, "felis-watchdog.service")); err != nil {
t.Fatal(err)
}
writeTestFile(t, filepath.Join(env.unitDir, "felis-watchdog.service"), "[Service]\nExecStart=/usr/local/bin/felis watchdog -config "+filepath.Join(t.TempDir(), "absent.toml")+"\n", 0o644)
env.run = fakeSystemctl("", map[string]string{"felis-velocity.service": "active", "felis-offsite.timer": "active"})
var out bytes.Buffer
if code := runDoctor(context.Background(), env, &out); code != 1 {
t.Errorf("exit %d, want 1", code)
}
for _, want := range []string{
"✗ configuration\n critical config: ",
"- Kubernetes cluster: not checked, the configuration did not load\n",
"- PostgreSQL: not checked, the configuration did not load\n",
"- disk space: not checked, the configuration did not load\n",
"✓ systemd units and timers\n",
"- alerting: not checked, the configuration did not load\n",
} {
if !strings.Contains(out.String(), want) {
t.Errorf("report lacks %q:\n%s", want, out.String())
}
}
if err := os.Remove(filepath.Join(env.unitDir, "felis-watchdog.service")); err != nil {
t.Fatal(err)
}
out.Reset()
runDoctor(context.Background(), env, &out)
if !strings.Contains(out.String(), "critical watchdog/unit: "+filepath.Join(env.unitDir, "felis-watchdog.service")+" is not installed") ||
!strings.Contains(out.String(), "checks run with the watchdog's defaults (config /etc/felis/felis.toml)") {
t.Errorf("a host without the watchdog's unit:\n%s", out.String())
}
unit := filepath.Join(env.unitDir, "felis-watchdog.service")
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis watchdog -no-such-flag x\n", 0o644)
out.Reset()
runDoctor(context.Background(), env, &out)
if !strings.Contains(out.String(), "critical watchdog/unit: cannot read the watchdog's settings: "+unit+": flag provided but not defined: -no-such-flag\n") ||
!strings.Contains(out.String(), "checks run with the watchdog's defaults (config /etc/felis/felis.toml)") {
t.Errorf("a watchdog unit this binary cannot read:\n%s", out.String())
}
}
func TestPrintDoctorReport(t *testing.T) {
var out bytes.Buffer
if code := printDoctorReport(&out, nil, map[string]string{"proxy": "no -proxy-addr"}, nil); code != 0 {
t.Errorf("exit %d with nothing found, want 0", code)
}
want := "✓ configuration\n✓ Kubernetes cluster\n✓ PostgreSQL\n- game proxy: not checked, no -proxy-addr\n✓ database backups\n✓ off-site copy\n" +
"✓ build scan database\n✓ disk space\n✓ memory\n✓ k3s certificates\n✓ node address\n✓ clock\n✓ systemd units and timers\n✓ alerting\n" +
"\nno problems found\n"
if out.String() != want {
t.Errorf("report:\n%s\nwant:\n%s", out.String(), want)
}
out.Reset()
code := printDoctorReport(&out, []watchdog.Finding{
{Key: "unit/systemctl", Severity: watchdog.Warning, SummaryEN: "systemctl list-units failed"},
{Key: "postgres", Severity: watchdog.Critical, SummaryEN: "PostgreSQL is down", Hint: "kubectl -n felis get pods"},
}, map[string]string{"postgres": "a skip loses to what was found"}, []string{"a note"})
if code != 1 {
t.Errorf("exit %d with problems, want 1", code)
}
for _, want := range []string{
"✗ PostgreSQL\n critical postgres: PostgreSQL is down\n → kubectl -n felis get pods\n✓ game proxy\n",
"! systemd units and timers\n warning unit/systemctl: systemctl list-units failed\n✓ alerting\n",
"✓ alerting\n\nnote: a note\n\n2 problem(s): 1 critical, 1 warning(s)\n",
} {
if !strings.Contains(out.String(), want) {
t.Errorf("report lacks %q:\n%s", want, out.String())
}
}
}
-1359
View File
File diff suppressed because it is too large. Load diff
-892
View File
@@ -1,892 +0,0 @@
package main
import (
"bytes"
"context"
"crypto/rand"
"crypto/rsa"
"crypto/tls"
"crypto/x509"
"crypto/x509/pkix"
"encoding/pem"
"errors"
"math/big"
"net"
"os"
"path/filepath"
"sort"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/platform"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
)
// installerTOML is felis.toml as deploy/bootstrap.sh write_felis_toml renders it,
// comments included: the domain move has to leave all of it but three values alone.
func installerTOML(root, dbHost string) string {
return `# Generated by deploy/bootstrap.sh; rerun the installer to regenerate. Hand edits are
# overwritten, except [smtp], [[auth_source]], [offsite], and the operator-owned
# [registry] / [archive] overrides, which carry forward.
[server]
listen = "0.0.0.0:8080"
root_domain = "` + root + `"
[database]
url = "postgres://felis:pw@` + dbHost + `:5432/felis?sslmode=disable"
[k8s]
namespace = "minecraft"
egress_mode = "nodeport"
[velocity]
# The two always-on system servers that felis setup provisions.
login_image = "felis/limbo:1"
lobby_image = "felis/lobby:1"
game_port = 25565
[registry]
url = "registry.felis.svc:5000"
build_namespace = "felis-build"
[archive]
store = "tarLocal"
local_path = "/var/lib/felis/archives"
[auth]
admin_hostname = "op.console.` + root + `"
panel_hostname = "console.` + root + `"
access_jwt_aud = "aud123"
# Third-party Yggdrasil sources federated by the hasJoined multiplexer.
[[auth_source]]
tag = "littleskin"
prefix = "LS"
url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined"
`
}
const linkPropsBody = `# Generated by deploy/bootstrap.sh — do not edit by hand; rerun the installer.
api-base-url=http://10.43.0.9:8081
service-token=TOKEN-NOT-TO-TOUCH
root-domain=old.example
panel-hostname=console.old.example
admin-hostname=op.console.old.example
login-server=login
lobby-server=lobby
`
var oldNames = domainNames{root: "old.example", panel: "console.old.example", admin: "op.console.old.example"}
var newNames = domainNames{root: "new.example", panel: "console.new.example", admin: "op.console.new.example"}
// domainRig models the host: the files, the cluster, and a felis-api, proxy and
// operator that pick up config the way the real ones do — the api serves what it
// read at its last restart, the proxy runs since its last restart, and the login
// pod carries the env its MinecraftServer had when it was last rolled.
type domainRig struct {
h domainHost
cl client.Client
out *bytes.Buffer
dir string
events []string
served domainNames
servedCert *x509.Certificate
proxySince time.Time
proxyLoaded bool
unresolved map[string]bool
// The fake operator: the CR env the login pod was last rolled to, and how
// many looks at the pod since the CR moved on.
rolledTo string
pending int
}
func (rig *domainRig) path(name string) string { return filepath.Join(rig.dir, name) }
func newDomainRig(t *testing.T) *domainRig {
t.Helper()
rig := &domainRig{out: &bytes.Buffer{}, dir: t.TempDir(), proxyLoaded: true, unresolved: map[string]bool{}}
writeTestFile(t, rig.path("felis.host.toml"), installerTOML("old.example", "127.0.0.1"), 0o600)
writeTestFile(t, rig.path("felis.pod.toml"), installerTOML("old.example", "10.211.55.6"), 0o600)
if err := os.Symlink(rig.path("felis.host.toml"), rig.path("felis.toml")); err != nil {
t.Fatal(err)
}
certPEM, keyPEM, err := issuePanelCert(oldNames, []net.IP{net.ParseIP("10.211.55.6")}, time.Now())
if err != nil {
t.Fatal(err)
}
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
writeTestFile(t, rig.path("panel-tls.key"), string(keyPEM), 0o600)
writeTestFile(t, rig.path("felis-link.properties"), linkPropsBody, 0o640)
// The proxy started before its config was last written, which is how it
// stands after an install.
rig.proxySince = time.Now().Add(-time.Hour)
pod := []byte(installerTOML("old.example", "10.211.55.6"))
login, err := loginSystemServer("felis/limbo:1", "minecraft", platform.InternalAPIBaseURL("felis"), oldNames.root, oldNames.panel)
if err != nil {
t.Fatal(err)
}
lobby, err := lobbySystemServer("felis/lobby:1", "minecraft")
if err != nil {
t.Fatal(err)
}
loginPod := &corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: naming.SystemLoginServer + "-0"},
Spec: corev1.PodSpec{Containers: []corev1.Container{{Name: "minecraft", Env: podEnv(login.Spec.Env)}}},
Status: corev1.PodStatus{Conditions: []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}}},
}
rig.rolledTo = envKey(login.Spec.Env)
rig.cl = fake.NewClientBuilder().WithScheme(haltScheme(t)).WithInterceptorFuncs(interceptor.Funcs{
Get: func(ctx context.Context, c client.WithWatch, key client.ObjectKey, obj client.Object, opts ...client.GetOption) error {
if key.Name == naming.SystemLoginServer+"-0" {
rig.operatorTick(t, c)
}
return c.Get(ctx, key, obj, opts...)
},
}).WithObjects(
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.ConfigSecretName},
Data: map[string][]byte{platform.ConfigSecretKey: pod}},
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: platform.ConfigSecretName},
Data: map[string][]byte{platform.ConfigSecretKey: pod}},
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.APITLSSecretName},
Type: corev1.SecretTypeTLS, Data: map[string][]byte{corev1.TLSCertKey: certPEM, corev1.TLSPrivateKeyKey: keyPEM}},
login, lobby, loginPod,
).Build()
rig.served = oldNames
rig.servedCert, _ = x509.ParseCertificate(mustCertDER(certPEM))
rig.h = domainHost{
paths: domainPaths{
hostTOML: rig.path("felis.host.toml"), podTOML: rig.path("felis.pod.toml"), defaultTOML: rig.path("felis.toml"),
cert: rig.path("panel-tls.crt"), key: rig.path("panel-tls.key"),
linkProps: rig.path("felis-link.properties"), tunnelConfig: rig.path("cloudflared.yml"),
},
cl: rig.cl,
controlNS: "felis",
rollAPI: func(ctx context.Context) error {
rig.events = append(rig.events, "roll-api")
rig.restartAPI(t)
return nil
},
restartUnit: func(_ context.Context, unit string) error {
rig.events = append(rig.events, "restart "+unit)
rig.proxySince = time.Now().Add(time.Second)
return nil
},
unitState: func(context.Context, string) (unitStatus, error) {
return unitStatus{loaded: rig.proxyLoaded, active: rig.proxyLoaded, since: rig.proxySince}, nil
},
liveAPI: func(context.Context, string) (liveAPIView, error) {
return liveAPIView{names: rig.served, cert: rig.servedCert}, nil
},
lookupHost: func(_ context.Context, host string) ([]string, error) {
if rig.unresolved[host] {
return nil, errors.New("no such host")
}
return []string{"10.211.55.6"}, nil
},
passkeys: func(context.Context) (int, int, error) { return 3, 2, nil },
now: time.Now,
out: rig.out,
loginWait: 50 * time.Millisecond,
pollEvery: time.Millisecond,
}
return rig
}
func podEnv(env []v1alpha1.EnvVar) []corev1.EnvVar {
out := make([]corev1.EnvVar, len(env))
for i, e := range env {
out[i] = corev1.EnvVar{Name: e.Name, Value: e.Value}
}
return out
}
// restartAPI makes the fake felis-api load the config and certificate its
// Secrets hold now.
func (rig *domainRig) restartAPI(t *testing.T) {
t.Helper()
var cfg, tlsSec corev1.Secret
ctx := context.Background()
if err := rig.cl.Get(ctx, client.ObjectKey{Namespace: "felis", Name: platform.ConfigSecretName}, &cfg); err != nil {
t.Fatal(err)
}
if err := rig.cl.Get(ctx, client.ObjectKey{Namespace: "felis", Name: platform.APITLSSecretName}, &tlsSec); err != nil {
t.Fatal(err)
}
names, err := tomlDomainNames(cfg.Data[platform.ConfigSecretKey])
if err != nil {
t.Fatal(err)
}
rig.served = names
rig.servedCert, err = x509.ParseCertificate(mustCertDER(tlsSec.Data[corev1.TLSCertKey]))
if err != nil {
t.Fatal(err)
}
}
// rollLoginPod is the operator restarting the login pod onto its CR's env.
// operatorTick is the operator as the login pod is watched: once the CR's env
// changes it takes operatorLag looks at the pod before the restarted pod
// carries the new env, the way a real rollout lags the CR.
func (rig *domainRig) operatorTick(t *testing.T, c client.Client) {
t.Helper()
ctx := context.Background()
var ms v1alpha1.MinecraftServer
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms); err != nil {
t.Fatal(err)
}
if envKey(ms.Spec.Env) == rig.rolledTo {
return
}
if rig.pending++; rig.pending < operatorLag {
return
}
var pod corev1.Pod
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer + "-0"}, &pod); err != nil {
t.Fatal(err)
}
pod.Spec.Containers[0].Env = podEnv(ms.Spec.Env)
if err := c.Update(ctx, &pod); err != nil {
t.Fatal(err)
}
rig.rolledTo, rig.pending = envKey(ms.Spec.Env), 0
}
const operatorLag = 3
func envKey(env []v1alpha1.EnvVar) string {
var b strings.Builder
for _, e := range env {
b.WriteString(e.Name + "=" + e.Value + "\n")
}
return b.String()
}
func (rig *domainRig) read(t *testing.T, name string) string {
t.Helper()
b, err := os.ReadFile(rig.path(name))
if err != nil {
t.Fatal(err)
}
return string(b)
}
func (rig *domainRig) secret(t *testing.T, ns, name string) map[string][]byte {
t.Helper()
var s corev1.Secret
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
t.Fatal(err)
}
return s.Data
}
func (rig *domainRig) crEnv(t *testing.T, name string) map[string]string {
t.Helper()
var ms v1alpha1.MinecraftServer
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
t.Fatal(err)
}
env := map[string]string{}
for _, e := range ms.Spec.Env {
env[e.Name] = e.Value
}
return env
}
// snapshot is every byte `set` may touch, for proving a refused or dry run
// touched none of it.
func (rig *domainRig) snapshot(t *testing.T) string {
t.Helper()
var b strings.Builder
entries, _ := os.ReadDir(rig.dir)
for _, e := range entries {
b.WriteString(e.Name() + "\n" + rig.read(t, e.Name()) + "\n")
}
var lines []string
for _, s := range []struct{ ns, name string }{{"felis", platform.ConfigSecretName}, {"minecraft", platform.ConfigSecretName}, {"felis", platform.APITLSSecretName}} {
for k, v := range rig.secret(t, s.ns, s.name) {
lines = append(lines, s.ns+"/"+s.name+"/"+k+"\n"+string(v))
}
}
for k, v := range rig.crEnv(t, naming.SystemLoginServer) {
lines = append(lines, "env "+k+"="+v)
}
sort.Strings(lines)
b.WriteString(strings.Join(lines, "\n"))
return b.String()
}
func TestNormalizeRootDomain(t *testing.T) {
for in, want := range map[string]string{
"Example.COM.": "example.com",
" mc.example.org ": "mc.example.org",
"10.211.55.6.nip.io": "10.211.55.6.nip.io",
"xn--bcher-kva.example": "xn--bcher-kva.example",
} {
got, err := normalizeRootDomain(in)
if err != nil || got != want {
t.Errorf("normalizeRootDomain(%q) = %q, %v; want %q", in, got, err, want)
}
}
for _, in := range []string{"", "https://example.com", "example.com:443", "example.com/x", "10.0.0.1", "::1",
"localhost", "a_b.example", "-a.example", "a-.example", strings.Repeat("a", 64) + ".example",
strings.Repeat("abcdefghi.", 25) + "example"} {
if got, err := normalizeRootDomain(in); err == nil {
t.Errorf("normalizeRootDomain(%q) = %q, want an error", in, got)
}
}
}
func TestPlanDomainChangeMovesDefaultsAndKeepsHandSetNames(t *testing.T) {
p := planDomainChange(oldNames, "new.example")
if p.to != newNames || p.customPanel || p.customAdmin {
t.Fatalf("defaults: %+v", p)
}
p = planDomainChange(domainNames{root: "old.example", panel: "play.corp.net", admin: "op.console.old.example"}, "new.example")
if p.to.panel != "play.corp.net" || !p.customPanel || p.to.admin != "op.console.new.example" || p.customAdmin {
t.Fatalf("hand-set panel: %+v", p)
}
p = planDomainChange(domainNames{root: "old.example", panel: "console.old.example", admin: "admin.corp.net"}, "new.example")
if p.to.admin != "admin.corp.net" || !p.customAdmin || p.to.panel != "console.new.example" {
t.Fatalf("hand-set admin: %+v", p)
}
}
func TestEditTOMLStringsChangesOnlyTheDomainLines(t *testing.T) {
orig := installerTOML("old.example", "127.0.0.1")
out, err := editTOMLStrings([]byte(orig), domainTOMLEdits(newNames))
if err != nil {
t.Fatal(err)
}
a, b := strings.Split(orig, "\n"), strings.Split(string(out), "\n")
if len(a) != len(b) {
t.Fatalf("line count %d → %d:\n%s", len(a), len(b), out)
}
changed := map[string]string{}
for i := range a {
if a[i] != b[i] {
changed[a[i]] = b[i]
}
}
want := map[string]string{
`root_domain = "old.example"`: `root_domain = "new.example"`,
`admin_hostname = "op.console.old.example"`: `admin_hostname = "op.console.new.example"`,
`panel_hostname = "console.old.example"`: `panel_hostname = "console.new.example"`,
}
if len(changed) != len(want) {
t.Fatalf("changed lines %v, want %v", changed, want)
}
for k, v := range want {
if changed[k] != v {
t.Errorf("%q → %q, want %q", k, changed[k], v)
}
}
}
func TestEditTOMLStringsAddsMissingKeysInTheirTable(t *testing.T) {
in := "[server]\nroot_domain = \"old.example\"\n\n[auth]\naccess_jwt_aud = \"x\"\n\n[smtp]\nhost = \"relay\"\n"
out, err := editTOMLStrings([]byte(in), domainTOMLEdits(newNames))
if err != nil {
t.Fatal(err)
}
want := "[server]\nroot_domain = \"new.example\"\n\n[auth]\naccess_jwt_aud = \"x\"\npanel_hostname = \"console.new.example\"\nadmin_hostname = \"op.console.new.example\"\n\n[smtp]\nhost = \"relay\"\n"
if string(out) != want {
t.Fatalf("got:\n%s\nwant:\n%s", out, want)
}
out, err = editTOMLStrings([]byte("[server]\nroot_domain = \"old.example\"\n\n[[auth_source]]\ntag = \"ls\"\n"), domainTOMLEdits(newNames))
if err != nil {
t.Fatal(err)
}
got, err := tomlDomainNames(out)
if err != nil || got != newNames || !strings.Contains(string(out), "[[auth_source]]\ntag = \"ls\"\n") {
t.Fatalf("no [auth] table: %v %+v\n%s", err, got, out)
}
}
func TestEditTOMLStringsRefusesWhatItCannotEditExactly(t *testing.T) {
for name, in := range map[string]string{
"multi-line value": "[server]\nroot_domain = \"\"\"\nold.example\"\"\"\n[auth]\n",
// The key's line sits inside another value; the real key is absent.
"key inside a string": "[server]\nmotd = \"\"\"\nroot_domain = \"old.example\"\n\"\"\"\n[auth]\n",
"quoted header": "[server]\nroot_domain = \"old.example\"\n[\"auth\"]\npanel_hostname = \"console.old.example\"\n",
"dotted key": "server.root_domain = \"old.example\"\n",
"inline table": "server = { root_domain = \"old.example\" }\n",
} {
if out, err := editTOMLStrings([]byte(in), domainTOMLEdits(newNames)); err == nil {
t.Errorf("%s: edited instead of refusing:\n%s", name, out)
}
}
}
// caSignedCert is an operator's certificate from their own CA; it names
// localhost too, so only the issuer tells it apart from the installer's.
func caSignedCert(t *testing.T, hosts ...string) (certPEM, keyPEM []byte) {
t.Helper()
caKey, _ := rsa.GenerateKey(rand.Reader, 2048)
ca := &x509.Certificate{SerialNumber: big.NewInt(1), Subject: pkix.Name{CommonName: "Corp CA"}, IsCA: true,
BasicConstraintsValid: true, KeyUsage: x509.KeyUsageCertSign, NotBefore: time.Now().Add(-time.Hour), NotAfter: time.Now().Add(time.Hour)}
caDER, err := x509.CreateCertificate(rand.Reader, ca, ca, &caKey.PublicKey, caKey)
if err != nil {
t.Fatal(err)
}
caCert, _ := x509.ParseCertificate(caDER)
key, _ := rsa.GenerateKey(rand.Reader, 2048)
leaf := &x509.Certificate{SerialNumber: big.NewInt(2), Subject: pkix.Name{CommonName: hosts[0]},
DNSNames: append(hosts, "localhost"), IPAddresses: []net.IP{net.IPv4(127, 0, 0, 1)},
NotBefore: time.Now().Add(-time.Hour), NotAfter: time.Now().Add(time.Hour)}
der, err := x509.CreateCertificate(rand.Reader, leaf, caCert, &key.PublicKey, caKey)
if err != nil {
t.Fatal(err)
}
pk, _ := x509.MarshalPKCS8PrivateKey(key)
return pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: der}), pem.EncodeToMemory(&pem.Block{Type: "PRIVATE KEY", Bytes: pk})
}
func TestFelisIssuedCert(t *testing.T) {
mine, _, err := issuePanelCert(oldNames, nil, time.Now())
if err != nil {
t.Fatal(err)
}
c, _ := x509.ParseCertificate(mustCertDER(mine))
if !felisIssuedCert(c) {
t.Error("the installer's kind of certificate is not recognised as Felis-issued")
}
theirs, _ := caSignedCert(t, "console.old.example")
c, _ = x509.ParseCertificate(mustCertDER(theirs))
if felisIssuedCert(c) {
t.Error("a CA-signed certificate is taken for Felis-issued")
}
// Self-signed but without one of the installer's localhost names: someone
// else's.
key, _ := rsa.GenerateKey(rand.Reader, 2048)
for name, self := range map[string]*x509.Certificate{
"no localhost": {DNSNames: []string{"console.old.example"}, IPAddresses: []net.IP{net.IPv4(127, 0, 0, 1)}},
"no 127.0.0.1": {DNSNames: []string{"console.old.example", "localhost"}},
} {
self.SerialNumber, self.Subject = big.NewInt(3), pkix.Name{CommonName: "x"}
self.NotBefore, self.NotAfter = time.Now().Add(-time.Hour), time.Now().Add(time.Hour)
der, _ := x509.CreateCertificate(rand.Reader, self, self, &key.PublicKey, key)
c, _ = x509.ParseCertificate(der)
if felisIssuedCert(c) {
t.Errorf("%s: a self-signed certificate is taken for Felis-issued", name)
}
}
}
func TestIssuePanelCertIsTheInstallersShape(t *testing.T) {
now := time.Now()
certPEM, keyPEM, err := issuePanelCert(newNames, []net.IP{net.ParseIP("10.211.55.6"), net.IPv4(127, 0, 0, 1)}, now)
if err != nil {
t.Fatal(err)
}
if _, err := tls.X509KeyPair(certPEM, keyPEM); err != nil {
t.Fatalf("key does not match the certificate: %v", err)
}
if b, _ := pem.Decode(keyPEM); b == nil || b.Type != "PRIVATE KEY" {
t.Fatalf("key is not PKCS#8 PEM like openssl writes")
}
c, _ := x509.ParseCertificate(mustCertDER(certPEM))
if !certCovers(c, newNames.panel, newNames.admin, "localhost") || !felisIssuedCert(c) {
t.Fatalf("names %v", c.DNSNames)
}
if len(c.IPAddresses) != 2 || !c.IPAddresses[1].Equal(net.ParseIP("10.211.55.6")) {
t.Fatalf("addresses %v, want 127.0.0.1 and the node address once each", c.IPAddresses)
}
if c.Subject.CommonName != newNames.admin || c.NotAfter.Sub(now) < 824*24*time.Hour || c.NotAfter.Sub(now) > 826*24*time.Hour {
t.Fatalf("CN %q, valid until %s", c.Subject.CommonName, c.NotAfter)
}
if len(c.ExtKeyUsage) != 1 || c.ExtKeyUsage[0] != x509.ExtKeyUsageServerAuth || c.IsCA {
t.Fatalf("usage %v, CA %v", c.ExtKeyUsage, c.IsCA)
}
}
func TestDomainSetMovesEverySurface(t *testing.T) {
rig := newDomainRig(t)
hostBefore := rig.read(t, "felis.host.toml")
code, err := rig.h.set(context.Background(), "New.Example", true)
if err != nil || code != 0 {
t.Fatalf("set = %d, %v\n%s", code, err, rig.out)
}
// The configs: the three values moved, everything else — comments, the
// database host of each copy, the Access audience — is as it was.
host := rig.read(t, "felis.host.toml")
if want := strings.NewReplacer("old.example", "new.example").Replace(hostBefore); host != want {
t.Fatalf("host toml:\n%s", host)
}
pod := rig.read(t, "felis.pod.toml")
if got, _ := tomlDomainNames([]byte(pod)); got != newNames || !strings.Contains(pod, "@10.211.55.6:5432") {
t.Fatalf("pod toml:\n%s", pod)
}
if link, err := os.Readlink(rig.path("felis.toml")); err != nil || link != rig.path("felis.host.toml") {
t.Fatalf("felis.toml is no longer the link to the host copy: %q %v", link, err)
}
if st, _ := os.Stat(rig.path("felis.host.toml")); st.Mode().Perm() != 0o600 {
t.Fatalf("host toml mode %v", st.Mode().Perm())
}
// The certificate: reissued for the new names, the node address kept, the old
// pair beside it.
c, err := readCertFile(rig.path("panel-tls.crt"))
if err != nil || !certCovers(c, newNames.panel, newNames.admin) || !felisIssuedCert(c) {
t.Fatalf("certificate: %v %v", err, c.DNSNames)
}
if !c.IPAddresses[len(c.IPAddresses)-1].Equal(net.ParseIP("10.211.55.6")) {
t.Fatalf("addresses %v", c.IPAddresses)
}
if _, err := tls.LoadX509KeyPair(rig.path("panel-tls.crt"), rig.path("panel-tls.key")); err != nil {
t.Fatalf("new pair: %v", err)
}
if st, _ := os.Stat(rig.path("panel-tls.key")); st.Mode().Perm() != 0o600 {
t.Fatalf("key mode %v", st.Mode().Perm())
}
backups, _ := filepath.Glob(rig.path("panel-tls.*.pre-domain-*"))
if len(backups) != 2 {
t.Fatalf("old pair kept as %v", backups)
}
for _, b := range backups {
if st, _ := os.Stat(b); strings.Contains(b, ".key.") && st.Mode().Perm() != 0o600 {
t.Fatalf("kept key %s has mode %v", b, st.Mode().Perm())
}
old, _ := os.ReadFile(b)
if strings.Contains(b, ".crt.") {
oc, _ := x509.ParseCertificate(mustCertDER(old))
if oc == nil || !certCovers(oc, oldNames.panel) {
t.Fatalf("kept certificate is not the old one")
}
}
}
// The Secrets carry the files.
for _, ns := range []string{"felis", "minecraft"} {
if got := rig.secret(t, ns, platform.ConfigSecretName)[platform.ConfigSecretKey]; string(got) != pod {
t.Fatalf("%s/felis-config is not felis.pod.toml", ns)
}
}
tlsData := rig.secret(t, "felis", platform.APITLSSecretName)
if string(tlsData[corev1.TLSCertKey]) != rig.read(t, "panel-tls.crt") || string(tlsData[corev1.TLSPrivateKeyKey]) != rig.read(t, "panel-tls.key") {
t.Fatal("felis-api-tls does not hold the new pair")
}
// The login gate's env moved and nothing else did.
env := rig.crEnv(t, naming.SystemLoginServer)
if env[envRootDomain] != newNames.root || env[envPanelHostname] != newNames.panel || env[envAPIBaseURL] != platform.InternalAPIBaseURL("felis") {
t.Fatalf("login env %v", env)
}
// The proxy's file: the three keys moved, its token and mode did not.
props := rig.read(t, "felis-link.properties")
if want := strings.NewReplacer("old.example", "new.example").Replace(linkPropsBody); props != want {
t.Fatalf("felis-link.properties:\n%s", props)
}
if st, _ := os.Stat(rig.path("felis-link.properties")); st.Mode().Perm() != 0o640 {
t.Fatalf("felis-link.properties mode %v", st.Mode().Perm())
}
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
t.Fatalf("events %v", rig.events)
}
if !strings.Contains(rig.out.String(), "Every surface is on new.example.") {
t.Fatalf("the closing check did not pass:\n%s", rig.out)
}
}
func TestDomainSetWithoutYesChangesNothing(t *testing.T) {
rig := newDomainRig(t)
before := rig.snapshot(t)
code, err := rig.h.set(context.Background(), "new.example", false)
if err != nil || code != 0 {
t.Fatalf("set = %d, %v", code, err)
}
if rig.snapshot(t) != before || len(rig.events) != 0 {
t.Fatalf("a dry run changed something (events %v)", rig.events)
}
out := rig.out.String()
for _, want := range []string{
"console.old.example → console.new.example",
"3 passkey(s) of 2 user(s) are bound to console.old.example",
"does not cover op.console.new.example",
"No [smtp] relay is configured",
"sudo felis domain set -yes new.example",
} {
if !strings.Contains(out, want) {
t.Errorf("plan lacks %q:\n%s", want, out)
}
}
}
func TestDomainSetKeepsAHandSetPanelHostname(t *testing.T) {
rig := newDomainRig(t)
for _, f := range []string{"felis.host.toml", "felis.pod.toml"} {
writeTestFile(t, rig.path(f), strings.Replace(rig.read(t, f), `panel_hostname = "console.old.example"`, `panel_hostname = "play.corp.net"`, 1), 0o600)
}
code, err := rig.h.set(context.Background(), "new.example", true)
if err != nil {
t.Fatalf("set: %v\n%s", err, rig.out)
}
got, _ := tomlDomainNames([]byte(rig.read(t, "felis.host.toml")))
if got != (domainNames{root: "new.example", panel: "play.corp.net", admin: "op.console.new.example"}) {
t.Fatalf("names %+v", got)
}
c, _ := readCertFile(rig.path("panel-tls.crt"))
if !certCovers(c, "play.corp.net", "op.console.new.example") {
t.Fatalf("certificate names %v", c.DNSNames)
}
if env := rig.crEnv(t, naming.SystemLoginServer); env[envPanelHostname] != "play.corp.net" {
t.Fatalf("login env %v", env)
}
if code != 0 || !strings.Contains(rig.out.String(), "play.corp.net (set by hand, kept") {
t.Fatalf("code %d:\n%s", code, rig.out)
}
}
func TestDomainSetRefusesAnOperatorCertificateForOtherNames(t *testing.T) {
rig := newDomainRig(t)
certPEM, keyPEM := caSignedCert(t, "console.old.example", "op.console.old.example")
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
writeTestFile(t, rig.path("panel-tls.key"), string(keyPEM), 0o600)
before := rig.snapshot(t)
if _, err := rig.h.set(context.Background(), "new.example", true); err == nil || !strings.Contains(err.Error(), "not issued by Felis") {
t.Fatalf("err = %v", err)
}
if rig.snapshot(t) != before || len(rig.events) != 0 {
t.Fatal("a refused move changed something")
}
// The operator's certificate for the new names is kept as it is.
certPEM, keyPEM = caSignedCert(t, "console.new.example", "op.console.new.example")
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
writeTestFile(t, rig.path("panel-tls.key"), string(keyPEM), 0o600)
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
t.Fatalf("set: %v", err)
}
if rig.read(t, "panel-tls.crt") != string(certPEM) {
t.Fatal("the operator's certificate was replaced")
}
}
func TestDomainSetRefusesAConfigItCannotEdit(t *testing.T) {
rig := newDomainRig(t)
writeTestFile(t, rig.path("felis.pod.toml"), strings.Replace(rig.read(t, "felis.pod.toml"), "[auth]", "[\"auth\"]", 1), 0o600)
before := rig.snapshot(t)
if _, err := rig.h.set(context.Background(), "new.example", true); err == nil || !strings.Contains(err.Error(), "by hand") {
t.Fatalf("err = %v", err)
}
if rig.snapshot(t) != before || len(rig.events) != 0 {
t.Fatal("a refused move changed something")
}
}
func TestDomainSetAgainOnlyConvergesWhatIsBehind(t *testing.T) {
rig := newDomainRig(t)
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
t.Fatal(err)
}
rig.events, rig.out = nil, &bytes.Buffer{}
rig.h.out = rig.out
before := rig.snapshot(t)
code, err := rig.h.set(context.Background(), "new.example", true)
if err != nil || code != 0 {
t.Fatalf("second set = %d, %v\n%s", code, err, rig.out)
}
if len(rig.events) != 0 || rig.snapshot(t) != before {
t.Fatalf("a converged install was touched again: %v\n%s", rig.events, rig.out)
}
// A proxy that was not restarted after the move is restarted by a re-run, and
// an api still on the old config is rolled.
rig.proxySince = time.Now().Add(-time.Hour)
rig.served = oldNames
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
t.Fatal(err)
}
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
t.Fatalf("events %v", rig.events)
}
// An api on the new names that still presents the old certificate is rolled.
rig.events = nil
oldCert, _, _ := issuePanelCert(oldNames, nil, time.Now())
rig.servedCert, _ = x509.ParseCertificate(mustCertDER(oldCert))
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
t.Fatal(err)
}
if strings.Join(rig.events, ",") != "roll-api" {
t.Fatalf("events %v", rig.events)
}
}
func TestDomainSetRefusesALoginServerItDoesNotOwn(t *testing.T) {
rig := newDomainRig(t)
var ms v1alpha1.MinecraftServer
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms); err != nil {
t.Fatal(err)
}
delete(ms.Labels, v1alpha1.LabelSystemRole)
if err := rig.cl.Update(context.Background(), &ms); err != nil {
t.Fatal(err)
}
if _, err := rig.h.set(context.Background(), "new.example", true); err == nil || !strings.Contains(err.Error(), "system role") {
t.Fatalf("err = %v", err)
}
if env := rig.crEnv(t, naming.SystemLoginServer); env[envRootDomain] != oldNames.root {
t.Fatalf("a server not marked as the login gate was changed: %v", env)
}
}
func TestDomainCheckNamesTheSurfaceThatIsBehind(t *testing.T) {
cases := []struct {
surface string
breakIt func(t *testing.T, rig *domainRig)
}{
{"felis.pod.toml", func(t *testing.T, rig *domainRig) {
writeTestFile(t, rig.path("felis.pod.toml"), installerTOML("old.example", "10.211.55.6"), 0o600)
}},
{"Secret minecraft/felis-config", func(t *testing.T, rig *domainRig) {
rig.putSecret(t, "minecraft", platform.ConfigSecretName, platform.ConfigSecretKey, installerTOML("old.example", "x"))
}},
{"Secret felis/felis-config", func(t *testing.T, rig *domainRig) {
rig.putSecret(t, "felis", platform.ConfigSecretName, platform.ConfigSecretKey, installerTOML("old.example", "x"))
}},
{"panel certificate", func(t *testing.T, rig *domainRig) {
// The Secret follows the file, so only the certificate's names are wrong.
certPEM, _, _ := issuePanelCert(oldNames, nil, time.Now())
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
rig.putSecret(t, "felis", platform.APITLSSecretName, corev1.TLSCertKey, string(certPEM))
}},
{"Secret felis/felis-api-tls", func(t *testing.T, rig *domainRig) {
certPEM, _, _ := issuePanelCert(newNames, nil, time.Now())
rig.putSecret(t, "felis", platform.APITLSSecretName, corev1.TLSCertKey, string(certPEM))
}},
{"felis-api", func(t *testing.T, rig *domainRig) { rig.served = oldNames }},
{"felis-api", func(t *testing.T, rig *domainRig) {
certPEM, _, _ := issuePanelCert(oldNames, nil, time.Now())
rig.servedCert, _ = x509.ParseCertificate(mustCertDER(certPEM))
}},
{"proxy", func(t *testing.T, rig *domainRig) {
writeTestFile(t, rig.path("felis-link.properties"), linkPropsBody, 0o640)
rig.proxySince = time.Now().Add(time.Hour)
}},
{"proxy", func(t *testing.T, rig *domainRig) { rig.proxySince = time.Now().Add(-time.Hour) }},
{"login gate", func(t *testing.T, rig *domainRig) {
var ms v1alpha1.MinecraftServer
_ = rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms)
for i := range ms.Spec.Env {
if ms.Spec.Env[i].Name == envRootDomain {
ms.Spec.Env[i].Value = "old.example"
}
}
if err := rig.cl.Update(context.Background(), &ms); err != nil {
t.Fatal(err)
}
}},
{"login gate", func(t *testing.T, rig *domainRig) { rig.setPodEnv(t, envRootDomain, "old.example") }},
{"login gate", func(t *testing.T, rig *domainRig) { rig.setPodEnv(t, envPanelHostname, "console.old.example") }},
{"Cloudflare tunnel", func(t *testing.T, rig *domainRig) {
writeTestFile(t, rig.path("cloudflared.yml"), "tunnel: abc\ningress:\n- hostname: console.new.example\n service: https://127.0.0.1:30443\n- hostname: op.console.old.example\n service: https://127.0.0.1:30443\n- service: http_status:404\n", 0o644)
}},
}
for _, tc := range cases {
rig := newDomainRig(t)
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
t.Fatal(err)
}
rig.out.Reset()
tc.breakIt(t, rig)
if code := rig.h.check(context.Background()); code != 1 {
t.Errorf("%s behind: check = %d\n%s", tc.surface, code, rig.out)
continue
}
var failed []string
for _, ln := range strings.Split(rig.out.String(), "\n") {
if strings.HasPrefix(ln, " FAIL ") {
failed = append(failed, ln)
}
}
if len(failed) != 1 || !strings.Contains(failed[0], tc.surface) {
t.Errorf("%s behind: FAIL lines %q", tc.surface, failed)
}
}
}
func TestDomainCheckPassesAConvergedInstallAndWarnsOnDNS(t *testing.T) {
rig := newDomainRig(t)
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
t.Fatal(err)
}
writeTestFile(t, rig.path("cloudflared.yml"), "tunnel: abc\ningress:\n- hostname: console.new.example\n service: https://127.0.0.1:30443\n- hostname: op.console.new.example\n service: https://127.0.0.1:30443\n- service: http_status:404\n", 0o644)
rig.unresolved["op.console.new.example"] = true
rig.unresolved[dnsProbeLabel+".new.example"] = true
rig.out.Reset()
if code := rig.h.check(context.Background()); code != 0 {
t.Fatalf("check = %d\n%s", code, rig.out)
}
out := rig.out.String()
for _, want := range []string{" ok Cloudflare tunnel: routes both names", " warn DNS: op.console.new.example, *.new.example do not resolve from this host\n"} {
if !strings.Contains(out, want) {
t.Errorf("check lacks %q:\n%s", want, out)
}
}
if strings.Contains(out, "TOKEN-NOT-TO-TOUCH") {
t.Fatal("check printed the proxy's service token")
}
// A zone with only the wildcard: the admin name alone is missing, and why is said.
delete(rig.unresolved, dnsProbeLabel+".new.example")
rig.out.Reset()
rig.h.check(context.Background())
if want := " warn DNS: op.console.new.example does not resolve from this host: the *.new.example wildcard does not cover op.console.new.example, which needs its own record\n"; !strings.Contains(rig.out.String(), want) {
t.Errorf("check lacks %q:\n%s", want, rig.out)
}
}
func (rig *domainRig) setPodEnv(t *testing.T, name, value string) {
t.Helper()
var pod corev1.Pod
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "login-0"}, &pod); err != nil {
t.Fatal(err)
}
for i, e := range pod.Spec.Containers[0].Env {
if e.Name == name {
pod.Spec.Containers[0].Env[i].Value = value
}
}
if err := rig.cl.Update(context.Background(), &pod); err != nil {
t.Fatal(err)
}
}
func (rig *domainRig) putSecret(t *testing.T, ns, name, key, val string) {
t.Helper()
var s corev1.Secret
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
t.Fatal(err)
}
s.Data[key] = []byte(val)
if err := rig.cl.Update(context.Background(), &s); err != nil {
t.Fatal(err)
}
}
func TestParseUnitShow(t *testing.T) {
st := parseUnitShow("LoadState=loaded\nActiveState=active\nActiveEnterTimestamp=@1790000000\n")
if !st.loaded || !st.active || !st.since.Equal(time.Unix(1790000000, 0)) {
t.Fatalf("%+v", st)
}
st = parseUnitShow("LoadState=not-found\nActiveState=inactive\nActiveEnterTimestamp=\n")
if st.loaded || st.active || !st.since.IsZero() {
t.Fatalf("%+v", st)
}
}
+11 -22
View File
@@ -17,28 +17,23 @@ var (
egressPollInterval = 200 * time.Millisecond
)
// cmdEgressGate is the first initContainer of every build pod and the last of
// every game server pod. A pod's NetworkPolicy is programmed asynchronously after
// the pod starts (live on k3s: a build-labelled pod reached the internet and the
// Kubernetes API for its first ~0.7 s, a server-labelled one felis-api's internal
// face on its first request), so the gate dials a destination the policy denies
// until it stops answering, and only then lets the pod's next container, the
// untrusted Dockerfile or server image, start.
// cmdEgressGate is the first initContainer of every build pod. The pod's
// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s:
// a build-labelled pod reached the internet and the Kubernetes API for its first
// ~0.7 s), so the gate dials a destination the policy denies until it stops
// answering, and only then lets the pod's next container, eventually the
// untrusted Dockerfile, start.
//
// The default probe is the Kubernetes API Service, which the kubelet names in
// every pod's environment and neither policy admits. A probe that still answers
// after --wait means the policy is not enforced at all (a CNI without
// every pod's environment and the build policy never admits. A probe that still
// answers after --wait means the policy is not enforced at all (a CNI without
// NetworkPolicy support, or k3s run with --disable-network-policy), and the
// build fails closed. A server passes --fail-open: an operator's
// --server-egress-allow-cidr may cover the node the API Service leads to, so a
// probe that keeps answering does not prove the fence is missing, and by then
// the policy has had --wait to land.
// build fails closed.
func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError)
fs.SetOutput(stderr)
probe := fs.String("probe", "", "host:port the pod's NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the gate gives up")
failOpen := fs.Bool("fail-open", false, "when --wait runs out, warn and let the pod go on instead of refusing it")
probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused")
if err := fs.Parse(args); err != nil {
return 2
}
@@ -61,12 +56,6 @@ func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
}
_ = conn.Close()
if time.Since(start) >= *wait {
if *failOpen {
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s; starting anyway. Either this namespace's "+
"NetworkPolicy is not enforced (a CNI without NetworkPolicy support, or k3s started with "+
"--disable-network-policy), or an allowed CIDR admits the address behind it\n", *probe, *wait)
return 0
}
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+
"(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n",
*probe, *wait)
-34
View File
@@ -75,40 +75,6 @@ func TestEgressGateRefusesAnOpenNetwork(t *testing.T) {
}
}
// A server's gate waits out --wait all the same, then lets the pod start with a
// warning in its log.
func TestEgressGateFailOpenWaitsThenWarns(t *testing.T) {
shrinkEgressGate(t)
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
defer ln.Close()
go func() {
for {
c, err := ln.Accept()
if err != nil {
return
}
_ = c.Close()
}
}()
var out, errb bytes.Buffer
start := time.Now()
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "150ms", "--fail-open"}, &out, &errb); code != 0 {
t.Fatalf("exit %d, want 0: %s", code, errb.String())
}
if waited := time.Since(start); waited < 150*time.Millisecond {
t.Errorf("gave up after %s, before --wait ran out", waited)
}
if !strings.Contains(errb.String(), ln.Addr().String()+" still answers after 150ms; starting anyway") {
t.Errorf("stderr = %q", errb.String())
}
if out.Len() != 0 {
t.Errorf("stdout = %q, want nothing: the lock was never seen", out.String())
}
}
func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) {
shrinkEgressGate(t)
t.Setenv("KUBERNETES_SERVICE_HOST", "")
-302
View File
@@ -1,302 +0,0 @@
package main
import (
"context"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"encoding/json"
"errors"
"flag"
"fmt"
"hash"
"io"
"io/fs"
"net/http"
"os"
"os/signal"
"strconv"
"strings"
"syscall"
"time"
"felis.lolicon.best/internal/backup"
"felis.lolicon.best/internal/fileedit"
"felis.lolicon.best/internal/worldexport"
)
// cmdExport is the in-Pod entrypoint the export Job runs. internal/worldexport
// renders a Pod whose command is `/usr/local/bin/felis export`. It archives the
// mounted world, re-streams one archive from the mounted backup store, or sends
// one file or folder of the world, PUTs it to felis-api's internal face, and
// exits once felis-api says the owner's browser got all of it. It is NOT a
// user-facing command and is never invoked by hand.
//
// Like cmdRestore it holds no database credentials and never calls config.Load:
// felis-api made every decision (who may download what, that the server is
// stopped, which archive) before the Job existed. Its input is the flags below
// plus the one-time upload token in the environment, which opens this one
// export and nothing else.
//
// Whatever leaves goes through the same guards as the file editor
// (fileedit.Guard): the proxy forwarding secret, which every server on the
// install shares, never leaves, and server.properties leaves with its RCON
// password redacted. A backup is stored with both, since a restore must bring
// the world back whole, so it is filtered on the way out rather than handed
// over as stored.
//
// Exit status: 0 once felis-api answers 204 (the download completed), 1 when
// the export could not be read or handed over, a backup failed its digest
// check, or felis-api refused it (the browser never came or left early), 2 on
// bad flags. The last stderr line reaches the export's status and the jobs list.
func cmdExport(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("export", flag.ContinueOnError)
fs.SetOutput(stderr)
mode := fs.String("mode", "", "what to export: world, backup or files")
server := fs.String("server", "", "server name being exported (for logging)")
target := fs.String("target-url", "", "felis-api URL to PUT the export to")
ref := fs.String("ref", "", "backup only: absolute path to the archive on the backup mount")
backupRoot := fs.String("backup-root", "/backups", "backup only: mount path of the backup PVC (the ref must resolve under it)")
sum := fs.String("sha256", "", "backup only: the sha256 recorded when the archive was written; a mismatch fails the export before its end is sent")
worldsRoot := fs.String("worlds-root", "/world", "world and files: mount path of the world PVC")
path := fs.String("path", "", "files only: the file or folder to send, relative to the world root")
dir := fs.Bool("dir", false, "files only: the path is a folder, sent as a zip")
if err := fs.Parse(args); err != nil {
return 2
}
token := os.Getenv(worldexport.TokenEnv)
if *target == "" || token == "" {
fmt.Fprintf(stderr, "felis export: --target-url and %s are required\n", worldexport.TokenEnv)
return 2
}
limitHeapToCgroup()
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
var err error
switch *mode {
case worldexport.ModeBackup:
if *ref == "" {
fmt.Fprintln(stderr, "felis export: --ref is required for a backup")
return 2
}
err = exportBackup(ctx, *target, token, *ref, *backupRoot, *sum, stdout)
case worldexport.ModeWorld:
err = exportWorld(ctx, *target, token, *worldsRoot, stdout)
case worldexport.ModeFiles:
if *path == "" {
fmt.Fprintln(stderr, "felis export: --path is required for files")
return 2
}
err = exportFiles(ctx, *target, token, *worldsRoot, *path, *dir, stdout)
default:
fmt.Fprintf(stderr, "felis export: --mode must be %s, %s or %s\n", worldexport.ModeWorld, worldexport.ModeBackup, worldexport.ModeFiles)
return 2
}
if err != nil {
fmt.Fprintf(stderr, "felis export: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "felis export: server=%s mode=%s downloaded\n", *server, *mode)
return 0
}
// archiveType is the content type of a world or backup export.
const archiveType = "application/gzip"
// errBackupDigest fails a backup export whose stored archive no longer hashes
// to what was recorded when it was written.
var errBackupDigest = errors.New("the backup archive does not match the sha256 recorded when it was written")
// exportBackup re-streams one stored archive through the export guards
// (backup.FilterTarGz with archiveFilter). Its length changes on the way, so it
// goes chunked. With want set, the stored bytes are hashed as they are read,
// and FilterTarGz reads them to their end before it closes its own archive: a
// mismatch aborts the upload while what felis-api has passed on still lacks
// its end, so the browser never keeps a complete-looking corrupt file.
func exportBackup(ctx context.Context, target, token, ref, root, want string, stdout io.Writer) error {
// Defense in depth, as in cmdRestore: the ref comes from felis-api, but this
// process opens it, so it confirms the ref stays on the backup mount.
if !refWithinRoot(ref, root) {
return fmt.Errorf("ref %q is not under backup root %q", ref, root)
}
f, err := os.Open(ref)
if err != nil {
return err
}
defer f.Close()
var src io.Reader = f
if want != "" {
src = &digestReader{r: f, sum: sha256.New(), want: want}
}
var withheld []string
err = streamExport(ctx, target, token, archiveType, -1, func(w io.Writer) error {
var err error
withheld, err = backup.FilterTarGz(ctx, w, src, archiveFilter)
return err
})
if errors.Is(err, errBackupDigest) {
return errBackupDigest // the jobs list shows it as it is, not wrapped as a read error
}
reportWithheld(stdout, len(withheld))
return err
}
// exportWorld archives the world straight into the request body: nothing is
// staged, so a world bigger than the Pod's memory or any scratch disk exports
// the same.
func exportWorld(ctx context.Context, target, token, root string, stdout io.Writer) error {
r, err := os.OpenRoot(root)
if err != nil {
return err
}
guard := fileedit.NewGuard(r)
r.Close()
var skipped, withheld []string
err = streamExport(ctx, target, token, archiveType, -1, func(w io.Writer) error {
var err error
skipped, withheld, err = backup.WriteTarGz(ctx, w, root, worldFilter(guard))
return err
})
if len(skipped) > 0 {
fmt.Fprintf(stdout, "felis export: left out %d entries a tar cannot hold (symbolic links, devices, sockets)\n", len(skipped))
}
reportWithheld(stdout, len(withheld))
return err
}
// exportFiles sends one file or folder of the world (fileedit.OpenDownload): a
// file with its exact length, a folder as a zip made as it streams. dir is
// what the owner saw at path when they asked.
func exportFiles(ctx context.Context, target, token, root, path string, dir bool, stdout io.Writer) error {
d, err := fileedit.OpenDownload(root, path, dir)
if err != nil {
return err
}
defer d.Close()
err = streamExport(ctx, target, token, d.ContentType, d.Size, func(w io.Writer) error { return d.WriteTo(ctx, w) })
if d.Skipped > 0 {
fmt.Fprintf(stdout, "felis export: left out %d entries a zip does not carry (symbolic links, devices, sockets)\n", d.Skipped)
}
reportWithheld(stdout, d.Withheld)
return err
}
func reportWithheld(stdout io.Writer, n int) {
if n > 0 {
fmt.Fprintf(stdout, "felis export: left out %d files that hold platform secrets\n", n)
}
}
// worldFilter guards a live world by file identity, so a link to a guarded
// file under another name is caught as well.
func worldFilter(g fileedit.Guard) backup.Filter {
return func(_ string, info fs.FileInfo) (bool, func([]byte) []byte) {
return guardAction(g.Rule(info))
}
}
// archiveFilter guards a stored archive, which has only names.
func archiveFilter(name string, _ fs.FileInfo) (bool, func([]byte) []byte) {
return guardAction(fileedit.ArchiveRule(name))
}
func guardAction(withhold, redact bool) (bool, func([]byte) []byte) {
if redact {
return withhold, fileedit.RedactProps
}
return withhold, nil
}
// digestReader passes r through, hashing it, and turns r's EOF into
// errBackupDigest when the bytes do not hash to want.
type digestReader struct {
r io.Reader
sum hash.Hash
want string
}
func (d *digestReader) Read(p []byte) (int, error) {
n, err := d.r.Read(p)
d.sum.Write(p[:n])
if err == io.EOF && !strings.EqualFold(hex.EncodeToString(d.sum.Sum(nil)), d.want) {
return n, errBackupDigest
}
return n, err
}
// streamExport runs write straight into the body of the PUT, hashing it as it
// goes. Once write has finished, the SHA-256 of all it wrote rides the
// request's trailer (worldexport.DigestTrailer), and felis-api holds back the
// last bytes from the browser until what it received hashes the same. An error
// from write aborts the chunked body before the trailer, and felis-api then
// cuts the browser's download off rather than end it; that error is the one
// reported, since the PUT's own error only wraps it. When the PUT ends first,
// write is stopped.
func streamExport(ctx context.Context, target, token, contentType string, size int64, write func(io.Writer) error) error {
pr, pw := io.Pipe()
trailer := http.Header{worldexport.DigestTrailer: nil}
werr := make(chan error, 1)
go func() {
sum := sha256.New()
err := write(io.MultiWriter(pw, sum))
if err == nil {
// Set before the body ends: the transport reads the trailer once it
// has read the body to its end.
trailer.Set(worldexport.DigestTrailer, "sha-256=:"+base64.StdEncoding.EncodeToString(sum.Sum(nil))+":")
}
pw.CloseWithError(err)
werr <- err
}()
err := putExport(ctx, target, token, contentType, pr, size, trailer)
pr.CloseWithError(io.ErrClosedPipe)
if w := <-werr; w != nil && !errors.Is(w, io.ErrClosedPipe) {
return w
}
return err
}
// putExport PUTs the export to felis-api. There is no retry: the token opens
// the export once, so a second attempt could only be refused. Redirects are
// refused because the request carries the token and the internal face never
// redirects. felis-api answers only after the whole download, which the Job's
// activeDeadlineSeconds bounds, so the header timeout is a backstop for a
// wedged endpoint and not the real limit.
//
// The body always goes chunked, which is what lets it end with a trailer; a
// size the Job knows (-1 when it does not) goes as worldexport.LengthHeader in
// place of Content-Length.
func putExport(ctx context.Context, target, token, contentType string, body io.Reader, size int64, trailer http.Header) error {
req, err := http.NewRequestWithContext(ctx, http.MethodPut, target, body)
if err != nil {
return err
}
req.ContentLength = -1
req.Trailer = trailer
if size >= 0 {
req.Header.Set(worldexport.LengthHeader, strconv.FormatInt(size, 10))
}
req.Header.Set("Authorization", "Bearer "+token)
req.Header.Set("Content-Type", contentType)
client := &http.Client{
Transport: &http.Transport{ResponseHeaderTimeout: 10 * time.Minute},
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
}
resp, err := client.Do(req)
if err != nil {
return err
}
defer resp.Body.Close()
if resp.StatusCode == http.StatusNoContent {
return nil
}
var e struct {
Error struct {
Message string `json:"message"`
} `json:"error"`
}
if json.NewDecoder(io.LimitReader(resp.Body, 4<<10)).Decode(&e) == nil && e.Error.Message != "" {
return fmt.Errorf("felis-api answered %s: %s", resp.Status, e.Error.Message)
}
return fmt.Errorf("felis-api answered %s", resp.Status)
}
-532
View File
@@ -1,532 +0,0 @@
package main
import (
"archive/tar"
"archive/zip"
"bytes"
"compress/gzip"
"context"
"crypto/rand"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"errors"
"io"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"reflect"
"slices"
"strconv"
"strings"
"sync/atomic"
"testing"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/worldexport"
)
// exportReceiver stands in for felis-api's internal upload route: it records
// the PUT it gets (or the error reading it ended on) and answers with reply.
type exportReceiver struct {
srv *httptest.Server
hits atomic.Int32
req *http.Request
body []byte
readErr error
served chan struct{} // one send per request, once it is answered
}
func receiveExport(t *testing.T, reply func(w http.ResponseWriter)) *exportReceiver {
t.Helper()
rcv := &exportReceiver{served: make(chan struct{}, 1)}
rcv.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
rcv.hits.Add(1)
rcv.req = r
rcv.body, rcv.readErr = io.ReadAll(r.Body)
reply(w)
rcv.served <- struct{}{}
}))
t.Cleanup(rcv.srv.Close)
return rcv
}
func noContent(w http.ResponseWriter) { w.WriteHeader(http.StatusNoContent) }
// sentWhole fails unless the upload rcv got ended with the Content-Digest
// trailer of its own bytes, and declared length as its size (-1: none).
func sentWhole(t *testing.T, rcv *exportReceiver, length int64) {
t.Helper()
sum := sha256.Sum256(rcv.body)
want := "sha-256=:" + base64.StdEncoding.EncodeToString(sum[:]) + ":"
wantLength := ""
if length >= 0 {
wantLength = strconv.FormatInt(length, 10)
}
r := rcv.req
if rcv.readErr != nil || r.Trailer.Get(worldexport.DigestTrailer) != want || r.Header.Get(worldexport.LengthHeader) != wantLength ||
r.ContentLength != -1 || strings.Join(r.TransferEncoding, ",") != "chunked" {
t.Fatalf("upload read %v, trailer %v, %s %q, length %d, encoding %v; want trailer %q and %s %q, chunked",
rcv.readErr, r.Trailer, worldexport.LengthHeader, r.Header.Get(worldexport.LengthHeader), r.ContentLength, r.TransferEncoding,
want, worldexport.LengthHeader, wantLength)
}
}
func tarEntries(t *testing.T, archive []byte) map[string]string {
t.Helper()
gz, err := gzip.NewReader(bytes.NewReader(archive))
if err != nil {
t.Fatalf("not gzip: %v", err)
}
out := map[string]string{}
tr := tar.NewReader(gz)
for {
h, err := tr.Next()
if err == io.EOF {
return out
}
if err != nil {
t.Fatalf("tar: %v", err)
}
b, _ := io.ReadAll(tr)
out[h.Name] = string(b)
}
}
// writeTree writes name → body under root, making the folders on the way.
func writeTree(t *testing.T, root string, files map[string]string) {
t.Helper()
for name, body := range files {
p := filepath.Join(root, name)
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(p, []byte(body), 0o600); err != nil {
t.Fatal(err)
}
}
}
// The two secrets a world holds, and what server.properties reads as once
// redacted.
const (
secretProps = "motd=hi\nrcon.password=hunter2\n"
redactedProps = "motd=hi\nrcon.password=<redacted by felis>\n"
forwardingKey = "secret: aVeryRealForwardingKey\n"
)
// secretWorld is a world holding both secrets, with a hard link to the
// forwarding secret under a name nothing would guard by.
func secretWorld(t *testing.T) string {
t.Helper()
root := t.TempDir()
writeTree(t, root, map[string]string{
"server.properties": secretProps,
"config/paper-global.yml": forwardingKey,
"world/region/r.0.0.mca": "chunks",
})
if err := os.MkdirAll(filepath.Join(root, "plugins"), 0o755); err != nil {
t.Fatal(err)
}
if err := os.Link(filepath.Join(root, "config/paper-global.yml"), filepath.Join(root, "plugins/copy.yml")); err != nil {
t.Fatal(err)
}
return root
}
func TestCmdExportWorld(t *testing.T) {
root := secretWorld(t)
if err := os.Symlink("server.properties", filepath.Join(root, "props-link")); err != nil {
t.Fatal(err)
}
rcv := receiveExport(t, noContent)
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
code := cmdExport([]string{"--mode", "world", "--server", "survival", "--target-url", rcv.srv.URL + "/api/v1/internal/exports/ab",
"--worlds-root", root}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
r := rcv.req
if r.Method != http.MethodPut || r.URL.Path != "/api/v1/internal/exports/ab" || r.Header.Get("Authorization") != "Bearer tok" ||
r.Header.Get("Content-Type") != "application/gzip" || r.ContentLength != -1 || strings.Join(r.TransferEncoding, ",") != "chunked" {
t.Fatalf("request = %s %s, headers %v, length %d, encoding %v", r.Method, r.URL.Path, r.Header, r.ContentLength, r.TransferEncoding)
}
want := map[string]string{
"server.properties": redactedProps, "config/": "", "plugins/": "",
"world/": "", "world/region/": "", "world/region/r.0.0.mca": "chunks",
}
if got := tarEntries(t, rcv.body); !reflect.DeepEqual(got, want) {
t.Fatalf("archive holds %v\nwant %v", got, want)
}
sentWhole(t, rcv, -1)
want2 := "felis export: left out 1 entries a tar cannot hold (symbolic links, devices, sockets)\n" +
"felis export: left out 2 files that hold platform secrets\n" +
"felis export: server=survival mode=world downloaded\n"
if stdout.String() != want2 {
t.Errorf("stdout = %q, want %q", stdout.String(), want2)
}
}
// A world root that cannot be opened fails before anything reaches felis-api.
func TestCmdExportWorldUnreadable(t *testing.T) {
rcv := receiveExport(t, noContent)
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
code := cmdExport([]string{"--mode", "world", "--target-url", rcv.srv.URL, "--worlds-root", filepath.Join(t.TempDir(), "missing")}, &stdout, &stderr)
if code != 1 || rcv.hits.Load() != 0 {
t.Fatalf("exit %d with %d requests, want 1 and none", code, rcv.hits.Load())
}
}
// An export that fails part-way must never reach felis-api as a complete body:
// the chunked upload is cut off, so felis-api aborts the browser's download,
// and the failure itself is what the Job reports.
func TestStreamExportWriteErrorAbortsTheUpload(t *testing.T) {
broken := errors.New("disk read failed")
rcv := receiveExport(t, noContent)
err := streamExport(context.Background(), rcv.srv.URL, "tok", "application/gzip", -1, func(w io.Writer) error {
if _, err := w.Write(bytes.Repeat([]byte("x"), 100_000)); err != nil {
return err
}
return broken
})
if err != broken {
t.Fatalf("err = %v, want the write's own error, unwrapped", err)
}
select {
case <-rcv.served:
if rcv.readErr == nil {
t.Fatalf("felis-api read a complete %d-byte body from a failed export", len(rcv.body))
}
case <-time.After(5 * time.Second):
t.Fatal("the request never reached felis-api")
}
}
// When felis-api refuses first, its reason is reported, not the closed pipe
// that then stops the writer.
func TestStreamExportRefusalStopsTheWriter(t *testing.T) {
refusing := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
w.WriteHeader(http.StatusNotFound)
}))
defer refusing.Close()
stopped := make(chan error, 1)
err := streamExport(context.Background(), refusing.URL, "tok", "application/gzip", -1, func(w io.Writer) error {
for {
if _, err := w.Write(make([]byte, 32<<10)); err != nil {
stopped <- err
return err
}
}
})
if err == nil || err.Error() != "felis-api answered 404 Not Found" {
t.Fatalf("err = %v, want felis-api's answer", err)
}
if werr := <-stopped; !errors.Is(werr, io.ErrClosedPipe) {
t.Fatalf("the writer stopped on %v, want the closed pipe", werr)
}
// A PUT that never starts leaves no transport to close the body: the writer
// is still stopped, and the export fails rather than hangs.
done := make(chan error, 1)
go func() {
done <- streamExport(context.Background(), "http://[::1", "tok", "application/gzip", -1, func(w io.Writer) error {
_, err := w.Write([]byte("x"))
return err
})
}()
select {
case err := <-done:
if err == nil || !strings.Contains(err.Error(), "missing ']'") {
t.Fatalf("err = %v, want the bad URL", err)
}
case <-time.After(5 * time.Second):
t.Fatal("an export whose PUT never started hung")
}
}
// storedBackup writes, at path, a gzip+tar like one the backup store holds:
// the world whole, both secrets included, and a region file that does not
// compress. It returns the archive's sha256.
func storedBackup(t *testing.T, path string) string {
t.Helper()
region := make([]byte, 64<<10)
if _, err := rand.Read(region); err != nil {
t.Fatal(err)
}
var buf bytes.Buffer
zw := gzip.NewWriter(&buf)
tw := tar.NewWriter(zw)
for _, e := range []struct{ name, body string }{
{"server.properties", secretProps},
{"config/paper-global.yml", forwardingKey},
{"world/level.dat", "level"},
{"world/region/r.0.0.mca", string(region)},
} {
if err := tw.WriteHeader(&tar.Header{Name: e.name, Typeflag: tar.TypeReg, Mode: 0o600, Size: int64(len(e.body))}); err != nil {
t.Fatal(err)
}
if _, err := io.WriteString(tw, e.body); err != nil {
t.Fatal(err)
}
}
if err := tw.Close(); err != nil {
t.Fatal(err)
}
if err := zw.Close(); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(path, buf.Bytes(), 0o600); err != nil {
t.Fatal(err)
}
sum := sha256.Sum256(buf.Bytes())
return hex.EncodeToString(sum[:])
}
func TestCmdExportBackup(t *testing.T) {
root := t.TempDir()
ref := filepath.Join(root, "survival-1.tar.gz")
sum := storedBackup(t, ref)
stored, err := os.ReadFile(ref)
if err != nil {
t.Fatal(err)
}
region := tarEntries(t, stored)["world/region/r.0.0.mca"]
args := func(url, ref string) []string {
return []string{"--mode", "backup", "--server", "survival", "--target-url", url, "--ref", ref, "--backup-root", root}
}
for name, extra := range map[string][]string{
"no digest recorded": nil,
"recorded digest matches": {"--sha256", sum},
} {
t.Run(name+": re-streamed through the guards", func(t *testing.T) {
rcv := receiveExport(t, noContent)
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdExport(append(args(rcv.srv.URL, ref), extra...), &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
r := rcv.req
if r.ContentLength != -1 || r.Header.Get("Content-Type") != "application/gzip" || r.Header.Get("Authorization") != "Bearer tok" {
t.Fatalf("length %d, headers %v", r.ContentLength, r.Header)
}
want := map[string]string{"server.properties": redactedProps, "world/level.dat": "level", "world/region/r.0.0.mca": region}
if got := tarEntries(t, rcv.body); !reflect.DeepEqual(got, want) {
t.Fatalf("archive holds %d entries, want exactly the redacted properties, level.dat and the region file", len(got))
}
sentWhole(t, rcv, -1)
if want := "felis export: left out 1 files that hold platform secrets\nfelis export: server=survival mode=backup downloaded\n"; stdout.String() != want {
t.Errorf("stdout = %q, want %q", stdout.String(), want)
}
})
}
// The stored bytes are checked as they stream, and the archive the Job sends
// is only closed once they are all read: a mismatch cuts the upload off
// short of its end, so felis-api never passes on a complete-looking copy.
t.Run("a digest mismatch cuts the upload off before its end", func(t *testing.T) {
rcv := receiveExport(t, noContent)
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdExport(append(args(rcv.srv.URL, ref), "--sha256", strings.Repeat("ab", 32)), &stdout, &stderr); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
if want := "felis export: the backup archive does not match the sha256 recorded when it was written\n"; stderr.String() != want {
t.Fatalf("stderr = %q, want %q", stderr.String(), want)
}
select {
case <-rcv.served:
case <-time.After(5 * time.Second):
t.Fatal("the upload never reached felis-api")
}
if rcv.readErr == nil {
t.Fatalf("felis-api read a complete %d-byte body", len(rcv.body))
}
zr, err := gzip.NewReader(bytes.NewReader(rcv.body))
if err == nil {
_, err = io.ReadAll(zr)
}
if err == nil {
t.Fatal("what felis-api got is a complete archive")
}
})
t.Run("a ref outside the backup root is refused before any request", func(t *testing.T) {
outside := filepath.Join(t.TempDir(), "secret.tar.gz")
if err := os.WriteFile(outside, []byte("secret"), 0o600); err != nil {
t.Fatal(err)
}
rcv := receiveExport(t, noContent)
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdExport(args(rcv.srv.URL, outside), &stdout, &stderr); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
if n := rcv.hits.Load(); n != 0 {
t.Fatalf("felis-api got %d requests", n)
}
})
t.Run("a refusal exits 1 with felis-api's reason", func(t *testing.T) {
rcv := receiveExport(t, func(w http.ResponseWriter) {
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusGone)
io.WriteString(w, `{"error":{"code":"export_expired","message":"nobody opened the download"}}`)
})
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdExport(args(rcv.srv.URL, ref), &stdout, &stderr); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
if got := stderr.String(); got != "felis export: felis-api answered 410 Gone: nobody opened the download\n" {
t.Fatalf("stderr = %q", got)
}
})
// The request carries the token and the internal face never redirects, so a
// redirect is refused rather than followed with the token attached. A 302 or
// 303 is the one net/http would follow on its own (as a GET, and to the same
// host with the Authorization header still on it).
for _, status := range []int{http.StatusFound, http.StatusSeeOther, http.StatusTemporaryRedirect, http.StatusPermanentRedirect} {
t.Run("a redirect is not followed: "+strconv.Itoa(status), func(t *testing.T) {
elsewhere := receiveExport(t, noContent)
redirecting := httptest.NewServer(http.RedirectHandler(elsewhere.srv.URL, status))
defer redirecting.Close()
t.Setenv(worldexport.TokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdExport(args(redirecting.URL, ref), &stdout, &stderr); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
if n := elsewhere.hits.Load(); n != 0 {
t.Fatalf("the redirect target got %d requests", n)
}
if got, want := stderr.String(), "felis export: felis-api answered "+strconv.Itoa(status)+" "+http.StatusText(status)+"\n"; got != want {
t.Fatalf("stderr = %q, want %q", got, want)
}
})
}
}
func TestCmdExportFiles(t *testing.T) {
root := secretWorld(t)
writeTree(t, root, map[string]string{"plugins/Essentials/config.yml": "x: 1"})
if err := os.Symlink("config.yml", filepath.Join(root, "plugins/Essentials/link.yml")); err != nil {
t.Fatal(err)
}
export := func(t *testing.T, path string, dir bool) (*exportReceiver, int, string, string) {
t.Helper()
rcv := receiveExport(t, noContent)
t.Setenv(worldexport.TokenEnv, "tok")
args := []string{"--mode", "files", "--server", "survival", "--target-url", rcv.srv.URL, "--worlds-root", root, "--path", path}
if dir {
args = append(args, "--dir")
}
var stdout, stderr bytes.Buffer
code := cmdExport(args, &stdout, &stderr)
return rcv, code, stdout.String(), stderr.String()
}
for path, want := range map[string]string{
"world/region/r.0.0.mca": "chunks",
"server.properties": redactedProps,
} {
t.Run("a file goes with its exact length: "+path, func(t *testing.T) {
rcv, code, stdout, stderr := export(t, path, false)
if code != 0 || stdout != "felis export: server=survival mode=files downloaded\n" {
t.Fatalf("exit %d, stdout %q, stderr %q", code, stdout, stderr)
}
if string(rcv.body) != want || rcv.req.Header.Get("Content-Type") != "application/octet-stream" {
t.Fatalf("body %q, type %q; want %q", rcv.body, rcv.req.Header.Get("Content-Type"), want)
}
sentWhole(t, rcv, int64(len(want)))
})
}
t.Run("a folder goes as a zip, guarded", func(t *testing.T) {
rcv, code, stdout, stderr := export(t, "plugins", true)
if code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr)
}
if rcv.req.ContentLength != -1 || rcv.req.Header.Get("Content-Type") != "application/zip" {
t.Fatalf("length %d, type %q", rcv.req.ContentLength, rcv.req.Header.Get("Content-Type"))
}
zr, err := zip.NewReader(bytes.NewReader(rcv.body), int64(len(rcv.body)))
if err != nil {
t.Fatal(err)
}
var names []string
for _, f := range zr.File {
names = append(names, f.Name)
}
if want := []string{"plugins/", "plugins/Essentials/", "plugins/Essentials/config.yml"}; !slices.Equal(names, want) {
t.Fatalf("zip holds %v, want %v", names, want)
}
want := "felis export: left out 1 entries a zip does not carry (symbolic links, devices, sockets)\n" +
"felis export: left out 1 files that hold platform secrets\n" +
"felis export: server=survival mode=files downloaded\n"
if stdout != want {
t.Errorf("stdout = %q, want %q", stdout, want)
}
})
for _, c := range []struct {
path string
dir bool
want string
}{
{"config/paper-global.yml", false, "forwarding secret"},
{"plugins/copy.yml", false, "forwarding secret"},
{"plugins", false, "is a folder now"},
{"server.properties", true, "is not a folder now"},
} {
t.Run("refused before any request: "+c.path, func(t *testing.T) {
rcv, code, _, stderr := export(t, c.path, c.dir)
if code != 1 || rcv.hits.Load() != 0 || !strings.Contains(stderr, c.want) {
t.Fatalf("exit %d, %d requests, stderr %q; want 1, none, and %q", code, rcv.hits.Load(), stderr, c.want)
}
})
}
}
func TestCmdExportUsage(t *testing.T) {
for name, tc := range map[string]struct {
token string
args []string
}{
"no token": {"", []string{"--mode", "world", "--target-url", "http://api/x"}},
"no target": {"tok", []string{"--mode", "world"}},
"unknown mode": {"tok", []string{"--mode", "both", "--target-url", "http://api/x"}},
"backup without ref": {"tok", []string{"--mode", "backup", "--target-url", "http://api/x"}},
"files without path": {"tok", []string{"--mode", "files", "--target-url", "http://api/x"}},
} {
t.Run(name, func(t *testing.T) {
t.Setenv(worldexport.TokenEnv, tc.token)
var stdout, stderr bytes.Buffer
if code := cmdExport(tc.args, &stdout, &stderr); code != 2 {
t.Fatalf("exit %d, want 2; stderr %q", code, stderr.String())
}
})
}
}
// TestCmdExportWiring: the Job's `felis export` reaches cmdExport, and
// felis-api's executor mounts the backup store at the path the archives were
// written under, since a backup's ref is an absolute path there.
func TestCmdExportWiring(t *testing.T) {
t.Setenv(worldexport.TokenEnv, "")
var stdout, stderr bytes.Buffer
if code := run([]string{"export"}, &stdout, &stderr); code != 2 ||
stderr.String() != "felis export: --target-url and "+worldexport.TokenEnv+" are required\n" {
t.Fatalf("felis export = %d, stderr %q", code, stderr.String())
}
cfg := &config.Config{}
cfg.K8s.Namespace, cfg.Archive.LocalPath = "games", "/srv/felis-backups"
want := worldexport.Config{Namespace: "games", Image: "felis:1", BackupPVC: "felis-backups", BackupRoot: "/srv/felis-backups"}
if got := exportConfig(cfg, "felis:1", "felis-backups"); got != want {
t.Fatalf("exportConfig = %+v, want %+v", got, want)
}
}
+18 -111
View File
@@ -1,15 +1,11 @@
package main
import (
"context"
"encoding/base64"
"flag"
"fmt"
"io"
"net/http"
"os"
"os/signal"
"syscall"
"time"
"felis.lolicon.best/internal/fileedit"
)
@@ -24,43 +20,32 @@ import (
// config.Load: felis-api made the authorization decision (the caller owns this
// server, and the server is stopped so the RWO world volume is free); this process
// is the unprivileged hands that touch bytes. Its entire input is the flags
// below plus, for a write, the content variables and, for an upload, one token.
// Every isolation guarantee lives in the Pod spec (internal/fileedit/jobspec.go),
// and the path-containment guarantee lives in fileedit.Execute, which resolves
// every path through os.Root and therefore cannot be walked out of the world
// mount.
// below plus, for a write, one environment variable. Every isolation guarantee
// lives in the Pod spec (internal/fileedit/jobspec.go), and the path-containment
// guarantee lives in fileedit.Execute, which resolves the path through os.Root and
// therefore cannot be walked out of the world mount.
//
// Exit status carries a specific meaning that felis-api depends on: a CALLER-fault
// outcome — a path that escapes the root, a file that is missing or too large — is
// a SUCCESSFUL run that prints a Result carrying an error code, so the API can map
// it to a precise 4xx. A non-zero exit means the operation could not be attempted
// at all (the world mount is unreadable, an upload's bytes could not be fetched
// intact, the result unprintable), which the API reports as a 500.
// at all (the world mount is unreadable, the result unprintable), which the API
// reports as a 500.
func cmdFiles(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("files", flag.ContinueOnError)
fs.SetOutput(stderr)
op := fs.String("op", "", "operation: list, read, write, mkdir, delete, rename, upload or unzip")
op := fs.String("op", "", "operation: list, read, or write")
path := fs.String("path", "", "path to operate on, relative to the world root (empty = the root itself)")
worldsRoot := fs.String("worlds-root", "/data", "mount path of the world PVC; every path resolves under it")
expect := fs.String("expect-sha256", "", "write only: refuse unless the file's current SHA-256 (hex) is this")
createOnly := fs.Bool("create-only", false, "write only: refuse a path that already exists")
to := fs.String("to", "", "rename only: the destination path")
sourceURL := fs.String("source-url", "", "upload only: felis-api URL to fetch the bytes from")
size := fs.Int64("size", -1, "upload only: the byte count the fetched file must have")
sum := fs.String("sha256", "", "write and upload: the SHA-256 (hex) the content or the fetched file must have")
overwrite := fs.Bool("overwrite", false, "upload and unzip: replace files already there")
if err := fs.Parse(args); err != nil {
return 2
}
if *op == "" {
fmt.Fprintln(stderr, "felis files: --op is required")
fmt.Fprintln(stderr, "felis files: --op is required (list, read, or write)")
return 2
}
limitHeapToCgroup()
req := fileedit.Request{
Op: *op, Path: *path, To: *to, Expect: *expect, CreateOnly: *createOnly, Overwrite: *overwrite,
}
// New content arrives base64-encoded in the environment rather than in argv:
// a process's arguments are world-readable on the node (/proc/<pid>/cmdline),
@@ -68,46 +53,22 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
// secrets — an RCON password in server.properties is the obvious case. The
// encoding is what lets arbitrary bytes (CRLF endings, a BOM, a NUL) survive a
// channel that must be a valid string.
switch *op {
case fileedit.OpWrite:
// The content's SHA-256 comes with it, so bytes that changed on the way
// to this Job are refused rather than written (Request.ContentSHA256).
if *sum == "" {
fmt.Fprintln(stderr, "felis files: a write needs --sha256")
var content []byte
if *op == fileedit.OpWrite {
raw, ok := os.LookupEnv(fileedit.ContentEnv)
if !ok {
fmt.Fprintf(stderr, "felis files: a write needs %s in the environment\n", fileedit.ContentEnv)
return 2
}
content, err := fileedit.ContentFromEnv(os.LookupEnv)
decoded, err := base64.StdEncoding.DecodeString(raw)
if err != nil {
fmt.Fprintf(stderr, "felis files: %v\n", err)
fmt.Fprintf(stderr, "felis files: %s is not valid base64: %v\n", fileedit.ContentEnv, err)
return 2
}
req.Content, req.ContentSHA256 = content, *sum
case fileedit.OpUpload:
token := os.Getenv(fileedit.UploadTokenEnv)
if *sourceURL == "" || token == "" {
fmt.Fprintf(stderr, "felis files: an upload needs --source-url and %s\n", fileedit.UploadTokenEnv)
return 2
content = decoded
}
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
req.Upload = &fileedit.Upload{
Size: *size, SHA256: *sum,
Open: func() (io.ReadCloser, error) { return fetchUpload(ctx, *sourceURL, token) },
Landed: func() {
if err := reportLanded(ctx, *sourceURL, token); err != nil {
// The file is in place; felis-api drops its copy when it
// has sat idle long enough, and the panel cancels it too.
fmt.Fprintf(stderr, "felis files: tell felis-api the upload landed: %v\n", err)
}
},
}
}
// An upload or an unzip (the only ops that report progress) can run long
// enough that felis-api does not wait on its Job, and the panel shows how far
// it has got from the latest of these lines (fileedit.K8sRunner.Ops).
req.Progress = fileedit.ThrottledProgress(stdout, time.Second, time.Now)
res, err := fileedit.Execute(*worldsRoot, req)
res, err := fileedit.Execute(*worldsRoot, *op, *path, content, *expect)
if err != nil {
// The operation could not be attempted — infrastructure, not caller fault.
fmt.Fprintf(stderr, "felis files: %v\n", err)
@@ -122,57 +83,3 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
}
return 0
}
// fetchUpload opens the staged upload on felis-api's internal face. There is no
// retry: the token opens the upload once (fileedit.Stage), so a second attempt
// could only be refused, and the caller retries the failed Job whole (a file
// sent in parts stays staged until its Job reports it landed, so that retry
// does not send it again). Redirects are refused because the request carries the
// token and the internal face never redirects; the header timeout catches a
// wedged endpoint, and the Job's activeDeadlineSeconds bounds the body.
func fetchUpload(ctx context.Context, url, token string) (io.ReadCloser, error) {
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
return nil, err
}
req.Header.Set("Authorization", "Bearer "+token)
client := &http.Client{
Transport: &http.Transport{ResponseHeaderTimeout: 30 * time.Second},
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
}
resp, err := client.Do(req)
if err != nil {
return nil, err
}
if resp.StatusCode != http.StatusOK {
resp.Body.Close()
return nil, fmt.Errorf("GET returned %s", resp.Status)
}
return resp.Body, nil
}
// reportLanded tells felis-api the upload's file is in place (DELETE on the URL
// it was fetched from, with the same token), so it deletes the copy it staged.
// One try: the file has landed whatever the answer, and a copy nobody deletes
// is dropped once it has sat idle for fileedit.SessionIdle.
func reportLanded(ctx context.Context, url, token string) error {
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
req, err := http.NewRequestWithContext(ctx, http.MethodDelete, url, nil)
if err != nil {
return err
}
req.Header.Set("Authorization", "Bearer "+token)
client := &http.Client{
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
}
resp, err := client.Do(req)
if err != nil {
return err
}
resp.Body.Close()
if resp.StatusCode != http.StatusNoContent {
return fmt.Errorf("DELETE returned %s", resp.Status)
}
return nil
}
-358
View File
@@ -1,358 +0,0 @@
package main
import (
"archive/zip"
"bytes"
"cmp"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"encoding/json"
"io"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"strings"
"sync/atomic"
"testing"
"felis.lolicon.best/internal/fileedit"
)
// filesResult is the Result a `felis files` run printed on its marked line,
// the last it prints.
func filesResult(t *testing.T, stdout string) fileedit.Result {
t.Helper()
lines := strings.Split(strings.TrimSpace(stdout), "\n")
line, ok := strings.CutPrefix(lines[len(lines)-1], fileedit.ResultPrefix)
if !ok {
t.Fatalf("stdout has no result line: %q", stdout)
}
var res fileedit.Result
if err := json.Unmarshal([]byte(line), &res); err != nil {
t.Fatalf("result line %q: %v", line, err)
}
return res
}
// stagedSource is felis-api's internal face for one staged upload. It serves
// body to a GET carrying Bearer token and 404 to any other, and answers the
// DELETE that reports the file landed with landedCode (204 when unset),
// redirecting to landedTo when that is a redirect. reports counts those
// DELETEs, each with the token and at the path the bytes came from.
type stagedSource struct {
*httptest.Server
reports, strays atomic.Int32
landedCode int
landedTo string
}
func stagedUpload(t *testing.T, token string, body []byte) *stagedSource {
t.Helper()
s := &stagedSource{}
s.Server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.Header.Get("Authorization") != "Bearer "+token || r.URL.Path != "/u" {
s.strays.Add(1)
http.Error(w, "no such upload", http.StatusNotFound)
return
}
switch r.Method {
case http.MethodGet:
w.Write(body)
case http.MethodDelete:
s.reports.Add(1)
if s.landedTo != "" {
w.Header().Set("Location", s.landedTo)
}
w.WriteHeader(cmp.Or(s.landedCode, http.StatusNoContent))
default:
s.strays.Add(1)
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
}
}))
t.Cleanup(s.Close)
return s
}
func uploadArgs(root, sourceURL string, body []byte) []string {
sum := sha256.Sum256(body)
return []string{
"--op", "upload", "--path", "plugins/a.jar", "--worlds-root", root,
"--source-url", sourceURL, "--size", "4", "--sha256", hex.EncodeToString(sum[:]),
}
}
// uploadRoot is a world with the plugins folder an upload lands in.
func uploadRoot(t *testing.T) string {
t.Helper()
root := t.TempDir()
if err := os.Mkdir(filepath.Join(root, "plugins"), 0o755); err != nil {
t.Fatal(err)
}
return root
}
func TestCmdFilesUpload(t *testing.T) {
body := []byte("PK\x03\x04")
t.Run("fetches the staged bytes with its token and lands them", func(t *testing.T) {
root := uploadRoot(t)
srv := stagedUpload(t, "tok", body)
t.Setenv(fileedit.UploadTokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if !strings.HasPrefix(stdout.String(), fileedit.ProgressPrefix+`{"done":4,"total":4}`+"\n") {
t.Fatalf("stdout %q does not start with the progress to the last byte", stdout.String())
}
if res := filesResult(t, stdout.String()); res.Code != "" {
t.Fatalf("result = %+v", res)
}
got, err := os.ReadFile(filepath.Join(root, "plugins", "a.jar"))
if err != nil || !bytes.Equal(got, body) {
t.Fatalf("landed %q, %v", got, err)
}
if n, strays := srv.reports.Load(), srv.strays.Load(); n != 1 || strays != 0 || stderr.Len() != 0 {
t.Fatalf("reported landed %d times, %d stray requests, stderr %q; want once", n, strays, stderr.String())
}
})
t.Run("a file already there is a result, and nothing is reported landed", func(t *testing.T) {
root := uploadRoot(t)
if err := os.WriteFile(filepath.Join(root, "plugins", "a.jar"), []byte("old!"), 0o644); err != nil {
t.Fatal(err)
}
srv := stagedUpload(t, "tok", body)
t.Setenv(fileedit.UploadTokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeExists || srv.reports.Load() != 0 {
t.Fatalf("result = %+v, reported landed %d times", res, srv.reports.Load())
}
})
// The file is in place whatever felis-api answers, so the Job still succeeds
// and says why the staged copy may linger. A redirect is not followed, since
// the request carries the token.
for name, tc := range map[string]struct {
code int
stderr string
}{
"refused": {http.StatusNotFound, "felis files: tell felis-api the upload landed: DELETE returned 404 Not Found\n"},
"redirected": {http.StatusFound, "felis files: tell felis-api the upload landed: DELETE returned 302 Found\n"},
} {
t.Run("a landed report "+name+" still lands the file", func(t *testing.T) {
var elsewhere atomic.Int32
away := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) { elsewhere.Add(1) }))
defer away.Close()
root := uploadRoot(t)
srv := stagedUpload(t, "tok", body)
srv.landedCode, srv.landedTo = tc.code, away.URL+"/u"
t.Setenv(fileedit.UploadTokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != "" || stderr.String() != tc.stderr || elsewhere.Load() != 0 {
t.Fatalf("result = %+v, stderr %q, redirect followed %d times", res, stderr.String(), elsewhere.Load())
}
if got, err := os.ReadFile(filepath.Join(root, "plugins", "a.jar")); err != nil || !bytes.Equal(got, body) {
t.Fatalf("landed %q, %v", got, err)
}
})
}
// A refused fetch is the Job failing, never a Result: the API answers it with a
// 500 the caller retries whole.
t.Run("a refused fetch exits 1 and lands nothing", func(t *testing.T) {
root := uploadRoot(t)
srv := stagedUpload(t, "tok", body)
t.Setenv(fileedit.UploadTokenEnv, "wrong")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 1 {
t.Fatalf("exit %d, want 1; stdout %q", code, stdout.String())
}
if !strings.Contains(stderr.String(), "404") {
t.Fatalf("stderr %q does not name the status", stderr.String())
}
if srv.reports.Load() != 0 {
t.Fatal("a refused fetch was reported landed")
}
if _, err := os.Lstat(filepath.Join(root, "plugins", "a.jar")); !os.IsNotExist(err) {
t.Fatalf("a refused fetch left a file: %v", err)
}
})
// The request carries the token, and the internal face never redirects, so a
// redirect is refused rather than followed with the token attached.
t.Run("a redirect is not followed", func(t *testing.T) {
root := uploadRoot(t)
var hits atomic.Int32
elsewhere := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
hits.Add(1)
w.Write(body)
}))
defer elsewhere.Close()
redirecting := httptest.NewServer(http.RedirectHandler(elsewhere.URL+"/u", http.StatusFound))
defer redirecting.Close()
t.Setenv(fileedit.UploadTokenEnv, "tok")
var stdout, stderr bytes.Buffer
if code := cmdFiles(uploadArgs(root, redirecting.URL+"/u", body), &stdout, &stderr); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
if n := hits.Load(); n != 0 {
t.Fatalf("the redirect target was fetched %d times", n)
}
})
for name, tc := range map[string]struct {
token string
drop string
}{
"no token": {"", ""},
"no source URL": {"tok", "--source-url"},
} {
t.Run(name+" exits 2", func(t *testing.T) {
srv := stagedUpload(t, "tok", body)
t.Setenv(fileedit.UploadTokenEnv, tc.token)
args := uploadArgs(uploadRoot(t), srv.URL+"/u", body)
if tc.drop != "" {
for i, a := range args {
if a == tc.drop {
args = append(args[:i:i], args[i+2:]...)
break
}
}
}
var stdout, stderr bytes.Buffer
if code := cmdFiles(args, &stdout, &stderr); code != 2 {
t.Fatalf("exit %d, want 2", code)
}
})
}
}
func TestCmdFilesWrite(t *testing.T) {
root := t.TempDir()
content := []byte("[]\r\n")
sum := sha256.Sum256(content)
args := []string{"--op", "write", "--path", "ops.json", "--worlds-root", root, "--sha256", hex.EncodeToString(sum[:])}
t.Run("reassembles the content parts", func(t *testing.T) {
t.Setenv(fileedit.ContentPartsEnv, "1")
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString(content))
var stdout, stderr bytes.Buffer
if code := cmdFiles(args, &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != "" {
t.Fatalf("result = %+v", res)
}
if got, err := os.ReadFile(filepath.Join(root, "ops.json")); err != nil || !bytes.Equal(got, content) {
t.Fatalf("wrote %q, %v", got, err)
}
})
// Writing what did arrive of an incomplete spec would truncate the file.
t.Run("an incomplete content spec exits 2 and writes nothing", func(t *testing.T) {
t.Setenv(fileedit.ContentPartsEnv, "2")
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString([]byte("x")))
var stdout, stderr bytes.Buffer
if code := cmdFiles([]string{"--op", "write", "--path", "new.txt", "--worlds-root", root, "--sha256", hex.EncodeToString(sum[:])}, &stdout, &stderr); code != 2 {
t.Fatalf("exit %d, want 2", code)
}
if _, err := os.Lstat(filepath.Join(root, "new.txt")); !os.IsNotExist(err) {
t.Fatalf("an incomplete spec wrote a file: %v", err)
}
})
// Without the content's SHA-256 the Job could not tell bytes changed on the
// way from the bytes felis-api sent.
t.Run("a write without its SHA-256 exits 2 and writes nothing", func(t *testing.T) {
t.Setenv(fileedit.ContentPartsEnv, "1")
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString(content))
var stdout, stderr bytes.Buffer
if code := cmdFiles([]string{"--op", "write", "--path", "new.txt", "--worlds-root", root}, &stdout, &stderr); code != 2 {
t.Fatalf("exit %d, want 2", code)
}
if _, err := os.Lstat(filepath.Join(root, "new.txt")); !os.IsNotExist(err) {
t.Fatalf("a write without its SHA-256 wrote a file: %v", err)
}
})
t.Run("content that changed on the way is a result and writes nothing", func(t *testing.T) {
t.Setenv(fileedit.ContentPartsEnv, "1")
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString([]byte("[]\n")))
var stdout, stderr bytes.Buffer
if code := cmdFiles([]string{"--op", "write", "--path", "new.txt", "--worlds-root", root, "--sha256", hex.EncodeToString(sum[:])}, &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeDigestMismatch {
t.Fatalf("result = %+v, want %s", res, fileedit.CodeDigestMismatch)
}
if _, err := os.Lstat(filepath.Join(root, "new.txt")); !os.IsNotExist(err) {
t.Fatalf("changed content wrote a file: %v", err)
}
})
}
// A caller-fault outcome is a successful run carrying a code, so felis-api can
// answer the precise 4xx instead of a 500.
func TestCmdFilesCallerFaultIsAResult(t *testing.T) {
root := t.TempDir()
var stdout, stderr bytes.Buffer
if code := cmdFiles([]string{"--op", "mkdir", "--path", "../out", "--worlds-root", root}, &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeBadPath {
t.Fatalf("result = %+v, want code %s", res, fileedit.CodeBadPath)
}
stdout.Reset()
if code := cmdFiles([]string{"--worlds-root", root}, &stdout, &stderr); code != 2 {
t.Fatalf("no --op: exit %d, want 2", code)
}
}
// TestCmdFilesUnzip checks an unzip extracts next to the archive and reports its
// progress before its result, the same way an upload does.
func TestCmdFilesUnzip(t *testing.T) {
root := t.TempDir()
if err := os.Mkdir(filepath.Join(root, "maps"), 0o755); err != nil {
t.Fatal(err)
}
var zb bytes.Buffer
zw := zip.NewWriter(&zb)
for name, body := range map[string]string{"world/level.dat": "level", "world/region/r.0.0.mca": "region!"} {
w, err := zw.Create(name)
if err != nil {
t.Fatal(err)
}
io.WriteString(w, body)
}
if err := zw.Close(); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(root, "maps", "a.zip"), zb.Bytes(), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
if code := cmdFiles([]string{"--op", "unzip", "--path", "maps/a.zip", "--worlds-root", root}, &stdout, &stderr); code != 0 {
t.Fatalf("exit %d, stderr %q", code, stderr.String())
}
if res := filesResult(t, stdout.String()); res.Code != "" || res.Files != 2 || res.Bytes != 12 {
t.Fatalf("result = %+v", res)
}
if !strings.HasPrefix(stdout.String(), fileedit.ProgressPrefix) ||
!strings.Contains(stdout.String(), fileedit.ProgressPrefix+`{"done":12,"total":12}`+"\n") {
t.Fatalf("stdout %q does not report the progress to the last byte", stdout.String())
}
got, err := os.ReadFile(filepath.Join(root, "maps", "world", "region", "r.0.0.mca"))
if err != nil || string(got) != "region!" {
t.Fatalf("extracted %q, %v", got, err)
}
}
-84
View File
@@ -1,84 +0,0 @@
package main
import (
"context"
"errors"
"io/fs"
"os"
"path/filepath"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// Host copies of the credentials `felis setup` takes at the keyboard: the [smtp]
// relay password and the uploads bucket's keys. The cluster reads them from the
// felis-smtp and felis-uploads-s3 Secrets, and a Secret lives in k3s's datastore,
// which a reinstall (uninstall.sh keeps /etc/felis) or a host rebuilt from a
// database bundle's state/ starts empty. Each file holds the bare value, mode
// 0600, directly in /etc/felis beside secrets.env: every installer run applies
// the Secrets from these files, and every database bundle, so the off-site copy
// too, carries them.
const (
hostSMTPPasswordPath = "/etc/felis/smtp-password"
hostUploadsS3AccessKeyPath = "/etc/felis/uploads-s3-access-key"
hostUploadsS3SecretKeyPath = "/etc/felis/uploads-s3-secret-key"
)
// writeHostCredential replaces the file at path with value, mode 0600, through a
// temporary file in the same directory, so a crash leaves the old value or the
// new one and never a partial one.
func writeHostCredential(path, value string) error {
tmp, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".*")
if err != nil {
return err
}
tmpPath := tmp.Name()
defer os.Remove(tmpPath)
// CreateTemp already makes the file 0600; the Chmod states it rather than
// leaning on that.
if err := tmp.Chmod(0o600); err != nil {
_ = tmp.Close()
return err
}
if _, err := tmp.WriteString(value); err != nil {
_ = tmp.Close()
return err
}
if err := tmp.Sync(); err != nil {
_ = tmp.Close()
return err
}
if err := tmp.Close(); err != nil {
return err
}
return os.Rename(tmpPath, path)
}
// readHostCredential returns the value in path; ok is false when there is no
// such file. An empty file is a value: the relay password of a relay without AUTH.
func readHostCredential(path string) (value string, ok bool, err error) {
b, err := os.ReadFile(path)
if errors.Is(err, fs.ErrNotExist) {
return "", false, nil
}
if err != nil {
return "", false, err
}
return string(b), true, nil
}
// relayPassword is the [smtp] relay password as the host holds it: the copy
// `felis setup` keeps at path, else, on an install from before that copy, the
// felis-smtp Secret in ns, whose absence means a relay without AUTH. cl is only
// used when the file is missing; a nil cl then reports errClusterUnreachable.
func relayPassword(ctx context.Context, path string, cl client.Client, ns string) (string, error) {
if pw, ok, err := readHostCredential(path); err != nil || ok {
return pw, err
}
if cl == nil {
return "", errClusterUnreachable
}
return smtpSecretPassword(ctx, cl, ns)
}
var errClusterUnreachable = errors.New("the cluster did not answer")
-198
View File
@@ -1,198 +0,0 @@
package main
import (
"bytes"
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/mail"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/watchdog"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
func TestHostCredentialFile(t *testing.T) {
dir := t.TempDir()
path := filepath.Join(dir, "smtp-password")
if _, ok, err := readHostCredential(path); ok || err != nil {
t.Fatalf("a missing file read as ok=%v err=%v; want not there", ok, err)
}
// A copy an operator put there by hand, readable by everyone, is tightened.
if err := os.WriteFile(path, []byte("by hand"), 0o644); err != nil {
t.Fatal(err)
}
for _, v := range []string{"first secret", "a \"quoted\" $second\nsecret", ""} {
if err := writeHostCredential(path, v); err != nil {
t.Fatal(err)
}
got, ok, err := readHostCredential(path)
if err != nil || !ok || got != v {
t.Fatalf("read back (%q, %v, %v); want (%q, true, nil)", got, ok, err, v)
}
fi, err := os.Stat(path)
if err != nil {
t.Fatal(err)
}
if mode := fi.Mode().Perm(); mode != 0o600 {
t.Fatalf("mode %v; want 0600", mode)
}
}
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatal(err)
}
if len(entries) != 1 {
t.Fatalf("the directory holds %d entries; want the credential alone, no leftover temporary file", len(entries))
}
if _, _, err := readHostCredential(dir); err == nil {
t.Fatal("an unreadable credential read as fine")
}
}
func smtpSecretClient(t *testing.T, password string) client.Client {
t.Helper()
return fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(&corev1.Secret{
ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.SMTPSecretName},
Data: map[string][]byte{platform.SMTPSecretPasswordKey: []byte(password)},
}).Build()
}
func TestRelayPassword(t *testing.T) {
ctx := context.Background()
dir := t.TempDir()
host := filepath.Join(dir, "smtp-password")
cl := smtpSecretClient(t, "from-secret")
if pw, err := relayPassword(ctx, host, cl, "felis"); err != nil || pw != "from-secret" {
t.Fatalf("without a host copy = (%q, %v); want the Secret's", pw, err)
}
if _, err := relayPassword(ctx, host, nil, "felis"); !errors.Is(err, errClusterUnreachable) {
t.Fatalf("without a host copy or a cluster err = %v; want errClusterUnreachable", err)
}
if err := writeHostCredential(host, "from-host"); err != nil {
t.Fatal(err)
}
for _, c := range []client.Client{cl, nil} {
if pw, err := relayPassword(ctx, host, c, "felis"); err != nil || pw != "from-host" {
t.Fatalf("with a host copy (cluster %v) = (%q, %v); want the host copy", c != nil, pw, err)
}
}
if _, err := relayPassword(ctx, dir, cl, "felis"); err == nil {
t.Fatal("an unreadable host copy fell through to the Secret")
}
}
// The watchdog mails the most while the cluster is down: the host copy must
// reach it then, and without one the password the last good run cached stays.
func TestRefreshSMTPPassword(t *testing.T) {
ctx := context.Background()
dir := t.TempDir()
host := filepath.Join(dir, "smtp-password")
var stderr bytes.Buffer
state := &watchdog.State{SMTPPassword: "cached"}
refreshSMTPPassword(ctx, host, nil, "felis", state, &stderr)
if state.SMTPPassword != "cached" || stderr.Len() != 0 {
t.Fatalf("cluster down, no host copy: password %q, stderr %q; want the cached one kept quietly", state.SMTPPassword, stderr.String())
}
refreshSMTPPassword(ctx, host, smtpSecretClient(t, "from-secret"), "felis", state, &stderr)
if state.SMTPPassword != "from-secret" {
t.Fatalf("cluster up, no host copy: password %q; want the Secret's", state.SMTPPassword)
}
if err := writeHostCredential(host, "from-host"); err != nil {
t.Fatal(err)
}
refreshSMTPPassword(ctx, host, nil, "felis", state, &stderr)
if state.SMTPPassword != "from-host" {
t.Fatalf("cluster down, host copy: password %q; want the host copy", state.SMTPPassword)
}
refreshSMTPPassword(ctx, dir, nil, "felis", state, &stderr)
if state.SMTPPassword != "from-host" || !strings.Contains(stderr.String(), "keeping the cached one") {
t.Fatalf("unreadable host copy: password %q, stderr %q; want the cached one kept and the failure said", state.SMTPPassword, stderr.String())
}
}
func TestHostRecoveryMailerHostCopy(t *testing.T) {
ctx := context.Background()
// No kubeconfig anywhere: reaching for the cluster fails, so a pass proves
// the host copy was enough.
t.Setenv("KUBECONFIG", filepath.Join(t.TempDir(), "no-kubeconfig"))
dir := t.TempDir()
host := filepath.Join(dir, "smtp-password")
if err := writeHostCredential(host, "from-host"); err != nil {
t.Fatal(err)
}
off := false
c := config.SMTPConfig{Host: "mail.example.com", Port: 2525, From: "[email protected]", Username: "felis", PasswordRef: "FELIS_TEST_UNSET_RELAY_PW", RequireTLS: &off}
got, err := hostRecoveryMailer(c, host, "felis")(ctx)
if err != nil {
t.Fatalf("with the host copy: %v", err)
}
if relay, ok := got.(*mail.SMTP); !ok || relay.Password != "from-host" {
t.Fatalf("relay = %#v; want the host copy's password", got)
}
if _, err := hostRecoveryMailer(c, dir, "felis")(ctx); err == nil || !strings.Contains(err.Error(), "read the relay password") {
t.Fatalf("unreadable host copy: err = %v; want it named", err)
}
if _, err := hostRecoveryMailer(c, filepath.Join(dir, "none"), "felis")(ctx); err == nil || !strings.Contains(err.Error(), "reach the cluster") {
t.Fatalf("no host copy and no cluster: err = %v; want the cluster named", err)
}
}
// A whole watchdog run with the API server and PostgreSQL both down still
// takes the relay password from the host copy, so the outage mail can
// authenticate even when no earlier run cached it.
func TestWatchdogReadsHostCopyWhileClusterDown(t *testing.T) {
dir := t.TempDir()
t.Setenv("KUBECONFIG", filepath.Join(dir, "no-kubeconfig"))
cfgPath := filepath.Join(dir, "felis.toml")
if err := os.WriteFile(cfgPath, []byte(`[database]
url = "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"
[server]
root_domain = "example.com"
[archive]
store = "tarLocal"
[k8s]
egress_mode = "nodeport"
[smtp]
host = "127.0.0.1"
port = 1
from = "[email protected]"
username = "felis"
password_ref = "FELIS_TEST_UNSET_RELAY_PW"
`), 0o600); err != nil {
t.Fatal(err)
}
pwPath := filepath.Join(dir, "smtp-password")
if err := writeHostCredential(pwPath, "from-host"); err != nil {
t.Fatal(err)
}
statePath := filepath.Join(dir, "state.json")
var stdout, stderr bytes.Buffer
cmdWatchdog([]string{
"-config", cfgPath, "-state", statePath, "-quiet-file", filepath.Join(dir, "quiet"),
"-backup-dir", "", "-disk-paths", dir, "-smtp-password-file", pwPath, "-heartbeat-file", filepath.Join(dir, "no-heartbeat"),
}, &stdout, &stderr)
if !strings.Contains(stdout.String(), "kube-api") {
t.Fatalf("the run found the API server up; the test needs it down (stdout %s)", stdout.String())
}
state, err := watchdog.LoadState(statePath)
if err != nil {
t.Fatalf("load state: %v (stderr %s)", err, stderr.String())
}
if state.SMTPPassword != "from-host" {
t.Fatalf("cached relay password %q; want the host copy (stdout %s, stderr %s)", state.SMTPPassword, stdout.String(), stderr.String())
}
}
-109
View File
@@ -1,109 +0,0 @@
package main
import (
"bufio"
"context"
"errors"
"flag"
"fmt"
"io"
"os"
"os/signal"
"strings"
"syscall"
"felis.lolicon.best/internal/imagepush"
)
// cmdImageBundle writes a release's image bundle: one OCI layout tar holding every
// image an install runs, for one platform, plus its listing (one "role name
// manifest-digest config-digest" line per image). deploy/build-release-artifacts.sh
// runs it in CI; deploy/bootstrap.sh imports the tar into k3s's containerd and
// pushes it into the platform registry with push-image --image.
//
// --layout role=name=path an image buildx wrote with --output type=oci
// --pull role=ref a digest-pinned public image, named repository@digest
//
// The tar and the listing are written beside their final paths and renamed into
// place, so a failed run leaves neither behind.
func cmdImageBundle(args []string, _, stderr io.Writer) int {
fs := flag.NewFlagSet("image-bundle", flag.ContinueOnError)
fs.SetOutput(stderr)
platform := fs.String("platform", "", "os/arch the bundle is for, e.g. linux/arm64")
out := fs.String("out", "", "path of the bundle tar to write")
list := fs.String("list", "", "path of the listing to write")
var images []imagepush.BundleImage
fs.Func("layout", "role=name=path of an OCI layout tar (repeatable)", func(v string) error {
role, rest, ok := strings.Cut(v, "=")
name, path, ok2 := strings.Cut(rest, "=")
if !ok || !ok2 || role == "" || name == "" || path == "" {
return fmt.Errorf("want role=name=path, got %q", v)
}
images = append(images, imagepush.BundleImage{Role: role, Name: name, Layout: path})
return nil
})
fs.Func("pull", "role=ref of a digest-pinned public image (repeatable)", func(v string) error {
role, ref, ok := strings.Cut(v, "=")
if !ok || role == "" || !strings.Contains(ref, "@sha256:") {
return fmt.Errorf("want role=ref with ref pinned by digest, got %q", v)
}
images = append(images, imagepush.BundleImage{Role: role, Name: imagepush.PinnedName(ref), Source: ref})
return nil
})
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if *platform == "" || *out == "" || *list == "" || len(images) == 0 {
fmt.Fprintln(stderr, "felis image-bundle: --platform, --out, --list and at least one --layout or --pull are required")
return 2
}
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
if err := writeImageBundle(ctx, &imagepush.Source{Platform: *platform}, images, *out, *list); err != nil {
fmt.Fprintf(stderr, "felis image-bundle: %v\n", err)
return 1
}
return 0
}
func writeImageBundle(ctx context.Context, s *imagepush.Source, images []imagepush.BundleImage, out, list string) (err error) {
tmpOut, tmpList := out+".tmp", list+".tmp"
defer func() {
if err != nil {
os.Remove(tmpOut)
os.Remove(tmpList)
}
}()
f, err := os.Create(tmpOut)
if err != nil {
return err
}
w := bufio.NewWriterSize(f, 1<<20)
entries, err := imagepush.WriteBundle(ctx, s, images, w)
if err == nil {
err = w.Flush()
}
if err == nil {
err = f.Sync()
}
if cerr := f.Close(); err == nil {
err = cerr
}
if err != nil {
return err
}
var b strings.Builder
for _, e := range entries {
fmt.Fprintf(&b, "%s %s %s %s\n", e.Role, e.Name, e.Digest, e.Config)
}
if err := os.WriteFile(tmpList, []byte(b.String()), 0o644); err != nil {
return err
}
if err := os.Rename(tmpOut, out); err != nil {
return err
}
return os.Rename(tmpList, list)
}
-131
View File
@@ -1,131 +0,0 @@
package main
import (
"archive/tar"
"bytes"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
)
// writeTestLayout writes an OCI layout tar holding one linux/arch image with one
// layer, the way buildx --output type=oci leaves a single-platform build, and
// returns its path with the manifest and config digests.
func writeTestLayout(t *testing.T, dir, arch, layer string) (path, manifestDigest, configDigest string) {
t.Helper()
blobs := map[string][]byte{}
add := func(b []byte) (string, int) {
sum := sha256.Sum256(b)
d := "sha256:" + hex.EncodeToString(sum[:])
blobs[d] = b
return d, len(b)
}
cfgDigest, cfgSize := add([]byte(`{"architecture":"` + arch + `","os":"linux","rootfs":{"type":"layers"}}`))
layerDigest, layerSize := add([]byte(layer))
manifest := fmt.Sprintf(`{"schemaVersion":2,"mediaType":"application/vnd.oci.image.manifest.v1+json",`+
`"config":{"mediaType":"application/vnd.oci.image.config.v1+json","digest":%q,"size":%d},`+
`"layers":[{"mediaType":"application/vnd.oci.image.layer.v1.tar+gzip","digest":%q,"size":%d}]}`,
cfgDigest, cfgSize, layerDigest, layerSize)
mDigest, mSize := add([]byte(manifest))
index, _ := json.Marshal(map[string]any{
"schemaVersion": 2,
"manifests": []map[string]any{{
"mediaType": "application/vnd.oci.image.manifest.v1+json", "digest": mDigest, "size": mSize,
}},
})
var buf bytes.Buffer
tw := tar.NewWriter(&buf)
put := func(name string, b []byte) {
tw.WriteHeader(&tar.Header{Name: name, Mode: 0o644, Size: int64(len(b)), Typeflag: tar.TypeReg})
tw.Write(b)
}
put("oci-layout", []byte(`{"imageLayoutVersion":"1.0.0"}`))
put("index.json", index)
for d, b := range blobs {
put("blobs/sha256/"+strings.TrimPrefix(d, "sha256:"), b)
}
tw.Close()
path = filepath.Join(dir, arch+"-"+layer+".tar")
if err := os.WriteFile(path, buf.Bytes(), 0o644); err != nil {
t.Fatal(err)
}
return path, mDigest, cfgDigest
}
func TestImageBundleWritesTheListingTheInstallerReads(t *testing.T) {
dir := t.TempDir()
limbo, limboDigest, limboConfig := writeTestLayout(t, dir, "arm64", "limbo")
lobby, lobbyDigest, lobbyConfig := writeTestLayout(t, dir, "arm64", "lobby")
out, list := filepath.Join(dir, "images.tar"), filepath.Join(dir, "images.txt")
var stderr bytes.Buffer
code := cmdImageBundle([]string{"--platform", "linux/arm64", "--out", out, "--list", list,
"--layout", "limbo=registry.felis.svc:5000/felis/limbo:demo=" + limbo,
"--layout", "lobby=registry.felis.svc:5000/felis/lobby:demo=" + lobby,
}, nil, &stderr)
if code != 0 {
t.Fatalf("image-bundle = %d: %s", code, stderr.String())
}
// deploy/bootstrap.sh reads this with `read -r role name digest config`.
got, _ := os.ReadFile(list)
want := "limbo registry.felis.svc:5000/felis/limbo:demo " + limboDigest + " " + limboConfig + "\n" +
"lobby registry.felis.svc:5000/felis/lobby:demo " + lobbyDigest + " " + lobbyConfig + "\n"
if string(got) != want {
t.Errorf("listing:\n%s\nwant:\n%s", got, want)
}
if fi, err := os.Stat(out); err != nil || fi.Size() == 0 {
t.Errorf("bundle: %v", err)
}
if leftovers, _ := filepath.Glob(filepath.Join(dir, "*.tmp")); len(leftovers) != 0 {
t.Errorf("left behind %v", leftovers)
}
}
func TestImageBundleLeavesNothingWhenAnImageIsRefused(t *testing.T) {
dir := t.TempDir()
amd, _, _ := writeTestLayout(t, dir, "amd64", "limbo")
out, list := filepath.Join(dir, "images.tar"), filepath.Join(dir, "images.txt")
// The pair an earlier run wrote stays as it was: a listing beside a bundle it
// does not describe would have the installer look for images that are not there.
os.WriteFile(out, []byte("old bundle"), 0o644)
os.WriteFile(list, []byte("old listing\n"), 0o644)
var stderr bytes.Buffer
code := cmdImageBundle([]string{"--platform", "linux/arm64", "--out", out, "--list", list,
"--layout", "limbo=registry.felis.svc:5000/felis/limbo:demo=" + amd}, nil, &stderr)
if code != 1 || !strings.Contains(stderr.String(), "is a linux/amd64 image") {
t.Fatalf("image-bundle = %d: %s", code, stderr.String())
}
if got, _ := os.ReadFile(out); string(got) != "old bundle" {
t.Errorf("bundle was replaced with %d bytes", len(got))
}
if got, _ := os.ReadFile(list); string(got) != "old listing\n" {
t.Errorf("listing was replaced with %q", got)
}
if leftovers, _ := filepath.Glob(filepath.Join(dir, "*.tmp")); len(leftovers) != 0 {
t.Errorf("left behind %v", leftovers)
}
}
func TestPushImageReadsABundleByImageName(t *testing.T) {
dir := t.TempDir()
limbo, _, _ := writeTestLayout(t, dir, "arm64", "limbo")
out, list := filepath.Join(dir, "images.tar"), filepath.Join(dir, "images.txt")
if code := cmdImageBundle([]string{"--platform", "linux/arm64", "--out", out, "--list", list,
"--layout", "limbo=registry.felis.svc:5000/felis/limbo:demo=" + limbo}, nil, &bytes.Buffer{}); code != 0 {
t.Fatal("image-bundle failed")
}
t.Setenv("FELIS_REGISTRY_USERNAME", "platform")
t.Setenv("FELIS_REGISTRY_PASSWORD", "x")
// The name is looked up in the bundle's index before the registry is contacted,
// so a name the bundle lacks fails here with what it does hold.
var stderr bytes.Buffer
code := cmdPushImage([]string{"--tar", out, "--image", "registry.felis.svc:5000/felis/lobby:demo",
"--ref", "127.0.0.1:1/felis/lobby:demo"}, &bytes.Buffer{}, &stderr)
if code != 1 || !strings.Contains(stderr.String(), "holds no image named registry.felis.svc:5000/felis/lobby:demo (it holds: registry.felis.svc:5000/felis/limbo:demo") {
t.Fatalf("push-image = %d: %s", code, stderr.String())
}
}
-24
View File
@@ -6,32 +6,8 @@ package main
import (
"os"
"path/filepath"
)
func main() {
ensureHostBinDirOnPath()
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
}
// hostBinDir is where deploy/bootstrap.sh installs felis, k3s and cloudflared.
const hostBinDir = "/usr/local/bin"
// ensureHostBinDirOnPath appends hostBinDir to PATH when it is missing, so the
// k3s and cloudflared this binary execs are found beside it. sudo's secure_path
// on EL leaves /usr/local/bin out: `sudo /usr/local/bin/felis db backup` would
// otherwise run with no k3s to reach the database's pod through. Appended, so a
// PATH that names another k3s first keeps it.
func ensureHostBinDirOnPath() {
path := os.Getenv("PATH")
for _, dir := range filepath.SplitList(path) {
if dir == hostBinDir {
return
}
}
if path == "" {
os.Setenv("PATH", hostBinDir)
return
}
os.Setenv("PATH", path+string(os.PathListSeparator)+hostBinDir)
}
-33
View File
@@ -1,33 +0,0 @@
package main
import (
"os"
"regexp"
"testing"
)
func TestEnsureHostBinDirOnPath(t *testing.T) {
for _, c := range []struct{ in, want string }{
{"/usr/sbin:/usr/bin", "/usr/sbin:/usr/bin:/usr/local/bin"},
{"/usr/local/bin:/usr/bin", "/usr/local/bin:/usr/bin"},
{"/opt/k3s:/usr/bin:/usr/local/bin", "/opt/k3s:/usr/bin:/usr/local/bin"},
{"", "/usr/local/bin"},
} {
t.Setenv("PATH", c.in)
ensureHostBinDirOnPath()
if got := os.Getenv("PATH"); got != c.want {
t.Errorf("PATH %q became %q, want %q", c.in, got, c.want)
}
}
}
// run is what the tests drive, so the PATH fix must sit in main, before it.
func TestMainFixesPathBeforeRunning(t *testing.T) {
b, err := os.ReadFile("main.go")
if err != nil {
t.Fatal(err)
}
if !regexp.MustCompile(`func main\(\) \{\n\tensureHostBinDirOnPath\(\)\n\tos\.Exit\(run\(`).Match(b) {
t.Fatal("main does not call ensureHostBinDirOnPath before run")
}
}
+1 -25
View File
@@ -35,11 +35,6 @@ func (m *multiFlag) Set(v string) error {
// --velocity-cidr records the proxy host addresses allowed by the game NetworkPolicy.
// Kubernetes permits resident-node traffic regardless, but remote proxy deployments
// need an explicit CIDR, so the renderer refuses to guess.
//
// --only postgres renders just the control-plane database (platform.PostgresObjects),
// which the installer brings up before migrations, before it has anything else
// to render the full bundle with; it needs neither --felis-image nor
// --velocity-cidr.
func cmdManifests(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("manifests", flag.ContinueOnError)
fs.SetOutput(stderr)
@@ -51,8 +46,6 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
panelNodePort := fs.Int("panel-node-port", int(platform.DefaultPanelNodePort), "NodePort that exposes the built-in HTTPS panel/API origin")
felisImage := fs.String("felis-image", "", "container image the felis-api/operator Deployments run, also passed through as FELIS_IMAGE (REQUIRED)")
registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry 2.8.3, pinned by digest)")
postgresImage := fs.String("postgres-image", "", "control-plane database image (default: PostgreSQL 18.6, pinned by digest)")
only := fs.String("only", "", `render one part of the bundle instead of all of it; "postgres" is the control-plane database`)
backupPVC := fs.String("backup-pvc", "felis-backups", "name of the world-archive PVC this bundle renders in the Minecraft namespace and advertises to the backup/restore executors via FELIS_BACKUP_PVC (default: felis-backups; pass an empty value to render none, leaving backup/restore answering 503)")
worldsHostPath := fs.String("worlds-host-path", "", "node directory the reaper reads worlds from: each world PVC resolves as <path>/<pvc>, or as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (k3s storage root: /var/lib/rancher/k3s/storage); enables the reaper CronJob (requires --archive-local-path and a non-empty --backup-pvc)")
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path. With the backup PVC alone it renders the retention-only CronJob, which deletes backups past their expiry and never touches a world")
@@ -71,18 +64,6 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
if err := fs.Parse(args); err != nil {
return 2
}
switch *only {
case "":
case "postgres":
return renderManifests(stdout, stderr, platform.PostgresObjects(platform.Params{
ControlNamespace: *controlNS,
MinecraftNamespace: *minecraftNS,
PostgresImage: *postgresImage,
}))
default:
fmt.Fprintf(stderr, "felis manifests: --only %q: the one part that renders alone is \"postgres\"\n", *only)
return 2
}
// Keep proxy placement explicit. This matters for remote proxies and documents
// the expected source even when Velocity runs on the resident node.
@@ -184,7 +165,6 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
PanelNodePort: int32(*panelNodePort),
FelisImage: *felisImage,
RegistryImage: *registryImage,
PostgresImage: *postgresImage,
BackupPVC: *backupPVC,
WorldsHostPath: *worldsHostPath,
ReaperNode: *reaperNode,
@@ -203,11 +183,7 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis manifests: %v\n", err)
return 2
}
return renderManifests(stdout, stderr, platform.Objects(params))
}
func renderManifests(stdout, stderr io.Writer, objs []platform.Object) int {
out, err := platform.RenderObjects(objs)
out, err := platform.RenderYAML(params)
if err != nil {
fmt.Fprintf(stderr, "felis manifests: render: %v\n", err)
return 1
-34
View File
@@ -278,37 +278,3 @@ func TestManifestsStorageSizes(t *testing.T) {
t.Errorf("--registry-storage lots: exit %d, stderr %q; want a refusal naming the flag", code, errBuf.String())
}
}
// TestManifestsOnlyPostgres: the installer renders the database before it has
// an image or a proxy address for the rest of the bundle, so --only postgres
// must render without them, and render the database and nothing else (a stray
// Deployment in that apply would start without its identities).
func TestManifestsOnlyPostgres(t *testing.T) {
var out, errBuf bytes.Buffer
code := run([]string{"manifests", "--only", "postgres", "--control-namespace", "ctl", "--postgres-image", "example/pg:18@sha256:abc"}, &out, &errBuf)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String())
}
var kinds []string
for _, doc := range strings.Split(out.String(), "\n---\n") {
for _, line := range strings.Split(doc, "\n") {
if strings.HasPrefix(line, "kind: ") {
kinds = append(kinds, strings.TrimPrefix(line, "kind: "))
}
}
}
if got, want := strings.Join(kinds, ","), "Namespace,ConfigMap,NetworkPolicy,Deployment,Service"; got != want {
t.Errorf("rendered kinds %s, want %s", got, want)
}
for _, want := range []string{"name: felis-postgres", "namespace: ctl", "image: example/pg:18@sha256:abc"} {
if !strings.Contains(out.String(), want) {
t.Errorf("rendered database missing %q", want)
}
}
out.Reset()
errBuf.Reset()
if code := run([]string{"manifests", "--only", "registry"}, &out, &errBuf); code != 2 || out.Len() != 0 {
t.Errorf("--only registry: exit %d with %d bytes of YAML, want 2 and none", code, out.Len())
}
}
-48
View File
@@ -1,48 +0,0 @@
package main
import (
"os"
"runtime/debug"
"strconv"
"strings"
)
// cgroupMemoryFiles are where a container reads the memory it is allowed:
// cgroup v2 first, then v1.
var cgroupMemoryFiles = []string{"/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory/memory.limit_in_bytes"}
// limitHeapToCgroup sets the Go heap's soft limit from the container's memory
// limit, so the collector works harder as a Job nears it and the kernel does not
// kill the Job first. An extraction or a folder zipped for download keeps a few
// hundred bytes per entry for as long as it runs; without the limit the heap
// grows to twice that before a collection, and a 256 MiB Job was killed at
// 400,000 entries whose live heap was 115 MB. GOMEMLIMIT set by hand wins.
func limitHeapToCgroup() {
if os.Getenv("GOMEMLIMIT") != "" {
return
}
for _, f := range cgroupMemoryFiles {
b, err := os.ReadFile(f)
if err != nil {
continue
}
if n, ok := softMemoryLimit(string(b)); ok {
debug.SetMemoryLimit(n)
}
return
}
}
// softMemoryLimit answers three fifths of the limit a cgroup memory file holds,
// or false for "max" (no limit) and anything unreadable. The rest is left for
// what the kernel charges the container beyond the Go heap: the page cache of
// the files it reads and writes, and the inodes it creates. Under a 256 MiB
// limit, 400,000 extracted entries peaked at 184 MB resident with the heap held
// to 150 MiB.
func softMemoryLimit(content string) (int64, bool) {
n, err := strconv.ParseInt(strings.TrimSpace(content), 10, 64)
if err != nil || n <= 0 {
return 0, false
}
return n / 5 * 3, true
}
-58
View File
@@ -1,58 +0,0 @@
package main
import (
"math"
"os"
"path/filepath"
"runtime/debug"
"testing"
)
func TestSoftMemoryLimit(t *testing.T) {
for _, c := range []struct {
in string
want int64
ok bool
}{
{"268435456\n", 161061273, true}, // 256 MiB, as memory.max holds it
{"max\n", 0, false},
{"0\n", 0, false},
{"-1", 0, false},
{"", 0, false},
} {
got, ok := softMemoryLimit(c.in)
if got != c.want || ok != c.ok {
t.Errorf("softMemoryLimit(%q) = %d %v, want %d %v", c.in, got, ok, c.want, c.ok)
}
}
}
// TestLimitHeapToCgroup checks the limit comes from the first cgroup file there
// is, and that GOMEMLIMIT set by hand leaves the heap alone.
func TestLimitHeapToCgroup(t *testing.T) {
prevFiles, prevLimit := cgroupMemoryFiles, debug.SetMemoryLimit(-1)
t.Cleanup(func() { cgroupMemoryFiles = prevFiles; debug.SetMemoryLimit(prevLimit) })
dir := t.TempDir()
v1, v1b := filepath.Join(dir, "v1"), filepath.Join(dir, "v1b")
if err := os.WriteFile(v1, []byte("268435456\n"), 0o600); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(v1b, []byte("536870912\n"), 0o600); err != nil {
t.Fatal(err)
}
cgroupMemoryFiles = []string{filepath.Join(dir, "missing"), v1, v1b}
t.Setenv("GOMEMLIMIT", "")
debug.SetMemoryLimit(math.MaxInt64)
limitHeapToCgroup()
if got := debug.SetMemoryLimit(-1); got != 161061273 {
t.Fatalf("limit = %d, want three fifths of 256 MiB", got)
}
t.Setenv("GOMEMLIMIT", "1GiB")
debug.SetMemoryLimit(math.MaxInt64)
limitHeapToCgroup()
if got := debug.SetMemoryLimit(-1); got != math.MaxInt64 {
t.Fatalf("limit = %d with GOMEMLIMIT set, want it left alone", got)
}
}
+7 -52
View File
@@ -2,11 +2,9 @@ package main
import (
"context"
"errors"
"flag"
"fmt"
"io"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
@@ -20,8 +18,7 @@ import (
// database that already holds a schema and has migrations pending is bundled
// first (internal/dbbackup, label pre-migrate). A failed snapshot stops the
// upgrade; -no-backup is the explicit way past it, e.g. for an external
// database (no [database] deployment, so the host's own pg_dump runs) whose
// server is newer than that pg_dump.
// database whose server is newer than the host's pg_dump.
func cmdMigrate(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("migrate", flag.ContinueOnError)
fs.SetOutput(stderr)
@@ -61,7 +58,7 @@ func cmdMigrate(args []string, stdout, stderr io.Writer) int {
}
if !*noBackup {
path, err := preMigrateBackup(ctx, drv, migrations, cfg.Database, *backupDir, stderr)
path, err := preMigrateBackup(ctx, drv, migrations, cfg.Database.URL, *backupDir, stderr)
if err != nil {
fmt.Fprintf(stderr, "felis migrate: pre-migration backup failed, nothing applied: %v\n", err)
fmt.Fprintln(stderr, " fix the backup, or re-run with -no-backup to migrate without one")
@@ -85,14 +82,10 @@ func cmdMigrate(args []string, stdout, stderr io.Writer) int {
return 0
}
// preMigrateStateDir is the host state a pre-migrate bundle carries; tests
// point it at a directory of their own.
var preMigrateStateDir = dbbackup.DefaultStateDir
// preMigrateBackup bundles the database when it already carries a schema and
// some of migrations are not applied yet, and returns the bundle's path ("" when
// there was nothing to protect: a fresh database, or nothing pending).
func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.Migration, db config.DatabaseConfig, dir string, log io.Writer) (string, error) {
func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.Migration, dbURL, dir string, log io.Writer) (string, error) {
if err := drv.EnsureVersionTable(ctx); err != nil {
return "", fmt.Errorf("ensure version table: %w", err)
}
@@ -103,22 +96,11 @@ func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.
if len(done) == 0 || !hasPending(done, migrations) {
return "", nil
}
tools, err := dbTools(db)
if err != nil {
return "", err
}
path, err := dbbackup.Backup(ctx, dbbackup.BackupOptions{
DatabaseURL: db.URL, Tools: tools, Dir: dir, Label: dbbackup.LabelPreMigrate,
Keep: defaultKeep[dbbackup.LabelPreMigrate], StateDir: preMigrateStateDir,
Version: resolvedVersion(), ExportServers: exportMinecraftServers, Log: log, Record: true,
return dbbackup.Backup(ctx, dbbackup.BackupOptions{
DatabaseURL: dbURL, Dir: dir, Label: dbbackup.LabelPreMigrate,
Keep: defaultKeep[dbbackup.LabelPreMigrate], StateDir: dbbackup.DefaultStateDir,
Version: resolvedVersion(), Log: log, Record: true,
})
if errors.Is(err, dbbackup.ErrServersMissing) {
// Rolling the migration back needs the database alone. Backup logged
// the gap, and the panel and the watchdog show it while this is the
// newest bundle.
return path, nil
}
return path, err
}
func hasPending(done map[int]struct{}, migrations []store.Migration) bool {
@@ -139,33 +121,6 @@ func openStore(ctx context.Context, url string, allowPending bool) (*store.Postg
if err != nil {
return nil, err
}
return checkSchema(ctx, drv, allowPending)
}
// podDBWindow and podDBInterval bound how long a pod that has just started retries its
// first database dial while the network policy has yet to admit it
// (store.OpenRetrying). A minute is far past the sync lag and far inside every Job's
// deadline. Vars so a test can shrink them.
var (
podDBWindow = time.Minute
podDBInterval = time.Second
)
// openPodStore is openStore for felis-api and the reaper and backup Jobs, whose first
// dial comes milliseconds after their pod starts.
func openPodStore(ctx context.Context, url, prog string, stderr io.Writer) (*store.PostgresDriver, error) {
drv, err := store.OpenRetrying(ctx, url, podDBWindow, podDBInterval, func(err error) {
fmt.Fprintf(stderr, "felis %s: %v; retrying (a pod that has just started waits for the network policy to admit it)\n", prog, err)
})
if err != nil {
return nil, err
}
return checkSchema(ctx, drv, false)
}
// checkSchema closes drv and fails when its schema is not the one this build was
// written against (see openStore).
func checkSchema(ctx context.Context, drv *store.PostgresDriver, allowPending bool) (*store.PostgresDriver, error) {
s, err := store.ReadSchema(ctx, drv)
if err == nil {
err = s.Err()
-48
View File
@@ -1,48 +0,0 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"net"
"strings"
"syscall"
"testing"
"time"
)
// openPodStore retries a refused first dial for podDBWindow, saying so on stderr
// under the calling command's name each time.
func TestOpenPodStoreRetriesARefusedDial(t *testing.T) {
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
addr := ln.Addr().String()
ln.Close()
window, interval := podDBWindow, podDBInterval
podDBWindow, podDBInterval = 200*time.Millisecond, 20*time.Millisecond
t.Cleanup(func() { podDBWindow, podDBInterval = window, interval })
var stderr bytes.Buffer
start := time.Now()
_, err = openPodStore(context.Background(), fmt.Sprintf("postgres://felis@%s/felis?sslmode=disable", addr), "reaper", &stderr)
if !errors.Is(err, syscall.ECONNREFUSED) {
t.Fatalf("err = %v, want a refused dial", err)
}
if elapsed := time.Since(start); elapsed < podDBWindow {
t.Fatalf("gave up after %s, inside the %s window", elapsed, podDBWindow)
}
lines := strings.Split(strings.TrimSuffix(stderr.String(), "\n"), "\n")
if len(lines) < 2 || len(lines) > 11 {
t.Fatalf("%d retry lines in a 200ms window at 20ms:\n%s", len(lines), stderr.String())
}
for _, l := range lines {
if !strings.HasPrefix(l, "felis reaper: failed to connect to `user=felis database=felis`: ") ||
!strings.HasSuffix(l, "; retrying (a pod that has just started waits for the network policy to admit it)") {
t.Fatalf("retry line %q", l)
}
}
}
+42 -335
View File
@@ -33,25 +33,12 @@ const offsiteUsage = `usage:
felis offsite fetch-worlds [-config path] [-archive-dir dir]
felis offsite fetch-images [-config path] [-registry host:port] [-at version]
felis offsite fetch-uploads [-config path] [-uploads-dir dir] [-at version]
felis offsite check-key [-config path]
felis offsite take-over [-config path] [-status-file path] [-yes]
felis offsite keygen
Every verb but keygen reads the bucket credentials and the encryption key from
the variables [offsite] names (default FELIS_OFFSITE_ACCESS_KEY,
FELIS_OFFSITE_SECRET_KEY, FELIS_OFFSITE_KEY), taking any that are unset from
-env-file (default /etc/felis/offsite.env).
check-key tells whether the key is the one the bucket's objects are sealed
with, writing nothing; it exits 3 when they are sealed with another key, and
sync then refuses to write or prune anything in the bucket.
take-over names the host that writes the bucket, writing nothing; it exits 4
when that is another host and this one never wrote it, and 5 when another
host took the bucket over from this one. A host built from another host's
backup (a rehearsal, or a rebuild) copies nothing into that host's bucket
until -yes makes it the writer; the host it replaces then stops copying and
says so.
`
// defaultOffsiteEnvFile is where bootstrap keeps the [offsite] secrets; the
@@ -87,10 +74,6 @@ func cmdOffsite(args []string, stdout, stderr io.Writer) int {
return offsiteFetchImages(fs, rest, stdout, stderr)
case "fetch-uploads":
return offsiteFetchUploads(fs, rest, stdout, stderr)
case "check-key":
return offsiteCheckKey(fs, rest, stdout, stderr)
case "take-over":
return offsiteTakeOver(fs, rest, stdout, stderr)
case "keygen":
k, err := offsite.NewKey()
if err != nil {
@@ -211,7 +194,6 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory bundled into the database bundle taken after archives are copied ("" for none)`)
registry := fs.String("registry", "", `host[:port] of the registry whose user images are copied (default: the in-cluster registry's loopback hostPort; "off" copies none)`)
uploadsDir := fs.String("uploads-dir", "", "host directory of the submission uploads volume (default: resolved from the uploads PVC through the cluster)")
uploadsPVC := fs.String("uploads-pvc", platform.UploadsPVCName, `the submission uploads PVC, in the control-plane namespace ("" copies no uploads)`)
@@ -224,32 +206,33 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
return 1
}
st, lease := startRun(env.cfg, env.key, *statusFile, time.Now())
st := offsite.Status{
LastAttempt: time.Now().UTC(), Endpoint: env.cfg.Endpoint, Bucket: env.cfg.Bucket,
Prefix: env.cfg.Prefix, KeyID: offsite.KeyID(env.key),
}
if prev, _ := offsite.ReadStatus(*statusFile); prev != nil {
st.LastSuccess = prev.LastSuccess
}
res, err := runOffsiteSync(cfg, env, offsiteSources{
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir, stateDir: *stateDir,
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir,
registry: offsiteRegistryEndpoint(*registry, cfg.Registry),
uploadsDir: *uploadsDir, uploadsPVC: *uploadsPVC,
}, &lease, stderr)
recordRun(&st, res, err, lease)
}, stderr)
st.Result = res
if err != nil {
st.LastError = err.Error()
} else {
st.LastSuccess = st.LastAttempt
}
if werr := offsite.WriteStatus(*statusFile, st); werr != nil {
fmt.Fprintf(stderr, "felis offsite sync: record status: %v\n", werr)
}
return reportRun(res, err, stdout, stderr)
}
// reportRun prints one pass's outcome and its exit code. A run stopped before
// it copied anything (a bucket that did not answer, another key's objects,
// another host writing the bucket) prints no counts: its zeros would read as
// an empty bucket.
func reportRun(res offsite.Result, err error, stdout, stderr io.Writer) int {
if err == nil || len(res.Errors) > 0 {
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; images copied=%d blobs=%d pruned=%d; uploads copied=%d pruned=%d; bucket holds %d worlds (%s), %d bundles, %d images in %d repositories (%s), %d uploads (%s)\n",
res.WorldsUploaded, res.WorldsPending, len(res.WorldsMissing), res.WorldsExpired,
res.DBUploaded, res.DBPruned, res.ImagesUploaded, res.ImageBlobsUploaded, res.ImageObjectsPruned,
res.UploadsUploaded, res.UploadObjectsPruned,
res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB, res.Images, res.ImageRepos, offsite.HumanBytes(res.RemoteImageBytes),
res.Uploads, offsite.HumanBytes(res.RemoteUploadBytes))
}
for _, m := range res.WorldsMissing {
fmt.Fprintf(stderr, "felis offsite sync: recorded archive not on the volume, nothing to copy: %s\n", m)
}
@@ -263,62 +246,19 @@ func reportRun(res offsite.Result, err error, stdout, stderr io.Writer) int {
return 0
}
// startRun begins a pass: its status record, in this release's format and
// carrying the last success over, and this host's lease, read from the record
// the last pass left.
func startRun(cfg config.OffsiteConfig, key []byte, statusFile string, now time.Time) (offsite.Status, offsite.Lease) {
st := offsite.Status{
LastAttempt: now.UTC(), Endpoint: cfg.Endpoint, Bucket: cfg.Bucket,
Prefix: cfg.Prefix, KeyID: offsite.KeyID(key), Format: offsite.StatusFormat,
}
if prev, _ := offsite.ReadStatus(statusFile); prev != nil {
st.LastSuccess = prev.LastSuccess
}
return st, offsite.HostLease(statusFile)
}
// recordRun puts one pass's outcome into its status record. A host that
// inherited the bucket from an older release keeps that until it has an id.
func recordRun(st *offsite.Status, res offsite.Result, err error, lease offsite.Lease) {
st.Result = res
if err != nil {
st.LastError = err.Error()
st.KeyMismatch = errors.Is(err, offsite.ErrKeyMismatch)
var we *offsite.WriterError
if errors.As(err, &we) {
st.Standby = errors.Is(err, offsite.ErrStandby)
st.Displaced = errors.Is(err, offsite.ErrDisplaced)
st.Writer = we.Writer
}
} else {
st.LastSuccess = st.LastAttempt
}
if lease.Inherited {
id, _ := lease.ID()
st.Inherited = id == ""
}
}
// offsiteSources is where one sync pass reads from: the world archive volume
// (archiveDir, or the backupPVC's directory), the bundle directory (with the
// host state the pass bundles, stateDir), the registry's loopback endpoint and
// the uploads volume (uploadsDir, or the uploadsPVC's directory). An empty
// source is skipped.
// (archiveDir, or the backupPVC's directory), the bundle directory, the
// registry's loopback endpoint and the uploads volume (uploadsDir, or the
// uploadsPVC's directory). An empty source is skipped.
type offsiteSources struct {
archiveDir, backupPVC string
dbDir, stateDir string
dbDir string
registry string
uploadsDir, uploadsPVC string
}
// offsiteRunLimit backstops one sync pass. Each upload has its own deadline,
// scaled to its size (internal/offsite), so a pass over a big archive may run
// for hours; the timer starts no second pass while one runs, and the unit's
// TimeoutStartSec sits above this.
const offsiteRunLimit = 23 * time.Hour
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, lease *offsite.Lease, log io.Writer) (offsite.Result, error) {
ctx, cancel := context.WithTimeout(context.Background(), offsiteRunLimit)
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, log io.Writer) (offsite.Result, error) {
ctx, cancel := context.WithTimeout(context.Background(), 50*time.Minute)
defer cancel()
checkCtx, checkCancel := context.WithTimeout(ctx, 30*time.Second)
err := env.bucket.Check(checkCtx)
@@ -348,8 +288,10 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, lea
return offsite.Result{}, fmt.Errorf("open database: %w", err)
}
defer drv.Close()
s := offsiteSyncer(cfg, env, src, archiveDir, uploadsDir, lease, log)
s.Catalog = offsite.PGCatalog{DB: drv.DB()}
s := &offsite.Syncer{
Bucket: env.bucket, Catalog: offsite.PGCatalog{DB: drv.DB()}, Key: env.key,
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Log: log,
}
if src.registry != "" {
s.Images = newRegistryImages(src.registry)
s.ImagePins = imagePins(drv.DB(), cfg.Registry.URL)
@@ -357,55 +299,6 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, lea
return s.Run(ctx)
}
// offsiteSyncer is the pass runOffsiteSync runs over the resolved archive and
// uploads directories, before its catalog and registry are attached. It
// snapshots the database into the bundle directory after copying archives, and
// sweeps world objects no backup records once they outlive every retention in
// [archive]; a retention that does not parse sweeps none.
func offsiteSyncer(cfg *config.Config, env *offsiteEnv, src offsiteSources, archiveDir, uploadsDir string, lease *offsite.Lease, log io.Writer) *offsite.Syncer {
s := &offsite.Syncer{
Bucket: env.bucket, Key: env.key,
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Lease: lease, Log: log,
}
if src.dbDir != "" {
s.Snapshot = offsiteSnapshot(cfg.Database, src.dbDir, src.stateDir, log)
}
if rc, err := reaperConfig(cfg); err != nil {
fmt.Fprintf(log, "felis offsite: world objects no backup records are kept: %v\n", err)
} else {
s.OrphanAfter = max(rc.Retention, rc.ManualRetention, rc.ScheduledRetention)
}
return s
}
// offsiteSnapshot takes the bundle a pass sends after copying world archives
// (offsite.Syncer.Snapshot): what `felis db backup` takes, labelled offsite,
// with the newest one kept in dir. It is not recorded for the panel, whose
// backup card watches felis-db-backup.timer: snapshots come only when archives
// are copied, and would hide a daily timer that stopped. It requires the
// MinecraftServer objects: it becomes the newest bundle in the bucket, which a
// lost host restores from, and a pass that cannot take a whole one fails and
// tries again next hour.
func offsiteSnapshot(db config.DatabaseConfig, dir, stateDir string, log io.Writer) func(context.Context) error {
return func(ctx context.Context) error {
tools, err := dbTools(db)
if err != nil {
return err
}
ctx, cancel := context.WithTimeout(ctx, 30*time.Minute)
defer cancel()
path, err := dbbackup.Backup(ctx, dbbackup.BackupOptions{
DatabaseURL: db.URL, Tools: tools, Dir: dir, Label: dbbackup.LabelOffsite,
Keep: defaultKeep[dbbackup.LabelOffsite], StateDir: stateDir, Version: resolvedVersion(),
ExportServers: exportMinecraftServers, RequireServers: true, Log: log,
})
if err == nil {
fmt.Fprintf(log, "felis offsite: took database bundle %s, which lists the archives just copied\n", filepath.Base(path))
}
return err
}
}
// volumeKind names a PVC the off-site copy reads or restores, for messages,
// with the flag that bypasses finding it through the cluster.
type volumeKind struct{ what, dirFlag, empty string }
@@ -539,27 +432,6 @@ func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) in
if st.LastError != "" {
fmt.Fprintf(stdout, "last error: %s\n", st.LastError)
}
if st.KeyMismatch {
fmt.Fprintf(stdout, "\nThe last run was refused: the bucket's objects are sealed with another key than this host's (key id %s). No sync copies or prunes anything there until FELIS_OFFSITE_KEY in %s is theirs (sudo felis offsite check-key).\n", st.KeyID, defaultOffsiteEnvFile)
return 1
}
if st.Displaced && st.Writer != nil {
fmt.Fprintf(stdout, "\nThe last run was refused: %s took the bucket over (it last wrote it at %s), and this host copies nothing there any more. If that host is a rehearsal machine, take the bucket back: sudo felis offsite take-over -yes\n",
st.Writer, st.Writer.At.Local().Format(time.DateTime))
return 1
}
if st.Standby {
switch w := st.StandsBy(now); {
case w != nil:
fmt.Fprintf(stdout, "\nThis host stands by: %s writes the bucket (last at %s). This host was built from its backup, copies nothing into the bucket and, while that host keeps writing, mails no watchdog alert.", w, w.At.Local().Format(time.DateTime))
case st.Writer != nil:
fmt.Fprintf(stdout, "\nThis host copies nothing into the bucket: %s wrote it, last at %s, and this host was built from its backup.", st.Writer, st.Writer.At.Local().Format(time.DateTime))
default:
fmt.Fprint(stdout, "\nThis host copies nothing into the bucket: it holds copies this host did not write, and names no host writing it.")
}
fmt.Fprintln(stdout, " Once this host replaces that one for good: sudo felis offsite take-over -yes")
return 1
}
r := st.Result
fmt.Fprintf(stdout, "bucket holds: %d world archives (%s), %d database bundles, newest %s\n",
r.RemoteWorlds, offsite.HumanBytes(r.RemoteBytes), r.RemoteDB, orNone(r.NewestDB))
@@ -623,7 +495,10 @@ func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
return 1
}
printDBBundles(ctx, env.bucket, env.key, bundles, stdout)
fmt.Fprintf(stdout, "database bundles (%d, newest first):\n", len(bundles))
for _, b := range bundles {
fmt.Fprintf(stdout, " %s %s\n", b.Key, offsite.HumanBytes(b.Size))
}
var total int64
for _, w := range worlds {
total += w.Size
@@ -660,164 +535,6 @@ func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
return 0
}
// printDBBundles lists the database bundles with what each one's database
// held, read off the front of each, so a restore can pick one by its contents.
func printDBBundles(ctx context.Context, b offsite.Bucket, key []byte, bundles []offsite.Object, stdout io.Writer) {
fmt.Fprintf(stdout, "database bundles (%d, newest first; restore one with fetch-db):\n", len(bundles))
for _, o := range bundles {
m, err := offsite.PeekDB(ctx, b, key, o.Key)
if err != nil {
fmt.Fprintf(stdout, " %s %s unreadable: %v\n", o.Key, offsite.HumanBytes(o.Size), err)
continue
}
gap := ""
if m.ServersError != "" {
gap = ", no MinecraftServer objects"
}
fmt.Fprintf(stdout, " %s %s %s%s\n", o.Key, offsite.HumanBytes(o.Size), m.Counts.String(), gap)
}
}
func offsiteCheckKey(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
if err := fs.Parse(args); err != nil {
return 2
}
_, env, err := loadOffsite(*cfgPath, *envFile)
if err != nil {
fmt.Fprintf(stderr, "felis offsite check-key: %v\n", err)
return 1
}
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
if err := env.bucket.Check(ctx); err != nil {
fmt.Fprintf(stderr, "felis offsite check-key: %v\n", err)
return 1
}
return checkKey(ctx, env.bucket, env.key, stdout, stderr)
}
// checkKey is check-key once the bucket is open: 0 when the key fits, 3 when
// the bucket's objects are sealed with another one, 1 when it cannot tell.
func checkKey(ctx context.Context, b offsite.Bucket, key []byte, stdout, stderr io.Writer) int {
fit, err := offsite.CheckKey(ctx, b, key)
if err != nil {
fmt.Fprintf(stderr, "felis offsite check-key: %v\n", err)
if errors.Is(err, offsite.ErrKeyMismatch) {
return 3
}
return 1
}
id := offsite.KeyID(key)
switch fit {
case offsite.KeyRecorded:
fmt.Fprintf(stdout, "felis offsite check-key: the bucket records key id %s, this key's\n", id)
case offsite.KeyOpens:
fmt.Fprintf(stdout, "felis offsite check-key: the bucket's newest objects open with this key (key id %s); the next sync records it\n", id)
case offsite.KeyUnused:
fmt.Fprintf(stdout, "felis offsite check-key: the bucket holds no sealed object yet; the first sync records key id %s\n", id)
}
return 0
}
func offsiteTakeOver(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "the record `sync` writes; this host's id is kept next to it")
yes := fs.Bool("yes", false, "make this host the one that writes the bucket")
if err := fs.Parse(args); err != nil {
return 2
}
_, env, err := loadOffsite(*cfgPath, *envFile)
if err != nil {
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
return 1
}
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
if err := env.bucket.Check(ctx); err != nil {
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
return 1
}
return takeOver(ctx, env.bucket, env.key, offsite.HostLease(*statusFile), *statusFile, *yes, time.Now(), stdout, stderr)
}
// takeOver is take-over once the bucket is open: without yes it says which
// host writes the bucket, 0 for this one (or none yet), 4 for another and 5
// for one that took the bucket over from this host; with
// yes it records this host as the writer. A key the bucket's objects refuse
// is 3, as in check-key: taking over a bucket this host cannot copy into
// would only stop the host that can.
func takeOver(ctx context.Context, b offsite.Bucket, key []byte, lease offsite.Lease, statusFile string, yes bool, now time.Time, stdout, stderr io.Writer) int {
fit, err := offsite.CheckKey(ctx, b, key)
if err != nil {
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
if errors.Is(err, offsite.ErrKeyMismatch) {
return 3
}
return 1
}
role, w, err := lease.Plan(ctx, b, fit == offsite.KeyUnused)
if err != nil {
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
return 1
}
switch role {
case offsite.RoleWrites:
id, _ := lease.ID()
fmt.Fprintf(stdout, "felis offsite take-over: this host (id %s) writes the bucket; nothing to take over\n", id)
return 0
case offsite.RoleClaims:
fmt.Fprintln(stdout, "felis offsite take-over: the bucket names no host writing it; this host's next sync records itself")
return 0
}
who := "another host"
if w != nil {
who = w.String()
fmt.Fprintf(stdout, "felis offsite take-over: %s writes the bucket, last at %s (%s ago)\n", w, w.At.Local().Format(time.DateTime), dbbackup.Age(now.Sub(w.At)))
} else {
fmt.Fprintln(stdout, "felis offsite take-over: the bucket holds copies this host did not write, and names no host writing it")
}
if !yes {
if role == offsite.RoleDisplaced {
fmt.Fprintf(stdout, "It took the bucket over from this host: this host copies nothing there any more, and its watchdog mails the owners about it. If that host is a rehearsal machine, take the bucket back:\n sudo felis offsite take-over -yes\n")
return 5
}
fmt.Fprintf(stdout, "This host was built from its backup and copies nothing into the bucket.\n")
fmt.Fprintf(stdout, "Taking it over makes this host the one that copies into the bucket and prunes it; %s stops at its next copy and mails its owners. Do it once that host is gone for good, or is a rehearsal machine you are done with:\n sudo felis offsite take-over -yes\n", who)
return 4
}
if _, err := lease.TakeOver(ctx, b, now); err != nil {
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
return 1
}
// The refusal the last sync recorded is over: the watchdog mails again
// from now on, and status shows the next run's outcome.
if st, err := offsite.ReadStatus(statusFile); err == nil && st != nil && (st.Standby || st.Displaced) {
st.Standby, st.Displaced, st.Writer, st.LastError, st.Inherited = false, false, nil, "", false
if err := offsite.WriteStatus(statusFile, *st); err != nil {
fmt.Fprintf(stderr, "felis offsite take-over: record status: %v\n", err)
}
}
id, _ := lease.ID()
fmt.Fprintf(stdout, "felis offsite take-over: this host (id %s) writes the bucket now; %s stops at its next copy.\nStart the first copy: sudo systemctl start felis-offsite.service\n", id, who)
return 0
}
// keyHint explains an object the key cannot open when the bucket records
// another key's id, "" otherwise.
func keyHint(ctx context.Context, b offsite.Bucket, key []byte, err error) string {
if !errors.Is(err, offsite.ErrAuth) {
return ""
}
id, ierr := offsite.BucketKeyID(ctx, b)
if ierr != nil || id == "" || id == offsite.KeyID(key) {
return ""
}
return fmt.Sprintf("\n the bucket records key id %s, and this key is %s: set FELIS_OFFSITE_KEY to the key the bucket was written with", id, offsite.KeyID(key))
}
func offsiteFetchDB(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml; on a host with no install yet, give -endpoint and -bucket instead")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
@@ -860,47 +577,37 @@ func offsiteFetchDB(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) i
}
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
defer cancel()
return fetchDB(ctx, env.bucket, env.key, arg, *dir, time.Now(), stdout, stderr)
}
// fetchDB is fetch-db once the bucket is open: arg is a bundle name or latest.
func fetchDB(ctx context.Context, b offsite.Bucket, key []byte, arg, dir string, now time.Time, stdout, stderr io.Writer) int {
name := arg
if name == "latest" {
var err error
if name, _, err = offsite.ChooseDB(ctx, b, key); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v%s\n", err, keyHint(ctx, b, key, err))
bundles, err := offsite.ListDB(ctx, env.bucket)
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
if len(bundles) == 0 {
fmt.Fprintln(stderr, "felis offsite fetch-db: the bucket holds no database bundle")
return 1
}
name = bundles[0].Key
}
if _, _, ok := dbbackup.ParseBundleName(name); !ok {
fmt.Fprintf(stderr, "felis offsite fetch-db: %q is not a bundle name (felis-db-<stamp>-<label>.tar); see `felis offsite list`\n", name)
return 2
}
if err := os.MkdirAll(dir, 0o700); err != nil {
if err := os.MkdirAll(*dir, 0o700); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
dst := filepath.Join(dir, name)
if err := offsite.FetchObject(ctx, b, key, offsite.DBKey(name), dst, 0o600); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v%s\n", err, keyHint(ctx, b, key, err))
dst := filepath.Join(*dir, name)
if err := offsite.FetchObject(ctx, env.bucket, env.key, offsite.DBKey(name), dst, 0o600); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
m, err := dbbackup.Verify(dst)
if err != nil {
if _, err := dbbackup.Verify(dst); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: fetched %s but it does not verify: %v\n", dst, err)
return 1
}
fmt.Fprintf(stdout, "felis offsite fetch-db: wrote %s (verified)\n", dst)
fmt.Fprintf(stdout, " taken %s (%s, %s ago)\n felis %s, schema %d\n holds %s\n",
m.CreatedAt.Format(time.RFC3339), m.Label, dbbackup.Age(now.Sub(m.CreatedAt)),
orUnknown(m.FelisVersion), m.SchemaVersion, m.Counts.String())
if m.Counts.Fresh() {
fmt.Fprintln(stdout, " This database holds no servers and at most one account, like a new install's. Check it is the state to restore before `felis db restore`.")
}
if m.ServersError != "" {
fmt.Fprintf(stdout, " This bundle lacks the MinecraftServer objects (%s): `felis db restore` brings back the database, and the servers come from k8s/minecraftservers.json in the newest bundle `felis offsite list` shows without that gap.\n", m.ServersError)
}
return 0
}
-653
View File
@@ -1,24 +1,14 @@
package main
import (
"archive/tar"
"bytes"
"context"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/imagepush"
"felis.lolicon.best/internal/offsite"
)
@@ -130,646 +120,3 @@ func TestRegistryGoneMarksNotFound(t *testing.T) {
t.Fatalf("503 = %v, want it kept an ordinary failure", err)
}
}
// mapBucket is an in-memory offsite.Bucket.
type mapBucket map[string][]byte
func (b mapBucket) Put(_ context.Context, key string, r io.Reader, _ int64) error {
data, err := io.ReadAll(r)
b[key] = data
return err
}
func (b mapBucket) Get(_ context.Context, key string) (io.ReadCloser, error) {
data, ok := b[key]
if !ok {
return nil, offsite.ErrNotFound
}
return io.NopCloser(bytes.NewReader(data)), nil
}
func (b mapBucket) List(_ context.Context, prefix string) ([]offsite.Object, error) {
var out []offsite.Object
for k, v := range b {
if strings.HasPrefix(k, prefix) {
out = append(out, offsite.Object{Key: k, Size: int64(len(v))})
}
}
return out, nil
}
func (b mapBucket) Remove(_ context.Context, key string) error {
delete(b, key)
return nil
}
var fetchT0 = time.Date(2026, 9, 20, 3, 30, 0, 0, time.UTC)
// putBundle seals a bundle that verifies, taken daysAgo days before fetchT0,
// into b and returns its name.
func putBundle(t *testing.T, b mapBucket, key []byte, daysAgo int, counts *dbbackup.Counts) string {
t.Helper()
return putBundleWith(t, b, key, daysAgo, counts, "")
}
// putBundleWith is putBundle for a bundle whose server export failed with
// serversError, when that is not empty.
func putBundleWith(t *testing.T, b mapBucket, key []byte, daysAgo int, counts *dbbackup.Counts, serversError string) string {
t.Helper()
created := fetchT0.AddDate(0, 0, -daysAgo)
dump := []byte("PGDMP " + created.String())
sum := sha256.Sum256(dump)
manifest, err := json.Marshal(dbbackup.Manifest{
Format: 1, CreatedAt: created, Label: dbbackup.LabelDaily, FelisVersion: "v1.2.3", SchemaVersion: 21, Counts: counts, ServersError: serversError,
Files: []dbbackup.ManifestEntry{{Name: "db.dump", Size: int64(len(dump)), SHA256: hex.EncodeToString(sum[:]), Mode: 0o600}},
})
if err != nil {
t.Fatal(err)
}
var plain bytes.Buffer
tw := tar.NewWriter(&plain)
for _, f := range []struct {
name string
data []byte
}{{"MANIFEST.json", manifest}, {"db.dump", dump}} {
if err := tw.WriteHeader(&tar.Header{Name: f.name, Mode: 0o600, Size: int64(len(f.data)), Typeflag: tar.TypeReg}); err != nil {
t.Fatal(err)
}
if _, err := tw.Write(f.data); err != nil {
t.Fatal(err)
}
}
if err := tw.Close(); err != nil {
t.Fatal(err)
}
var sealed bytes.Buffer
if err := offsite.Encrypt(&sealed, &plain, key); err != nil {
t.Fatal(err)
}
name := dbbackup.BundleName(created, dbbackup.LabelDaily)
b[offsite.DBKey(name)] = sealed.Bytes()
return name
}
func TestOffsiteFetchDB(t *testing.T) {
rawKey, _ := offsite.NewKey()
key, _ := offsite.ParseKey(rawKey)
now := fetchT0.Add(2 * time.Hour)
fetch := func(b mapBucket, arg string) (dir string, code int, stdout, stderr string) {
dir = t.TempDir()
var out, errb bytes.Buffer
code = fetchDB(context.Background(), b, key, arg, dir, now, &out, &errb)
return dir, code, out.String(), errb.String()
}
fetched := func(t *testing.T, dir string) []string {
t.Helper()
var names []string
entries, _ := os.ReadDir(dir)
for _, e := range entries {
names = append(names, e.Name())
}
return names
}
t.Run("latest skips a rebuilt host's empty bundle", func(t *testing.T) {
b := mapBucket{}
full := putBundle(t, b, key, 3, &dbbackup.Counts{Users: 5, Servers: 3})
empty := putBundle(t, b, key, 0, &dbbackup.Counts{})
dir, code, out, errb := fetch(b, "latest")
if code != 1 || !strings.Contains(errb, empty) || !strings.Contains(errb, full+" (5 accounts, 3 servers)") {
t.Fatalf("exit %d, stdout %q, stderr %q; want a refusal naming %s", code, out, errb, full)
}
if got := fetched(t, dir); len(got) != 0 {
t.Errorf("a refused fetch wrote %v", got)
}
})
t.Run("latest takes the newest bundle and says what it holds", func(t *testing.T) {
b := mapBucket{}
putBundle(t, b, key, 3, &dbbackup.Counts{Users: 5, Servers: 2})
newest := putBundle(t, b, key, 1, &dbbackup.Counts{Users: 5, Servers: 3})
dir, code, out, errb := fetch(b, "latest")
if code != 0 {
t.Fatalf("exit %d: %s", code, errb)
}
if got := fetched(t, dir); !slices.Equal(got, []string{newest}) {
t.Errorf("wrote %v, want %s", got, newest)
}
for _, want := range []string{
"wrote " + filepath.Join(dir, newest) + " (verified)",
"taken 2026-09-19T03:30:00Z (daily, 26h0m ago)",
"felis v1.2.3, schema 21",
"holds 5 accounts, 3 servers",
} {
if !strings.Contains(out, want) {
t.Errorf("stdout lacks %q:\n%s", want, out)
}
}
if strings.Contains(out, "new install") {
t.Errorf("a bundle with servers flagged as a new install's:\n%s", out)
}
})
t.Run("an empty bundle named outright is fetched with a warning", func(t *testing.T) {
b := mapBucket{}
putBundle(t, b, key, 3, &dbbackup.Counts{Users: 5, Servers: 3})
empty := putBundle(t, b, key, 0, &dbbackup.Counts{Users: 1})
dir, code, out, errb := fetch(b, empty)
if code != 0 || !slices.Equal(fetched(t, dir), []string{empty}) {
t.Fatalf("exit %d, wrote %v: %s", code, fetched(t, dir), errb)
}
if !strings.Contains(out, "holds 1 account, 0 servers") || !strings.Contains(out, "like a new install's") {
t.Errorf("stdout = %s", out)
}
})
t.Run("a bundle without the servers says where they come from", func(t *testing.T) {
b := mapBucket{}
gapped := putBundleWith(t, b, key, 1, &dbbackup.Counts{Users: 5, Servers: 3}, "connection refused (tried 3 times)")
_, code, out, errb := fetch(b, gapped)
if code != 0 || !strings.Contains(out, "This bundle lacks the MinecraftServer objects (connection refused (tried 3 times))") ||
!strings.Contains(out, "k8s/minecraftservers.json in the newest bundle `felis offsite list` shows without that gap") {
t.Errorf("exit %d, stdout %q, stderr %q", code, out, errb)
}
whole := putBundle(t, b, key, 0, &dbbackup.Counts{Users: 5, Servers: 3})
if _, _, out, _ := fetch(b, whole); strings.Contains(out, "lacks the MinecraftServer objects") {
t.Errorf("a whole bundle flagged:\n%s", out)
}
})
t.Run("a bundle from before counts says so", func(t *testing.T) {
b := mapBucket{}
old := putBundle(t, b, key, 0, nil)
_, code, out, errb := fetch(b, "latest")
if code != 0 || !strings.Contains(out, old) || !strings.Contains(out, "holds not recorded") || strings.Contains(out, "new install") {
t.Errorf("exit %d, stdout %q, stderr %q", code, out, errb)
}
})
}
func TestOffsiteCheckKey(t *testing.T) {
newKey := func() []byte {
raw, _ := offsite.NewKey()
k, _ := offsite.ParseKey(raw)
return k
}
key, other := newKey(), newKey()
marked := func(k []byte) mapBucket { return mapBucket{"felis-key-id": []byte(offsite.KeyID(k) + "\n")} }
unmarked := func(k []byte) mapBucket {
b := mapBucket{}
putBundle(t, b, k, 0, nil)
return b
}
for _, tc := range []struct {
what string
bucket mapBucket
code int
says []string
}{
{"the recorded key", marked(key), 0, []string{"records key id " + offsite.KeyID(key)}},
{"an unmarked bucket the key opens", unmarked(key), 0, []string{"open with this key", "the next sync records it"}},
{"an empty bucket", mapBucket{}, 0, []string{"no sealed object yet", "records key id " + offsite.KeyID(key)}},
{"another recorded key", marked(other), 3, []string{offsite.KeyID(other), offsite.KeyID(key), "FELIS_OFFSITE_KEY"}},
{"an unmarked bucket under another key", unmarked(other), 3, []string{"opens none of db/felis-db-"}},
{"a marker Felis did not write", mapBucket{"felis-key-id": []byte("hello")}, 1, []string{"not a key id"}},
} {
t.Run(tc.what, func(t *testing.T) {
before := len(tc.bucket)
var out, errb bytes.Buffer
code := checkKey(context.Background(), tc.bucket, key, &out, &errb)
if code != tc.code {
t.Fatalf("exit %d, want %d; stdout %q, stderr %q", code, tc.code, out.String(), errb.String())
}
said := out.String() + errb.String()
for _, s := range tc.says {
if !strings.Contains(said, s) {
t.Errorf("output lacks %q: %s", s, said)
}
}
if len(tc.bucket) != before {
t.Errorf("check-key wrote to the bucket: %d objects, had %d", len(tc.bucket), before)
}
})
}
// fetch-db names both ids when the bucket records another key.
b := marked(other)
name := putBundle(t, b, other, 0, &dbbackup.Counts{Users: 5, Servers: 3})
for _, arg := range []string{"latest", name} {
var out, errb bytes.Buffer
if code := fetchDB(context.Background(), b, key, arg, t.TempDir(), fetchT0, &out, &errb); code != 1 ||
!strings.Contains(errb.String(), "the bucket records key id "+offsite.KeyID(other)+", and this key is "+offsite.KeyID(key)) {
t.Errorf("fetch-db %s under another key: exit %d, stderr %q", arg, code, errb.String())
}
}
// A bundle the bucket lacks is not the key's fault.
var missOut, missErr bytes.Buffer
if code := fetchDB(context.Background(), b, key, "felis-db-20200101T000000Z-daily.tar", t.TempDir(), fetchT0, &missOut, &missErr); code != 1 || strings.Contains(missErr.String(), "records key id") {
t.Errorf("missing bundle: exit %d, stderr %q", code, missErr.String())
}
// A bundle damaged under the recorded key gets no such hint.
b = marked(key)
name = putBundle(t, b, key, 0, &dbbackup.Counts{Users: 5, Servers: 3})
b[offsite.DBKey(name)][60] ^= 1
var out, errb bytes.Buffer
if code := fetchDB(context.Background(), b, key, name, t.TempDir(), fetchT0, &out, &errb); code != 1 || strings.Contains(errb.String(), "records key id") {
t.Errorf("damaged bundle: exit %d, stderr %q", code, errb.String())
}
}
func TestRecordRun(t *testing.T) {
t0 := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: t0}
for _, tc := range []struct {
err error
mismatch, standby, displaced, success bool
}{
{nil, false, false, false, true},
{errors.New("list worlds/ in the bucket: connection reset"), false, false, false, false},
{fmt.Errorf("%w: the bucket records key id 0123456789abcdef", offsite.ErrKeyMismatch), true, false, false, false},
{&offsite.WriterError{Kind: offsite.ErrStandby, Writer: w}, false, true, false, false},
{&offsite.WriterError{Kind: offsite.ErrDisplaced, Writer: w}, false, false, true, false},
} {
st := offsite.Status{LastAttempt: t0}
recordRun(&st, offsite.Result{RemoteDB: 2}, tc.err, offsite.Lease{})
if st.KeyMismatch != tc.mismatch || st.Standby != tc.standby || st.Displaced != tc.displaced ||
(st.Writer != nil) != (tc.standby || tc.displaced) || st.LastSuccess.Equal(t0) != tc.success || st.Result.RemoteDB != 2 {
t.Errorf("err %v: status %+v", tc.err, st)
}
}
// A host that copied before writers were recorded keeps that claim over
// failed runs until it has an id.
l := offsite.Lease{IDFile: filepath.Join(t.TempDir(), offsite.HostIDFile), Inherited: true}
st := offsite.Status{}
recordRun(&st, offsite.Result{}, errors.New("cannot reach bucket"), l)
if !st.Inherited {
t.Error("a failed run on a host with no id dropped the older release's claim")
}
writeTestFile(t, l.IDFile, "aaaaaaaaaaaaaaaa\n", 0o600)
st = offsite.Status{Inherited: true}
recordRun(&st, offsite.Result{}, nil, l)
if st.Inherited {
t.Error("a host with an id still carries the older release's claim")
}
}
// A refused run has copied nothing and listed nothing: its zero counts would
// tell the journal the bucket is empty. A pass that ran and failed a step
// shows what it did get to.
func TestReportRun(t *testing.T) {
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)}
for _, tc := range []struct {
name string
res offsite.Result
err error
code int
counts bool
}{
{"a pass", offsite.Result{DBUploaded: 1, RemoteDB: 3}, nil, 0, true},
{"a pass with a failed step", offsite.Result{RemoteDB: 3, Errors: []string{"copy db/x: timeout"}}, errors.New("1 of this run's steps failed; first: copy db/x: timeout"), 1, true},
{"standing by", offsite.Result{}, &offsite.WriterError{Kind: offsite.ErrStandby, Writer: w}, 1, false},
{"another key", offsite.Result{}, fmt.Errorf("%w: the bucket records key id 1111111111111111", offsite.ErrKeyMismatch), 1, false},
{"no bucket", offsite.Result{}, errors.New("bucket: access denied"), 1, false},
} {
t.Run(tc.name, func(t *testing.T) {
var out, errOut bytes.Buffer
code := reportRun(tc.res, tc.err, &out, &errOut)
if code != tc.code {
t.Errorf("exit %d, want %d", code, tc.code)
}
if got := strings.Contains(out.String(), "bucket holds 0 worlds (0 B), 3 bundles"); got != tc.counts {
t.Errorf("counts shown = %v, want %v: %q", got, tc.counts, out.String())
}
if !tc.counts && out.Len() > 0 {
t.Errorf("a refused run printed %q", out.String())
}
if tc.err != nil && !strings.Contains(errOut.String(), "felis offsite sync: "+tc.err.Error()) {
t.Errorf("stderr %q lacks the error", errOut.String())
}
})
}
}
func TestOffsiteTakeOver(t *testing.T) {
newKey := func() []byte {
raw, _ := offsite.NewKey()
k, _ := offsite.ParseKey(raw)
return k
}
key, other := newKey(), newKey()
const mine, theirs = "aaaaaaaaaaaaaaaa", "bbbbbbbbbbbbbbbb"
t0 := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
sealed := func(k []byte, writer string) mapBucket {
b := mapBucket{}
putBundle(t, b, k, 0, nil)
b["felis-key-id"] = []byte(offsite.KeyID(k) + "\n")
if writer != "" {
raw, _ := json.Marshal(offsite.Writer{HostID: writer, Host: "prod-1", At: t0.Add(-20 * time.Minute)})
b["felis-writer"] = raw
}
return b
}
lease := func(id string) offsite.Lease {
l := offsite.Lease{IDFile: filepath.Join(t.TempDir(), offsite.HostIDFile), Host: "spare-1"}
if id != "" {
writeTestFile(t, l.IDFile, id+"\n", 0o600)
}
return l
}
run := func(b mapBucket, l offsite.Lease, statusFile string, yes bool) (int, string, string) {
t.Helper()
var out, errb bytes.Buffer
code := takeOver(context.Background(), b, key, l, statusFile, yes, t0, &out, &errb)
return code, out.String(), errb.String()
}
for _, tc := range []struct {
what string
bucket mapBucket
id string
code int
says []string
}{
{"this host writes the bucket", sealed(key, mine), mine, 0, []string{"this host (id " + mine + ") writes the bucket; nothing to take over"}},
{"an empty bucket", mapBucket{}, "", 0, []string{"names no host writing it; this host's next sync records itself"}},
{"a host built from the writer's backup", sealed(key, theirs), "", 4, []string{"host prod-1 (id " + theirs + ") writes the bucket, last at", "(20m ago)", "built from its backup", "sudo felis offsite take-over -yes"}},
{"another host's copies, no writer named", sealed(key, ""), "", 4, []string{"holds copies this host did not write, and names no host writing it", "take-over -yes"}},
{"a host another one took over from", sealed(key, theirs), mine, 5, []string{"took the bucket over from this host"}},
{"a key the bucket refuses", sealed(other, theirs), "", 3, []string{"sealed with another key"}},
} {
t.Run(tc.what, func(t *testing.T) {
before := string(tc.bucket["felis-writer"])
l := lease(tc.id)
code, out, errb := run(tc.bucket, l, filepath.Join(t.TempDir(), "status.json"), false)
if code != tc.code {
t.Fatalf("exit %d, want %d\n%s%s", code, tc.code, out, errb)
}
for _, s := range tc.says {
if !strings.Contains(out+errb, s) {
t.Errorf("output lacks %q:\n%s%s", s, out, errb)
}
}
if string(tc.bucket["felis-writer"]) != before {
t.Error("take-over without -yes wrote the bucket's writer")
}
if tc.id == "" {
if _, err := os.Stat(l.IDFile); !errors.Is(err, os.ErrNotExist) {
t.Errorf("take-over without -yes made this host an id: %v", err)
}
}
})
}
// -yes on a standby host: the bucket names it, and the refusal the last
// sync recorded is cleared, so the watchdog mails again at once.
b := sealed(key, theirs)
l := lease("")
statusFile := filepath.Join(t.TempDir(), "status.json")
lastSuccess := t0.Add(-48 * time.Hour)
if err := offsite.WriteStatus(statusFile, offsite.Status{
LastAttempt: t0.Add(-time.Hour), LastSuccess: lastSuccess, LastError: "offsite: another host writes this bucket", Format: offsite.StatusFormat,
Standby: true, Writer: &offsite.Writer{HostID: theirs, Host: "prod-1", At: t0}, Inherited: true,
}); err != nil {
t.Fatal(err)
}
code, out, errb := run(b, l, statusFile, true)
id, _ := l.ID()
if code != 0 || id == "" || !strings.Contains(out, "this host (id "+id+") writes the bucket now; host prod-1 (id "+theirs+") stops at its next copy") || !strings.Contains(out, "systemctl start felis-offsite.service") {
t.Fatalf("take-over -yes: exit %d, id %q\n%s%s", code, id, out, errb)
}
if w, err := offsite.BucketWriter(context.Background(), b); err != nil || w.HostID != id || w.Host != "spare-1" || !w.At.Equal(t0) {
t.Errorf("writer after take-over -yes = %+v, %v", w, err)
}
st, err := offsite.ReadStatus(statusFile)
if err != nil || st.Standby || st.Writer != nil || st.LastError != "" || st.Inherited || !st.LastSuccess.Equal(lastSuccess) {
t.Errorf("status after take-over -yes = %+v, %v; want the refusal cleared and the last success kept", st, err)
}
// -yes with a key the bucket refuses writes nothing.
b = sealed(other, theirs)
before := string(b["felis-writer"])
if code, _, _ := run(b, lease(""), filepath.Join(t.TempDir(), "status.json"), true); code != 3 || string(b["felis-writer"]) != before {
t.Errorf("take-over -yes under another key: exit %d, writer %s", code, b["felis-writer"])
}
}
func TestOffsiteStatusSaysWhoWritesTheBucket(t *testing.T) {
dir := t.TempDir()
cfg := filepath.Join(dir, "felis.toml")
writeTestFile(t, cfg, installerTOML("example.com", "127.0.0.1")+"\n[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis-backups\"\n", 0o600)
statusFile := filepath.Join(dir, "status.json")
now := time.Now()
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: now.Add(-30 * time.Minute)}
stale := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: now.Add(-offsite.WriterLive - time.Hour)}
for _, tc := range []struct {
what string
st offsite.Status
says []string
not string
}{
{"standing by for a live writer", offsite.Status{Standby: true, Writer: w}, []string{"This host stands by: host prod-1 (id bbbbbbbbbbbbbbbb) writes the bucket", "mails no watchdog alert", "take-over -yes"}, "wrote it, last at"},
{"standing by for a writer gone quiet", offsite.Status{Standby: true, Writer: stale}, []string{"copies nothing into the bucket: host prod-1 (id bbbbbbbbbbbbbbbb) wrote it, last at", "take-over -yes"}, "mails no watchdog alert"},
{"standing by, no writer named", offsite.Status{Standby: true}, []string{"names no host writing it", "take-over -yes"}, "stands by:"},
{"displaced", offsite.Status{Displaced: true, Writer: w}, []string{"host prod-1 (id bbbbbbbbbbbbbbbb) took the bucket over", "rehearsal machine", "take-over -yes"}, "stands by"},
} {
t.Run(tc.what, func(t *testing.T) {
tc.st.LastAttempt, tc.st.LastSuccess, tc.st.LastError = now.Add(-time.Minute), now.Add(-time.Hour), "offsite: another host writes this bucket"
if err := offsite.WriteStatus(statusFile, tc.st); err != nil {
t.Fatal(err)
}
var out, errb bytes.Buffer
code := cmdOffsite([]string{"status", "-config", cfg, "-status-file", statusFile}, &out, &errb)
if code != 1 || strings.Contains(out.String(), "bucket holds:") || strings.Contains(out.String(), tc.not) {
t.Errorf("exit %d\n%s%s", code, out.String(), errb.String())
}
for _, s := range tc.says {
if !strings.Contains(out.String(), s) {
t.Errorf("output lacks %q:\n%s", s, out.String())
}
}
})
}
}
func TestOffsiteStatusSaysTheKeyWasRefused(t *testing.T) {
dir := t.TempDir()
cfg := filepath.Join(dir, "felis.toml")
writeTestFile(t, cfg, installerTOML("example.com", "127.0.0.1")+"\n[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis-backups\"\n", 0o600)
statusFile := filepath.Join(dir, "status.json")
st := offsite.Status{LastAttempt: time.Now().Add(-time.Minute), LastSuccess: time.Now().Add(-time.Hour), KeyID: "0123456789abcdef"}
status := func() (int, string) {
t.Helper()
if err := offsite.WriteStatus(statusFile, st); err != nil {
t.Fatal(err)
}
var out, errb bytes.Buffer
code := cmdOffsite([]string{"status", "-config", cfg, "-status-file", statusFile}, &out, &errb)
return code, out.String() + errb.String()
}
if code, out := status(); code != 0 || strings.Contains(out, "refused") {
t.Fatalf("a recent success: exit %d\n%s", code, out)
}
st.LastError, st.KeyMismatch = "offsite: the bucket's objects are sealed with another key", true
code, out := status()
if code != 1 || !strings.Contains(out, "The last run was refused") || !strings.Contains(out, "key id 0123456789abcdef") || strings.Contains(out, "bucket holds:") {
t.Fatalf("a refused run: exit %d\n%s", code, out)
}
}
func TestPrintDBBundlesSaysWhatEachHolds(t *testing.T) {
rawKey, _ := offsite.NewKey()
key, _ := offsite.ParseKey(rawKey)
otherRaw, _ := offsite.NewKey()
other, _ := offsite.ParseKey(otherRaw)
b := mapBucket{}
gapped := putBundleWith(t, b, key, 4, &dbbackup.Counts{Users: 5, Servers: 3}, "connection refused")
old := putBundle(t, b, key, 3, nil)
full := putBundle(t, b, key, 2, &dbbackup.Counts{Users: 5, Servers: 3})
sealedElsewhere := putBundle(t, b, other, 1, &dbbackup.Counts{Users: 5, Servers: 3})
empty := putBundle(t, b, key, 0, &dbbackup.Counts{})
bundles, err := offsite.ListDB(context.Background(), b)
if err != nil {
t.Fatal(err)
}
var out bytes.Buffer
printDBBundles(context.Background(), b, key, bundles, &out)
lines := strings.Split(strings.TrimSpace(out.String()), "\n")
want := []struct{ name, holds string }{
{empty, "0 accounts, 0 servers"},
{sealedElsewhere, "unreadable: offsite: object does not decrypt with this key"},
{full, "5 accounts, 3 servers"},
{old, "not recorded"},
{gapped, "5 accounts, 3 servers, no MinecraftServer objects"},
}
if len(lines) != len(want)+1 || !strings.HasPrefix(lines[0], "database bundles (5, newest first") {
t.Fatalf("output:\n%s", out.String())
}
for i, w := range want {
if l := lines[i+1]; !strings.HasPrefix(l, " "+w.name+" ") || !strings.Contains(l, w.holds) || strings.Contains(l, "no MinecraftServer") != (w.name == gapped) {
t.Errorf("line %d = %q, want %s with %q", i+1, l, w.name, w.holds)
}
}
}
// TestRestoredHostKeepsStandingBy walks the status file across runs: a host
// restored from the writer's backup stands by on its first run and on every
// run after it, a host an older release left copying claims the bucket once,
// and a failed first run after the upgrade keeps that claim.
func TestRestoredHostKeepsStandingBy(t *testing.T) {
rawKey, _ := offsite.NewKey()
key, _ := offsite.ParseKey(rawKey)
cfg := config.OffsiteConfig{Endpoint: "https://s3.example.com", Bucket: "felis-backups"}
t0 := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
standby := &offsite.WriterError{Kind: offsite.ErrStandby, Writer: &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: t0}}
pass := func(statusFile string, at time.Time, err error) offsite.Lease {
t.Helper()
st, lease := startRun(cfg, key, statusFile, at)
recordRun(&st, offsite.Result{}, err, lease)
if werr := offsite.WriteStatus(statusFile, st); werr != nil {
t.Fatal(werr)
}
return lease
}
restored := filepath.Join(t.TempDir(), "status.json")
for i := range 3 {
if l := pass(restored, t0.Add(time.Duration(i)*time.Hour), standby); l.Inherited {
t.Fatalf("run %d of a restored host claims the bucket", i+1)
}
}
if st, _ := offsite.ReadStatus(restored); !st.Standby || st.Format != offsite.StatusFormat || st.KeyID != offsite.KeyID(key) || st.Bucket != "felis-backups" {
t.Errorf("restored host's status = %+v", st)
}
upgraded := filepath.Join(t.TempDir(), "status.json")
if err := offsite.WriteStatus(upgraded, offsite.Status{LastAttempt: t0.Add(-time.Hour), LastSuccess: t0.Add(-time.Hour)}); err != nil {
t.Fatal(err)
}
if l := pass(upgraded, t0, errors.New("cannot reach bucket")); !l.Inherited {
t.Fatal("the first run after the upgrade does not claim the bucket")
}
l := pass(upgraded, t0.Add(time.Hour), nil)
if !l.Inherited {
t.Fatal("a failed first run after the upgrade lost the claim")
}
if st, _ := offsite.ReadStatus(upgraded); !st.LastSuccess.Equal(t0.Add(time.Hour)) {
t.Errorf("upgraded host's status = %+v", st)
}
}
// TestOffsiteSyncerSnapshotsAndSweeps: the pass `offsite sync` runs takes its
// snapshot the way `felis db backup` does, in the database pod, into the
// bundle directory, keeping the newest one there; and it sweeps unrecorded
// world objects only past the longest retention [archive] gives any backup.
func TestOffsiteSyncerSnapshotsAndSweeps(t *testing.T) {
dir := newPodRig(t)
bundles := filepath.Join(dir, "bundles")
env := &offsiteEnv{cfg: config.OffsiteConfig{DBKeep: 5}}
var log bytes.Buffer
cfg := &config.Config{Database: podDB, Archive: config.ArchiveConfig{Retention: "120d"}}
s := offsiteSyncer(cfg, env, offsiteSources{dbDir: bundles}, "/archives", "/uploads", nil, &log)
if s.DBDir != bundles || s.DBKeep != 5 || s.ArchiveDir != "/archives" || s.UploadsDir != "/uploads" {
t.Fatalf("syncer = %+v", s)
}
if s.OrphanAfter != 120*24*time.Hour {
t.Errorf("OrphanAfter = %s, want the 120d retention", s.OrphanAfter)
}
if s.Snapshot == nil {
t.Fatal("the pass takes no snapshot after copying archives")
}
for i := 0; i < 2; i++ {
if err := s.Snapshot(context.Background()); err != nil {
t.Fatalf("snapshot %d: %v", i, err)
}
}
got, err := dbbackup.List(bundles)
if err != nil || len(got) != 1 || got[0].Label != dbbackup.LabelOffsite {
t.Fatalf("bundle directory = %+v, %v; want the newest offsite bundle alone", got, err)
}
if _, err := dbbackupVerify(got[0].Path); err != nil {
t.Fatalf("the snapshot does not verify: %v", err)
}
// The MinecraftServer objects are exported alongside, as in the daily bundle.
argv, _ := os.ReadFile(filepath.Join(dir, "k3s.args"))
if ran := string(argv); !strings.Contains(ran, podExecPrefix+"pg_dump --format=custom") {
t.Errorf("k3s ran %q, want pg_dump in the pod", ran)
}
if !strings.Contains(string(bundleServers(t, got[0].Path)), `"name": "lobby"`) {
t.Errorf("the snapshot holds no MinecraftServer objects")
}
if !strings.Contains(log.String(), "took database bundle "+got[0].Name) {
t.Errorf("the snapshot is not logged:\n%s", log.String())
}
// It becomes the newest bundle in the bucket, so with the cluster away it
// fails, leaves no bundle, and the next pass tries again.
noServerExportWait(t)
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
if err := s.Snapshot(context.Background()); err == nil || !strings.Contains(err.Error(), "export the MinecraftServer objects") {
t.Errorf("snapshot with the cluster away: %v, want a failure", err)
}
if again, _ := dbbackup.List(bundles); len(again) != 1 || again[0].Name != got[0].Name {
t.Errorf("bundle directory after the failed snapshot = %+v, want %s alone", again, got[0].Name)
}
for _, c := range []struct {
archive config.ArchiveConfig
want time.Duration
}{
{config.ArchiveConfig{}, 90 * 24 * time.Hour},
{config.ArchiveConfig{ScheduledRetention: "200d"}, 200 * 24 * time.Hour},
{config.ArchiveConfig{ManualRetention: "150d", Retention: "30d", ScheduledRetention: "60d"}, 150 * 24 * time.Hour},
} {
s := offsiteSyncer(&config.Config{Database: podDB, Archive: c.archive}, env, offsiteSources{dbDir: bundles}, "", "", nil, io.Discard)
if s.OrphanAfter != c.want {
t.Errorf("%+v: OrphanAfter = %s, want %s", c.archive, s.OrphanAfter, c.want)
}
}
log.Reset()
s = offsiteSyncer(&config.Config{Database: podDB, Archive: config.ArchiveConfig{Retention: "soon"}}, env, offsiteSources{}, "", "", nil, &log)
if s.OrphanAfter != 0 || !strings.Contains(log.String(), "world objects no backup records are kept") {
t.Errorf("a retention that does not parse: OrphanAfter %s, log %q; want no sweep, said", s.OrphanAfter, log.String())
}
if s.Snapshot != nil {
t.Error("a pass that copies no bundles takes a snapshot")
}
}
-3
View File
@@ -121,9 +121,6 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
// Uncached too: RCON Secrets are read by name, so the Role grants
// secrets:get without the list/watch an informer would need.
Secrets: mgr.GetAPIReader(),
// And pods: pod-0 is read by name, so the Role grants pods:get and no
// namespace-wide pod informer runs.
Pods: mgr.GetAPIReader(),
Recorder: mgr.GetEventRecorderFor("felis-operator"),
Watch: watch,
}
+5 -26
View File
@@ -78,7 +78,7 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
return 1
}
drv, err := openPodStore(ctx, cfg.Database.URL, "reaper", stderr)
drv, err := openStore(ctx, cfg.Database.URL, false)
if err != nil {
fmt.Fprintf(stderr, "felis reaper: open database: %v\n", err)
return 1
@@ -144,8 +144,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
// FelisWorldJobFailed rule) reach the operator: a world that cannot be archived
// is kept, and without this nobody would learn that it is never reaped.
func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int {
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d released=%d deleted=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
sum.Evaluated, sum.WorldsReaped, sum.Released, sum.ServersDeleted, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed,
sum.Verified, sum.Corrupt, sum.VerifyFailed, sum.Swept, sum.OrphanArchives)
if !sum.Failed() {
@@ -199,9 +199,8 @@ func (w *mailWarner) Warn(ctx context.Context, ownerID, server, remaining string
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
// idle deadline is fixed by §18; only the warning offsets, retention, the
// store soft-cap, the on-demand backup bounds and the scheduled restore points
// are configurable (§24). The backup Job and felis-api read the manual_* and
// scheduled_* keys through it too.
// store soft-cap and the on-demand backup bounds are configurable (§24). The
// backup Job and felis-api read the manual_* bounds through it too.
func reaperConfig(cfg *config.Config) (reaper.Config, error) {
rc := reaper.DefaultConfig()
if v := cfg.Archive.Retention; v != "" {
@@ -249,26 +248,6 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
}
rc.ManualCooldown = d
}
if v := cfg.Archive.ScheduledEvery; v != "" {
d, err := parseSpanDuration(v)
if err != nil || d < 0 {
return rc, fmt.Errorf("[archive] scheduled_every %q: want a span such as 1d (0s for none)", v)
}
rc.ScheduledEvery = d
}
switch n := cfg.Archive.ScheduledKeep; {
case n < 0:
return rc, fmt.Errorf("[archive] scheduled_keep %d: want 1 or more", n)
case n > 0:
rc.ScheduledKeep = n
}
if v := cfg.Archive.ScheduledRetention; v != "" {
d, err := parseSpanDuration(v)
if err != nil || d <= 0 {
return rc, fmt.Errorf("[archive] scheduled_retention %q: want a positive span such as 90d", v)
}
rc.ScheduledRetention = d
}
rc.RequireOffsite = cfg.Offsite.Enabled()
return rc, nil
}
+1 -40
View File
@@ -4,7 +4,6 @@ import (
"bytes"
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
@@ -28,7 +27,6 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
want int
}{
{"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0},
{"retirements", reaper.Summary{Evaluated: 5, WorldsReaped: 3, Released: 2, ServersDeleted: 1}, 0},
{"waiting for a stop", reaper.Summary{Evaluated: 3, AwaitingStop: 1}, 0},
{"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1},
{"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1},
@@ -42,8 +40,7 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
if got := reportReaperRun(tc.sum, &out, &errb); got != tc.want {
t.Errorf("%s: exit %d, want %d", tc.name, got, tc.want)
}
if !strings.Contains(out.String(), "skipped=") || !strings.Contains(out.String(), "expire_failed=") ||
!strings.Contains(out.String(), fmt.Sprintf(" released=%d deleted=%d ", tc.sum.Released, tc.sum.ServersDeleted)) {
if !strings.Contains(out.String(), "skipped=") || !strings.Contains(out.String(), "expire_failed=") {
t.Errorf("%s: summary line = %q", tc.name, out.String())
}
if (tc.want == 1) != (errb.Len() > 0) {
@@ -88,42 +85,6 @@ func TestReaperConfigManualKeys(t *testing.T) {
}
}
// TestReaperConfigScheduledKeys: the scheduled restore points default to one a
// day, seven per server and the reaper's 90 days, accept overrides ("0s" turns
// them off), and refuse values that would keep nothing or run backwards.
func TestReaperConfigScheduledKeys(t *testing.T) {
rc, err := reaperConfig(&config.Config{})
if err != nil {
t.Fatal(err)
}
if rc.ScheduledEvery != reaper.Day || rc.ScheduledKeep != 7 || rc.ScheduledRetention != 90*reaper.Day {
t.Fatalf("defaults = %v / %d / %v", rc.ScheduledEvery, rc.ScheduledKeep, rc.ScheduledRetention)
}
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{
ScheduledEvery: "12h", ScheduledKeep: 3, ScheduledRetention: "14d"}})
if err != nil {
t.Fatal(err)
}
if rc.ScheduledEvery != 12*time.Hour || rc.ScheduledKeep != 3 || rc.ScheduledRetention != 14*reaper.Day {
t.Fatalf("overrides = %v / %d / %v", rc.ScheduledEvery, rc.ScheduledKeep, rc.ScheduledRetention)
}
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{ScheduledEvery: "0s"}})
if err != nil || rc.ScheduledEvery != 0 {
t.Fatalf("scheduled_every 0s = %v, %v; want off", rc.ScheduledEvery, err)
}
for _, bad := range []config.ArchiveConfig{
{ScheduledEvery: "-1h"},
{ScheduledEvery: "daily"},
{ScheduledKeep: -1},
{ScheduledRetention: "0d"},
{ScheduledRetention: "forever"},
} {
if _, err := reaperConfig(&config.Config{Archive: bad}); err == nil {
t.Errorf("%+v was accepted", bad)
}
}
}
func TestResolveWorldDir(t *testing.T) {
ctx := context.Background()
root := t.TempDir()
+2 -9
View File
@@ -131,8 +131,7 @@ func loopbackAddr(addr string) bool {
func cmdPushImage(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("push-image", flag.ContinueOnError)
fs.SetOutput(stderr)
tarPath := fs.String("tar", "", "image tarball Kaniko wrote with --tar-path, or with --image an OCI layout tar")
image := fs.String("image", "", "push the image this name (io.containerd.image.name) marks in the OCI layout tar --tar, e.g. a release's image bundle")
tarPath := fs.String("tar", "", "image tarball Kaniko wrote with --tar-path")
ref := fs.String("ref", "", "host/repository:tag to publish it as")
scheme := fs.String("scheme", "http", "registry scheme: http for the in-cluster registry, https otherwise")
if err := fs.Parse(args); err != nil {
@@ -155,13 +154,7 @@ func cmdPushImage(args []string, stdout, stderr io.Writer) int {
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
p := &imagepush.Pusher{Scheme: *scheme, Username: user, Password: pass, Log: stderr}
var digest string
var err error
if *image != "" {
digest, err = p.PushLayout(ctx, *tarPath, *image, *ref)
} else {
digest, err = p.Push(ctx, *tarPath, *ref)
}
digest, err := p.Push(ctx, *tarPath, *ref)
if err != nil {
fmt.Fprintf(stderr, "felis push-image: %v\n", err)
return 1
+68 -613
View File
@@ -2,70 +2,40 @@ package main
import (
"context"
"crypto/hmac"
"crypto/pbkdf2"
"crypto/rand"
"crypto/sha256"
"encoding/base64"
"encoding/hex"
"errors"
"flag"
"fmt"
"io"
"io/fs"
neturl "net/url"
"os"
"os/exec"
"path/filepath"
"strconv"
"strings"
"syscall"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/maintenance"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/operator"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/registrygate"
"github.com/BurntSushi/toml"
"github.com/jackc/pgx/v5"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// rotate-token replaces one credential the installer generated: an internal
// caller's token (naming.CallerTokens), the registry's write tokens, the
// Velocity forwarding secret or the database password. The new value goes into
// the installer's record first, so that whatever fails later a re-run of the
// installer puts it everywhere; then into the Secrets and files that carry it;
// then whatever read the old value at start restarts. Without -yes it prints
// what it would change and what that interrupts, and changes nothing.
// rotate-token replaces one internal caller's token (naming.CallerTokens): a new
// value goes into the installer's record, the control-namespace Secret and the
// replica the caller's pods mount, felis-api rolls so it accepts only the new
// value, and then the caller restarts so it presents it. Between the api's
// rollout and the caller's restart the caller is turned away with 401; for the
// login gate and the proxy that is the few seconds of a pod or unit restart.
const (
defaultSecretsEnvPath = "/etc/felis/secrets.env"
defaultLinkPropsPath = "/opt/felis/velocity/plugins/felis-link/felis-link.properties"
defaultForwardingPath = "/opt/felis/velocity/forwarding.secret"
velocityUnit = "felis-velocity"
apiDeployment = "felis-api"
registryDeployment = "registry"
kindRegistry = "registry"
kindForwarding = "forwarding"
kindDB = "db"
// proxyReloadWait is how long the host proxy gets, once felis-api has rolled,
// to show it re-read its token: it reads the file again on its next call to
// felis-api, and it calls every 15 seconds.
proxyReloadWait = 60 * time.Second
// apiRestartNote is in every plan: felis-api runs as one replica replaced in
// place, so its restart is a short outage of everything that talks to it.
apiRestartNote = "felis-api restarts (a single replica): the panel, sign-in and the proxy's calls are unavailable for the few seconds that takes"
)
// installerTokenKeys names each caller's token in the installer's secrets.env
@@ -79,54 +49,20 @@ var installerTokenKeys = map[string]string{
"ops": "OPS_TOKEN",
}
// installerRegistryKeys names the registry principals' tokens in secrets.env,
// in registrygate.Principals order.
var installerRegistryKeys = map[string]string{
registrygate.PrincipalPlatform: "REGISTRY_PLATFORM_TOKEN",
registrygate.PrincipalBuild: "REGISTRY_BUILD_TOKEN",
registrygate.PrincipalPrune: "REGISTRY_PRUNE_TOKEN",
}
// installerForwardingKey and installerDBKey name the forwarding secret and the
// database password in secrets.env.
const (
installerForwardingKey = "FORWARDING_SECRET"
installerDBKey = "DB_PASSWORD"
)
type tokenRotator struct {
cl client.Client
controlNS string
minecraftNS string
buildNS string
// secretsEnv, linkProps and forwardingFile are the installer's record and the
// host proxy's felis-link.properties and forwarding.secret; a missing file is
// reported and skipped.
// secretsEnv and linkProps are the installer's record and the proxy's
// felis-link.properties; a missing file is reported and skipped.
secretsEnv string
linkProps string
forwardingFile string
// hostTOML, podTOML and defaultTOML are the config copies that carry the
// database URL (defaultTOML only when it is a file of its own).
hostTOML string
podTOML string
defaultTOML string
newToken func() (string, error)
// rollout restarts a control-namespace Deployment and waits for it.
rollout func(ctx context.Context, deployment string) error
// rollAPI restarts felis-api and waits for the rollout.
rollAPI func(ctx context.Context) error
// restartUnit restarts a systemd unit on this host.
restartUnit func(ctx context.Context, unit string) error
// proxyLog is what the host proxy has logged since a moment, as far as it
// can be read.
proxyLog func(ctx context.Context, since time.Time) string
// alterRole stores a password verifier for a role of the database that runs
// as deployment ("namespace/name"); verifyDB connects with a URL.
alterRole func(ctx context.Context, deployment, role, verifier string) error
verifyDB func(ctx context.Context, url string) error
now func() time.Time
// reloadWait bounds the wait for the proxy to re-read its token, polled
// every pollEvery.
reloadWait time.Duration
pollEvery time.Duration
out io.Writer
}
@@ -136,12 +72,9 @@ func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
secretsEnv := fs.String("secrets-env", defaultSecretsEnvPath, "the installer's secrets file, updated so a re-run keeps the new value")
linkProps := fs.String("link-properties", defaultLinkPropsPath, "the host proxy's felis-link.properties (velocity only)")
forwarding := fs.String("forwarding-secret", defaultForwardingPath, "the host proxy's forwarding secret file (forwarding only)")
yes := fs.Bool("yes", false, "rotate; without it the plan is printed and nothing changes")
fs.Usage = func() {
fmt.Fprintf(stderr, "Usage: felis rotate-token [-yes] [flags] <%s>\n\n", strings.Join(rotationKinds(), "|"))
fmt.Fprintln(stderr, "Replaces one generated credential: the installer's record, the Secrets and files that carry it, then what reads it.")
fmt.Fprintln(stderr, "Without -yes it prints what would change and what that interrupts.")
fmt.Fprintf(stderr, "Usage: felis rotate-token [flags] <%s>\n\n", strings.Join(callerNames(), "|"))
fmt.Fprintln(stderr, "Replaces one internal caller's token: the Secrets, felis-api, then the caller itself.")
fs.PrintDefaults()
}
if err := fs.Parse(args); err != nil {
@@ -154,13 +87,12 @@ func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
fs.Usage()
return 2
}
kind := fs.Arg(0)
if !knownRotation(kind) {
fmt.Fprintf(stderr, "felis rotate-token: unknown credential %q (one of %s)\n", kind, strings.Join(rotationKinds(), ", "))
if _, ok := callerToken(fs.Arg(0)); !ok {
fmt.Fprintf(stderr, "felis rotate-token: unknown caller %q (one of %s)\n", fs.Arg(0), strings.Join(callerNames(), ", "))
return 2
}
if os.Geteuid() != 0 {
fmt.Fprintln(stderr, "felis rotate-token: refused — rotating writes the cluster Secrets and the installer's secrets file, so it must run as root (try: sudo felis rotate-token "+kind+")")
fmt.Fprintln(stderr, "felis rotate-token: refused — rotating writes the cluster Secrets and the installer's secrets file, so it must run as root (try: sudo felis rotate-token "+fs.Arg(0)+")")
return 1
}
cfg, err := config.Load(*cfgPath)
@@ -177,43 +109,24 @@ func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
if buildNS == "" {
buildNS = platform.DefaultBuildNamespace
}
controlNS := platform.DefaultControlNamespace
r := tokenRotator{
cl: cl,
controlNS: controlNS,
controlNS: platform.DefaultControlNamespace,
minecraftNS: cfg.K8s.Namespace,
buildNS: buildNS,
secretsEnv: *secretsEnv,
linkProps: *linkProps,
forwardingFile: *forwarding,
hostTOML: hostSetupConfigPath,
podTOML: podSetupConfigPath,
defaultTOML: defaultSetupConfigPath,
newToken: randomToken,
rollout: func(ctx context.Context, deployment string) error {
if err := kubectl(ctx, "-n", controlNS, "rollout", "restart", "deployment/"+deployment); err != nil {
rollAPI: func(ctx context.Context) error {
if err := kubectl(ctx, "-n", platform.DefaultControlNamespace, "rollout", "restart", "deployment/felis-api"); err != nil {
return err
}
return kubectl(ctx, "-n", controlNS, "rollout", "status", "deployment/"+deployment, "--timeout=180s")
return kubectl(ctx, "-n", platform.DefaultControlNamespace, "rollout", "status", "deployment/felis-api", "--timeout=180s")
},
restartUnit: func(ctx context.Context, unit string) error { return systemctl(ctx, "restart", unit) },
proxyLog: journalSince,
alterRole: alterRoleInPod,
verifyDB: func(ctx context.Context, url string) error {
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
defer cancel()
conn, err := pgx.Connect(ctx, url)
if err != nil {
return err
}
return conn.Close(ctx)
},
now: time.Now,
reloadWait: proxyReloadWait,
pollEvery: 3 * time.Second,
out: stdout,
}
if err := r.rotate(context.Background(), kind, *yes); err != nil {
if err := r.rotate(context.Background(), fs.Arg(0)); err != nil {
fmt.Fprintf(stderr, "felis rotate-token: %v\n", err)
return 1
}
@@ -237,21 +150,6 @@ func callerToken(name string) (naming.CallerToken, bool) {
return naming.CallerToken{}, false
}
// rotationKinds is every credential rotate-token replaces: the callers' tokens
// and the installer's other generated secrets.
func rotationKinds() []string {
return append(callerNames(), kindRegistry, kindForwarding, kindDB)
}
func knownRotation(kind string) bool {
for _, k := range rotationKinds() {
if k == kind {
return true
}
}
return false
}
// randomToken is 32 random bytes in hex, the shape the installer generates.
func randomToken() (string, error) {
b := make([]byte, 32)
@@ -261,527 +159,99 @@ func randomToken() (string, error) {
return hex.EncodeToString(b), nil
}
// tokenFingerprint is how the proxy names the token it reloaded in its log
// (plugins/shared FileToken.fingerprint): the first twelve hex digits of its
// SHA-256, enough to tell tokens apart and useless for finding one.
func tokenFingerprint(token string) string {
sum := sha256.Sum256([]byte(token))
return hex.EncodeToString(sum[:])[:12]
}
func (r tokenRotator) rotate(ctx context.Context, kind string, apply bool) error {
switch kind {
case kindRegistry:
return r.rotateRegistry(ctx, apply)
case kindForwarding:
return r.rotateForwarding(ctx, apply)
case kindDB:
return r.rotateDB(ctx, apply)
}
ct, ok := callerToken(kind)
func (r tokenRotator) rotate(ctx context.Context, caller string) error {
ct, ok := callerToken(caller)
if !ok {
return fmt.Errorf("unknown credential %q (one of %s)", kind, strings.Join(rotationKinds(), ", "))
return fmt.Errorf("unknown caller %q", caller)
}
return r.rotateCaller(ctx, ct, apply)
tok, err := r.newToken()
if err != nil {
return fmt.Errorf("generate a token: %w", err)
}
// confirm ends the plan: without apply it says nothing changed and how to go
// ahead, and reports false.
func (r tokenRotator) confirm(kind string, apply bool) bool {
if !apply {
fmt.Fprintf(r.out, "\nNothing was changed. To rotate: sudo felis rotate-token -yes %s\n", kind)
return false
}
fmt.Fprintln(r.out, "\nRotating:")
return true
}
// record writes new values into the installer's secrets.env. It comes first in
// every rotation: from then on, whatever fails, a re-run of the installer puts
// the new values everywhere.
func (r tokenRotator) record(kv ...[2]string) error {
keys := make([]string, len(kv))
for i, p := range kv {
keys[i] = p[0]
}
switch err := setKeyValueLines(r.secretsEnv, "=", kv); {
// The installer's record first: from here on, whatever fails, a re-run of the
// installer puts the new value everywhere.
switch err := setKeyValueLine(r.secretsEnv, installerTokenKeys[ct.Caller], "=", tok); {
case errors.Is(err, fs.ErrNotExist):
fmt.Fprintf(r.out, " - %s: not found, skipped (this host was not installed by deploy/bootstrap.sh)\n", r.secretsEnv)
case err != nil:
return fmt.Errorf("record the new value in %s: %w", r.secretsEnv, err)
return fmt.Errorf("record the new token in %s: %w", r.secretsEnv, err)
default:
fmt.Fprintf(r.out, " - %s: %s updated\n", r.secretsEnv, strings.Join(keys, ", "))
}
return nil
fmt.Fprintf(r.out, " - %s: %s updated\n", r.secretsEnv, installerTokenKeys[ct.Caller])
}
func (r tokenRotator) putSecret(ctx context.Context, ns, name string, data map[string][]byte) error {
if _, err := putSecretKeys(ctx, r.cl, ns, name, corev1.SecretTypeOpaque, data); err != nil {
return fmt.Errorf("write Secret %s/%s: %w", ns, name, err)
}
fmt.Fprintf(r.out, " - Secret %s/%s: updated\n", ns, name)
return nil
}
func (r tokenRotator) rollAPI(ctx context.Context) error {
if err := r.rollout(ctx, apiDeployment); err != nil {
return fmt.Errorf("roll felis-api: %w", err)
}
fmt.Fprintln(r.out, " - felis-api: rolled out on the new value")
return nil
}
// withMinecraft is the control namespace plus the minecraft namespace when that
// is a different one: where the Secrets game pods and Jobs mount are mirrored.
func (r tokenRotator) withMinecraft() []string {
if r.minecraftNS == "" || r.minecraftNS == r.controlNS {
return []string{r.controlNS}
}
return []string{r.controlNS, r.minecraftNS}
}
func (r tokenRotator) rotateCaller(ctx context.Context, ct naming.CallerToken, apply bool) error {
namespaces := []string{r.controlNS}
replica := map[string]string{"minecraft": r.minecraftNS, "build": r.buildNS}[ct.Replica]
if replica != "" && replica != r.controlNS {
namespaces = append(namespaces, replica)
}
key := installerTokenKeys[ct.Caller]
fmt.Fprintf(r.out, "felis rotate-token %s: a new internal token for the %s caller\n", ct.Caller, ct.Caller)
secrets := make([]string, len(namespaces))
for i, ns := range namespaces {
secrets[i] = "Secret " + ns + "/" + ct.Secret
}
writes := append([]string{r.secretsEnv + " (" + key + ")"}, secrets...)
hostProxy := ct.Caller == "velocity" && fileExists(r.linkProps)
if hostProxy {
writes = append(writes, r.linkProps+" (service-token)")
}
fmt.Fprintf(r.out, " - writes %s\n", strings.Join(writes, ", "))
fmt.Fprintf(r.out, " - %s\n", apiRestartNote)
switch ct.Caller {
case "velocity":
if hostProxy {
fmt.Fprintf(r.out, " - the proxy re-reads its token from the file and keeps its players; if it has not within %s of felis-api's restart, it is restarted, which disconnects everyone online\n", r.reloadWait)
} else {
fmt.Fprintf(r.out, " - no proxy on this host (%s): set service-token in your proxy's felis-link.properties to the value in Secret %s/%s afterwards; it re-reads the file without a restart\n",
r.linkProps, r.controlNS, ct.Secret)
}
case "limbo":
fmt.Fprintln(r.out, " - the login gate's pod restarts: a player signing in at that moment reconnects")
case "build":
fmt.Fprintln(r.out, " - a build fetching its context at that moment fails and can be submitted again")
case "ops":
fmt.Fprintln(r.out, " - felis backup-now presents the new token on its next run")
}
if !r.confirm(ct.Caller, apply) {
return nil
}
tok, err := r.newToken()
if err != nil {
return fmt.Errorf("generate a token: %w", err)
}
if err := r.record([2]string{key, tok}); err != nil {
return err
}
for _, ns := range namespaces {
if err := r.putSecret(ctx, ns, ct.Secret, map[string][]byte{naming.ServiceTokenSecretKey: []byte(tok)}); err != nil {
return err
if err := writeTokenSecret(ctx, r.cl, ns, ct.Secret, tok); err != nil {
return fmt.Errorf("write Secret %s/%s: %w", ns, ct.Secret, err)
}
fmt.Fprintf(r.out, " - Secret %s/%s: updated\n", ns, ct.Secret)
}
// The proxy's file changes before felis-api rolls, so the file never falls
// behind the Secret; the proxy's log is read from just before the write,
// since it may pick the new token up before the rollout ends.
since := r.now()
if hostProxy {
if err := setProxyToken(r.linkProps, tok); err != nil {
hostProxy := false
if ct.Caller == "velocity" {
switch err := setKeyValueLine(r.linkProps, "service-token", "=", tok); {
case errors.Is(err, fs.ErrNotExist):
fmt.Fprintf(r.out, " - %s: not found; set service-token in your proxy's felis-link.properties to the value in Secret %s/%s and restart it\n",
r.linkProps, r.controlNS, ct.Secret)
case err != nil:
return fmt.Errorf("write the proxy's token into %s: %w", r.linkProps, err)
}
default:
hostProxy = true
fmt.Fprintf(r.out, " - %s: service-token updated\n", r.linkProps)
}
}
if err := r.rollAPI(ctx); err != nil {
return err
return fmt.Errorf("roll felis-api: %w", err)
}
fmt.Fprintln(r.out, " - felis-api: rolled out, accepting only the new token")
switch ct.Caller {
case "velocity":
if !hostProxy {
break
}
if r.proxyReloaded(ctx, since, tokenFingerprint(tok)) {
fmt.Fprintf(r.out, " - %s: took the new token from its properties; players stayed connected\n", velocityUnit)
break
}
if hostProxy {
if err := r.restartUnit(ctx, velocityUnit); err != nil {
return fmt.Errorf("restart %s: %w", velocityUnit, err)
}
fmt.Fprintf(r.out, " - %s: had not taken the new token within %s, restarted (players on the proxy were disconnected and can rejoin)\n", velocityUnit, r.reloadWait)
fmt.Fprintf(r.out, " - %s: restarted (players on the proxy were disconnected and can rejoin)\n", velocityUnit)
}
case "limbo":
// Only the operator's server pods carry its managed-by label; a backup or
// restore Job's pod carries the server label too, and app.kubernetes.io ones.
if err := r.cl.DeleteAllOf(ctx, &corev1.Pod{}, client.InNamespace(r.minecraftNS), client.MatchingLabels{
v1alpha1.LabelServer: naming.SystemLoginServer,
v1alpha1.LabelManagedBy: operator.ManagedByValue,
}); err != nil {
if err := r.cl.DeleteAllOf(ctx, &corev1.Pod{}, client.InNamespace(r.minecraftNS),
client.MatchingLabels{v1alpha1.LabelServer: naming.SystemLoginServer}); err != nil {
return fmt.Errorf("restart the login gate: %w", err)
}
fmt.Fprintln(r.out, " - login gate: pod restarted to read the new token")
case "build":
fmt.Fprintln(r.out, " - builds: the next build Job reads the new token")
fmt.Fprintln(r.out, " - builds: the next build Job reads the new token; one fetching its context right now fails and can be submitted again")
case "ops":
fmt.Fprintln(r.out, " - felis backup-now reads the new token on its next run")
}
return nil
}
// proxyReloaded waits up to reloadWait for the host proxy to log that it
// reloaded the token with this fingerprint (plugins/shared FileToken).
func (r tokenRotator) proxyReloaded(ctx context.Context, since time.Time, fingerprint string) bool {
want := "(fingerprint " + fingerprint + ")"
deadline := r.now().Add(r.reloadWait)
for {
if strings.Contains(r.proxyLog(ctx, since), want) {
return true
// writeTokenSecret sets the token in a Secret, creating it when absent.
func writeTokenSecret(ctx context.Context, cl client.Client, namespace, name, token string) error {
var sec corev1.Secret
err := cl.Get(ctx, client.ObjectKey{Namespace: namespace, Name: name}, &sec)
if apierrors.IsNotFound(err) {
return cl.Create(ctx, &corev1.Secret{
ObjectMeta: metav1.ObjectMeta{Namespace: namespace, Name: name},
Type: corev1.SecretTypeOpaque,
Data: map[string][]byte{naming.ServiceTokenSecretKey: []byte(token)},
})
}
if !r.now().Before(deadline) {
return false
}
select {
case <-ctx.Done():
return false
case <-time.After(r.pollEvery):
}
}
}
// journalSince is the proxy unit's journal from the second since falls in. A
// journal that cannot be read reads as one without the line, which ends in the
// restart a rotation made before the proxy could reload.
func journalSince(ctx context.Context, since time.Time) string {
out, _ := exec.CommandContext(ctx, "journalctl", "-u", velocityUnit, "--since", "@"+strconv.FormatInt(since.Unix(), 10),
"-o", "cat", "--no-pager", "-q").Output()
return string(out)
}
// setProxyToken writes the proxy's token into felis-link.properties and puts
// the file's modification time back. The proxy re-reads its token by itself,
// while `felis domain check` and `felis domain set` read a file newer than the
// proxy's start as config it has not loaded (the installer's
// install_if_changed keeps the time for the same reason).
func setProxyToken(path, token string) error {
info, err := os.Stat(path)
if err != nil {
return err
}
if err := setKeyValueLine(path, "service-token", "=", token); err != nil {
return err
if sec.Data == nil {
sec.Data = map[string][]byte{}
}
return os.Chtimes(path, time.Time{}, info.ModTime())
}
func (r tokenRotator) rotateRegistry(ctx context.Context, apply bool) error {
keys := make([]string, len(registrygate.Principals))
for i, p := range registrygate.Principals {
keys[i] = installerRegistryKeys[p]
}
fmt.Fprintf(r.out, "felis rotate-token %s: new write tokens for the image registry's principals (%s)\n", kindRegistry, strings.Join(registrygate.Principals, ", "))
fmt.Fprintf(r.out, " - writes %s (%s), Secret %s/%s, Secret %s/%s\n", r.secretsEnv, strings.Join(keys, ", "),
r.controlNS, naming.RegistryAuthSecretName, r.buildNS, naming.RegistryPushSecretName)
fmt.Fprintln(r.out, " - the registry restarts to load them: a build pushing its image at that moment fails and can be submitted again, and an image pull in that moment retries")
fmt.Fprintf(r.out, " - %s (it presents the prune token)\n", apiRestartNote)
if !r.confirm(kindRegistry, apply) {
return nil
}
tokens := map[string][]byte{}
kv := make([][2]string, 0, len(registrygate.Principals))
for _, p := range registrygate.Principals {
tok, err := r.newToken()
if err != nil {
return fmt.Errorf("generate a token: %w", err)
}
tokens[p] = []byte(tok)
kv = append(kv, [2]string{installerRegistryKeys[p], tok})
}
if err := r.record(kv...); err != nil {
return err
}
if err := r.putSecret(ctx, r.controlNS, naming.RegistryAuthSecretName, tokens); err != nil {
return err
}
if err := r.putSecret(ctx, r.buildNS, naming.RegistryPushSecretName, map[string][]byte{
naming.RegistryPushUsernameKey: []byte(registrygate.PrincipalBuild),
naming.RegistryPushPasswordKey: tokens[registrygate.PrincipalBuild],
}); err != nil {
return err
}
if err := r.rollout(ctx, registryDeployment); err != nil {
return fmt.Errorf("restart the registry: %w", err)
}
fmt.Fprintln(r.out, " - registry: restarted on the new tokens")
return r.rollAPI(ctx)
}
func (r tokenRotator) rotateForwarding(ctx context.Context, apply bool) error {
restart, held, err := r.gamePods(ctx)
if err != nil {
return err
}
hostProxy := fileExists(r.forwardingFile)
fmt.Fprintf(r.out, "felis rotate-token %s: a new Velocity forwarding secret, the key a server checks each player's identity with\n", kindForwarding)
secrets := []string{}
for _, ns := range r.withMinecraft() {
secrets = append(secrets, "Secret "+ns+"/"+naming.ForwardingSecretName)
}
writes := append([]string{r.secretsEnv + " (" + installerForwardingKey + ")"}, secrets...)
if hostProxy {
writes = append(writes, r.forwardingFile)
}
fmt.Fprintf(r.out, " - writes %s\n", strings.Join(writes, ", "))
fmt.Fprintf(r.out, " - every running server restarts to read it (%d now), saving its world on the way down, and the proxy restarts: everyone online is disconnected and can rejoin once their server is back\n", len(restart))
if len(held) > 0 {
fmt.Fprintf(r.out, " - left running, because a backup, restore or file write holds its world: %s. Players cannot join it until it restarts: stop and start it from the panel once that finishes\n", strings.Join(held, ", "))
}
if !hostProxy {
fmt.Fprintf(r.out, " - no proxy on this host (%s): put the value in Secret %s/%s into your proxy's forwarding secret file and restart it\n",
r.forwardingFile, r.controlNS, naming.ForwardingSecretName)
}
if !r.confirm(kindForwarding, apply) {
return nil
}
tok, err := r.newToken()
if err != nil {
return fmt.Errorf("generate a secret: %w", err)
}
if err := r.record([2]string{installerForwardingKey, tok}); err != nil {
return err
}
if hostProxy {
info, err := os.Stat(r.forwardingFile)
if err != nil {
return err
}
if err := replaceFileKeepingMode(r.forwardingFile, info, []byte(tok)); err != nil {
return fmt.Errorf("write %s: %w", r.forwardingFile, err)
}
fmt.Fprintf(r.out, " - %s: updated\n", r.forwardingFile)
}
for _, ns := range r.withMinecraft() {
if err := r.putSecret(ctx, ns, naming.ForwardingSecretName, map[string][]byte{naming.ForwardingSecretKey: []byte(tok)}); err != nil {
return err
}
}
// The servers go first: their pods read the Secret as they are recreated,
// and a player who rejoins through the restarted proxy meets a server on the
// new secret, or one still starting.
for i := range restart {
p := &restart[i]
if err := r.cl.Delete(ctx, p, client.Preconditions{UID: &p.UID}); err != nil && !apierrors.IsNotFound(err) && !apierrors.IsConflict(err) {
return fmt.Errorf("restart %s: %w", p.Labels[v1alpha1.LabelServer], err)
}
}
fmt.Fprintf(r.out, " - servers: %d restarting on the new secret\n", len(restart))
if hostProxy {
if err := r.restartUnit(ctx, velocityUnit); err != nil {
return fmt.Errorf("restart %s: %w", velocityUnit, err)
}
fmt.Fprintf(r.out, " - %s: restarted on the new secret\n", velocityUnit)
}
return nil
}
// gamePods lists the running game server pods: those a rotation may restart,
// and the servers left alone because a backup, restore or file write holds
// their world (internal/maintenance), which a restart in the middle of would
// break.
func (r tokenRotator) gamePods(ctx context.Context) ([]corev1.Pod, []string, error) {
var pods corev1.PodList
// The operator's managed-by label is on its server pods alone (see rotateCaller).
if err := r.cl.List(ctx, &pods, client.InNamespace(r.minecraftNS), client.MatchingLabels{v1alpha1.LabelManagedBy: operator.ManagedByValue}); err != nil {
return nil, nil, fmt.Errorf("list the servers' pods: %w", err)
}
var jobs batchv1.JobList
if err := r.cl.List(ctx, &jobs, client.InNamespace(r.minecraftNS)); err != nil {
return nil, nil, fmt.Errorf("list the maintenance Jobs: %w", err)
}
var restart []corev1.Pod
var held []string
for _, p := range pods.Items {
if p.DeletionTimestamp != nil {
continue
}
server := p.Labels[v1alpha1.LabelServer]
var ms v1alpha1.MinecraftServer
if err := r.cl.Get(ctx, client.ObjectKey{Namespace: r.minecraftNS, Name: server}, &ms); client.IgnoreNotFound(err) != nil {
return nil, nil, fmt.Errorf("read server %s: %w", server, err)
}
if kind, ok := maintenance.Holder(server, ms.Annotations, jobs.Items, r.now()); ok {
held = append(held, server+" ("+kind+")")
continue
}
restart = append(restart, p)
}
return restart, held, nil
}
func (r tokenRotator) rotateDB(ctx context.Context, apply bool) error {
cfg, err := config.Load(r.hostTOML)
if err != nil {
return err
}
if cfg.Database.Deployment == "" {
return fmt.Errorf("[database] deployment is unset in %s, so the database is not the installer's felis-postgres: change the role's password where it runs, then in [database] url of each config copy", r.hostTOML)
}
u, err := neturl.Parse(cfg.Database.URL)
if err != nil || u.User == nil || u.User.Username() == "" {
return fmt.Errorf("[database] url in %s names no role", r.hostTOML)
}
role := u.User.Username()
targets, err := tomlTargetsOf(r.hostTOML, r.podTOML, r.defaultTOML)
if err != nil {
return err
}
paths := make([]string, len(targets))
for i, t := range targets {
paths[i] = t.path
}
secrets := []string{}
for _, ns := range r.withMinecraft() {
secrets = append(secrets, ns+"/"+platform.ConfigSecretName)
}
fmt.Fprintf(r.out, "felis rotate-token %s: a new password for the database role %q\n", kindDB, role)
fmt.Fprintf(r.out, " - writes %s (%s), the role in %s, [database] url in %s, Secret %s\n",
r.secretsEnv, installerDBKey, cfg.Database.Deployment, strings.Join(paths, " and "), strings.Join(secrets, " and "))
fmt.Fprintf(r.out, " - %s\n", apiRestartNote)
fmt.Fprintln(r.out, " - a backup, restore or file Job that connects in the seconds between the password change and felis-api's restart fails and can be run again; the host's timers read the new config on their next run")
if !r.confirm(kindDB, apply) {
return nil
}
password, err := r.newToken()
if err != nil {
return fmt.Errorf("generate a password: %w", err)
}
// Every config copy is edited in memory first, so one this cannot edit stops
// the rotation before the role's password changes.
edited := make([][]byte, len(targets))
var hostURL string
var podConfig []byte
for i, t := range targets {
raw, err := os.ReadFile(t.real)
if err != nil {
return err
}
var doc struct {
Database struct {
URL string `toml:"url"`
} `toml:"database"`
}
if _, err := toml.Decode(string(raw), &doc); err != nil {
return fmt.Errorf("%s: %w", t.path, err)
}
next, err := withPassword(doc.Database.URL, password)
if err != nil {
return fmt.Errorf("%s: %w", t.path, err)
}
if edited[i], err = editTOMLStrings(raw, []tomlStringEdit{{"database", "url", next}}); err != nil {
return fmt.Errorf("%s: %w; set the password in its [database] url by hand", t.path, err)
}
switch t.path {
case r.hostTOML:
hostURL = next
case r.podTOML:
podConfig = edited[i]
}
}
if podConfig == nil {
return fmt.Errorf("%s resolves to the same file as %s; the pods reach the database at another address and need a copy of their own", r.podTOML, r.hostTOML)
}
if err := r.record([2]string{installerDBKey, password}); err != nil {
return err
}
salt := make([]byte, 16)
if _, err := rand.Read(salt); err != nil {
return err
}
verifier, err := scramVerifier(password, salt, scramIterations)
if err != nil {
return err
}
if err := r.alterRole(ctx, cfg.Database.Deployment, role, verifier); err != nil {
return fmt.Errorf("set the role's password: %w", err)
}
if err := r.verifyDB(ctx, hostURL); err != nil {
return fmt.Errorf("the database does not accept the new password (%v); the config copies still hold the old one: run the installer again (sudo bash deploy/bootstrap.sh), which sets the password in %s everywhere", err, r.secretsEnv)
}
fmt.Fprintf(r.out, " - role %s: password changed, and the database accepts it\n", role)
for i, t := range targets {
info, err := os.Stat(t.real)
if err != nil {
return err
}
if err := replaceFileKeepingMode(t.real, info, edited[i]); err != nil {
return fmt.Errorf("write %s: %w", t.path, err)
}
fmt.Fprintf(r.out, " - %s: [database] url updated\n", t.path)
}
for _, ns := range r.withMinecraft() {
if err := r.putSecret(ctx, ns, platform.ConfigSecretName, map[string][]byte{platform.ConfigSecretKey: podConfig}); err != nil {
return err
}
}
return r.rollAPI(ctx)
}
// withPassword is a database URL with its password replaced.
func withPassword(raw, password string) (string, error) {
u, err := neturl.Parse(raw)
if err != nil || u.User == nil || u.User.Username() == "" {
return "", errors.New("its [database] url names no role")
}
u.User = neturl.UserPassword(u.User.Username(), password)
return u.String(), nil
}
// scramIterations is PostgreSQL's default scram_iterations.
const scramIterations = 4096
// scramVerifier is the SCRAM-SHA-256 verifier PostgreSQL stores for a password
// (RFC 5802 and RFC 7677, in the form libpq's PQencryptPasswordConn makes).
// ALTER ROLE stores a verifier as it is given, so the password itself never
// reaches the server, where a failing statement is logged with its text.
func scramVerifier(password string, salt []byte, iterations int) (string, error) {
salted, err := pbkdf2.Key(sha256.New, password, salt, iterations, sha256.Size)
if err != nil {
return "", err
}
mac := func(msg string) []byte {
h := hmac.New(sha256.New, salted)
h.Write([]byte(msg))
return h.Sum(nil)
}
stored := sha256.Sum256(mac("Client Key"))
b64 := base64.StdEncoding.EncodeToString
return fmt.Sprintf("SCRAM-SHA-256$%d:%s$%s:%s", iterations, b64(salt), b64(stored[:]), b64(mac("Server Key"))), nil
}
// alterRoleInPod runs ALTER ROLE as the superuser inside the database's
// container, over its socket, with the statement on stdin.
func alterRoleInPod(ctx context.Context, deployment, role, verifier string) error {
ns, name, _ := strings.Cut(deployment, "/")
return kubectlWithInput(ctx, []byte(alterRoleSQL(role, verifier)), "-n", ns, "exec", "-i", "deploy/"+name, "-c", platform.PostgresContainer, "--",
"psql", "-X", "-q", "-v", "ON_ERROR_STOP=1", "-U", "postgres", "-d", "postgres")
}
// alterRoleSQL sets role's password to a SCRAM verifier, which holds no quote.
func alterRoleSQL(role, verifier string) string {
return "ALTER ROLE \"" + strings.ReplaceAll(role, `"`, `""`) + "\" WITH PASSWORD '" + verifier + "';\n"
sec.Data[naming.ServiceTokenSecretKey] = []byte(token)
return cl.Update(ctx, &sec)
}
// setKeyValueLine rewrites the `key<sep>value` line of a flat key/value file
@@ -789,11 +259,6 @@ func alterRoleSQL(role, verifier string) string {
// file is replaced atomically and keeps its mode and owner: felis-link.properties
// is root:felis-velocity 0640, and the proxy must still be able to read it.
func setKeyValueLine(path, key, sep, value string) error {
return setKeyValueLines(path, sep, [][2]string{{key, value}})
}
// setKeyValueLines is setKeyValueLine for several keys in one rewrite.
func setKeyValueLines(path, sep string, kv [][2]string) error {
info, err := os.Stat(path)
if err != nil {
return err
@@ -803,8 +268,6 @@ func setKeyValueLines(path, sep string, kv [][2]string) error {
return err
}
lines := strings.Split(strings.TrimRight(string(raw), "\n"), "\n")
for _, p := range kv {
key, value := p[0], p[1]
found := false
for i, ln := range lines {
k, _, ok := strings.Cut(ln, sep)
@@ -816,14 +279,6 @@ func setKeyValueLines(path, sep string, kv [][2]string) error {
if !found {
lines = append(lines, key+sep+value)
}
}
return replaceFileKeepingMode(path, info, []byte(strings.Join(lines, "\n")+"\n"))
}
// replaceFileKeepingMode atomically replaces path with data, keeping the mode and
// owner info describes: these files are read by other users (the proxy's) and
// some hold credentials, so a rewrite must not widen or narrow who can read them.
func replaceFileKeepingMode(path string, info os.FileInfo, data []byte) error {
tmp, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".*")
if err != nil {
return err
@@ -839,7 +294,7 @@ func replaceFileKeepingMode(path string, info os.FileInfo, data []byte) error {
return err
}
}
if _, err := tmp.Write(data); err != nil {
if _, err := tmp.WriteString(strings.Join(lines, "\n") + "\n"); err != nil {
tmp.Close()
return err
}
+50 -650
View File
@@ -3,20 +3,13 @@ package main
import (
"bytes"
"context"
"encoding/base64"
"errors"
"fmt"
"os"
"path/filepath"
"regexp"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/maintenance"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
@@ -29,44 +22,23 @@ type rotationRig struct {
out *bytes.Buffer
events []string
dir string
clock time.Time
}
func tokenSecret(ns, name, val string) *corev1.Secret {
return keySecret(ns, name, "token", val)
}
func keySecret(ns, name, key, val string) *corev1.Secret {
return &corev1.Secret{
ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name},
Data: map[string][]byte{key: []byte(val)},
Data: map[string][]byte{"token": []byte(val)},
}
}
// serverPod is a game server's pod as the operator labels it.
func serverPod(ns, name, server string) *corev1.Pod {
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name,
Labels: map[string]string{
"felis.lolicon.best/server": server,
"felis.lolicon.best/managed-by": "felis-operator",
"felis.lolicon.best/component": "server",
}}}
}
// backupPod is a backup Job's pod as internal/backupjob labels it: it carries
// the server's label too, and no rotation may take it for the server's own.
func backupPod(ns, name, server string) *corev1.Pod {
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name,
Labels: map[string]string{
"felis.lolicon.best/server": server,
"app.kubernetes.io/managed-by": "felis-backup",
"app.kubernetes.io/component": "world-backup",
}}}
Labels: map[string]string{"felis.lolicon.best/server": server}}}
}
func newRotationRig(t *testing.T, objs ...client.Object) *rotationRig {
t.Helper()
rig := &rotationRig{out: &bytes.Buffer{}, dir: t.TempDir(), clock: time.Unix(1_800_000_000, 0)}
rig := &rotationRig{out: &bytes.Buffer{}, dir: t.TempDir()}
rig.cl = fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(objs...).Build()
rig.r = tokenRotator{
cl: rig.cl,
@@ -75,65 +47,27 @@ func newRotationRig(t *testing.T, objs ...client.Object) *rotationRig {
buildNS: "felis-build",
secretsEnv: filepath.Join(rig.dir, "secrets.env"),
linkProps: filepath.Join(rig.dir, "felis-link.properties"),
forwardingFile: filepath.Join(rig.dir, "forwarding.secret"),
hostTOML: filepath.Join(rig.dir, "felis.host.toml"),
podTOML: filepath.Join(rig.dir, "felis.pod.toml"),
defaultTOML: filepath.Join(rig.dir, "felis.toml"),
newToken: func() (string, error) { return "NEWTOKEN", nil },
rollout: func(_ context.Context, deployment string) error {
rig.events = append(rig.events, "roll "+deployment)
rollAPI: func(context.Context) error {
rig.events = append(rig.events, "roll-api")
return nil
},
restartUnit: func(_ context.Context, unit string) error {
rig.events = append(rig.events, "restart "+unit)
return nil
},
proxyLog: func(context.Context, time.Time) string { return "" },
alterRole: func(context.Context, string, string, string) error {
rig.events = append(rig.events, "alter-role")
return nil
},
verifyDB: func(context.Context, string) error {
rig.events = append(rig.events, "verify-db")
return nil
},
// Every look at the clock moves it a second on, so a wait measured with it
// ends after a known number of looks.
now: func() time.Time {
rig.clock = rig.clock.Add(time.Second)
return rig.clock
},
reloadWait: 5 * time.Second,
out: rig.out,
}
return rig
}
func (rig *rotationRig) secretKey(t *testing.T, ns, name, key string) string {
func (rig *rotationRig) secret(t *testing.T, ns, name string) string {
t.Helper()
var s corev1.Secret
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
return "<missing>"
}
return string(s.Data[key])
}
func (rig *rotationRig) secret(t *testing.T, ns, name string) string {
t.Helper()
return rig.secretKey(t, ns, name, "token")
}
func (rig *rotationRig) pods(t *testing.T) string {
t.Helper()
var pods corev1.PodList
if err := rig.cl.List(context.Background(), &pods, client.InNamespace("minecraft")); err != nil {
t.Fatal(err)
}
names := make([]string, len(pods.Items))
for i, p := range pods.Items {
names[i] = p.Name
}
return strings.Join(names, ",")
return string(s.Data["token"])
}
func writeTestFile(t *testing.T, path, body string, mode os.FileMode) {
@@ -146,15 +80,6 @@ func writeTestFile(t *testing.T, path, body string, mode os.FileMode) {
}
}
func readTestFile(t *testing.T, path string) string {
t.Helper()
raw, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
return string(raw)
}
func TestRotateLimboToken(t *testing.T) {
rig := newRotationRig(t,
tokenSecret("felis", "felis-limbo-token", "old"),
@@ -162,15 +87,14 @@ func TestRotateLimboToken(t *testing.T) {
tokenSecret("felis", "felis-service-token", "proxy"),
serverPod("minecraft", "login-0", "login"),
serverPod("minecraft", "survival-0", "survival"),
backupPod("minecraft", "login-backup-x", "login"),
)
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=old\nOPS_TOKEN=ops\n", 0o600)
// felis-api must roll only after both copies hold the new value, and the login
// pod must still be there then: restarting it earlier would have it present
// the new token to an api that does not know it yet.
rig.r.rollout = func(_ context.Context, deployment string) error {
rig.events = append(rig.events, "roll "+deployment)
rig.r.rollAPI = func(context.Context) error {
rig.events = append(rig.events, "roll-api")
if got := rig.secret(t, "felis", "felis-limbo-token"); got != "NEWTOKEN" {
t.Errorf("api rolled while the control Secret held %q", got)
}
@@ -183,12 +107,13 @@ func TestRotateLimboToken(t *testing.T) {
}
return nil
}
if err := rig.r.rotate(context.Background(), "limbo", true); err != nil {
if err := rig.r.rotate(context.Background(), "limbo"); err != nil {
t.Fatal(err)
}
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=NEWTOKEN\nOPS_TOKEN=ops\n" {
t.Errorf("secrets.env = %q", got)
raw, _ := os.ReadFile(rig.r.secretsEnv)
if string(raw) != "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=NEWTOKEN\nOPS_TOKEN=ops\n" {
t.Errorf("secrets.env = %q", raw)
}
if info, _ := os.Stat(rig.r.secretsEnv); info.Mode().Perm() != 0o600 {
t.Errorf("secrets.env mode = %v, want 0600", info.Mode().Perm())
@@ -196,12 +121,14 @@ func TestRotateLimboToken(t *testing.T) {
if got := rig.secret(t, "felis", "felis-service-token"); got != "proxy" {
t.Errorf("the proxy's token changed to %q", got)
}
// The login pod restarted; a user server and the backup Job's pod, which
// carries the login server's label too, are untouched.
if got := rig.pods(t); got != "login-backup-x,survival-0" {
t.Errorf("pods left = %s, want login-backup-x,survival-0", got)
var pods corev1.PodList
if err := rig.cl.List(context.Background(), &pods, client.InNamespace("minecraft")); err != nil {
t.Fatal(err)
}
if strings.Join(rig.events, ",") != "roll felis-api" {
if len(pods.Items) != 1 || pods.Items[0].Name != "survival-0" {
t.Errorf("pods left = %v, want only survival-0 (the login pod restarted, user servers untouched)", pods.Items)
}
if strings.Join(rig.events, ",") != "roll-api" {
t.Errorf("events = %v, want only the api roll (no unit restart for limbo)", rig.events)
}
if strings.Contains(rig.out.String(), "NEWTOKEN") {
@@ -209,56 +136,21 @@ func TestRotateLimboToken(t *testing.T) {
}
}
const testLinkProps = "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=old\nroot-domain=example.com\n"
func reloadLine(token string) string {
return "[12:00:00 INFO] [felis-link]: Felis: service-token reloaded from /x/felis-link.properties (fingerprint " + tokenFingerprint(token) + ")\n"
}
// The host proxy re-reads its token: the rotation waits for it to say so and
// leaves it running.
func TestRotateVelocityTokenReloadsTheHostProxy(t *testing.T) {
func TestRotateVelocityTokenOnTheHostProxy(t *testing.T) {
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=old\n", 0o600)
writeTestFile(t, rig.r.linkProps, testLinkProps, 0o640)
written := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
if err := os.Chtimes(rig.r.linkProps, written, written); err != nil {
t.Fatal(err)
}
writeTestFile(t, rig.r.linkProps, "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=old\nroot-domain=example.com\n", 0o640)
// The proxy logs its reload on its next call to felis-api, which may be
// while the api rolls: the log must be read from before that.
var reloadedAt time.Time
rig.r.rollout = func(_ context.Context, deployment string) error {
rig.events = append(rig.events, "roll "+deployment)
if got := readTestFile(t, rig.r.linkProps); !strings.Contains(got, "service-token=NEWTOKEN\n") {
t.Errorf("api rolled before the proxy's file held the new token: %q", got)
}
reloadedAt = rig.clock
return nil
}
looks := 0
rig.r.proxyLog = func(_ context.Context, since time.Time) string {
looks++
log := reloadLine("old") // an earlier rotation's
if looks >= 3 && !since.After(reloadedAt) {
log += reloadLine("NEWTOKEN")
}
return log
}
if err := rig.r.rotate(context.Background(), "velocity", true); err != nil {
if err := rig.r.rotate(context.Background(), "velocity"); err != nil {
t.Fatal(err)
}
if got := readTestFile(t, rig.r.linkProps); got != "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=NEWTOKEN\nroot-domain=example.com\n" {
t.Errorf("felis-link.properties = %q", got)
raw, _ := os.ReadFile(rig.r.linkProps)
if string(raw) != "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=NEWTOKEN\nroot-domain=example.com\n" {
t.Errorf("felis-link.properties = %q", raw)
}
info, _ := os.Stat(rig.r.linkProps)
if info.Mode().Perm() != 0o640 {
if info, _ := os.Stat(rig.r.linkProps); info.Mode().Perm() != 0o640 {
t.Errorf("properties mode = %v, want 0640 (the proxy's group must still read it)", info.Mode().Perm())
}
if !info.ModTime().Equal(written) {
t.Errorf("properties mtime = %v, want it kept at %v (felis domain check would read the proxy as stale)", info.ModTime(), written)
}
if got := rig.secret(t, "felis", "felis-service-token"); got != "NEWTOKEN" {
t.Errorf("control Secret = %q, want NEWTOKEN", got)
}
@@ -266,49 +158,9 @@ func TestRotateVelocityTokenReloadsTheHostProxy(t *testing.T) {
if got := rig.secret(t, "minecraft", "felis-service-token"); got != "<missing>" {
t.Errorf("rotation copied the proxy token into minecraft (%q)", got)
}
if strings.Join(rig.events, ",") != "roll felis-api" {
t.Errorf("events = %v, want the api roll and no proxy restart", rig.events)
}
if looks != 3 {
t.Errorf("the log was read %d times, want 3 (until the line appeared)", looks)
}
if out := rig.out.String(); !strings.Contains(out, "felis-velocity: took the new token from its properties; players stayed connected") {
t.Errorf("output does not say the players stayed: %s", out)
}
}
// A proxy that never logs the new fingerprint (an older plugin, a stuck proxy)
// is restarted once the wait is over.
func TestRotateVelocityTokenRestartsAProxyThatDoesNotReload(t *testing.T) {
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=old\n", 0o600)
writeTestFile(t, rig.r.linkProps, testLinkProps, 0o640)
looks := 0
rig.r.proxyLog = func(context.Context, time.Time) string {
looks++
return reloadLine("old")
}
if err := rig.r.rotate(context.Background(), "velocity", true); err != nil {
t.Fatal(err)
}
if strings.Join(rig.events, ",") != "roll felis-api,restart felis-velocity" {
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
t.Errorf("events = %v, want the api roll then the proxy restart", rig.events)
}
// reloadWait is 5s and each look moves the clock a second.
if looks != 5 {
t.Errorf("the log was read %d times, want 5", looks)
}
if out := rig.out.String(); !strings.Contains(out, "felis-velocity: had not taken the new token within 5s, restarted") {
t.Errorf("output does not explain the restart: %s", out)
}
}
// The proxy and the Java plugin name a token by the same fingerprint
// (plugins/shared LinkConfigLoaderTest fileTokenFollowsTheFile).
func TestTokenFingerprintMatchesThePlugin(t *testing.T) {
if got := tokenFingerprint("new-token"); got != "348e9df2a42b" {
t.Errorf("tokenFingerprint(new-token) = %q, want 348e9df2a42b", got)
}
}
// An external proxy has no felis-link.properties here: the Secret still rotates,
@@ -316,13 +168,13 @@ func TestTokenFingerprintMatchesThePlugin(t *testing.T) {
// without printing it.
func TestRotateVelocityTokenForAnExternalProxy(t *testing.T) {
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
if err := rig.r.rotate(context.Background(), "velocity", true); err != nil {
if err := rig.r.rotate(context.Background(), "velocity"); err != nil {
t.Fatal(err)
}
if got := rig.secret(t, "felis", "felis-service-token"); got != "NEWTOKEN" {
t.Errorf("control Secret = %q, want NEWTOKEN", got)
}
if strings.Join(rig.events, ",") != "roll felis-api" {
if strings.Join(rig.events, ",") != "roll-api" {
t.Errorf("events = %v, want no proxy restart", rig.events)
}
out := rig.out.String()
@@ -335,7 +187,7 @@ func TestRotateBuildTokenReachesTheBuildNamespace(t *testing.T) {
rig := newRotationRig(t, tokenSecret("felis", "felis-build-token", "old"))
// An install from before per-caller tokens has no BUILD_TOKEN line yet.
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=proxy", 0o600)
if err := rig.r.rotate(context.Background(), "build", true); err != nil {
if err := rig.r.rotate(context.Background(), "build"); err != nil {
t.Fatal(err)
}
if got := rig.secret(t, "felis-build", "felis-build-token"); got != "NEWTOKEN" {
@@ -344,8 +196,9 @@ func TestRotateBuildTokenReachesTheBuildNamespace(t *testing.T) {
if got := rig.secret(t, "felis", "felis-build-token"); got != "NEWTOKEN" {
t.Errorf("control Secret = %q, want NEWTOKEN", got)
}
if got := readTestFile(t, rig.r.secretsEnv); got != "SERVICE_TOKEN=proxy\nBUILD_TOKEN=NEWTOKEN\n" {
t.Errorf("secrets.env = %q", got)
raw, _ := os.ReadFile(rig.r.secretsEnv)
if string(raw) != "SERVICE_TOKEN=proxy\nBUILD_TOKEN=NEWTOKEN\n" {
t.Errorf("secrets.env = %q", raw)
}
}
@@ -356,453 +209,31 @@ func TestRotateStopsWhenTheAPIDoesNotRoll(t *testing.T) {
tokenSecret("felis", "felis-limbo-token", "old"),
serverPod("minecraft", "login-0", "login"),
)
rig.r.rollout = func(context.Context, string) error { return errors.New("rollout timed out") }
err := rig.r.rotate(context.Background(), "limbo", true)
rig.r.rollAPI = func(context.Context) error { return errors.New("rollout timed out") }
err := rig.r.rotate(context.Background(), "limbo")
if err == nil || !strings.Contains(err.Error(), "rollout timed out") {
t.Fatalf("err = %v, want the rollout failure", err)
}
if got := rig.pods(t); got != "login-0" {
t.Errorf("pods left = %s: the login pod was restarted although the api never rolled", got)
var pod corev1.Pod
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "login-0"}, &pod); err != nil {
t.Error("the login pod was restarted although the api never rolled")
}
}
func TestRotateRefusesAnUnknownCredential(t *testing.T) {
func TestRotateRefusesAnUnknownCaller(t *testing.T) {
rig := newRotationRig(t)
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=proxy\n", 0o600)
if err := rig.r.rotate(context.Background(), "admin", true); err == nil {
t.Fatal("rotated an unknown credential")
if err := rig.r.rotate(context.Background(), "admin"); err == nil {
t.Fatal("rotated a token for an unknown caller")
}
if got := readTestFile(t, rig.r.secretsEnv); got != "SERVICE_TOKEN=proxy\n" {
t.Errorf("secrets.env = %q, want it untouched", got)
if raw, _ := os.ReadFile(rig.r.secretsEnv); string(raw) != "SERVICE_TOKEN=proxy\n" {
t.Errorf("secrets.env = %q, want it untouched", raw)
}
if len(rig.events) != 0 {
t.Errorf("events = %v, want nothing touched", rig.events)
}
}
const (
testHostTOML = "# Written by the installer.\n[database]\nurl = \"postgres://felis:[email protected]:30432/felis?sslmode=disable\"\ndeployment = \"felis/felis-postgres\"\n\n[server]\nroot_domain = \"example.com\"\n"
testPodTOML = "[database]\n# The Service address.\nurl = \"postgres://felis:[email protected]:5432/felis?sslmode=disable\"\n\n[server]\nroot_domain = \"example.com\"\n"
)
// installedRig is a host as the installer leaves it, with every credential in
// place, for the plan test.
func installedRig(t *testing.T) *rotationRig {
t.Helper()
rig := newRotationRig(t,
tokenSecret("felis", "felis-service-token", "old"),
tokenSecret("felis", "felis-limbo-token", "old"),
tokenSecret("minecraft", "felis-limbo-token", "old"),
tokenSecret("felis", "felis-build-token", "old"),
tokenSecret("felis-build", "felis-build-token", "old"),
tokenSecret("felis", "felis-ops-token", "old"),
keySecret("felis", "felis-registry-auth", "platform", "old"),
keySecret("felis-build", "felis-registry-push", "password", "old"),
keySecret("felis", "felis-forwarding-secret", "secret", "old"),
keySecret("minecraft", "felis-forwarding-secret", "secret", "old"),
keySecret("felis", "felis-config", "felis.toml", testPodTOML),
keySecret("minecraft", "felis-config", "felis.toml", testPodTOML),
serverPod("minecraft", "login-0", "login"),
serverPod("minecraft", "survival-0", "survival"),
)
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=oldpw\nSERVICE_TOKEN=old\n", 0o600)
writeTestFile(t, rig.r.linkProps, testLinkProps, 0o640)
writeTestFile(t, rig.r.forwardingFile, "old", 0o640)
writeTestFile(t, rig.r.hostTOML, testHostTOML, 0o600)
writeTestFile(t, rig.r.podTOML, testPodTOML, 0o600)
return rig
}
// Without -yes every rotation prints its plan and changes nothing.
func TestRotatePlanChangesNothing(t *testing.T) {
for _, kind := range rotationKinds() {
t.Run(kind, func(t *testing.T) {
rig := installedRig(t)
files := map[string]string{}
for _, p := range []string{rig.r.secretsEnv, rig.r.linkProps, rig.r.forwardingFile, rig.r.hostTOML, rig.r.podTOML} {
files[p] = readTestFile(t, p)
}
rig.r.newToken = func() (string, error) {
t.Error("a plan generated a token")
return "NEWTOKEN", nil
}
if err := rig.r.rotate(context.Background(), kind, false); err != nil {
t.Fatal(err)
}
for p, before := range files {
if got := readTestFile(t, p); got != before {
t.Errorf("%s changed to %q", filepath.Base(p), got)
}
}
for _, s := range [][3]string{
{"felis", "felis-service-token", "token"}, {"felis", "felis-limbo-token", "token"},
{"felis-build", "felis-build-token", "token"}, {"felis", "felis-ops-token", "token"},
{"felis", "felis-registry-auth", "platform"}, {"felis-build", "felis-registry-push", "password"},
{"minecraft", "felis-forwarding-secret", "secret"},
} {
if got := rig.secretKey(t, s[0], s[1], s[2]); got != "old" {
t.Errorf("Secret %s/%s changed to %q", s[0], s[1], got)
}
}
if got := rig.secretKey(t, "minecraft", "felis-config", "felis.toml"); got != testPodTOML {
t.Errorf("felis-config changed to %q", got)
}
if len(rig.events) != 0 {
t.Errorf("events = %v, want none", rig.events)
}
if got := rig.pods(t); got != "login-0,survival-0" {
t.Errorf("pods left = %s, want all", got)
}
out := rig.out.String()
if !strings.HasSuffix(out, "\nNothing was changed. To rotate: sudo felis rotate-token -yes "+kind+"\n") {
t.Errorf("the plan does not end with how to go ahead: %s", out)
}
// Each plan names the interruption it causes.
want := apiRestartNote
if kind == kindForwarding {
want = "everyone online is disconnected"
}
if !strings.Contains(out, want) {
t.Errorf("the plan does not say %q: %s", want, out)
}
})
}
}
func TestRotateRegistryTokens(t *testing.T) {
rig := newRotationRig(t,
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: "felis-registry-auth"},
Data: map[string][]byte{"platform": []byte("a"), "build": []byte("b"), "prune": []byte("c")}},
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis-build", Name: "felis-registry-push"},
Data: map[string][]byte{"username": []byte("build"), "password": []byte("b")}},
)
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nREGISTRY_PLATFORM_TOKEN=a\nREGISTRY_BUILD_TOKEN=b\nREGISTRY_PRUNE_TOKEN=c\n", 0o600)
n := 0
rig.r.newToken = func() (string, error) {
n++
return fmt.Sprintf("NEWTOKEN%d", n), nil
}
// The registry reads its tokens as it starts: it restarts after the Secret
// holds them, and felis-api (the prune token) after that.
rig.r.rollout = func(_ context.Context, deployment string) error {
rig.events = append(rig.events, "roll "+deployment)
if got := rig.secretKey(t, "felis", "felis-registry-auth", "prune"); got != "NEWTOKEN3" {
t.Errorf("%s rolled while the prune token was %q", deployment, got)
}
return nil
}
if err := rig.r.rotate(context.Background(), kindRegistry, true); err != nil {
t.Fatal(err)
}
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=db\nREGISTRY_PLATFORM_TOKEN=NEWTOKEN1\nREGISTRY_BUILD_TOKEN=NEWTOKEN2\nREGISTRY_PRUNE_TOKEN=NEWTOKEN3\n" {
t.Errorf("secrets.env = %q", got)
}
for key, want := range map[string]string{"platform": "NEWTOKEN1", "build": "NEWTOKEN2", "prune": "NEWTOKEN3"} {
if got := rig.secretKey(t, "felis", "felis-registry-auth", key); got != want {
t.Errorf("felis-registry-auth %s = %q, want %s", key, got, want)
}
}
if got := rig.secretKey(t, "felis-build", "felis-registry-push", "password"); got != "NEWTOKEN2" {
t.Errorf("the build Jobs' push password = %q, want the build token", got)
}
if got := rig.secretKey(t, "felis-build", "felis-registry-push", "username"); got != "build" {
t.Errorf("the build Jobs' push username = %q, want build", got)
}
if strings.Join(rig.events, ",") != "roll registry,roll felis-api" {
t.Errorf("events = %v, want the registry then felis-api", rig.events)
}
if strings.Contains(rig.out.String(), "NEWTOKEN") {
t.Errorf("a new token was printed: %s", rig.out.String())
}
}
func TestRotateForwardingSecret(t *testing.T) {
lock := time.Unix(1_800_000_000, 0)
rig := newRotationRig(t,
keySecret("felis", "felis-forwarding-secret", "secret", "old"),
keySecret("minecraft", "felis-forwarding-secret", "secret", "old"),
serverPod("minecraft", "survival-0", "survival"),
serverPod("minecraft", "lobby-0", "lobby"),
// creative is being backed up: a running Job holds its world.
serverPod("minecraft", "creative-0", "creative"),
&batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "creative-backup",
Labels: map[string]string{maintenance.LabelServer: "creative", maintenance.LabelManagedBy: "felis-backup"}}},
// survival's last backup is over; its finished pod stays until the Job's TTL.
&batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "survival-backup",
Labels: map[string]string{maintenance.LabelServer: "survival", maintenance.LabelManagedBy: "felis-backup"}},
Status: batchv1.JobStatus{Conditions: []batchv1.JobCondition{{Type: batchv1.JobComplete, Status: corev1.ConditionTrue}}}},
backupPod("minecraft", "survival-backup-x", "survival"),
// A pod already on its way out is neither counted nor deleted again.
func() *corev1.Pod {
p := serverPod("minecraft", "lobby-old", "lobby")
p.DeletionTimestamp = &metav1.Time{Time: lock}
p.Finalizers = []string{"felis.lolicon.best/test"}
return p
}(),
// skyblock's restore has just been admitted, its Job not created yet.
serverPod("minecraft", "skyblock-0", "skyblock"),
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "skyblock",
Annotations: map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lock)}}},
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "survival"}},
)
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nFORWARDING_SECRET=old\n", 0o600)
writeTestFile(t, rig.r.forwardingFile, "old", 0o640)
// The proxy restarts after the servers were told to: a player who rejoins
// must not meet a server still on the old secret.
rig.r.restartUnit = func(_ context.Context, unit string) error {
rig.events = append(rig.events, "restart "+unit)
if got := rig.pods(t); strings.Contains(got, "survival-0") {
t.Errorf("the proxy restarted before the servers (pods %s)", got)
}
if got := readTestFile(t, rig.r.forwardingFile); got != "NEWTOKEN" {
t.Errorf("the proxy restarted on forwarding.secret %q", got)
}
return nil
}
if err := rig.r.rotate(context.Background(), kindForwarding, true); err != nil {
t.Fatal(err)
}
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=db\nFORWARDING_SECRET=NEWTOKEN\n" {
t.Errorf("secrets.env = %q", got)
}
if info, _ := os.Stat(rig.r.forwardingFile); info.Mode().Perm() != 0o640 {
t.Errorf("forwarding.secret mode = %v, want 0640", info.Mode().Perm())
}
for _, ns := range []string{"felis", "minecraft"} {
if got := rig.secretKey(t, ns, "felis-forwarding-secret", "secret"); got != "NEWTOKEN" {
t.Errorf("Secret %s/felis-forwarding-secret = %q, want NEWTOKEN", ns, got)
}
}
if got := rig.pods(t); got != "creative-0,lobby-old,skyblock-0,survival-backup-x" {
t.Errorf("pods left = %s, want the held servers, the terminating pod and the backup's pod", got)
}
if strings.Join(rig.events, ",") != "restart felis-velocity" {
t.Errorf("events = %v, want only the proxy restart (felis-api does not hold this secret)", rig.events)
}
out := rig.out.String()
for _, want := range []string{
"every running server restarts to read it (2 now)",
"left running, because a backup, restore or file write holds its world: creative (backup), skyblock (restore).",
" - servers: 2 restarting on the new secret\n",
} {
if !strings.Contains(out, want) {
t.Errorf("output lacks %q: %s", want, out)
}
}
if strings.Contains(out, "NEWTOKEN") {
t.Errorf("the new secret was printed: %s", out)
}
}
func TestRotateForwardingSecretForAnExternalProxy(t *testing.T) {
rig := newRotationRig(t, keySecret("felis", "felis-forwarding-secret", "secret", "old"))
if err := rig.r.rotate(context.Background(), kindForwarding, true); err != nil {
t.Fatal(err)
}
if got := rig.secretKey(t, "felis", "felis-forwarding-secret", "secret"); got != "NEWTOKEN" {
t.Errorf("control Secret = %q, want NEWTOKEN", got)
}
if len(rig.events) != 0 {
t.Errorf("events = %v, want no proxy restart on this host", rig.events)
}
if out := rig.out.String(); !strings.Contains(out, "put the value in Secret felis/felis-forwarding-secret into your proxy's forwarding secret file") {
t.Errorf("output does not say where the value is: %s", out)
}
}
// dbRig is an installed host for the database rotation: felis.toml links to the
// host copy, as the installer makes it.
func dbRig(t *testing.T) *rotationRig {
t.Helper()
rig := newRotationRig(t,
keySecret("felis", "felis-config", "felis.toml", testPodTOML),
keySecret("minecraft", "felis-config", "felis.toml", testPodTOML),
)
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=oldpw\nSERVICE_TOKEN=s\n", 0o600)
writeTestFile(t, rig.r.hostTOML, testHostTOML, 0o600)
writeTestFile(t, rig.r.podTOML, testPodTOML, 0o600)
if err := os.Symlink("felis.host.toml", rig.r.defaultTOML); err != nil {
t.Fatal(err)
}
return rig
}
const (
wantHostTOML = "# Written by the installer.\n[database]\nurl = \"postgres://felis:[email protected]:30432/felis?sslmode=disable\"\ndeployment = \"felis/felis-postgres\"\n\n[server]\nroot_domain = \"example.com\"\n"
wantPodTOML = "[database]\n# The Service address.\nurl = \"postgres://felis:[email protected]:5432/felis?sslmode=disable\"\n\n[server]\nroot_domain = \"example.com\"\n"
)
func TestRotateDatabasePassword(t *testing.T) {
rig := dbRig(t)
rig.r.alterRole = func(_ context.Context, deployment, role, verifier string) error {
rig.events = append(rig.events, "alter-role")
if deployment != "felis/felis-postgres" || role != "felis" {
t.Errorf("alterRole(%q, %q), want felis/felis-postgres and felis", deployment, role)
}
if !strings.Contains(readTestFile(t, rig.r.secretsEnv), "DB_PASSWORD=NEWTOKEN\n") {
t.Error("the role changed before secrets.env recorded the new password")
}
// The verifier is the new password's, under the salt it carries.
m := regexp.MustCompile(`^SCRAM-SHA-256\$4096:([^$]+)\$`).FindStringSubmatch(verifier)
if m == nil {
t.Fatalf("verifier %q is not a SCRAM-SHA-256 verifier", verifier)
}
salt, err := base64.StdEncoding.DecodeString(m[1])
if err != nil || len(salt) != 16 {
t.Fatalf("verifier salt %q: %v", m[1], err)
}
if want, _ := scramVerifier("NEWTOKEN", salt, 4096); verifier != want {
t.Errorf("verifier = %q, want %q", verifier, want)
}
return nil
}
// The configs change only once the database accepts the new password.
rig.r.verifyDB = func(_ context.Context, url string) error {
rig.events = append(rig.events, "verify-db")
if url != "postgres://felis:[email protected]:30432/felis?sslmode=disable" {
t.Errorf("verified with %q, want the host URL with the new password", url)
}
if got := readTestFile(t, rig.r.hostTOML); got != testHostTOML {
t.Errorf("the host config changed before the database accepted the password: %q", got)
}
return nil
}
rig.r.rollout = func(_ context.Context, deployment string) error {
rig.events = append(rig.events, "roll "+deployment)
if got := rig.secretKey(t, "felis", "felis-config", "felis.toml"); got != wantPodTOML {
t.Errorf("felis-api rolled on felis-config %q", got)
}
return nil
}
if err := rig.r.rotate(context.Background(), kindDB, true); err != nil {
t.Fatal(err)
}
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=NEWTOKEN\nSERVICE_TOKEN=s\n" {
t.Errorf("secrets.env = %q", got)
}
if got := readTestFile(t, rig.r.hostTOML); got != wantHostTOML {
t.Errorf("host config = %q", got)
}
if got := readTestFile(t, rig.r.podTOML); got != wantPodTOML {
t.Errorf("pod config = %q", got)
}
if info, err := os.Lstat(rig.r.defaultTOML); err != nil || info.Mode()&os.ModeSymlink == 0 {
t.Errorf("felis.toml is no longer the link to the host copy (%v)", err)
}
for _, ns := range []string{"felis", "minecraft"} {
if got := rig.secretKey(t, ns, "felis-config", "felis.toml"); got != wantPodTOML {
t.Errorf("Secret %s/felis-config = %q", ns, got)
}
}
if strings.Join(rig.events, ",") != "alter-role,verify-db,roll felis-api" {
t.Errorf("events = %v", rig.events)
}
if strings.Contains(rig.out.String(), "NEWTOKEN") {
t.Errorf("the new password was printed: %s", rig.out.String())
}
}
// A password the database does not accept leaves the configs on the old one
// and felis-api running: the installer's record, already new, is the way back.
func TestRotateDatabasePasswordStopsWhenTheDatabaseRefuses(t *testing.T) {
rig := dbRig(t)
rig.r.verifyDB = func(context.Context, string) error {
rig.events = append(rig.events, "verify-db")
return errors.New("password authentication failed")
}
err := rig.r.rotate(context.Background(), kindDB, true)
if err == nil || !strings.Contains(err.Error(), "password authentication failed") || !strings.Contains(err.Error(), "run the installer again") {
t.Fatalf("err = %v, want the refusal and the way back", err)
}
if got := readTestFile(t, rig.r.hostTOML); got != testHostTOML {
t.Errorf("host config = %q, want it unchanged", got)
}
if got := readTestFile(t, rig.r.podTOML); got != testPodTOML {
t.Errorf("pod config = %q, want it unchanged", got)
}
if got := rig.secretKey(t, "felis", "felis-config", "felis.toml"); got != testPodTOML {
t.Errorf("felis-config = %q, want it unchanged", got)
}
if strings.Join(rig.events, ",") != "alter-role,verify-db" {
t.Errorf("events = %v, want no api roll", rig.events)
}
}
// Everything that can refuse does so before the installer's record or the
// role change; what the host config alone shows is refused by the plan too.
func TestRotateDatabasePasswordRefusesEarly(t *testing.T) {
for _, tc := range []struct {
name string
apply bool
setup func(t *testing.T, rig *rotationRig)
want string
}{
{"a database the installer does not run", false, func(t *testing.T, rig *rotationRig) {
writeTestFile(t, rig.r.hostTOML, strings.Replace(testHostTOML, "deployment = \"felis/felis-postgres\"\n", "", 1), 0o600)
}, "[database] deployment is unset"},
{"a database url without a role", false, func(t *testing.T, rig *rotationRig) {
writeTestFile(t, rig.r.hostTOML, strings.Replace(testHostTOML, "felis:oldpw@", "", 1), 0o600)
}, "names no role"},
{"a config copy the line editor cannot edit", true, func(t *testing.T, rig *rotationRig) {
writeTestFile(t, rig.r.podTOML, "[database]\nurl = \"\"\"\npostgres://felis:[email protected]:5432/felis\"\"\"\n", 0o600)
}, "set the password in its [database] url by hand"},
{"a pod copy that is the host copy", true, func(t *testing.T, rig *rotationRig) {
if err := os.Remove(rig.r.podTOML); err != nil {
t.Fatal(err)
}
if err := os.Symlink("felis.host.toml", rig.r.podTOML); err != nil {
t.Fatal(err)
}
}, "resolves to the same file as"},
} {
t.Run(tc.name, func(t *testing.T) {
rig := dbRig(t)
tc.setup(t, rig)
err := rig.r.rotate(context.Background(), kindDB, tc.apply)
if err == nil || !strings.Contains(err.Error(), tc.want) {
t.Fatalf("err = %v, want %q", err, tc.want)
}
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=oldpw\nSERVICE_TOKEN=s\n" {
t.Errorf("secrets.env = %q, want it untouched", got)
}
if len(rig.events) != 0 {
t.Errorf("events = %v, want nothing done", rig.events)
}
})
}
}
// The RFC 7677 example (user "user", password "pencil"), with the stored and
// server keys computed by openssl's PBKDF2 and HMAC rather than this code.
func TestScramVerifier(t *testing.T) {
salt, _ := base64.StdEncoding.DecodeString("W22ZaJ0SNY7soEsUEjb6gQ==")
got, err := scramVerifier("pencil", salt, 4096)
if err != nil {
t.Fatal(err)
}
if want := "SCRAM-SHA-256$4096:W22ZaJ0SNY7soEsUEjb6gQ==$WG5d8oPm3OtcPnkdi4Uo7BkeZkBFzpcXkuLmtbsT4qY=:wfPLwcE6nTWhTAmQ7tl2KeoiWGPlZqQxSrmfPwDl2dU="; got != want {
t.Errorf("scramVerifier = %q\nwant %q", got, want)
}
}
func TestAlterRoleSQL(t *testing.T) {
if got := alterRoleSQL(`fe"lis`, "SCRAM-SHA-256$4096:c2FsdA==$a:b"); got != "ALTER ROLE \"fe\"\"lis\" WITH PASSWORD 'SCRAM-SHA-256$4096:c2FsdA==$a:b';\n" {
t.Errorf("alterRoleSQL = %q", got)
}
}
// installerSecretKeys is every secrets.env key a rotation writes.
func installerSecretKeys() map[string]bool {
keys := map[string]bool{installerForwardingKey: true, installerDBKey: true}
for _, k := range installerTokenKeys {
keys[k] = true
}
for _, k := range installerRegistryKeys {
keys[k] = true
}
return keys
}
// The installer and rotate-token must agree on where each caller's token lives:
// rotate-token writes installerTokenKeys into secrets.env, and the installer
// applies those same keys to the Secrets on its next run. A key the installer
@@ -818,6 +249,12 @@ func TestInstallerProvisionsEveryCallerToken(t *testing.T) {
if !ok {
t.Fatalf("installerTokenKeys names unknown caller %q", caller)
}
if !strings.Contains(script, key+`="${`+key+`:-$(openssl rand -hex 32)}"`) {
t.Errorf("bootstrap.sh does not generate %s", key)
}
if !strings.Contains(script, key+"=${"+key+"}\n") {
t.Errorf("bootstrap.sh does not persist %s to secrets.env", key)
}
apply := regexp.MustCompile(`apply_literal_secret "\$CONTROL_NS" ` + regexp.QuoteMeta(ct.Secret) + ` token "\$` + key + `"`)
if !apply.MatchString(script) {
t.Errorf("bootstrap.sh does not apply %s from %s in the control namespace", ct.Secret, key)
@@ -844,40 +281,3 @@ func TestInstallerProvisionsEveryCallerToken(t *testing.T) {
t.Errorf("installerTokenKeys covers %d callers, naming.CallerTokens lists %d", len(installerTokenKeys), len(callerNames()))
}
}
// Every value the installer keeps in secrets.env has a rotation that rewrites
// that very key, and every key a rotation writes is one the installer
// generates, keeps and so re-applies on its next run.
func TestInstallerSecretsAreAllRotatable(t *testing.T) {
raw, err := os.ReadFile(filepath.Join("..", "..", "deploy", "bootstrap.sh"))
if err != nil {
t.Fatal(err)
}
script := string(raw)
m := regexp.MustCompile(`(?s)write_file_atomic "\$SECRETS_ENV" 0600 <<EOF\n(.*?)\nEOF\n`).FindStringSubmatch(script)
if m == nil {
t.Fatal("bootstrap.sh: no secrets.env heredoc found")
}
persisted := map[string]bool{}
for _, line := range strings.Split(m[1], "\n") {
key, value, _ := strings.Cut(line, "=")
if value != "${"+key+"}" {
t.Errorf("secrets.env line %q is not KEY=${KEY}", line)
}
persisted[key] = true
}
rotated := installerSecretKeys()
for key := range persisted {
if !rotated[key] {
t.Errorf("secrets.env keeps %s, which no rotation replaces", key)
}
}
for key := range rotated {
if !persisted[key] {
t.Errorf("a rotation writes %s, which the installer does not keep in secrets.env", key)
}
if !regexp.MustCompile(`\n ` + key + `="\$\{` + key + `:-\$\(openssl rand -hex [0-9]+\)\}"\n`).MatchString(script) {
t.Errorf("bootstrap.sh does not generate %s when secrets.env lacks it", key)
}
}
}
+3 -16
View File
@@ -20,10 +20,8 @@ Commands:
reaper Run the world reaper / backup batch
restore Extract a world archive into a world volume (internal Job entrypoint)
backup Archive a world into the backup store and record it (internal Job entrypoint)
backup-now Archive every user server's world now, one at a time (or the named ones; -stop stops running ones first; prints the plan, -yes applies; requires root/sudo)
files List, read, write, mkdir, delete, rename, upload or unzip one path in a stopped server's world (internal Job entrypoint)
export Archive a stopped server's world, or read one of its backups, and hand it to felis-api for download (internal Job entrypoint)
egress-gate Hold a build or game server pod until its egress NetworkPolicy is enforced (internal init container entrypoint)
files List/read/write one file in a stopped server's world (internal Job entrypoint)
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
scan-gate Apply the scan policy to a build's Trivy report and hand felis-api the report and SBOM (internal Job entrypoint)
push-image Push a scanned image tarball to the registry (internal Job entrypoint)
@@ -33,11 +31,7 @@ Commands:
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
converge Fill in fields a newer desired spec added to already-installed system servers
rotate-token Replace a generated credential and restart what reads it (velocity|limbo|build|ops|registry|forwarding|db; prints the plan, -yes applies; requires root/sudo)
domain Move the install to a new root domain on every surface that carries it, or check each one (set|check; requires root/sudo)
status Print the platform at a glance: node, control plane, proxy, servers, backups, host, open alerts (requires root/sudo)
doctor Run every health check once, grouped by area, with where to look next; mails nothing (requires root/sudo)
support-bundle Collect status, doctor, logs and cluster state into one redacted tar.gz to share when asking for help (requires root/sudo)
rotate-token Replace one internal caller's token and restart what holds it (velocity|limbo|build|ops; requires root/sudo)
watchdog Check the platform once and mail the owners what has gone wrong (run by felis-watchdog.timer)
version Print the build stamp of this binary
update Report which platform components have updates available
@@ -65,9 +59,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
"reaper": cmdReaper,
"restore": cmdRestore,
"backup": cmdBackup,
"backup-now": cmdBackupNow,
"files": cmdFiles,
"export": cmdExport,
"egress-gate": cmdEgressGate,
"fetch-context": cmdFetchContext,
"scan-gate": cmdScanGate,
@@ -79,19 +71,14 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
"setup": cmdSetup,
"converge": cmdConverge,
"rotate-token": cmdRotateToken,
"domain": cmdDomain,
"breakGlass": cmdBreakGlass,
"bootstrap-assets": cmdBootstrapAssets,
"init-forwarding": cmdInitForwarding,
"init-volume": cmdInitVolume,
"pin-images": cmdPinImages,
"image-bundle": cmdImageBundle,
"version": cmdVersion,
"update": cmdUpdate,
"watchdog": cmdWatchdog,
"status": cmdStatus,
"doctor": cmdDoctor,
"support-bundle": cmdSupportBundle,
}
// run dispatches a subcommand. It is separate from main so the router is
+1 -1
View File
@@ -43,7 +43,7 @@ func TestRunUnknownCommand(t *testing.T) {
// decision rather than an oversight.
var undocumentedCommands = map[string]bool{
"bootstrap-assets": true, "init-forwarding": true, "init-volume": true,
"pin-images": true, "image-bundle": true,
"pin-images": true,
}
// The usage text and the dispatch table must describe the same set of commands.
+1 -1
View File
@@ -112,7 +112,7 @@ func cmdSetup(args []string, stdout, stderr io.Writer) int {
return 1
}
res, err := runSetupTUI(ctx, setup.repo, setup.cfg.Database, setup.cfg.Server.RootDomain, setup.cfg.Auth.AdminHostname, setup.cfg.Auth.PanelHostname, setup.cfg.Auth.AccessJWTAud, setup.cfg.K8s.Namespace, accountableOSUser(), setup.adminExists)
res, err := runSetupTUI(ctx, setup.repo, setup.cfg.Database.URL, setup.cfg.Server.RootDomain, setup.cfg.Auth.AdminHostname, setup.cfg.Auth.PanelHostname, setup.cfg.Auth.AccessJWTAud, setup.cfg.K8s.Namespace, accountableOSUser(), setup.adminExists)
if err != nil {
fmt.Fprintf(stderr, "felis setup: %v\n", err)
return 1
-438
View File
@@ -1,438 +0,0 @@
package main
import (
"bufio"
"context"
"errors"
"flag"
"fmt"
"io"
"net"
"os"
"path/filepath"
"sort"
"strconv"
"strings"
"syscall"
"text/tabwriter"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
"felis.lolicon.best/internal/watchdog"
appsv1 "k8s.io/api/apps/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/labels"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// cmdStatus prints the platform at a glance: the release and the node, the
// control plane's workloads, the game proxy, every server with its players
// and newest world backup, the database backups and the off-site copy, the
// host's disks and memory, and what the watchdog has open. It changes
// nothing, and a part that is down (the cluster, PostgreSQL) reads as such
// while the rest still prints. felis doctor says what is wrong and where to
// look.
func cmdStatus(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("status", flag.ContinueOnError)
fs.SetOutput(stderr)
unitDir := fs.String("systemd-dir", systemdUnitDir, "where the installer's systemd units are")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if os.Geteuid() != 0 {
fmt.Fprintln(stderr, "felis status: run as root (sudo felis status): it reads root-only state under /etc/felis and /var/lib/felis")
return 1
}
env, err := hostStatusEnv(*unitDir)
if err != nil {
fmt.Fprintf(stderr, "felis status: %v\n", err)
return 1
}
ctx, cancel := context.WithTimeout(context.Background(), time.Minute)
defer cancel()
printStatus(ctx, env, stdout)
return 0
}
// statusEnv is what one status report reads the host through.
type statusEnv struct {
cfg *config.Config
// w is how felis-watchdog.service runs the watchdog: where the backups,
// the off-site record and the proxy are.
w watchdogFlags
cl client.Client // nil while the API server is unreachable
clErr error
backups func(ctx context.Context) (map[string]time.Time, error)
run func(ctx context.Context, name string, args ...string) ([]byte, error)
unitDir string
meminfo string
host string
now time.Time
}
// hostStatusEnv reads this host: the watchdog's settings, the configuration
// they name, and the cluster.
func hostStatusEnv(unitDir string) (statusEnv, error) {
w, _, err := watchdogUnitFlags(filepath.Join(unitDir, "felis-watchdog.service"))
if err != nil {
return statusEnv{}, err
}
cfg, err := config.Load(w.cfgPath)
if err != nil {
return statusEnv{}, err
}
cl, clErr := buildSystemServerClient()
if clErr != nil {
cl = nil
}
host, _ := os.Hostname()
return statusEnv{
cfg: cfg, w: w, cl: cl, clErr: clErr,
backups: func(ctx context.Context) (map[string]time.Time, error) {
return newestWorldBackups(ctx, cfg.Database.URL)
},
run: hostCommand, unitDir: unitDir, meminfo: "/proc/meminfo", host: host, now: time.Now(),
}, nil
}
func printStatus(ctx context.Context, env statusEnv, out io.Writer) {
fmt.Fprintf(out, "felis %s on %s at %s\n", resolvedVersion(), env.host, env.now.UTC().Format("2006-01-02 15:04 UTC"))
if env.cl == nil {
fmt.Fprintf(out, "cluster: unreachable (%v)\n", env.clErr)
} else {
statusCluster(ctx, env, out)
}
statusProxy(ctx, env, out)
statusServers(ctx, env, out)
statusBackups(env, out)
statusHost(env, out)
statusWatchdog(ctx, env, out)
}
func statusCluster(ctx context.Context, env statusEnv, out io.Writer) {
var nodes corev1.NodeList
if err := env.cl.List(ctx, &nodes); err != nil {
fmt.Fprintf(out, "cluster: unreachable (%v)\n", err)
return
}
for _, n := range nodes.Items {
ready := "NotReady"
for _, c := range n.Status.Conditions {
if c.Type == corev1.NodeReady && c.Status == corev1.ConditionTrue {
ready = "Ready"
}
}
info := n.Status.NodeInfo
fmt.Fprintf(out, "node: %s %s, k3s %s, %s, kernel %s\n", n.Name, ready, info.KubeletVersion, info.OSImage, info.KernelVersion)
}
ns := env.w.controlNS
var deps appsv1.DeploymentList
var pods corev1.PodList
err := env.cl.List(ctx, &deps, client.InNamespace(ns))
if err == nil {
err = env.cl.List(ctx, &pods, client.InNamespace(ns))
}
fmt.Fprintf(out, "\ncontrol plane (namespace %s):\n", ns)
if err != nil {
fmt.Fprintf(out, " cannot list it: %v\n", err)
return
}
sort.Slice(deps.Items, func(i, j int) bool { return deps.Items[i].Name < deps.Items[j].Name })
tw := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0)
for _, d := range deps.Items {
want := int32(1)
if d.Spec.Replicas != nil {
want = *d.Spec.Replicas
}
image := "-"
if cs := d.Spec.Template.Spec.Containers; len(cs) > 0 {
image = shortImage(cs[0].Image)
}
fmt.Fprintf(tw, " %s\t%d/%d ready\t%s\trestarts %d\n", d.Name, d.Status.ReadyReplicas, want, image, podRestarts(d.Spec.Selector, pods.Items))
}
tw.Flush()
}
// podRestarts adds up the container restarts of the pods selector picks (the
// API server refuses a Deployment whose selector is empty).
func podRestarts(selector *metav1.LabelSelector, pods []corev1.Pod) int32 {
sel, err := metav1.LabelSelectorAsSelector(selector)
if err != nil {
return 0
}
var n int32
for _, p := range pods {
if !sel.Matches(labels.Set(p.Labels)) {
continue
}
for _, cs := range p.Status.ContainerStatuses {
n += cs.RestartCount
}
}
return n
}
// shortImage is an image reference without its registry and repository path,
// and with its digest cut to 12 characters.
func shortImage(ref string) string {
if i := strings.LastIndex(ref, "/"); i >= 0 {
ref = ref[i+1:]
}
if name, digest, ok := strings.Cut(ref, "@sha256:"); ok && len(digest) > 12 {
ref = name + "@" + digest[:12]
}
return ref
}
func statusProxy(ctx context.Context, env statusEnv, out io.Writer) {
var parts []string
if _, err := os.Stat(filepath.Join(env.unitDir, "felis-velocity.service")); err == nil {
parts = append(parts, "felis-velocity "+unitActiveState(ctx, doctorEnv{run: env.run}, "felis-velocity.service"))
}
if env.w.proxyAddr != "" {
d := net.Dialer{Timeout: 3 * time.Second}
if conn, err := d.DialContext(ctx, "tcp", env.w.proxyAddr); err != nil {
parts = append(parts, fmt.Sprintf("%s refuses connections (%v)", env.w.proxyAddr, err))
} else {
conn.Close()
parts = append(parts, env.w.proxyAddr+" accepts connections")
}
}
if len(parts) == 0 {
parts = append(parts, "not on this host")
}
fmt.Fprintf(out, "\nproxy: %s\n", strings.Join(parts, ", "))
}
func statusServers(ctx context.Context, env statusEnv, out io.Writer) {
ns := env.cfg.K8s.Namespace
if ns == "" {
ns = platform.DefaultMinecraftNamespace
}
if env.cl == nil {
fmt.Fprintf(out, "\nservers (namespace %s): unknown while the cluster is unreachable\n", ns)
return
}
var list v1alpha1.MinecraftServerList
if err := env.cl.List(ctx, &list, client.InNamespace(ns)); err != nil {
fmt.Fprintf(out, "\nservers (namespace %s): cannot list them: %v\n", ns, err)
return
}
newest, backupErr := env.backups(ctx)
running, online := 0, int32(0)
for _, ms := range list.Items {
if ms.Status.Phase == v1alpha1.PhaseRunning {
running++
online += ms.Status.Players.Online
}
}
fmt.Fprintf(out, "\nservers (namespace %s): %d, %d running, %d players online\n", ns, len(list.Items), running, online)
if len(list.Items) == 0 {
return
}
sort.Slice(list.Items, func(i, j int) bool { return list.Items[i].Name < list.Items[j].Name })
tw := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0)
fmt.Fprintln(tw, " NAME\tROLE\tDESIRED\tPHASE\tPLAYERS\tNEWEST WORLD BACKUP")
for _, ms := range list.Items {
role := ms.Labels[v1alpha1.LabelSystemRole]
if role == "" {
role = "-"
}
phase := string(ms.Status.Phase)
if phase == "" {
phase = "-"
}
players := "-"
if ms.Status.Phase == v1alpha1.PhaseRunning {
players = fmt.Sprintf("%d/%d", ms.Status.Players.Online, ms.Status.Players.Max)
}
backup := "none"
switch at, ok := newest[ms.Name]; {
case backupErr != nil:
backup = "?"
case ok:
backup = dbbackup.Age(env.now.Sub(at)) + " ago"
}
fmt.Fprintf(tw, " %s\t%s\t%s\t%s\t%s\t%s\n", ms.Name, role, orDash(string(ms.Spec.DesiredState)), phase, players, backup)
}
tw.Flush()
if backupErr != nil {
fmt.Fprintf(out, " world backups unknown: %v\n", backupErr)
}
}
func orDash(s string) string {
if s == "" {
return "-"
}
return s
}
// newestWorldBackups is when each server's newest world backup that a restore
// can use was taken.
func newestWorldBackups(ctx context.Context, url string) (map[string]time.Time, error) {
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
defer cancel()
drv, err := store.Open(ctx, url)
if err != nil {
return nil, err
}
defer drv.Close()
rows, err := drv.DB().QueryContext(ctx, `SELECT server_name, max(created_at) FROM world_backups
WHERE status = 'present' AND corrupt_at IS NULL GROUP BY server_name`)
if err != nil {
return nil, err
}
defer rows.Close()
out := map[string]time.Time{}
for rows.Next() {
var name string
var at time.Time
if err := rows.Scan(&name, &at); err != nil {
return nil, err
}
out[name] = at
}
return out, rows.Err()
}
func statusBackups(env statusEnv, out io.Writer) {
fmt.Fprintln(out, "\nbackups:")
switch bundles, err := dbbackup.List(env.w.backupDir); {
case env.w.backupDir == "":
fmt.Fprintln(out, " database: not checked (felis-watchdog.service names no -backup-dir)")
case err != nil:
fmt.Fprintf(out, " database: cannot read %s: %v\n", env.w.backupDir, err)
case len(bundles) == 0:
fmt.Fprintf(out, " database: none in %s\n", env.w.backupDir)
default:
fmt.Fprintf(out, " database: newest %s, %s ago; %d bundles in %s\n",
bundles[0].Name, dbbackup.Age(env.now.Sub(bundles[0].Created)), len(bundles), env.w.backupDir)
}
if !env.cfg.Offsite.Enabled() {
fmt.Fprintln(out, " off-site: not configured, every backup is on this machine only")
return
}
switch st, err := offsite.ReadStatus(env.w.offsiteStatus); {
case err != nil:
fmt.Fprintf(out, " off-site: %v\n", err)
case st == nil:
fmt.Fprintln(out, " off-site: never synced")
case st.LastSuccess.IsZero():
fmt.Fprintf(out, " off-site: never succeeded; last attempt %s ago: %s\n", dbbackup.Age(env.now.Sub(st.LastAttempt)), st.LastError)
default:
line := fmt.Sprintf(" off-site: last good sync %s ago to %s", dbbackup.Age(env.now.Sub(st.LastSuccess)), st.Bucket)
if st.LastAttempt.After(st.LastSuccess) { // a failed run records no success
line += fmt.Sprintf("; the last attempt, %s ago, failed: %s", dbbackup.Age(env.now.Sub(st.LastAttempt)), st.LastError)
}
fmt.Fprintln(out, line)
}
}
func statusHost(env statusEnv, out io.Writer) {
fmt.Fprintln(out, "\nhost:")
seen := map[uint64]bool{}
for _, p := range splitList(env.w.diskPaths) {
var st syscall.Stat_t
if err := syscall.Stat(p, &st); err != nil {
continue
}
dev := uint64(st.Dev) // int32 on darwin
if seen[dev] {
continue
}
seen[dev] = true
var fs syscall.Statfs_t
if err := syscall.Statfs(p, &fs); err != nil || fs.Blocks == 0 {
continue
}
bsize := uint64(fs.Bsize) // uint32 on darwin
total, avail := uint64(fs.Blocks)*bsize, uint64(fs.Bavail)*bsize
fmt.Fprintf(out, " disk %s: %s free of %s (%.0f%% free)\n", p, offsite.HumanBytes(int64(avail)), offsite.HumanBytes(int64(total)), float64(fs.Bavail)/float64(fs.Blocks)*100)
}
if total, avail, ok := readMeminfo(env.meminfo); ok {
fmt.Fprintf(out, " memory: %s available of %s\n", offsite.HumanBytes(int64(avail)), offsite.HumanBytes(int64(total)))
}
}
// readMeminfo reads MemTotal and MemAvailable, in bytes, from a /proc/meminfo
// style file.
func readMeminfo(path string) (total, avail uint64, ok bool) {
f, err := os.Open(path)
if err != nil {
return 0, 0, false
}
defer f.Close()
sc := bufio.NewScanner(f)
for sc.Scan() {
fields := strings.Fields(sc.Text())
if len(fields) < 2 {
continue
}
v, _ := strconv.ParseUint(fields[1], 10, 64) // the kernel writes numbers
switch fields[0] {
case "MemTotal:":
total = v * 1024
case "MemAvailable:":
avail = v * 1024
}
}
return total, avail, total > 0
}
func statusWatchdog(ctx context.Context, env statusEnv, out io.Writer) {
fmt.Fprintln(out, "\nwatchdog:")
if _, err := os.Stat(filepath.Join(env.unitDir, "felis-watchdog.timer")); err != nil {
fmt.Fprintln(out, " not installed: nothing checks this host")
return
}
line := " timer " + unitActiveState(ctx, doctorEnv{run: env.run}, "felis-watchdog.timer")
show, _ := env.run(ctx, "systemctl", "show", "--timestamp=unix", "-p", "Result", "-p", "ExecMainExitTimestamp", "felis-watchdog.service")
props := map[string]string{}
for _, ln := range strings.Split(string(show), "\n") {
if k, v, ok := strings.Cut(strings.TrimSpace(ln), "="); ok {
props[k] = v
}
}
// ExecMainExitTimestamp is empty until the service has run.
if sec, _ := strconv.ParseInt(strings.TrimPrefix(props["ExecMainExitTimestamp"], "@"), 10, 64); sec > 0 {
line += fmt.Sprintf(", last run %s ago (%s)", dbbackup.Age(env.now.Sub(time.Unix(sec, 0))), orDash(props["Result"]))
}
fmt.Fprintln(out, line)
state, err := watchdog.LoadState(watchdog.NewestState(env.w.statePath, env.w.fallbackState))
if err != nil {
fmt.Fprintf(out, " alerts: unknown (%v)\n", err)
return
}
var open []string
for key, a := range state.Alerts {
if !a.ClearedAt.IsZero() {
continue
}
if a.Notified.IsZero() {
open = append(open, fmt.Sprintf("%s (%s, seen %s ago, not mailed yet)", key, a.Severity, dbbackup.Age(env.now.Sub(a.FirstSeen))))
} else {
open = append(open, fmt.Sprintf("%s (%s, mailed %s ago)", key, a.Severity, dbbackup.Age(env.now.Sub(a.Notified))))
}
}
if len(open) == 0 {
fmt.Fprintln(out, " alerts: none open")
return
}
sort.Strings(open)
fmt.Fprintf(out, " alerts: %d open (sudo felis doctor says where to look)\n", len(open))
for _, o := range open {
fmt.Fprintf(out, " %s\n", o)
}
}
-490
View File
@@ -1,490 +0,0 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"net"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/watchdog"
appsv1 "k8s.io/api/apps/v1"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/meta"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
)
// statusCluster is a one-node install: felis-api with a restarted pod, three
// servers (one of them the lobby), and a pod of another app whose restarts
// are not felis-api's.
func statusClusterObjects() []client.Object {
replicas := int32(1)
apiLabels := map[string]string{"app": "felis-api"}
return []client.Object{
&corev1.Node{
ObjectMeta: metav1.ObjectMeta{Name: "felis-1"},
Status: corev1.NodeStatus{
Conditions: []corev1.NodeCondition{{Type: corev1.NodeReady, Status: corev1.ConditionTrue}},
NodeInfo: corev1.NodeSystemInfo{KubeletVersion: "v1.36.4+k3s1", OSImage: "CentOS Stream 9", KernelVersion: "5.14.0-630.el9.aarch64"},
},
},
&appsv1.Deployment{
ObjectMeta: metav1.ObjectMeta{Name: "felis-api", Namespace: "felis"},
Spec: appsv1.DeploymentSpec{
Replicas: &replicas,
Selector: &metav1.LabelSelector{MatchLabels: apiLabels},
Template: corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{{
Name: "api", Image: "registry.felis.svc:5000/felis/felis-api:v1.4.0@sha256:0123456789abcdef0123456789abcdef",
}}}},
},
Status: appsv1.DeploymentStatus{ReadyReplicas: 1},
},
&corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Name: "felis-api-7d9", Namespace: "felis", Labels: apiLabels},
Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{Name: "api", RestartCount: 2}}},
},
&corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Name: "other-1", Namespace: "felis", Labels: map[string]string{"app": "other"}},
Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{Name: "x", RestartCount: 5}}},
},
&v1alpha1.MinecraftServer{
ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"},
Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredRunning},
Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseRunning, Players: v1alpha1.PlayersStatus{Online: 2, Max: 20}},
},
&v1alpha1.MinecraftServer{
ObjectMeta: metav1.ObjectMeta{Name: "creative", Namespace: "minecraft"},
Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredStopped},
Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStopped},
},
&v1alpha1.MinecraftServer{
ObjectMeta: metav1.ObjectMeta{Name: "lobby", Namespace: "minecraft", Labels: map[string]string{v1alpha1.LabelSystemRole: "lobby"}},
Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredRunning},
Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting},
},
// Another namespace's server is not this install's.
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Name: "elsewhere", Namespace: "other"}},
}
}
// statusTestEnv is a host with the cluster above, a database backup nine hours
// old, an off-site copy last good two hours ago, and one alert open.
func statusTestEnv(t *testing.T) statusEnv {
t.Helper()
now := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
dir := t.TempDir()
var w watchdogFlags
w.controlNS = "felis"
w.backupDir = filepath.Join(dir, "db-backups")
w.offsiteStatus = filepath.Join(dir, "offsite-status.json")
w.statePath = filepath.Join(dir, "state.json")
w.fallbackState = filepath.Join(dir, "fallback.json")
w.diskPaths = "/nonexistent-felis-status-test"
unitDir := filepath.Join(dir, "systemd")
for _, d := range []string{w.backupDir, unitDir} {
if err := os.MkdirAll(d, 0o700); err != nil {
t.Fatal(err)
}
}
writeTestFile(t, filepath.Join(w.backupDir, dbbackup.BundleName(now.Add(-9*time.Hour), dbbackup.LabelDaily)), "x", 0o600)
writeTestFile(t, filepath.Join(w.backupDir, dbbackup.BundleName(now.Add(-33*time.Hour), dbbackup.LabelDaily)), "x", 0o600)
if err := offsite.WriteStatus(w.offsiteStatus, offsite.Status{
LastAttempt: now.Add(-2 * time.Hour), LastSuccess: now.Add(-2 * time.Hour), Bucket: "felis-dr", Endpoint: "https://s3.example.com",
}); err != nil {
t.Fatal(err)
}
state := &watchdog.State{Alerts: map[string]*watchdog.Alert{
"db-backup-servers": {Finding: watchdog.Finding{Key: "db-backup-servers", Severity: watchdog.Warning}, FirstSeen: now.Add(-3 * time.Hour), Notified: now.Add(-90 * time.Minute)},
"proxy": {Finding: watchdog.Finding{Key: "proxy", Severity: watchdog.Critical}, FirstSeen: now.Add(-time.Hour), Notified: now.Add(-time.Hour), ClearedAt: now.Add(-10 * time.Minute)},
"disk//var": {Finding: watchdog.Finding{Key: "disk//var", Severity: watchdog.Critical}, FirstSeen: now.Add(-4 * time.Minute)},
}}
if err := watchdog.SaveState(w.statePath, state); err != nil {
t.Fatal(err)
}
meminfo := filepath.Join(dir, "meminfo")
writeTestFile(t, meminfo, "MemTotal: 8000000 kB\nMemFree: 100000 kB\nMemAvailable: 2000000 kB\n", 0o644)
writeTestFile(t, filepath.Join(unitDir, "felis-watchdog.timer"), "[Unit]\n", 0o644)
cfg := &config.Config{Offsite: config.OffsiteConfig{Endpoint: "https://s3.example.com", Bucket: "felis-dr"}}
return statusEnv{
cfg: cfg, w: w, cl: fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(statusClusterObjects()...).Build(),
backups: func(context.Context) (map[string]time.Time, error) {
return map[string]time.Time{"survival": now.Add(-3 * time.Hour)}, nil
},
run: func(_ context.Context, name string, args ...string) ([]byte, error) {
switch strings.Join(append([]string{name}, args...), " ") {
case "systemctl is-active felis-watchdog.timer":
return []byte("active\n"), nil
case "systemctl show --timestamp=unix -p Result -p ExecMainExitTimestamp felis-watchdog.service":
return []byte("Result=success\nExecMainExitTimestamp=@" + strconv.FormatInt(now.Add(-70*time.Second).Unix(), 10) + "\n"), nil
}
return nil, errors.New("unexpected")
},
unitDir: unitDir, meminfo: meminfo, host: "felis-test", now: now,
}
}
// lineFields is the words of the first line of out that starts with prefix
// once trimmed.
func lineFields(out, prefix string) []string {
for _, l := range strings.Split(out, "\n") {
if strings.HasPrefix(strings.TrimSpace(l), prefix) {
return strings.Fields(l)
}
}
return nil
}
func TestStatusReport(t *testing.T) {
env := statusTestEnv(t)
var out bytes.Buffer
printStatus(context.Background(), env, &out)
got := out.String()
for _, want := range []string{
"on felis-test at 2026-09-27 12:00 UTC\n",
"node: felis-1 Ready, k3s v1.36.4+k3s1, CentOS Stream 9, kernel 5.14.0-630.el9.aarch64\n",
"\ncontrol plane (namespace felis):\n",
"\nproxy: not on this host\n",
"\nservers (namespace minecraft): 3, 1 running, 2 players online\n",
" database: newest " + dbbackup.BundleName(env.now.Add(-9*time.Hour), dbbackup.LabelDaily) + ", 9h0m ago; 2 bundles in " + env.w.backupDir + "\n",
" off-site: last good sync 2h0m ago to felis-dr\n",
" memory: 1.9 GiB available of 7.6 GiB\n",
" timer active, last run 1m ago (success)\n",
" alerts: 2 open (sudo felis doctor says where to look)\n" +
" db-backup-servers (warning, mailed 1h30m ago)\n" +
" disk//var (critical, seen 4m ago, not mailed yet)\n",
} {
if !strings.Contains(got, want) {
t.Errorf("status lacks %q:\n%s", want, got)
}
}
for prefix, want := range map[string]string{
"felis-api": "felis-api 1/1 ready felis-api:v1.4.0@0123456789ab restarts 2",
"NAME": "NAME ROLE DESIRED PHASE PLAYERS NEWEST WORLD BACKUP",
"creative": "creative - Stopped Stopped - none",
"lobby": "lobby lobby Running Starting - none",
"survival": "survival - Running Running 2/20 3h0m ago",
} {
if f := strings.Join(lineFields(got, prefix), " "); f != want {
t.Errorf("row %q = %q, want %q\n%s", prefix, f, want, got)
}
}
if strings.Contains(got, "elsewhere") || strings.Contains(got, "other-1") {
t.Errorf("status shows what is not this install's:\n%s", got)
}
}
// Whatever is down reads as down, and the rest of the report still prints.
func TestStatusWithPartsDown(t *testing.T) {
env := statusTestEnv(t)
env.cl, env.clErr = nil, errors.New("connection refused")
env.cfg = &config.Config{}
var out bytes.Buffer
printStatus(context.Background(), env, &out)
for _, want := range []string{
"cluster: unreachable (connection refused)\n",
"\nservers (namespace minecraft): unknown while the cluster is unreachable\n",
" off-site: not configured, every backup is on this machine only\n",
" timer active, last run",
} {
if !strings.Contains(out.String(), want) {
t.Errorf("status lacks %q:\n%s", want, out.String())
}
}
env = statusTestEnv(t)
env.backups = func(context.Context) (map[string]time.Time, error) { return nil, errors.New("postgres is down") }
out.Reset()
printStatus(context.Background(), env, &out)
if f := strings.Join(lineFields(out.String(), "survival"), " "); f != "survival - Running Running 2/20 ?" {
t.Errorf("survival with PostgreSQL down = %q", f)
}
if !strings.Contains(out.String(), " world backups unknown: postgres is down\n") {
t.Errorf("status does not say why the backups are unknown:\n%s", out.String())
}
}
func TestShortImage(t *testing.T) {
for in, want := range map[string]string{
"registry.felis.svc:5000/felis/felis-api:v1.4.0": "felis-api:v1.4.0",
"docker.io/library/postgres:17@sha256:0123456789abcdef0123": "postgres:17@0123456789ab",
"felis-operator@sha256:fedcba9876543210fedcba9876543210fedcba9876543210fedcba98765432": "felis-operator@fedcba987654",
"busybox": "busybox",
} {
if got := shortImage(in); got != want {
t.Errorf("shortImage(%q) = %q, want %q", in, got, want)
}
}
}
// A node that is not ready, a Deployment without replicas or containers, and
// lists the API server refuses each read as such.
func TestStatusClusterEdges(t *testing.T) {
env := statusTestEnv(t)
env.cfg = &config.Config{K8s: config.K8sConfig{Namespace: "games"}}
env.cl = fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(
&corev1.Node{ObjectMeta: metav1.ObjectMeta{Name: "felis-1"}, Status: corev1.NodeStatus{
Conditions: []corev1.NodeCondition{{Type: corev1.NodeReady, Status: corev1.ConditionFalse}, {Type: corev1.NodeMemoryPressure, Status: corev1.ConditionTrue}}}},
&appsv1.Deployment{ObjectMeta: metav1.ObjectMeta{Name: "felis-bare", Namespace: "felis"}},
&corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "p", Namespace: "felis"}, Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{RestartCount: 4}}}},
).Build()
var out bytes.Buffer
printStatus(context.Background(), env, &out)
got := out.String()
for _, want := range []string{
"node: felis-1 NotReady, k3s , , kernel \n",
"\nservers (namespace games): 0, 0 running, 0 players online\n\nbackups:\n",
} {
if !strings.Contains(got, want) {
t.Errorf("status lacks %q:\n%s", want, got)
}
}
if f := strings.Join(lineFields(got, "felis-bare"), " "); f != "felis-bare 0/1 ready - restarts 0" {
t.Errorf("row felis-bare = %q, want its one replica wanted, no image, and no pod of its own:\n%s", f, got)
}
failing := func(what client.ObjectList) {
env.cl = fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(statusClusterObjects()...).
WithInterceptorFuncs(interceptor.Funcs{List: func(ctx context.Context, c client.WithWatch, list client.ObjectList, opts ...client.ListOption) error {
if fmt.Sprintf("%T", list) == fmt.Sprintf("%T", what) {
return errors.New("forbidden")
}
return c.List(ctx, list, opts...)
}}).Build()
out.Reset()
printStatus(context.Background(), env, &out)
}
for _, tc := range []struct {
list client.ObjectList
want string
}{
{&corev1.NodeList{}, "cluster: unreachable (forbidden)\n\nproxy:"},
{&appsv1.DeploymentList{}, "\ncontrol plane (namespace felis):\n cannot list it: forbidden\n\nproxy:"},
{&corev1.PodList{}, "\ncontrol plane (namespace felis):\n cannot list it: forbidden\n\nproxy:"},
{&v1alpha1.MinecraftServerList{}, "\nservers (namespace games): cannot list them: forbidden\n\nbackups:"},
} {
failing(tc.list)
if !strings.Contains(out.String(), tc.want) {
t.Errorf("listing %T refused: status lacks %q:\n%s", tc.list, tc.want, out.String())
}
}
}
func TestStatusProxy(t *testing.T) {
env := statusTestEnv(t)
writeTestFile(t, filepath.Join(env.unitDir, "felis-velocity.service"), "[Unit]\n", 0o644)
env.run = fakeSystemctl("", map[string]string{"felis-velocity.service": "active"})
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
env.w.proxyAddr = ln.Addr().String()
var out bytes.Buffer
statusProxy(context.Background(), env, &out)
if want := "\nproxy: felis-velocity active, " + env.w.proxyAddr + " accepts connections\n"; out.String() != want {
t.Errorf("proxy listening: %q, want %q", out.String(), want)
}
ln.Close()
out.Reset()
statusProxy(context.Background(), env, &out)
if want := "\nproxy: felis-velocity active, " + env.w.proxyAddr + " refuses connections (dial tcp " + env.w.proxyAddr + ": "; !strings.HasPrefix(out.String(), want) {
t.Errorf("proxy gone: %q, want it to start %q", out.String(), want)
}
}
func TestStatusBackups(t *testing.T) {
env := statusTestEnv(t)
status := func() string {
var out bytes.Buffer
statusBackups(env, &out)
return out.String()
}
for _, tc := range []struct {
what, dir, want string
}{
{"no -backup-dir", "", " database: not checked (felis-watchdog.service names no -backup-dir)\n"},
{"an empty directory", t.TempDir(), " database: none in %s\n"},
{"a file where the directory should be", filepath.Join(env.w.backupDir, dbbackup.BundleName(env.now.Add(-9*time.Hour), dbbackup.LabelDaily)), " database: cannot read %s: "},
} {
env.w.backupDir = tc.dir
want := tc.want
if strings.Contains(want, "%s") {
want = fmt.Sprintf(want, tc.dir)
}
if got := status(); !strings.Contains(got, want) {
t.Errorf("%s: %q, want %q", tc.what, got, want)
}
}
for _, tc := range []struct {
what string
status *offsite.Status
raw string
want string
}{
{"never synced", nil, "", " off-site: never synced\n"},
{"never succeeded", &offsite.Status{LastAttempt: env.now.Add(-time.Hour), LastError: "403 Forbidden"}, "", " off-site: never succeeded; last attempt 1h0m ago: 403 Forbidden\n"},
{"the last attempt failed", &offsite.Status{LastAttempt: env.now.Add(-30 * time.Minute), LastSuccess: env.now.Add(-26 * time.Hour), LastError: "timeout", Bucket: "felis-dr"}, "",
" off-site: last good sync 26h0m ago to felis-dr; the last attempt, 30m ago, failed: timeout\n"},
{"an error an earlier attempt left", &offsite.Status{LastAttempt: env.now.Add(-2 * time.Hour), LastSuccess: env.now.Add(-time.Hour), LastError: "timeout", Bucket: "felis-dr"}, "",
" off-site: last good sync 1h0m ago to felis-dr\n"},
{"an unreadable record", nil, "{", " off-site: unexpected end of JSON input\n"},
} {
os.Remove(env.w.offsiteStatus)
switch {
case tc.status != nil:
if err := offsite.WriteStatus(env.w.offsiteStatus, *tc.status); err != nil {
t.Fatal(err)
}
case tc.raw != "":
writeTestFile(t, env.w.offsiteStatus, tc.raw, 0o600)
}
if got := status(); !strings.HasSuffix(got, tc.want) {
t.Errorf("%s: %q, want it to end %q", tc.what, got, tc.want)
}
}
}
func TestStatusHost(t *testing.T) {
env := statusTestEnv(t)
dir := t.TempDir()
env.w.diskPaths = "/nonexistent-felis-status-test," + dir + "," + dir
env.meminfo = filepath.Join(dir, "meminfo")
var out bytes.Buffer
statusHost(env, &out)
lines := strings.Split(strings.TrimSuffix(out.String(), "\n"), "\n")
if len(lines) != 3 || lines[0] != "" || lines[1] != "host:" || !strings.HasPrefix(lines[2], " disk "+dir+": ") || !strings.HasSuffix(lines[2], "% free)") {
t.Errorf("host: %q, want one line for %s (a path on a disk already shown, and one that is not there, print none) and no memory line", lines, dir)
}
writeTestFile(t, env.meminfo, "MemTotal: 4096 kB\ngarbage\nMemAvailable: 1024 kB\n", 0o644)
if total, avail, ok := readMeminfo(env.meminfo); !ok || total != 4096*1024 || avail != 1024*1024 {
t.Errorf("readMeminfo = %d, %d, %v", total, avail, ok)
}
writeTestFile(t, env.meminfo, "MemAvailable: 1024 kB\n", 0o644)
if _, _, ok := readMeminfo(env.meminfo); ok {
t.Error("readMeminfo without MemTotal: ok")
}
}
func TestStatusWatchdog(t *testing.T) {
env := statusTestEnv(t)
status := func() string {
var out bytes.Buffer
statusWatchdog(context.Background(), env, &out)
return out.String()
}
run := env.run
env.run = func(ctx context.Context, name string, args ...string) ([]byte, error) {
if len(args) > 0 && args[0] == "show" {
return []byte("Result=success\nExecMainExitTimestamp=\n"), nil
}
return run(ctx, name, args...)
}
if err := watchdog.SaveState(env.w.statePath, &watchdog.State{Alerts: map[string]*watchdog.Alert{
"proxy": {Finding: watchdog.Finding{Key: "proxy", Severity: watchdog.Critical}, FirstSeen: env.now.Add(-time.Hour), ClearedAt: env.now.Add(-time.Minute)},
}}); err != nil {
t.Fatal(err)
}
if got, want := status(), "\nwatchdog:\n timer active\n alerts: none open\n"; got != want {
t.Errorf("a watchdog that has not run and has nothing open: %q, want %q", got, want)
}
writeTestFile(t, env.w.statePath, "{", 0o600)
if got := status(); !strings.Contains(got, "\n alerts: unknown (") {
t.Errorf("an unreadable state: %q", got)
}
if err := os.Remove(filepath.Join(env.unitDir, "felis-watchdog.timer")); err != nil {
t.Fatal(err)
}
if got, want := status(), "\nwatchdog:\n not installed: nothing checks this host\n"; got != want {
t.Errorf("no watchdog timer: %q, want %q", got, want)
}
}
// Rows print by name in whatever order the API server lists them, a server
// the operator has not reconciled yet reads as dashes, and open alerts print
// by key in whatever order the state's map yields them.
func TestStatusOrder(t *testing.T) {
env := statusTestEnv(t)
objs := append(statusClusterObjects(),
&appsv1.Deployment{ObjectMeta: metav1.ObjectMeta{Name: "felis-operator", Namespace: "felis"}},
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Name: "fresh", Namespace: "minecraft"}})
env.cl = fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(objs...).
WithInterceptorFuncs(interceptor.Funcs{List: func(ctx context.Context, c client.WithWatch, list client.ObjectList, opts ...client.ListOption) error {
// The fake lists by name; the other way round, then.
if err := c.List(ctx, list, opts...); err != nil {
return err
}
items, err := meta.ExtractList(list)
if err != nil {
return err
}
slices.Reverse(items)
return meta.SetList(list, items)
}}).Build()
var out bytes.Buffer
printStatus(context.Background(), env, &out)
var rows []string
for _, l := range strings.Split(out.String(), "\n") {
if f := strings.Fields(l); len(f) > 0 && (strings.HasPrefix(f[0], "felis-") || slices.Contains([]string{"creative", "fresh", "lobby", "survival"}, f[0])) {
rows = append(rows, strings.Join(f, " "))
}
}
want := []string{
"felis-api 1/1 ready felis-api:v1.4.0@0123456789ab restarts 2",
"felis-operator 0/1 ready - restarts 0",
"creative - Stopped Stopped - none",
"fresh - - - - none",
"lobby lobby Running Starting - none",
"survival - Running Running 2/20 3h0m ago",
}
if !slices.Equal(rows, want) {
t.Errorf("rows %q, want %q:\n%s", rows, want, out.String())
}
alerts := map[string]*watchdog.Alert{}
for i, key := range []string{"a", "b", "c", "d"} {
alerts[key] = &watchdog.Alert{Finding: watchdog.Finding{Key: key, Severity: watchdog.Warning}, FirstSeen: env.now.Add(-time.Duration(i+1) * time.Minute)}
}
if err := watchdog.SaveState(env.w.statePath, &watchdog.State{Alerts: alerts}); err != nil {
t.Fatal(err)
}
wantAlerts := " alerts: 4 open (sudo felis doctor says where to look)\n" +
" a (warning, seen 1m ago, not mailed yet)\n b (warning, seen 2m ago, not mailed yet)\n" +
" c (warning, seen 3m ago, not mailed yet)\n d (warning, seen 4m ago, not mailed yet)\n"
// A map starts its walk at random: fifty walks all in order by chance
// is out of the question.
for range 50 {
out.Reset()
statusWatchdog(context.Background(), env, &out)
if !strings.HasSuffix(out.String(), wantAlerts) {
t.Fatalf("alerts %q, want them to end %q", out.String(), wantAlerts)
}
}
}
// A selector the API server would have refused picks no pods.
func TestPodRestartsBadSelector(t *testing.T) {
pods := []corev1.Pod{{Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{RestartCount: 3}}}}}
if n := podRestarts(&metav1.LabelSelector{MatchExpressions: []metav1.LabelSelectorRequirement{{Key: "app", Operator: "Near"}}}, pods); n != 0 {
t.Errorf("podRestarts with a bad selector = %d, want 0", n)
}
if n := podRestarts(&metav1.LabelSelector{}, pods); n != 3 {
t.Errorf("podRestarts with a selector that picks all = %d, want 3", n)
}
}
-862
View File
@@ -1,862 +0,0 @@
package main
import (
"archive/tar"
"bytes"
"compress/gzip"
"context"
"errors"
"flag"
"fmt"
"io"
"io/fs"
"net/url"
"os"
"path"
"path/filepath"
"regexp"
"sort"
"strconv"
"strings"
"text/tabwriter"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/watchdog"
appsv1 "k8s.io/api/apps/v1"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
networkingv1 "k8s.io/api/networking/v1"
"k8s.io/apimachinery/pkg/api/meta"
"k8s.io/client-go/kubernetes"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/yaml"
)
// supportBundleDir is where felis support-bundle writes unless told otherwise.
const supportBundleDir = "/var/lib/felis/support"
// redacted stands in for every value the bundle leaves out.
const redacted = "<redacted>"
// minScrubLen is the shortest known secret value the bundle searches for: a
// shorter one would blank ordinary words, and every secret the installer
// generates is far longer.
const minScrubLen = 8
// hostSecretSources are the files on a Felis host that hold secrets; the
// bundle reads them only to take each value out of what it collects.
var hostSecretSources = secretSources{
envFiles: []string{"/etc/felis/secrets.env", defaultOffsiteEnvFile},
valueFiles: []string{hostSMTPPasswordPath, hostUploadsS3AccessKeyPath, hostUploadsS3SecretKeyPath, defaultHeartbeatFile, "/opt/felis/velocity/forwarding.secret"},
propsFiles: []string{"/opt/felis/velocity/plugins/felis-link/felis-link.properties"},
tokenFiles: []string{"/var/lib/rancher/k3s/server/token", "/var/lib/rancher/k3s/server/agent-token"},
}
// cmdSupportBundle collects what someone helping with this host needs into
// one tar.gz: felis status and felis doctor, the logs of the control plane,
// the builds and the systemd units, the cluster's workloads and events, and a
// summary of the configuration. It never collects a Secret, a ConfigMap, the
// contents of a configuration file, the database or a world, and takes every
// value of the host's secret files out of what it does collect. Game server
// logs, which carry player names, addresses and chat, only with -server-logs.
func cmdSupportBundle(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("support-bundle", flag.ContinueOnError)
fs.SetOutput(stderr)
outDir := fs.String("o", supportBundleDir, "directory to write the bundle to (created mode 0700 when missing)")
logLines := fs.Int64("log-lines", 2000, "lines kept from the end of each pod log and each unit's journal")
since := fs.Duration("since", 48*time.Hour, "how far back each unit's journal is read")
serverLogs := fs.Bool("server-logs", false, "also collect the game servers' own logs, which carry player names, IP addresses and chat")
unitDir := fs.String("systemd-dir", systemdUnitDir, "where the installer's systemd units are")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if os.Geteuid() != 0 {
fmt.Fprintln(stderr, "felis support-bundle: run as root (sudo felis support-bundle): it reads root-only logs and state")
return 1
}
b := hostSupportBundle(*unitDir, *logLines, *since, *serverLogs)
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
defer cancel()
path, err := b.write(ctx, *outDir)
if err != nil {
fmt.Fprintf(stderr, "felis support-bundle: %v\n", err)
return 1
}
size := int64(0)
if st, err := os.Stat(path); err == nil {
size = st.Size()
}
fmt.Fprintf(stdout, "wrote %s (%s, mode 0600)\n", path, humanSize(size))
fmt.Fprintln(stdout, "MANIFEST.txt inside says what it holds and what was taken out. Read it through before you send it anywhere:")
fmt.Fprintln(stdout, "redaction finds this host's known secrets and the common ways a secret is logged, and a secret logged another way stays in.")
if !*serverLogs {
fmt.Fprintln(stdout, "Game server logs are left out; -server-logs adds them (player names, IP addresses, chat).")
}
return 0
}
func humanSize(n int64) string {
switch {
case n >= 1<<20:
return fmt.Sprintf("%.1f MiB", float64(n)/(1<<20))
case n >= 1<<10:
return fmt.Sprintf("%.1f KiB", float64(n)/(1<<10))
}
return fmt.Sprintf("%d B", n)
}
// shortDuration is d as Duration.String writes it, less the zero minutes and
// seconds after whole hours or minutes: 48h, and 90m as 1h30m.
func shortDuration(d time.Duration) string {
s := d.String()
if strings.HasSuffix(s, "m0s") {
s = strings.TrimSuffix(s, "0s")
}
if strings.HasSuffix(s, "h0m") {
s = strings.TrimSuffix(s, "0m")
}
return s
}
// supportBundle is one collection and what it reads the host through.
type supportBundle struct {
host string
now time.Time
unitDir string
// w is how felis-watchdog.service runs the watchdog, wErr why it could
// not be read (w then holds the defaults).
w watchdogFlags
wErr error
cfg *config.Config // nil when cfgErr
cfgErr error
cl client.Client // nil when clErr
clErr error
logs func(ctx context.Context, ns, pod, container string, previous bool) ([]byte, error)
run func(ctx context.Context, name string, args ...string) ([]byte, error)
backups func(ctx context.Context) (map[string]time.Time, error)
doctor func(ctx context.Context, out io.Writer)
secrets secretSources
logLines int64
since time.Duration
serverLogs bool
// Host files read whole, and the directory whose listing is kept.
meminfo, osRelease, procVersion, stateDir string
}
func hostSupportBundle(unitDir string, logLines int64, since time.Duration, serverLogs bool) *supportBundle {
host, _ := os.Hostname()
b := &supportBundle{
host: host, now: time.Now(), unitDir: unitDir, run: hostCommand, secrets: hostSecretSources,
logLines: logLines, since: since, serverLogs: serverLogs,
meminfo: "/proc/meminfo", osRelease: "/etc/os-release", procVersion: "/proc/version", stateDir: "/etc/felis",
}
var found bool
b.w, found, b.wErr = watchdogUnitFlags(filepath.Join(unitDir, "felis-watchdog.service"))
if b.wErr == nil && !found {
b.wErr = fmt.Errorf("%s is not installed; read the watchdog's defaults", filepath.Join(unitDir, "felis-watchdog.service"))
}
if b.cfg, b.cfgErr = config.Load(b.w.cfgPath); b.cfgErr == nil {
cfg := b.cfg
b.backups = func(ctx context.Context) (map[string]time.Time, error) {
return newestWorldBackups(ctx, cfg.Database.URL)
}
} else {
b.cfg = nil
}
if b.cl, b.clErr = buildSystemServerClient(); b.clErr != nil {
b.cl = nil
} else if rc, err := hostRESTConfig(); err != nil {
b.clErr = err
b.cl = nil
} else if cs, err := kubernetes.NewForConfig(rc); err != nil {
b.clErr = err
b.cl = nil
} else {
limit := int64(8 << 20)
b.logs = func(ctx context.Context, ns, pod, container string, previous bool) ([]byte, error) {
return cs.CoreV1().Pods(ns).GetLogs(pod, &corev1.PodLogOptions{
Container: container, Previous: previous, Timestamps: true, TailLines: &logLines, LimitBytes: &limit,
}).DoRaw(ctx)
}
}
b.doctor = func(ctx context.Context, out io.Writer) {
runDoctor(ctx, doctorEnv{unitDir: unitDir, run: hostCommand, now: b.now, host: host}, out)
}
return b
}
// bundleWriter streams scrubbed files into the archive and keeps what went
// wrong while collecting, for MANIFEST.txt.
type bundleWriter struct {
tw *tar.Writer
prefix string
now time.Time
scrub *scrubber
errs []string
}
// logText adds a log, or a report that quotes errors: known secrets and
// anything logged as one come out.
func (bw *bundleWriter) logText(name string, data []byte) error {
return bw.add(name, bw.scrub.text(data))
}
// plain adds a file whose secrets were already taken out by structure (the
// cluster's objects, the configuration summary) or that names none (the
// release, disk use, addresses): known secret values still come out, and
// nothing else is rewritten.
func (bw *bundleWriter) plain(name string, data []byte) error {
return bw.add(name, bw.scrub.values(data))
}
func (bw *bundleWriter) add(name string, data []byte) error {
hdr := &tar.Header{Name: path.Join(bw.prefix, name), Mode: 0o600, Size: int64(len(data)), ModTime: bw.now, Typeflag: tar.TypeReg}
if err := bw.tw.WriteHeader(hdr); err != nil {
return err
}
_, err := bw.tw.Write(data)
return err
}
func (bw *bundleWriter) failed(what string, err error) {
bw.errs = append(bw.errs, string(bw.scrub.text([]byte(fmt.Sprintf("%s: %v", what, err)))))
}
// write collects the bundle into dir and returns its path.
func (b *supportBundle) write(ctx context.Context, dir string) (string, error) {
if _, err := os.Stat(dir); errors.Is(err, fs.ErrNotExist) {
if err := os.MkdirAll(dir, 0o700); err != nil {
return "", err
}
}
stamp := b.now.UTC().Format("20060102T150405Z")
base := fmt.Sprintf("felis-support-%s-%s", safeName(b.host), stamp)
final := filepath.Join(dir, base+".tar.gz")
f, err := os.CreateTemp(dir, "."+base+".*.partial") // mode 0600
if err != nil {
return "", err
}
keep := false
defer func() {
if !keep {
f.Close()
os.Remove(f.Name())
}
}()
gz := gzip.NewWriter(f)
bw := &bundleWriter{tw: tar.NewWriter(gz), prefix: base, now: b.now, scrub: b.scrubber()}
if err := b.collect(ctx, bw); err != nil {
return "", err
}
if err := bw.tw.Close(); err != nil {
return "", err
}
if err := gz.Close(); err != nil {
return "", err
}
if err := f.Sync(); err != nil {
return "", err
}
if err := f.Close(); err != nil {
return "", err
}
if err := os.Rename(f.Name(), final); err != nil {
return "", err
}
keep = true
return final, nil
}
// safeName keeps a host name usable in a file name.
func safeName(s string) string {
s = strings.Map(func(r rune) rune {
if r == '-' || r == '.' || r == '_' || (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z') || (r >= '0' && r <= '9') {
return r
}
return '_'
}, s)
if s == "" {
return "host"
}
return s
}
func (b *supportBundle) collect(ctx context.Context, bw *bundleWriter) error {
var buf bytes.Buffer
cmdVersion(nil, &buf, io.Discard)
for _, p := range []string{b.osRelease, b.procVersion} {
if raw, err := os.ReadFile(p); err == nil {
fmt.Fprintf(&buf, "\n# %s\n%s", p, raw)
}
}
if err := bw.plain("version.txt", buf.Bytes()); err != nil {
return err
}
buf.Reset()
switch {
case b.cfgErr != nil:
fmt.Fprintf(&buf, "felis status: the configuration did not load: %v\n", b.cfgErr)
default:
printStatus(ctx, statusEnv{
cfg: b.cfg, w: b.w, cl: b.cl, clErr: b.clErr, backups: b.backups, run: b.run,
unitDir: b.unitDir, meminfo: b.meminfo, host: b.host, now: b.now,
}, &buf)
}
if err := bw.logText("status.txt", buf.Bytes()); err != nil {
return err
}
buf.Reset()
b.doctor(ctx, &buf)
if err := bw.logText("doctor.txt", buf.Bytes()); err != nil {
return err
}
if b.cfg != nil {
if err := bw.plain("config.txt", configSummary(b.cfg, b.w.cfgPath)); err != nil {
return err
}
}
if err := b.collectHost(ctx, bw); err != nil {
return err
}
if err := b.collectJournal(ctx, bw); err != nil {
return err
}
if b.cl == nil {
bw.failed("cluster", b.clErr)
} else if err := b.collectCluster(ctx, bw); err != nil {
return err
}
return bw.add("MANIFEST.txt", b.manifest(bw))
}
func (b *supportBundle) collectHost(ctx context.Context, bw *bundleWriter) error {
commands := []struct {
name string
argv []string
}{
{"host/systemd-units.txt", []string{"systemctl", "list-units", "--all", "--no-pager", "--plain", "felis-*", "k3s.service"}},
{"host/systemd-timers.txt", []string{"systemctl", "list-timers", "--all", "--no-pager", "felis-*"}},
{"host/df.txt", []string{"df", "-h"}},
{"host/addresses.txt", []string{"ip", "-brief", "address"}},
}
for _, c := range commands {
out, err := b.run(ctx, c.argv[0], c.argv[1:]...)
if err != nil && len(out) == 0 {
bw.failed(strings.Join(c.argv, " "), err)
continue
}
if err := bw.plain(c.name, out); err != nil {
return err
}
}
if raw, err := os.ReadFile(b.meminfo); err == nil {
if err := bw.plain("host/meminfo.txt", raw); err != nil {
return err
}
}
listing, err := dirListing(b.stateDir)
if err != nil {
bw.failed("list "+b.stateDir, err)
return nil
}
return bw.plain("host/etc-felis.txt", listing)
}
// dirListing names every file under dir with its mode, size and time, and
// holds nothing of what is in them.
func dirListing(dir string) ([]byte, error) {
var buf bytes.Buffer
tw := tabwriter.NewWriter(&buf, 0, 0, 2, ' ', 0)
fmt.Fprintf(tw, "# %s: names, modes, sizes and times only; no contents\n", dir)
err := filepath.WalkDir(dir, func(p string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
info, err := d.Info()
if err != nil {
return err
}
fmt.Fprintf(tw, "%s\t%d\t%s\t%s\n", info.Mode(), info.Size(), info.ModTime().UTC().Format(time.RFC3339), p)
return nil
})
tw.Flush()
return buf.Bytes(), err
}
func (b *supportBundle) collectJournal(ctx context.Context, bw *bundleWriter) error {
units, _ := filepath.Glob(filepath.Join(b.unitDir, "felis-*.service"))
if _, err := os.Stat(filepath.Join(b.unitDir, "k3s.service")); err == nil {
units = append(units, filepath.Join(b.unitDir, "k3s.service"))
}
since := "@" + strconv.FormatInt(b.now.Add(-b.since).Unix(), 10)
for _, u := range units {
unit := filepath.Base(u)
out, err := b.run(ctx, "journalctl", "-u", unit, "--since", since, "-n", strconv.FormatInt(b.logLines, 10), "--no-pager", "-o", "short-iso")
if err != nil && len(out) == 0 {
bw.failed("journalctl -u "+unit, err)
continue
}
if err := bw.logText("journal/"+strings.TrimSuffix(unit, ".service")+".log", out); err != nil {
return err
}
}
return nil
}
// bundleNamespaces are the namespaces whose objects and logs the bundle
// collects: the control plane, the builds and the game servers.
func (b *supportBundle) bundleNamespaces() (control, build, minecraft string) {
control, build, minecraft = b.w.controlNS, platform.DefaultBuildNamespace, platform.DefaultMinecraftNamespace
if b.cfg != nil {
if b.cfg.Registry.BuildNamespace != "" {
build = b.cfg.Registry.BuildNamespace
}
if b.cfg.K8s.Namespace != "" {
minecraft = b.cfg.K8s.Namespace
}
}
return control, build, minecraft
}
func (b *supportBundle) collectCluster(ctx context.Context, bw *bundleWriter) error {
control, build, minecraft := b.bundleNamespaces()
dump := func(name string, list client.ObjectList, opts ...client.ListOption) error {
if err := b.cl.List(ctx, list, opts...); err != nil {
bw.failed("list "+name, err)
return nil
}
redactList(list)
out, err := yaml.Marshal(list)
if err != nil {
bw.failed("encode "+name, err)
return nil
}
return bw.plain(name, out)
}
if err := dump("cluster/nodes.yaml", &corev1.NodeList{}); err != nil {
return err
}
if err := dump("cluster/persistentvolumes.yaml", &corev1.PersistentVolumeList{}); err != nil {
return err
}
if err := dump("cluster/minecraftservers.yaml", &v1alpha1.MinecraftServerList{}, client.InNamespace(minecraft)); err != nil {
return err
}
for _, ns := range []string{control, build, minecraft} {
for _, k := range []struct {
name string
list client.ObjectList
}{
{"pods", &corev1.PodList{}},
{"deployments", &appsv1.DeploymentList{}},
{"statefulsets", &appsv1.StatefulSetList{}},
{"jobs", &batchv1.JobList{}},
{"cronjobs", &batchv1.CronJobList{}},
{"services", &corev1.ServiceList{}},
{"persistentvolumeclaims", &corev1.PersistentVolumeClaimList{}},
{"networkpolicies", &networkingv1.NetworkPolicyList{}},
{"events", &corev1.EventList{}},
} {
if err := dump("cluster/"+ns+"/"+k.name+".yaml", k.list, client.InNamespace(ns)); err != nil {
return err
}
}
}
var all corev1.PodList
if err := b.cl.List(ctx, &all); err != nil {
bw.failed("list every pod", err)
} else if err := bw.plain("cluster/pods-all-namespaces.txt", podTable(all.Items, b.now)); err != nil {
return err
}
for _, ns := range []string{control, build, minecraft} {
var pods corev1.PodList
if err := b.cl.List(ctx, &pods, client.InNamespace(ns)); err != nil {
continue // already recorded by the dump above
}
for _, p := range pods.Items {
if err := b.collectPodLogs(ctx, bw, p, ns == minecraft && !b.serverLogs); err != nil {
return err
}
}
}
return nil
}
// collectPodLogs keeps the tail of each container's log, and of its previous
// run when it restarted. initOnly keeps only the init containers: a game
// server's own log carries player names, addresses and chat.
func (b *supportBundle) collectPodLogs(ctx context.Context, bw *bundleWriter, p corev1.Pod, initOnly bool) error {
type ctr struct {
name string
restarts int32
}
var ctrs []ctr
restarts := map[string]int32{}
for _, cs := range append(append([]corev1.ContainerStatus(nil), p.Status.InitContainerStatuses...), p.Status.ContainerStatuses...) {
restarts[cs.Name] = cs.RestartCount
}
for _, c := range p.Spec.InitContainers {
ctrs = append(ctrs, ctr{c.Name, restarts[c.Name]})
}
if !initOnly {
for _, c := range p.Spec.Containers {
ctrs = append(ctrs, ctr{c.Name, restarts[c.Name]})
}
}
for _, c := range ctrs {
for _, previous := range []bool{false, true} {
if previous && c.restarts == 0 {
continue
}
name := fmt.Sprintf("logs/%s/%s/%s.log", p.Namespace, p.Name, c.name)
if previous {
name = fmt.Sprintf("logs/%s/%s/%s.previous.log", p.Namespace, p.Name, c.name)
}
out, err := b.logs(ctx, p.Namespace, p.Name, c.name, previous)
if err != nil {
bw.failed(name, err)
continue
}
if err := bw.logText(name, out); err != nil {
return err
}
}
}
return nil
}
// podTable is every pod on the node, one line each.
func podTable(pods []corev1.Pod, now time.Time) []byte {
sort.Slice(pods, func(i, j int) bool {
if pods[i].Namespace != pods[j].Namespace {
return pods[i].Namespace < pods[j].Namespace
}
return pods[i].Name < pods[j].Name
})
var buf bytes.Buffer
tw := tabwriter.NewWriter(&buf, 0, 0, 2, ' ', 0)
fmt.Fprintln(tw, "NAMESPACE\tNAME\tPHASE\tREADY\tRESTARTS\tAGE")
for _, p := range pods {
var ready, restarts int32
for _, cs := range p.Status.ContainerStatuses {
if cs.Ready {
ready++
}
restarts += cs.RestartCount
}
age := "-"
if !p.CreationTimestamp.IsZero() {
age = now.Sub(p.CreationTimestamp.Time).Round(time.Minute).String()
}
fmt.Fprintf(tw, "%s\t%s\t%s\t%d/%d\t%d\t%s\n", p.Namespace, p.Name, p.Status.Phase, ready, len(p.Spec.Containers), restarts, age)
}
tw.Flush()
return buf.Bytes()
}
// redactList takes out of every object what the bundle must not carry: the
// literal env values of pod specs and of MinecraftServers (valueFrom
// references stay, naming the Secret without its contents), the
// last-applied-configuration annotation that repeats them, and managedFields.
func redactList(list client.ObjectList) {
items, err := meta.ExtractList(list)
if err != nil {
return
}
for _, it := range items {
if acc, err := meta.Accessor(it); err == nil {
acc.SetManagedFields(nil)
if ann := acc.GetAnnotations(); ann != nil {
delete(ann, corev1.LastAppliedConfigAnnotation)
acc.SetAnnotations(ann)
}
}
switch o := it.(type) {
case *corev1.Pod:
redactPodSpec(&o.Spec)
case *appsv1.Deployment:
redactPodSpec(&o.Spec.Template.Spec)
case *appsv1.StatefulSet:
redactPodSpec(&o.Spec.Template.Spec)
case *batchv1.Job:
redactPodSpec(&o.Spec.Template.Spec)
case *batchv1.CronJob:
redactPodSpec(&o.Spec.JobTemplate.Spec.Template.Spec)
case *v1alpha1.MinecraftServer:
for i := range o.Spec.Env {
if o.Spec.Env[i].Value != "" {
o.Spec.Env[i].Value = redacted
}
}
}
}
}
func redactPodSpec(s *corev1.PodSpec) {
blank := func(cs []corev1.Container) {
for i := range cs {
for j := range cs[i].Env {
if cs[i].Env[j].Value != "" {
cs[i].Env[j].Value = redacted
}
}
}
}
blank(s.InitContainers)
blank(s.Containers)
for i := range s.EphemeralContainers {
for j := range s.EphemeralContainers[i].Env {
if s.EphemeralContainers[i].Env[j].Value != "" {
s.EphemeralContainers[i].Env[j].Value = redacted
}
}
}
}
// configSummary is felis.toml without a single credential: the hostnames,
// namespaces and which features are on. Fields are picked one by one, so a
// field added later stays out until someone decides it is safe.
func configSummary(c *config.Config, path string) []byte {
var buf bytes.Buffer
line := func(k string, v any) { fmt.Fprintf(&buf, "%-34s %v\n", k, v) }
fmt.Fprintf(&buf, "# a summary of %s; no password, key or token is in it\n", path)
line("server.root_domain", c.Server.RootDomain)
line("server.listen", c.Server.Listen)
db := "(unparsable)"
if u, err := url.Parse(c.Database.URL); err == nil {
db = u.Scheme + "://" + u.User.Username() + "@" + u.Host + u.Path
}
line("database.url (no password)", db)
line("database.deployment", c.Database.Deployment)
line("velocity.public_ip", c.Velocity.PublicIP)
line("velocity.game_port", c.Velocity.GamePort)
line("velocity.login_image", c.Velocity.LoginImage)
line("velocity.lobby_image", c.Velocity.LobbyImage)
line("auth.panel_hostname", c.Auth.PanelHostname)
line("auth.admin_hostname", c.Auth.AdminHostname)
line("auth.client_ip_header", c.Auth.ClientIPHeader)
line("k8s.namespace", c.K8s.Namespace)
line("k8s.egress_mode", c.K8s.EgressMode)
line("k8s.metallb_pool", c.K8s.MetalLBPool)
line("registry.url", c.Registry.URL)
line("registry.build_namespace", c.Registry.BuildNamespace)
line("registry.trivy_db_repository", c.Registry.TrivyDBRepository)
line("registry.build_user_namespaces", c.Registry.BuildUserNamespaces)
line("registry.build_runtime_class", c.Registry.BuildRuntimeClass)
line("registry.max_concurrent_builds", c.Registry.MaxConcurrentBuilds)
line("registry.user_uploads_context", c.Registry.UserUploadsContext)
line("archive.store", c.Archive.Store)
line("archive.local_path", c.Archive.LocalPath)
line("archive.retention", c.Archive.Retention)
line("archive.scheduled_every", c.Archive.ScheduledEvery)
line("archive.scheduled_keep", c.Archive.ScheduledKeep)
line("offsite (configured)", c.Offsite.Enabled())
if c.Offsite.Enabled() {
line("offsite.endpoint", c.Offsite.Endpoint)
line("offsite.bucket", c.Offsite.Bucket)
line("offsite.prefix", c.Offsite.Prefix)
}
line("smtp.host", c.SMTP.Host)
if c.SMTP.Host != "" {
line("smtp.port", c.SMTP.Port)
line("smtp.require_tls", c.SMTP.TLSRequired())
line("smtp.max_per_hour", c.SMTP.MaxPerHour)
}
for i, s := range c.AuthSources {
line(fmt.Sprintf("auth_source[%d]", i), s.Tag+" "+s.Prefix+" "+s.URL)
}
return buf.Bytes()
}
func (b *supportBundle) manifest(bw *bundleWriter) []byte {
control, build, minecraft := b.bundleNamespaces()
var buf bytes.Buffer
fmt.Fprintf(&buf, "Felis support bundle\nhost %s, collected %s, felis %s\n\n", b.host, b.now.UTC().Format(time.RFC3339), resolvedVersion())
fmt.Fprintf(&buf, `What it holds:
status.txt, doctor.txt felis status and felis doctor at collection time
version.txt the release, the OS and the kernel
config.txt a summary of felis.toml: hostnames, namespaces, which features are on
host/ systemd units and timers, disk use, memory, addresses, and the names,
modes, sizes and times of the files under %s (not what is in them)
journal/ up to %d lines per Felis unit and k3s, from the last %s
logs/ up to %d lines of each container of the pods in %s and %s, and of the
init containers of the game server pods in %s; the previous run too
where a container restarted
cluster/ nodes, volumes, the MinecraftServers, and the pods, workloads, services,
volume claims, network policies and events of %s, %s and %s
`, b.stateDir, b.logLines, shortDuration(b.since), b.logLines, control, build, minecraft, control, build, minecraft)
if b.serverLogs {
fmt.Fprintln(&buf, "\nThe game servers' own logs are in logs/ (-server-logs): they carry player names, IP addresses and chat.")
} else {
fmt.Fprintln(&buf, "\nThe game servers' own logs are left out (they carry player names, IP addresses and chat; -server-logs adds them).")
}
fmt.Fprint(&buf, `
Never collected: Kubernetes Secrets and ConfigMaps, what is in /etc/felis or any configuration
file, the database, worlds, uploads.
Taken out:
`)
if len(bw.scrub.sources) > 0 {
fmt.Fprintf(&buf, " - %d secret values, wherever they appear, read from:\n", len(bw.scrub.vals))
for _, s := range bw.scrub.sources {
fmt.Fprintf(&buf, " %s\n", s)
}
} else {
fmt.Fprintln(&buf, " - no secret file was found on this host to take values from")
}
fmt.Fprint(&buf, ` - passwords in URLs, private keys, Bearer and Basic credentials
- in logs and command output, whatever follows password=, secret=, token=, api_key=,
access_key=, private_key= or credentials= (and the same with a colon)
- every literal env value in pod specs and MinecraftServers (valueFrom references stay)
Read it through before you send it anywhere: redaction finds this host's known secrets and the
common ways a secret is logged, and a secret logged another way stays in.
`)
if b.wErr != nil {
fmt.Fprintf(&buf, "\nThe watchdog's settings: %v\n", b.wErr)
}
if len(bw.errs) > 0 {
fmt.Fprintln(&buf, "\nNot collected:")
for _, e := range bw.errs {
fmt.Fprintf(&buf, " - %s\n", e)
}
}
// Its own words name what is redacted, and would be redacted themselves;
// what it quotes was scrubbed as it was recorded.
return buf.Bytes()
}
// secretSources are files that hold secrets, by how each is laid out.
type secretSources struct {
envFiles []string // KEY=VALUE lines, every value a secret (a commented-out one too)
valueFiles []string // one secret, the whole file
propsFiles []string // key=value lines; the keys naming a token, secret, password or key hold one
tokenFiles []string // k3s join tokens: the whole token and its secret part after the last ':'
}
// scrubber takes secrets out of what the bundle collects.
type scrubber struct {
vals []string // longest first, so a secret that contains another goes whole
sources []string // the files vals came from
}
var (
pemPrivateKey = regexp.MustCompile(`(?s)-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----.*?-----END [A-Z0-9 ]*PRIVATE KEY-----`)
urlUserinfo = regexp.MustCompile(`([A-Za-z][A-Za-z0-9+.-]*://[^/\s:@]*:)[^/\s@]+@`)
authScheme = regexp.MustCompile(`(?i)\b(bearer|basic)\s+[A-Za-z0-9._~+/=-]{8,}`)
secretAssign = regexp.MustCompile(`(?i)((?:password|passwd|secret|token|api[_-]?key|access[_-]?key|private[_-]?key|credentials?)[A-Za-z0-9_.-]*"?[ \t]*[=:][ \t]*"?)([^\s"',;&]{4,})`)
secretPropKey = regexp.MustCompile(`(?i)token|secret|password|key`)
)
func (b *supportBundle) scrubber() *scrubber {
s := &scrubber{}
s.readSources(b.secrets)
if b.cfg != nil {
if u, err := url.Parse(b.cfg.Database.URL); err == nil {
if pw, ok := u.User.Password(); ok {
s.add(pw)
}
}
for _, ref := range []string{
b.cfg.Velocity.ServiceTokenRef, b.cfg.SMTP.PasswordRef,
b.cfg.Offsite.AccessKeyRef, b.cfg.Offsite.SecretKeyRef, b.cfg.Offsite.KeyRef,
b.cfg.Registry.S3.AccessKeyRef, b.cfg.Registry.S3.SecretKeyRef,
b.cfg.Archive.S3.AccessKeyRef, b.cfg.Archive.S3.SecretKeyRef,
} {
if ref != "" {
s.add(os.Getenv(ref))
}
}
}
if st, err := watchdog.LoadState(watchdog.NewestState(b.w.statePath, b.w.fallbackState)); err == nil {
s.add(st.SMTPPassword)
}
return s
}
func (s *scrubber) add(v string) bool {
v = strings.TrimSpace(v)
if len(v) < minScrubLen {
return false
}
for _, have := range s.vals {
if have == v {
return true
}
}
s.vals = append(s.vals, v)
sort.SliceStable(s.vals, func(i, j int) bool { return len(s.vals[i]) > len(s.vals[j]) })
return true
}
func (s *scrubber) readSources(src secretSources) {
read := func(p string, take func(content string) bool) {
raw, err := os.ReadFile(p)
if err != nil {
return
}
if take(string(raw)) {
s.sources = append(s.sources, p)
}
}
keyValues := func(content string, keep func(key string) bool) bool {
found := false
for _, line := range strings.Split(content, "\n") {
k, v, ok := strings.Cut(line, "=")
if !ok || !keep(strings.TrimSpace(k)) {
continue
}
v = strings.TrimSpace(v)
if len(v) >= 2 && (v[0] == '\'' || v[0] == '"') && v[len(v)-1] == v[0] {
v = v[1 : len(v)-1]
}
found = s.add(v) || found
}
return found
}
for _, p := range src.envFiles {
read(p, func(c string) bool { return keyValues(c, func(string) bool { return true }) })
}
for _, p := range src.propsFiles {
read(p, func(c string) bool { return keyValues(c, secretPropKey.MatchString) })
}
for _, p := range src.valueFiles {
read(p, func(c string) bool { return s.add(c) })
}
for _, p := range src.tokenFiles {
read(p, func(c string) bool {
c = strings.TrimSpace(c)
whole := s.add(c)
part := false
if i := strings.LastIndex(c, ":"); i >= 0 {
part = s.add(c[i+1:])
}
return whole || part
})
}
}
// values takes every known secret value out of data.
func (s *scrubber) values(data []byte) []byte {
t := pemPrivateKey.ReplaceAllString(string(data), "<redacted private key>")
for _, v := range s.vals {
t = strings.ReplaceAll(t, v, redacted)
}
t = urlUserinfo.ReplaceAllString(t, "${1}"+redacted+"@")
t = authScheme.ReplaceAllString(t, "${1} "+redacted)
return []byte(t)
}
// text is values, and whatever a log names as a secret as well.
func (s *scrubber) text(data []byte) []byte {
return secretAssign.ReplaceAll(s.values(data), []byte("${1}"+redacted))
}
-433
View File
@@ -1,433 +0,0 @@
package main
import (
"archive/tar"
"compress/gzip"
"context"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/watchdog"
appsv1 "k8s.io/api/apps/v1"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
// The secrets a bundle host holds, each of which must be nowhere in the bundle.
var plantedSecrets = map[string]string{
"secrets.env SERVICE_TOKEN": "svc-token-planted-0a1b2c3d",
"secrets.env DB_PASSWORD": "db-pass-planted-4e5f6a7b",
"offsite.env FELIS_OFFSITE_KEY": "offsite-key-planted+8c9d/0e1f=",
"smtp-password": "smtp-pass-planted-2a3b",
"felis-link.properties token": "props-token-planted-4c5d",
"k3s token secret part": "k3s-secret-part-planted-6e7f",
"watchdog state relay password": "cached-relay-pw-planted-8a9b",
"pod env literal": "env-literal-planted-0c1d",
"deployment env literal": "deploy-env-planted-2e3f",
"MinecraftServer env literal": "cr-env-planted-4a5b",
"last-applied annotation": "annotation-planted-6c7d",
"logged password": "hunter2-planted-8e9f",
"logged bearer": "bearerplanted0a1b2c3d",
"URL password": "urlpass-planted-4e5f",
"token in an error": "errtoken-planted-0f1e",
"felis.host.toml DB password": "cfg-db-pass-planted-1c2d",
"service token by its env ref": "env-ref-token-planted-3e4f",
"init container env literal": "init-env-planted-5a6b",
"ephemeral container env": "ephemeral-env-planted-7c8d",
"statefulset env literal": "sts-env-planted-9e0f",
"job env literal": "job-env-planted-1a2b",
"cronjob env literal": "cronjob-env-planted-3c4d",
"commented-out offsite key": "old-offsite-key-planted-5e6f",
"doctor quoting a password": "doctor-pw-planted-7a8b",
"status quoting a password": "status-pw-planted-9c0d",
"journal quoting a password": "journal-pw-planted-1e2f",
}
// bundleHost is a Felis host for felis support-bundle: its secret files, its
// watchdog unit and state, a cluster with a control-plane pod that restarted
// and a game server, and logs and a journal that name secrets.
func bundleHost(t *testing.T) *supportBundle {
t.Helper()
p := plantedSecrets
dir := t.TempDir()
etc := filepath.Join(dir, "etc-felis")
for _, d := range []string{etc, filepath.Join(dir, "systemd"), filepath.Join(dir, "k3s")} {
if err := os.MkdirAll(d, 0o700); err != nil {
t.Fatal(err)
}
}
writeTestFile(t, filepath.Join(etc, "secrets.env"), "SERVICE_TOKEN="+p["secrets.env SERVICE_TOKEN"]+"\nDB_PASSWORD="+p["secrets.env DB_PASSWORD"]+"\nSHORT=abc\n", 0o600)
writeTestFile(t, filepath.Join(etc, "offsite.env"), "# before the rotation\n# FELIS_OFFSITE_KEY="+p["commented-out offsite key"]+"\nFELIS_OFFSITE_KEY='"+p["offsite.env FELIS_OFFSITE_KEY"]+"'\n", 0o600)
writeTestFile(t, filepath.Join(etc, "smtp-password"), p["smtp-password"]+"\n", 0o600)
writeTestFile(t, filepath.Join(etc, "felis-link.properties"), "api-base-url=http://10.43.0.10:8081\nservice-token="+p["felis-link.properties token"]+"\nroot-domain=games.example.org\n", 0o640)
writeTestFile(t, filepath.Join(dir, "k3s", "token"), "K10deadbeefcafe::server:"+p["k3s token secret part"]+"\n", 0o600)
cfgPath := filepath.Join(etc, "felis.host.toml")
writeTestFile(t, cfgPath, "[database]\nurl = \"postgres://felis:"+p["felis.host.toml DB password"]+"@127.0.0.1:1/felis?sslmode=disable&connect_timeout=1\"\n"+
"[server]\nroot_domain = \"games.example.org\"\n[archive]\nstore = \"tarLocal\"\n[k8s]\negress_mode = \"nodeport\"\n"+
"[velocity]\nservice_token_ref = \"FELIS_BUNDLE_TEST_SERVICE_TOKEN\"\n", 0o600)
t.Setenv("FELIS_BUNDLE_TEST_SERVICE_TOKEN", p["service token by its env ref"])
statePath := filepath.Join(dir, "state.json")
if err := watchdog.SaveState(statePath, &watchdog.State{SMTPPassword: p["watchdog state relay password"]}); err != nil {
t.Fatal(err)
}
unitDir := filepath.Join(dir, "systemd")
writeTestFile(t, filepath.Join(unitDir, "felis-watchdog.service"), "[Service]\nExecStart=/usr/local/bin/felis watchdog -config "+cfgPath+" -state "+statePath+"\n", 0o644)
writeTestFile(t, filepath.Join(unitDir, "felis-offsite.service"), "[Unit]\n", 0o644)
writeTestFile(t, filepath.Join(unitDir, "k3s.service"), "[Unit]\n", 0o644)
meminfo := filepath.Join(dir, "meminfo")
writeTestFile(t, meminfo, "MemTotal: 8000000 kB\nMemAvailable: 2000000 kB\n", 0o644)
cfg, err := config.Load(cfgPath)
if err != nil {
t.Fatal(err)
}
var w watchdogFlags
w, _, err = watchdogUnitFlags(filepath.Join(unitDir, "felis-watchdog.service"))
if err != nil {
t.Fatal(err)
}
w.diskPaths = "/nonexistent-felis-bundle-test"
secretEnv := corev1.EnvVar{Name: "FELIS_SMTP_PASSWORD", ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{Name: "felis-smtp"}, Key: "password"}}}
replicas := int32(1)
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(
&corev1.Pod{
ObjectMeta: metav1.ObjectMeta{
Name: "felis-api-7d9", Namespace: "felis",
Annotations: map[string]string{corev1.LastAppliedConfigAnnotation: `{"env":"` + p["last-applied annotation"] + `"}`, "felis.lolicon.best/kept": "yes"},
ManagedFields: []metav1.ManagedFieldsEntry{{Manager: "kubectl-client-side-apply"}},
},
Spec: corev1.PodSpec{
InitContainers: []corev1.Container{{Name: "wait-db", Image: "busybox", Env: []corev1.EnvVar{{Name: "FELIS_INIT_SETTING", Value: p["init container env literal"]}}}},
Containers: []corev1.Container{{Name: "api", Image: "felis-api:v1", Env: []corev1.EnvVar{
{Name: "FELIS_PLAIN_SETTING", Value: p["pod env literal"]}, secretEnv,
}}},
EphemeralContainers: []corev1.EphemeralContainer{{EphemeralContainerCommon: corev1.EphemeralContainerCommon{
Name: "debug", Env: []corev1.EnvVar{{Name: "FELIS_DEBUG_SETTING", Value: p["ephemeral container env"]}}}}},
},
Status: corev1.PodStatus{Phase: corev1.PodRunning, ContainerStatuses: []corev1.ContainerStatus{{Name: "api", RestartCount: 1, Ready: true}}},
},
&appsv1.Deployment{
ObjectMeta: metav1.ObjectMeta{Name: "felis-api", Namespace: "felis"},
Spec: appsv1.DeploymentSpec{Replicas: &replicas, Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app": "felis-api"}},
Template: corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{{Name: "api", Env: []corev1.EnvVar{
{Name: "FELIS_DEPLOY_SETTING", Value: p["deployment env literal"]},
}}}}}},
},
&corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Name: "survival-0", Namespace: "minecraft"},
Spec: corev1.PodSpec{
InitContainers: []corev1.Container{{Name: "prepare-data"}, {Name: "egress-gate"}},
Containers: []corev1.Container{{Name: "server"}},
},
Status: corev1.PodStatus{Phase: corev1.PodRunning},
},
&v1alpha1.MinecraftServer{
ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"},
Spec: v1alpha1.MinecraftServerSpec{Env: []v1alpha1.EnvVar{{Name: "DISCORD_WEBHOOK", Value: p["MinecraftServer env literal"]}}},
},
&appsv1.StatefulSet{
ObjectMeta: metav1.ObjectMeta{Name: "felis-postgres", Namespace: "felis"},
Spec: appsv1.StatefulSetSpec{Template: envTemplate("FELIS_STS_SETTING", p["statefulset env literal"])},
},
&batchv1.Job{
ObjectMeta: metav1.ObjectMeta{Name: "build-1", Namespace: "felis-build"},
Spec: batchv1.JobSpec{Template: envTemplate("FELIS_JOB_SETTING", p["job env literal"])},
},
&batchv1.CronJob{
ObjectMeta: metav1.ObjectMeta{Name: "felis-reaper", Namespace: "felis"},
Spec: batchv1.CronJobSpec{JobTemplate: batchv1.JobTemplateSpec{Spec: batchv1.JobSpec{Template: envTemplate("FELIS_CRON_SETTING", p["cronjob env literal"])}}},
},
&corev1.Node{ObjectMeta: metav1.ObjectMeta{Name: "felis-1"}},
).Build()
secretLog := fmt.Sprintf("started with SERVICE_TOKEN=%s\nlogin password=%s ok\nAuthorization: Bearer %s\ndial postgres://felis:%s@db:5432/felis\n",
p["secrets.env SERVICE_TOKEN"], p["logged password"], p["logged bearer"], p["URL password"])
return &supportBundle{
host: "felis-test", now: time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC), unitDir: unitDir,
w: w, cfg: cfg, cl: cl,
logs: func(_ context.Context, ns, pod, container string, previous bool) ([]byte, error) {
if container == "wait-db" {
return nil, errors.New("container \"wait-db\" is waiting to start; token=" + p["token in an error"])
}
return []byte(fmt.Sprintf("log of %s/%s/%s previous=%v\n%s", ns, pod, container, previous, secretLog)), nil
},
run: func(_ context.Context, name string, args ...string) ([]byte, error) {
switch name {
case "journalctl":
return []byte("journal of " + args[1] + "\nFELIS_OFFSITE_KEY=" + p["offsite.env FELIS_OFFSITE_KEY"] + "\nrelay " + p["smtp-password"] + " refused\n" +
"relay login with the cached " + p["watchdog state relay password"] + "\ndatabase auth failed for " + p["felis.host.toml DB password"] +
"\nservice token " + p["service token by its env ref"] + " rejected\nnode joined with " + p["k3s token secret part"] + "\n" +
"old copies sealed with " + p["commented-out offsite key"] + "\nretrying with password=" + p["journal quoting a password"] + "\n"), nil
case "systemctl", "df", "ip":
return []byte(name + " output\n"), nil
}
return nil, errors.New("unexpected " + name)
},
backups: func(context.Context) (map[string]time.Time, error) {
return nil, errors.New("connect: password=" + p["status quoting a password"])
},
doctor: func(_ context.Context, out io.Writer) {
fmt.Fprintf(out, "doctor report; the k3s token K10deadbeefcafe::server:%s leaked here\nprobe said password=%s\n", p["k3s token secret part"], p["doctor quoting a password"])
},
secrets: secretSources{
envFiles: []string{filepath.Join(etc, "secrets.env"), filepath.Join(etc, "offsite.env"), filepath.Join(etc, "absent.env")},
valueFiles: []string{filepath.Join(etc, "smtp-password"), filepath.Join(etc, "uploads-s3-secret-key")},
propsFiles: []string{filepath.Join(etc, "felis-link.properties")},
tokenFiles: []string{filepath.Join(dir, "k3s", "token")},
},
logLines: 500, since: 48 * time.Hour,
meminfo: meminfo, osRelease: filepath.Join(dir, "os-release"), procVersion: filepath.Join(dir, "version"), stateDir: etc,
}
}
// envTemplate is a pod template whose one container sets name to a literal.
func envTemplate(name, value string) corev1.PodTemplateSpec {
return corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{{Name: "main", Env: []corev1.EnvVar{{Name: name, Value: value}}}}}}
}
// readBundle is every file in the bundle at path, by its name inside the
// bundle's directory, and each one's mode.
func readBundle(t *testing.T, path string) (map[string]string, map[string]int64) {
t.Helper()
f, err := os.Open(path)
if err != nil {
t.Fatal(err)
}
defer f.Close()
gz, err := gzip.NewReader(f)
if err != nil {
t.Fatal(err)
}
tr := tar.NewReader(gz)
files, modes := map[string]string{}, map[string]int64{}
prefix := strings.TrimSuffix(filepath.Base(path), ".tar.gz") + "/"
for {
hdr, err := tr.Next()
if err == io.EOF {
break
}
if err != nil {
t.Fatal(err)
}
name, ok := strings.CutPrefix(hdr.Name, prefix)
if !ok {
t.Errorf("%s is outside the bundle's directory %s", hdr.Name, prefix)
}
raw, err := io.ReadAll(tr)
if err != nil {
t.Fatal(err)
}
files[name], modes[name] = string(raw), hdr.Mode
}
return files, modes
}
func TestSupportBundle(t *testing.T) {
b := bundleHost(t)
out := filepath.Join(t.TempDir(), "support")
path, err := b.write(context.Background(), out)
if err != nil {
t.Fatal(err)
}
if want := filepath.Join(out, "felis-support-felis-test-20260927T120000Z.tar.gz"); path != want {
t.Errorf("path %s, want %s", path, want)
}
for p, want := range map[string]os.FileMode{out: 0o700 | os.ModeDir, path: 0o600} {
if st, err := os.Stat(p); err != nil || st.Mode() != want {
t.Errorf("%s: mode %v (err %v), want %v", p, st.Mode(), err, want)
}
}
if left, _ := filepath.Glob(filepath.Join(out, ".*partial")); len(left) != 0 {
t.Errorf("left behind %v", left)
}
files, modes := readBundle(t, path)
var names []string
for n := range files {
names = append(names, n)
if modes[n] != 0o600 {
t.Errorf("%s: mode %o in the archive, want 600", n, modes[n])
}
}
sort.Strings(names)
want := []string{
"MANIFEST.txt",
"cluster/felis-build/cronjobs.yaml", "cluster/felis-build/deployments.yaml", "cluster/felis-build/events.yaml", "cluster/felis-build/jobs.yaml",
"cluster/felis-build/networkpolicies.yaml", "cluster/felis-build/persistentvolumeclaims.yaml", "cluster/felis-build/pods.yaml",
"cluster/felis-build/services.yaml", "cluster/felis-build/statefulsets.yaml",
"cluster/felis/cronjobs.yaml", "cluster/felis/deployments.yaml", "cluster/felis/events.yaml", "cluster/felis/jobs.yaml",
"cluster/felis/networkpolicies.yaml", "cluster/felis/persistentvolumeclaims.yaml", "cluster/felis/pods.yaml",
"cluster/felis/services.yaml", "cluster/felis/statefulsets.yaml",
"cluster/minecraft/cronjobs.yaml", "cluster/minecraft/deployments.yaml", "cluster/minecraft/events.yaml", "cluster/minecraft/jobs.yaml",
"cluster/minecraft/networkpolicies.yaml", "cluster/minecraft/persistentvolumeclaims.yaml", "cluster/minecraft/pods.yaml",
"cluster/minecraft/services.yaml", "cluster/minecraft/statefulsets.yaml",
"cluster/minecraftservers.yaml", "cluster/nodes.yaml", "cluster/persistentvolumes.yaml", "cluster/pods-all-namespaces.txt",
"config.txt", "doctor.txt",
"host/addresses.txt", "host/df.txt", "host/etc-felis.txt", "host/meminfo.txt", "host/systemd-timers.txt", "host/systemd-units.txt",
"journal/felis-offsite.log", "journal/felis-watchdog.log", "journal/k3s.log",
"logs/felis/felis-api-7d9/api.log", "logs/felis/felis-api-7d9/api.previous.log",
"logs/minecraft/survival-0/egress-gate.log", "logs/minecraft/survival-0/prepare-data.log",
"status.txt", "version.txt",
}
if strings.Join(names, "\n") != strings.Join(want, "\n") {
t.Errorf("bundle holds\n %s\nwant\n %s", strings.Join(names, "\n "), strings.Join(want, "\n "))
}
for what, secret := range plantedSecrets {
for n, body := range files {
if strings.Contains(body, secret) {
t.Errorf("%s (%s) is in %s:\n%s", what, secret, n, body)
}
}
}
// What was taken out leaves what a reader needs around it.
for name, wants := range map[string][]string{
"logs/felis/felis-api-7d9/api.previous.log": {"log of felis/felis-api-7d9/api previous=true\n", "SERVICE_TOKEN=<redacted>\n", "login password=<redacted> ok\n", "Authorization: Bearer <redacted>\n", "postgres://felis:<redacted>@db:5432/felis\n"},
"journal/k3s.log": {"journal of k3s.service\n", "FELIS_OFFSITE_KEY=<redacted>\n", "relay <redacted> refused\n",
"relay login with the cached <redacted>\n", "database auth failed for <redacted>\n", "service token <redacted> rejected\n", "node joined with <redacted>\n"},
"cluster/felis/statefulsets.yaml": {"name: FELIS_STS_SETTING\n"},
"cluster/felis/cronjobs.yaml": {"name: FELIS_CRON_SETTING\n"},
"cluster/felis-build/jobs.yaml": {"name: FELIS_JOB_SETTING\n"},
"cluster/felis/pods.yaml": {"name: FELIS_PLAIN_SETTING\n value: <redacted>\n", "name: FELIS_SMTP_PASSWORD\n valueFrom:\n secretKeyRef:\n key: password\n name: felis-smtp\n", "felis.lolicon.best/kept: \"yes\"", "name: FELIS_INIT_SETTING\n", "name: FELIS_DEBUG_SETTING\n"},
"cluster/felis/deployments.yaml": {"name: FELIS_DEPLOY_SETTING\n value: <redacted>\n"},
"cluster/minecraftservers.yaml": {"name: DISCORD_WEBHOOK\n value: <redacted>\n"},
"config.txt": {fmt.Sprintf("%-34s %s\n", "server.root_domain", "games.example.org"), fmt.Sprintf("%-34s %s\n", "database.url (no password)", "postgres://[email protected]:1/felis")},
"doctor.txt": {"the k3s token <redacted> leaked here\nprobe said password=<redacted>\n"},
"status.txt": {" world backups unknown: connect: password=<redacted>\n"},
"host/etc-felis.txt": {"secrets.env\n", "smtp-password\n", "felis-link.properties\n"},
"MANIFEST.txt": {
"The game servers' own logs are left out",
" journal/ up to 500 lines per Felis unit and k3s, from the last 48h\n",
" - 11 secret values, wherever they appear, read from:\n",
"secrets.env\n", "offsite.env\n", "smtp-password\n", "felis-link.properties\n", "token\n",
"logs/felis/felis-api-7d9/wait-db.log: container \"wait-db\" is waiting to start; token=<redacted>\n",
// Its own words about what is redacted come through whole.
" - passwords in URLs, private keys, Bearer and Basic credentials\n",
"access_key=, private_key= or credentials= (and the same with a colon)\n",
"Read it through before you send it anywhere",
},
} {
for _, w := range wants {
if !strings.Contains(files[name], w) {
t.Errorf("%s lacks %q:\n%s", name, w, files[name])
}
}
}
for _, gone := range []string{"managedFields", "last-applied-configuration"} {
if strings.Contains(files["cluster/felis/pods.yaml"], gone) {
t.Errorf("pods.yaml keeps %s:\n%s", gone, files["cluster/felis/pods.yaml"])
}
}
if strings.Contains(files["MANIFEST.txt"], "absent.env") || strings.Contains(files["MANIFEST.txt"], "uploads-s3-secret-key") {
t.Errorf("MANIFEST names files this host does not have:\n%s", files["MANIFEST.txt"])
}
}
// -server-logs adds the game servers' own logs, and says so.
func TestSupportBundleServerLogs(t *testing.T) {
b := bundleHost(t)
b.serverLogs = true
path, err := b.write(context.Background(), t.TempDir())
if err != nil {
t.Fatal(err)
}
files, _ := readBundle(t, path)
if _, ok := files["logs/minecraft/survival-0/server.log"]; !ok {
t.Error("no server.log with -server-logs")
}
if !strings.Contains(files["MANIFEST.txt"], "The game servers' own logs are in logs/ (-server-logs)") {
t.Errorf("MANIFEST does not say server logs are in:\n%s", files["MANIFEST.txt"])
}
}
// With the cluster and the configuration gone the bundle still holds what the
// host shows, and MANIFEST says what it could not collect.
func TestSupportBundleWithTheClusterDown(t *testing.T) {
b := bundleHost(t)
b.cl, b.clErr = nil, errors.New("connection refused")
b.cfg, b.cfgErr = nil, errors.New("felis.host.toml: no such file")
b.wErr = errors.New("felis-watchdog.service is not installed; read the watchdog's defaults")
path, err := b.write(context.Background(), t.TempDir())
if err != nil {
t.Fatal(err)
}
files, _ := readBundle(t, path)
for _, n := range []string{"status.txt", "doctor.txt", "host/etc-felis.txt", "journal/k3s.log"} {
if _, ok := files[n]; !ok {
t.Errorf("no %s", n)
}
}
for n := range files {
if strings.HasPrefix(n, "cluster/") || strings.HasPrefix(n, "logs/") || n == "config.txt" {
t.Errorf("%s with no cluster and no configuration", n)
}
}
if !strings.Contains(files["status.txt"], "felis status: the configuration did not load: felis.host.toml: no such file") {
t.Errorf("status.txt:\n%s", files["status.txt"])
}
if !strings.Contains(files["MANIFEST.txt"], "\nThe watchdog's settings: felis-watchdog.service is not installed; read the watchdog's defaults\n") ||
!strings.Contains(files["MANIFEST.txt"], " - cluster: connection refused\n") {
t.Errorf("MANIFEST.txt:\n%s", files["MANIFEST.txt"])
}
}
func TestShortDuration(t *testing.T) {
for d, want := range map[time.Duration]string{
48 * time.Hour: "48h",
90 * time.Minute: "1h30m",
45 * time.Minute: "45m",
30 * time.Second: "30s",
time.Hour + 30*time.Second: "1h0m30s",
} {
if got := shortDuration(d); got != want {
t.Errorf("shortDuration(%v) = %q, want %q", d, got, want)
}
}
}
func TestScrubber(t *testing.T) {
s := &scrubber{}
for _, v := range []string{"known-secret-value", "known-secret-value-longer", "short", "known-secret-value"} {
s.add(v)
}
if got := strings.Join(s.vals, ","); got != "known-secret-value-longer,known-secret-value" {
t.Errorf("kept %q; want each value once, longest first, none under %d characters", s.vals, minScrubLen)
}
for in, want := range map[string]string{
"a known-secret-value-longer b": "a <redacted> b",
"a known-secret-value b": "a <redacted> b",
"password=abcd1234 next": "password=<redacted> next",
`{"token": "abcd1234"}`: `{"token": "<redacted>"}`,
"SMTP_PASSWORD: s3cr3t!x": "SMTP_PASSWORD: <redacted>",
"x-api-key=zzzz9999&q=1": "x-api-key=<redacted>&q=1",
"Authorization: Basic dXNlcjpwYXNzd29yZA==": "Authorization: Basic <redacted>",
"s3://AKIA:secretpart123@bucket/key": "s3://AKIA:<redacted>@bucket/key",
"https://user@host/path": "https://user@host/path",
"-----BEGIN EC PRIVATE KEY-----\nMHc\n-----END EC PRIVATE KEY-----": "<redacted private key>",
"tokens: 3": "tokens: 3",
"read the relay password (keeping the cached one)": "read the relay password (keeping the cached one)",
`secrets "felis-smtp" not found`: `secrets "felis-smtp" not found`,
} {
if got := string(s.text([]byte(in))); got != want {
t.Errorf("text(%q) = %q, want %q", in, got, want)
}
}
// Structured files keep what only looks like a secret.
if got := string(s.values([]byte("secretName: felis-forwarding-secret and known-secret-value"))); got != "secretName: felis-forwarding-secret and <redacted>" {
t.Errorf("values() = %q", got)
}
}
+1 -11
View File
@@ -15,7 +15,6 @@ import (
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"k8s.io/client-go/rest"
"k8s.io/client-go/tools/clientcmd"
"k8s.io/client-go/util/retry"
ctrl "sigs.k8s.io/controller-runtime"
@@ -244,15 +243,6 @@ func buildSystemServerClient() (client.Client, error) {
if err := v1alpha1.AddToScheme(scheme); err != nil {
return nil, err
}
cfg, err := hostRESTConfig()
if err != nil {
return nil, err
}
return client.New(cfg, client.Options{Scheme: scheme})
}
// hostRESTConfig is the cluster connection buildSystemServerClient describes.
func hostRESTConfig() (*rest.Config, error) {
cfg, err := ctrl.GetConfig()
if err != nil {
cfg, err = clientcmd.BuildConfigFromFlags("", hostBootstrapKubeconfigPath)
@@ -260,7 +250,7 @@ func hostRESTConfig() (*rest.Config, error) {
return nil, fmt.Errorf("no reachable kubeconfig (tried in-cluster/$KUBECONFIG/~/.kube and %s): %w", hostBootstrapKubeconfigPath, err)
}
}
return cfg, nil
return client.New(cfg, client.Options{Scheme: scheme})
}
// systemServerOutcome records what ensureSystemServers did with one service so
-202
View File
@@ -1,202 +0,0 @@
package main
import (
"context"
"errors"
"path/filepath"
"strings"
"testing"
tea "github.com/charmbracelet/bubbletea"
"github.com/charmbracelet/lipgloss"
"felis.lolicon.best/internal/config"
)
// TestAlertRouteLines: the summary's alert rows say where the watchdog's
// alerts go, and flag every route that reaches no one.
func TestAlertRouteLines(t *testing.T) {
cases := []struct {
name string
r alertRoute
alerts string
alertsOK bool
beat string
beatOK bool
}{
{
name: "fresh install",
alerts: "only logged: no email relay (press e)",
beat: "none: no outside check (troubleshooting.md §14)",
},
{
name: "relay, no verified owner",
r: alertRoute{relay: "smtp.example.com:587", heartbeat: "https://hc-ping.com/..."},
alerts: "only logged: no verified Owner email (panel → Account)",
beat: "https://hc-ping.com/... every 2 minutes", beatOK: true,
},
{
name: "relay, owners unreadable",
r: alertRoute{relay: "smtp.example.com:587", lookupErr: errors.New("connection refused")},
alerts: "via smtp.example.com:587; could not read the Owner addresses",
beat: "none: no outside check (troubleshooting.md §14)",
},
{
name: "owners but no relay",
r: alertRoute{recipients: []string{"[email protected]"}},
alerts: "only logged: no email relay (press e)",
beat: "none: no outside check (troubleshooting.md §14)",
},
{
name: "mailed",
r: alertRoute{relay: "smtp.example.com:587", recipients: []string{"[email protected]", "[email protected]"}, heartbeatErr: errors.New("/etc/felis/watchdog-heartbeat-url: the heartbeat URL is not an http:// or https:// URL")},
alerts: "mailed to [email protected], [email protected] via smtp.example.com:587", alertsOK: true,
beat: "unreadable: /etc/felis/watchdog-heartbeat-url: the heartbeat URL is not an http:// or https:// URL",
},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if line, ok := c.r.alertsLine(); line != c.alerts || ok != c.alertsOK {
t.Errorf("alerts = %q, %v; want %q, %v", line, ok, c.alerts, c.alertsOK)
}
if line, ok := c.r.heartbeatLine(); line != c.beat || ok != c.beatOK {
t.Errorf("heartbeat = %q, %v; want %q, %v", line, ok, c.beat, c.beatOK)
}
})
}
}
// TestSummaryAlertRows: the rows show on the card, a route that needs action
// marked so it reads without colour, and stay off a summary with no route.
func TestSummaryAlertRows(t *testing.T) {
m := &summaryModel{ownerUsername: "owner", alerts: &alertRoute{relay: "smtp.example.com:587", recipients: []string{"[email protected]"}}}
v := m.View()
for _, want := range []string{"alerts mailed to [email protected] via smtp.example.com:587", "heartbeat ⚠ none: no outside check"} {
if !strings.Contains(v, want) {
t.Errorf("summary lacks %q:\n%s", want, v)
}
}
if strings.Contains(v, "⚠ mailed") {
t.Errorf("a route that reaches the owners is marked:\n%s", v)
}
m.alerts = &alertRoute{heartbeat: "https://hc-ping.com/..."}
v = m.View()
for _, want := range []string{"alerts ⚠ only logged: no email relay (press e)", "heartbeat https://hc-ping.com/... every 2 minutes"} {
if !strings.Contains(v, want) {
t.Errorf("summary lacks %q:\n%s", want, v)
}
}
m.alerts = nil
if v = m.View(); strings.Contains(v, "alerts") || strings.Contains(v, "heartbeat") {
t.Errorf("a summary with no route shows alert rows:\n%s", v)
}
}
// TestRootSummaryReadsAlertRoute: the summary and the re-run status screen
// read the route each time they open, so a relay configured with e shows at
// once.
func TestRootSummaryReadsAlertRoute(t *testing.T) {
m := newTestRoot(false, consoleModeSetup, "")
reads := 0
route := alertRoute{}
m.alertRoute = func(context.Context) alertRoute {
reads++
return route
}
m = drive(t, m, storageResultMsg{method: storageLocal, detail: "local disk"})
sum, ok := m.screen.(*summaryModel)
if !ok || sum.alerts == nil || sum.alerts.relay != "" {
t.Fatalf("summary after storage: %T %+v", m.screen, sum)
}
route = alertRoute{relay: "smtp.example.com:587", recipients: []string{"[email protected]"}}
m = drive(t, m, smtpResultMsg{configured: true})
sum, ok = m.screen.(*summaryModel)
if !ok || sum.alerts == nil || sum.alerts.relay != "smtp.example.com:587" {
t.Fatalf("summary after configuring email: %T %+v", m.screen, sum)
}
if reads != 2 {
t.Errorf("route read %d times, want 2", reads)
}
status := newTestRoot(true, consoleModeSetup, "")
status.alertRoute = m.alertRoute
status.showStatus()
if sum, ok := status.screen.(*summaryModel); !ok || sum.alerts == nil || sum.alerts.relay != "smtp.example.com:587" {
t.Fatalf("status screen: %T %+v", status.screen, status.screen)
}
}
// TestHostAlertRoute reads the relay from the host config and the heartbeat
// file, shows the heartbeat by its host only, and reports a database it cannot
// reach.
func TestHostAlertRoute(t *testing.T) {
dir := t.TempDir()
cfg := filepath.Join(dir, "felis.host.toml")
writeTestFile(t, cfg, testWatchdogConfig+testWatchdogSMTP, 0o600)
beat := filepath.Join(dir, "watchdog-heartbeat-url")
writeTestFile(t, beat, "https://hc-ping.com/check-key\n", 0o600)
dbURL := "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"
r := hostAlertRoute(context.Background(), cfg, dbURL, beat)
if r.relay != "smtp.config.example:2525" {
t.Errorf("relay = %q", r.relay)
}
if r.heartbeat != "https://hc-ping.com/..." || r.heartbeatErr != nil {
t.Errorf("heartbeat = %q, %v", r.heartbeat, r.heartbeatErr)
}
if r.lookupErr == nil {
t.Errorf("an unreachable database reads as %v", r.recipients)
}
noRelay := filepath.Join(dir, "no-relay.toml")
writeTestFile(t, noRelay, testWatchdogConfig, 0o600)
writeTestFile(t, beat, "hc-ping.com/check-key\n", 0o600)
r = hostAlertRoute(context.Background(), noRelay, dbURL, beat)
if r.relay != "" {
t.Errorf("no [smtp]: relay = %q", r.relay)
}
if r.heartbeatErr == nil || strings.Contains(r.heartbeatErr.Error(), "check-key") {
t.Errorf("a bad heartbeat file: %v", r.heartbeatErr)
}
r = hostAlertRoute(context.Background(), noRelay, dbURL, filepath.Join(dir, "none"))
if r.heartbeat != "" || r.heartbeatErr != nil {
t.Errorf("no heartbeat file: %q, %v", r.heartbeat, r.heartbeatErr)
}
}
// TestSummaryWithAlertsFitsTerminal: the two rows keep the summary inside the
// terminal, with the longest route lines.
func TestSummaryWithAlertsFitsTerminal(t *testing.T) {
for _, w := range []int{60, 80, 90} {
for _, h := range []int{24, 30, 45} {
m := newTestRoot(false, consoleModeSetup, "")
m.alertRoute = func(context.Context) alertRoute {
return alertRoute{relay: "smtp.example.com:587", lookupErr: errors.New("dial tcp 127.0.0.1:5432: connect: connection refused")}
}
m = drive(t, m, tea.WindowSizeMsg{Width: w, Height: h})
m = drive(t, m, preflightDoneMsg{})
m = drive(t, m, ownerResultMsg{username: "owner", setupTokenURL: "https://op.console.example.com/setup?token=t0ken"})
m = drive(t, m, connectResultMsg{method: connectLocal, panelHostname: "panel.example.com"})
m = drive(t, m, storageResultMsg{method: storageLocal, detail: "local disk · /var/lib/felis/uploads"})
if _, ok := m.screen.(*summaryModel); !ok {
t.Fatalf("screen = %T, want the summary", m.screen)
}
if got := lipgloss.Height(m.View()); got > h {
t.Errorf("terminal %dx%d: summary with alert rows = %d rows (exceeds height)", w, h, got)
}
}
}
}
// TestConsoleRootReadsHostAlertRoute: the console the host runs gives its
// summary this host's route, read against its database.
func TestConsoleRootReadsHostAlertRoute(t *testing.T) {
db := config.DatabaseConfig{URL: "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"}
rm := newConsoleRoot(context.Background(), &fakeOwnerStore{}, db, "felis.example.com", "admin.felis.example.com", "panel.felis.example.com", "", "minecraft", "root", false, consoleModeSetup, recoveryConfig{})
if rm.alertRoute == nil {
t.Fatal("the console's summary reads no alert route")
}
if r := rm.alertRoute(context.Background()); r.lookupErr == nil {
t.Errorf("the route did not query the console's database: %+v", r)
}
}
+3 -4
View File
@@ -6,7 +6,6 @@ import (
"io"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/store"
)
@@ -14,10 +13,10 @@ import (
// applyMigrations runs any pending schema migrations and returns the resulting
// applied count. Used by the preflight stage to self-heal a freshly bootstrapped
// (or upgraded) database.
func applyMigrations(db config.DatabaseConfig) (int, error) {
func applyMigrations(dbURL string) (int, error) {
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
defer cancel()
drv, err := store.Open(ctx, db.URL)
drv, err := store.Open(ctx, dbURL)
if err != nil {
return 0, fmt.Errorf("open db: %w", err)
}
@@ -28,7 +27,7 @@ func applyMigrations(db config.DatabaseConfig) (int, error) {
}
// Same guard as `felis migrate up`: never roll a populated database forward
// without a snapshot to roll back to.
if _, err := preMigrateBackup(ctx, drv, migrations, db, dbbackup.DefaultDir, io.Discard); err != nil {
if _, err := preMigrateBackup(ctx, drv, migrations, dbURL, dbbackup.DefaultDir, io.Discard); err != nil {
return 0, fmt.Errorf("pre-migration backup: %w", err)
}
if _, err := store.Up(ctx, drv, migrations); err != nil {
+25 -144
View File
@@ -5,7 +5,6 @@ import (
"errors"
"fmt"
"strings"
"time"
"felis.lolicon.best/internal/api"
@@ -15,7 +14,8 @@ import (
)
type owAuthMsg struct {
start recoveryStart
matched string
ok bool
err error
}
@@ -29,15 +29,14 @@ type owStep int
const (
owAuth owStep = iota
owOverride
owCode // typing the recovery code mailed to the named admin
owProvision
owWorking
owDone
owError
)
// ownerModel collects the owner account. The input phases (admin name, recovery
// code, root override, owner details) are huh forms; the async phases (verifying,
// ownerModel collects the owner account. The input phases (admin auth, root
// override, owner details) are huh forms; the async phases (verifying,
// provisioning) show a spinner; the done phase shows the credential card. The
// outward contract is unchanged: it emits an ownerResultMsg when finished.
//
@@ -56,18 +55,6 @@ type ownerModel struct {
mode string // "bootstrap", "recovery", "root_override"
accountable string
attempt string
recovery recoveryConfig
// Recovery proof: the admin the typed name resolved to, the code mailed to it,
// and once settled either how it was proven or why the run fell back to the
// override.
admin *api.StaffUser
code *recoveryCode
codeNote string // "wrong code" line shown above a rebuilt code form
verifiedBy string
codeSentTo string
skip string // otpSkip*
skipDetail string
step owStep
form *huh.Form
@@ -79,7 +66,6 @@ type ownerModel struct {
// huh-bound form values
authUser string
codeInput string
overrideTok string
ownerUser string
ownerEmail string
@@ -138,12 +124,6 @@ func newOperatorModel(ctx context.Context, store ownerStore, osUser string) *own
return m
}
// withRecovery hands the model the relay its recovery codes go through.
func (m *ownerModel) withRecovery(r recoveryConfig) *ownerModel {
m.recovery = r
return m
}
func (m *ownerModel) Init() tea.Cmd { return m.form.Init() }
// subject is the human label for the account being provisioned, branching every
@@ -172,7 +152,7 @@ func (m *ownerModel) sized(f *huh.Form) *huh.Form {
}
func (m *ownerModel) isFormStep() bool {
return m.step == owAuth || m.step == owOverride || m.step == owCode || m.step == owProvision
return m.step == owAuth || m.step == owOverride || m.step == owProvision
}
func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
@@ -181,15 +161,16 @@ func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
if msg.err != nil {
return m, m.failCmd(msg.err)
}
m.admin = msg.start.admin
if msg.start.code != nil {
m.code = msg.start.code
m.codeNote = ""
m.step = owCode
m.form = m.sized(m.buildCodeForm())
if msg.ok {
m.mode = "recovery"
m.accountable = msg.matched
m.step = owProvision
m.form = m.sized(m.buildProvisionForm())
return m, m.form.Init()
}
return m.toOverride(msg.start.skip, msg.start.detail)
m.step = owOverride
m.form = m.sized(m.buildOverrideForm())
return m, m.form.Init()
case owProvisionMsg:
if msg.err != nil {
@@ -240,9 +221,7 @@ func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
case "ctrl+c":
return m, tea.Quit
case "esc":
if m.step == owOverride || m.step == owCode {
// Start over: a code mailed for the old attempt dies with it.
m.admin, m.code, m.skip, m.skipDetail = nil, nil, "", ""
if m.step == owOverride {
m.step = owAuth
m.form = m.sized(m.buildAuthForm())
return m, m.form.Init()
@@ -274,41 +253,18 @@ func (m *ownerModel) onFormComplete() (tea.Model, tea.Cmd) {
case owAuth:
m.attempt = strings.TrimSpace(m.authUser)
m.step = owWorking
m.working = "Sending a recovery code…"
user, rc, osUser, op := m.authUser, m.recovery, m.osUser, m.operation
m.working = "Verifying admin…"
user := m.authUser
return m, tea.Batch(m.sp.Tick, func() tea.Msg {
start, err := beginRecovery(m.ctx, m.store, rc, user, osUser, op)
return owAuthMsg{start: start, err: err}
matched, ok, err := authenticateAdmin(m.ctx, m.store, user)
return owAuthMsg{matched: matched, ok: ok, err: err}
})
case owCode:
typed := strings.TrimSpace(m.codeInput)
m.codeInput = ""
if typed == breakGlassOverrideToken {
m.skip, m.skipDetail = otpSkipByOperator, ""
m.code = nil
return m.proceedAsRoot()
}
switch m.code.check(typed, m.recovery.clock()) {
case codeAccepted:
m.mode = "recovery"
m.accountable = m.admin.Username
m.verifiedBy = verifiedByEmailOTP
m.codeSentTo = m.admin.Email
m.code = nil
case owOverride:
m.mode = "root_override"
m.accountable = m.osUser
m.step = owProvision
m.form = m.sized(m.buildProvisionForm())
return m, m.form.Init()
case codeWrong:
m.codeNote = fmt.Sprintf("That code is wrong. %d attempts left.", m.code.attemptsLeft())
m.form = m.sized(m.buildCodeForm())
return m, m.form.Init()
case codeExpired:
return m.toOverride(otpSkipCodeExpired, "")
default:
return m.toOverride(otpSkipCodeRejected, fmt.Sprintf("%d wrong codes", recoveryCodeAttempts))
}
case owOverride:
return m.proceedAsRoot()
case owProvision:
m.username = strings.TrimSpace(m.ownerUser)
m.step = owWorking
@@ -318,24 +274,6 @@ func (m *ownerModel) onFormComplete() (tea.Model, tea.Cmd) {
return m, nil
}
// toOverride records why no code proved an admin and asks for the typed OVERRIDE.
func (m *ownerModel) toOverride(skip, detail string) (tea.Model, tea.Cmd) {
m.skip, m.skipDetail = skip, detail
m.code = nil
m.step = owOverride
m.form = m.sized(m.buildOverrideForm())
return m, m.form.Init()
}
// proceedAsRoot is the typed OVERRIDE: the run goes on as the OS user, unverified.
func (m *ownerModel) proceedAsRoot() (tea.Model, tea.Cmd) {
m.mode = "root_override"
m.accountable = m.osUser
m.step = owProvision
m.form = m.sized(m.buildProvisionForm())
return m, m.form.Init()
}
func (m *ownerModel) provisionCmd() tea.Cmd {
op := breakGlassOp{
mode: m.mode,
@@ -344,10 +282,6 @@ func (m *ownerModel) provisionCmd() tea.Cmd {
ownerUsername: m.username,
ownerEmail: m.ownerEmail,
attemptedAdmin: m.attempt,
verifiedBy: m.verifiedBy,
codeSentTo: m.codeSentTo,
otpSkipped: m.skip,
otpSkipDetail: m.skipDetail,
}
// performAddOperator and performBreakGlass share a signature; the operation
// discriminator selects which one runs. The operator path is insert-only and
@@ -386,7 +320,7 @@ func (m *ownerModel) buildAuthForm() *huh.Form {
return m.sized(newFelisForm(huh.NewGroup(
huh.NewNote().
Title("Admin authentication").
Description("A staff account already exists. Name yours: a one-time code goes to its verified email address."),
Description("A staff account already exists. Identify yourself to continue."),
huh.NewInput().
Title("Admin username").
Value(&m.authUser).
@@ -394,64 +328,11 @@ func (m *ownerModel) buildAuthForm() *huh.Form {
)))
}
func (m *ownerModel) buildCodeForm() *huh.Form {
desc := fmt.Sprintf("A recovery code went to %s, the verified address of %q. It works for %d minutes.\n\n"+
"No mail? Type %s to go on as OS user %q with root authority; the audit log records that as an unverified override. Esc starts over.",
maskEmail(m.admin.Email), m.admin.Username, int(recoveryCodeTTL/time.Minute), breakGlassOverrideToken, m.osUser)
if m.codeNote != "" {
desc = m.codeNote + "\n\n" + desc
}
return m.sized(newFelisForm(huh.NewGroup(
huh.NewNote().Title("Email verification").Description(desc),
huh.NewInput().
Title("Recovery code").
Value(&m.codeInput).
Validate(func(s string) error {
s = strings.TrimSpace(s)
if s == breakGlassOverrideToken || isRecoveryCodeShape(s) {
return nil
}
return errors.New("enter the 6-digit code, or " + breakGlassOverrideToken)
}),
)))
}
func isRecoveryCodeShape(s string) bool {
if len(s) != 6 {
return false
}
for _, c := range s {
if c < '0' || c > '9' {
return false
}
}
return true
}
// overrideReason says why no code proved an admin, first line of the override form.
func (m *ownerModel) overrideReason() string {
switch m.skip {
case otpSkipUnknownAdmin:
return fmt.Sprintf("No staff account is named %q.", m.attempt)
case otpSkipNoVerifiedEmail:
return fmt.Sprintf("%q has no verified email address, so no recovery code can reach it.", m.admin.Username)
case otpSkipNoRelay:
return "No mail relay can send a recovery code: " + m.skipDetail + "."
case otpSkipSendFailed:
return "The recovery code could not be sent: " + m.skipDetail + "."
case otpSkipCodeExpired:
return "The recovery code expired."
case otpSkipCodeRejected:
return fmt.Sprintf("%d wrong codes; that code no longer works.", recoveryCodeAttempts)
}
return "No admin was verified."
}
func (m *ownerModel) buildOverrideForm() *huh.Form {
return m.sized(newFelisForm(huh.NewGroup(
huh.NewNote().
Title("Root override").
Description(m.overrideReason()+"\n\n"+fmt.Sprintf("Proceed as OS user %q with root authority by typing the confirmation token. The audit log records this run as an unverified root override and the reason above. Esc starts over.", m.osUser)),
Description(fmt.Sprintf("That credential did not match. Proceed as OS user %q with root authority by typing the confirmation token.", m.osUser)),
huh.NewInput().
Title("Type "+breakGlassOverrideToken+" to confirm").
Value(&m.overrideTok).
@@ -471,14 +352,14 @@ func (m *ownerModel) buildProvisionForm() *huh.Form {
desc := fmt.Sprintf("Create the first Owner — recorded as OS user %q.", m.osUser)
switch m.mode {
case "recovery":
desc = fmt.Sprintf("Verified as %q by an email code.", m.accountable)
desc = fmt.Sprintf("Authenticated as %q.", m.accountable)
case "root_override":
desc = "Root override — the Owner will be reset."
}
if m.operation == bgAddOperator {
switch m.mode {
case "recovery":
desc = fmt.Sprintf("Add an Operator — verified as %q by an email code.", m.accountable)
desc = fmt.Sprintf("Add an Operator — authenticated as %q.", m.accountable)
case "root_override":
desc = "Add an Operator (root override)."
}
+6 -8
View File
@@ -3,8 +3,6 @@ package main
import (
"fmt"
"felis.lolicon.best/internal/config"
"github.com/charmbracelet/bubbles/spinner"
tea "github.com/charmbracelet/bubbletea"
)
@@ -16,7 +14,7 @@ import (
// of the wizard can assume a healthy backend. It is non-interactive — it runs
// to completion and reports back to the root via preflightDoneMsg.
type preflightModel struct {
db config.DatabaseConfig
dbURL string
rootDomain string
adminHost string
@@ -59,11 +57,11 @@ type pfMigApplyMsg struct {
type pfPanelMsg struct{ err error }
func newPreflightModel(db config.DatabaseConfig, rootDomain, adminHostname string) *preflightModel {
func newPreflightModel(dbURL, rootDomain, adminHostname string) *preflightModel {
sp := spinner.New()
sp.Spinner = spinner.Dot
sp.Style = tuiLabel
return &preflightModel{db: db, rootDomain: rootDomain, adminHost: adminHostname, sp: sp, state: pfCheckDB}
return &preflightModel{dbURL: dbURL, rootDomain: rootDomain, adminHost: adminHostname, sp: sp, state: pfCheckDB}
}
func (m *preflightModel) Init() tea.Cmd {
@@ -202,7 +200,7 @@ func (m *preflightModel) errText() string {
}
func (m *preflightModel) checkDB() tea.Cmd {
return func() tea.Msg { return pfDBMsg{err: checkPostgres(m.db.URL)} }
return func() tea.Msg { return pfDBMsg{err: checkPostgres(m.dbURL)} }
}
func (m *preflightModel) checkMigrations() tea.Cmd {
@@ -210,7 +208,7 @@ func (m *preflightModel) checkMigrations() tea.Cmd {
// The sets, not their sizes: a database a newer release migrated can hold as
// many rows as this build has migrations, and must stop here rather than be
// "healed" by an older binary.
s, err := readSchema(m.db.URL)
s, err := readSchema(m.dbURL)
if err == nil {
err = s.Newer()
}
@@ -220,7 +218,7 @@ func (m *preflightModel) checkMigrations() tea.Cmd {
func (m *preflightModel) applyMigrations() tea.Cmd {
return func() tea.Msg {
applied, err := applyMigrations(m.db)
applied, err := applyMigrations(m.dbURL)
return pfMigApplyMsg{applied: applied, err: err}
}
}
+6 -23
View File
@@ -5,7 +5,6 @@ import (
"strings"
"felis.lolicon.best/internal/cfsetup"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/platform"
tea "github.com/charmbracelet/bubbletea"
@@ -134,7 +133,7 @@ type rootModel struct {
err error
mode consoleMode
db config.DatabaseConfig
dbURL string
store ownerStore
osUser string
rootDomain string
@@ -143,17 +142,13 @@ type rootModel struct {
accessAud string
namespace string // minecraft workload namespace (cfg.K8s.Namespace); target of the halt op
adminExists bool
recovery recoveryConfig // how the account operations mail a recovery code
// alertRoute reads where the watchdog's alerts go for the summary; nil
// leaves those rows out.
alertRoute func(context.Context) alertRoute
}
func newRootModel(ctx context.Context, store ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode) *rootModel {
func newRootModel(ctx context.Context, store ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode) *rootModel {
rm := &rootModel{
ctx: ctx,
reviewing: -1,
db: db,
dbURL: dbURL,
store: store,
osUser: osUser,
rootDomain: rootDomain,
@@ -184,7 +179,7 @@ func newRootModel(ctx context.Context, store ownerStore, db config.DatabaseConfi
}
} else {
rm.stage = stagePreflight
rm.screen = newPreflightModel(db, rootDomain, adminHostname)
rm.screen = newPreflightModel(dbURL, rootDomain, adminHostname)
}
return rm
}
@@ -226,7 +221,7 @@ func (m *rootModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
m.stage = stageOwner
switch msg.op {
case bgAddOperator:
return m.adopt(newOperatorModel(m.ctx, m.store, m.osUser).withRecovery(m.recovery))
return m.adopt(newOperatorModel(m.ctx, m.store, m.osUser))
case bgHaltServer:
return m.adopt(newHaltModel(m.ctx, m.store, m.namespace, m.osUser))
case bgSyncBackup:
@@ -234,7 +229,7 @@ func (m *rootModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
// Service + service token the peer dials live in the control namespace.
return m.adopt(newBackupModel(m.ctx, m.namespace, platform.DefaultControlNamespace, m.osUser))
default:
return m.adopt(newOwnerModel(m.ctx, m.store, m.osUser, m.adminExists).withRecovery(m.recovery))
return m.adopt(newOwnerModel(m.ctx, m.store, m.osUser, m.adminExists))
}
case haltResultMsg:
@@ -559,7 +554,6 @@ func (m *rootModel) showSummary() (tea.Model, tea.Cmd) {
routedHosts: routed,
localHint: m.result.connectMethod == connectLocal,
alreadySetUp: m.result.alreadySetUp,
alerts: m.readAlertRoute(),
})
}
@@ -580,20 +574,9 @@ func (m *rootModel) showStatus() (tea.Model, tea.Cmd) {
accessLabel: accessLabel,
alreadySetUp: true,
localHint: m.accessAud == "" && rootDomainEmbeddedIP(m.rootDomain) != "",
alerts: m.readAlertRoute(),
})
}
// readAlertRoute reads the alert route afresh, so the summary shows a relay the
// Owner just configured with e.
func (m *rootModel) readAlertRoute() *alertRoute {
if m.alertRoute == nil {
return nil
}
r := m.alertRoute(m.ctx)
return &r
}
func panelURLFor(method connectMethod, panelHostname, rootDomain, adminHostname string) string {
if method != connectLocal && panelHostname != "" {
return "https://" + panelHostname
+1 -3
View File
@@ -5,8 +5,6 @@ import (
"strings"
"testing"
"felis.lolicon.best/internal/config"
tea "github.com/charmbracelet/bubbletea"
)
@@ -32,7 +30,7 @@ func newTestRoot(adminExists bool, mode consoleMode, accessAud string) *rootMode
return newRootModel(
context.Background(),
&fakeOwnerStore{},
config.DatabaseConfig{URL: "postgres://localhost/felis"},
"postgres://localhost/felis",
"felis.example.com",
"admin.felis.example.com",
"panel.felis.example.com",
+5 -14
View File
@@ -276,14 +276,11 @@ func validateSMTPFrom(s string) error {
}
// currentSMTPInputs reads the relay already recorded in felis.toml so the form
// pre-fills the non-secret fields. The password (the felis-smtp Secret and its
// host copy) is deliberately never read back — it must be re-entered to change.
// pre-fills the non-secret fields. The password lives only in the felis-smtp
// Secret and is deliberately never read back — it must be re-entered to change.
// Any read error falls back to a blank form rather than blocking reconfig.
func currentSMTPInputs() smtpInputs { return smtpInputsFrom(hostSetupConfigPath) }
// smtpInputsFrom is currentSMTPInputs for the config at path.
func smtpInputsFrom(path string) smtpInputs {
cfg, err := config.Load(path)
func currentSMTPInputs() smtpInputs {
cfg, err := config.Load(hostSetupConfigPath)
if err != nil || cfg.SMTP.Host == "" {
return smtpInputs{}
}
@@ -298,8 +295,7 @@ func smtpInputsFrom(path string) smtpInputs {
// applySMTPConfig proves the relay works, then persists it and rolls felis-api:
// Ping (a full transaction — connect/STARTTLS/AUTH/MAIL FROM/RCPT/DATA, which
// delivers one self-test message to the From address) → [smtp] into both config
// files → the password into /etc/felis/smtp-password (hostcreds.go) → the
// felis-smtp Secret → the config Secret → rollout. A failed Ping
// files → the felis-smtp Secret → the config Secret → rollout. A failed Ping
// leaves the install untouched, so a bad relay dies at the keyboard, not at a
// player's OTP.
//
@@ -334,11 +330,6 @@ func applySMTPConfig(ctx context.Context, in smtpInputs) error {
return err
}
}
// The host copy first: should the Secret fail, the next installer run applies
// it from this file.
if err := writeHostCredential(hostSMTPPasswordPath, in.password); err != nil {
return fmt.Errorf("keep the relay password in %s: %w", hostSMTPPasswordPath, err)
}
if err := applySMTPSecret(ctx, in.password); err != nil {
return err
}
+3 -13
View File
@@ -53,17 +53,8 @@ func applyStorageConfig(ctx context.Context, method storageMethod, in s3Inputs)
return err
}
// S3: land the credentials in their own Secret BEFORE the roll, so the optional
// env refs resolve on the fresh pod, and on the host before that, so the next
// installer run can apply the Secret again (hostcreds.go). Local needs no Secret.
// env refs resolve on the fresh pod. Local needs no Secret.
if method == storageS3 {
for _, c := range []struct{ path, value string }{
{hostUploadsS3AccessKeyPath, in.accessKey},
{hostUploadsS3SecretKeyPath, in.secretKey},
} {
if err := writeHostCredential(c.path, c.value); err != nil {
return fmt.Errorf("keep the bucket credentials in %s: %w", c.path, err)
}
}
if err := applyUploadsS3Secret(ctx, in.accessKey, in.secretKey); err != nil {
return err
}
@@ -79,9 +70,8 @@ func applyStorageConfig(ctx context.Context, method storageMethod, in s3Inputs)
// currentStorageInputs reads the storage backend already recorded in felis.toml so
// the reconfigure flow can pre-select the method and pre-fill the non-secret S3
// fields (endpoint/bucket/region). The credentials (the felis-uploads-s3 Secret
// and its host copies) are deliberately never read back — they must be re-entered
// to change.
// fields (endpoint/bucket/region). Credentials live only in the felis-uploads-s3
// Secret and are deliberately never read back — they must be re-entered to change.
// Any read error falls back to a blank local default rather than blocking reconfig.
func currentStorageInputs() (storageMethod, s3Inputs) {
cfg, err := config.Load(hostSetupConfigPath)
-77
View File
@@ -1,9 +1,7 @@
package main
import (
"context"
"strings"
"time"
tea "github.com/charmbracelet/bubbletea"
)
@@ -22,8 +20,6 @@ type summaryModel struct {
routedHosts []string
alreadySetUp bool // re-run: Owner pre-existed
localHint bool // show the self-signed-cert note
// alerts is where the watchdog's alerts go; nil leaves the rows out.
alerts *alertRoute
}
func (m *summaryModel) Init() tea.Cmd { return nil }
@@ -77,12 +73,6 @@ func (m *summaryModel) View() string {
if m.panelURL != "" {
card.WriteString(tuiLabel.Render("panel ") + m.panelURL + "\n")
}
if m.alerts != nil {
line, ok := m.alerts.alertsLine()
card.WriteString(routeRow("alerts ", line, ok))
line, ok = m.alerts.heartbeatLine()
card.WriteString(routeRow("heartbeat ", line, ok))
}
b.WriteString(tuiCardStyle.Render(strings.TrimRight(card.String(), "\n")) + "\n\n")
b.WriteString(tuiHint.Render("ℹ Everything else — servers, users, plugins — is configured in the panel. You won't need this console again.") + "\n")
@@ -93,70 +83,3 @@ func (m *summaryModel) View() string {
b.WriteString("\n" + tuiAction("c", "change connection", "s", "change storage", "e", "configure email", "enter/esc", "exit"))
return b.String()
}
// alertRoute is where this host's watchdog alerts go, as the summary shows it:
// by mail through the [smtp] relay to the Owners' verified addresses, and the
// heartbeat that notices the host itself going down (docs/troubleshooting.md
// §14). Setup runs mail-less by design, so a fresh install has neither; the
// summary says so where the Owner can press e.
type alertRoute struct {
relay string // "host:port", "" with no [smtp] relay
recipients []string // enabled Owners' verified addresses
lookupErr error // the recipients could not be read
heartbeat string // the heartbeat URL's scheme and host, "" with none
heartbeatErr error // the heartbeat file does not read
}
// hostAlertRoute reads the route from the host config at cfgPath, the database
// at dbURL and the heartbeat file at heartbeatPath: what the next watchdog run
// uses.
func hostAlertRoute(ctx context.Context, cfgPath, dbURL, heartbeatPath string) alertRoute {
var r alertRoute
if in := smtpInputsFrom(cfgPath); in.host != "" {
r.relay = in.host + ":" + in.port
}
ctx, cancel := context.WithTimeout(ctx, 5*time.Second)
defer cancel()
r.recipients, r.lookupErr = ownerEmails(ctx, dbURL)
u, err := readHeartbeatURL(heartbeatPath)
switch {
case err != nil:
r.heartbeatErr = err
case u != "":
r.heartbeat = redactURL(u)
}
return r
}
// alertsLine is the summary's alerts row; ok is false when the alerts reach no one.
func (r alertRoute) alertsLine() (line string, ok bool) {
switch {
case r.relay == "":
return "only logged: no email relay (press e)", false
case r.lookupErr != nil:
return "via " + r.relay + "; could not read the Owner addresses", false
case len(r.recipients) == 0:
return "only logged: no verified Owner email (panel → Account)", false
}
return "mailed to " + strings.Join(r.recipients, ", ") + " via " + r.relay, true
}
// heartbeatLine is the summary's heartbeat row; ok is false with no heartbeat.
func (r alertRoute) heartbeatLine() (line string, ok bool) {
switch {
case r.heartbeatErr != nil:
return "unreadable: " + r.heartbeatErr.Error(), false
case r.heartbeat == "":
return "none: no outside check (troubleshooting.md §14)", false
}
return r.heartbeat + " every 2 minutes", true
}
// routeRow renders one alert row, marked and in the warning style when it
// needs action: the mark reads without colour too.
func routeRow(label, line string, ok bool) string {
if !ok {
line = tuiWarn.Render("⚠ " + line)
}
return tuiLabel.Render(label) + line + "\n"
}
+4 -6
View File
@@ -128,9 +128,8 @@ var updateTargets = []updateTarget{
selector: "postgres",
help: "select PostgreSQL",
component: "postgresql",
note: "PostgreSQL runs as the felis-postgres Deployment from the image the Felis release pins by digest; a newer minor reaches the host with a release that moves the pin, and the installer re-run restarts the database on it (a few seconds without the API). A new major is a dump and restore: docs/operations.md §4",
command: installerRerun,
installer: true,
note: "PostgreSQL comes from the distribution's packages, so a minor release is a package update followed by a restart (a few seconds without the API). A new major needs pg_upgrade first: docs/operations.md §4",
command: "sudo dnf upgrade 'postgresql*' || sudo apt-get install --only-upgrade 'postgresql*'; sudo systemctl restart postgresql",
},
{
selector: "mc",
@@ -150,9 +149,8 @@ var updateTargets = []updateTarget{
// reinstall/repair case.
//
// It never applies anything and never mutates the node, so unlike setup/breakGlass
// it needs no root, apart from PostgreSQL's version, which the felis-postgres container
// answers through the cluster's admin kubeconfig. The versions it reads come from this
// host: k3s, cloudflared and PostgreSQL answer `--version`, Velocity's version is read out of the installed jar's
// it needs no root. The versions it reads come from this host: k3s, cloudflared and
// PostgreSQL answer `--version`, Velocity's version is read out of the installed jar's
// manifest, the JRE's out of its release file, and felis-api's is this binary's own
// build stamp — the same value `felis version` prints, which is what the user asked
// to be the source of truth.
+4 -4
View File
@@ -215,8 +215,8 @@ func TestUpdateSelectorsAreFlags(t *testing.T) {
}
}
// k3s and cloudflared move only when the re-run is told to; the release pins the JRE
// build and the PostgreSQL image, so their guidance is the plain re-run.
// k3s and cloudflared move only when the re-run is told to; PostgreSQL is the package
// manager's, so its guidance carries no installer trailer.
func TestApplyGuidanceForHostDependencies(t *testing.T) {
notify := func(c string) updater.Result {
return planResult([]updates.Action{{Component: c, Kind: updates.ActionNotify, LatestKnown: true}})
@@ -232,8 +232,8 @@ func TestApplyGuidanceForHostDependencies(t *testing.T) {
t.Errorf("--jre guidance is the plain installer re-run:\n%s", jre)
}
pg := renderApplyGuidance(notify("postgresql"), map[string]bool{"postgres": true}, false)
if !strings.Contains(pg, "| sudo bash") || strings.Contains(pg, "FELIS_UPGRADE_DEPS") || strings.Contains(pg, "apt-get") || strings.Contains(pg, "systemctl") {
t.Errorf("--postgres guidance is the plain installer re-run that moves the image pin:\n%s", pg)
if !strings.Contains(pg, "apt-get install --only-upgrade") || strings.Contains(pg, "Re-running the installer") {
t.Errorf("--postgres guidance is the package manager, without the installer trailer:\n%s", pg)
}
}
+106 -348
View File
@@ -2,18 +2,17 @@ package main
import (
"context"
"database/sql"
"errors"
"flag"
"fmt"
"io"
"net"
"net/http"
"os"
"strings"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/mail"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
@@ -29,134 +28,42 @@ const proxyFor = 3 * time.Minute
// cmdWatchdog runs one pass of the platform watchdog (internal/watchdog): it
// checks the cluster, PostgreSQL, the game proxy, the database backups and the
// host (disks, memory, its address and clock), prints every finding, and mails
// the platform owners what came due.
// host, prints every finding, and mails the platform owners what came due.
// deploy/bootstrap.sh runs it every two minutes from felis-watchdog.timer.
func cmdWatchdog(args []string, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("watchdog", flag.ContinueOnError)
fs.SetOutput(stderr)
var w watchdogFlags
w.register(fs)
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
statePath := fs.String("state", "/var/lib/felis/watchdog/state.json", "state kept between runs (root only: it caches the relay password)")
quietPath := fs.String("quiet-file", "/run/felis/watchdog-quiet-until", "Unix time before which nothing is mailed; the installer writes it while it restarts things on purpose")
backupDir := fs.String("backup-dir", "/var/lib/felis/db-backups", `control-plane database backups to check for freshness ("" skips the check)`)
diskPaths := fs.String("disk-paths", "/,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis", "comma-separated paths whose filesystems must keep free space")
proxyAddr := fs.String("proxy-addr", "", `game proxy address to dial, e.g. 127.0.0.1:25565 ("" skips the check)`)
controlNS := fs.String("control-namespace", platform.DefaultControlNamespace, "namespace of the control plane")
offsiteStatus := fs.String("offsite-status", offsite.DefaultStatusFile, "the record `felis offsite sync` leaves, checked when [offsite] is configured")
toolsStatus := fs.String("build-tools-status", defaultBuildToolsStatus, "the record `felis mirror-build-tools` leaves, checked when builds scan against the registry's DB copy")
dryRun := fs.Bool("dry-run", false, "print every finding and the mail that is due; send nothing and keep the state as it was")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
now := time.Now()
if w.unitFailed {
return watchdogUnitFailed(unitFailedRun{
cfgPath: w.cfgPath, statePath: w.statePath, fallbackPath: w.fallbackState, quietPath: w.quietPath,
offsiteStatus: w.offsiteStatus, heartbeatFile: w.heartbeatFile,
result: os.Getenv("MONITOR_SERVICE_RESULT"), exitStatus: os.Getenv("MONITOR_EXIT_STATUS"),
send: watchdogSender, client: http.DefaultClient, now: now,
}, stdout, stderr)
}
cfg, err := config.Load(w.cfgPath)
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis watchdog: %v\n", err)
return 1
}
loadPath := watchdog.NewestState(w.statePath, w.fallbackState)
state, aside, err := watchdog.RecoverState(loadPath, now)
state, err := watchdog.LoadState(*statePath)
if err != nil {
fmt.Fprintf(stderr, "felis watchdog: %v\n", err)
return 1
}
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
defer cancel()
now := time.Now()
var report watchdog.Report
if aside != "" {
fmt.Fprintf(stderr, "felis watchdog: %s was unreadable; moved it to %s and started over\n", loadPath, aside)
report.Findings = append(report.Findings, watchdog.StateSetAside(aside))
}
cl, owners, ownersErr := watchdogProbes(ctx, w, cfg, now, &report)
if cfg.SMTP.Host != "" {
refreshSMTPPassword(ctx, w.smtpPasswordFile, cl, w.controlNS, state, stderr)
}
state.Relay = cachedRelay(cfg.SMTP)
if ownersErr == nil {
state.Recipients = owners
}
if len(report.Findings) == 0 {
fmt.Fprintln(stdout, "felis watchdog: every check passed")
}
for _, f := range report.Findings {
fmt.Fprintf(stdout, "felis watchdog: [%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
}
plan := state.Observe(report, now)
host, _ := os.Hostname()
subject, body := plan.Message(host, now)
beat := heartbeat{
standby: standsBy(cfg.Offsite.Enabled(), w.offsiteStatus),
quiet: now.Before(watchdog.QuietUntil(w.quietPath)),
}
if beat.url, err = readHeartbeatURL(w.heartbeatFile); err != nil {
fmt.Fprintf(stderr, "felis watchdog: %v; pinging no heartbeat\n", err)
}
if w.dryRun {
if plan.Empty() {
fmt.Fprintln(stdout, "felis watchdog: nothing is due to be mailed")
} else {
fmt.Fprintf(stdout, "felis watchdog: due to be mailed to %s:\nSubject: %s\n\n%s", strings.Join(state.Recipients, ", "), subject, strings.ReplaceAll(body, "\r\n", "\n"))
}
if beat.url != "" {
fmt.Fprintf(stdout, "felis watchdog: a run pings the heartbeat at %s\n", redactURL(beat.url))
}
return 0
}
m := configMailer(cfg.SMTP, state.SMTPPassword, watchdogSender)
unheard, mailFailed := m.deliver(ctx, state, plan, subject, body, mailHold(w.quietPath, cfg.Offsite.Enabled(), w.offsiteStatus, now), now, stdout, stderr)
saveErr := watchdog.SaveStateOr(w.statePath, w.fallbackState, state)
if saveErr != nil {
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", saveErr)
}
beat.report = failureReport(unheard, mailFailed, saveErr, state.Open(), report)
beat.fail = beat.report != ""
beat.send(http.DefaultClient, stdout, stderr)
if mailFailed || saveErr != nil {
return 1
}
return 0
}
// watchdogFlags are felis watchdog's flags. felis doctor reads them back from
// the ExecStart= line of felis-watchdog.service, so it checks what the timer's
// runs check, with the same paths.
type watchdogFlags struct {
cfgPath, statePath, fallbackState, smtpPasswordFile, quietPath string
backupDir, diskPaths, certDirs, proxyAddr, nodeIP, controlNS string
offsiteStatus, toolsStatus, heartbeatFile string
dryRun, unitFailed bool
}
func (w *watchdogFlags) register(fs *flag.FlagSet) {
fs.StringVar(&w.cfgPath, "config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
fs.StringVar(&w.statePath, "state", "/var/lib/felis/watchdog/state.json", "state kept between runs (root only: it caches the relay password)")
fs.StringVar(&w.fallbackState, "fallback-state", watchdog.FallbackStatePath, "where a run keeps its state while -state cannot be written, so what it mailed is not mailed again (tmpfs: until the host restarts; \"\" keeps none)")
fs.StringVar(&w.smtpPasswordFile, "smtp-password-file", hostSMTPPasswordPath, "the relay password `felis setup` keeps on the host; the felis-smtp Secret stands in while it is missing")
fs.StringVar(&w.quietPath, "quiet-file", "/run/felis/watchdog-quiet-until", "Unix time before which nothing is mailed; the installer writes it while it restarts things on purpose")
fs.StringVar(&w.backupDir, "backup-dir", "/var/lib/felis/db-backups", `control-plane database backups to check for freshness ("" skips the check)`)
fs.StringVar(&w.diskPaths, "disk-paths", "/,/var/lib/rancher/k3s,/var/lib/felis", "comma-separated paths whose filesystems must keep free space")
fs.StringVar(&w.certDirs, "k3s-cert-dirs", strings.Join(watchdog.K3sCertDirs, ","), `k3s certificate directories whose *.crt files must not be near expiry ("" skips the check)`)
fs.StringVar(&w.proxyAddr, "proxy-addr", "", `game proxy address to dial, e.g. 127.0.0.1:25565 ("" skips the check)`)
fs.StringVar(&w.nodeIP, "node-ip", "", `the node address the install was made on, which must stay on this host ("" skips the check)`)
fs.StringVar(&w.controlNS, "control-namespace", platform.DefaultControlNamespace, "namespace of the control plane")
fs.StringVar(&w.offsiteStatus, "offsite-status", offsite.DefaultStatusFile, "the record `felis offsite sync` leaves, checked when [offsite] is configured")
fs.StringVar(&w.toolsStatus, "build-tools-status", defaultBuildToolsStatus, "the record `felis mirror-build-tools` leaves, checked when builds scan against the registry's DB copy")
fs.BoolVar(&w.dryRun, "dry-run", false, "print every finding and the mail that is due; send nothing and keep the state as it was")
fs.StringVar(&w.heartbeatFile, "heartbeat-file", defaultHeartbeatFile, "file holding the heartbeat URL each run pings, a dead man's switch at a monitoring service that alerts when the pings stop (no file pings nothing)")
fs.BoolVar(&w.unitFailed, "unit-failed", false, "report a failed run of felis-watchdog.service instead of checking; felis-watchdog-failed.service runs this through OnFailure=")
}
// watchdogProbes is one pass of every check, appended to report. cl is the
// cluster client, nil while the API server is unreachable; owners is who the
// alerts go to, and ownersErr why PostgreSQL did not say.
func watchdogProbes(ctx context.Context, w watchdogFlags, cfg *config.Config, now time.Time, report *watchdog.Report) (cl client.Client, owners []string, ownersErr error) {
add := func(f *watchdog.Finding) {
if f != nil {
report.Findings = append(report.Findings, *f)
@@ -171,237 +78,90 @@ func watchdogProbes(ctx context.Context, w watchdogFlags, cfg *config.Config, no
cl, err := buildSystemServerClient()
var found []watchdog.Finding
if err == nil {
found, err = watchdog.Cluster{Client: cl, ControlNamespace: w.controlNS, MinecraftNamespace: minecraftNS}.Check(ctx, now)
found, err = watchdog.Cluster{Client: cl, ControlNamespace: *controlNS, MinecraftNamespace: minecraftNS}.Check(ctx, now)
}
if err != nil {
cl = nil
f := watchdog.KubeAPIDown(err)
add(&f)
report.Unknown = append(report.Unknown, watchdog.ClusterPrefixes...)
} else {
report.Findings = append(report.Findings, found...)
if cfg.SMTP.Host != "" {
refreshSMTPPassword(ctx, cl, *controlNS, state, stderr)
}
}
if owners, ownersErr = ownerEmails(ctx, cfg.Database.URL); ownersErr != nil {
f := watchdog.PostgresDown(ownersErr)
if recipients, err := ownerEmails(ctx, cfg.Database.URL); err != nil {
f := watchdog.PostgresDown(err)
add(&f)
} else {
state.Recipients = recipients
}
if w.proxyAddr != "" {
add(proxyFinding(ctx, w.proxyAddr))
if *proxyAddr != "" {
add(proxyFinding(ctx, *proxyAddr))
}
if w.backupDir != "" {
add(watchdog.BackupFinding(w.backupDir, now))
if *backupDir != "" {
add(watchdog.BackupFinding(*backupDir, now))
}
if cfg.Offsite.Enabled() {
add(watchdog.OffsiteFinding(w.offsiteStatus, now))
add(watchdog.OffsiteFinding(*offsiteStatus, now))
}
if usesMirroredScanDB(cfg) {
add(watchdog.ScanDBFinding(w.toolsStatus, now))
add(watchdog.ScanDBFinding(*toolsStatus, now))
}
report.Findings = append(report.Findings, watchdog.DiskFindings(splitList(w.diskPaths))...)
report.Findings = append(report.Findings, watchdog.DiskFindings(splitList(*diskPaths))...)
add(watchdog.MemoryFinding("/proc/meminfo"))
add(watchdog.CertFinding(splitList(w.certDirs), now))
if w.nodeIP != "" {
if held, err := watchdog.HostAddresses(); err == nil {
add(watchdog.AddressFinding(w.nodeIP, held))
if len(report.Findings) == 0 {
fmt.Fprintln(stdout, "felis watchdog: every check passed")
}
}
add(watchdog.ClockFinding(watchdog.ClockStatus()))
return cl, owners, ownersErr
for _, f := range report.Findings {
fmt.Fprintf(stdout, "felis watchdog: [%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
}
// failureReport is what the heartbeat's failure ping carries, "" when the run
// pings success: the alerts this run knows of reach no one (a mail that
// failed, or no relay or recipient while something is open), or the state did
// not save to its file: kept on tmpfs it holds until the host restarts, which
// forgets what was mailed, and with nowhere to keep it the next run mails the
// same alerts again.
func failureReport(unheard string, mailFailed bool, saveErr error, open bool, r watchdog.Report) string {
var why []string
if unheard != "" && (mailFailed || open) {
why = append(why, "the alerts reach no one: "+unheard)
plan := state.Observe(report, now)
host, _ := os.Hostname()
subject, body := plan.Message(host, now)
if *dryRun {
if plan.Empty() {
fmt.Fprintln(stdout, "felis watchdog: nothing is due to be mailed")
} else {
fmt.Fprintf(stdout, "felis watchdog: due to be mailed to %s:\nSubject: %s\n\n%s", strings.Join(state.Recipients, ", "), subject, strings.ReplaceAll(body, "\r\n", "\n"))
}
if saveErr != nil {
why = append(why, "the watchdog state did not save: "+saveErr.Error())
}
if len(why) == 0 {
return ""
}
return strings.Join(why, "\n") + "\n\n" + findingLines(r)
return 0
}
// findingLines is the report as the journal shows it.
func findingLines(r watchdog.Report) string {
var b strings.Builder
for _, f := range r.Findings {
fmt.Fprintf(&b, "[%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
}
return b.String()
}
// mailer is how a run reaches the owners.
type mailer struct {
relay *watchdog.Relay // nil: no [smtp] relay
password string
send func(*mail.SMTP) alertSender
}
// deliver mails plan to the owners unless hold says why it waits, and commits
// it once it reached them, or once it is logged because nothing can reach them.
// unheard is why the owners hear nothing of this run's alerts, "" when they do;
// mailFailed is a mail that did not go out, left uncommitted so the same
// alerts come due again next run.
func (m mailer) deliver(ctx context.Context, state *watchdog.State, plan watchdog.Plan, subject, body, hold string, now time.Time, stdout, stderr io.Writer) (unheard string, mailFailed bool) {
switch {
case m.relay == nil:
unheard = "no [smtp] relay is configured"
case len(state.Recipients) == 0:
unheard = "no owner account has a verified email"
}
switch {
case plan.Empty():
case hold != "":
fmt.Fprintf(stdout, "felis watchdog: %s; holding this mail: %s\n", hold, subject)
case unheard != "":
fmt.Fprintf(stdout, "felis watchdog: %s, so this is logged only: %s\n", unheard, subject)
state.Commit(plan, now)
default:
relay := &mail.SMTP{Host: m.relay.Host, Port: m.relay.Port, From: m.relay.From, Username: m.relay.Username, Password: m.password, RequireTLS: m.relay.RequireTLS}
if err := sendAlert(ctx, m.send(relay), state.Recipients, subject, body); err != nil {
fmt.Fprintf(stderr, "felis watchdog: mail %q: %v\n", subject, err)
return "the alert mail failed: " + err.Error(), true
}
fmt.Fprintf(stdout, "felis watchdog: mailed %s: %s\n", strings.Join(state.Recipients, ", "), subject)
state.Commit(plan, now)
}
return unheard, false
}
// configMailer reaches the owners through the relay felis.toml configures.
func configMailer(c config.SMTPConfig, cachedPassword string, send func(*mail.SMTP) alertSender) mailer {
return mailer{relay: cachedRelay(c), password: smtpPassword(c, cachedPassword), send: send}
}
// cachedRelay is what State.Relay keeps of c, nil when no relay is configured.
func cachedRelay(c config.SMTPConfig) *watchdog.Relay {
if c.Host == "" {
return nil
}
return &watchdog.Relay{Host: c.Host, Port: c.Port, From: c.From, Username: c.Username, RequireTLS: c.TLSRequired()}
}
// smtpPassword is the relay password a mail signs in with: the env var [smtp]
// password_ref names when it is set, else the one the state caches.
func smtpPassword(c config.SMTPConfig, cached string) string {
if ref := c.PasswordRef; ref != "" && os.Getenv(ref) != "" {
return os.Getenv(ref)
}
return cached
}
// unitFailedRun is one run of felis-watchdog-failed.service.
type unitFailedRun struct {
cfgPath, statePath, fallbackPath, quietPath, offsiteStatus, heartbeatFile string
// result and exitStatus are what systemd hands an OnFailure= unit
// (MONITOR_SERVICE_RESULT, MONITOR_EXIT_STATUS; systemd 251 and later).
result, exitStatus string
send func(*mail.SMTP) alertSender
client *http.Client
now time.Time
}
// watchdogUnitFailed is felis-watchdog-failed.service, which systemd starts
// through OnFailure= when a run of felis-watchdog.service fails: a crash, a
// felis.toml that no longer loads, a hang past the unit's timeout, a mail
// that did not go out. Such a run checks and mails nothing, so this records
// the failure as the alert watchdog/run, due after five failed runs in a row
// and cleared by the next run that succeeds; mails it through the relay the
// last good run cached when felis.toml does not load; and pings the
// heartbeat's failure endpoint. Every other alert keeps its state.
func watchdogUnitFailed(r unitFailedRun, stdout, stderr io.Writer) int {
detail := failureDetail(r.result, r.exitStatus)
cfg, cfgErr := config.Load(r.cfgPath)
if cfgErr != nil {
detail += "; " + cfgErr.Error()
}
fmt.Fprintf(stdout, "felis watchdog: felis-watchdog.service failed: %s\n", detail)
offsiteOn := cfgErr != nil || cfg.Offsite.Enabled()
beat := heartbeat{
fail: true,
report: "felis-watchdog.service failed: " + detail,
standby: standsBy(offsiteOn, r.offsiteStatus),
quiet: r.now.Before(watchdog.QuietUntil(r.quietPath)),
}
var err error
if beat.url, err = readHeartbeatURL(r.heartbeatFile); err != nil {
fmt.Fprintf(stderr, "felis watchdog: %v; pinging no heartbeat\n", err)
}
state, err := watchdog.LoadState(watchdog.NewestState(r.statePath, r.fallbackPath))
if err != nil {
// The next run that gets that far moves a state that does not parse
// aside (watchdog.RecoverState).
fmt.Fprintf(stderr, "felis watchdog: %v; mailing nothing\n", err)
beat.report += "\nThe watchdog state does not load either, so nothing was mailed: " + err.Error()
beat.send(r.client, stdout, stderr)
save := func() int {
if err := watchdog.SaveState(*statePath, state); err != nil {
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", err)
return 1
}
plan := state.Observe(watchdog.Report{Findings: []watchdog.Finding{watchdog.WatchdogFailed(detail)}, Unknown: []string{""}}, r.now)
host, _ := os.Hostname()
subject, body := plan.Message(host, r.now)
m := mailer{relay: state.Relay, password: state.SMTPPassword, send: r.send}
if cfgErr == nil {
m = configMailer(cfg.SMTP, state.SMTPPassword, r.send)
return 0
}
ctx, cancel := context.WithTimeout(context.Background(), time.Minute)
defer cancel()
unheard, mailFailed := m.deliver(ctx, state, plan, subject, body, mailHold(r.quietPath, offsiteOn, r.offsiteStatus, r.now), r.now, stdout, stderr)
code := 0
if mailFailed {
code = 1
if plan.Empty() {
return save()
}
if err := watchdog.SaveStateOr(r.statePath, r.fallbackPath, state); err != nil {
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", err)
code = 1
if until := watchdog.QuietUntil(*quietPath); now.Before(until) {
fmt.Fprintf(stdout, "felis watchdog: quiet until %s (installer running); holding this mail: %s\n", until.UTC().Format(time.RFC3339), subject)
return save()
}
if unheard != "" {
beat.report += "\nThe alerts reach no one: " + unheard
}
beat.send(r.client, stdout, stderr)
return code
}
// failureDetail names how felis-watchdog.service failed.
func failureDetail(result, exitStatus string) string {
switch {
case result == "":
return "systemd named no cause (journalctl -u felis-watchdog -n 50)"
case exitStatus == "":
return "result " + result
case cfg.SMTP.Host == "":
fmt.Fprintf(stdout, "felis watchdog: no [smtp] relay configured, so this is logged only: %s\n", subject)
case len(state.Recipients) == 0:
fmt.Fprintf(stdout, "felis watchdog: no owner account has a verified email, so this is logged only: %s\n", subject)
default:
if err := sendAlert(ctx, cfg, state, subject, body); err != nil {
// Not committed: the same alerts come due again next run.
fmt.Fprintf(stderr, "felis watchdog: mail %q: %v\n", subject, err)
save()
return 1
}
return "result " + result + ", exit status " + exitStatus
fmt.Fprintf(stdout, "felis watchdog: mailed %s: %s\n", strings.Join(state.Recipients, ", "), subject)
}
// mailHold is why this run's mail waits, "" when it goes out: the installer's
// quiet window, or this host standing by for another host's off-site bucket
// (offsite.Status.StandsBy). A standby host is a rehearsal, or a rebuild not
// yet taken over, and the owners its restored database names are that host's,
// which mails them itself.
func mailHold(quietPath string, offsiteOn bool, offsiteStatus string, now time.Time) string {
if until := watchdog.QuietUntil(quietPath); now.Before(until) {
return fmt.Sprintf("quiet until %s (installer running)", until.UTC().Format(time.RFC3339))
}
if !offsiteOn {
return ""
}
st, err := offsite.ReadStatus(offsiteStatus)
if err != nil {
return ""
}
if w := st.StandsBy(now); w != nil {
return fmt.Sprintf("this host stands by for %s, which wrote the off-site bucket at %s and mails its owners itself", w, w.At.UTC().Format(time.RFC3339))
}
return ""
state.Commit(plan, now)
return save()
}
// usesMirroredScanDB reports whether build scans read the vulnerability DB copy
@@ -415,40 +175,24 @@ func usesMirroredScanDB(cfg *config.Config) bool {
return repo == "" || strings.HasPrefix(repo, cfg.Registry.URL+"/mirror/")
}
// refreshSMTPPassword caches the relay password (relayPassword: the host copy at
// path, else the felis-smtp Secret), or forgets it when the Secret is gone (a
// relay without AUTH). The host copy is read even while the cluster is down, the
// time an alert matters most; without one, a down cluster keeps the cached
// password. An env var named by [smtp] password_ref, when set, wins at send time
// instead.
func refreshSMTPPassword(ctx context.Context, path string, cl client.Client, ns string, state *watchdog.State, stderr io.Writer) {
password, err := relayPassword(ctx, path, cl, ns)
switch {
case errors.Is(err, errClusterUnreachable):
return
case err != nil:
fmt.Fprintf(stderr, "felis watchdog: read the relay password (keeping the cached one): %v\n", err)
return
}
state.SMTPPassword = password
}
// smtpSecretPassword reads the relay password from the felis-smtp Secret. A missing
// Secret is a relay without AUTH and reads as "".
func smtpSecretPassword(ctx context.Context, cl client.Client, ns string) (string, error) {
// refreshSMTPPassword caches the relay password from the felis-smtp Secret, or
// forgets it when the Secret is gone (a relay without AUTH). An env var named by
// [smtp] password_ref, when set, wins at send time instead.
func refreshSMTPPassword(ctx context.Context, cl client.Client, ns string, state *watchdog.State, stderr io.Writer) {
var sec corev1.Secret
err := cl.Get(ctx, client.ObjectKey{Namespace: ns, Name: platform.SMTPSecretName}, &sec)
if apierrors.IsNotFound(err) {
return "", nil
switch {
case apierrors.IsNotFound(err):
state.SMTPPassword = ""
case err != nil:
fmt.Fprintf(stderr, "felis watchdog: read %s/%s (keeping the cached relay password): %v\n", ns, platform.SMTPSecretName, err)
default:
state.SMTPPassword = string(sec.Data[platform.SMTPSecretPasswordKey])
}
if err != nil {
return "", err
}
return string(sec.Data[platform.SMTPSecretPasswordKey]), nil
}
// ownerEmails pings PostgreSQL and returns the owners an alert goes to
// (watchdog.OwnerEmails).
// ownerEmails pings PostgreSQL and returns the verified addresses of the
// enabled owner accounts, the people who can act on an alert.
func ownerEmails(ctx context.Context, url string) ([]string, error) {
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
defer cancel()
@@ -457,7 +201,24 @@ func ownerEmails(ctx context.Context, url string) ([]string, error) {
return nil, err
}
defer drv.Close()
return watchdog.OwnerEmails(ctx, drv.DB())
rows, err := drv.DB().QueryContext(ctx,
`SELECT email FROM users
WHERE role = 'owner' AND email_verified AND COALESCE(email, '') <> ''
AND NOT disabled AND deleted_at IS NULL
ORDER BY email`)
if err != nil {
return nil, err
}
defer rows.Close()
var out []string
for rows.Next() {
var email sql.NullString
if err := rows.Scan(&email); err != nil {
return nil, err
}
out = append(out, email.String)
}
return out, rows.Err()
}
// proxyFinding dials the game proxy; players reach every server through it.
@@ -476,24 +237,21 @@ func proxyFinding(ctx context.Context, addr string) *watchdog.Finding {
}
}
// alertSender mails one alert; smtpSender is the real one.
type alertSender func(ctx context.Context, to, subject, body string) error
func smtpSender(relay *mail.SMTP) alertSender { return relay.SendNotice }
// watchdogSender is what a run mails through; tests stand a recorder in.
var watchdogSender = smtpSender
// sendAlert mails subject/body to every recipient; it fails only when no
// recipient got it.
func sendAlert(ctx context.Context, send alertSender, recipients []string, subject, body string) error {
func sendAlert(ctx context.Context, cfg *config.Config, state *watchdog.State, subject, body string) error {
password := state.SMTPPassword
if ref := cfg.SMTP.PasswordRef; ref != "" && os.Getenv(ref) != "" {
password = os.Getenv(ref)
}
relay := smtpRelay(cfg.SMTP, password)
var errs []error
for _, to := range recipients {
if err := send(ctx, to, subject, body); err != nil {
for _, to := range state.Recipients {
if err := relay.SendNotice(ctx, to, subject, body); err != nil {
errs = append(errs, fmt.Errorf("%s: %w", to, err))
}
}
if len(errs) == len(recipients) {
if len(errs) == len(state.Recipients) {
return errors.Join(errs...)
}
return nil
-164
View File
@@ -1,164 +0,0 @@
package main
import (
"context"
"errors"
"fmt"
"io"
"net/http"
"net/url"
"os"
"strings"
"time"
"felis.lolicon.best/internal/offsite"
)
// defaultHeartbeatFile holds the heartbeat URL deploy/bootstrap.sh writes from
// FELIS_WATCHDOG_HEARTBEAT_URL. It lives in /etc/felis, so a host rebuilt from
// a bundle pings the same check once it takes the off-site bucket over.
const defaultHeartbeatFile = "/etc/felis/watchdog-heartbeat-url"
const (
// heartbeatTimeout bounds one ping; a monitoring service slower than that
// is as good as down for the run.
heartbeatTimeout = 10 * time.Second
// heartbeatReportMax caps the report a failure ping carries: monitoring
// services keep about the first 10 KB of a ping's body.
heartbeatReportMax = 10000
)
// heartbeat is the ping one run sends to a dead man's switch at a monitoring
// service (Healthchecks.io, and the services that copy its API), which alerts
// its own users when the pings stop: the host down, the timer gone, the
// watchdog failing before it can mail. No check on the host can report those.
// A run whose alerts reach the owners GETs url; one whose alerts reach no one
// POSTs report to url/fail, or withholds the ping when url has a query, where
// no /fail can be added and the missing ping trips the check instead.
type heartbeat struct {
url string // "" pings nothing
fail bool
report string
// standby is a host standing by for another host's off-site bucket: a
// rehearsal, or a rebuild not taken over. It pings nothing, or it would
// keep the check of the host that writes the bucket green after that host
// died.
standby bool
// quiet is the installer's quiet window, which restarts things on
// purpose: no failure is pinged.
quiet bool
}
// send pings the heartbeat and logs the outcome.
func (b heartbeat) send(cl *http.Client, stdout, stderr io.Writer) {
switch {
case b.url == "":
return
case b.standby:
fmt.Fprintln(stdout, "felis watchdog: this host stands by for the off-site bucket; pinging no heartbeat (the host that writes the bucket pings it)")
return
case b.fail && b.quiet:
fmt.Fprintln(stdout, "felis watchdog: quiet while the installer runs; withholding the failure ping")
return
}
sent, err := b.ping(cl)
switch {
case err != nil:
fmt.Fprintf(stderr, "felis watchdog: heartbeat: %v\n", err)
case !sent:
fmt.Fprintf(stdout, "felis watchdog: withholding the heartbeat ping (the URL has a query, so it has no /fail endpoint)\n")
case b.fail:
fmt.Fprintf(stdout, "felis watchdog: pinged the heartbeat's failure endpoint at %s\n", redactURL(b.url))
}
}
// ping sends the request; sent is false when a failure withholds it.
func (b heartbeat) ping(cl *http.Client) (sent bool, err error) {
ctx, cancel := context.WithTimeout(context.Background(), heartbeatTimeout)
defer cancel()
method, target, body := http.MethodGet, b.url, io.Reader(nil)
if b.fail {
if strings.Contains(b.url, "?") {
return false, nil
}
method, target = http.MethodPost, strings.TrimSuffix(b.url, "/")+"/fail"
body = strings.NewReader(clipUTF8(b.report, heartbeatReportMax))
}
req, err := http.NewRequestWithContext(ctx, method, target, body)
if err != nil {
return false, fmt.Errorf("%s %s: bad URL", method, redactURL(target))
}
resp, err := cl.Do(req)
if err != nil {
// A url.Error quotes the whole URL, the check's key among it.
var ue *url.Error
if errors.As(err, &ue) {
err = ue.Err
}
return false, fmt.Errorf("%s %s: %w", method, redactURL(target), err)
}
io.Copy(io.Discard, io.LimitReader(resp.Body, 4096))
resp.Body.Close()
if resp.StatusCode/100 != 2 {
return false, fmt.Errorf("%s %s: %s", method, redactURL(target), resp.Status)
}
return true, nil
}
// clipUTF8 cuts s to at most n bytes without splitting a character.
func clipUTF8(s string, n int) string {
if len(s) <= n {
return s
}
return strings.ToValidUTF8(s[:n], "")
}
// readHeartbeatURL reads the heartbeat URL from path; no file is no heartbeat.
func readHeartbeatURL(path string) (string, error) {
if path == "" {
return "", nil
}
raw, err := os.ReadFile(path)
if errors.Is(err, os.ErrNotExist) {
return "", nil
}
if err != nil {
return "", err
}
s := strings.TrimSpace(string(raw))
if err := checkHeartbeatURL(s); err != nil {
return "", fmt.Errorf("%s: %w", path, err)
}
return s, nil
}
// checkHeartbeatURL accepts an http(s) URL with a host. Its error leaves the
// URL out: the path is the check's key, which anyone who reads it can ping in
// the host's name.
func checkHeartbeatURL(s string) error {
u, err := url.Parse(s)
if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" || strings.ContainsAny(s, " \t\r\n") {
return errors.New("the heartbeat URL is not an http:// or https:// URL")
}
return nil
}
// redactURL is a heartbeat URL as logs show it: its scheme and host.
func redactURL(s string) string {
u, err := url.Parse(s)
if err != nil || u.Host == "" {
return "(the heartbeat URL)"
}
return u.Scheme + "://" + u.Host + "/..."
}
// standsBy reports whether this host stands by for another host's off-site
// bucket (offsite.Status.Standby), however old that record is: the
// installer's take-over or the host's first write ends it.
func standsBy(offsiteOn bool, statusPath string) bool {
if !offsiteOn {
return false
}
st, err := offsite.ReadStatus(statusPath)
return err == nil && st != nil && st.Standby
}
-615
View File
@@ -1,615 +0,0 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"io"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"strings"
"sync"
"testing"
"time"
"unicode/utf8"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/mail"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/watchdog"
)
// pingLog is a monitoring service that records every ping it gets.
type pingLog struct {
mu sync.Mutex
pings []string // "METHOD path?query"
bodies []string
status int
}
func newPingServer(t *testing.T) (*pingLog, *httptest.Server) {
t.Helper()
l := &pingLog{status: http.StatusOK}
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
body, _ := io.ReadAll(r.Body)
l.mu.Lock()
defer l.mu.Unlock()
p := r.Method + " " + r.URL.Path
if r.URL.RawQuery != "" {
p += "?" + r.URL.RawQuery
}
l.pings = append(l.pings, p)
l.bodies = append(l.bodies, string(body))
w.WriteHeader(l.status)
}))
t.Cleanup(srv.Close)
return l, srv
}
func (l *pingLog) got() ([]string, []string) {
l.mu.Lock()
defer l.mu.Unlock()
return append([]string(nil), l.pings...), append([]string(nil), l.bodies...)
}
// TestHeartbeatSend: a run whose alerts reach the owners GETs the URL, one
// whose alerts reach no one POSTs its report to /fail, and a standby host, a
// failure in the installer's quiet window and a query URL that has no /fail
// send nothing.
func TestHeartbeatSend(t *testing.T) {
const key = "/ping/5f1e0c2a-check-key"
for _, tc := range []struct {
what string
beat heartbeat
path string // appended to the server URL
want string // the ping, "" for none
wantBody string
wantOut string
}{
{what: "success", beat: heartbeat{}, path: key, want: "GET " + key},
{what: "failure", beat: heartbeat{fail: true, report: "the alerts reach no one: x"}, path: key, want: "POST " + key + "/fail", wantBody: "the alerts reach no one: x", wantOut: "pinged the heartbeat's failure endpoint"},
{what: "failure, URL with a trailing slash", beat: heartbeat{fail: true, report: "r"}, path: key + "/", want: "POST " + key + "/fail", wantBody: "r"},
{what: "success, URL with a query", beat: heartbeat{}, path: key + "?rid=7", want: "GET " + key + "?rid=7"},
{what: "failure, URL with a query", beat: heartbeat{fail: true, report: "r"}, path: key + "?rid=7", wantOut: "withholding the heartbeat ping"},
{what: "standby, success", beat: heartbeat{standby: true}, path: key, wantOut: "stands by for the off-site bucket"},
{what: "standby, failure", beat: heartbeat{standby: true, fail: true}, path: key, wantOut: "stands by for the off-site bucket"},
{what: "quiet, failure", beat: heartbeat{quiet: true, fail: true}, path: key, wantOut: "withholding the failure ping"},
{what: "quiet, success", beat: heartbeat{quiet: true}, path: key, want: "GET " + key},
} {
log, srv := newPingServer(t)
b := tc.beat
b.url = srv.URL + tc.path
var stdout, stderr bytes.Buffer
b.send(srv.Client(), &stdout, &stderr)
pings, bodies := log.got()
switch {
case tc.want == "" && len(pings) != 0:
t.Errorf("%s: pinged %v, want nothing", tc.what, pings)
case tc.want != "" && (len(pings) != 1 || pings[0] != tc.want):
t.Errorf("%s: pinged %v, want %q", tc.what, pings, tc.want)
case tc.want != "" && bodies[0] != tc.wantBody:
t.Errorf("%s: body %q, want %q", tc.what, bodies[0], tc.wantBody)
}
if !strings.Contains(stdout.String(), tc.wantOut) || stderr.Len() != 0 {
t.Errorf("%s: stdout %q (want %q), stderr %q", tc.what, stdout.String(), tc.wantOut, stderr.String())
}
}
}
// TestHeartbeatErrors: a ping the service refuses or that cannot connect is
// logged without the URL's path, the check's key.
func TestHeartbeatErrors(t *testing.T) {
const key = "5f1e0c2a-check-key"
log, srv := newPingServer(t)
log.status = http.StatusNotFound
var stdout, stderr bytes.Buffer
heartbeat{url: srv.URL + "/" + key}.send(srv.Client(), &stdout, &stderr)
if got := stderr.String(); !strings.Contains(got, "heartbeat: GET http://127.0.0.1") || !strings.Contains(got, "404") || strings.Contains(got, key) {
t.Errorf("refused ping: stderr %q; want the host and status, not the key", got)
}
closed := httptest.NewServer(http.NotFoundHandler())
dead := closed.URL + "/" + key
closed.Close()
stderr.Reset()
heartbeat{url: dead, fail: true, report: "r"}.send(http.DefaultClient, &stdout, &stderr)
if got := stderr.String(); !strings.Contains(got, "heartbeat: POST http://127.0.0.1") || strings.Contains(got, key) {
t.Errorf("unreachable service: stderr %q; want the host, not the key", got)
}
}
// TestHeartbeatReportClipped: a long report is cut to what the service keeps,
// on a character boundary.
func TestHeartbeatReportClipped(t *testing.T) {
log, srv := newPingServer(t)
report := strings.Repeat("磁盘", heartbeatReportMax)
heartbeat{url: srv.URL + "/k", fail: true, report: report}.send(srv.Client(), io.Discard, io.Discard)
_, bodies := log.got()
if len(bodies) != 1 || len(bodies[0]) > heartbeatReportMax || len(bodies[0]) < heartbeatReportMax-3 || !utf8.ValidString(bodies[0]) {
t.Fatalf("clipped body: %d pings, %d bytes, valid UTF-8 %v", len(bodies), len(bodies[0]), utf8.ValidString(bodies[0]))
}
if got := clipUTF8("short", heartbeatReportMax); got != "short" {
t.Errorf("clipUTF8(short) = %q", got)
}
}
func TestReadHeartbeatURL(t *testing.T) {
dir := t.TempDir()
path := filepath.Join(dir, "watchdog-heartbeat-url")
if u, err := readHeartbeatURL(path); u != "" || err != nil {
t.Fatalf("no file: %q, %v", u, err)
}
if u, err := readHeartbeatURL(""); u != "" || err != nil {
t.Fatalf("no path: %q, %v", u, err)
}
writeTestFile(t, path, "https://hc-ping.com/5f1e0c2a\n", 0o600)
if u, err := readHeartbeatURL(path); u != "https://hc-ping.com/5f1e0c2a" || err != nil {
t.Fatalf("good file: %q, %v", u, err)
}
for _, bad := range []string{"", "hc-ping.com/5f1e0c2a", "ftp://hc-ping.com/5f1e0c2a", "https:///5f1e0c2a", "https://hc-ping.com/5f1e 0c2a"} {
writeTestFile(t, path, bad, 0o600)
u, err := readHeartbeatURL(path)
if u != "" || err == nil || !strings.Contains(err.Error(), path) || (bad != "" && strings.Contains(err.Error(), "5f1e")) {
t.Errorf("%q: %q, %v; want an error naming the file and not the URL", bad, u, err)
}
}
}
func TestRedactURL(t *testing.T) {
for in, want := range map[string]string{
"https://hc-ping.com/5f1e0c2a": "https://hc-ping.com/...",
"http://user:[email protected]:8080/k?x=1": "http://status.example:8080/...",
"not a url": "(the heartbeat URL)",
} {
if got := redactURL(in); got != want {
t.Errorf("redactURL(%q) = %q, want %q", in, got, want)
}
}
}
// TestStandsBy: a standby record keeps the host from pinging however old it
// is; a displaced host, the writer, or a host without [offsite] pings.
func TestStandsBy(t *testing.T) {
status := filepath.Join(t.TempDir(), "status.json")
if standsBy(true, status) {
t.Fatal("no status file stands by")
}
old := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: time.Now().Add(-30 * 24 * time.Hour)}
for _, tc := range []struct {
what string
st offsite.Status
offsiteOn bool
want bool
}{
{"standing by for a writer gone quiet a month ago", offsite.Status{Standby: true, Writer: old}, true, true},
{"standing by, [offsite] since removed", offsite.Status{Standby: true, Writer: old}, false, false},
{"displaced", offsite.Status{Displaced: true, Writer: old}, true, false},
{"the writer itself", offsite.Status{LastSuccess: time.Now()}, true, false},
} {
if err := offsite.WriteStatus(status, tc.st); err != nil {
t.Fatal(err)
}
if got := standsBy(tc.offsiteOn, status); got != tc.want {
t.Errorf("%s: standsBy = %v, want %v", tc.what, got, tc.want)
}
}
}
func TestFailureReport(t *testing.T) {
r := watchdog.Report{Findings: []watchdog.Finding{{Key: "postgres", Severity: watchdog.Critical, SummaryEN: "PostgreSQL is down"}}}
for _, tc := range []struct {
what string
unheard string
mailFailed bool
saveErr error
open bool
want string // "" for a success ping
}{
{what: "all good", open: true},
{what: "no relay, nothing mailed yet", unheard: "no [smtp] relay is configured"},
{what: "no relay, an alert open", unheard: "no [smtp] relay is configured", open: true, want: "the alerts reach no one: no [smtp] relay is configured"},
{what: "the mail failed", unheard: "the alert mail failed: refused", mailFailed: true, want: "the alerts reach no one: the alert mail failed: refused"},
{what: "the state did not save", saveErr: errors.New("disk full"), want: "the watchdog state did not save: disk full"},
} {
got := failureReport(tc.unheard, tc.mailFailed, tc.saveErr, tc.open, r)
if (got == "") != (tc.want == "") || !strings.Contains(got, tc.want) || (got != "" && !strings.Contains(got, "[critical] postgres: PostgreSQL is down")) {
t.Errorf("%s: report %q, want %q and the findings", tc.what, got, tc.want)
}
}
}
// alertRecorder records the mail a run hands the relay.
type alertRecorder struct {
relays []*mail.SMTP
sent []string // "to: subject"
fail map[string]bool
}
func (f *alertRecorder) sender(relay *mail.SMTP) alertSender {
f.relays = append(f.relays, relay)
return func(_ context.Context, to, subject, _ string) error {
if f.fail[to] {
return errors.New("550 refused")
}
f.sent = append(f.sent, to+": "+subject)
return nil
}
}
// TestMailerDeliver: a plan goes out and is committed; held, it waits
// uncommitted; with no relay or recipient it is logged and committed; a mail
// no recipient got is left uncommitted to come due again.
func TestMailerDeliver(t *testing.T) {
now := time.Now()
relay := &watchdog.Relay{Host: "smtp.example.com", Port: 587, From: "[email protected]", Username: "felis"}
for _, tc := range []struct {
what string
relay *watchdog.Relay
recipients []string
hold string
fail map[string]bool
wantUnheard string
wantFailed bool
committed bool
sent int
}{
{what: "mailed", relay: relay, recipients: []string{"[email protected]", "[email protected]"}, committed: true, sent: 2},
{what: "one recipient refused", relay: relay, recipients: []string{"[email protected]", "[email protected]"}, fail: map[string]bool{"[email protected]": true}, committed: true, sent: 1},
{what: "every recipient refused", relay: relay, recipients: []string{"[email protected]"}, fail: map[string]bool{"[email protected]": true}, wantUnheard: "the alert mail failed", wantFailed: true},
{what: "held", relay: relay, recipients: []string{"[email protected]"}, hold: "quiet until later"},
{what: "no relay", recipients: []string{"[email protected]"}, wantUnheard: "no [smtp] relay is configured", committed: true},
{what: "no recipient", relay: relay, wantUnheard: "no owner account has a verified email", committed: true},
} {
state := &watchdog.State{Recipients: tc.recipients}
plan := state.Observe(watchdog.Report{Findings: []watchdog.Finding{{Key: "postgres", Severity: watchdog.Critical, SummaryEN: "down"}}}, now)
f := &alertRecorder{fail: tc.fail}
m := mailer{relay: tc.relay, password: "relay-pw", send: f.sender}
var stdout, stderr bytes.Buffer
unheard, failed := m.deliver(context.Background(), state, plan, "subject", "body", tc.hold, now, &stdout, &stderr)
if !strings.HasPrefix(unheard, tc.wantUnheard) || (tc.wantUnheard == "") != (unheard == "") || failed != tc.wantFailed {
t.Errorf("%s: unheard %q, failed %v; want %q, %v", tc.what, unheard, failed, tc.wantUnheard, tc.wantFailed)
}
if got := !state.Alerts["postgres"].Notified.IsZero(); got != tc.committed {
t.Errorf("%s: committed %v, want %v", tc.what, got, tc.committed)
}
if len(f.sent) != tc.sent {
t.Errorf("%s: sent %v, want %d mails", tc.what, f.sent, tc.sent)
}
if len(f.relays) > 0 {
if got := *f.relays[0]; got != (mail.SMTP{Host: "smtp.example.com", Port: 587, From: "[email protected]", Username: "felis", Password: "relay-pw"}) {
t.Errorf("%s: relay %+v", tc.what, got)
}
}
}
}
// TestCachedRelay: the cache keeps the relay's coordinates and its TLS rule,
// and nothing when no relay is configured.
func TestCachedRelay(t *testing.T) {
if r := cachedRelay(testSMTPConfig("")); r != nil {
t.Fatalf("no relay cached %+v", r)
}
got := cachedRelay(testSMTPConfig("smtp.example.com"))
if got == nil || *got != (watchdog.Relay{Host: "smtp.example.com", Port: 2525, From: "[email protected]", Username: "felis", RequireTLS: true}) {
t.Fatalf("cachedRelay = %+v", got)
}
if got := cachedRelay(testSMTPConfig("127.0.0.1")); got == nil || got.RequireTLS {
t.Fatalf("a relay on this host: %+v, want TLS not required", got)
}
}
func testSMTPConfig(host string) config.SMTPConfig {
if host == "" {
return config.SMTPConfig{}
}
return config.SMTPConfig{Host: host, Port: 2525, From: "[email protected]", Username: "felis"}
}
// TestSMTPPassword: the env var password_ref names wins over the cache once
// it is set.
func TestSMTPPassword(t *testing.T) {
c := testSMTPConfig("smtp.example.com")
c.PasswordRef = "FELIS_TEST_WATCHDOG_RELAY_PW"
t.Setenv(c.PasswordRef, "")
if got := smtpPassword(c, "cached"); got != "cached" {
t.Errorf("env var unset: %q", got)
}
t.Setenv(c.PasswordRef, "from-env")
if got := smtpPassword(c, "cached"); got != "from-env" {
t.Errorf("env var set: %q", got)
}
}
const testWatchdogConfig = `[database]
url = "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"
[server]
root_domain = "example.com"
[archive]
store = "tarLocal"
[k8s]
egress_mode = "nodeport"
`
const testWatchdogSMTP = `[smtp]
host = "smtp.config.example"
port = 2525
from = "[email protected]"
username = "felis"
password_ref = "FELIS_TEST_WATCHDOG_RELAY_PW"
`
// unitFailedFixture is a host whose watchdog last ran well: one alert open,
// one owner and the relay cached.
func unitFailedFixture(t *testing.T, cfg string) (unitFailedRun, *alertRecorder, *pingLog) {
t.Helper()
dir := t.TempDir()
log, srv := newPingServer(t)
beatFile := filepath.Join(dir, "watchdog-heartbeat-url")
writeTestFile(t, beatFile, srv.URL+"/check-key\n", 0o600)
cfgPath := filepath.Join(dir, "felis.toml")
writeTestFile(t, cfgPath, cfg, 0o600)
statePath := filepath.Join(dir, "state.json")
state := &watchdog.State{
Recipients: []string{"[email protected]"},
SMTPPassword: "cached-pw",
Relay: &watchdog.Relay{Host: "smtp.cached.example", Port: 587, From: "[email protected]", RequireTLS: true},
}
mem := watchdog.Finding{Key: "memory", Severity: watchdog.Warning, SummaryEN: "memory low"}
state.Commit(state.Observe(watchdog.Report{Findings: []watchdog.Finding{mem}}, time.Now().Add(-time.Hour)), time.Now().Add(-time.Hour))
if err := watchdog.SaveState(statePath, state); err != nil {
t.Fatal(err)
}
f := &alertRecorder{}
return unitFailedRun{
cfgPath: cfgPath, statePath: statePath, fallbackPath: filepath.Join(t.TempDir(), "watchdog-state.json"), quietPath: filepath.Join(dir, "quiet"),
offsiteStatus: filepath.Join(dir, "offsite-status.json"), heartbeatFile: beatFile,
result: "exit-code", exitStatus: "1", send: f.sender, client: srv.Client(), now: time.Now(),
}, f, log
}
// TestWatchdogUnitFailedBrokenConfig: failed runs of a watchdog whose
// felis.toml no longer loads are mailed through the relay the last good run
// cached, after five in a row; every one pings /fail; the open alert keeps
// its state.
func TestWatchdogUnitFailedBrokenConfig(t *testing.T) {
r, f, log := unitFailedFixture(t, "[database\n")
start := r.now
for i := 0; i <= 5; i++ {
r.now = start.Add(time.Duration(i) * 2 * time.Minute)
var stdout, stderr bytes.Buffer
if code := watchdogUnitFailed(r, &stdout, &stderr); code != 0 {
t.Fatalf("run %d: exit %d (stdout %s, stderr %s)", i, code, stdout.String(), stderr.String())
}
if i < 5 && len(f.sent) != 0 {
t.Fatalf("run %d, %v after the first failure: mailed %v, want nothing yet", i, r.now.Sub(start), f.sent)
}
}
if len(f.sent) != 1 || !strings.Contains(f.sent[0], "[email protected]: ") {
t.Fatalf("mailed %v, want one alert to the cached owner", f.sent)
}
if got := *f.relays[0]; got.Host != "smtp.cached.example" || got.Password != "cached-pw" || !got.RequireTLS {
t.Fatalf("relay %+v, want the cached one", got)
}
pings, bodies := log.got()
if len(pings) != 6 || pings[0] != "POST /check-key/fail" || !strings.Contains(bodies[0], "felis-watchdog.service failed: result exit-code, exit status 1; ") {
t.Fatalf("pings %v, bodies %q", pings, bodies)
}
state, err := watchdog.LoadState(r.statePath)
if err != nil {
t.Fatal(err)
}
if a := state.Alerts["watchdog/run"]; a == nil || a.Notified.IsZero() || !strings.Contains(a.SummaryEN, "result exit-code, exit status 1") {
t.Fatalf("watchdog/run alert = %+v", a)
}
if a := state.Alerts["memory"]; a == nil || !a.ClearedAt.IsZero() || !a.Notified.Before(start) {
t.Fatalf("memory alert = %+v, want it untouched", a)
}
}
// TestWatchdogUnitFailedStateThatDoesNotSave: with the state file read-only,
// each report keeps the state in the fallback and the next one reads it, so
// the failure is mailed once, after five in a row, and never again.
func TestWatchdogUnitFailedStateThatDoesNotSave(t *testing.T) {
if os.Geteuid() == 0 {
t.Skip("root writes into a read-only directory")
}
r, f, _ := unitFailedFixture(t, "[database\n")
dir := filepath.Dir(r.statePath)
if err := os.Chmod(dir, 0o500); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { os.Chmod(dir, 0o700) })
start := r.now
for i := 0; i <= 8; i++ {
r.now = start.Add(time.Duration(i) * 2 * time.Minute)
var stdout, stderr bytes.Buffer
if code := watchdogUnitFailed(r, &stdout, &stderr); code != 1 || !strings.Contains(stderr.String(), "; kept in "+r.fallbackPath) {
t.Fatalf("run %d: exit %d, stderr %s; want exit 1 and the state kept in the fallback", i, code, stderr.String())
}
if want := min(max(i-4, 0), 1); len(f.sent) != want {
t.Fatalf("run %d, %v after the first failure: mailed %d, want %d", i, r.now.Sub(start), len(f.sent), want)
}
}
}
// TestWatchdogUnitFailedConfigRelay: with felis.toml loading, the relay is
// the configured one and signs in with the env var password_ref names.
func TestWatchdogUnitFailedConfigRelay(t *testing.T) {
r, f, _ := unitFailedFixture(t, testWatchdogConfig+testWatchdogSMTP)
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
start := r.now
for _, at := range []time.Duration{0, 10 * time.Minute} {
r.now = start.Add(at)
var stdout, stderr bytes.Buffer
if code := watchdogUnitFailed(r, &stdout, &stderr); code != 0 {
t.Fatalf("exit %d (stdout %s, stderr %s)", code, stdout.String(), stderr.String())
}
}
if len(f.relays) != 1 || f.relays[0].Host != "smtp.config.example" || f.relays[0].Port != 2525 || f.relays[0].Password != "env-pw" {
t.Fatalf("relays %+v, want the configured one with the env password", f.relays)
}
}
// TestWatchdogUnitFailedHeld: in the installer's quiet window the alert waits
// and no failure is pinged; a host standing by for the off-site bucket pings
// nothing either.
func TestWatchdogUnitFailedHeld(t *testing.T) {
r, f, log := unitFailedFixture(t, "[database\n")
writeTestFile(t, r.quietPath, fmt.Sprintf("%d\n", r.now.Add(time.Hour).Unix()), 0o644)
start := r.now
for _, at := range []time.Duration{0, 10 * time.Minute} {
r.now = start.Add(at)
watchdogUnitFailed(r, io.Discard, io.Discard)
}
if pings, _ := log.got(); len(f.sent) != 0 || len(pings) != 0 {
t.Fatalf("quiet window: mailed %v, pinged %v", f.sent, pings)
}
os.Remove(r.quietPath)
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: r.now.Add(-time.Hour)}
if err := offsite.WriteStatus(r.offsiteStatus, offsite.Status{Standby: true, Writer: w}); err != nil {
t.Fatal(err)
}
var stdout bytes.Buffer
watchdogUnitFailed(r, &stdout, io.Discard)
if pings, _ := log.got(); len(f.sent) != 0 || len(pings) != 0 || !strings.Contains(stdout.String(), "stands by for host prod-1") {
t.Fatalf("standby: mailed %v, pinged %v, stdout %s", f.sent, pings, stdout.String())
}
}
// TestWatchdogUnitFailedStateUnreadable: with no state to mail from, the
// failure still reaches the heartbeat.
func TestWatchdogUnitFailedStateUnreadable(t *testing.T) {
r, f, log := unitFailedFixture(t, "[database\n")
writeTestFile(t, r.statePath, "{", 0o600)
if code := watchdogUnitFailed(r, io.Discard, io.Discard); code != 1 {
t.Fatalf("exit %d, want 1", code)
}
pings, bodies := log.got()
if len(f.sent) != 0 || len(pings) != 1 || pings[0] != "POST /check-key/fail" || !strings.Contains(bodies[0], "state does not load") {
t.Fatalf("mailed %v, pinged %v %q", f.sent, pings, bodies)
}
}
func TestFailureDetail(t *testing.T) {
for _, tc := range [][3]string{
{"exit-code", "1", "result exit-code, exit status 1"},
{"timeout", "", "result timeout"},
{"", "", "systemd named no cause (journalctl -u felis-watchdog -n 50)"},
} {
if got := failureDetail(tc[0], tc[1]); got != tc[2] {
t.Errorf("failureDetail(%q, %q) = %q, want %q", tc[0], tc[1], got, tc[2])
}
}
}
// TestWatchdogRunHeartbeat runs whole watchdog passes with the API server and
// PostgreSQL down and the relay a recorder: a pass whose alerts reach the
// owners pings success; one whose mail fails, or that has no relay while an
// alert is open, posts its report to /fail; a host standing by for the
// off-site bucket pings nothing; a state file that does not parse is moved
// aside and reported.
func TestWatchdogRunHeartbeat(t *testing.T) {
due := func(statePath string) {
// PostgreSQL has been down for an hour and nobody was told yet.
s := &watchdog.State{Recipients: []string{"[email protected]"}, SMTPPassword: "cached-pw"}
s.Observe(watchdog.Report{Findings: []watchdog.Finding{watchdog.PostgresDown(errors.New("refused"))}}, time.Now().Add(-time.Hour))
if err := watchdog.SaveState(statePath, s); err != nil {
t.Fatal(err)
}
}
open := func(statePath string) {
// The owners were told of PostgreSQL half an hour ago.
s := &watchdog.State{Recipients: []string{"[email protected]"}}
at := time.Now().Add(-30 * time.Minute)
r := watchdog.Report{Findings: []watchdog.Finding{watchdog.PostgresDown(errors.New("refused"))}}
s.Observe(r, at.Add(-time.Hour))
s.Commit(s.Observe(r, at), at)
if err := watchdog.SaveState(statePath, s); err != nil {
t.Fatal(err)
}
}
const offsiteTable = "[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis\"\n"
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
for _, tc := range []struct {
what string
cfg string
state func(path string)
standby bool
quiet bool
refused bool
wantCode int
want string // the ping, "" for none
wantBody string
wantOut string
wantSent int
}{
{what: "nothing due", cfg: testWatchdogConfig + testWatchdogSMTP, want: "GET /check-key"},
{what: "mailed", cfg: testWatchdogConfig + testWatchdogSMTP, state: due, want: "GET /check-key", wantSent: 1},
{what: "the mail fails", cfg: testWatchdogConfig + testWatchdogSMTP, state: due, refused: true, wantCode: 1, want: "POST /check-key/fail", wantBody: "the alerts reach no one: the alert mail failed: [email protected]: 550 refused"},
{what: "no relay, an alert open", cfg: testWatchdogConfig, state: due, want: "POST /check-key/fail", wantBody: "the alerts reach no one: no [smtp] relay is configured"},
{what: "no relay, an alert open, the installer running", cfg: testWatchdogConfig, state: open, quiet: true, wantOut: "withholding the failure ping"},
{what: "standing by for the off-site bucket", cfg: testWatchdogConfig + testWatchdogSMTP + offsiteTable, state: due, standby: true, wantOut: "stands by for the off-site bucket"},
{what: "a state that does not parse", cfg: testWatchdogConfig, state: func(p string) { writeTestFile(t, p, "{", 0o600) }, want: "GET /check-key", wantOut: "[warning] watchdog/state: the watchdog's state file was unreadable and was moved to "},
} {
dir := t.TempDir()
t.Setenv("KUBECONFIG", filepath.Join(dir, "no-kubeconfig"))
log, srv := newPingServer(t)
beatFile := filepath.Join(dir, "watchdog-heartbeat-url")
writeTestFile(t, beatFile, srv.URL+"/check-key\n", 0o600)
cfgPath := filepath.Join(dir, "felis.toml")
writeTestFile(t, cfgPath, tc.cfg, 0o600)
statePath := filepath.Join(dir, "state.json")
if tc.state != nil {
tc.state(statePath)
}
statusPath := filepath.Join(dir, "offsite-status.json")
if tc.standby {
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: time.Now().Add(-time.Hour)}
if err := offsite.WriteStatus(statusPath, offsite.Status{Standby: true, Writer: w}); err != nil {
t.Fatal(err)
}
}
if tc.quiet {
writeTestFile(t, filepath.Join(dir, "quiet"), fmt.Sprintf("%d\n", time.Now().Add(time.Hour).Unix()), 0o644)
}
rec := &alertRecorder{}
if tc.refused {
rec.fail = map[string]bool{"[email protected]": true}
}
watchdogSender = rec.sender
var stdout, stderr bytes.Buffer
code := cmdWatchdog([]string{
"-config", cfgPath, "-state", statePath, "-quiet-file", filepath.Join(dir, "quiet"),
"-backup-dir", "", "-disk-paths", dir, "-smtp-password-file", filepath.Join(dir, "smtp-password"),
"-offsite-status", statusPath, "-heartbeat-file", beatFile,
}, &stdout, &stderr)
watchdogSender = smtpSender
pings, bodies := log.got()
fail := code != tc.wantCode || len(rec.sent) != tc.wantSent || !strings.Contains(stdout.String(), tc.wantOut)
if tc.want == "" {
fail = fail || len(pings) != 0
} else {
fail = fail || len(pings) != 1 || pings[0] != tc.want || !strings.Contains(bodies[0], tc.wantBody)
}
if fail {
t.Errorf("%s: exit %d, mailed %v, pings %v %q; want exit %d, %d mails, %q with %q\nstdout %s\nstderr %s", tc.what, code, rec.sent, pings, bodies, tc.wantCode, tc.wantSent, tc.want, tc.wantBody, stdout.String(), stderr.String())
}
if tc.wantSent > 0 {
if got := *rec.relays[0]; got.Host != "smtp.config.example" || got.Port != 2525 || got.Password != "env-pw" {
t.Errorf("%s: relay %+v, want the configured one with the env password", tc.what, got)
}
if s, err := watchdog.LoadState(statePath); err != nil || s.Relay == nil || s.Relay.Host != "smtp.config.example" {
t.Errorf("%s: the state caches relay %+v (%v), want the configured one", tc.what, s.Relay, err)
}
}
if strings.Contains(tc.wantOut, "watchdog/state") {
if aside, _ := filepath.Glob(statePath + ".unreadable-*"); len(aside) != 1 {
t.Errorf("%s: moved aside %v, want one file", tc.what, aside)
}
}
}
}
-289
View File
@@ -1,20 +1,10 @@
package main
import (
"bytes"
"context"
"errors"
"fmt"
"io/fs"
"net"
"os"
"path/filepath"
"strings"
"testing"
"time"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/watchdog"
)
// TestProxyFinding: a listening proxy is healthy; a closed port is the critical
@@ -50,282 +40,3 @@ func TestSplitList(t *testing.T) {
t.Fatalf("splitList = %q", got)
}
}
// TestMailHold: mail waits through the installer's quiet window, and on a host
// standing by for another host's off-site bucket while that host writes it.
func TestMailHold(t *testing.T) {
dir := t.TempDir()
quiet, status := filepath.Join(dir, "quiet"), filepath.Join(dir, "status.json")
now := time.Now()
if got := mailHold(quiet, true, status, now); got != "" {
t.Errorf("no quiet file, no status: %q", got)
}
writeTestFile(t, quiet, fmt.Sprintf("%d\n", now.Add(time.Hour).Unix()), 0o644)
if got := mailHold(quiet, false, status, now); !strings.Contains(got, "quiet until") {
t.Errorf("inside the quiet window: %q", got)
}
os.Remove(quiet)
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: now.Add(-40 * time.Minute)}
for _, tc := range []struct {
what string
st offsite.Status
offsiteOn bool
held bool
}{
{"standing by for a live writer", offsite.Status{Standby: true, Writer: w}, true, true},
{"standing by, [offsite] since removed", offsite.Status{Standby: true, Writer: w}, false, false},
{"standing by for a writer gone quiet", offsite.Status{Standby: true, Writer: &offsite.Writer{HostID: w.HostID, Host: w.Host, At: now.Add(-offsite.WriterLive - time.Minute)}}, true, false},
{"standing by, no writer named", offsite.Status{Standby: true}, true, false},
{"displaced", offsite.Status{Displaced: true, Writer: w}, true, false},
{"the writer itself", offsite.Status{LastSuccess: now}, true, false},
} {
if err := offsite.WriteStatus(status, tc.st); err != nil {
t.Fatal(err)
}
got := mailHold(quiet, tc.offsiteOn, status, now)
if (got != "") != tc.held || (tc.held && !strings.Contains(got, "stands by for host prod-1 (id bbbbbbbbbbbbbbbb)")) {
t.Errorf("%s: hold = %q, want held %v", tc.what, got, tc.held)
}
}
writeTestFile(t, status, "{", 0o600)
if got := mailHold(quiet, true, status, now); got != "" {
t.Errorf("an unreadable status held the mail: %q", got)
}
}
// watchdogHost is a host whole watchdog passes run on: the API server and PostgreSQL
// are down (no kubeconfig, nothing on the database port), the relay is a recorder and
// the heartbeat URL points at a ping log.
type watchdogHost struct {
dir, statePath string
// fallbackPath stands in for /run/felis, in a directory of its own.
fallbackPath string
args []string
rec *alertRecorder
pings *pingLog
url string
}
func newWatchdogHost(t *testing.T, cfg string, state *watchdog.State) *watchdogHost {
t.Helper()
dir := t.TempDir()
t.Setenv("KUBECONFIG", filepath.Join(dir, "no-kubeconfig"))
pings, srv := newPingServer(t)
h := &watchdogHost{dir: dir, statePath: filepath.Join(dir, "state.json"), fallbackPath: filepath.Join(t.TempDir(), "watchdog-state.json"),
rec: &alertRecorder{}, pings: pings, url: srv.URL}
writeTestFile(t, filepath.Join(dir, "felis.toml"), cfg, 0o600)
writeTestFile(t, filepath.Join(dir, "watchdog-heartbeat-url"), srv.URL+"/check-key\n", 0o600)
if state != nil {
if err := watchdog.SaveState(h.statePath, state); err != nil {
t.Fatal(err)
}
}
h.args = []string{
"-config", filepath.Join(dir, "felis.toml"), "-state", h.statePath, "-fallback-state", h.fallbackPath, "-quiet-file", filepath.Join(dir, "quiet"),
"-backup-dir", "", "-disk-paths", dir, "-k3s-cert-dirs", "", "-smtp-password-file", filepath.Join(dir, "smtp-password"),
"-offsite-status", filepath.Join(dir, "offsite-status.json"), "-build-tools-status", filepath.Join(dir, "build-tools.json"),
"-heartbeat-file", filepath.Join(dir, "watchdog-heartbeat-url"),
}
return h
}
// run is one pass; flags given here override the host's own.
func (h *watchdogHost) run(flags ...string) (code int, stdout, stderr string) {
watchdogSender = h.rec.sender
defer func() { watchdogSender = smtpSender }()
var out, errOut bytes.Buffer
code = cmdWatchdog(append(append([]string(nil), h.args...), flags...), &out, &errOut)
return code, out.String(), errOut.String()
}
// duePostgres is a state whose owner has not yet been told of PostgreSQL, down
// for an hour.
func duePostgres(cachedPassword string) *watchdog.State {
s := &watchdog.State{Recipients: []string{"[email protected]"}, SMTPPassword: cachedPassword}
s.Observe(watchdog.Report{Findings: []watchdog.Finding{watchdog.PostgresDown(errors.New("refused"))}}, time.Now().Add(-time.Hour))
return s
}
// TestWatchdogRunKeepsClusterAlertsWhileTheAPIIsDown: with the API server down,
// the alerts under the cluster checks keep their state, since nothing looked
// at them; an alert of a check that did run and found nothing reads as cleared.
func TestWatchdogRunKeepsClusterAlertsWhileTheAPIIsDown(t *testing.T) {
told := time.Now().Add(-30 * time.Minute)
cluster := []string{"deployment/felis-api", "node/felis-1/NotReady", "server-failed/lobby"}
var r watchdog.Report
for _, key := range append([]string{"proxy"}, cluster...) {
r.Findings = append(r.Findings, watchdog.Finding{Key: key, Severity: watchdog.Critical, SummaryEN: key + " is down"})
}
s := &watchdog.State{Recipients: []string{"[email protected]"}}
s.Commit(s.Observe(r, told), told)
h := newWatchdogHost(t, testWatchdogConfig, s)
code, stdout, stderr := h.run()
if code != 0 || len(h.rec.relays) != 0 {
t.Fatalf("exit %d, mailed %v\nstdout %s\nstderr %s", code, h.rec.sent, stdout, stderr)
}
for _, want := range []string{"felis watchdog: [critical] kube-api: ", "felis watchdog: [critical] postgres: "} {
if !strings.Contains(stdout, want) {
t.Errorf("stdout lacks %q:\n%s", want, stdout)
}
}
got, err := watchdog.LoadState(h.statePath)
if err != nil {
t.Fatal(err)
}
for _, key := range cluster {
if a := got.Alerts[key]; a == nil || !a.ClearedAt.IsZero() || !a.Notified.Equal(told) {
t.Errorf("%s with the API server down: %+v, want it kept open as mailed at %s", key, a, told)
}
}
if a := got.Alerts["proxy"]; a == nil || a.ClearedAt.IsZero() {
t.Errorf("proxy, whose check ran and found nothing: %+v, want it cleared", a)
}
}
// TestWatchdogDryRunMailsAndSavesNothing: a dry run prints the mail that is due
// and the heartbeat it would ping, and sends, pings and saves nothing.
func TestWatchdogDryRunMailsAndSavesNothing(t *testing.T) {
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
h := newWatchdogHost(t, testWatchdogConfig+testWatchdogSMTP, duePostgres(""))
before, err := os.ReadFile(h.statePath)
if err != nil {
t.Fatal(err)
}
code, stdout, stderr := h.run("-dry-run")
after, err := os.ReadFile(h.statePath)
if err != nil {
t.Fatal(err)
}
pings, _ := h.pings.got()
if code != 0 || len(h.rec.relays) != 0 || len(pings) != 0 || !bytes.Equal(before, after) {
t.Errorf("dry run: exit %d, relays %d, pings %v, state changed %v\nstdout %s\nstderr %s", code, len(h.rec.relays), pings, !bytes.Equal(before, after), stdout, stderr)
}
host, _ := os.Hostname()
for _, want := range []string{
"felis watchdog: due to be mailed to [email protected]:\nSubject: Felis 严重告警(" + host + "):1 项异常 · 1 firing\n\n",
"felis watchdog: a run pings the heartbeat at " + h.url + "/...\n",
} {
if !strings.Contains(stdout, want) {
t.Errorf("stdout lacks %q:\n%s", want, stdout)
}
}
if strings.Contains(stdout, "\r") {
t.Errorf("the mail body printed with CRLF:\n%q", stdout)
}
h = newWatchdogHost(t, testWatchdogConfig, nil)
code, stdout, stderr = h.run("-dry-run")
if code != 0 || !strings.Contains(stdout, "felis watchdog: nothing is due to be mailed\n") {
t.Errorf("dry run with nothing due: exit %d\nstdout %s\nstderr %s", code, stdout, stderr)
}
if _, err := os.Stat(h.statePath); !errors.Is(err, fs.ErrNotExist) {
t.Errorf("a dry run wrote the state (%v)", err)
}
}
// TestWatchdogRunSignsInWithTheHostRelayPassword: a pass caches the relay
// password from the host copy even while the cluster is down, and signs in with
// it; with no host copy and no cluster it keeps the one it had.
func TestWatchdogRunSignsInWithTheHostRelayPassword(t *testing.T) {
const smtpNoRef = "[smtp]\nhost = \"smtp.config.example\"\nport = 2525\nfrom = \"[email protected]\"\nusername = \"felis\"\n"
for _, tc := range []struct{ what, hostCopy, want string }{
{"the host copy", "host-pw", "host-pw"},
{"no host copy, the cluster down", "", "cached-pw"},
} {
h := newWatchdogHost(t, testWatchdogConfig+smtpNoRef, duePostgres("cached-pw"))
if tc.hostCopy != "" {
writeTestFile(t, filepath.Join(h.dir, "smtp-password"), tc.hostCopy, 0o600)
}
code, stdout, stderr := h.run()
if code != 0 || len(h.rec.relays) != 1 || h.rec.relays[0].Password != tc.want {
t.Errorf("%s: exit %d, relays %+v; want one mail signed in with %q\nstdout %s\nstderr %s", tc.what, code, h.rec.relays, tc.want, stdout, stderr)
continue
}
if s, err := watchdog.LoadState(h.statePath); err != nil || s.SMTPPassword != tc.want {
t.Errorf("%s: the state caches another password (%v)", tc.what, err)
}
}
}
// TestWatchdogRunChecksWhatIsConfigured: the proxy, the database backups, the
// off-site copy and the build lane's scan DB are each checked when the host
// has them, and only then.
func TestWatchdogRunChecksWhatIsConfigured(t *testing.T) {
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
closed := ln.Addr().String()
ln.Close()
const offsiteTable = "[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis\"\n"
const registry = "[registry]\nurl = \"registry.felis.svc:5000\"\n"
optional := []string{"proxy", "db-backup", "offsite", "scan-db"}
for _, tc := range []struct {
what string
cfg string
flags func(dir string) []string
want string // the one optional check that reports, "" for none
}{
{what: "none of them", cfg: testWatchdogConfig},
{what: "a proxy address", cfg: testWatchdogConfig, flags: func(string) []string { return []string{"-proxy-addr", closed} }, want: "proxy"},
{what: "a backup directory", cfg: testWatchdogConfig, flags: func(dir string) []string { return []string{"-backup-dir", dir} }, want: "db-backup"},
{what: "[offsite]", cfg: testWatchdogConfig + offsiteTable, want: "offsite"},
{what: "a registry with the default scan DB", cfg: testWatchdogConfig + registry, want: "scan-db"},
{what: "a registry with its scan DB under mirror/", cfg: testWatchdogConfig + registry + "trivy_db_repository = \"registry.felis.svc:5000/mirror/trivy-db\"\n", want: "scan-db"},
{what: "a registry with the scan DB elsewhere", cfg: testWatchdogConfig + registry + "trivy_db_repository = \"ghcr.io/aquasecurity/trivy-db\"\n"},
} {
h := newWatchdogHost(t, tc.cfg, nil)
var flags []string
if tc.flags != nil {
flags = tc.flags(t.TempDir())
}
code, stdout, stderr := h.run(append(flags, "-dry-run")...)
var reported []string
for _, key := range optional {
if strings.Contains(stdout, "] "+key+": ") {
reported = append(reported, key)
}
}
if code != 0 || strings.Join(reported, ",") != tc.want {
t.Errorf("%s: exit %d, reported %v; want %q\nstdout %s\nstderr %s", tc.what, code, reported, tc.want, stdout, stderr)
}
}
}
// TestWatchdogRunStateThatDoesNotSave: a pass whose state does not save mails
// as usual, keeps its state in the fallback, exits 1 and posts the failure to
// the heartbeat. The next pass reads the fallback and mails nothing again; the
// first one whose state saves drops the fallback.
func TestWatchdogRunStateThatDoesNotSave(t *testing.T) {
if os.Geteuid() == 0 {
t.Skip("root writes into a read-only directory")
}
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
h := newWatchdogHost(t, testWatchdogConfig+testWatchdogSMTP, duePostgres(""))
if err := os.Chmod(h.dir, 0o500); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { os.Chmod(h.dir, 0o700) })
kept := "; kept in " + h.fallbackPath + " until the host restarts"
for i := range 2 {
code, stdout, stderr := h.run()
pings, bodies := h.pings.got()
if code != 1 || len(h.rec.sent) != 1 || !strings.Contains(stderr, "felis watchdog: save state: ") || !strings.Contains(stderr, kept) ||
len(pings) != i+1 || pings[i] != "POST /check-key/fail" || !strings.HasPrefix(bodies[i], "the watchdog state did not save: ") {
t.Fatalf("pass %d: exit %d, mailed %v, pings %v %q; want exit 1, one mail in all and the save failure posted to /fail\nstdout %s\nstderr %s", i, code, h.rec.sent, pings, bodies, stdout, stderr)
}
}
if err := os.Chmod(h.dir, 0o700); err != nil {
t.Fatal(err)
}
code, stdout, stderr := h.run()
if _, err := os.Stat(h.fallbackPath); code != 0 || len(h.rec.sent) != 1 || !errors.Is(err, fs.ErrNotExist) {
t.Fatalf("once the state saves: exit %d, mailed %v, fallback %v; want exit 0, no new mail, the fallback gone\nstdout %s\nstderr %s", code, h.rec.sent, err, stdout, stderr)
}
if s, err := watchdog.LoadState(h.statePath); err != nil || s.Alerts["postgres"] == nil || s.Alerts["postgres"].Notified.IsZero() {
t.Fatalf("saved state = %+v, %v; want the mailed PostgreSQL alert", s, err)
}
}
+422 -2170
View File
File diff suppressed because it is too large. Load diff
+182 -2759
View File
File diff suppressed because it is too large. Load diff
-198
View File
@@ -1,198 +0,0 @@
#!/usr/bin/env bash
# Builds everything a Felis release installs, for every architecture, into one directory:
#
# felis-linux-<arch> the felis binary, panel included
# felis-image-felis-linux-<arch>.tar the control-plane image (OCI layout tars:
# felis-image-game-linux-<arch>.tar the limbo, lobby and paper images `ctr images import`
# felis-image-base-linux-<arch>.tar the registry and PostgreSQL reads them as-is)
# felis-images-linux-<arch>.txt one "bundle role name manifest-digest config-digest"
# line per image in the three tars
# felis-velocity.jar the proxy plugin (JVM bytecode, one for every arch)
# SHA256SUMS the sha256 of every file above
#
# Every name above is a contract with deploy/bootstrap.sh, which installs from a release's
# assets (or from a directory like this one, FELIS_ARTIFACT_DIR) instead of building on the
# host: with them a host needs neither Docker, Gradle, Go nor Docker Hub. The images are split
# in three because they change at different rates: the control plane with every release, the
# game images when game-stack.lock or a plugin changes, the base images almost never. An
# upgrade downloads only the tars holding an image the host does not have yet.
#
# .github/workflows/release.yml runs this on a tag and publishes the directory; e2e.yml runs
# it on a branch and installs from the directory.
#
# Needs docker with buildx, and binfmt/QEMU for the architectures other than the builder's:
# the game images' runtime stages run apt-get on the target platform. The builder's own felis
# binary writes the image tars, so Go is not needed here either.
#
# Usage: deploy/build-release-artifacts.sh <version> <out-dir>
# FELIS_RELEASE_ARCHES the architectures to build (default "amd64 arm64")
set -Eeuo pipefail
log() { printf '\033[1;36m[release]\033[0m %s\n' "$*"; }
die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; }
[ "$#" -eq 2 ] || die "usage: $0 <version> <out-dir>"
VERSION="$1"
OUT="$2"
ARCHES="${FELIS_RELEASE_ARCHES:-amd64 arm64}"
cd "$(dirname "$0")/.."
# The Gradle image the plugin builds run in: deploy/{limbo,lobby}/Dockerfile and bootstrap's
# Velocity build name the same one (bootstrap_asset_test.go holds them together).
PLUGIN_BUILD_IMAGE="gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01"
# The names and pins bootstrap uses, read from bootstrap itself so the two cannot drift.
bootstrap_value() {
local v
v="$(sed -n "s/^$1=\"\(.*\)\"\$/\1/p" deploy/bootstrap.sh)"
[ -n "$v" ] || die "deploy/bootstrap.sh sets no $1"
printf '%s' "$v"
}
REGISTRY_URL="$(bootstrap_value REGISTRY_URL)"
REGISTRY_IMAGE="$(bootstrap_value REGISTRY_IMAGE)"
POSTGRES_IMAGE="$(bootstrap_value POSTGRES_IMAGE)"
# The final stage of the repo Dockerfile, the base every felis image runs on.
FELIS_BASE_IMAGE="$(awk '$1 == "FROM" && $2 ~ /^gcr\.io\/distroless\// { print $2 }' Dockerfile)"
[ -n "$FELIS_BASE_IMAGE" ] || die "the Dockerfile names no distroless base"
# lock_value reads one KEY=value line of deploy/game-stack.lock, which bootstrap reads the
# same way: never sourced, values limited to URL and version characters.
lock_value() {
local v
v="$(awk -F= -v k="$1" '$1 == k { print substr($0, length(k) + 2); exit }' deploy/game-stack.lock)"
case "$v" in
''|*[!A-Za-z0-9._:/+%-]*) die "deploy/game-stack.lock: $1 is missing or malformed" ;;
esac
printf '%s' "$v"
}
# image_tag_for_version is bootstrap's: the tag bootstrap names this release's image with.
image_tag_for_version() {
local v
v="$(printf '%s' "$1" | tr -c 'A-Za-z0-9_.-' '-')"
case "$v" in
""|dev|[!A-Za-z0-9_]*) printf 'demo' ;;
*) printf '%s' "${v:0:128}" ;;
esac
}
host_arch() {
case "$(uname -m)" in
x86_64|amd64) printf 'amd64' ;;
aarch64|arm64) printf 'arm64' ;;
*) die "unsupported builder architecture $(uname -m)" ;;
esac
}
mkdir -p "$OUT"
[ -z "$(ls -A "$OUT")" ] || die "${OUT} is not empty; SHA256SUMS must describe exactly what one run built"
WORK="$(mktemp -d)"
trap 'rm -rf "$WORK"' EXIT
platforms=""
for arch in $ARCHES; do
case "$arch" in amd64|arm64) ;; *) die "unsupported architecture ${arch}" ;; esac
platforms="${platforms:+${platforms},}linux/${arch}"
done
# ---- the felis binaries --------------------------------------------------------------
# Through the repo Dockerfile, the one recipe that runs the panel build before go build:
# a plain `go build` compiles against the placeholder panel and ships it.
log "building felis ${VERSION} for ${platforms}"
docker buildx build --platform "$platforms" \
--build-arg FELIS_VERSION="$VERSION" \
--output "type=local,dest=${WORK}/bin" .
for arch in $ARCHES; do
src="${WORK}/bin/usr/local/bin/felis"
# buildx nests the output per platform only when it builds more than one.
[ -f "${WORK}/bin/linux_${arch}/usr/local/bin/felis" ] && src="${WORK}/bin/linux_${arch}/usr/local/bin/felis"
install -m 0755 "$src" "${OUT}/felis-linux-${arch}"
# By ELF machine, never by running it: binfmt would run the wrong architecture happily.
case "$arch" in
amd64) want='x86-64' ;;
arm64) want='ARM aarch64' ;;
esac
file "${OUT}/felis-linux-${arch}" | grep -q "$want" \
|| die "felis-linux-${arch} is not a ${want} ELF: TARGETARCH did not reach the go build"
done
HOST_FELIS="${OUT}/felis-linux-$(host_arch)"
[ -x "$HOST_FELIS" ] || die "no felis binary for the builder's own architecture ($(host_arch)); add it to FELIS_RELEASE_ARCHES"
got="$("$HOST_FELIS" version | head -n 1)"
[ "$got" = "felis ${VERSION}" ] || die "felis reports '${got}', want 'felis ${VERSION}': the version stamp did not reach the binary"
# ---- the Velocity plugin -------------------------------------------------------------
# The command bootstrap's source path runs, on a copy of the tree so the checkout stays clean.
log "building felis-velocity.jar in ${PLUGIN_BUILD_IMAGE%%@*}"
mkdir -p "${WORK}/velocity"
tar -C . --exclude=build --exclude=.gradle -cf - plugins/velocity plugins/shared | tar -C "${WORK}/velocity" -xf -
docker run --rm --user "$(id -u):$(id -g)" -e HOME=/tmp -e GRADLE_USER_HOME=/tmp/gradle \
-v "${WORK}/velocity:/src" -w /src/plugins/velocity \
"$PLUGIN_BUILD_IMAGE" gradle --no-daemon clean build
jars=( "${WORK}"/velocity/plugins/velocity/build/libs/felis-velocity-*.jar )
[ "${#jars[@]}" -eq 1 ] && [ -f "${jars[0]}" ] || die "the felis-velocity build must produce exactly one plugin jar"
install -m 0644 "${jars[0]}" "${OUT}/felis-velocity.jar"
# ---- the images ----------------------------------------------------------------------
FELIS_IMAGE="${REGISTRY_URL}/felis/felis:$(image_tag_for_version "$VERSION")"
game_build_args=(
--build-arg "LIMBO_JAR_URL=$(lock_value LIMBO_JAR_URL)"
--build-arg "LIMBO_JAR_SHA256=$(lock_value LIMBO_JAR_SHA256)"
--build-arg "LIMBO_SCHEM_URL=$(lock_value LIMBO_SCHEM_URL)"
--build-arg "LIMBO_SCHEM_SHA256=$(lock_value LIMBO_SCHEM_SHA256)"
--build-arg "LIMBO_VERSION=$(lock_value LIMBO_VERSION)"
--build-arg "PAPER_JAR_URL=$(lock_value PAPER_JAR_URL)"
--build-arg "PAPER_JAR_SHA256=$(lock_value PAPER_JAR_SHA256)"
--build-arg "LUCKPERMS_JAR_URL=$(lock_value LUCKPERMS_JAR_URL)"
--build-arg "LUCKPERMS_JAR_SHA256=$(lock_value LUCKPERMS_JAR_SHA256)"
)
# oci_image <arch> <dest> <docker build args...> builds one image for one platform into an
# OCI layout tar. No attestations: the bundle carries images only.
oci_image() {
local arch="$1" dest="$2"
shift 2
docker buildx build --platform "linux/${arch}" --provenance=false --sbom=false \
--output "type=oci,dest=${dest}" "$@"
}
# bundle <arch> <group> <image-bundle flags...> writes one image tar and appends its lines
# to the architecture's listing.
bundle() {
local arch="$1" group="$2" name
shift 2
name="felis-image-${group}-linux-${arch}.tar"
"$HOST_FELIS" image-bundle --platform "linux/${arch}" \
--out "${OUT}/${name}" --list "${WORK}/${name}.txt" "$@"
awk -v b="$name" '{ print b, $0 }' "${WORK}/${name}.txt" >> "${OUT}/felis-images-linux-${arch}.txt"
}
for arch in $ARCHES; do
# The control-plane image wraps the released binary itself, byte for byte, the way
# bootstrap wraps a downloaded binary.
log "building the linux/${arch} images"
mkdir -p "${WORK}/felis-${arch}"
cp "${OUT}/felis-linux-${arch}" "${WORK}/felis-${arch}/felis"
cat > "${WORK}/felis-${arch}/Dockerfile" <<EOF
FROM ${FELIS_BASE_IMAGE}
ENV PATH=/usr/local/bin:/usr/bin:/bin
COPY --chmod=0755 felis /usr/local/bin/felis
USER 65532:65532
ENTRYPOINT ["/usr/local/bin/felis"]
EOF
oci_image "$arch" "${WORK}/felis-${arch}.tar" "${WORK}/felis-${arch}"
for role in limbo lobby paper; do
oci_image "$arch" "${WORK}/${role}-${arch}.tar" -f "deploy/${role}/Dockerfile" "${game_build_args[@]}" .
done
bundle "$arch" felis --layout "felis=${FELIS_IMAGE}=${WORK}/felis-${arch}.tar"
bundle "$arch" game \
--layout "limbo=${REGISTRY_URL}/felis/limbo:demo=${WORK}/limbo-${arch}.tar" \
--layout "lobby=${REGISTRY_URL}/felis/lobby:demo=${WORK}/lobby-${arch}.tar" \
--layout "paper=${REGISTRY_URL}/felis/paper:demo=${WORK}/paper-${arch}.tar"
bundle "$arch" base --pull "registry=${REGISTRY_IMAGE}" --pull "postgres=${POSTGRES_IMAGE}"
rm -f "${WORK}"/*-"${arch}".tar
done
log "writing SHA256SUMS"
(cd "$OUT" && sha256sum -- *) > "${WORK}/SHA256SUMS"
mv "${WORK}/SHA256SUMS" "${OUT}/SHA256SUMS"
ls -l "$OUT"
@@ -370,10 +370,8 @@ spec:
description: |-
EmptySince is when the operator first observed 0 online players during
a Running phase (spec §8 idle auto-stop). It is reset when a player joins
or the server restarts or stops, so the empty-duration counter starts fresh
each time the server becomes unoccupied. A zero sampled within three minutes
of a run's first ready probe stamps nothing: the players it came up for may
not be in yet.
or the server stops, so the empty-duration counter starts fresh each time
the server becomes unoccupied.
format: date-time
type: string
endpoint:
@@ -422,9 +420,7 @@ spec:
status ping is never sufficient — spec §5).
type: boolean
readySignalAt:
description: |-
ReadySignalAt is when the first RCON probe of the current run succeeded.
Every Starting or Stopping pass clears it, so each start is measured once.
description: ReadySignalAt is when the first RCON probe succeeded.
format: date-time
type: string
startRequestedAt:
@@ -437,14 +433,6 @@ spec:
because the two endpoints fall in different reconcile passes.
format: date-time
type: string
stopNoticeAt:
description: |-
StopNoticeAt is when the operator told the players on a server that it is
about to stop (desiredState flipped to Stopped with players online). The stop
itself waits until StopNoticeWindow has passed since then; the stamp is cleared
once the server is scaled down, or when desiredState goes back to Running first.
format: date-time
type: string
type: object
type: object
served: true
+10 -124
View File
@@ -7,12 +7,10 @@
# sudo bash deploy/e2e_check.sh release # after the newest release installed
# sudo bash deploy/e2e_check.sh upgrade # after this commit ran over a release
#
# It asks what an operator's first minutes ask: the binary runs, the control plane and its
# database are rolled out and ready, a database backup can be taken and restored, the panel
# answers on its NodePort, the proxy answers a Minecraft status ping, the host timers are
# all there and waiting, and the backup and watchdog runs the installer made succeeded.
# A rerun must also leave the proxy running (it restarts only when what it runs changed)
# and keep every earlier answer.
# It asks what an operator's first minutes ask: the binary runs, the control plane is
# rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft
# status ping, and the host timers are there. A rerun must also leave the proxy running
# (it restarts only when what it runs changed) and keep every earlier answer.
set -euo pipefail
phase="${1:?usage: e2e_check.sh install|rerun|release|upgrade}"
@@ -31,12 +29,7 @@ check() { # label command...
check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "'
deploys=(felis-api felis-operator registry)
# A release may still run the database on the host; the upgrade moves it into felis-postgres.
if [ "$phase" != release ] || "${KUBECTL[@]}" -n felis get deploy/felis-postgres >/dev/null 2>&1; then
deploys=(felis-postgres "${deploys[@]}")
fi
for d in "${deploys[@]}"; do
for d in felis-api felis-operator registry; do
check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
done
@@ -48,120 +41,13 @@ internal="$("${KUBECTL[@]}" -n felis get svc felis-api-internal -o jsonpath='{.s
check "felis-api is ready (database and cluster reachable)" \
curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"
for unit in k3s felis-velocity; do
for unit in k3s postgresql felis-velocity; do
check "${unit} is active" systemctl is-active --quiet "$unit"
done
# pod_psql runs one statement in the database's pod, over its socket, as the felis role on
# the felis database.
pod_psql() {
"${KUBECTL[@]}" -n felis exec -i deploy/felis-postgres -c postgres -- \
psql -X -q -At -v ON_ERROR_STOP=1 -U felis -d felis -c "$1"
}
# restore_drill walks troubleshooting.md's "Restore on the same host": refused while the
# control plane is connected; with it scaled to 0 the bundle comes back (a row written after
# it is gone) and the database it replaced is kept; migrate up runs; the control plane serves
# again.
restore_drill() { # dir bundle
local dir="$1" bundle="$2" out rc
local sel="app.kubernetes.io/part-of=felis-control-plane,app.kubernetes.io/component in (api,operator)"
if ! pod_psql "INSERT INTO platform_settings (key, value) VALUES ('e2e_restore_drill', '1')" >/dev/null; then
fail "write a row after the bundle"
return
fi
rc=0
out="$(/usr/local/bin/felis db restore -dir "$dir" -yes "$bundle" 2>&1)" || rc=$?
if [ "$rc" -eq 1 ] && grep -q "other clients are connected to the database" <<<"$out"; then
pass "felis db restore refuses while the control plane is connected"
else
fail "felis db restore refuses while the control plane is connected (exit ${rc}): ${out}"
fi
"${KUBECTL[@]}" -n felis scale deployment felis-api felis-operator --replicas=0 >/dev/null
for _ in $(seq 60); do
[ -z "$("${KUBECTL[@]}" -n felis get pods -l "$sel" -o name)" ] && break
sleep 2
done
rc=0
out="$(/usr/local/bin/felis db restore -dir "$dir" -yes "$bundle" 2>&1)" || rc=$?
if [ "$rc" -eq 0 ]; then
pass "felis db restore replays the bundle with the control plane scaled to 0"
else
fail "felis db restore replays the bundle with the control plane scaled to 0 (exit ${rc}): ${out}"
fi
check "the restore dropped the row written after the bundle" \
test "$(pod_psql "SELECT count(*) FROM platform_settings WHERE key = 'e2e_restore_drill'")" = 0
check "the restore kept the database it replaced in a pre-restore bundle" \
sh -c "ls '${dir}' | grep -q -- '-pre-restore\.tar\$'"
check "felis migrate up runs on the restored database" \
/usr/local/bin/felis migrate up -config /etc/felis/felis.host.toml
"${KUBECTL[@]}" -n felis scale deployment felis-api felis-operator --replicas=1 >/dev/null
for d in felis-api felis-operator; do
check "deployment ${d} is rolled out again after the restore" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
done
check "felis-api is ready on the restored database" \
curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"
}
# The database runs in k3s; a release may still run it on the host, and the upgrade moved
# it. The host has no PostgreSQL client: a bundle that verifies proves felis reaches the
# database's pod through kubectl exec, and that pg_dump there reads every table.
# A release may predate a timer; what this commit installs has them all.
if [ "$phase" != release ]; then
check "the host's own postgresql is stopped" sh -c '! systemctl is-active --quiet postgresql'
bundle_dir="$(mktemp -d)"
if out="$(/usr/local/bin/felis db backup -dir "$bundle_dir" -state-dir "" -no-servers 2>&1)"; then
bundle="$(printf '%s\n' "$out" | sed -n 's/^felis db backup: wrote //p' | tail -n 1)"
check "felis db backup writes a bundle that verifies" /usr/local/bin/felis db verify "$bundle"
# A rerun keeps what the install left; the install and the upgrade restore it.
[ "$phase" = rerun ] || restore_drill "$bundle_dir" "$bundle"
else
fail "felis db backup writes a bundle: ${out}"
fi
rm -rf "$bundle_dir"
fi
# The daily reaper is what deletes expired archives. Run it once from its CronJob: the API
# accepts a pod template naming a ServiceAccount that does not exist, and only the Job's
# pod creation fails, so rendering it proves nothing. A release may carry exactly that bug.
if [ "$phase" != release ]; then
reaper_job="felis-e2e-reaper-${phase}"
"${KUBECTL[@]}" -n minecraft delete job "$reaper_job" --ignore-not-found >/dev/null
if "${KUBECTL[@]}" -n minecraft create job --from=cronjob/felis-reaper "$reaper_job" >/dev/null; then
check "the reaper CronJob runs to completion" \
"${KUBECTL[@]}" -n minecraft wait --for=condition=complete "job/${reaper_job}" --timeout=180s
"${KUBECTL[@]}" -n minecraft delete job "$reaper_job" --ignore-not-found >/dev/null
else
fail "a Job can be created from the reaper CronJob"
fi
fi
# A release may predate a timer. What this commit installs has every timer bootstrap.sh
# names, enabled and waiting, and no felis timer it does not name. The off-site copy's
# comes only with an [offsite] bucket, which no e2e host has.
if [ "$phase" != release ]; then
timers="$(sed -n 's|^[A-Z_]*_TIMER="/etc/systemd/system/\(felis-[a-z-]*\.timer\)"$|\1|p' "$(dirname "$0")/bootstrap.sh")"
check "bootstrap.sh names the felis timers" test -n "$timers"
for timer in $timers; do
if [ "$timer" = felis-offsite.timer ]; then
check "${timer} is not installed without an [offsite] bucket" test ! -e "/etc/systemd/system/${timer}"
continue
fi
check "${timer} is enabled" systemctl is-enabled --quiet "$timer"
check "${timer} is waiting" systemctl is-active --quiet "$timer"
done
for timer in $(systemctl list-unit-files --no-legend 'felis-*.timer' | awk '{print $1}'); do
check "${timer} is a timer bootstrap.sh installs" grep -qxF "$timer" <<<"$timers"
done
# The installer runs the backup and the watchdog once itself and only warns when that
# fails: a unit that cannot run (a flag the binary lacks, a path its sandbox hides) shows
# here. Without an [smtp] relay the watchdog has no mail to fail on.
for unit in felis-db-backup.service felis-watchdog.service; do
check "${unit} has run" test "$(systemctl show -p ExecMainStartTimestampMonotonic --value "$unit")" != 0
check "${unit}'s last run succeeded" test "$(systemctl show -p Result --value "$unit")" = success
done
# A failed watchdog run starts the unit its OnFailure= names, which must be installed.
on_failure="$(systemctl show -p OnFailure --value felis-watchdog.service)"
check "felis-watchdog.service names an OnFailure= unit" test -n "$on_failure"
for unit in $on_failure; do
check "${unit} is installed" test "$(systemctl show -p LoadState --value "$unit")" = loaded
for timer in felis-db-backup.timer felis-watchdog.timer felis-update-check.timer; do
check "${timer} is scheduled" systemctl is-enabled --quiet "$timer"
done
fi
@@ -240,5 +126,5 @@ if [ "$fails" -eq 0 ]; then
echo "ALL PASS (${phase})"
else
echo "${fails} FAILED (${phase})"
exit 1
fi
exit "$fails"
-112
View File
@@ -1,112 +0,0 @@
#!/bin/bash
# The newest published release, for the e2e workflow's readme and upgrade jobs, and what
# their installer logs must show about how that release reached the host:
#
# bash deploy/e2e_release.sh find # tag, binary, sums into $GITHUB_OUTPUT
# bash deploy/e2e_release.sh check-own LOG # the release's own installer (upgrade)
# bash deploy/e2e_release.sh check-readme LOG # this commit's installer on its default
# # channel, the README's command (readme)
#
# The checks read TAG, BINARY and SUMS from the environment, as find wrote them. The
# runners are x86_64, so the binary asset is felis-linux-amd64. deploy/e2e_release_test.sh
# holds this script's own checks, against bootstrap.sh's own messages.
set -euo pipefail
ASSET=felis-linux-amd64
HOST_BIN=/usr/local/bin/felis
fails=0
pass() { printf 'PASS %s\n' "$*"; }
fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); }
has() { # label fixed-string log
if grep -qF -- "$2" "$3"; then pass "$1"; else fail "$1: no line with <$2> in $3"; fi
}
has_re() { # label regex log
if grep -qE -- "$2" "$3"; then pass "$1"; else fail "$1: no line matching <$2> in $3"; fi
}
lacks_re() { # label regex log
local hit
if hit="$(grep -E -m 1 -- "$2" "$3")"; then fail "$1: ${hit}"; else pass "$1"; fi
}
# find_release: `gh release view` with no tag answers with the newest release that is not a
# prerelease, the one the installer's release channel resolves.
#
# An empty tag skips the readme and upgrade jobs, green. Only gh's own `release not found`
# means there is no release; any other failure (a token it refused, a rate limit, a network
# error) fails the step, or every release's upgrade would go untested without a word.
find_release() {
local lines err rc=0 tag names binary="" sums=""
err="$(mktemp)"
lines="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName,assets --jq '.tagName, .assets[].name' 2>"$err")" || rc=$?
if [ "$rc" -ne 0 ] && grep -qx 'release not found' "$err"; then
rm -f "$err"
echo "::notice::no published release yet; the readme and upgrade jobs have nothing to install"
printf 'tag=\nbinary=\nsums=\n' >> "${GITHUB_OUTPUT:-/dev/stdout}"
return 0
fi
if [ "$rc" -ne 0 ]; then
echo "::error::gh release view failed (exit ${rc}), so the newest release is unknown: $(tr '\n' ' ' < "$err")"
rm -f "$err"
fails=$((fails + 1))
return
fi
rm -f "$err"
tag="$(printf '%s\n' "$lines" | head -n 1)"
names="$(printf '%s\n' "$lines" | tail -n +2)"
if [ -z "$tag" ]; then
echo "::error::gh release view answered without a tag"
fails=$((fails + 1))
return
fi
if printf '%s\n' "$names" | grep -qxF "$ASSET"; then binary=yes; fi
if printf '%s\n' "$names" | grep -qxF SHA256SUMS; then sums=yes; fi
printf 'tag=%s\nbinary=%s\nsums=%s\n' "$tag" "$binary" "$sums" >> "${GITHUB_OUTPUT:-/dev/stdout}"
}
# check_own: a release's installer, however old, downloads the release's binary when the
# release publishes one, and says so in the same words.
check_own() {
local log="$1"
if [ "${BINARY:-}" != yes ]; then
echo "::notice::release ${TAG} publishes no ${ASSET}; its installer builds it from source"
return 0
fi
has "the release's installer installed the release's binary" "installed ${ASSET} ${TAG} at ${HOST_BIN}" "$log"
}
# check_readme: this commit's installer takes everything from a release that publishes its
# SHA256SUMS and builds nothing on the host; from one without, it builds the tag from source
# and says why.
check_readme() {
local log="$1" role
if [ "${BINARY:-}" = yes ] && [ "${SUMS:-}" = yes ]; then
has "the binary is the release's" "installed ${ASSET} ${TAG} at ${HOST_BIN}" "$log"
has "the images and plugin come from the release" "release ${TAG}'s prebuilt images and Velocity plugin are installed as published" "$log"
for role in felis limbo lobby paper; do
has_re "felis/${role} is the release's" "felis/${role}:[^ ]* is the release's" "$log"
done
has_re "the registry image is the release's" "docker.io/library/registry@sha256:[0-9a-f]* is the release's" "$log"
has_re "the postgres image is the release's" "docker.io/library/postgres@sha256:[0-9a-f]* is the release's" "$log"
has "felis-velocity.jar is the release's" "felis-velocity.jar is the release's" "$log"
lacks_re "nothing fell back to a build or a pull" "on this host instead|from Docker Hub instead|building felis from source" "$log"
lacks_re "Docker was left alone" "docker already installed|installing docker|docker running" "$log"
elif [ "${BINARY:-}" = yes ]; then
has "the source build says why" "release ${TAG} publishes no SHA256SUMS, so ${ASSET} cannot be verified; building ${TAG} from source on this host instead" "$log"
echo "::warning::release ${TAG} publishes no SHA256SUMS, so the README's install builds ${TAG} from source on the host, Docker included; a release cut by release.yml puts it on the assets"
else
has "the source build says why" "release ${TAG} publishes no usable ${ASSET}; building ${TAG} from source on this host instead" "$log"
echo "::warning::release ${TAG} publishes no ${ASSET}, so the README's install builds ${TAG} from source on the host, Docker included; a release cut by release.yml puts it on the assets"
fi
}
case "${1:-}" in
find) find_release ;;
check-own) check_own "${2:?usage: e2e_release.sh check-own LOG}" ;;
check-readme) check_readme "${2:?usage: e2e_release.sh check-readme LOG}" ;;
*)
echo "usage: e2e_release.sh find | check-own LOG | check-readme LOG" >&2
exit 2
;;
esac
[ "$fails" -eq 0 ] || exit 1
-187
View File
@@ -1,187 +0,0 @@
#!/bin/bash
# Checks for deploy/e2e_release.sh. Run it as: bash deploy/e2e_release_test.sh
#
# The installer logs it reads are built from bootstrap.sh's own ok/warn messages, expanded
# with a release's values, so rewording one of them there fails here rather than in a
# two-hour e2e run. gh is a stub that prints what a release listing would.
set -u
here="$(dirname "$0")"
ER="${1:-${here}/e2e_release.sh}"
BS="${2:-${here}/bootstrap.sh}"
[ -f "$ER" ] || { echo "no such script: $ER"; exit 1; }
[ -f "$BS" ] || { echo "no such script: $BS"; exit 1; }
fails=0
expect() { # label needle haystack
case "$3" in
*"$2"*) echo "PASS $1" ;;
*) echo "FAIL $1: expected <$2> in:"; echo "$3"; fails=$((fails + 1)) ;;
esac
}
status() { # label want got
if [ "$2" = "$3" ]; then echo "PASS $1"; else echo "FAIL $1: exit $3, want $2"; fails=$((fails + 1)); fi
}
root="$(mktemp -d)"
trap 'rm -rf "$root"' EXIT
# msg <marker> prints bootstrap.sh's first ok/warn message holding <marker>, expanded with
# the variables below the way the installer expands it. They are read only through that eval.
# shellcheck disable=SC2034
{
tag=v1.2.3
FELIS_REF="$tag"
v="$tag"
asset=felis-linux-amd64
name="$asset"
HOST_BIN=/usr/local/bin/felis
digest=sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef
}
msg() {
local line body
line="$(grep -F -- "$1" "$BS" | grep -E '^[[:space:]]*(ok|warn) "' | head -n 1)"
if [ -z "$line" ]; then
echo "FAIL bootstrap.sh prints no message holding <$1>" >&2
fails=$((fails + 1))
return
fi
body="${line#*\"}"
body="${body%\"*}"
eval "printf '%s\n' \"${body}\""
}
# unusable <marker> prints the warning of bootstrap.sh's first artifact_unusable call holding
# <marker>; artifact_unusable joins its two arguments with "; ".
unusable() {
local line
line="$(grep -F -- "$1" "$BS" | grep -E '^[[:space:]]*artifact_unusable "' | head -n 1)"
if [ -z "$line" ]; then
echo "FAIL bootstrap.sh has no artifact_unusable call holding <$1>" >&2
fails=$((fails + 1))
return
fi
artifact_unusable() { printf '%s; %s\n' "$1" "$2"; }
eval "$line"
}
image_line() { # target
# shellcheck disable=SC2034 # read by msg's eval
local target="$1"
msg 'is the release'"'"'s (${digest:7:12})'
}
# What an install from a complete release prints, in the installer's order.
{
msg 'ok "installed ${asset} ${FELIS_REF} at ${HOST_BIN}"'
msg 'matches release ${tag}'"'"'s SHA256SUMS'
msg 'prebuilt images and Velocity plugin are installed as published'
for t in felis/felis:v1.2.3 felis/limbo:v1.2.3 felis/lobby:v1.2.3 felis/paper:v1.2.3 \
docker.io/library/registry@sha256:aa docker.io/library/postgres@sha256:bb; do
image_line "$t"
done
msg 'felis-velocity.jar is the release'"'"'s"'
} > "$root/complete.log"
readme() { # log binary sums
TAG="$tag" BINARY="$2" SUMS="$3" bash "$ER" check-readme "$1" 2>&1
}
own() { # log binary
TAG="$tag" BINARY="$2" bash "$ER" check-own "$1" 2>&1
}
out="$(readme "$root/complete.log" yes yes)"
status "a complete release's install passes" 0 $?
expect " and every image is checked" "PASS felis/paper is the release's" "$out"
grep -v 'felis/limbo:' "$root/complete.log" > "$root/nolimbo.log"
out="$(readme "$root/nolimbo.log" yes yes)"
status "an image missing from the log fails" 1 $?
expect " and names it" "FAIL felis/limbo is the release's" "$out"
{
cat "$root/complete.log"
name=felis-images-linux-amd64.txt unusable '} cannot be used" "building its images on this host instead"'
} > "$root/fallback.log"
out="$(readme "$root/fallback.log" yes yes)"
status "an image built on the host fails" 1 $?
expect " and quotes the fallback" "building its images on this host instead" "$out"
{ cat "$root/complete.log"; msg 'ok "docker already installed"'; } > "$root/docker.log"
out="$(readme "$root/docker.log" yes yes)"
status "Docker touched fails" 1 $?
sed 's/installed felis-linux-amd64 v1.2.3/installed felis-linux-amd64 v1.2.2/' "$root/complete.log" > "$root/other.log"
out="$(readme "$root/other.log" yes yes)"
status "another release's binary fails" 1 $?
# A release that publishes its binary but no SHA256SUMS: this commit's installer builds the
# tag from source and says so.
msg 'publishes no SHA256SUMS, so ${name} cannot be verified' > "$root/nosums.log"
out="$(readme "$root/nosums.log" yes "")"
status "a release without SHA256SUMS passes when the fallback is announced" 0 $?
expect " and warns in the run" "::warning::release v1.2.3 publishes no SHA256SUMS" "$out"
out="$(readme "$root/complete.log" yes "")"
status "a release without SHA256SUMS fails when nothing announced the fallback" 1 $?
msg 'publishes no usable ${asset}' > "$root/nobinary.log"
out="$(readme "$root/nobinary.log" "" "")"
status "a release without a binary passes when the fallback is announced" 0 $?
out="$(readme "$root/nosums.log" "" "")"
status "a release without a binary fails when the log says otherwise" 1 $?
# check-own: the release's own installer, which may predate SHA256SUMS.
out="$(own "$root/complete.log" yes)"
status "the release's installer downloading its binary passes" 0 $?
out="$(own "$root/nobinary.log" yes)"
status "the release's installer building from source fails" 1 $?
out="$(own "$root/other.log" yes)"
status "the release's installer downloading another release fails" 1 $?
out="$(own "$root/nobinary.log" "")"
status "a release without a binary asks nothing of its installer" 0 $?
# find: the listing gh answers with, in the step's outputs, exactly. GH_FAIL is what a
# failing gh prints on stderr before it exits 1.
mkdir -p "$root/bin"
cat > "$root/bin/gh" <<'STUB'
#!/bin/sh
if [ -n "${GH_FAIL:-}" ]; then
printf '%s\n' "$GH_FAIL" >&2
exit 1
fi
printf '%s\n' $GH_LISTING
STUB
chmod +x "$root/bin/gh"
find_run() { # listing [gh's error]: prints what find says; its outputs land in $root/out
: > "$root/out"
GH_LISTING="$1" GH_FAIL="${2:-}" GITHUB_REPOSITORY=FelisMC/Felis GITHUB_OUTPUT="$root/out" PATH="$root/bin:$PATH" bash "$ER" find 2>&1
}
same() { # label want got
if [ "$2" = "$3" ]; then echo "PASS $1"; else printf 'FAIL %s: got\n%s\nwant\n%s\n' "$1" "$3" "$2"; fails=$((fails + 1)); fi
}
find_run "v1.2.3 felis-linux-amd64 felis-linux-arm64 SHA256SUMS felis-velocity.jar" >/dev/null
same "find: a complete release" "$(printf 'tag=v1.2.3\nbinary=yes\nsums=yes')" "$(cat "$root/out")"
find_run "v0.1.0 felis-linux-amd64 felis-linux-arm64" >/dev/null
same "find: a release without SHA256SUMS" "$(printf 'tag=v0.1.0\nbinary=yes\nsums=')" "$(cat "$root/out")"
find_run "v0.1.0 felis-linux-arm64 felis-linux-amd64.cdx.json SHA256SUMS.sig" >/dev/null
same "find: only exact asset names count" "$(printf 'tag=v0.1.0\nbinary=\nsums=')" "$(cat "$root/out")"
out="$(find_run "" "release not found")"
status "find: no release passes" 0 $?
same " with an empty tag, which skips the release jobs" "$(printf 'tag=\nbinary=\nsums=')" "$(cat "$root/out")"
same " and says so" "::notice::no published release yet; the readme and upgrade jobs have nothing to install" "$out"
# Anything else gh fails on leaves the newest release unknown: the step fails, and writes no
# tag that would skip the jobs.
for e in "HTTP 401: Bad credentials (https://api.github.com/graphql)" \
"API rate limit exceeded for installation ID 1." \
"error connecting to api.github.com"; do
out="$(find_run "" "$e")"
status "find: gh failing with <$e> fails" 1 $?
same " and quotes gh" "::error::gh release view failed (exit 1), so the newest release is unknown: $e " "$out"
same " and writes no outputs" "" "$(cat "$root/out")"
done
out="$(find_run "")"
status "find: gh answering nothing fails" 1 $?
same " and says so" "::error::gh release view answered without a tag" "$out"
same " and writes no outputs" "" "$(cat "$root/out")"
echo
if [ "$fails" -eq 0 ]; then echo "ALL PASS"; else echo "${fails} FAILED"; exit 1; fi
-208
View File
@@ -1,208 +0,0 @@
#!/bin/bash
# Seeds the database of the release the e2e upgrade job installed, and checks after the
# upgrade that every seeded row came through unchanged. The job runs it around the upgrade:
#
# sudo bash deploy/e2e_seed.sh seed # after the release installed
# sudo bash deploy/e2e_seed.sh check # after this commit ran over it
#
# A fresh install's database holds only the rows its migrations write, so without the seed
# the upgrade moves and migrates an almost empty database: the move's row-count comparison
# compares zeros and the pending migrations never meet an existing row. The seed writes the
# columns the oldest release has (v0.1.0), so it applies to every release since, and the
# check reads the same columns back: a migration that rewrites one of them on purpose
# updates the snapshot query here.
#
# Every row is inert. Nothing in it is a credential anyone holds: the session and the
# challenge carry the hash of bytes nobody kept and are spent already. Nothing is due for
# the reaper or the operator, and the retention sweep keeps spent rows for 30 days after
# they were spent. servers stays empty: its rows mirror MinecraftServer objects the
# operator and the reaper reconcile, and a row without one would not stay as written.
set -euo pipefail
phase="${1:?usage: e2e_seed.sh seed|check}"
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
KUBECTL=(/usr/local/bin/k3s kubectl)
STATE_DIR=/var/tmp/felis-e2e-seed
PG_MOVED_MARKER=/var/lib/felis/postgres-moved
TABLES=(users account_links quotas sessions webauthn_challenges platform_settings audit_logs
world_backups image_builds image_submissions account_migrations op_login_requests)
fails=0
pass() { printf 'PASS %s\n' "$*"; }
fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); }
# db_where says where the platform's database lives: in felis-postgres once the install
# has it, else in the host PostgreSQL a release before it ran.
db_where() {
if "${KUBECTL[@]}" -n felis get deploy/felis-postgres >/dev/null 2>&1; then
echo pod
else
echo host
fi
}
# db_psql runs the SQL on its stdin in the felis database at <where>.
db_psql() { # where
case "$1" in
pod)
"${KUBECTL[@]}" -n felis exec -i deploy/felis-postgres -c postgres -- \
psql -X -q -At -v ON_ERROR_STOP=1 -U felis -d felis
;;
# From /, which the postgres user can always enter: psql warns about any other cwd.
host) (cd / && runuser -u postgres -- psql -X -q -At -v ON_ERROR_STOP=1 -d felis) ;;
esac
}
# snapshot prints every seeded row, one table after the other, each in key order. The host
# server's time zone is the host's and the pod's is UTC, so timestamps print in UTC.
snapshot() { # where
db_psql "$1" <<'EOF'
SET TimeZone = 'UTC';
SELECT 'users ' || row(id, username, email, role, email_verified, disabled, deleted_at, created_at, updated_at)::text
FROM users WHERE id LIKE 'e2e-seed-%' ORDER BY id;
SELECT 'account_links ' || row(user_id, mc_uuid, auth_source, verified_at)::text
FROM account_links WHERE user_id LIKE 'e2e-seed-%' ORDER BY mc_uuid;
SELECT 'quotas ' || row(user_id, max_servers, max_cpu_milli, max_memory_mb, max_storage_gb, updated_by, updated_at)::text
FROM quotas WHERE user_id LIKE 'e2e-seed-%' ORDER BY user_id;
SELECT 'sessions ' || row(token_hash, user_id, created_at, expires_at, revoked_at)::text
FROM sessions WHERE user_id LIKE 'e2e-seed-%' ORDER BY token_hash;
SELECT 'webauthn_challenges ' || row(id, user_id, purpose, session_data, expires_at, consumed_at, created_at)::text
FROM webauthn_challenges WHERE id LIKE 'e2e-seed-%' ORDER BY id;
SELECT 'platform_settings ' || row(key, value, updated_at)::text
FROM platform_settings WHERE key = 'e2e_seed';
SELECT 'audit_logs ' || row(id, actor, source, action, server_name, request_id, payload, created_at)::text
FROM audit_logs WHERE actor = 'e2e-seed-staff' ORDER BY id;
SELECT 'world_backups ' || row(id, server_name, former_owner, backup_ref, size_bytes, reason, status, created_at, expires_at, deleted_at)::text
FROM world_backups WHERE id LIKE 'e2e-seed-%' ORDER BY id;
SELECT 'image_builds ' || row(id, image_ref, status, dockerfile, context_ref, base_image, requested_by, job_name, log_ref, error, created_at, finished_at)::text
FROM image_builds WHERE id LIKE 'e2e-seed-%' ORDER BY id;
SELECT 'image_submissions ' || row(id, submitted_by, display_name, context_ref, status, image_ref, build_id, reviewed_by, reject_reason, created_at, reviewed_at)::text
FROM image_submissions WHERE id LIKE 'e2e-seed-%' ORDER BY id;
SELECT 'account_migrations ' || row(id, source_user_id, target_user_id, state, confirm_factor, confirmed_at, code_hash, code_expires_at, redeemed_at, created_at, updated_at)::text
FROM account_migrations WHERE id LIKE 'e2e-seed-%' ORDER BY id;
SELECT 'op_login_requests ' || row(id, user_id, email, expires_at, created_at, consumed_at, approved_at, approved_by)::text
FROM op_login_requests WHERE id LIKE 'e2e-seed-%' ORDER BY id;
EOF
}
# seed writes the rows in one transaction and keeps where they went and what they read as.
# Text carries quotes, backslashes, control characters and non-ASCII, and the bytea a NUL,
# so a dump or restore that mangles any of them shows in the check. Rows that the retention
# sweep would judge by age are dated now.
seed() {
local where before t
where="$(db_where)"
if ! db_psql "$where" <<'EOF'; then
BEGIN;
INSERT INTO users (id, username, email, role, email_verified, disabled, created_at, updated_at) VALUES
('e2e-seed-player', 'e2e_seed_player', '[email protected]', 'user', true, false,
'2026-01-02 03:04:05.678901+00', '2026-01-02 03:04:05.678901+00'),
('e2e-seed-staff', 'e2e_seed_staff', '[email protected]', 'admin', false, true,
'2026-01-03 00:00:00+00', '2026-01-03 00:00:00+00');
INSERT INTO account_links (user_id, mc_uuid, auth_source, verified_at) VALUES
('e2e-seed-player', '5eed0000-e2e0-4000-8000-000000000001', 'thirdparty', '2026-01-02 04:00:00+00');
INSERT INTO quotas (user_id, max_servers, max_cpu_milli, max_memory_mb, max_storage_gb, updated_by, updated_at) VALUES
('e2e-seed-player', 2, 4000, 8192, NULL, 'e2e-seed-staff', '2026-01-04 00:00:00+00');
INSERT INTO sessions (token_hash, user_id, created_at, expires_at, revoked_at) VALUES
(encode(sha256(convert_to(gen_random_uuid()::text, 'UTF8')), 'hex'), 'e2e-seed-player',
now(), now() + interval '30 days', now());
INSERT INTO webauthn_challenges (id, user_id, purpose, session_data, expires_at, consumed_at, created_at) VALUES
('e2e-seed-challenge', 'e2e-seed-player', 'passkey_register', '\x00ff0a0d5c27'::bytea,
now() + interval '5 minutes', now(), now());
INSERT INTO platform_settings (key, value, updated_at) VALUES
('e2e_seed', '{"text": "quote '' dq \" backslash \\ tab\t newline\n 猫 🐱", "n": 1.50, "list": [1, null, true, {"k": "v"}]}',
'2026-01-05 00:00:00+00');
INSERT INTO audit_logs (actor, source, action, server_name, request_id, payload, created_at) VALUES
('e2e-seed-staff', 'panel', 'e2e.seed', NULL, 'e2e-seed-req-1', '{"reason": "种子 \"quoted\"", "ids": [1, 2, 3]}', now()),
('e2e-seed-staff', 'cli', 'e2e.seed', 'e2e-seed-world', 'e2e-seed-req-2', NULL, now());
INSERT INTO world_backups (id, server_name, former_owner, backup_ref, size_bytes, reason, status, created_at, expires_at, deleted_at) VALUES
('e2e-seed-backup', 'e2e-seed-world', 'e2e-seed-player', 'e2e-seed/world.tar.zst', 123456789012, 'manual', 'deleted',
'2026-01-06 00:00:00+00', '2026-04-06 00:00:00+00', '2026-04-07 00:00:00+00');
INSERT INTO image_builds (id, image_ref, status, dockerfile, context_ref, requested_by, error, created_at, finished_at) VALUES
('e2e-seed-build', 'registry.felis.svc:5000/user-uploads/e2e-seed:latest', 'failed',
E'FROM scratch\nLABEL note="e2e seed"\n', 'e2e-seed/context', 'e2e-seed-staff', 'e2e seed: never built',
'2026-01-07 00:00:00+00', '2026-01-07 00:01:00+00');
INSERT INTO image_submissions (id, submitted_by, display_name, context_ref, status, reviewed_by, reject_reason, created_at, reviewed_at) VALUES
('e2e-seed-submission', 'e2e-seed-player', '种子整合包 e2e', 'e2e-seed/submission', 'rejected', 'e2e-seed-staff', 'e2e seed',
'2026-01-08 00:00:00+00', '2026-01-08 01:00:00+00');
INSERT INTO account_migrations (id, source_user_id, target_user_id, state, confirm_factor, confirmed_at, redeemed_at, created_at, updated_at) VALUES
('e2e-seed-migration', 'e2e-seed-staff', 'e2e-seed-player', 'redeemed', 'email_otp', '2026-01-09 00:00:00+00',
'2026-01-09 00:05:00+00', '2026-01-09 00:00:00+00', '2026-01-09 00:05:00+00');
INSERT INTO op_login_requests (id, user_id, email, expires_at, created_at, consumed_at, approved_at, approved_by) VALUES
('e2e-seed-oplogin', 'e2e-seed-staff', '[email protected]', now() + interval '10 minutes', now(), now(), now(), 'e2e-seed-staff');
COMMIT;
EOF
fail "seed the release's database (${where})"
return
fi
mkdir -p "$STATE_DIR"
printf '%s\n' "$where" > "${STATE_DIR}/where"
if ! before="$(snapshot "$where")"; then
fail "read the seeded rows back"
return
fi
printf '%s\n' "$before" > "${STATE_DIR}/before"
for t in "${TABLES[@]}"; do
if grep -q "^${t} " <<<"$before"; then
pass "seeded ${t} (${where})"
else
fail "seeded ${t} (${where}): the snapshot has no row of it"
fi
done
}
# check_upgrade reads the seeded rows from felis-postgres and compares them with what the
# release's database held. A database the upgrade moved off the host also left the marker.
check_upgrade() {
local where after seq
if [ ! -s "${STATE_DIR}/before" ]; then
fail "the seed step left its snapshot in ${STATE_DIR}/before"
return
fi
where="$(db_where)"
if [ "$where" != pod ]; then
fail "the database lives in felis-postgres after the upgrade"
return
fi
if [ "$(cat "${STATE_DIR}/where")" = host ]; then
if [ -f "$PG_MOVED_MARKER" ]; then
pass "the upgrade moved the seeded host database into felis-postgres"
else
fail "the upgrade moved the seeded host database into felis-postgres: no ${PG_MOVED_MARKER}"
fi
fi
if ! after="$(snapshot pod)"; then
fail "read the seeded rows from felis-postgres"
return
fi
if [ "$after" = "$(cat "${STATE_DIR}/before")" ]; then
pass "every seeded row came through the upgrade unchanged"
else
fail "the seeded rows changed across the upgrade (< release, > felis-postgres):"
diff "${STATE_DIR}/before" <(printf '%s\n' "$after") || true
fi
# A restore that loses a sequence's position hands out ids that are taken: the next
# audit row would fail on its primary key.
seq="$(db_psql pod <<<"SELECT coalesce(pg_sequence_last_value(pg_get_serial_sequence('audit_logs', 'id')::regclass), 0) >= (SELECT max(id) FROM audit_logs)")" || seq=error
if [ "$seq" = t ]; then
pass "audit_logs' id sequence is past every seeded id"
else
fail "audit_logs' id sequence is past every seeded id (${seq})"
fi
}
case "$phase" in
seed) seed ;;
check) check_upgrade ;;
*)
echo "usage: e2e_seed.sh seed|check" >&2
exit 2
;;
esac
if [ "$fails" -eq 0 ]; then
echo "ALL PASS (seed ${phase})"
exit 0
fi
echo "${fails} FAILED (seed ${phase})"
exit 1
+1 -4
View File
@@ -33,10 +33,7 @@
# must be >= 21 because current LOOHP/Limbo releases ship Java 21 API classes
# (class-file major 65); a JDK 17 fails to read them with "wrong version 65.0, should be
# 61.0". build.gradle still targets release 17 bytecode so the plugin loads on Java 17+.
# On the build platform: the jar is plain bytecode, identical for every architecture, so a
# release building the arm64 image on an amd64 runner compiles it natively instead of under
# QEMU (deploy/build-release-artifacts.sh). The runtime stage below stays on the target.
FROM --platform=$BUILDPLATFORM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
FROM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
WORKDIR /src
# Copy what the limbo module needs: its own tree plus the shared link core it
# srcDir-includes (../shared/src/main/java → /src/plugins/shared/src/main/java), so
+1 -7
View File
@@ -58,8 +58,6 @@ Configuration (deployment inputs, never compiled in; env wins over a
| `FELIS_LOBBY_SERVER` | Velocity server name to transfer to | `lobby` |
| `FELIS_LOGIN_TIMEOUT_SECONDS` | login window (clamped 30–3600) | `600` |
| `FELIS_HEALTH_PORT` | readiness port | `8080` |
| `FELIS_API_CONNECT_TIMEOUT_SECONDS` | felis-api connect timeout (1–120) | `10` |
| `FELIS_API_REQUEST_TIMEOUT_SECONDS` | felis-api call timeout (1–120) | `10` |
If the API config **or** the root domain is absent the login flow stays **OFF** and
the plugin runs readiness-only (the same "load un-crippled" fail-safe the other
@@ -116,10 +114,6 @@ LOOHP/Limbo would otherwise default to `30000`, unreachable through the Velocity
idempotent, so a persisted world volume keeps all its other `server.properties`
settings. Do **not** override `FELIS_GAME_PORT` except in lockstep with the operator.
It also pins `max-players=-1` (no cap, Limbo's own default): unbound players wait at
the gate for up to ten minutes and a stopped server's players all fall back here at
once, so a cap left on the volume would turn players away at the door.
## Configure (deployer's responsibility)
One setting this image does **not** guess (it keeps the release's own default):
@@ -144,7 +138,7 @@ set them by hand:
the minecraft namespace (and `felis setup` refreshes that replica from the control
namespace), and the operator injects it into the `login` pod (only) as
`FELIS_SERVICE_TOKEN` via a `secretKeyRef`, keyed off the reserved `login` name.
`sudo felis rotate-token -yes limbo` replaces it and restarts the pod. Until the token is
`sudo felis rotate-token limbo` replaces it and restarts the pod. Until the token is
present the plugin fail-safes to readiness-only, so the gate is never broken — it
simply does not authenticate yet.
- **Service:** the login pod dials `FELIS_API_BASE_URL`, which resolves to the
+1 -10
View File
@@ -61,11 +61,7 @@ set_prop() {
# The secret is base64/hex-ish, but a '/' or '&' would still break a bare sed s///.
# '|' as the delimiter plus escaping it is enough for every value we write.
esc=$(printf '%s' "$2" | sed 's/[|\\&]/\\&/g')
# Through a temp file rather than sed -i, which BSD sed reads differently, so the
# same function runs under the entrypoint tests on any machine.
sed "s|^$1=.*|$1=${esc}|" "$PROPS" > "$PROPS.tmp"
cat "$PROPS.tmp" > "$PROPS"
rm -f "$PROPS.tmp"
sed -i "s|^$1=.*|$1=${esc}|" "$PROPS"
else
printf '%s=%s\n' "$1" "$2" >> "$PROPS"
fi
@@ -79,11 +75,6 @@ set_prop bungeecord false
set_prop bungee-guard false
set_prop velocity-modern true
set_prop forwarding-secrets "$SECRET"
# Unbound players wait here for up to ten minutes, and a stopped server's players all
# fall back here at once, so the gate must never be full. -1 (no cap) is Limbo's own
# default; it is pinned so a hand-edited properties file on the volume cannot bring
# a cap back.
set_prop max-players -1
echo "felis-limbo: server-port=${PORT}, velocity-modern=true (forwarding secret loaded, UUIDs are Mojang-verified)"
JAVA_MEMORY_ARG=""
+1 -4
View File
@@ -32,10 +32,7 @@
# build rather than the module's wrapper, which would download the same distribution
# again on every image build. The build checks every dependency against
# plugins/paper/gradle/verification-metadata.xml and fails on a mismatch.
# On the build platform: the jar is plain bytecode, identical for every architecture, so a
# release building the arm64 image on an amd64 runner compiles it natively instead of under
# QEMU (deploy/build-release-artifacts.sh). The runtime stage below stays on the target.
FROM --platform=$BUILDPLATFORM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
FROM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
WORKDIR /src
COPY plugins/paper/ ./plugins/paper/
COPY plugins/shared/ ./plugins/shared/
-22
View File
@@ -46,25 +46,3 @@ sudo felis setup
- Align `online-mode` / player forwarding with the off-cluster Velocity proxy.
- The lobby speaks only the `felis:control` plugin-message channel; it holds no
felis-api token by design (spec §12).
## What the lobby allows
felis-paper's `LobbyGuard` keeps the lobby a hub that nobody can hurt, get hurt in,
or leave a mark on:
- every world is peaceful, with natural spawning, PvP, mob griefing and TNT off,
time frozen at noon, clear weather and inventories kept;
- players take no damage and never go hungry; a fall into the void lands at spawn;
- a player without `felis.lobby.build` joins at spawn in adventure mode and cannot
break or place blocks, use buckets, trample farmland, light fires, or harm mobs,
item frames, paintings, armor stands or vehicles. Buttons, doors, pressure plates
and containers keep working;
- every join gets a chat line with a click that runs `/menu`.
`felis.lobby.build` defaults to ops. To let an admin build the lobby, grant it with
LuckPerms (`lp user <name> permission set felis.lobby.build true` on the lobby console)
or op them.
The entrypoint pins `max-players=200` on every boot, over Paper's default of 20: every
authenticated player passes through here, and a stopped server's players arrive all at
once.
+1 -13
View File
@@ -57,11 +57,7 @@ set_prop() {
# otherwise corrupt this bare sed s||| and silently break the key. Same escaping as
# deploy/limbo — without it an unlucky password kills the console/permission channel.
esc=$(printf '%s' "$2" | sed 's/[|\\&]/\\&/g')
# Through a temp file rather than sed -i, which BSD sed reads differently, so the
# same function runs under the entrypoint tests on any machine.
sed "s|^$1=.*|$1=${esc}|" "$PROPS" > "$PROPS.tmp"
cat "$PROPS.tmp" > "$PROPS"
rm -f "$PROPS.tmp"
sed -i "s|^$1=.*|$1=${esc}|" "$PROPS"
else
printf '%s=%s\n' "$1" "$2" >> "$PROPS"
fi
@@ -70,14 +66,6 @@ set_prop() {
set_prop server-port "$PORT"
set_prop online-mode false
# Every authenticated player passes through the lobby, and a stopped server's players
# arrive together (they fall back to the login gate, which sends them straight on).
# Paper's default cap of 20 would turn the 21st away at the door. 200 is far above
# what one node serves at once, and a flood beyond it is refused at the door instead
# of running the 1Gi lobby out of memory. What the world itself allows (no damage, no
# building, the /menu hint) is felis-paper's LobbyGuard.
set_prop max-players 200
# RCON is the control plane's write channel (spec §8 写=RCON): the operator probes it
# for readiness and the player tally, and felis-api runs console/permission commands over
# it. Paper only reads these three keys from server.properties, so the operator's injected
+16 -132
View File
@@ -6,20 +6,17 @@
# curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes
#
# Options:
# --purge also delete /etc/felis and /var/lib/felis (the database's cluster, the
# database bundles, and anything an earlier keep-data run set aside), and
# drop the felis database and role from a host PostgreSQL an earlier
# release installed. Asks for the word "purge" unless --yes is given.
# --purge also drop the felis database and role, and delete /etc/felis and
# /var/lib/felis (the database bundles, and anything an earlier keep-data
# run set aside). Asks for the word "purge" unless --yes is given.
# --keep-k3s leave k3s installed and remove only Felis's namespaces and CRD.
# --remove-k3s run k3s's own uninstaller even when other workloads live in the cluster.
# --no-backup skip the final database bundle keep-data mode takes first.
# --yes do not ask.
#
# Keep-data mode (the default) first takes a database bundle (`felis db backup -label
# manual`) and stops if that fails. It leaves /etc/felis (the secrets, felis.toml,
# offsite.env) and /var/lib/felis in place, and with it the felis database: felis-postgres
# keeps its cluster in /var/lib/felis/postgres, stopped cleanly before k3s goes, and a
# host PostgreSQL an earlier release installed keeps its copy. The world, archive,
# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the
# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive,
# registry and upload volumes live under k3s's storage directory, which k3s's uninstaller
# deletes, so they are moved to /var/lib/felis/retained/k3s-storage-<UTC stamp> first; with
# --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go.
@@ -44,12 +41,6 @@ CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}"
VELOCITY_USER="felis-velocity"
DB_NAME="felis"
DB_USER="felis"
CONTROL_NS="felis"
# The database bootstrap runs in k3s, its cluster on a hostPath under DATA_DIR, and the
# marker bootstrap leaves once it moved a host PostgreSQL's felis database into it.
PG_DEPLOYMENT="felis-postgres"
PG_DATA_DIR="${DATA_DIR}/postgres"
PG_MOVED_MARKER="${DATA_DIR}/postgres-moved"
POD_CIDR="10.42.0.0/16"
SERVICE_CIDR="10.43.0.0/16"
FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}"
@@ -60,7 +51,7 @@ FELIS_CRD="minecraftservers.felis.lolicon.best"
# service that is already gone.
FELIS_UNITS=(
felis-db-backup.timer felis-watchdog.timer felis-offsite.timer felis-build-tools.timer felis-update-check.timer
felis-db-backup.service felis-watchdog.service felis-watchdog-failed.service felis-offsite.service felis-build-tools.service felis-update-check.service
felis-db-backup.service felis-watchdog.service felis-offsite.service felis-build-tools.service felis-update-check.service
felis-velocity.service felis-nano.service cloudflared-felis.service
felis-postgres-firewall.service
)
@@ -133,19 +124,18 @@ confirm() {
print_plan() {
log "this will remove from $(uname -n):"
log " the felis-* systemd units, cloudflared-felis.service, the ${VELOCITY_USER} user,"
log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld and ufw openings"
log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld openings"
case "$K3S_MODE" in
remove) log " k3s, with everything in it (${K3S_BIN_DIR}/k3s-uninstall.sh)" ;;
keep) log " Felis's namespaces (${FELIS_NAMESPACES[*]}) and the ${FELIS_CRD} CRD; k3s stays" ;;
absent) ;;
esac
if [ "$PURGE" = 1 ]; then
log " PURGE: ${STATE_DIR} (secrets), ${DATA_DIR} (the ${DB_NAME} database's cluster, the database"
log " bundles and anything set aside before), the ${DB_NAME} database and role in a host PostgreSQL,"
log " every world and archive, the Felis images and Docker's build cache"
log " PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles"
log " and anything set aside before), every world and archive, the Felis images and Docker's build cache"
else
[ "$BACKUP" = 1 ] && log " after a final database bundle into ${DATA_DIR}/db-backups"
log " kept: ${STATE_DIR}, ${DATA_DIR} (the ${DB_NAME} database in ${PG_DATA_DIR}); the volumes move to ${RETAIN_DIR}/"
log " kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/"
fi
}
@@ -244,35 +234,6 @@ remove_firewalld_rules() { # game-port nano-port
fi
}
# remove_ufw_rules takes back the ufw rules the installer added, found by their felis-
# comments (configure_k3s_firewall and its neighbours in bootstrap.sh). The k3s ranges stay
# when k3s does, and go with it or when it is already gone. ufw lists a rule's IPv6 twin
# under its own number, and each delete renumbers the rules after it, so they go from the
# highest number down.
remove_ufw_rules() {
command -v ufw >/dev/null 2>&1 || return 0
local status nums n keep_k3s=1
status="$(LC_ALL=C ufw status numbered 2>/dev/null)" || return 0
[ "$(printf '%s\n' "$status" | head -n 1)" = "Status: active" ] || return 0
[ "$K3S_MODE" = keep ] || keep_k3s=""
nums="$(printf '%s\n' "$status" | awk -v keep_k3s="$keep_k3s" '
match($0, /# felis-[a-z0-9-]+ *$/) {
tag = substr($0, RSTART + 2)
sub(/ +$/, "", tag)
if (keep_k3s != "" && tag ~ /^felis-k3s-/) next
if (match($0, /^\[ *[0-9]+\]/)) {
n = substr($0, RSTART + 1, RLENGTH - 2)
gsub(/ /, "", n)
print n
}
}' | sort -rn)"
[ -n "$nums" ] || return 0
for n in $nums; do
ufw --force delete "$n" >/dev/null
done
ok "ufw rules removed"
}
# retain_volumes_in_cluster keeps every volume Felis's claims are bound to when the
# namespaces go: local-path deletes a Delete-policy volume's directory with its claim.
retain_volumes_in_cluster() {
@@ -312,18 +273,8 @@ remove_from_cluster() {
ok "Felis removed from the cluster; k3s stays"
}
# stop_database_pod shuts felis-postgres down cleanly: k3s-killall.sh SIGKILLs every
# container, and the cluster it leaves in PG_DATA_DIR is what a reinstall starts from.
stop_database_pod() {
kube -n "$CONTROL_NS" scale deployment "$PG_DEPLOYMENT" --replicas=0 >/dev/null 2>&1 || return 0
kube -n "$CONTROL_NS" wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres \
--timeout=120s >/dev/null 2>&1 \
|| warn "${PG_DEPLOYMENT} did not stop within 2 minutes; its cluster recovers from its WAL on the next start"
}
remove_k3s() {
local stamp
[ "$PURGE" = 1 ] || stop_database_pod
if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then
# k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a
# world while it moves.
@@ -362,9 +313,6 @@ remove_host_files() {
fi
rm -rf "$OPT_DIR"
rm -f "$HOST_BIN" "${HOST_BIN}.new" "${HOST_BIN}.prev"
# The release assets an install that stopped part way left for its rerun: downloads, not
# data, so keep-data mode drops them too.
rm -rf "${DATA_DIR}/artifacts"
remove_cloudflared_binary
if [ "$PURGE" = 0 ]; then
# What describes the removed install goes; what a reinstall reuses stays. Without
@@ -377,8 +325,7 @@ remove_host_files() {
as_postgres() { (cd / && runuser -u postgres -- "$@"); }
# remove_hba_block drops the block bootstrap heads pg_hba.conf with (the rules of an
# install on the host server, or the lockout the move into k3s left), and nothing else.
# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else.
remove_hba_block() { # file
local tmp
tmp="$(mktemp)"
@@ -393,80 +340,23 @@ remove_hba_block() { # file
rm -f "$tmp"
}
# check_database_purge runs before anything is removed. DROP ROLE refuses a role that
# still owns a database or holds anything in one besides felis (the felis_pgint database
# CONTRIBUTING.md has developers make for the PG contract tests, a grant made by hand),
# and by the time purge_database runs the units and k3s are already gone. It lists what
# holds the role instead, so the purge either runs to the end or not at all.
check_database_purge() {
[ "$PURGE" = 1 ] || return 0
systemctl is-active --quiet postgresql 2>/dev/null || return 0
local sql held err
read -r -d '' sql <<SQL || true
WITH r AS (SELECT oid FROM pg_roles WHERE rolname = '${DB_USER}'),
f AS (SELECT oid FROM pg_database WHERE datname = '${DB_NAME}')
SELECT DISTINCT CASE
WHEN s.classid = 'pg_database'::regclass THEN 'database ' || (SELECT datname FROM pg_database WHERE oid = s.objid)
WHEN s.classid = 'pg_tablespace'::regclass THEN 'tablespace ' || (SELECT spcname FROM pg_tablespace WHERE oid = s.objid)
ELSE 'objects in database ' || (SELECT datname FROM pg_database WHERE oid = s.dbid)
END || CASE s.deptype WHEN 'o' THEN ' (owned)' ELSE ' (privileges)' END
FROM pg_shdepend s
WHERE s.refclassid = 'pg_authid'::regclass AND s.refobjid = (SELECT oid FROM r)
AND s.dbid IS DISTINCT FROM (SELECT oid FROM f)
AND NOT (s.classid = 'pg_database'::regclass AND s.objid IS NOT DISTINCT FROM (SELECT oid FROM f))
ORDER BY 1;
SQL
err="$(mktemp)"
if ! held="$(as_postgres psql -v ON_ERROR_STOP=1 -tAq 2>"$err" <<<"$sql")"; then
held="$(cat "$err")"
rm -f "$err"
die "could not ask PostgreSQL what the ${DB_USER} role still holds, so nothing was removed: ${held}"
fi
rm -f "$err"
[ -n "$held" ] || return 0
die "the ${DB_USER} role still holds $(printf '%s' "$held" | paste -sd ';' - | sed 's/;/; /g'), so DROP ROLE would fail halfway through the purge; nothing was removed. Hand them to postgres first (sudo -u postgres psql -c 'ALTER DATABASE <name> OWNER TO postgres', or REASSIGN OWNED BY ${DB_USER} TO postgres; DROP OWNED BY ${DB_USER}; inside that database), or rerun without --purge"
}
# purge_database drops the felis database and role from a host PostgreSQL an earlier
# release installed; the cluster felis-postgres runs goes with DATA_DIR (purge_state). That
# host server either still serves the platform, or the move into felis-postgres stopped it
# with the pre-move copy left in it for a rollback: a purge takes that copy too and leaves
# the server stopped. The units and k3s are gone by now, so a failure here is a warning
# with the commands to finish by hand, and the purge goes on.
purge_database() {
[ "$PURGE" = 1 ] || return 0
local hba started=0
if ! systemctl is-active --quiet postgresql 2>/dev/null; then
[ -e "$PG_MOVED_MARKER" ] && systemctl cat postgresql >/dev/null 2>&1 || return 0
if ! systemctl start postgresql >/dev/null 2>&1; then
warn "could not start the host PostgreSQL, so its copy of the ${DB_NAME} database from before the move into k3s stays in it; drop it once it runs: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}"
warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it"
return 0
fi
started=1
fi
local hba
hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
if ! as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = '${DB_NAME}' AND pid <> pg_backend_pid();
DROP DATABASE IF EXISTS ${DB_NAME};
DROP ROLE IF EXISTS ${DB_USER};
ALTER SYSTEM RESET listen_addresses;
SQL
then
[ "$started" = 0 ] || systemctl stop postgresql >/dev/null 2>&1 || true
warn "could not drop the ${DB_NAME} database and role from the host PostgreSQL; drop them by hand: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}"
return 0
fi
if [ -n "$hba" ] && [ -f "$hba" ]; then
remove_hba_block "$hba"
rm -f "${hba}.pre-pg-move"
fi
if [ "$started" = 1 ]; then
systemctl stop postgresql
ok "the host PostgreSQL's copy of '${DB_NAME}' from before the move into k3s dropped; the server stays stopped"
else
[ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba"
systemctl restart postgresql
ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again"
fi
}
purge_images() {
@@ -495,10 +385,6 @@ purge_state() {
&& cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)"
if [ -n "$cred" ]; then rm -f "$cred"; fi
rm -f "$TUNNEL_CONFIG"
# The file-context rule bootstrap gave the database's cluster directory.
if command -v semanage >/dev/null 2>&1; then
semanage fcontext -d "${PG_DATA_DIR}(/.*)?" >/dev/null 2>&1 || true
fi
rm -rf "$STATE_DIR" "$DATA_DIR"
ok "${STATE_DIR} and ${DATA_DIR} removed"
}
@@ -509,7 +395,6 @@ main() {
[ -e "$STATE_DIR" ] || [ -e "$HOST_BIN" ] || [ -e "$OPT_DIR" ] \
|| die "no Felis install here (${STATE_DIR}, ${HOST_BIN} and ${OPT_DIR} are all absent)"
decide_k3s
check_database_purge
print_plan
confirm
final_backup
@@ -525,7 +410,6 @@ main() {
esac
remove_nft_tables
remove_firewalld_rules "$game" "$nano"
remove_ufw_rules
remove_host_files
purge_database
purge_images
@@ -534,7 +418,7 @@ main() {
if [ "$PURGE" = 1 ]; then
ok "Felis is gone from this host"
else
ok "Felis is removed; the data stays in ${STATE_DIR} and ${DATA_DIR}, the ${DB_NAME} database in ${PG_DATA_DIR}"
ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}"
log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\""
fi
}
+4 -174
View File
@@ -44,10 +44,6 @@ fresh_host() {
printf 'x\n' > "$root/h/etc/$f"
done
printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat"
mkdir -p "$root/h/data/artifacts"
printf 'bundle\n' > "$root/h/data/artifacts/felis-image-game-linux-amd64.tar.partial"
mkdir -p "$root/h/data/postgres/18/docker"
printf '18\n' > "$root/h/data/postgres/18/docker/PG_VERSION"
for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do
printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b"
chmod +x "$root/h/bin/$b"
@@ -76,19 +72,11 @@ run_uninstall() {
. "$0"
calls="$ROOT/calls"
id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; }
# PG_HOST: the host PostgreSQL is active (the default), stopped, or not installed (none);
# PG_START=fail for one that will not start.
systemctl() {
echo "SYSTEMCTL $*" >> "$calls"
case "$*" in
"is-active --quiet firewalld") return 1 ;;
"is-active --quiet postgresql") [ "${PG_HOST:-active}" = active ] ;;
"cat postgresql") [ "${PG_HOST:-active}" != none ] ;;
"start postgresql") [ "${PG_START:-}" != fail ] ;;
*) return 0 ;;
esac
case "$*" in "is-active --quiet firewalld") return 1 ;; esac
return 0
}
semanage() { echo "SEMANAGE $*" >> "$calls"; }
kube() {
echo "KUBE $*" >> "$calls"
case "$*" in
@@ -101,25 +89,7 @@ run_uninstall() {
shift 3
case "$*" in
*"SHOW hba_file"*) echo "$ROOT/h/hba.conf" ;;
*)
sql="$(cat)"
case "$sql" in
# What the felis role still holds: PG_HELD, one line each; PG_CHECK=fail for a
# server that refuses the query.
*pg_shdepend*)
echo "PSQL-CHECK" >> "$calls"
[ "${PG_CHECK:-}" = fail ] && { echo "psql: error: connection refused" >&2; return 2; }
[ -z "${PG_HELD:-}" ] || printf "%s\n" "$PG_HELD" ;;
*) echo "PSQL $* $sql" >> "$calls"; [ "${PG_DROP:-}" != fail ] ;;
esac ;;
esac
}
# UFW_STATUS: what `ufw status numbered` answers; ufw translates it outside the C locale.
ufw() {
case "$*" in
"status numbered")
if [ "${LC_ALL:-}" = C ]; then printf "%s\n" "${UFW_STATUS:-Status: inactive}"; else echo "状态:激活"; fi ;;
*) echo "UFW $*" >> "$calls" ;;
*) echo "PSQL $* $(cat)" >> "$calls" ;;
esac
}
docker() { echo "DOCKER $*" >> "$calls"; }
@@ -146,9 +116,6 @@ expect "the volumes are set aside before k3s deletes them" "k3s-storage-" "$kept
[ ! -e "$root/h/etc/bootstrap.done" ] && [ ! -e "$root/h/etc/velocity.fingerprint" ] \
&& echo "PASS the markers of the removed install go, so a reinstall starts fresh" \
|| { echo "FAIL bootstrap.done or the proxy fingerprint was left"; fails=$((fails + 1)); }
[ ! -e "$root/h/data/artifacts" ] && [ -d "$root/h/data/postgres" ] \
&& echo "PASS keep-data drops the downloaded release assets and keeps the database" \
|| { echo "FAIL keep-data left the release-asset cache or took the database: $(ls "$root/h/data")"; fails=$((fails + 1)); }
refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
[ ! -e "$root/h/opt" ] && [ ! -e "$root/h/bin/felis" ] \
&& echo "PASS /opt/felis and the host binary are removed" \
@@ -158,15 +125,6 @@ refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls"
expect "the velocity user is removed" "USERDEL felis-velocity" "$calls"
expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out"
[ -f "$root/h/data/postgres/18/docker/PG_VERSION" ] && echo "PASS the database's cluster stays for the reinstall" \
|| { echo "FAIL keep-data removed the database's cluster"; fails=$((fails + 1)); }
stop_at="$(grep -n 'KUBE -n felis scale deployment felis-postgres --replicas=0' "$root/calls" | head -n 1 | cut -d: -f1)"
kill_at="$(grep -n 'RUN k3s-killall.sh' "$root/calls" | head -n 1 | cut -d: -f1)"
[ -n "$stop_at" ] && [ -n "$kill_at" ] && [ "$stop_at" -lt "$kill_at" ] \
&& echo "PASS the database shuts down cleanly before k3s-killall.sh kills what is left" \
|| { echo "FAIL felis-postgres was not stopped before k3s-killall.sh: $calls"; fails=$((fails + 1)); }
expect "and the uninstall waits for it to stop" "KUBE -n felis wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres" "$calls"
refute "keep-data keeps the cluster directory's SELinux rule" "SEMANAGE" "$calls"
# --- a failed final bundle stops everything ------------------------------------------------
fresh_host
@@ -203,7 +161,6 @@ fresh_host
out="$(run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
refute "purge takes no bundle" "db backup" "$calls"
expect "purge asks what the role holds first" "PSQL-CHECK" "$calls"
expect "purge drops the database" "DROP DATABASE IF EXISTS felis;" "$calls"
expect "purge drops the role" "DROP ROLE IF EXISTS felis;" "$calls"
expect "purge puts listen_addresses back" "ALTER SYSTEM RESET listen_addresses;" "$calls"
@@ -221,71 +178,6 @@ case "$hba" in
*) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;;
esac
expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls"
expect "purge drops the cluster directory's SELinux rule" "SEMANAGE fcontext -d $root/h/data/postgres(/.*)?" "$calls"
refute "purge leaves k3s's pods to k3s-uninstall.sh" "scale deployment felis-postgres" "$calls"
# --- purge after the move into felis-postgres ---------------------------------------------
# The move stopped the host server with the felis database left in it for a rollback.
moved_host() { fresh_host; printf 'moved\n' > "$root/h/data/postgres-moved"; printf 'saved\n' > "$root/h/hba.conf.pre-pg-move"; }
moved_host
out="$(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
expect "purge starts the stopped host server to drop the pre-move copy" "SYSTEMCTL start postgresql" "$calls"
expect "and drops it" "DROP DATABASE IF EXISTS felis;" "$calls"
expect "the server stays stopped afterwards" "SYSTEMCTL stop postgresql" "$calls"
refute "and is not restarted" "SYSTEMCTL restart postgresql" "$calls"
refute "the lockout leaves pg_hba.conf" "FELIS MANAGED" "$(cat "$root/h/hba.conf")"
[ ! -e "$root/h/hba.conf.pre-pg-move" ] && echo "PASS the pg_hba.conf the move saved goes with the copy" \
|| { echo "FAIL purge left pg_hba.conf.pre-pg-move"; fails=$((fails + 1)); }
moved_host
out="$(PG_HOST=stopped PG_START=fail run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
expect "a host server that will not start is named" "could not start the host PostgreSQL" "$out"
refute "so nothing is dropped" "DROP DATABASE" "$calls"
[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS and the purge goes on" \
|| { echo "FAIL the purge stopped at the host server"; fails=$((fails + 1)); }
moved_host
out="$(PG_HOST=stopped PG_DROP=fail run_uninstall "default felis minecraft" --purge --yes)"
calls="$(cat "$root/calls")"
expect "a drop that fails says how to finish by hand" "sudo -u postgres dropdb felis" "$out"
expect "and stops the server it started" "SYSTEMCTL stop postgresql" "$calls"
[ ! -e "$root/h/data" ] && echo "PASS a failed drop does not stop the purge halfway" \
|| { echo "FAIL the purge stopped at a failed drop"; fails=$((fails + 1)); }
fresh_host
(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes >/dev/null) # a subshell: sh keeps a prefix assignment to a function
refute "a stopped host server the move never touched is left alone" "SYSTEMCTL start postgresql" "$(cat "$root/calls")"
moved_host
(PG_HOST=none run_uninstall "default felis minecraft" --purge --yes >/dev/null) # a subshell: sh keeps a prefix assignment to a function
refute "a host server removed since the move is not started" "SYSTEMCTL start postgresql" "$(cat "$root/calls")"
# --- a purge DROP ROLE would refuse -------------------------------------------------------
# The VM drill: the PG contract tests' felis_pgint was owned by felis, the purge removed the
# units and k3s, then stopped at DROP ROLE with half the host gone.
untouched() { # label
[ -d "$root/h/opt" ] && [ -f "$root/h/units/felis-velocity.service" ] && [ -d "$root/h/etc" ] \
&& ! grep -q "k3s-uninstall.sh\|DROP DATABASE\|SYSTEMCTL disable" "$root/calls" \
&& echo "PASS $1" \
|| { echo "FAIL $1: $(cat "$root/calls")"; fails=$((fails + 1)); }
}
fresh_host
out="$(PG_HELD="database felis_pgint (owned)
objects in database shop (privileges)" run_uninstall "default felis minecraft" --purge --yes)"
expect "a purge the role cannot survive is refused" "the felis role still holds database felis_pgint (owned); objects in database shop (privileges)" "$out"
expect "with the way to hand the database over" "ALTER DATABASE <name> OWNER TO postgres" "$out"
untouched "nothing is removed when DROP ROLE would fail"
fresh_host
out="$(PG_HELD="database felis_pgint (owned)" CONFIRM_TTY="$root/no-tty/x" run_uninstall "default felis minecraft" --purge)"
expect "the check comes before the plan and the prompt" "the felis role still holds database felis_pgint" "$out"
refute "so nobody confirms a purge that cannot finish" "this will remove" "$out"
fresh_host
out="$(PG_CHECK=fail run_uninstall "default felis minecraft" --purge --yes)"
expect "a server that cannot be asked stops the purge" "could not ask PostgreSQL what the felis role still holds, so nothing was removed: psql: error: connection refused" "$out"
untouched "nothing is removed when the check cannot run"
fresh_host
PG_HELD="database felis_pgint (owned)" run_uninstall "default felis minecraft" --yes >/dev/null
calls="$(cat "$root/calls")"
refute "keep-data drops no role, so it asks nothing" "PSQL-CHECK" "$calls"
expect "and goes on" "RUN k3s-uninstall.sh" "$calls"
# --- the pieces read before they are removed ---------------------------------------------
fresh_host
@@ -321,71 +213,9 @@ for u in $(sed -n 's|^[A-Z_]*="/etc/systemd/system/\([^"]*\)"$|\1|p' "$(dirname
expect "the uninstaller removes $u" " $u" " $(printf '%s' "$units" | tr '\n' ' ')"
done
# --- ufw: the rules bootstrap added, by the comments bootstrap gives them ------------------
# The listing is what `ufw status numbered` prints once bootstrap has run: a rule of the
# operator's own first (commented, as an operator may), then each felis- rule bootstrap.sh can
# add and its IPv6 twin.
tags="$(grep -o 'comment felis-[a-z0-9-]*' "$(dirname "$US")/bootstrap.sh" | awk '{ print $2 }' | sort -u)"
case " $(printf '%s ' $tags)" in
*" felis-k3s-pods "*" felis-proxy "*) echo "PASS bootstrap tags its ufw rules" ;;
*) echo "FAIL bootstrap.sh adds no felis-k3s-pods and felis-proxy ufw rules: <$tags>"; fails=$((fails + 1)) ;;
esac
listing="Status: active
To Action From
-- ------ ----
[ 1] 22/tcp ALLOW IN Anywhere # ssh"
n=1
for twin in "" " (v6)"; do
for t in $tags; do
n=$((n + 1))
listing="${listing}
$(printf '[%2d] Rule%-22s ALLOW IN Anywhere%-19s # %s' "$n" "$twin" "$twin" "$t")"
done
done
listing="${listing}
$(printf '[%2d] 22/tcp (v6) ALLOW IN Anywhere (v6)' "$((n + 1))")"
# numbers <listing> <regex>: the rule numbers whose line matches, highest first.
numbers() { printf '%s\n' "$1" | grep -E "$2" | sed 's/^\[ *\([0-9]*\)\].*/\1/' | sort -rn | paste -sd ' ' -; }
deletes() { grep '^UFW --force delete' "$root/calls" | awk '{ print $4 }' | paste -sd ' ' -; }
fresh_host
out="$(UFW_STATUS="$listing" run_uninstall "default felis minecraft" --yes)"
want="$(numbers "$listing" '# felis-')"
[ -n "$want" ] && [ "$(deletes)" = "$want" ] \
&& echo "PASS a Felis-only cluster's uninstall deletes every felis- ufw rule, highest first" \
|| { echo "FAIL a Felis-only cluster: deleted <$(deletes)>, want <$want>"; fails=$((fails + 1)); }
case " $(deletes) " in
*" 1 "* | *" $((n + 1)) "*) echo "FAIL the operator's own ufw rules were deleted: $(deletes)"; fails=$((fails + 1)) ;;
*) echo "PASS the operator's own ufw rules stay" ;;
esac
expect " and says so" "ufw rules removed" "$out"
fresh_host
UFW_STATUS="$listing" run_uninstall "default kube-system felis minecraft felis-build shop" --yes >/dev/null
[ "$(deletes)" = "$(numbers "$listing" '# felis-' | tr ' ' '\n' | grep -vxE "$(numbers "$listing" '# felis-k3s-' | tr ' ' '|')" | paste -sd ' ' -)" ] \
&& echo "PASS a k3s that stays keeps its pod and service ranges in ufw" \
|| { echo "FAIL a kept k3s: deleted <$(deletes)>, listing:"; echo "$listing"; fails=$((fails + 1)); }
fresh_host
UFW_STATUS="Status: inactive" run_uninstall "default felis minecraft" --yes >/dev/null
[ -z "$(deletes)" ] && echo "PASS an inactive ufw is left alone" \
|| { echo "FAIL an inactive ufw: deleted <$(deletes)>"; fails=$((fails + 1)); }
# The database's cluster and the move's marker are where bootstrap put them, or keep-data
# and purge act on a directory that is not there.
paths="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; printf "%s %s\n" "$PG_DATA_DIR" "$PG_MOVED_MARKER"' "$US")"
bspaths="$(sed -n 's/^PG_DATA_DIR="\(.*\)"$/\1/p; s/^PG_MOVED_MARKER="\(.*\)"$/\1/p' "$(dirname "$US")/bootstrap.sh" | paste -sd ' ' -)"
[ -n "$bspaths" ] && [ "$paths" = "$bspaths" ] && echo "PASS the database paths agree with bootstrap" \
|| { echo "FAIL uninstall's database paths <$paths> differ from bootstrap's <$bspaths>"; fails=$((fails + 1)); }
bscache="$(sed -n 's/^ARTIFACT_CACHE="\(.*\)"$/\1/p' "$(dirname "$US")/bootstrap.sh")"
uscache="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; printf "%s/artifacts\n" "$DATA_DIR"' "$US")"
[ -n "$bscache" ] && [ "$uscache" = "$bscache" ] && echo "PASS the release-asset cache is where bootstrap keeps it" \
|| { echo "FAIL uninstall removes <$uscache>, bootstrap caches release assets in <$bscache>"; fails=$((fails + 1)); }
if [ "$fails" -eq 0 ]; then
echo "ALL PASS"
else
echo "$fails FAILED"
exit 1
fi
exit "$fails"
+84 -2046
View File
File diff suppressed because it is too large. Load diff
+38 -521
View File
@@ -12,16 +12,13 @@ Evidence tags follow troubleshooting.md: **[VM-VERIFIED]** was run on a real hos
## 1. Supported hosts
`deploy/bootstrap.sh` provisions a single node. It needs systemd, root, and one of the
package managers below; everything else (k3s, the JRE, cloudflared, and Docker when an image
has to be built on the host; see "Where the binary and the images come from" below) it
installs.
PostgreSQL runs inside k3s as the `felis-postgres` Deployment, from the official image the
release pins by digest, with its data on the host in `/var/lib/felis/postgres`.
package managers below; everything else (Docker, k3s, PostgreSQL, the JRE, cloudflared)
it installs.
| OS family | Package manager | Architectures | Status |
|---|---|---|---|
| CentOS Stream 9 (firewalld active, SELinux enforcing) | dnf | aarch64 | **[VM-VERIFIED]** fresh install from release assets and its rerun, upgrade from v0.1.0 (moving the database off the host PostgreSQL 13 into felis-postgres), uninstall and reinstall |
| Ubuntu 24.04 LTS | apt | x86_64 | **[CI]** fresh install and same-commit rerun from the pushed commit's release assets; the README's one-line install as a new host runs it (the newest release's own assets); upgrade from the newest release, installed from its assets by its own installer and seeded with rows in twelve tables, onto them, every seeded row read back unchanged; the on-host build weekly |
| CentOS Stream 9 (firewalld active, PostgreSQL 13) | dnf | aarch64 | **[VM-VERIFIED]** install, same-version rerun, upgrade, uninstall and reinstall |
| Ubuntu 24.04 LTS | apt | x86_64 | **[CI]** fresh install, same-commit rerun, and upgrade from the newest release to the pushed commit |
| RHEL / Rocky / Alma 9, Fedora | dnf | x86_64, aarch64 | [CODE-ONLY] same code path as CentOS Stream |
| Debian 12, other Ubuntu releases | apt | x86_64, aarch64 | [CODE-ONLY] |
| openSUSE Leap / Tumbleweed | zypper | x86_64, aarch64 | [CODE-ONLY] |
@@ -37,66 +34,11 @@ cloudflared is left as it is, see §4):
| Temurin JRE (Velocity) | 25, patch build pinned | `FELIS_JRE_VERSION`, sha256 per architecture |
| Go (nano builds) | 1.26.8 | `GO_PINNED_VERSION`, sha256 per architecture |
| Minecraft / Limbo / Paper / Velocity / LuckPerms | `deploy/game-stack.lock` | §15b |
| PostgreSQL | 18.6, the official `postgres` image by digest | `POSTGRES_IMAGE` in `bootstrap.sh`, `defaultPostgresImage` in `internal/platform` |
| PostgreSQL | the distribution's package | 13 and 18 are exercised by the `pgint` CI job |
32-bit hosts are not supported: there is no k3s, JRE or Go build the installer will fetch
for them.
Before it changes anything the installer checks the host and reports every problem at
once, then stops with nothing touched **[SH-TESTED]**:
- the architecture, systemd as init, and the memory cgroup controller k3s needs;
- RAM: under 1.75 GiB is refused (a "2 GB" VPS passes), under 3.5 GiB is a warning;
- free disk on each filesystem it writes to, summed when they share one: on a bare host
about 17 GiB installing a release, 15 GiB from `FELIS_ARTIFACT_DIR` and 23 GiB when it
builds the images itself; 7 GiB for a rerun; a directory that already holds data
(Docker's cache, a reused k3s) counting at the rerun size; a filesystem that would end
over 85%, where k3s starts deleting cached images, is a warning;
- the ports it will listen on: the game port, the panel NodePort, k3s's 6443/6444 and
10248–10259 and the registry's loopback 5000. A port held by the installer's own
proxy or k3s is a rerun and passes;
- another Kubernetes (kubelet, RKE2, k0s, MicroK8s) or a k3s agent on the host;
- the node address or a routed network inside k3s's `10.42.0.0/16` and `10.43.0.0/16`
(a Docker network there is the usual case); a wider route such as a `10.0.0.0/8` VPN
is a warning;
- HTTPS to the hosts it downloads from: GitHub and PaperMC's download API always, Docker
Hub when it builds images on the host, Rancher's RPM repository where k3s's installer
adds it (the list is under "Where the binary and the images come from"). A host counts
as reachable once a TLS handshake with it completes, and each gets three tries two
seconds apart. Installing a release, an
unreachable Docker Hub is a warning (it is needed only if an asset turns out unusable);
from `FELIS_ARTIFACT_DIR` it is not checked.
`FELIS_PREFLIGHT=warn` reports the same problems as warnings and installs anyway, for a
host the checks misjudge.
A host firewall is opened, never turned off **[SH-TESTED]**. With firewalld active the
installer adds the panel NodePort, the game port and 6443, and puts k3s's pod and service
ranges in the trusted zone. With ufw enabled (common on Ubuntu and Debian, and enabled in
the CI install **[CI]**) it admits `10.42.0.0/16` and `10.43.0.0/16`, the panel NodePort and
the game port, each rule commented `felis-…`; 6443 stays closed to the network, since pods
reach the API server from their own range. Felis-nano opens its port to
`FELIS_NANO_PROXY_CIDR` alone in either. `uninstall.sh` removes these again, the k3s ranges
only when k3s goes too. Any other firewall in front of the host must admit the same:
dropped pod traffic shows up as the first rollout timing out ("control-plane rollout did
not complete").
Two things the host must keep for as long as the install lives:
- **Its address.** The install is bound to the IPv4 address it was made on (the k3s
node, the network policies, the panel certificate and the default nip.io domain all
carry it). Give the host a static address or a DHCP reservation before installing;
the installer warns when the address is a lease, and the watchdog reports
`host-address` when the host loses it (troubleshooting §13c). The k3s node name is
pinned at install time, so a hostname change is harmless.
- **A synchronized clock.** The installer turns NTP on (chrony where nothing else can)
and the watchdog reports a clock that stays unsynchronized. Allow outbound UDP 123,
or set `FELIS_MANAGE_TIME_SYNC=0` on a host whose clock is kept another way.
The installer also makes the system journal persistent (capped at
`FELIS_JOURNAL_MAX_USE`, default 1G; `FELIS_MANAGE_JOURNAL=0` skips it) and writes the
admin kubeconfig `/etc/rancher/k3s/k3s.yaml` root-only: run `sudo k3s kubectl`.
One node is the whole supported shape. A world volume is a ReadWriteOnce claim on the
node's local-path storage, so a game server's pod is pinned to the node that first
scheduled it and cannot move when that node fails; the operator and felis-api each run
@@ -106,78 +48,6 @@ and gains no failover. A multi-node shape would need, at least, storage that can
pod to another node and leader election in felis-operator (controller-runtime's
`LeaderElection`) so a second replica can stand by.
### Where the binary and the images come from
A release install (the default channel, and the setup console) takes everything Felis
builds from that release's assets, each checked against the release's `SHA256SUMS` before
it is used: the `felis` binary, the control-plane image, the limbo, lobby and paper images,
the registry and PostgreSQL images (at the digests `bootstrap.sh` pins), and
`felis-velocity.jar`. The images go into k3s's containerd with `k3s ctr images import` and
from there into the in-cluster registry, so the host needs no Docker, Gradle, Go or Docker
Hub for them. k3s's own images come from k3s's GitHub release
(`k3s-airgap-images-<arch>.tar.zst`, checked against k3s's sha256 list) before k3s first
starts. An upgrade downloads only the image tars holding an image the host lacks; they wait
in `/var/lib/felis/artifacts` until the registry has the images, and are deleted then.
`deploy/build-release-artifacts.sh` documents every asset. The decisions are **[SH-TESTED]**.
Installing from the assets is **[VM-VERIFIED]** on CentOS Stream 9 aarch64 through
`FELIS_ARTIFACT_DIR`: a fresh install and an upgrade over a release that built on the host
pulled no image and built nothing, and a rerun imported and uploaded nothing. Downloading them from a
release is [SH-TESTED] until a release publishes assets.
The release is the newest one unless `FELIS_RELEASE=<tag>` names an earlier one, which
installs from that release's assets the same way: the way back after a bad upgrade
(troubleshooting §16, "Roll back an upgrade that broke the database"), with the installer
read at that tag.
The installer builds on the host instead, installing Docker for it and stopping Docker once
the images are in the registry, when:
- the source is not a release: `FELIS_VERSION_BOOTSTRAP=dev`, a pinned `FELIS_REF`, or
`FELIS_SKIP_FETCH`;
- `FELIS_GAME_STACK=latest`, for the login, lobby and paper images (the rest still come
from the release);
- the release publishes no `SHA256SUMS` (one cut before release assets existed, or still
uploading), or an asset is missing, fails its checksum or is malformed. Only that image is
built (the registry and PostgreSQL images are pulled from Docker Hub instead), and a
warning names it; troubleshooting §15c lists the messages. Each download is tried
three times first, and a host without the room for the build stops before installing
Docker (troubleshooting §15c).
`FELIS_ARTIFACT_DIR=<absolute path>` installs from a directory instead of the release: a
release's assets downloaded there (every `felis-*` file and `SHA256SUMS`), or the directory
`deploy/build-release-artifacts.sh <version> <dir>` wrote. Nothing of Felis's own is
downloaded or built (except the game images under `FELIS_GAME_STACK=latest`, which no release
ships), so an asset the directory lacks, or one failing its checksum, stops the install. It
cannot be combined with `FELIS_REF` or `FELIS_SKIP_FETCH`, which name a source too.
The rest of the host's software still downloads, so the host needs outbound HTTPS to these,
directly or through `https_proxy`. A host with no outbound access cannot be installed yet
**[SH-TESTED]**:
| Host | What comes from it |
|---|---|
| `github.com`, and the githubusercontent.com hosts its release downloads redirect to | k3s and its images (`k3s-airgap-images-<arch>.tar.zst`), cloudflared, the Temurin JRE, ViaVersion, ViaBackwards and ViaRewind |
| `raw.githubusercontent.com` | k3s's install script, until k3s is installed |
| `rpm.rancher.io` | k3s-selinux, which k3s's install script adds on an SELinux host of the Red Hat or SUSE family (CentOS Stream, RHEL, Rocky, Alma, Fedora, openSUSE Leap), until k3s is installed |
| `fill-data.papermc.io` | the Velocity jar, unless `FELIS_VELOCITY_FORK_JAR` supplies one |
| the distribution's package mirrors | the base packages (CA certificates, OpenSSL, curl and tar where missing), and container-selinux beside k3s-selinux |
Preflight probes each named host above before it changes anything and lists every one it
cannot reach in one refusal (`cannot reach … over HTTPS`); the package manager reports its
own mirrors. An override adds a host the download itself tries: `FELIS_JRE_VERSION` reads
`api.adoptium.net`, a `FELIS_VELOCITY_VERSION` other than the pinned one reads
`fill.papermc.io`, and `FELIS_GAME_STACK=latest` builds its images on the host from Docker
Hub (which preflight probes), PaperMC, Limbo's CI and LuckPerms.
```
# on a machine with access: the release's assets for the host's architecture
gh release download v1.4.0 --repo FelisMC/Felis --dir felis-v1.4.0 \
--pattern 'felis-*linux-amd64*' --pattern felis-velocity.jar --pattern SHA256SUMS
# on the host, after copying the directory over
sudo FELIS_ARTIFACT_DIR=/root/felis-v1.4.0 bash bootstrap.sh
```
`SHA256SUMS` lists both architectures; the files of the other one may be left out.
### While felis-api restarts
An installer rerun that changes felis-api, a node restart or a crashed pod takes the API
@@ -200,17 +70,6 @@ the window:
`/opt/felis/velocity/plugins/felis-link/last-servers.json`, the last server list the
API answered with, until a refresh succeeds (every 15 s).
A longer outage reads like this in the logs [VM-VERIFIED]. The drill scaled felis-api
to 0 for about 8 minutes on the reference VM.
- The proxy logged `server list refresh failed ... keeping current registrations` 11 s
in, then `still failing: 22 failed attempts over 304 s` at the 5-minute mark.
- The watchdog found `deployment/felis-api` critical on its first run after the scale.
It raised the alert on the first run past 5 minutes, at about 7 minutes; with no
`[smtp]` that is logged only (`journalctl -u felis-watchdog`).
- The proxy logged `server list refresh recovered after 32 failed attempts over 469 s`
as soon as the new pod was Available.
## 2. Sizing
### What the platform itself uses
@@ -226,16 +85,15 @@ server running **[VM-VERIFIED]**:
| lobby (Paper, pod limit 1 GiB) | ~0.7–0.85 GB |
| login (Limbo, pod limit 512 MiB) | ~0.16 GB |
| felis-api, felis-operator, registry gate | ~50 MB each |
| PostgreSQL (the felis-postgres pod) | ~30 MB plus page cache |
| PostgreSQL | ~30 MB plus page cache |
| **Total in use** | **~3.4 GB** |
Every game server adds the memory its owner gave it: the pod's limit equals its request,
and the JVM heap is derived from it (§1a). Quotas cap it per user (panel → 管理 → 配额).
A release install builds nothing (§1). When the installer builds on the host its peak is
the image builds (Docker plus a Gradle container); it stops Docker afterwards so that memory
goes back to the servers. On a host under 2 GB of RAM without swap it adds a 2 GiB
`/swapfile`.
The installer's own peak is the image builds (Docker plus a Gradle container); it stops
Docker afterwards so that memory goes back to the servers. On a host under 2 GB of RAM
without swap it adds a 2 GiB `/swapfile`.
### Recommendations
@@ -269,107 +127,15 @@ curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo FELIS_VELOCITY_XMX=2G bash
| World archives | the `felis-backups` volume (`FELIS_BACKUP_STORAGE`, default 10Gi requested) | about one compressed world per backup kept |
| In-cluster registry | the `registry` volume (default 10Gi requested) | 2–3 GB for the stock images; grows with custom builds, pruned daily (§9) |
| k3s's containerd images | `/var/lib/rancher/k3s/agent/containerd` | 6–9 GB |
| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store), on a host that built its images (§1) | 5–10 GB after repeated upgrades |
| Release assets during an install | `/var/lib/felis/artifacts` | up to ~2 GB, deleted once the images are in the registry |
| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store) | 5–10 GB after repeated upgrades |
| Toolchains and sources | `/opt/felis` | ~2.5 GB |
| Database | `/var/lib/felis/postgres` (felis-postgres's cluster) | tens of MB; the audit log is most of it |
| Database bundles | `/var/lib/felis/db-backups` | a few MB each, 14 daily kept |
k3s's local-path volumes do not enforce the requested sizes (§9), so every volume shares
the root filesystem. Give the host at least **40 GB**, and 60 GB or more once worlds and
custom images accumulate. The watchdog mails the owners when a watched filesystem passes
its threshold, and §13b covers a full disk. On a host that built its images,
`docker builder prune -af` (with Docker started) reclaims the build cache when space is
short; the next upgrade rebuilds it.
### Growing the disk
Everything above shares the root filesystem, so more room means a bigger root
filesystem. It grows in place, with everything running: enlarge the virtual disk at the
provider, then the partition and the filesystem on it.
```bash
sudo felis backup-now -yes # a mistyped partition number is how a resize loses a disk
lsblk -f # which disk and partition hold /, and whether LVM sits on it
sudo growpart /dev/vda 3 # cloud-utils-growpart (RHEL) / cloud-guest-utils (Debian, Ubuntu)
# LVM (the RHEL-family default):
sudo pvresize /dev/vda3
sudo lvextend -r -l +100%FREE /dev/<vg>/root # -r grows the filesystem with it
# no LVM:
sudo xfs_growfs / # xfs
sudo resize2fs /dev/vda3 # ext4
df -h /
```
`felis backup-now` (troubleshooting.md §10) archives every stopped world; add `-stop` to
include the running ones.
### Moving the data to its own disk [VM-VERIFIED]
The bulk lives under `/var/lib/rancher/k3s`: the worlds, the world archives, the registry
and the images. On a disk of its own it grows without touching the system, and a full
world store leaves the root filesystem alone. The database and its bundles
(`/var/lib/felis`) are small and stay on the root disk. The move takes the platform down
for the copy plus a minute or two: the drill copied 4.2 GB in 18 s, and felis-api answered
`/readyz` 14 s after k3s started on the new disk.
1. Attach the disk and put a filesystem on it (the whole disk; `lsblk` shows it empty):
```bash
sudo mkfs.xfs /dev/vdb
U=$(sudo blkid -s UUID -o value /dev/vdb)
```
2. Archive every world, stopping the servers so each one saves, and keep the watchdog
quiet for the next hour (the marker the installer writes: no mail, no failure pings
to the heartbeat, until the time in it):
```bash
sudo felis backup-now -yes -stop
sudo install -d -m 0755 /run/felis
echo $(( $(date +%s) + 3600 )) | sudo tee /run/felis/watchdog-quiet-until
```
3. Stop k3s and copy:
```bash
sudo systemctl stop k3s
sudo /usr/local/bin/k3s-killall.sh # the containers k3s leaves running, and their mounts
sudo mkdir -p /mnt/felis-data
sudo mount UUID=$U /mnt/felis-data
sudo rsync -aHAX --numeric-ids /var/lib/rancher/k3s/ /mnt/felis-data/
sudo umount /mnt/felis-data
```
`-X` carries the SELinux labels k3s set itself. Leave `restorecon` out: it would reset
runc and the CNI binaries from `container_runtime_exec_t` to the policy default.
4. Mount it in place, and tie k3s to the mount:
```bash
sudo mv /var/lib/rancher/k3s /var/lib/rancher/k3s.old
sudo mkdir /var/lib/rancher/k3s
echo "UUID=$U /var/lib/rancher/k3s xfs defaults,nofail 0 0" | sudo tee -a /etc/fstab
sudo mkdir -p /etc/systemd/system/k3s.service.d
printf '[Unit]\nRequiresMountsFor=/var/lib/rancher/k3s\n' | sudo tee /etc/systemd/system/k3s.service.d/data-disk.conf
sudo systemctl daemon-reload
sudo mount /var/lib/rancher/k3s
sudo systemctl start k3s
```
The drop-in is what keeps the data safe: k3s started on the empty mount point creates
a new, empty cluster there. With it, a disk that does not come up fails the start with
`A dependency job for k3s.service failed`, and `nofail` keeps the host booting so you
can reach it. In the drill a detached disk left k3s inactive and the mount point empty;
reattached, `systemctl start k3s` mounted it and started.
5. Check that `sudo k3s kubectl -n felis get pods` shows every pod ready and
`findmnt /var/lib/rancher/k3s` names the new disk, then start the servers from the
panel and `sudo rm /run/felis/watchdog-quiet-until`. Once the host has run a day,
`sudo rm -rf /var/lib/rancher/k3s.old` frees the root disk.
The watchdog already watches `/var/lib/rancher/k3s` as a filesystem of its own (its
`-disk-paths`), so the new disk's fill level is mailed like the root's.
its threshold, and §13b covers a full disk. `docker builder prune -af` (with Docker
started) reclaims the build cache when space is short; the next upgrade rebuilds it.
## 3. Uninstall
@@ -385,31 +151,24 @@ curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --purge # remove th
With a private repository, fetch it the way the README fetches `bootstrap.sh`.
Both modes remove the `felis-*` systemd units and `cloudflared-felis.service`, the
Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the release assets an interrupted
install left in `/var/lib/felis/artifacts`, the installer's cloudflared binary (unless
another unit runs it), the `felis_edge` nftables table (and `felis_postgres`, which
releases before the database moved into k3s loaded), the firewalld ports the installer
opened and its `felis-`-commented ufw rules. k3s goes with k3s's own `k3s-uninstall.sh` when the cluster holds nothing but
Felis's namespaces; when it runs anything else only `felis`, `minecraft`, `felis-build`
and the MinecraftServer CRD are deleted.
Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the installer's cloudflared binary
(unless another unit runs it), the `felis_postgres` and `felis_edge` nftables tables and
the firewalld ports the installer opened. k3s goes with k3s's own `k3s-uninstall.sh` when
the cluster holds nothing but Felis's namespaces; when it runs anything else only
`felis`, `minecraft`, `felis-build` and the MinecraftServer CRD are deleted.
`--keep-k3s` and `--remove-k3s` override that choice.
| | keep data (default) | `--purge` |
|---|---|---|
| Final database bundle | taken first (`felis db backup -label manual`); a failure stops the uninstall before anything is removed. `--no-backup` skips it | none |
| The database (`/var/lib/felis/postgres`) | kept; felis-postgres is stopped cleanly before k3s goes | deleted with `/var/lib/felis` |
| A host PostgreSQL an earlier release ran the database on | kept as it is: stopped after the move into k3s (below, §4), with its old copy of `felis` | its `felis` database and role are dropped (the server is started for that and stopped again), and `listen_addresses` and `pg_hba.conf` go back to how they were. Checked before anything is removed: a role that still owns another database (the `felis_pgint` the PG contract tests use, CONTRIBUTING.md) or holds grants elsewhere stops the purge up front with the list and the `ALTER DATABASE … OWNER TO postgres` to run |
| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, the mail relay password and uploads bucket keys `felis setup` took, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file |
| `/var/lib/felis` (the database, its bundles) | kept | deleted |
| `felis` database and role | kept | dropped; `listen_addresses` and `pg_hba.conf` go back to how they were |
| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file |
| `/var/lib/felis` (database bundles) | kept | deleted |
| Worlds, archives, registry, uploads | moved to `/var/lib/felis/retained/k3s-storage-<stamp>/` (with `--keep-k3s`: their volumes switch to `Retain` and stay in place) | deleted |
| Felis images, Docker build cache | kept | deleted |
The two database rows are [SH-TESTED] (`deploy/uninstall_test.sh`); the VM runs above
predate felis-postgres.
Neither mode removes packages (Docker, git, nftables, and the PostgreSQL server an earlier
release installed) or the swap file: other software may use them. On a host that should
end up bare:
Neither mode removes packages (Docker, PostgreSQL, git, nftables) or the swap file: other
software may use them. On a host that should end up bare:
```
sudo swapoff /swapfile && sudo rm /swapfile && sudo sed -i '\|^/swapfile |d' /etc/fstab
@@ -424,9 +183,8 @@ DNS records for the panel hostnames, and the Access application.
A keep-data uninstall leaves everything a reinstall needs. The installer reuses
`/etc/felis/secrets.env`, so the database password and the forwarding and session
secrets are unchanged, and the installer migrates the kept database instead of creating
one **[VM-VERIFIED]** (with the host database of the releases before felis-postgres).
felis-postgres starts again on the cluster kept in `/var/lib/felis/postgres` [SH-TESTED].
secrets are unchanged, and it migrates the kept database instead of creating one
**[VM-VERIFIED]**.
Each step below was run on the reference VM after a keep-data uninstall, and the
restored worlds matched their kept `level.dat` checksums **[VM-VERIFIED]**. `kept` names
@@ -498,7 +256,7 @@ the version they were installed with unless noted:
| Temurin JRE | moves to the pinned patch build | rerun |
| k3s | left alone | rerun with `FELIS_UPGRADE_DEPS=1`: moves to the pinned release through that tag's install script, one minor version at a time (a bigger jump stops before anything changes and names the release to go through), never backwards |
| cloudflared | left alone | rerun with `FELIS_UPGRADE_DEPS=1`: swaps `/usr/local/bin/cloudflared` for the pinned, sha256-checked release and restarts `cloudflared-felis`; a cloudflared the distribution installed stays with its package manager |
| PostgreSQL | follows the image the release pins | a minor release comes with a Felis release, and the rerun restarts felis-postgres on it (a few seconds without the API); a major version is a dump and restore (below) |
| PostgreSQL | the distribution's package | the package manager for a minor release; a major version needs `pg_upgrade` first (below) |
| Docker, git, nftables | distribution packages | the package manager |
```sh
@@ -508,9 +266,8 @@ curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap
`sudo felis update` reports Felis, Velocity, k3s, cloudflared, the JRE and PostgreSQL
against their newest releases; `--k3s`, `--cloudflared`, `--jre` and `--postgres` narrow
it to one. PostgreSQL is read from the felis-postgres container and compared within its
major, since a minor release arrives with a Felis release, and a major past its end of life
gets a note naming the current one.
it to one. PostgreSQL is compared within its major, since a minor release is a package
update, and a major past its end of life gets a note naming the current one.
The installer also sets up `felis-update-check.timer`, which runs `felis update --record`
once a day around 05:30 (and at boot after a missed run). `--record` stores the result
@@ -526,115 +283,22 @@ journalctl -u felis-update-check -n 50 --no-pager
sudo felis update --record # record a fresh check now
```
### Bringing an older install up to date [VM-VERIFIED]
Three pieces of an install keep the shape they had on the day they were created, and
neither `felis setup` nor `kubectl rollout restart` reaches them: the felis-api
Deployment (an env var added later, such as `FELIS_SMTP_PASSWORD`, is absent until the
Deployment is rendered again), the lobby image (built with whatever plugins the recipe
had then; LuckPerms came later, and without it every permission change from the panel
answers `luckperms_missing`), and the `MinecraftServer` specs (a field added later stays
unset). Bring all three forward in this order, images first:
```sh
# 1. Rerun the installer: renders and applies the control-plane bundle, rebuilds and
# re-imports the login and lobby images, and recreates those two pods so they run
# the new images. The [smtp] relay the setup wizard wrote is carried forward.
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
# 2. Fill the spec fields the system servers gained since (troubleshooting §12b), then
# RCON for user servers created before it was the default. -user-rcon waits on
# each server's image opening RCON; see §12b before running it.
sudo felis converge
sudo felis converge -user-rcon
```
Check each piece:
```sh
kubectl -n felis get deploy felis-api \
-o jsonpath='{.spec.template.spec.containers[0].env[*].name}' | tr ' ' '\n' | grep SMTP
kubectl -n minecraft exec lobby-0 -- ls /data/plugins | grep -i luckperms
kubectl -n minecraft get minecraftserver \
-o custom-columns=NAME:.metadata.name,RCON:.spec.rcon.enabled,IDLE:.spec.idle.autoStopEnabled
```
The env var only carries the password; mail still needs the relay itself, set in
`felis setup` → email. A user server picks up its new RCON block at its next start.
### PostgreSQL major versions [CODE-ONLY]
felis-postgres keeps its cluster in `/var/lib/felis/postgres/<major>/docker`. A release that
moves the image to a new major finds the old major's cluster there and stops before it
changes anything: the new server would start an empty cluster beside it. The way across is
a bundle, taken on the release you run now, restored into the new major's empty cluster:
The installer takes the major the distribution ships (13 on EL9) and never moves it. To
go to a newer one, stop the writers, keep a dump, then use the distribution's upgrade
path:
```sh
# On the release you run now:
b="$(sudo felis db backup -label pre-upgrade | sed -n 's/^felis db backup: wrote //p')"
sudo k3s kubectl -n felis scale deploy/felis-postgres --replicas=0
sudo mv /var/lib/felis/postgres/18 /var/lib/felis/postgres-18.old # the old major's cluster, for a way back
# Install the new release: it starts an empty cluster on the new major and creates the schema.
curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo bash
# Put the data back and bring its schema up to the new release.
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator --replicas=0
sudo felis db restore -yes -no-safety-backup "$b"
sudo felis migrate up -config /etc/felis/felis.host.toml
sudo -u postgres pg_dumpall > /root/felis-pg-$(date +%F).sql
# EL9: sudo systemctl stop postgresql; sudo dnf module switch-to postgresql:16
# sudo dnf install postgresql-upgrade; sudo postgresql-setup --upgrade
# Debian/Ubuntu: sudo pg_upgradecluster <old-major> main
sudo systemctl start postgresql
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator --replicas=1
```
Delete `/var/lib/felis/postgres-18.old` once the new release has run for a while. To go back
instead, scale felis-postgres to 0, move the new major's directory out of
`/var/lib/felis/postgres`, move `postgres-18.old` back as `/var/lib/felis/postgres/18`, and
rerun the older release's installer.
### The database's move into k3s [VM-VERIFIED] [CI]
Releases before the move ran the database on a PostgreSQL the installer installed on the
host. The first rerun of a release with felis-postgres moves it, once:
1. It stops felis-api, felis-operator and the host timers, and heads the host's
`pg_hba.conf` with a block that refuses every connection to `felis` but its own copy
(the original is kept beside it as `pg_hba.conf.pre-pg-move`).
2. It takes a `pre-pg-move` bundle of the host database (`felis db backup`), restores it
into felis-postgres (`felis db restore`) and compares the row count of every table on
both servers. Any failure up to here puts `pg_hba.conf` and the control plane back and
the platform keeps running on the host database, untouched.
3. It stops and disables the host `postgresql` service, which stays installed with its
copy of the data, and writes `/var/lib/felis/postgres-moved`. A host server that also
holds other databases keeps running; its `felis` copy is then reachable over loopback
only.
The e2e upgrade job seeds the newest release's database with users, links, sessions, audit
rows, backups, builds and the rest (`deploy/e2e_seed.sh`), upgrades, and checks that
felis-postgres holds every seeded row with the same values after the pending migrations.
While that release is v0.1.0, the upgrade is this move.
From then on the host config points at felis-postgres (`127.0.0.1:15432`, and
`deployment = "felis/felis-postgres"`, through which `felis db` runs `pg_dump`, `psql`
and `pg_restore` inside the pod) and the pods at `felis-postgres.felis.svc:5432`.
To go back to the host database, for instance to reinstall the release before the move:
```sh
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator deploy/felis-postgres --replicas=0
hba="$(sudo -u postgres psql -XtAc 'SHOW hba_file' 2>/dev/null || echo /var/lib/pgsql/data/pg_hba.conf)"
sudo cp -p "${hba}.pre-pg-move" "$hba"
sudo systemctl enable --now postgresql
sudo rm /var/lib/felis/postgres-moved
curl -fsSL <raw-url-of-that-release>/deploy/bootstrap.sh | sudo bash
```
`SHOW hba_file` needs the server running; with it stopped, the fallback path is EL's
(Debian and Ubuntu keep it in `/etc/postgresql/<major>/main/`). Whatever the platform wrote
after the move lives only in felis-postgres; take a bundle there first
(`sudo felis db backup`) and restore it onto the host database afterwards if that matters.
Once the move has run for a while, drop the host copy:
`sudo systemctl start postgresql; sudo -u postgres dropdb felis; sudo -u postgres dropuser felis`,
or remove the server package altogether.
### The MinecraftServer CRD [VM-VERIFIED]
Every rerun applies the CRD embedded in the `felis` binary (`felis bootstrap-assets crd`).
@@ -666,38 +330,6 @@ version, in this order, each step one release:
3. A later release stops serving `v1alpha1`. Felis itself reads through one Go type at a
time, so the operator and felis-api switch in the release that moves storage.
### Legacy-forwarded backends [VM-VERIFIED]
A 1.8-era backend sits behind ViaVersion, which drops modern forwarding's login plugin
message on the way down to protocol 47, so the proxy has to hand that server the
player's identity BungeeCord-style, in the handshake address. Only the Felis-Legacy
Velocity fork can do that per server. Mark the server's CR and the proxy picks it up at
its next server-list refresh (every 15 s):
```sh
kubectl -n minecraft label minecraftserver <name> felis.lolicon.best/forwarding=legacy
kubectl -n minecraft label minecraftserver <name> felis.lolicon.best/forwarding- # back to modern
journalctl -u felis-velocity | grep 'legacy forwarding list'
```
The installer's `FELIS_LEGACY_FORWARDING_SERVERS` (default `legacy18`) stays in the list
whatever the labels say. What a label does depends on the proxy the host runs:
| Proxy | A label applies |
|---|---|
| Fork with patch 0004 (`build-velocity.sh` default arm) | from the next connection to that server |
| Fork with 0003 alone (`--deployed`) | at the next `systemctl restart felis-velocity` |
| Stock Velocity | never; the log line is a warning naming the server |
On the test VM (fork with 0004) labelling a server logged `legacy forwarding list is now
[legacy18,resolvecheck]` 12 s later, and removing the label logged the list back to
`[legacy18]`. The fork's own test (`FelisLegacyForwardingTest`) covers the next
connection following the rewritten list.
Legacy forwarding carries no secret. A marked server believes any identity that reaches
its game port, which `felis-allow-game-from-velocity` limits to the proxy and the node
itself; anything else running on the node can reach it too.
## 5. Disaster recovery
The procedures are in §16: what a database bundle holds, restoring one on the same host,
@@ -712,122 +344,7 @@ production install:
holds only sealed objects.
- **Keep one database bundle off the host** as well when there is no bucket. It contains
`secrets.env`, which a rebuild needs to read the rest.
- **Rehearse the rebuild** once on a spare VM: troubleshooting.md §16 "Rebuild on a new
host", every step but 8 (take-over) and 11 (the tunnel), then its checks: sign in with
an email code, restore one world and join it. `felis offsite status` and `felis db check` exit
non-zero when the copy or the newest daily bundle is stale; wire them into your monitoring,
- **Rehearse the rebuild** once on a spare VM: §16 "Rebuild on a new host", steps 1–5,
then log in and restore one world. `felis offsite status` and `felis db check` exit
non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
or rely on the watchdog's mail.
### Moving to another host (planned)
A planned move is the rebuild of troubleshooting.md §16, with the old host still there to
hand over a copy that misses nothing. It needs the off-site bucket: that is how the world
archives reach the new host (§16 step 7). The platform is down from step 1 until the new
host serves.
1. **On the old host**, stop everything that changes a world, then send the last copy:
```bash
sudo install -d -m 0755 /run/felis
echo $(( $(date +%s) + 4 * 3600 )) | sudo tee /run/felis/watchdog-quiet-until
sudo systemctl stop felis-velocity # no joins, so no server wakes
sudo felis backup-now -yes -stop # every world archived; the servers stay stopped
sudo k3s kubectl -n felis scale deploy/felis-operator --replicas=0 # nothing starts a server from here on
sudo felis db backup # a bundle that lists those archives
sudo systemctl start felis-offsite.service
sudo felis offsite status # again until nothing waits
```
The order matters. The new host fetches the archives its restored database lists, so
the bundle comes after the last archive. The operator goes after `backup-now`, which
needs it to stop the servers. The quiet marker keeps the watchdog from mailing the
owners about the stopped proxy and operator for the next 4 hours.
2. **On the new host**, follow troubleshooting.md §16 "Rebuild on a new host" from step 1;
`fetch-db latest` picks the bundle the old host just sent. Step 8 (`felis offsite
take-over`) makes the new host the one that writes the bucket, and from then on the
old host copies nothing more. Steps 10 and 11 move the names and the tunnel.
3. **Check the new host** before announcing it: sign in with an email code, restore one
world and join it, and see `sudo felis offsite status` show a recent `last success` and
no stand-by notice.
4. **Retire the old host.** It holds the last copy of every world outside the bucket, so
keep it powered off with its disk for a few days first, disabled so a boot brings
nothing up:
```bash
sudo systemctl disable k3s felis-velocity felis-watchdog.timer felis-offsite.timer \
felis-db-backup.timer felis-update-check.timer felis-build-tools.timer
sudo poweroff
```
Then uninstall it (§3) or wipe it.
Each step is covered where it is documented (backup-now in troubleshooting.md §10, the
rebuild in §16); the sequence as a whole has not been rehearsed as one move.
## 6. Changing the root domain [VM-VERIFIED] [GO-TESTED] [SH-TESTED]
The root domain is written into more places than the installer's config: the panel
certificate (`/etc/felis/panel-tls.crt`), the `felis-config` Secret in both namespaces,
the `felis-api-tls` Secret, the proxy's `felis-link.properties`, the login gate's
`MinecraftServer` env (`FELIS_ROOT_DOMAIN`, `FELIS_PANEL_HOSTNAME`), the Cloudflare tunnel
and DNS. `felis domain set` moves every one of them that lives on the host, in that
order, then restarts what reads them; `felis domain check` reports each surface on its
own line. The installer keeps the installed domain: a rerun with a different
`FELIS_ROOT_DOMAIN` stops and names this command.
```sh
sudo felis domain set new.example.net # the plan: every surface, what it moves to, what it costs
sudo felis domain set -yes new.example.net # do it
sudo felis domain check # one line per surface; exits 1 while any is behind
```
What it keeps:
- A panel or admin-console hostname set by hand in `[auth]` (anything other than
`console.<root>` / `op.console.<root>`) stays as it is; change it in
`/etc/felis/felis.host.toml` yourself if it should move, then run `set` again.
- The other `[auth]` keys (`access_jwt_aud`, `client_ip_header`) and every other line of
both config files. The edit refuses a file it cannot change line for line (a multi-line
value, a quoted or dotted key) and names what to fix.
- An operator's own certificate. The installer's self-signed certificate is reissued for
the new names (same shape, the old pair saved beside it as `*.pre-domain-<time>`); a
certificate from another issuer that does not cover the new names stops the command
before anything changes. Replace it with one that does, then run `set` again.
What it costs, which the plan prints before `-yes`:
- **DNS.** `<root>`, `console.<root>`, `op.console.<root>` and `*.<root>` must reach the
host. The wildcard does not cover `op.console.<root>`, a third-level name: give it its
own record. `check` resolves each name and warns on the ones that do not resolve yet.
- **Players.** Servers are reached as `<name>.<new root>`; the old addresses stop routing,
and the proxy restart disconnects everyone online. The first installer re-run after a
move restarts the proxy once more: the fingerprint it keeps of the proxy's files
predates the move.
- **Sign-in.** Session cookies belong to the old hostnames, so everyone signs in again.
Passkeys are bound to the panel hostname: when it changes, the plan counts the passkeys
that stop working, and their users sign in with an email code and register a new one.
Without an `[smtp]` relay no code is delivered; an Owner locked out that way recovers
with `sudo felis breakGlass`.
- **Cloudflare.** The tunnel's ingress and the Access application still carry the old
names. Re-run the Cloudflare step of `sudo felis setup` after the move; `check` lists
the tunnel's hostnames against the new ones.
- **A proxy on another host** (a remote `felis-link.properties`) is outside this host's
reach: `set` prints the three keys to put there.
`set` is safe to repeat: a second run changes only what is still behind, and on an
install that is already on the domain it converges whatever `check` reports. The same
holds after an interruption.
On the reference VM the move from `10.211.55.6.nip.io` to `10-211-55-6.nip.io` took 34
seconds. The certificate served on 30443, `/config.json` on both hostnames, the proxy's
`Felis routing ready: rootDomain=` log line and the login pod's env all carried the new
names afterwards. A second `set -yes` changed and restarted nothing; the installer run
with the old `FELIS_ROOT_DOMAIN` stopped at its first check; a full installer re-run kept
the moved domain and left `check` clean; moving back restored every surface
**[VM-VERIFIED]**.
`check` reads the proxy as behind when `felis-velocity` started before
`felis-link.properties` last changed. Installers before this command rewrote that file
on every run, so a host upgraded from one can show that line once with the file already
on the names; `sudo systemctl restart felis-velocity` clears it. The installer now leaves
the file alone when its content is the same.
+1 -2
View File
@@ -80,8 +80,7 @@ sequenceDiagram
API-->>Panel: 412 not_linked
else linked
Repo-->>API: true
API->>Repo: QuotaCheck(user_id, the server's real size)
Note over API,Repo: all four caps: servers, CPU, memory, storage
API->>Repo: QuotaAvailable(user_id)
alt quota exhausted
Repo-->>API: false
API-->>Panel: 403 quota_exceeded
+98 -1261
View File
File diff suppressed because it is too large. Load diff
-133
View File
@@ -1,133 +0,0 @@
package felis
import (
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
)
// The lobby and login gate take their player cap from server.properties, which the
// entrypoint rewrites on every boot over whatever the volume already holds. These run
// the shipped entrypoints the way a pod does (image and volume paths pointed into temp
// dirs, java replaced by a stub that exits) and read the file the server would start on.
// runEntrypoint runs the embedded entrypoint with its runtime dir holding the given
// files and server.properties seeded with props ("" for a first boot), and returns
// server.properties afterwards.
func runEntrypoint(t *testing.T, name, runtimeVar string, files []string, props string, env ...string) string {
t.Helper()
root := t.TempDir()
runtime, data, bin := filepath.Join(root, "image"), filepath.Join(root, "data"), filepath.Join(root, "bin")
for _, f := range files {
p := filepath.Join(runtime, f)
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(p, []byte("jar"), 0o644); err != nil {
t.Fatal(err)
}
}
for _, d := range []string{data, bin} {
if err := os.MkdirAll(d, 0o755); err != nil {
t.Fatal(err)
}
}
if props != "" {
if err := os.WriteFile(filepath.Join(data, "server.properties"), []byte(props), 0o644); err != nil {
t.Fatal(err)
}
}
if err := os.WriteFile(filepath.Join(bin, "java"), []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil {
t.Fatal(err)
}
script := readGameStackFile(t, name)
for old, repl := range map[string]string{
runtimeVar: `RUNTIME_DIR="` + runtime + `"`,
`DATA_DIR="/data"`: `DATA_DIR="` + data + `"`,
} {
if !strings.Contains(script, old) {
t.Fatalf("%s no longer sets %s", name, old)
}
script = strings.Replace(script, old, repl, 1)
}
path := filepath.Join(root, "entrypoint.sh")
if err := os.WriteFile(path, []byte(script), 0o755); err != nil {
t.Fatal(err)
}
cmd := exec.Command("sh", path)
cmd.Env = append([]string{"PATH=" + bin + ":" + os.Getenv("PATH"), "FELIS_FORWARDING_SECRET=fwd-test"}, env...)
if out, err := cmd.CombinedOutput(); err != nil {
t.Fatalf("%s failed: %v\n%s", name, err, out)
}
got, err := os.ReadFile(filepath.Join(data, "server.properties"))
if err != nil {
t.Fatal(err)
}
return string(got)
}
// propLines lists the key's lines, so a key written twice shows up as two.
func propLines(props, key string) []string {
var out []string
for _, line := range strings.Split(props, "\n") {
if strings.HasPrefix(line, key+"=") {
out = append(out, line)
}
}
return out
}
func assertProp(t *testing.T, props, key, want string) {
t.Helper()
got := propLines(props, key)
if len(got) != 1 || got[0] != key+"="+want {
t.Errorf("%s: got %q, want exactly [%s=%s]\nserver.properties:\n%s", key, got, key, want, props)
}
}
var lobbyImage = []string{"paper.jar", "plugins/felis-paper.jar", "plugins/LuckPerms.jar"}
func TestLobbyEntrypointLiftsThePlayerCap(t *testing.T) {
t.Run("over the cap Paper wrote on an earlier boot", func(t *testing.T) {
props := runEntrypoint(t, "deploy/lobby/entrypoint.sh", `RUNTIME_DIR="/paper"`, lobbyImage,
"#Minecraft server properties\nmax-players=20\nmotd=Kept as it was\n")
assertProp(t, props, "max-players", "200")
assertProp(t, props, "motd", "Kept as it was")
})
t.Run("on a first boot", func(t *testing.T) {
props := runEntrypoint(t, "deploy/lobby/entrypoint.sh", `RUNTIME_DIR="/paper"`, lobbyImage, "")
assertProp(t, props, "max-players", "200")
assertProp(t, props, "online-mode", "false")
})
}
// The RCON password is arbitrary bytes from a Secret; each of sed's special characters
// has to land in the file as itself when an earlier boot's line is replaced.
func TestLobbyEntrypointWritesTheRconPasswordVerbatim(t *testing.T) {
const password = `a|b\c&d/e`
props := runEntrypoint(t, "deploy/lobby/entrypoint.sh", `RUNTIME_DIR="/paper"`, lobbyImage,
"enable-rcon=false\nrcon.password=stale\n", "RCON_PASSWORD="+password)
assertProp(t, props, "enable-rcon", "true")
assertProp(t, props, "rcon.password", password)
}
var limboImage = []string{"Limbo.jar", "plugins/felis-limbo.jar"}
func TestLimboEntrypointNeverCapsTheGate(t *testing.T) {
t.Run("over a cap left on the volume", func(t *testing.T) {
props := runEntrypoint(t, "deploy/limbo/entrypoint.sh", `RUNTIME_DIR="/limbo"`, limboImage,
"max-players=10\nlevel-name=world;spawn.schem\n")
assertProp(t, props, "max-players", "-1")
assertProp(t, props, "level-name", "world;spawn.schem")
assertProp(t, props, "velocity-modern", "true")
})
t.Run("on a first boot", func(t *testing.T) {
props := runEntrypoint(t, "deploy/limbo/entrypoint.sh", `RUNTIME_DIR="/limbo"`, limboImage, "")
assertProp(t, props, "max-players", "-1")
assertProp(t, props, "forwarding-secrets", "fwd-test")
})
}
+2 -2
View File
@@ -16,7 +16,6 @@ require (
github.com/minio/minio-go/v7 v7.2.1
github.com/prometheus/client_golang v1.19.1
github.com/prometheus/client_model v0.6.1
golang.org/x/text v0.39.0
k8s.io/api v0.31.3
k8s.io/apimachinery v0.31.3
k8s.io/client-go v0.31.0
@@ -96,10 +95,11 @@ require (
golang.org/x/crypto v0.52.0 // indirect
golang.org/x/exp v0.0.0-20231006140011-7918f672742d // indirect
golang.org/x/net v0.55.0 // indirect
golang.org/x/oauth2 v0.27.0 // indirect
golang.org/x/oauth2 v0.21.0 // indirect
golang.org/x/sync v0.21.0 // indirect
golang.org/x/sys v0.45.0 // indirect
golang.org/x/term v0.43.0 // indirect
golang.org/x/text v0.39.0 // indirect
golang.org/x/time v0.3.0 // indirect
gomodules.xyz/jsonpatch/v2 v2.4.0 // indirect
google.golang.org/protobuf v1.36.10 // indirect
+2 -2
View File
@@ -249,8 +249,8 @@ golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLL
golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU=
golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8=
golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww=
golang.org/x/oauth2 v0.27.0 h1:da9Vo7/tDv5RH/7nZDz1eMGS/q1Vv1N/7FCrBhI9I3M=
golang.org/x/oauth2 v0.27.0/go.mod h1:onh5ek6nERTohokkhCD/y2cV4Do3fxFHFuAejCkRWT8=
golang.org/x/oauth2 v0.21.0 h1:tsimM75w1tF/uws5rbeHzIWxEqElMehnc+iW793zsZs=
golang.org/x/oauth2 v0.21.0/go.mod h1:XYTD2NtWslqkgxebSiOHnXEap4TF09sJSc7H1sXbhtI=
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
Loaded 100 of 508 files, more files were not shown because too many files have changed in this diff. Show more