Compare commits
184
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c0ee75efe4 | ||
|
|
118e3bc0c4 | ||
|
|
3922aae9a7 | ||
|
|
f7a826d635 | ||
|
|
8dce369190 | ||
|
|
6227b2dd1a | ||
|
|
79fcc30118 | ||
|
|
ac834aa196 | ||
|
|
02d39d773a | ||
|
|
3254d5ae76 | ||
|
|
ab4d8c2bf7 | ||
|
|
6644046016 | ||
|
|
3c0322a5c6 | ||
|
|
01cd69692f | ||
|
|
20c4ae5567 | ||
|
|
d31fbae3cf | ||
|
|
6b5bd7055b | ||
|
|
9b4c9bf4b5 | ||
|
|
94864473ce | ||
|
|
2b9cd1869c | ||
|
|
cb3065da1d | ||
|
|
1d1549cea2 | ||
|
|
3abf146321 | ||
|
|
5dffadb40d | ||
|
|
34b81ee8fe | ||
|
|
e8ff8f2fe5 | ||
|
|
c97dadc904 | ||
|
|
ab9a6f0bdb | ||
|
|
051cc1c9f5 | ||
|
|
b6751eac0f | ||
|
|
db7fbfec46 | ||
|
|
cc0672557c | ||
|
|
b959c560cc | ||
|
|
7718879bcd | ||
|
|
9fd699e301 | ||
|
|
125ba1c3a0 | ||
|
|
e97a73e539 | ||
|
|
95f4e79ff2 | ||
|
|
840d339760 | ||
|
|
46f99a7104 | ||
|
|
a32e8c19f7 | ||
|
|
de5fbb7283 | ||
|
|
7b28434068 | ||
|
|
2c6e29fa0d | ||
|
|
5282d55214 | ||
|
|
68387a3b12 | ||
|
|
d308a5edb1 | ||
|
|
ec861c10e0 | ||
|
|
5d95dc6b6b | ||
|
|
d13d99ed75 | ||
|
|
c0228f4e86 | ||
|
|
e129b04c67 | ||
|
|
9df1178022 | ||
|
|
c1bf1a81a2 | ||
|
|
b6357ad68b | ||
|
|
50ec181177 | ||
|
|
f6d586ed0c | ||
|
|
9910c78709 | ||
|
|
1e42e9a87b | ||
|
|
249d675bd7 | ||
|
|
32534e3d56 | ||
|
|
67028727ae | ||
|
|
56a4bbcf4c | ||
|
|
4839d52f88 | ||
|
|
e9e9a7a622 | ||
|
|
e491e063cf | ||
|
|
009e81b4cf | ||
|
|
1124413876 | ||
|
|
26d997cb40 | ||
|
|
c54bd2d9eb | ||
|
|
372c8972f9 | ||
|
|
0d485992c6 | ||
|
|
c272e2ea24 | ||
|
|
67a7e27f8d | ||
|
|
b8da4e2318 | ||
|
|
fef8d962c9 | ||
|
|
01408014b2 | ||
|
|
2c5a85a033 | ||
|
|
7f203b9245 | ||
|
|
e91886822c | ||
|
|
5b0686dc06 | ||
|
|
f692725bed | ||
|
|
fe48319866 | ||
|
|
f9312118c9 | ||
|
|
d7afdb4124 | ||
|
|
dfc2dd8888 | ||
|
|
28fd20c791 | ||
|
|
64476b0171 | ||
|
|
7b91f7c6c3 | ||
|
|
572d12d5f3 | ||
|
|
403a751c96 | ||
|
|
2d04e0e637 | ||
|
|
d112a43c65 | ||
|
|
9d0253a75b | ||
|
|
339224bdc2 | ||
|
|
c1cdef0d6a | ||
|
|
983727d9c0 | ||
|
|
c146ce84e2 | ||
|
|
82ffa55a8f | ||
|
|
149ab0be66 | ||
|
|
c9e5e2fe3e | ||
|
|
fa46733859 | ||
|
|
aa73a8ca28 | ||
|
|
6943d9c4c6 | ||
|
|
4291d08e2b | ||
|
|
e519549c8a | ||
|
|
42431210c3 | ||
|
|
75dd639cca | ||
|
|
9a238780e5 | ||
|
|
6d5aca79fb | ||
|
|
4ae64cc2e8 | ||
|
|
82310d02ec | ||
|
|
7ce1ba7174 | ||
|
|
d2730c6e5a | ||
|
|
2f920cf1c9 | ||
|
|
0e08fa7ced | ||
|
|
4a26bf4ca0 | ||
|
|
d807fd5887 | ||
|
|
7aa5034c1a | ||
|
|
ee4065b986 | ||
|
|
9a89254146 | ||
|
|
7e6272997d | ||
|
|
4fe58ef6f5 | ||
|
|
aad72ddb2b | ||
|
|
5c58104e09 | ||
|
|
3f04171424 | ||
|
|
21dd4baed2 | ||
|
|
d0ab983f0f | ||
|
|
2d82e8cca7 | ||
|
|
8ef7112dab | ||
|
|
d6e1a6464c | ||
|
|
a60ff31ef6 | ||
|
|
4425060d3e | ||
|
|
fda14e9e45 | ||
|
|
d65cf3af30 | ||
|
|
22d1e834ad | ||
|
|
c1025274e1 | ||
|
|
df8b09788c | ||
|
|
4afabc390d | ||
|
|
3843082153 | ||
|
|
4b386c5c2e | ||
|
|
6f9f0c56ba | ||
|
|
9a853098af | ||
|
|
3b58b39763 | ||
|
|
760966660e | ||
|
|
2ca2f5ca86 | ||
|
|
d34a118723 | ||
|
|
537639072a | ||
|
|
56344f313b | ||
|
|
54942f30f3 | ||
|
|
9931f64fff | ||
|
|
7beb43195b | ||
|
|
c24396d50a | ||
|
|
36b8a8b444 | ||
|
|
5d6b36d190 | ||
|
|
96b4f31bf9 | ||
|
|
40d6e4d0ca | ||
|
|
fe9f49987b | ||
|
|
912d129f14 | ||
|
|
34a4dffc50 | ||
|
|
b496e59b99 | ||
|
|
659127b271 | ||
|
|
a20840e7d8 | ||
|
|
b4737f4682 | ||
|
|
12d2697801 | ||
|
|
1f2a6130ac | ||
|
|
1944a44f94 | ||
|
|
54a533103a | ||
|
|
eaef592014 | ||
|
|
23b7c14b00 | ||
|
|
6a061859dc | ||
|
|
d601abb09b | ||
|
|
8bc3997a8b | ||
|
|
5099b2501f | ||
|
|
7c2fa08e85 | ||
|
|
201cfc864e | ||
|
|
aa6e487ff9 | ||
|
|
cde6502ab7 | ||
|
|
9ab7b27a30 | ||
|
|
6a4c30b3f1 | ||
|
|
298a1e635d | ||
|
|
c15eec6c2d | ||
|
|
b65c2c00b3 | ||
|
|
fcf5c305ea |
No files matched your search
@@ -17,11 +17,15 @@ name: ci
|
||||
#
|
||||
# release.yml calls this workflow (workflow_call) before it builds anything, so a tag passes
|
||||
# exactly these gates and there is one list of them.
|
||||
#
|
||||
# workflow_dispatch reruns the suite on a commit whose push produced no run, such as one
|
||||
# pushed while Actions was unavailable.
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
workflow_call:
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -72,6 +76,9 @@ jobs:
|
||||
# The business stores' SQL against a real PostgreSQL (internal/pgint): the unit suites run
|
||||
# on fakes, and PGRepo drifted from them three times while those stayed green. 13 is the
|
||||
# oldest server a supported distribution installs (EL9), 18 the newest (Arch).
|
||||
# `felis db backup` and `restore` run there too, with the tools inside the service
|
||||
# container, as production runs them inside felis-postgres: the runner's own client is one
|
||||
# major version, and pg_dump refuses a newer server.
|
||||
pgint:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
@@ -102,6 +109,7 @@ jobs:
|
||||
- run: go test -race -tags pgint -count=1 ./internal/pgint/
|
||||
env:
|
||||
FELIS_TEST_PG_URL: postgres://felis:pgint@localhost:5432/felis_pgint?sslmode=disable
|
||||
FELIS_TEST_PG_EXEC: docker exec -i ${{ job.services.postgres.id }}
|
||||
|
||||
shell:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -134,8 +142,34 @@ jobs:
|
||||
tar -xJf shellcheck.tar.xz
|
||||
./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh')
|
||||
|
||||
# An exit status is 8 bits, so `exit "$fails"` reads 256 failures as a pass. The suites
|
||||
# exit 1 on any failure; this keeps the next one from carrying its count out.
|
||||
- name: No script exits with its failure count
|
||||
run: |
|
||||
if git grep -nE 'exit +"?\$\{?[a-z_]*fail[a-z_]*\}?"?[[:space:]]*$' -- '*.sh'; then exit 1; fi
|
||||
|
||||
- run: sh deploy/bootstrap_test.sh
|
||||
- run: sh deploy/uninstall_test.sh
|
||||
- run: bash deploy/e2e_release_test.sh
|
||||
|
||||
# The shipped alert rules (deploy/alerts): promtool parses them and runs their unit tests,
|
||||
# which pin when each alert fires and that it stays quiet before. internal/metrics'
|
||||
# alerts_test.go pins what promtool cannot see from there: the PrometheusRule twin and the
|
||||
# metric names the rules read.
|
||||
alerts:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
# A pinned release, like shellcheck's, so a new Prometheus cannot change what fails.
|
||||
- name: promtool
|
||||
run: |
|
||||
curl -fsSL -o prometheus.tar.gz \
|
||||
https://github.com/prometheus/prometheus/releases/download/v3.15.0/prometheus-3.15.0.linux-amd64.tar.gz
|
||||
echo "2a542df32eac02ee17b9d844fb2aa1de00dafa5476579ba8a3ba862e9d572ea0 prometheus.tar.gz" | sha256sum -c
|
||||
tar -xzf prometheus.tar.gz --strip-components=1 prometheus-3.15.0.linux-amd64/promtool
|
||||
./promtool check rules deploy/alerts/felis-alerts.yaml
|
||||
./promtool test rules deploy/alerts/felis-alerts_test.yml
|
||||
|
||||
panel:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -206,7 +240,7 @@ jobs:
|
||||
# compiled by bootstrap on a live host, and the test mains under plugins/*/test
|
||||
# were run by hand. JDK 25 is what the plugin build image runs (paper-api 26.x
|
||||
# needs it); each module's wrapper brings the Gradle the image pins.
|
||||
- uses: actions/setup-java@de7274f081f381c8f8158605e0321c36c376e2e6 # v6.0.1
|
||||
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
@@ -225,7 +259,7 @@ jobs:
|
||||
# this job nothing ever built them: no install path touches them, and their
|
||||
# gradlew scripts were committed without the exec bit, so the README's
|
||||
# one-liners failed on a fresh clone.
|
||||
- uses: actions/setup-java@de7274f081f381c8f8158605e0321c36c376e2e6 # v6.0.1
|
||||
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '17'
|
||||
|
||||
+218
-39
@@ -1,15 +1,26 @@
|
||||
# deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every
|
||||
# host goes through, run for real instead of by hand on a VM.
|
||||
#
|
||||
# install a full install of this commit, then the same commit again (a rerun must
|
||||
# converge without restarting what did not change)
|
||||
# upgrade the newest published release, then this commit on top of it; skipped until
|
||||
# a release exists
|
||||
# artifacts this commit's release assets, built by deploy/build-release-artifacts.sh the
|
||||
# way release.yml builds a tag's (amd64 only, the runners' architecture)
|
||||
# install a full install from those assets, the way a release installs, on a host with
|
||||
# ufw enabled, then the same assets again (a rerun must converge without
|
||||
# restarting what did not change, importing or uploading an image again, or
|
||||
# touching Docker)
|
||||
# readme the README's one-line install as a new host runs it today: this commit's
|
||||
# installer on its default channel, which installs the newest published
|
||||
# release's binary, images and plugin from that release's assets; skipped until
|
||||
# a release exists
|
||||
# upgrade the newest published release through its own installer and its own assets,
|
||||
# rows of every kind seeded into its database, then this commit's assets on
|
||||
# top of it; skipped until a release exists
|
||||
# source a full install built on the host from this checkout (FELIS_SKIP_FETCH):
|
||||
# the fallback a release without assets, or an unsupported one, takes. It builds
|
||||
# everything with Docker, so it runs by hand and weekly only
|
||||
#
|
||||
# Each job builds the control plane, the game images and the proxy from scratch, about
|
||||
# half an hour of runner time, so this runs on pushes that touch what gets installed, by
|
||||
# hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium).
|
||||
# deploy/e2e_check.sh holds the assertions.
|
||||
# This runs on pushes that touch what gets installed, by hand, and weekly (a moving
|
||||
# upstream: apt mirrors, k3s's install script and release assets, Adoptium).
|
||||
# deploy/e2e_check.sh holds the assertions, deploy/e2e_seed.sh the upgrade's seed and its check.
|
||||
name: e2e
|
||||
|
||||
on:
|
||||
@@ -32,14 +43,47 @@ on:
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# An explicit bash runs with -o pipefail, so `bootstrap.sh | tee install.log` fails the step
|
||||
# when the installer fails; the default shell reports tee's status.
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
concurrency:
|
||||
group: e2e-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
install:
|
||||
artifacts:
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 90
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
fetch-depth: 0 # the newest tag is the version's base, as on the dev channel
|
||||
|
||||
- uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
# Stamped the way bootstrap stamps a build of main: "<newest tag>+g<sha>".
|
||||
- name: Build the release assets
|
||||
run: |
|
||||
base="$(git describe --tags --abbrev=0 --match 'v*' 2>/dev/null || echo v0.0.0)"
|
||||
FELIS_RELEASE_ARCHES=amd64 deploy/build-release-artifacts.sh "${base}+g$(git rev-parse --short HEAD)" dist
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
with:
|
||||
name: e2e-assets
|
||||
path: dist/
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
install:
|
||||
needs: artifacts
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
@@ -47,28 +91,50 @@ jobs:
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
# FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs
|
||||
# have always been done; the .git directory is what stamps the build. Root owns it
|
||||
# so git, run as root by the installer, does not refuse it as dubious.
|
||||
- name: Stage this commit as the installer's source
|
||||
run: |
|
||||
sudo mkdir -p /opt/felis
|
||||
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
|
||||
sudo chown -R root:root /opt/felis/src
|
||||
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
|
||||
with:
|
||||
name: e2e-assets
|
||||
path: dist
|
||||
|
||||
# The runner has Docker preinstalled, so its absence proves nothing: the log shows
|
||||
# whether bootstrap reached for it. From FELIS_ARTIFACT_DIR every image and the plugin
|
||||
# come out of the assets, so it must not have. (`! grep` is spelled `if grep ... exit 1`
|
||||
# below because bash -e ignores a failing `!` command.)
|
||||
# ufw, enabled on many Ubuntu hosts, drops every inbound packet no rule admits, the
|
||||
# pods' traffic to the API server among them; the runner ships it disabled. The
|
||||
# runner's own traffic is outbound, which ufw lets through.
|
||||
- name: Enable ufw
|
||||
run: sudo ufw --force enable
|
||||
|
||||
- name: Install
|
||||
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log
|
||||
run: sudo FELIS_ARTIFACT_DIR="$GITHUB_WORKSPACE/dist" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee install.log
|
||||
|
||||
- name: Check the install
|
||||
run: sudo bash deploy/e2e_check.sh install
|
||||
run: |
|
||||
sudo bash deploy/e2e_check.sh install
|
||||
for tag in felis-k3s-pods felis-k3s-services felis-panel felis-proxy; do
|
||||
sudo ufw status | grep -qE "# ${tag} *$"
|
||||
done
|
||||
if grep -E 'docker already installed|installing docker|docker running' install.log; then exit 1; fi
|
||||
for role in felis limbo lobby paper; do
|
||||
grep -q "felis/${role}:[^ ]* is the release's" install.log
|
||||
done
|
||||
grep -q "docker.io/library/registry@sha256:[0-9a-f]* is the release's" install.log
|
||||
grep -q "docker.io/library/postgres@sha256:[0-9a-f]* is the release's" install.log
|
||||
grep -q "felis-velocity.jar is the release's" install.log
|
||||
|
||||
- name: Rerun the same commit
|
||||
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log
|
||||
- name: Rerun the same assets
|
||||
run: sudo FELIS_ARTIFACT_DIR="$GITHUB_WORKSPACE/dist" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee rerun.log
|
||||
|
||||
# containerd and the registry already hold every image under the digest the listing
|
||||
# names, so nothing is imported or uploaded twice.
|
||||
- name: Check the rerun
|
||||
run: |
|
||||
sudo bash deploy/e2e_check.sh rerun
|
||||
grep -q 'felis-velocity unchanged; left running' rerun.log
|
||||
if grep -E 'docker already installed|installing docker|docker running' rerun.log; then exit 1; fi
|
||||
if grep -E 'importing felis-image-|mirroring .* into the internal registry' rerun.log; then exit 1; fi
|
||||
grep -q 'is already in the internal registry' rerun.log
|
||||
|
||||
- name: Diagnostics
|
||||
if: failure()
|
||||
@@ -81,7 +147,8 @@ jobs:
|
||||
k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true
|
||||
k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true
|
||||
done
|
||||
sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true
|
||||
k -n felis logs deploy/felis-postgres --tail=60 || true
|
||||
sudo journalctl -u k3s -u felis-velocity --no-pager -n 120 || true
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
if: always()
|
||||
@@ -90,7 +157,62 @@ jobs:
|
||||
path: '*.log'
|
||||
if-no-files-found: ignore
|
||||
|
||||
# The README's command on a fresh host: this commit's installer, as main serves it, on its
|
||||
# default channel with nothing pinned. It resolves the newest release and installs that
|
||||
# release's binary, images and plugin from its assets, each checked against its
|
||||
# SHA256SUMS; the private repo's token is the only thing added. FELIS_INSTALL_MODE picks the
|
||||
# mode the setup console would ask for.
|
||||
readme:
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
- name: Find the newest release
|
||||
id: release
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: bash deploy/e2e_release.sh find
|
||||
|
||||
- name: Install as the README does
|
||||
if: steps.release.outputs.tag != ''
|
||||
env:
|
||||
TOKEN: ${{ github.token }}
|
||||
run: sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee readme.log
|
||||
|
||||
# The binary is the release's, so the phase is `release`: its database backup and
|
||||
# timers are the release's to have or lack.
|
||||
- name: Check the install
|
||||
if: steps.release.outputs.tag != ''
|
||||
env:
|
||||
TAG: ${{ steps.release.outputs.tag }}
|
||||
BINARY: ${{ steps.release.outputs.binary }}
|
||||
SUMS: ${{ steps.release.outputs.sums }}
|
||||
run: |
|
||||
sudo bash deploy/e2e_check.sh release
|
||||
sudo /usr/local/bin/felis version | grep -qx "felis ${TAG}"
|
||||
bash deploy/e2e_release.sh check-readme readme.log
|
||||
|
||||
- name: Diagnostics
|
||||
if: failure()
|
||||
run: |
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
|
||||
sudo -E /usr/local/bin/k3s kubectl -n felis logs deploy/felis-postgres --tail=60 || true
|
||||
sudo journalctl -u k3s -u felis-velocity -u docker --no-pager -n 120 || true
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
if: always()
|
||||
with:
|
||||
name: e2e-readme-logs
|
||||
path: '*.log'
|
||||
if-no-files-found: ignore
|
||||
|
||||
upgrade:
|
||||
needs: artifacts
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
@@ -101,19 +223,22 @@ jobs:
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
|
||||
with:
|
||||
name: e2e-assets
|
||||
path: dist
|
||||
|
||||
- name: Find the newest release
|
||||
id: release
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)"
|
||||
if [ -z "$tag" ]; then
|
||||
echo "::notice::no published release yet; the upgrade path has nothing to start from"
|
||||
fi
|
||||
echo "tag=${tag}" >> "$GITHUB_OUTPUT"
|
||||
run: bash deploy/e2e_release.sh find
|
||||
|
||||
# The release's own installer, fetching the release's own binary: what a host that
|
||||
# installed that release is running today.
|
||||
# The release's own installer on its default channel, fetching the release's own
|
||||
# binary: what a host that installed that release is running today. FELIS_REF would
|
||||
# build the tag from source instead, a path no host takes by default. FELIS_RELEASE pins
|
||||
# the tag found above; an installer older than FELIS_RELEASE ignores it and resolves
|
||||
# the newest release, the same tag.
|
||||
- name: Install the newest release
|
||||
if: steps.release.outputs.tag != ''
|
||||
env:
|
||||
@@ -121,29 +246,44 @@ jobs:
|
||||
TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh
|
||||
sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log
|
||||
sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_RELEASE="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log
|
||||
|
||||
- name: Check the release install
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: sudo bash deploy/e2e_check.sh release
|
||||
env:
|
||||
TAG: ${{ steps.release.outputs.tag }}
|
||||
BINARY: ${{ steps.release.outputs.binary }}
|
||||
run: |
|
||||
sudo bash deploy/e2e_check.sh release
|
||||
bash deploy/e2e_release.sh check-own release.log
|
||||
|
||||
# A fresh install's database holds only what its migrations wrote: the seed gives the
|
||||
# upgrade's move and its pending migrations existing rows to carry.
|
||||
- name: Seed the release's database
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: sudo bash deploy/e2e_seed.sh seed
|
||||
|
||||
- name: Upgrade to this commit
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: |
|
||||
sudo rm -rf /opt/felis/src
|
||||
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
|
||||
sudo chown -R root:root /opt/felis/src
|
||||
sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log
|
||||
run: sudo FELIS_ARTIFACT_DIR="$GITHUB_WORKSPACE/dist" FELIS_INSTALL_MODE=full bash deploy/bootstrap.sh 2>&1 | tee upgrade.log
|
||||
|
||||
# Both checks run and report: the seeded rows go last, after the restore drill has
|
||||
# also put them back from a bundle.
|
||||
- name: Check the upgrade
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: sudo bash deploy/e2e_check.sh upgrade
|
||||
run: |
|
||||
rc=0
|
||||
sudo bash deploy/e2e_check.sh upgrade || rc=1
|
||||
sudo bash deploy/e2e_seed.sh check || rc=1
|
||||
exit "$rc"
|
||||
|
||||
- name: Diagnostics
|
||||
if: failure()
|
||||
run: |
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
|
||||
sudo -E /usr/local/bin/k3s kubectl -n felis logs deploy/felis-postgres --tail=60 || true
|
||||
# The release ran the database on the host; the upgrade moves it into k3s.
|
||||
sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
@@ -152,3 +292,42 @@ jobs:
|
||||
name: e2e-upgrade-logs
|
||||
path: '*.log'
|
||||
if-no-files-found: ignore
|
||||
|
||||
source:
|
||||
if: github.event_name != 'push'
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 90
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
# FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src; the .git directory is what
|
||||
# stamps the build. Root owns it so git, run as root by the installer, does not refuse
|
||||
# it as dubious.
|
||||
- name: Stage this commit as the installer's source
|
||||
run: |
|
||||
sudo mkdir -p /opt/felis
|
||||
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
|
||||
sudo chown -R root:root /opt/felis/src
|
||||
|
||||
- name: Install
|
||||
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee source.log
|
||||
|
||||
- name: Check the install
|
||||
run: sudo bash deploy/e2e_check.sh install
|
||||
|
||||
- name: Diagnostics
|
||||
if: failure()
|
||||
run: |
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
|
||||
sudo journalctl -u k3s -u felis-velocity -u docker --no-pager -n 120 || true
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
if: always()
|
||||
with:
|
||||
name: e2e-source-logs
|
||||
path: '*.log'
|
||||
if-no-files-found: ignore
|
||||
@@ -2,14 +2,15 @@
|
||||
#
|
||||
# Two things depend on this job. /repos/{repo}/releases/latest must ANSWER — that endpoint is
|
||||
# what `felis update` polls (internal/updater/github.go) and what deploy/bootstrap.sh's default
|
||||
# "release" channel resolves its ref from. And the binaries below are what that channel then
|
||||
# INSTALLS: bootstrap downloads felis-linux-<arch> instead of compiling on the target host, so
|
||||
# these are the shipped artifact, not a convenience.
|
||||
# "release" channel resolves its ref from. And the assets below are what that channel then
|
||||
# INSTALLS: bootstrap downloads the felis binary, the image tars, their listing and the Velocity
|
||||
# plugin instead of building anything on the target host, so these are the shipped artifact,
|
||||
# not a convenience. A host installing a release needs neither Docker, Gradle, Go nor Docker Hub.
|
||||
#
|
||||
# The asset NAME is a contract with deploy/bootstrap.sh (download_release_binary builds
|
||||
# "felis-linux-${arch}"). It is deliberately a plain literal on both sides: a GitHub Actions
|
||||
# YAML and a go:embed'ed shell script have no honest way to share a constant, and the failure
|
||||
# mode is benign — bootstrap warns and falls back to a source build of the same tag.
|
||||
# deploy/build-release-artifacts.sh builds every asset in one run and documents each name; the
|
||||
# names are a contract with deploy/bootstrap.sh. They are plain literals on both sides: a YAML
|
||||
# workflow and a go:embed'ed shell script have no honest way to share a constant, and the
|
||||
# failure mode is benign — bootstrap warns and falls back to building on the host.
|
||||
#
|
||||
# The binary is built through the repo Dockerfile rather than a plain `go build`.
|
||||
# internal/panel/static holds a tracked PLACEHOLDER index.html so the //go:embed
|
||||
@@ -18,13 +19,12 @@
|
||||
# build first, and is the same recipe bootstrap uses, so there is one way to build felis
|
||||
# rather than two that can drift.
|
||||
#
|
||||
# SHA256SUMS is a contract with bootstrap too: download_release_binary refuses a binary whose
|
||||
# hash is not listed there, BEFORE it runs it. A release without the file installs by source
|
||||
# build instead.
|
||||
# SHA256SUMS is a contract with bootstrap too: it refuses any asset whose hash is not listed
|
||||
# there, BEFORE it runs or imports it.
|
||||
#
|
||||
# The write token never meets the test suite: `gates` (ci.yml) and `build` run the tests,
|
||||
# Gradle and the Docker build (each of which executes third-party code) with a read-only
|
||||
# token, and `build` hands the binaries over as a workflow artifact; `publish` holds contents:write and runs only
|
||||
# Gradle and the Docker builds (each of which executes third-party code) with a read-only
|
||||
# token, and `build` hands the assets over as a workflow artifact; `publish` holds contents:write and runs only
|
||||
# pinned actions and gh. Every action is pinned to a commit SHA (the tag in the trailing
|
||||
# comment is for humans); .github/dependabot.yml proposes the bumps.
|
||||
name: release
|
||||
@@ -39,9 +39,10 @@ permissions:
|
||||
jobs:
|
||||
# A tag that ships red is worse than a tag that fails to ship. These are ci.yml's gates,
|
||||
# called rather than copied: Go (race, vet, staticcheck), govulncheck, the PostgreSQL
|
||||
# contract suite, shellcheck and the bootstrap tests, the panel, and the Java layer the
|
||||
# binary EMBEDS (bootstrap_asset.go ships the plugin sources, so a tag whose plugins do
|
||||
# not compile turns every install of that release into a failed bootstrap).
|
||||
# contract suite, shellcheck and the bootstrap tests, promtool on the alert rules, the
|
||||
# panel, and the Java layer the binary EMBEDS (bootstrap_asset.go ships the plugin
|
||||
# sources, so a tag whose plugins do not compile turns every install of that release into
|
||||
# a failed bootstrap).
|
||||
gates:
|
||||
uses: ./.github/workflows/ci.yml
|
||||
|
||||
@@ -52,69 +53,47 @@ jobs:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
# Both architectures, because bootstrap's default release channel DOWNLOADS these
|
||||
# rather than compiling on the target host — an arm64 host with no asset silently
|
||||
# falls back to a slow source build. Neither stage is emulated: the Dockerfile pins
|
||||
# both build stages to $BUILDPLATFORM and the Go stage cross-compiles via TARGETARCH,
|
||||
# so the second architecture costs about a minute.
|
||||
# rather than building on the target host — an arm64 host with no asset falls back to
|
||||
# a slow on-host build. The felis binary and the plugin jars cross-compile on the
|
||||
# runner (their build stages are pinned to $BUILDPLATFORM); the game images' runtime
|
||||
# stages run apt-get for the target platform, which for arm64 takes QEMU.
|
||||
- uses: docker/setup-qemu-action@99012661954931238ded8c8b007157a8430204e1 # v4.4.0
|
||||
- uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
|
||||
- name: Build the stamped binaries
|
||||
run: |
|
||||
docker buildx build --platform linux/amd64,linux/arm64 \
|
||||
--build-arg FELIS_VERSION="${GITHUB_REF_NAME}" \
|
||||
--output type=local,dest=out .
|
||||
mv out/linux_amd64/usr/local/bin/felis ./felis-linux-amd64
|
||||
mv out/linux_arm64/usr/local/bin/felis ./felis-linux-arm64
|
||||
chmod +x ./felis-linux-amd64 ./felis-linux-arm64
|
||||
# Two architectures of five images plus BuildKit's cache outgrow the runner's free disk
|
||||
# with its preinstalled SDKs in place; none of them is used here.
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
# The stamp is the whole point and it fails silently: an unstamped binary reports
|
||||
# "dev", which `felis update` refuses to compare, disabling update reporting for
|
||||
# every install built from it. Assert it end to end instead of trusting the ARG
|
||||
# reached the linker.
|
||||
- name: Verify the version stamp
|
||||
run: |
|
||||
got="$(./felis-linux-amd64 version | head -1)"
|
||||
echo "reported: ${got}"
|
||||
[ "$got" = "felis ${GITHUB_REF_NAME}" ] \
|
||||
|| { echo "expected 'felis ${GITHUB_REF_NAME}' — the -X main.version stamp did not reach the binary"; exit 1; }
|
||||
# The arm64 binary is checked by ELF machine type, NOT by running it. Runners have
|
||||
# binfmt/QEMU registered, so `./felis-linux-arm64 version` would happily succeed on
|
||||
# an amd64 binary misnamed arm64 — which is exactly the failure the Dockerfile's
|
||||
# ${TARGETARCH:-$(go env GOARCH)} fallback produces if buildx did not take. Both
|
||||
# binaries come out of one RUN with one -ldflags string, so the stamp is checked once.
|
||||
file ./felis-linux-arm64
|
||||
file ./felis-linux-arm64 | grep -q 'ARM aarch64' \
|
||||
|| { echo "felis-linux-arm64 is not an arm64 ELF — TARGETARCH did not reach the go build"; exit 1; }
|
||||
|
||||
- name: Checksum the binaries
|
||||
run: sha256sum felis-linux-amd64 felis-linux-arm64 | tee SHA256SUMS
|
||||
# The script checks what the old inline steps did: the host binary reports exactly
|
||||
# "felis <tag>" (unstamped, `felis update` refuses to compare), and the other one is an
|
||||
# ELF of its architecture, read with file(1) — binfmt would run a misnamed binary happily.
|
||||
- name: Build the release assets
|
||||
run: deploy/build-release-artifacts.sh "${GITHUB_REF_NAME}" dist
|
||||
|
||||
# A CycloneDX SBOM per binary: the Go modules (and versions) linked into it, read
|
||||
# from the build info the linker embeds.
|
||||
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0.24.0
|
||||
with:
|
||||
file: felis-linux-amd64
|
||||
file: dist/felis-linux-amd64
|
||||
format: cyclonedx-json
|
||||
output-file: felis-linux-amd64.cdx.json
|
||||
output-file: sbom/felis-linux-amd64.cdx.json
|
||||
upload-artifact: false
|
||||
upload-release-assets: false
|
||||
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0.24.0
|
||||
with:
|
||||
file: felis-linux-arm64
|
||||
file: dist/felis-linux-arm64
|
||||
format: cyclonedx-json
|
||||
output-file: felis-linux-arm64.cdx.json
|
||||
output-file: sbom/felis-linux-arm64.cdx.json
|
||||
upload-artifact: false
|
||||
upload-release-assets: false
|
||||
# Outside SHA256SUMS, which lists what bootstrap installs; beside it in the release.
|
||||
- run: mv sbom/*.cdx.json dist/
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
with:
|
||||
name: release-assets
|
||||
path: |
|
||||
felis-linux-amd64
|
||||
felis-linux-arm64
|
||||
felis-linux-amd64.cdx.json
|
||||
felis-linux-arm64.cdx.json
|
||||
SHA256SUMS
|
||||
path: dist/
|
||||
if-no-files-found: error
|
||||
retention-days: 7
|
||||
|
||||
@@ -134,7 +113,7 @@ jobs:
|
||||
- run: sha256sum -c SHA256SUMS
|
||||
|
||||
# Signed SLSA provenance: which workflow run, commit and repository produced each
|
||||
# binary. Check one with `gh attestation verify felis-linux-amd64 --repo FelisMC/Felis`.
|
||||
# asset. Check one with `gh attestation verify felis-linux-amd64 --repo FelisMC/Felis`.
|
||||
# GitHub only stores attestations for private repositories on Enterprise Cloud, and a
|
||||
# failure here would block the release, so a private repository skips the step and
|
||||
# relies on SHA256SUMS alone.
|
||||
@@ -143,8 +122,9 @@ jobs:
|
||||
uses: actions/attest-build-provenance@977bb373ede98d70efdf65b84cb5f73e068dcc2a # v3.0.0
|
||||
with:
|
||||
subject-path: |
|
||||
felis-linux-amd64
|
||||
felis-linux-arm64
|
||||
felis-linux-*
|
||||
felis-image-*.tar
|
||||
felis-velocity.jar
|
||||
|
||||
# --verify-tag refuses to invent a release for a tag that is not pushed.
|
||||
#
|
||||
@@ -164,7 +144,9 @@ jobs:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
run: |
|
||||
assets="felis-linux-amd64 felis-linux-arm64 felis-linux-amd64.cdx.json felis-linux-arm64.cdx.json SHA256SUMS"
|
||||
# Every file the build handed over: SHA256SUMS names the installable ones, and the
|
||||
# SBOMs ride along.
|
||||
assets="$(ls)"
|
||||
flags=""
|
||||
case "$GITHUB_REF_NAME" in *-*) flags="--prerelease" ;; esac
|
||||
if ! gh release view "$GITHUB_REF_NAME" >/dev/null 2>&1; then
|
||||
|
||||
+9
-2
@@ -78,8 +78,15 @@ FELIS_TEST_PG_URL='postgres://felis:***@127.0.0.1:5432/felis_pgint?sslmode=disab
|
||||
```
|
||||
|
||||
Run it after touching anything under `internal/api/pgrepo.go`, `internal/submit`,
|
||||
or `internal/build` that speaks SQL: the fakes encode the contract, and this
|
||||
suite exists to catch the drift between the fakes and the real queries.
|
||||
`internal/build` or `internal/dbbackup` that speaks SQL: the fakes encode the
|
||||
contract, and this suite exists to catch the drift between the fakes and the real
|
||||
queries. The `felis db backup` and `restore` tests also run `pg_dump`, `pg_restore`
|
||||
and `psql`, which must be the server's major version. For a server in a container,
|
||||
run them in it, as production does in felis-postgres:
|
||||
|
||||
```bash
|
||||
FELIS_TEST_PG_EXEC='docker exec -i <container>' FELIS_TEST_PG_URL=... go test -tags pgint ./internal/pgint/
|
||||
```
|
||||
|
||||
Build the CLI:
|
||||
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
# Felis
|
||||
|
||||
**此项目仍处于早期开发阶段,您不该在任何生产环境使用该项目。若产生任何问题,贵用户的使用行为与 FelisMC 团队无任何民事刑事法律关系。**
|
||||
**THIS PROJECT IS STILL WIP, YOU SHOULD DO NOT USE THIS PROJECT IN ANY PRODUCTION USAGE. WE ARE NOT RESPOND FOR ANY LEGAL OR HUMANLY PROBLEM.**
|
||||
|
||||
一款 Kubernetes 驱动的 Minecraft 服务器托管平台,一行命令部署,自动管理生命周期与安全。
|
||||
A Kubernetes-driven Minecraft server hosting platform — one command to deploy, automatic lifecycle, backup, and security.
|
||||
|
||||
@@ -19,6 +22,7 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
||||
- **Web 控制面板**:浏览器中查看服务器状态、在线玩家与资源用量,管理备份与恢复。
|
||||
- **备份与恢复**:一键把整服数据(世界、配置、插件/模组,即整个 /data 卷)打包进集群内的归档库,支持从任意备份点回滚;默认安装就已启用(归档 PVC 与路径由安装器一并生成)。
|
||||
- **控制面数据库备份**:账号、服务器归属、配额与存档索引所在的数据库每天自动备份,每次升级迁移前先快照,出错可用 `felis db restore` 整库原子回滚;面板「维护与备份」页显示备份是否新鲜(见 [故障排查 §16](docs/troubleshooting.md))。
|
||||
- **运维自检**:`sudo felis status` 一屏列出节点、控制面、游戏代理、每台服务器、备份与未解决的告警;`sudo felis doctor` 把健康检查全跑一遍,按区域给出问题和下一步去哪看,不发邮件;`sudo felis support-bundle` 打出一个脱敏的诊断包,求助时直接附上(见 [故障排查 §0](docs/troubleshooting.md))。
|
||||
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。过期备份无论是否开启都会每天清理。
|
||||
- **多核心支持**:兼容 Paper、Fabric、Forge、NeoForge,经由 Velocity 代理统一入口。
|
||||
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建;构建产物进入镜像白名单,可直接选用为服务器镜像完成部署。
|
||||
@@ -27,13 +31,17 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
||||
|
||||
## 使用方式
|
||||
|
||||
在准备好的 Linux 主机上执行(已验证的发行版与架构见 [运维手册 §1](docs/operations.md#1-supported-hosts):CentOS Stream 9 aarch64 实机验证,Ubuntu 24.04 x86_64 每次推送由 CI 跑全新安装、重跑与升级):
|
||||
在准备好的 Linux 主机上执行(已验证的发行版与架构见 [运维手册 §1](docs/operations.md#1-supported-hosts):CentOS Stream 9 aarch64 实机验证,Ubuntu 24.04 x86_64 每次推送由 CI 跑全新安装、重跑、升级和下面这条命令本身):
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
```
|
||||
|
||||
脚本将自动安装 K3s、部署控制平面并启动设置向导。完成后浏览器访问已配置的域名进入控制面板即可使用。
|
||||
脚本将自动安装 K3s,在 K3s 内部署 PostgreSQL 与控制平面,并启动设置向导。完成后浏览器访问已配置的域名进入控制面板即可使用。
|
||||
|
||||
安装发布版时,二进制、全部镜像与 Velocity 插件都取自该版本在 CI 里预构建好的 release 附件,逐个核对 `SHA256SUMS` 后导入,主机上无需 Docker、Gradle 或 Go,也不从 Docker Hub 拉取;某个附件缺失或校验不符时,只有那一个镜像退回到本机构建,并给出提示(见 [故障排查 §15c](docs/troubleshooting.md))。附件也可以先拷到本机,再用 `FELIS_ARTIFACT_DIR=<绝对路径>` 安装,Felis 自己的二进制、镜像和插件就都取自这个目录;k3s 及其镜像、JRE、cloudflared、Velocity 和 Via 插件照旧从 GitHub 与 PaperMC 下载,RHEL、Fedora、openSUSE Leap 这类开着 SELinux 的主机还要从 rpm.rancher.io 装 k3s-selinux,系统软件包来自发行版的源。所以出网受限的主机要放行这几处的 HTTPS(或设 `https_proxy`),preflight 会在改动主机之前逐个探测,完全断网的主机目前装不了(地址清单见 [运维手册 §1](docs/operations.md#1-supported-hosts))。旧版本装在宿主上的 PostgreSQL 会在重跑时整库迁进 K3s,宿主上的那份停用保留,供回退(见 [运维手册 §4](docs/operations.md#4-upgrading-the-pieces-around-felis))。
|
||||
|
||||
动手之前,脚本先检查内存、磁盘、端口、网段冲突、已有的 Kubernetes 和外网连通,把所有问题一次列出并停下,主机上什么都没改(检查项见 [运维手册 §1](docs/operations.md#1-supported-hosts))。
|
||||
|
||||
> **本仓库当前为私有**,上面这条会返回 404。请改用带凭据的形式;安装器自身也需要同一个 token
|
||||
> 去解析并下载 release,所以用 `sudo -E` 把它带进去:
|
||||
|
||||
+4
-1
@@ -19,6 +19,7 @@ Table of Contents
|
||||
- **Web Dashboard**: Monitor server status, online players, and resource usage from your browser, with backup and restore management.
|
||||
- **Backup & Restore**: One-click snapshots of a server's whole data volume (worlds, config, plugins/mods — the entire /data volume) into the cluster's archive store, with rollback from any backup point — enabled by default (the installer renders the archive PVC and its path).
|
||||
- **Control-plane database backups**: The database holding accounts, server ownership, quotas and the archive index is backed up daily and snapshotted before every upgrade migrates it; `felis db restore` rolls it back atomically, and the panel's Maintenance & Backups page shows whether the newest backup is fresh (see [troubleshooting §16](docs/troubleshooting.md)).
|
||||
- **Self-check for operators**: `sudo felis status` shows the node, the control plane, the game proxy, every server, the backups and the open alerts on one screen; `sudo felis doctor` runs every health check once and lists each problem by area with where to look next, mailing nothing; `sudo felis support-bundle` writes one redacted diagnostics archive to attach when asking for help (see [troubleshooting §0](docs/troubleshooting.md)).
|
||||
- **World Reaper** (opt in): Worlds idle for more than 15 days are automatically backed up and removed to free disk space. Enable it by setting `FELIS_WORLDS_HOST_PATH` at install time (on k3s: `/var/lib/rancher/k3s/storage`); without it, no world is ever deleted.
|
||||
- **Multi-core Support**: Compatible with Paper, Fabric, Forge, and NeoForge, federated behind a Velocity proxy.
|
||||
- **Modpack Submission**: Players submit custom modpacks; admin approval triggers an automatic build, and the result is whitelisted as a server image you can select to deploy.
|
||||
@@ -36,7 +37,9 @@ every push):
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
```
|
||||
|
||||
The script installs K3s, deploys the control plane, and launches a setup wizard. Once done, open your browser at the configured domain.
|
||||
The script installs K3s, deploys PostgreSQL and the control plane inside it, and launches a setup wizard. Once done, open your browser at the configured domain. A PostgreSQL an earlier release installed on the host is moved into K3s on the next rerun, and the host copy is stopped and kept for a rollback (see [Operations §4](docs/operations.md#4-upgrading-the-pieces-around-felis)).
|
||||
|
||||
Before it changes anything, the script checks RAM, disk, ports, network-range clashes, any Kubernetes already there and outbound access, lists every problem at once and stops with the host untouched (the checks are in [operations §1](docs/operations.md#1-supported-hosts)).
|
||||
|
||||
> **This repository is currently private**, so the command above returns 404. Use the
|
||||
> credentialed form instead; the installer itself needs the same token to resolve and
|
||||
|
||||
+18
-5
@@ -304,16 +304,19 @@ func TestDockerfileBaseImagesArePinnedByDigest(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The plugin jars are built in three places the installer controls — the lobby and limbo
|
||||
// image builds and bootstrap's Velocity build — and through each module's wrapper by a
|
||||
// developer or CI. A tag alone is whatever it points at on build day, and two Gradle
|
||||
// versions are two chances for a build to pass in one place and break in the other, so
|
||||
// all of them run one image, pinned by digest, whose Gradle is the wrappers' Gradle.
|
||||
// The plugin jars are built in four places the installer controls — the lobby and limbo
|
||||
// image builds, bootstrap's Velocity build and the release build of the same jar — and
|
||||
// through each module's wrapper by a developer or CI. A tag alone is whatever it points
|
||||
// at on build day, and two Gradle versions are two chances for a build to pass in one
|
||||
// place and break in the other, so all of them run one image, pinned by digest, whose
|
||||
// Gradle is the wrappers' Gradle.
|
||||
func TestPluginBuildsRunOnePinnedGradle(t *testing.T) {
|
||||
sources := map[string]string{
|
||||
"deploy/lobby/Dockerfile": readGameStackFile(t, "deploy/lobby/Dockerfile"),
|
||||
"deploy/limbo/Dockerfile": readGameStackFile(t, "deploy/limbo/Dockerfile"),
|
||||
"deploy/bootstrap.sh": BootstrapScript(),
|
||||
// The release build compiles felis-velocity.jar for hosts that install prebuilt.
|
||||
"deploy/build-release-artifacts.sh": readRepoFile(t, "deploy/build-release-artifacts.sh"),
|
||||
}
|
||||
anyRef := regexp.MustCompile(`gradle:[\w.-]+(@sha256:\w+)?`)
|
||||
pinned := regexp.MustCompile(`^gradle:(\d+\.\d+(?:\.\d+)?)-jdk\d+@sha256:[0-9a-f]{64}$`)
|
||||
@@ -487,6 +490,16 @@ func gameStackLock(t *testing.T) map[string]string {
|
||||
return lock
|
||||
}
|
||||
|
||||
// readRepoFile reads a file of the checkout that the binary does not embed.
|
||||
func readRepoFile(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile(name)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
func readGameStackFile(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
b, err := gameStackAssets.ReadFile(name)
|
||||
|
||||
+173
-18
@@ -33,6 +33,7 @@ import (
|
||||
"felis.lolicon.best/internal/restore"
|
||||
"felis.lolicon.best/internal/retention"
|
||||
"felis.lolicon.best/internal/submit"
|
||||
"felis.lolicon.best/internal/worldexport"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
@@ -49,6 +50,22 @@ import (
|
||||
// the code marks Identity (UUIDs trusted verbatim); config can never add another.
|
||||
const mojangSessionServer = "https://sessionserver.mojang.com/session/minecraft/hasJoined"
|
||||
|
||||
// passkeyRelyingParty is the one WebAuthn relying party both web faces share: its id
|
||||
// is the player console host, derived from server.root_domain the way the panel
|
||||
// handler derives it when auth.panel_hostname is unset, and its origins are that host
|
||||
// plus the operator host. An empty id means the install names no panel host at all.
|
||||
func passkeyRelyingParty(cfg *config.Config) (string, []string) {
|
||||
rpID := defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname)
|
||||
if rpID == "" {
|
||||
return "", nil
|
||||
}
|
||||
origins := []string{"https://" + rpID}
|
||||
if admin := defaultAdminHostname(cfg.Server.RootDomain, cfg.Auth.AdminHostname); admin != "" && admin != rpID {
|
||||
origins = append(origins, "https://"+admin)
|
||||
}
|
||||
return rpID, origins
|
||||
}
|
||||
|
||||
// authSourcesFromConfig builds the multiplexer's priority list from the configured
|
||||
// [[auth_source]] entries: Mojang leads as the code-owned identity anchor (正版优先, the ONLY
|
||||
// Identity source — config can only append namespace-rewritten third-party sources, never a
|
||||
@@ -100,7 +117,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
|
||||
// Before anything serves: an api on a schema it was not built for answers with
|
||||
// errors, or writes rows the other version cannot read.
|
||||
drv, err := openStore(ctx, cfg.Database.URL, false)
|
||||
drv, err := openPodStore(ctx, cfg.Database.URL, "api", stderr)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: open database: %v\n", err)
|
||||
return 1
|
||||
@@ -276,6 +293,17 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintln(stderr, "felis api: backup executor disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — backup endpoint returns 503")
|
||||
}
|
||||
|
||||
// World export: a weak-SA Job mounts the world PVC, or the backup PVC, read-only
|
||||
// and PUTs the archive to the internal face, which streams it on to the owner's
|
||||
// browser (internal/worldexport). A backup export mounts the backup PVC, so it
|
||||
// is wired under the restore gate; otherwise the export routes return 503.
|
||||
var exporter api.Exporter
|
||||
if felisImage != "" && backupPVC != "" {
|
||||
exporter = worldexport.New(clientset, exportConfig(cfg, felisImage, backupPVC))
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis api: world export disabled (needs FELIS_IMAGE and FELIS_BACKUP_PVC) — export endpoints return 503")
|
||||
}
|
||||
|
||||
// Server file editor: a weak-SA Job mounts ONLY the target world PVC and runs
|
||||
// `felis files`, printing its result for felis-api to read back through
|
||||
// pods/log (see internal/fileedit). It needs FELIS_IMAGE but — unlike restore
|
||||
@@ -284,12 +312,22 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// endpoints honestly return 503. It takes the typed clientset rather than the
|
||||
// controller-runtime client because the log subresource lives only on the typed
|
||||
// CoreV1 client, and one client covers its Job create, Pod list, and log read.
|
||||
//
|
||||
// Uploads additionally stage their bytes on this pod's disk until the Job
|
||||
// fetches them from the internal face; whatever a previous process staged is
|
||||
// orphaned (the index is in memory), so the stage starts empty.
|
||||
var files api.FileEditor
|
||||
var fileStage *fileedit.Stage
|
||||
if felisImage != "" {
|
||||
files = &fileedit.Editor{
|
||||
Runner: fileedit.NewK8sRunner(clientset),
|
||||
Config: fileEditConfig(cfg, felisImage),
|
||||
}
|
||||
fileStage = &fileedit.Stage{Dir: fileStagingDir()}
|
||||
if err := fileStage.Sweep(); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: %v — file uploads return 503\n", err)
|
||||
fileStage = nil
|
||||
}
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis api: file editor disabled (needs FELIS_IMAGE) — file endpoints return 503")
|
||||
}
|
||||
@@ -337,8 +375,15 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// restore behind each one.
|
||||
RestoreChains: jobStatus,
|
||||
Files: files,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
FileStage: fileStage,
|
||||
// The file Job fetches an upload from here; it runs in the minecraft
|
||||
// namespace, where the internal face is reachable like it is for the login
|
||||
// gate.
|
||||
InternalBaseURL: internalAPIBaseURL(),
|
||||
Exporter: exporter,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
Schedules: repo,
|
||||
// The external face authenticates the local session cookie the sign-in doors
|
||||
// mint, live once `felis breakGlass` flips local_auth_enabled on. Cloudflare
|
||||
// Access, when the install sits behind it, is enforced at the edge only.
|
||||
@@ -393,22 +438,15 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// enrolled once asserts on either face — one binding, usable on the player console
|
||||
// AND the operator console. Both hosts are therefore listed as permitted origins,
|
||||
// while the RP id stays the panel host so the credential's scope is ONE relying
|
||||
// party, not two. Wired only when auth.panel_hostname is configured; otherwise
|
||||
// a.Passkey stays nil and the passkey routes honestly return 503 (the authenticated
|
||||
// enrollment boundary is still enforced by the handlers).
|
||||
if cfg.Auth.PanelHostname != "" {
|
||||
origins := []string{"https://" + cfg.Auth.PanelHostname}
|
||||
if admin := defaultAdminHostname(cfg.Server.RootDomain, cfg.Auth.AdminHostname); admin != "" && admin != cfg.Auth.PanelHostname {
|
||||
origins = append(origins, "https://"+admin)
|
||||
}
|
||||
pv, err := passkey.New(cfg.Auth.PanelHostname, "Felis", origins)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: passkey verifier disabled: %v — passkey endpoints return 503\n", err)
|
||||
} else {
|
||||
a.Passkey = pv
|
||||
}
|
||||
// party, not two. Without a panel host (neither auth.panel_hostname nor
|
||||
// server.root_domain) a.Passkey stays nil and the passkey routes honestly return
|
||||
// 503 (the authenticated enrollment boundary is still enforced by the handlers).
|
||||
if rpID, origins := passkeyRelyingParty(cfg); rpID == "" {
|
||||
fmt.Fprintln(stderr, "felis api: passkey verifier disabled (no panel host: set server.root_domain or auth.panel_hostname) — passkey endpoints return 503")
|
||||
} else if pv, err := passkey.New(rpID, "Felis", origins); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: passkey verifier disabled: %v — passkey endpoints return 503\n", err)
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis api: passkey verifier disabled (auth.panel_hostname unset) — passkey endpoints return 503")
|
||||
a.Passkey = pv
|
||||
}
|
||||
|
||||
// Derive the console hostnames when felis.toml leaves them unset, exactly as the
|
||||
@@ -440,11 +478,25 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// reconciles it, but this loop converges builds nobody is polling.
|
||||
go reconcileBuilds(ctx, builder, stderr)
|
||||
go settleRestoreChains(ctx, a, stderr)
|
||||
go runSchedules(ctx, a, stderr)
|
||||
// A daily restore point of every world played since its last one, taken
|
||||
// once the server stops ([archive] scheduled_every; 0s turns it off).
|
||||
if backuper != nil && rcfg.ScheduledEvery > 0 {
|
||||
go scheduleBackups(ctx, &api.BackupScheduler{API: a, Store: repo, Jobs: jobStatus, Every: rcfg.ScheduledEvery}, stderr)
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis api: scheduled backups off (needs the backup executor and [archive] scheduled_every above 0s)")
|
||||
}
|
||||
|
||||
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
|
||||
go pruner.Loop(ctx, registryPruneInterval)
|
||||
}
|
||||
go reapRejectedContexts(ctx, submissions, stderr)
|
||||
if fileStage != nil {
|
||||
go expireFileSessions(ctx, fileStage, fileSessionSweep, stderr)
|
||||
}
|
||||
if exporter != nil {
|
||||
go expireExports(ctx, a, exportSweep)
|
||||
}
|
||||
go retention.Loop(ctx, drv.DB(), retention.Policy{Audit: auditRetention}, retentionInterval, slog.Default())
|
||||
|
||||
servers := []*http.Server{internalSrv, externalSrv}
|
||||
@@ -654,6 +706,17 @@ func backupConfig(cfg *config.Config, image, backupPVC string) backupjob.Config
|
||||
}
|
||||
}
|
||||
|
||||
// exportConfig builds the world export executor's config. BackupRoot mirrors
|
||||
// restoreConfig: the stored refs are absolute paths under [archive] local_path.
|
||||
func exportConfig(cfg *config.Config, image, backupPVC string) worldexport.Config {
|
||||
return worldexport.Config{
|
||||
Namespace: cfg.K8s.Namespace,
|
||||
Image: image,
|
||||
BackupPVC: backupPVC,
|
||||
BackupRoot: cfg.Archive.LocalPath,
|
||||
}
|
||||
}
|
||||
|
||||
// fileEditConfig builds the file editor's config from felis.toml plus the
|
||||
// deployment-supplied image. It is the shortest of the three: the editor mounts
|
||||
// only the world PVC, so it needs no archive coordinates at all, and everything
|
||||
@@ -703,6 +766,88 @@ func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
|
||||
}
|
||||
}
|
||||
|
||||
// runSchedules runs the servers' scheduled tasks (api.API.RunSchedules). The
|
||||
// interval is how late a task may start, and how often a restart or backup in
|
||||
// progress checks whether it can take its next step.
|
||||
func runSchedules(ctx context.Context, a *api.API, stderr io.Writer) {
|
||||
t := time.NewTicker(15 * time.Second)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
if err := a.RunSchedules(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: scheduled tasks: %v\n", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// scheduleBackups starts the scheduled backups (api.BackupScheduler). Each tick
|
||||
// starts at most one, so the interval also spaces the worlds that stopped at
|
||||
// the same time: a world that stops waits at most this long for its point to
|
||||
// start once the Jobs ahead of it are done.
|
||||
func scheduleBackups(ctx context.Context, s *api.BackupScheduler, stderr io.Writer) {
|
||||
t := time.NewTicker(2 * time.Minute)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
if err := s.Tick(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: scheduled backups: %v\n", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// fileSessionSweep is how often expireFileSessions looks for idle upload
|
||||
// sessions: small beside fileedit.SessionIdle, so an abandoned one gives its
|
||||
// room back within minutes of going stale.
|
||||
const fileSessionSweep = 10 * time.Minute
|
||||
|
||||
// expireFileSessions drops the file manager's upload sessions left untouched
|
||||
// for fileedit.SessionIdle. Each reserved room on the staging disk for its whole
|
||||
// file when it began, so one abandoned would otherwise hold that room until
|
||||
// felis-api restarts.
|
||||
func expireFileSessions(ctx context.Context, s *fileedit.Stage, every time.Duration, stderr io.Writer) {
|
||||
t := time.NewTicker(every)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
if n := s.Expire(); n > 0 {
|
||||
fmt.Fprintf(stderr, "felis api: dropped %d upload session(s) left idle for %s\n", n, fileedit.SessionIdle)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// exportSweep is how often expireExports runs: an export whose Job never
|
||||
// connected is stopped within a minute of going stale.
|
||||
const exportSweep = time.Minute
|
||||
|
||||
// expireExports runs the export sweep (api.API.ExpireExports) on a ticker. The
|
||||
// export routes sweep as they are called, and an owner who closed the tab calls
|
||||
// none; a Job whose Pod never got going would then keep the server from
|
||||
// starting until the Job's deadline.
|
||||
func expireExports(ctx context.Context, a interface{ ExpireExports() }, every time.Duration) {
|
||||
t := time.NewTicker(every)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
a.ExpireExports()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
|
||||
// submissions rejected more than submit.RejectedContextRetention ago, and the
|
||||
// chunked uploads left untouched for submit.StalePartRetention. Without it a
|
||||
@@ -912,6 +1057,16 @@ func uploadPartsDir(contextBase string) string {
|
||||
return filepath.Join(os.TempDir(), "felis-upload-parts")
|
||||
}
|
||||
|
||||
// fileStagingDir is where file uploads wait for their Job: on the uploads
|
||||
// volume, whose capacity is its own, or the pod's /tmp when run by hand without
|
||||
// it — /tmp is the node's disk, which a burst of uploads should not fill.
|
||||
func fileStagingDir() string {
|
||||
if fi, err := os.Stat(platform.UploadsLocalPath); err == nil && fi.IsDir() {
|
||||
return filepath.Join(platform.UploadsLocalPath, ".file-staging")
|
||||
}
|
||||
return filepath.Join(os.TempDir(), "felis-file-staging")
|
||||
}
|
||||
|
||||
// contextMaxBytes resolves [registry] context_max_bytes. 0 keeps the submit
|
||||
// package's own default (1 GiB). The Cloudflare edge refuses a single request
|
||||
// body over 100 MB, which the panel's chunked upload stays under, so the edge
|
||||
|
||||
@@ -1,12 +1,15 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"slices"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -14,8 +17,40 @@ import (
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/build"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
)
|
||||
|
||||
// The passkey relying party follows the panel host the SPA is served on: an install
|
||||
// that names only its root domain still gets passkeys, on console.<root>, with the
|
||||
// operator host as the second origin; only an install with no panel host goes without.
|
||||
func TestPasskeyRelyingParty(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
root, panel, admin string
|
||||
wantRP string
|
||||
wantOrigins []string
|
||||
}{
|
||||
{"root domain only", "example.net", "", "", "console.example.net",
|
||||
[]string{"https://console.example.net", "https://op.console.example.net"}},
|
||||
{"configured hosts", "example.net", " play.example.net ", "ops.example.net", "play.example.net",
|
||||
[]string{"https://play.example.net", "https://ops.example.net"}},
|
||||
{"operator host equal to the panel host", "example.net", "console.example.net", "console.example.net",
|
||||
"console.example.net", []string{"https://console.example.net"}},
|
||||
{"panel host without a root domain", "", "console.example.org", "", "console.example.org",
|
||||
[]string{"https://console.example.org"}},
|
||||
{"no host at all", "", "", "", "", nil},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
cfg := &config.Config{}
|
||||
cfg.Server.RootDomain, cfg.Auth.PanelHostname, cfg.Auth.AdminHostname = tc.root, tc.panel, tc.admin
|
||||
rp, origins := passkeyRelyingParty(cfg)
|
||||
if rp != tc.wantRP || !slices.Equal(origins, tc.wantOrigins) {
|
||||
t.Fatalf("relying party = %q %q, want %q %q", rp, origins, tc.wantRP, tc.wantOrigins)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestAuthSourcesFromConfig pins the one place the hasJoined identity anchor is decided:
|
||||
// Mojang is prepended in code, first, and is the only source whose UUIDs are trusted as-is.
|
||||
// The empty case matters on its own — both `felis api` and `felis nano` call this with a
|
||||
@@ -203,3 +238,76 @@ func TestInUseImageRefsCoversEverySource(t *testing.T) {
|
||||
t.Fatal("a failing whitelist read produced a reference list")
|
||||
}
|
||||
}
|
||||
|
||||
// TestExpireFileSessions runs the loop against a stage whose clock the test
|
||||
// holds: the session idle past fileedit.SessionIdle goes, the one touched since
|
||||
// stays, and the drop is said once.
|
||||
func TestExpireFileSessions(t *testing.T) {
|
||||
var mu sync.Mutex
|
||||
now := time.Date(2026, 9, 28, 10, 0, 0, 0, time.UTC)
|
||||
advance := func(d time.Duration) { mu.Lock(); now = now.Add(d); mu.Unlock() }
|
||||
st := &fileedit.Stage{Dir: t.TempDir(), MinFree: 1e-9, Now: func() time.Time {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
return now
|
||||
}}
|
||||
idle, err := st.Begin("u1", "survival", "a.jar", 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
advance(fileedit.SessionIdle - time.Minute)
|
||||
fresh, err := st.Begin("u1", "survival", "b.jar", 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
advance(2 * time.Minute)
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
var out bytes.Buffer
|
||||
done := make(chan struct{})
|
||||
go func() { expireFileSessions(ctx, st, time.Millisecond, &out); close(done) }()
|
||||
for deadline := time.Now().Add(5 * time.Second); ; time.Sleep(time.Millisecond) {
|
||||
if _, err := st.Status("u1", "survival", idle.ID); errors.Is(err, fileedit.ErrNotStaged) {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
cancel()
|
||||
t.Fatal("the idle session was never dropped")
|
||||
}
|
||||
}
|
||||
// A few more ticks with nothing idle, which must stay quiet.
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
cancel()
|
||||
<-done
|
||||
if _, err := st.Status("u1", "survival", fresh.ID); err != nil {
|
||||
t.Fatalf("the session touched since went too: %v", err)
|
||||
}
|
||||
if got := out.String(); got != "felis api: dropped 1 upload session(s) left idle for 6h0m0s\n" {
|
||||
t.Fatalf("said %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
type sweepCount struct{ n atomic.Int32 }
|
||||
|
||||
func (s *sweepCount) ExpireExports() { s.n.Add(1) }
|
||||
|
||||
// TestExpireExports: the loop sweeps on each tick, and returns once felis-api
|
||||
// shuts down.
|
||||
func TestExpireExports(t *testing.T) {
|
||||
var s sweepCount
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan struct{})
|
||||
go func() { expireExports(ctx, &s, time.Millisecond); close(done) }()
|
||||
for deadline := time.Now().Add(5 * time.Second); s.n.Load() < 3; time.Sleep(time.Millisecond) {
|
||||
if time.Now().After(deadline) {
|
||||
cancel()
|
||||
t.Fatalf("swept %d times in 5s at a 1ms tick", s.n.Load())
|
||||
}
|
||||
}
|
||||
cancel()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("the loop outlived its context")
|
||||
}
|
||||
}
|
||||
+16
-20
@@ -8,7 +8,9 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strings"
|
||||
"syscall"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
@@ -16,10 +18,6 @@ import (
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
@@ -111,20 +109,15 @@ func cmdApply(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
// ------- K8s client (one context, one client) -------
|
||||
// SetupSignalHandler must be called exactly once per process —
|
||||
// controller-runtime panics on a second call. We create ctx and the
|
||||
// K8s client here and thread both through every downstream call so no
|
||||
// callee ever needs to call SetupSignalHandler again.
|
||||
ctx := ctrl.SetupSignalHandler()
|
||||
|
||||
scheme := runtime.NewScheme()
|
||||
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
|
||||
utilruntime.Must(v1alpha1.AddToScheme(scheme))
|
||||
|
||||
cfg := ctrl.GetConfigOrDie()
|
||||
cl, err := client.New(cfg, client.Options{Scheme: scheme})
|
||||
// The operator runs this on the node, where the kubeconfig is k3s's own file and
|
||||
// neither $KUBECONFIG nor ~/.kube is set. buildSystemServerClient falls back to that
|
||||
// file and names what it tried; ctrl.GetConfigOrDie exited 1 there without a word,
|
||||
// because controller-runtime's logger is never set up in a CLI command.
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis apply: build client: %v\n", err)
|
||||
fmt.Fprintf(stderr, "felis apply: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
@@ -167,6 +160,10 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
|
||||
if err := naming.ValidateServerName(req.Subdomain); err != nil {
|
||||
return nil, fmt.Errorf("invalid subdomain: %w", err)
|
||||
}
|
||||
displayName, err := naming.CleanDisplayName(req.DisplayName)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("invalid displayName: %w", err)
|
||||
}
|
||||
if strings.TrimSpace(req.Image) == "" {
|
||||
return nil, fmt.Errorf("image is required")
|
||||
}
|
||||
@@ -236,7 +233,7 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
|
||||
},
|
||||
Spec: v1alpha1.MinecraftServerSpec{
|
||||
Subdomain: req.Subdomain,
|
||||
DisplayName: req.DisplayName,
|
||||
DisplayName: displayName,
|
||||
Image: req.Image,
|
||||
JavaMemory: deriveApplyJavaHeap(memLim),
|
||||
DesiredState: v1alpha1.DesiredStopped,
|
||||
@@ -264,8 +261,7 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
|
||||
// request if any CRD already carries the given spec.subdomain. metadata.name
|
||||
// uniqueness is enforced by K8s on Create, but spec.subdomain must be checked
|
||||
// here because two CRDs with different names could otherwise share a subdomain.
|
||||
// It reuses the caller's context and K8s client — it never calls
|
||||
// SetupSignalHandler or builds its own client.
|
||||
// It reuses the caller's context and K8s client.
|
||||
func checkSubdomainUnique(ctx context.Context, cl client.Client, namespace, subdomain string) error {
|
||||
var list v1alpha1.MinecraftServerList
|
||||
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
|
||||
|
||||
+41
-5
@@ -1,7 +1,10 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
@@ -137,11 +140,12 @@ func TestDeriveApplyJavaHeap(t *testing.T) {
|
||||
|
||||
func TestBuildMinecraftServerFromApplyRequest_Valid(t *testing.T) {
|
||||
req := applyRequest{
|
||||
Name: "test-server",
|
||||
Subdomain: "test-server",
|
||||
Image: "registry.felis.svc/paper:1.21",
|
||||
Memory: "4Gi",
|
||||
Storage: "20Gi",
|
||||
Name: "test-server",
|
||||
Subdomain: "test-server",
|
||||
DisplayName: " Test Server ",
|
||||
Image: "registry.felis.svc/paper:1.21",
|
||||
Memory: "4Gi",
|
||||
Storage: "20Gi",
|
||||
}
|
||||
ms, err := buildMinecraftServerFromApplyRequest(req, "minecraft")
|
||||
if err != nil {
|
||||
@@ -156,6 +160,9 @@ func TestBuildMinecraftServerFromApplyRequest_Valid(t *testing.T) {
|
||||
if ms.Spec.Subdomain != "test-server" {
|
||||
t.Errorf("Subdomain = %q", ms.Spec.Subdomain)
|
||||
}
|
||||
if ms.Spec.DisplayName != "Test Server" {
|
||||
t.Errorf("DisplayName = %q, want it trimmed to Test Server", ms.Spec.DisplayName)
|
||||
}
|
||||
if ms.Spec.Image != "registry.felis.svc/paper:1.21" {
|
||||
t.Errorf("Image = %q", ms.Spec.Image)
|
||||
}
|
||||
@@ -280,6 +287,11 @@ func TestBuildMinecraftServerFromApplyRequest_Errors(t *testing.T) {
|
||||
applyRequest{Name: ok, Subdomain: "", Image: "x", Memory: "1Gi", Storage: "1Gi"},
|
||||
"invalid subdomain",
|
||||
},
|
||||
{
|
||||
"display name with a tab",
|
||||
applyRequest{Name: ok, Subdomain: ok, DisplayName: "a" + string(rune(0x09)) + "b", Image: "x", Memory: "1Gi", Storage: "1Gi"},
|
||||
"invalid displayName",
|
||||
},
|
||||
{
|
||||
"empty image",
|
||||
applyRequest{Name: ok, Subdomain: ok, Image: "", Memory: "1Gi", Storage: "1Gi"},
|
||||
@@ -370,3 +382,27 @@ func resList(specs ...string) corev1.ResourceList {
|
||||
}
|
||||
return rl
|
||||
}
|
||||
|
||||
// TestApplyReportsAMissingKubeconfig pins the node-side failure: with no kubeconfig to
|
||||
// find, apply says which ones it tried and exits 1. It used to call
|
||||
// ctrl.GetConfigOrDie, which ended the process with exit 1 and nothing printed.
|
||||
func TestApplyReportsAMissingKubeconfig(t *testing.T) {
|
||||
if _, err := os.Stat(hostBootstrapKubeconfigPath); err == nil {
|
||||
t.Skipf("%s exists on this machine", hostBootstrapKubeconfigPath)
|
||||
}
|
||||
dir := t.TempDir()
|
||||
t.Setenv("KUBECONFIG", filepath.Join(dir, "missing"))
|
||||
t.Setenv("KUBERNETES_SERVICE_HOST", "")
|
||||
form := filepath.Join(dir, "server.json")
|
||||
if err := os.WriteFile(form, []byte(`{"name":"alpha","subdomain":"alpha","image":"registry.felis.svc:5000/felis/paper:demo","memory":"1Gi","storage":"1Gi"}`), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var out, errw bytes.Buffer
|
||||
if code := cmdApply([]string{"-f", form}, &out, &errw); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1; stderr %q", code, errw.String())
|
||||
}
|
||||
want := "felis apply: no reachable kubeconfig (tried in-cluster/$KUBECONFIG/~/.kube and " + hostBootstrapKubeconfigPath + "): stat " + hostBootstrapKubeconfigPath + ": no such file or directory\n"
|
||||
if errw.String() != want || out.Len() != 0 {
|
||||
t.Fatalf("stdout %q, stderr %q, want stderr %q", out.String(), errw.String(), want)
|
||||
}
|
||||
}
|
||||
+35
-20
@@ -15,7 +15,6 @@ import (
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
"felis.lolicon.best/internal/store"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
)
|
||||
|
||||
@@ -38,7 +37,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
server := fs.String("server", "", "server name whose world is being backed up")
|
||||
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
|
||||
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
|
||||
reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
|
||||
reason := fs.String("reason", reasonManual, "world_backups reason: manual, pre_restore for the safety snapshot in front of a restore, or scheduled for felis-api's daily restore point")
|
||||
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
@@ -47,9 +46,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintln(stderr, "felis backup: --server is required")
|
||||
return 2
|
||||
}
|
||||
keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
|
||||
if !ok {
|
||||
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
|
||||
if _, _, ok := backupPolicy(*reason, reaper.DefaultConfig()); !ok {
|
||||
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual, %s or %s)\n", *reason, backupjob.ReasonPreRestore, backupjob.ReasonScheduled)
|
||||
return 2
|
||||
}
|
||||
|
||||
@@ -62,16 +60,14 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
|
||||
return 1
|
||||
}
|
||||
// The [archive] parse the reaper uses; an on-demand backup takes its
|
||||
// manual_retention and manual_keep.
|
||||
// The [archive] parse the reaper uses: it holds each reason's keep and
|
||||
// retention.
|
||||
rcfg, err := reaperConfig(cfg)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if keep < 0 {
|
||||
keep = rcfg.ManualKeep
|
||||
}
|
||||
keep, retention, _ := backupPolicy(*reason, rcfg)
|
||||
|
||||
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
|
||||
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
|
||||
@@ -103,7 +99,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
len(a.Skipped), strings.Join(a.Skipped[:min(len(a.Skipped), 10)], ", "))
|
||||
}
|
||||
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
drv, err := openPodStore(ctx, cfg.Database.URL, "backup", stderr)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup: open database: %v\n", err)
|
||||
return 1
|
||||
@@ -117,7 +113,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
BackupRef: string(ref),
|
||||
SizeBytes: size,
|
||||
Reason: *reason,
|
||||
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
|
||||
ExpiresAt: time.Now().Add(retention),
|
||||
|
||||
SHA256: a.SHA256,
|
||||
SkippedEntries: len(a.Skipped),
|
||||
@@ -136,7 +132,7 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
||||
pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
|
||||
pruneBackups(ctx, st, archiver, *server, *formerOwner, *reason, keep, *protect, stdout, stderr)
|
||||
return 0
|
||||
}
|
||||
|
||||
@@ -148,13 +144,32 @@ const (
|
||||
preRestoreKeep = 3
|
||||
)
|
||||
|
||||
// pruneBackups keeps server's newest keep backups of this reason and removes the
|
||||
// rest, oldest first, so repeated backups of one world cannot fill the shared
|
||||
// archive store. protect is never removed: it is the backup a chained restore is
|
||||
// about to extract. The new backup is already recorded; a removal that fails is
|
||||
// reported and retried after the next backup.
|
||||
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
|
||||
excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
|
||||
// backupPolicy is how many backups of one reason a server keeps and how long
|
||||
// each lives: an owner's own backups and the safety snapshots in front of a
|
||||
// restore by [archive] manual_keep / manual_retention (the snapshots capped at
|
||||
// preRestoreKeep), felis-api's daily restore points by scheduled_keep /
|
||||
// scheduled_retention, so neither kind crowds out the other. ok is false for a
|
||||
// reason this command does not record.
|
||||
func backupPolicy(reason string, rcfg reaper.Config) (keep int, retention time.Duration, ok bool) {
|
||||
switch reason {
|
||||
case reasonManual:
|
||||
return rcfg.ManualKeep, rcfg.ManualRetention, true
|
||||
case backupjob.ReasonPreRestore:
|
||||
return preRestoreKeep, rcfg.ManualRetention, true
|
||||
case backupjob.ReasonScheduled:
|
||||
return rcfg.ScheduledKeep, rcfg.ScheduledRetention, true
|
||||
}
|
||||
return 0, 0, false
|
||||
}
|
||||
|
||||
// pruneBackups keeps the newest keep backups of this reason that owner holds of
|
||||
// server and removes the rest, oldest first, so repeated backups of one world
|
||||
// cannot fill the shared archive store and a new owner's backups never remove a
|
||||
// previous owner's. protect is never removed: it is the backup a chained restore
|
||||
// is about to extract. The new backup is already recorded; a removal that fails
|
||||
// is reported and retried after the next backup.
|
||||
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, owner, reason string, keep int, protect string, stdout, stderr io.Writer) {
|
||||
excess, err := st.ExcessBackups(ctx, server, owner, reason, keep, protect)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
|
||||
return
|
||||
|
||||
@@ -4,6 +4,9 @@ import (
|
||||
"bytes"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
)
|
||||
|
||||
// The reason decides which backups the new one's prune may remove, so an
|
||||
@@ -17,3 +20,29 @@ func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
|
||||
t.Fatalf("stderr = %q", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
// Each reason is pruned and expired by its own [archive] keys: a daily
|
||||
// restore point must never count against, or take the lifetime of, the
|
||||
// backups an owner asked for.
|
||||
func TestBackupPolicyPerReason(t *testing.T) {
|
||||
rcfg := reaper.DefaultConfig()
|
||||
rcfg.ManualKeep, rcfg.ManualRetention = 5, 30*reaper.Day
|
||||
rcfg.ScheduledKeep, rcfg.ScheduledRetention = 7, 90*reaper.Day
|
||||
for _, tc := range []struct {
|
||||
reason string
|
||||
keep int
|
||||
retention time.Duration
|
||||
}{
|
||||
{"manual", 5, 30 * reaper.Day},
|
||||
{"pre_restore", preRestoreKeep, 30 * reaper.Day},
|
||||
{"scheduled", 7, 90 * reaper.Day},
|
||||
} {
|
||||
keep, retention, ok := backupPolicy(tc.reason, rcfg)
|
||||
if !ok || keep != tc.keep || retention != tc.retention {
|
||||
t.Errorf("backupPolicy(%q) = (%d, %v, %v); want (%d, %v, true)", tc.reason, keep, retention, ok, tc.keep, tc.retention)
|
||||
}
|
||||
}
|
||||
if _, _, ok := backupPolicy("inactive_15d", rcfg); ok {
|
||||
t.Error("backupPolicy accepted inactive_15d; the reaper records those itself")
|
||||
}
|
||||
}
|
||||
+453
-2
@@ -4,15 +4,28 @@ import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/signal"
|
||||
"sort"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
@@ -22,7 +35,9 @@ import (
|
||||
// (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console
|
||||
// cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth)
|
||||
// while the API is alive, and the API renders the Job and audits the action. This file
|
||||
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue.
|
||||
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue. The
|
||||
// `felis backup-now` command (cmdBackupNow, below) takes the same route for every
|
||||
// user server in turn.
|
||||
|
||||
// backupNowOutcome is the durable result of a backup request, re-printed after the TUI
|
||||
// alt-screen tears down.
|
||||
@@ -75,7 +90,7 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o
|
||||
|
||||
resp, err := hc.Do(req)
|
||||
if err != nil {
|
||||
return backupNowOutcome{}, fmt.Errorf("felis-api unreachable (a backup needs it alive): %w", err)
|
||||
return backupNowOutcome{}, fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
@@ -145,3 +160,439 @@ func backupPickable(servers []haltableServer) []haltableServer {
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// cmdBackupNow is `felis backup-now`: the world of every user server (or of the
|
||||
// named ones) archived now, one at a time, through the internal backup route the
|
||||
// console's Sync uses. A world lives only in its volume and the off-site copy holds
|
||||
// only its archives, so this is the lever in front of a planned move to another
|
||||
// host, a disk swap or anything else that could lose a volume.
|
||||
func cmdBackupNow(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("backup-now", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
|
||||
yes := fs.Bool("yes", false, "back up; without it the plan is printed and nothing changes")
|
||||
stop := fs.Bool("stop", false, "stop the running servers first: their players are disconnected and the servers stay stopped")
|
||||
fs.Usage = func() {
|
||||
fmt.Fprintln(stderr, "Usage: felis backup-now [-yes] [-stop] [server ...]")
|
||||
fmt.Fprintln(stderr)
|
||||
fmt.Fprintln(stderr, "Archives the world of every user server, or of the named ones, one at a time, and waits for each archive.")
|
||||
fmt.Fprintln(stderr, "A running server is skipped unless -stop is given. Without -yes it prints what it would do.")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis backup-now: refused — it reads the cluster's ops token, so it must run as root (try: sudo felis backup-now)")
|
||||
return 1
|
||||
}
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup-now: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer cancel()
|
||||
|
||||
ns := cfg.K8s.Namespace
|
||||
osUser := accountableOSUser()
|
||||
jobs := api.NewK8sJobStatus(cl, ns)
|
||||
var baseURL, token string
|
||||
var repo ownerStore
|
||||
var drv *store.PostgresDriver
|
||||
defer func() {
|
||||
if drv != nil {
|
||||
_ = drv.Close()
|
||||
}
|
||||
}()
|
||||
hc := &http.Client{Timeout: 10 * time.Second}
|
||||
r := backupNowRun{
|
||||
out: stdout,
|
||||
errw: stderr,
|
||||
ns: ns,
|
||||
list: func(ctx context.Context) ([]backupNowWorld, error) { return listBackupNowWorlds(ctx, cl, ns) },
|
||||
stopped: func(ctx context.Context, name string) (bool, error) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &ms); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return backupNowStopped(ctx, cl, &ms)
|
||||
},
|
||||
halt: func(ctx context.Context, name string) error {
|
||||
// The database is opened only once a server is to be stopped: the halt
|
||||
// is audited like the console's, and a run with nothing running needs
|
||||
// no more than the API.
|
||||
if repo == nil {
|
||||
d, err := store.Open(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("open the database for the audit log: %w", err)
|
||||
}
|
||||
drv, repo = d, api.NewPGRepo(d.DB())
|
||||
}
|
||||
out, err := performHalt(ctx, cl, repo, ns, name, osUser)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if out.auditErr != nil {
|
||||
fmt.Fprintf(stderr, "felis backup-now: the audit row for stopping %s was not written: %v\n", name, out.auditErr)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
request: func(ctx context.Context, name string) error {
|
||||
if baseURL == "" {
|
||||
var err error
|
||||
if baseURL, token, err = resolveInternalAPI(ctx, cl, platform.DefaultControlNamespace); err != nil {
|
||||
return fmt.Errorf("%w: %w", errBackupAPIUnreachable, err)
|
||||
}
|
||||
}
|
||||
_, err := requestBackup(ctx, hc, baseURL, token, name, osUser)
|
||||
return err
|
||||
},
|
||||
jobs: jobs.LatestJobs,
|
||||
now: time.Now,
|
||||
sleep: func(ctx context.Context, d time.Duration) { sleepCtx(ctx, d) },
|
||||
}
|
||||
return r.run(ctx, fs.Args(), *yes, *stop)
|
||||
}
|
||||
|
||||
// sleepCtx waits d or until ctx ends.
|
||||
func sleepCtx(ctx context.Context, d time.Duration) {
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
case <-t.C:
|
||||
}
|
||||
}
|
||||
|
||||
// backupNowWorld is one user server as backup-now sees it.
|
||||
type backupNowWorld struct {
|
||||
name string
|
||||
phase string // the observed phase, or the desired state before the operator reconciled it
|
||||
stopped bool // the backup route's stopped gate admits it
|
||||
hasWorld bool // its world volume exists
|
||||
}
|
||||
|
||||
// listBackupNowWorlds lists the user servers of namespace in the API server's
|
||||
// order (by name), each with what the backup route checks. System servers are
|
||||
// left out: they have no row in the servers table, so the route refuses them
|
||||
// (backupPickable).
|
||||
func listBackupNowWorlds(ctx context.Context, cl client.Client, namespace string) ([]backupNowWorld, error) {
|
||||
var list v1alpha1.MinecraftServerList
|
||||
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var out []backupNowWorld
|
||||
for i := range list.Items {
|
||||
ms := &list.Items[i]
|
||||
if isSystemServer(ms.Name) {
|
||||
continue
|
||||
}
|
||||
stopped, err := backupNowStopped(ctx, cl, ms)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var pvc corev1.PersistentVolumeClaim
|
||||
err = cl.Get(ctx, types.NamespacedName{Namespace: namespace, Name: naming.WorldPVCName(ms.Name)}, &pvc)
|
||||
if err != nil && !apierrors.IsNotFound(err) {
|
||||
return nil, fmt.Errorf("look up the world volume of %s: %w", ms.Name, err)
|
||||
}
|
||||
phase := string(ms.Status.Phase)
|
||||
if phase == "" {
|
||||
phase = string(ms.Spec.DesiredState)
|
||||
}
|
||||
out = append(out, backupNowWorld{name: ms.Name, phase: phase, stopped: stopped, hasWorld: err == nil})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// backupNowStopped is the backup route's stopped gate (api.enqueueBackup and
|
||||
// K8sCluster.AcquireMaintenance together): desired Stopped, not ready, phase
|
||||
// Stopped and no game pod left.
|
||||
func backupNowStopped(ctx context.Context, cl client.Client, ms *v1alpha1.MinecraftServer) (bool, error) {
|
||||
if ms.Spec.DesiredState != v1alpha1.DesiredStopped || ms.Status.Ready || ms.Status.Phase != v1alpha1.PhaseStopped {
|
||||
return false, nil
|
||||
}
|
||||
var pods corev1.PodList
|
||||
if err := cl.List(ctx, &pods, client.InNamespace(ms.Namespace), client.MatchingLabels{
|
||||
v1alpha1.LabelServer: ms.Name, v1alpha1.LabelComponent: operator.ComponentValue,
|
||||
}); err != nil {
|
||||
return false, fmt.Errorf("look up the pod of %s: %w", ms.Name, err)
|
||||
}
|
||||
return len(pods.Items) == 0, nil
|
||||
}
|
||||
|
||||
// errBackupAPIUnreachable is a backup request that never reached felis-api. It
|
||||
// ends a backup-now run: every later world would fail the same way, and stopping
|
||||
// servers for backups that cannot be taken only takes them away from players.
|
||||
var errBackupAPIUnreachable = errors.New("felis-api unreachable (a backup needs it alive)")
|
||||
|
||||
// Polling of backup-now. A graceful stop saves the world first; the backup Job's
|
||||
// own deadline (backupjob, 30 minutes) ends a Job that hangs, so its wait needs no
|
||||
// cap of its own.
|
||||
const (
|
||||
backupNowPoll = 2 * time.Second
|
||||
backupNowStopWait = 10 * time.Minute
|
||||
backupNowJobAppear = time.Minute
|
||||
)
|
||||
|
||||
// backupNowRun is backup-now over seams, so the plan and the run are tested
|
||||
// without a cluster or felis-api.
|
||||
type backupNowRun struct {
|
||||
out, errw io.Writer
|
||||
ns string // where the servers and their Jobs live, for the kubectl hints
|
||||
list func(ctx context.Context) ([]backupNowWorld, error)
|
||||
stopped func(ctx context.Context, name string) (bool, error)
|
||||
halt func(ctx context.Context, name string) error
|
||||
request func(ctx context.Context, name string) error
|
||||
jobs func(ctx context.Context, name string) ([]api.AsyncJob, error)
|
||||
now func() time.Time
|
||||
sleep func(ctx context.Context, d time.Duration)
|
||||
}
|
||||
|
||||
// pickBackupNowWorlds narrows worlds to names, in the order given, or keeps them
|
||||
// all when names is empty.
|
||||
func pickBackupNowWorlds(worlds []backupNowWorld, names []string) ([]backupNowWorld, error) {
|
||||
if len(names) == 0 {
|
||||
return worlds, nil
|
||||
}
|
||||
byName := make(map[string]backupNowWorld, len(worlds))
|
||||
for _, w := range worlds {
|
||||
byName[w.name] = w
|
||||
}
|
||||
var out []backupNowWorld
|
||||
seen := map[string]bool{}
|
||||
for _, n := range names {
|
||||
if seen[n] {
|
||||
continue
|
||||
}
|
||||
seen[n] = true
|
||||
w, ok := byName[n]
|
||||
switch {
|
||||
case ok:
|
||||
out = append(out, w)
|
||||
case isSystemServer(n):
|
||||
return nil, fmt.Errorf("%s is a system server: its world is rebuilt by felis setup and has no backups", n)
|
||||
default:
|
||||
return nil, fmt.Errorf("no server named %q", n)
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// backupNowAction is what the plan does with a world.
|
||||
func backupNowAction(w backupNowWorld, stop bool) string {
|
||||
switch {
|
||||
case w.stopped && !w.hasWorld:
|
||||
return "skip: no world volume (never started, nothing to save)"
|
||||
case w.stopped:
|
||||
return "back up"
|
||||
case stop:
|
||||
return "stop, then back up"
|
||||
default:
|
||||
return "skip: running (stop it first, or pass -stop)"
|
||||
}
|
||||
}
|
||||
|
||||
func (r *backupNowRun) run(ctx context.Context, names []string, yes, stop bool) int {
|
||||
all, err := r.list(ctx)
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.errw, "felis backup-now: list the servers: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
worlds, err := pickBackupNowWorlds(all, names)
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.errw, "felis backup-now: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
if len(worlds) == 0 {
|
||||
fmt.Fprintln(r.out, "felis backup-now: there are no user servers")
|
||||
return 0
|
||||
}
|
||||
|
||||
// The stopped worlds go first, so a felis-api that cannot take a backup is
|
||||
// found before any server is stopped for one.
|
||||
sort.SliceStable(worlds, func(i, j int) bool { return worlds[i].stopped && !worlds[j].stopped })
|
||||
width := 0
|
||||
for _, w := range worlds {
|
||||
width = max(width, len(w.name))
|
||||
}
|
||||
fmt.Fprintf(r.out, "felis backup-now: %d server(s), backed up one at a time:\n", len(worlds))
|
||||
stopping, work := false, 0
|
||||
for _, w := range worlds {
|
||||
fmt.Fprintf(r.out, " %-*s %-8s %s\n", width, w.name, w.phase, backupNowAction(w, stop))
|
||||
stopping = stopping || (!w.stopped && stop)
|
||||
if (w.stopped && w.hasWorld) || (!w.stopped && stop) {
|
||||
work++
|
||||
}
|
||||
}
|
||||
if work > 0 {
|
||||
fmt.Fprintln(r.out, "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.")
|
||||
}
|
||||
if stopping {
|
||||
fmt.Fprintln(r.out, "Stopping disconnects the players on those servers, and they stay stopped afterwards.")
|
||||
}
|
||||
if !yes {
|
||||
if work == 0 {
|
||||
fmt.Fprintln(r.out, "Nothing to back up.")
|
||||
} else {
|
||||
fmt.Fprintln(r.out, "Nothing changed. Run again with -yes to back them up.")
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
var done, failed, running, empty int
|
||||
var leftStopped []string
|
||||
for _, w := range worlds {
|
||||
if err := ctx.Err(); err != nil {
|
||||
break
|
||||
}
|
||||
switch {
|
||||
case w.stopped && !w.hasWorld:
|
||||
empty++
|
||||
continue
|
||||
case !w.stopped && !stop:
|
||||
running++
|
||||
continue
|
||||
}
|
||||
if !w.stopped {
|
||||
halted, err := r.stopWorld(ctx, w.name)
|
||||
if halted {
|
||||
leftStopped = append(leftStopped, w.name)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
|
||||
failed++
|
||||
continue
|
||||
}
|
||||
}
|
||||
err := r.backUp(ctx, w.name)
|
||||
if errors.Is(err, errBackupAPIUnreachable) {
|
||||
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
|
||||
fmt.Fprintln(r.out, "Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).")
|
||||
failed++
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(r.out, " %s: failed: %v\n", w.name, err)
|
||||
failed++
|
||||
continue
|
||||
}
|
||||
done++
|
||||
}
|
||||
|
||||
interrupted := ctx.Err() != nil
|
||||
if interrupted {
|
||||
fmt.Fprintln(r.out, "Interrupted: a backup Job already started runs to its end.")
|
||||
}
|
||||
parts := []string{fmt.Sprintf("%d backed up", done)}
|
||||
if failed > 0 {
|
||||
parts = append(parts, fmt.Sprintf("%d failed", failed))
|
||||
}
|
||||
if running > 0 {
|
||||
parts = append(parts, fmt.Sprintf("%d skipped (running)", running))
|
||||
}
|
||||
if empty > 0 {
|
||||
parts = append(parts, fmt.Sprintf("%d without a world", empty))
|
||||
}
|
||||
fmt.Fprintf(r.out, "%s.\n", strings.Join(parts, ", "))
|
||||
if len(leftStopped) > 0 {
|
||||
fmt.Fprintf(r.out, "Left stopped: %s. Start them from the panel when you are done.\n", strings.Join(leftStopped, ", "))
|
||||
}
|
||||
if done > 0 {
|
||||
fmt.Fprintln(r.out, "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.")
|
||||
}
|
||||
if failed > 0 || running > 0 || interrupted {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// stopWorld stops one server and waits until the backup route would admit it.
|
||||
// halted is whether the stop was asked for: the server then stays stopped, even
|
||||
// when it takes longer than the wait.
|
||||
func (r *backupNowRun) stopWorld(ctx context.Context, name string) (halted bool, err error) {
|
||||
fmt.Fprintf(r.out, " %s: stopping\n", name)
|
||||
if err := r.halt(ctx, name); err != nil {
|
||||
return false, err
|
||||
}
|
||||
start := r.now()
|
||||
for {
|
||||
ok, err := r.stopped(ctx, name)
|
||||
if err != nil {
|
||||
return true, err
|
||||
}
|
||||
if ok {
|
||||
fmt.Fprintf(r.out, " %s: stopped after %s\n", name, r.now().Sub(start).Round(time.Second))
|
||||
return true, nil
|
||||
}
|
||||
if r.now().Sub(start) >= backupNowStopWait {
|
||||
return true, fmt.Errorf("did not stop within %s (kubectl -n %s describe minecraftserver %s)", backupNowStopWait, r.ns, name)
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return true, err
|
||||
}
|
||||
r.sleep(ctx, backupNowPoll)
|
||||
}
|
||||
}
|
||||
|
||||
// backUp requests one world's backup and waits for its Job to finish. The Job is
|
||||
// the one of this server that was not there before the request.
|
||||
func (r *backupNowRun) backUp(ctx context.Context, name string) error {
|
||||
before, err := r.jobs(ctx, name)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list its backup Jobs: %w", err)
|
||||
}
|
||||
known := make(map[string]bool, len(before))
|
||||
for _, j := range before {
|
||||
known[j.Name] = true
|
||||
}
|
||||
if err := r.request(ctx, name); err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(r.out, " %s: backing up\n", name)
|
||||
start := r.now()
|
||||
seen := ""
|
||||
for {
|
||||
jobs, err := r.jobs(ctx, name)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list its backup Jobs: %w", err)
|
||||
}
|
||||
var job *api.AsyncJob
|
||||
for i := range jobs {
|
||||
if jobs[i].Kind == "backup" && (jobs[i].Name == seen || seen == "" && !known[jobs[i].Name]) {
|
||||
job = &jobs[i]
|
||||
break
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case job == nil && seen != "":
|
||||
return fmt.Errorf("its backup Job %s was deleted before it finished", seen)
|
||||
case job == nil && r.now().Sub(start) >= backupNowJobAppear:
|
||||
return fmt.Errorf("felis-api accepted the backup, but no backup Job appeared within %s (kubectl -n %s get jobs)", backupNowJobAppear, r.ns)
|
||||
case job != nil && job.State == "succeeded":
|
||||
fmt.Fprintf(r.out, " %s: archived in %s\n", name, r.now().Sub(start).Round(time.Second))
|
||||
return nil
|
||||
case job != nil && job.State == "failed":
|
||||
msg := job.Message
|
||||
if msg == "" {
|
||||
msg = "the backup Job failed"
|
||||
}
|
||||
return fmt.Errorf("%s (kubectl -n %s logs job/%s)", msg, r.ns, job.Name)
|
||||
case job != nil:
|
||||
seen = job.Name
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
r.sleep(ctx, backupNowPoll)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,563 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// bnRig drives backupNowRun against scripted servers and Jobs on a fake clock that
|
||||
// moves only when the run sleeps.
|
||||
type bnRig struct {
|
||||
out, errw bytes.Buffer
|
||||
clock time.Time
|
||||
events []string
|
||||
worlds []backupNowWorld
|
||||
// stopAfter is how many polls a halted server takes to stop; -1 never does.
|
||||
stopAfter map[string]int
|
||||
polls map[string]int
|
||||
reqErr map[string]error
|
||||
// states is what the server's new backup Job reports on each poll after the
|
||||
// request, the last one repeating; "" is no Job.
|
||||
states map[string][]string
|
||||
messages map[string]string
|
||||
jobPolls map[string]int
|
||||
// stopErr fails a server's stop polls; jobsFail fails the nth (1-based) Job
|
||||
// list of a server.
|
||||
stopErr map[string]error
|
||||
jobsFail map[string]int
|
||||
jobCalls map[string]int
|
||||
// ctx is the run's context, and onAct runs after each halt and request.
|
||||
ctx context.Context
|
||||
onAct func(event string)
|
||||
}
|
||||
|
||||
func newBNRig(worlds ...backupNowWorld) *bnRig {
|
||||
return &bnRig{
|
||||
clock: time.Unix(1_800_000_000, 0),
|
||||
worlds: worlds,
|
||||
stopAfter: map[string]int{},
|
||||
polls: map[string]int{},
|
||||
reqErr: map[string]error{},
|
||||
states: map[string][]string{},
|
||||
messages: map[string]string{},
|
||||
jobPolls: map[string]int{},
|
||||
stopErr: map[string]error{},
|
||||
jobsFail: map[string]int{},
|
||||
jobCalls: map[string]int{},
|
||||
ctx: context.Background(),
|
||||
onAct: func(string) {},
|
||||
}
|
||||
}
|
||||
|
||||
func (g *bnRig) run(names []string, yes, stop bool) int {
|
||||
r := backupNowRun{
|
||||
out: &g.out, errw: &g.errw, ns: "minecraft",
|
||||
list: func(context.Context) ([]backupNowWorld, error) {
|
||||
return append([]backupNowWorld(nil), g.worlds...), nil
|
||||
},
|
||||
stopped: func(_ context.Context, name string) (bool, error) {
|
||||
g.polls[name]++
|
||||
if err := g.stopErr[name]; err != nil {
|
||||
return false, err
|
||||
}
|
||||
n := g.stopAfter[name]
|
||||
return n >= 0 && g.polls[name] > n, nil
|
||||
},
|
||||
halt: func(_ context.Context, name string) error {
|
||||
g.events = append(g.events, "halt "+name)
|
||||
g.onAct("halt " + name)
|
||||
return nil
|
||||
},
|
||||
request: func(_ context.Context, name string) error {
|
||||
g.events = append(g.events, "request "+name)
|
||||
g.onAct("request " + name)
|
||||
if err := g.reqErr[name]; err != nil {
|
||||
return err
|
||||
}
|
||||
g.jobPolls[name] = 0
|
||||
return nil
|
||||
},
|
||||
jobs: func(_ context.Context, name string) ([]api.AsyncJob, error) {
|
||||
// Every server has an older finished backup, and a restore Job that
|
||||
// shows up with the new backup: neither is the Job to wait for.
|
||||
g.jobCalls[name]++
|
||||
if g.jobCalls[name] == g.jobsFail[name] {
|
||||
return nil, errors.New("the apiserver is gone")
|
||||
}
|
||||
out := []api.AsyncJob{{Name: "backup-" + name + "-old", Kind: "backup", State: "succeeded"}}
|
||||
n, requested := g.jobPolls[name]
|
||||
if !requested {
|
||||
return out, nil
|
||||
}
|
||||
g.jobPolls[name] = n + 1
|
||||
states := g.states[name]
|
||||
if len(states) == 0 {
|
||||
states = []string{"succeeded"}
|
||||
}
|
||||
state := states[min(n, len(states)-1)]
|
||||
if state == "" {
|
||||
return out, nil
|
||||
}
|
||||
return append([]api.AsyncJob{
|
||||
{Name: "restore-" + name + "-x", Kind: "restore", State: "succeeded"},
|
||||
{Name: "backup-" + name + "-new", Kind: "backup", State: state, Message: g.messages[name]},
|
||||
}, out...), nil
|
||||
},
|
||||
now: func() time.Time { return g.clock },
|
||||
sleep: func(_ context.Context, d time.Duration) { g.clock = g.clock.Add(d) },
|
||||
}
|
||||
return r.run(g.ctx, names, yes, stop)
|
||||
}
|
||||
|
||||
func stoppedWorld(name string) backupNowWorld {
|
||||
return backupNowWorld{name: name, phase: "Stopped", stopped: true, hasWorld: true}
|
||||
}
|
||||
|
||||
func runningWorld(name string) backupNowWorld {
|
||||
return backupNowWorld{name: name, phase: "Running", hasWorld: true}
|
||||
}
|
||||
|
||||
func emptyWorld(name string) backupNowWorld {
|
||||
return backupNowWorld{name: name, phase: "Stopped", stopped: true}
|
||||
}
|
||||
|
||||
const (
|
||||
bnManualKeep = "Each archive is a manual backup: a server that already holds [archive] manual_keep of them loses its oldest.\n"
|
||||
bnStopping = "Stopping disconnects the players on those servers, and they stay stopped afterwards.\n"
|
||||
bnOffsite = "The archives reach the off-site bucket with the hourly copy; sudo systemctl start felis-offsite.service sends them now.\n"
|
||||
)
|
||||
|
||||
func TestBackupNowPlanChangesNothing(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
stop bool
|
||||
want string
|
||||
}{
|
||||
{false, "felis backup-now: 4 server(s), backed up one at a time:\n" +
|
||||
" alpha Stopped back up\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running skip: running (stop it first, or pass -stop)\n" +
|
||||
" delta Starting skip: running (stop it first, or pass -stop)\n" +
|
||||
bnManualKeep +
|
||||
"Nothing changed. Run again with -yes to back them up.\n"},
|
||||
{true, "felis backup-now: 4 server(s), backed up one at a time:\n" +
|
||||
" alpha Stopped back up\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running stop, then back up\n" +
|
||||
" delta Starting stop, then back up\n" +
|
||||
bnManualKeep + bnStopping +
|
||||
"Nothing changed. Run again with -yes to back them up.\n"},
|
||||
} {
|
||||
t.Run(fmt.Sprintf("stop=%v", tc.stop), func(t *testing.T) {
|
||||
delta := runningWorld("delta")
|
||||
delta.phase = "Starting"
|
||||
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), delta)
|
||||
if code := g.run(nil, false, tc.stop); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0; stderr %q", code, g.errw.String())
|
||||
}
|
||||
if g.out.String() != tc.want {
|
||||
t.Fatalf("plan =\n%s\nwant\n%s", g.out.String(), tc.want)
|
||||
}
|
||||
if len(g.events) != 0 || len(g.polls) != 0 {
|
||||
t.Fatalf("the plan acted: events %v, polls %v", g.events, g.polls)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A plan that saves nothing says so, without the manual_keep warning; a running
|
||||
// server counts as something to save once -stop is given.
|
||||
func TestBackupNowPlanWithNothingToSave(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
stop bool
|
||||
want string
|
||||
}{
|
||||
{false, "felis backup-now: 2 server(s), backed up one at a time:\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running skip: running (stop it first, or pass -stop)\n" +
|
||||
"Nothing to back up.\n"},
|
||||
{true, "felis backup-now: 2 server(s), backed up one at a time:\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
" bravo Running stop, then back up\n" +
|
||||
bnManualKeep + bnStopping +
|
||||
"Nothing changed. Run again with -yes to back them up.\n"},
|
||||
} {
|
||||
g := newBNRig(runningWorld("bravo"), emptyWorld("charlie"))
|
||||
if code := g.run(nil, false, tc.stop); code != 0 {
|
||||
t.Fatalf("stop=%v: exit = %d, want 0", tc.stop, code)
|
||||
}
|
||||
if g.out.String() != tc.want {
|
||||
t.Fatalf("stop=%v: plan =\n%s\nwant\n%s", tc.stop, g.out.String(), tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowBacksUpEachWorldInTurn(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"), emptyWorld("charlie"), stoppedWorld("delta"))
|
||||
g.states["alpha"] = []string{"", "running", "succeeded"}
|
||||
g.states["delta"] = []string{"running", "failed"}
|
||||
g.messages["delta"] = "felis backup: not enough free disk for the archive"
|
||||
g.stopAfter["bravo"] = 2
|
||||
g.states["bravo"] = []string{"running", "running", "running", "succeeded"}
|
||||
|
||||
code := g.run(nil, true, true)
|
||||
if code != 1 {
|
||||
t.Fatalf("exit = %d, want 1 (delta failed)", code)
|
||||
}
|
||||
if want := []string{"request alpha", "request delta", "halt bravo", "request bravo"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping+"")
|
||||
want := " alpha: backing up\n" +
|
||||
" alpha: archived in 4s\n" +
|
||||
" delta: backing up\n" +
|
||||
" delta: failed: felis backup: not enough free disk for the archive (kubectl -n minecraft logs job/backup-delta-new)\n" +
|
||||
" bravo: stopping\n" +
|
||||
" bravo: stopped after 4s\n" +
|
||||
" bravo: backing up\n" +
|
||||
" bravo: archived in 6s\n" +
|
||||
"2 backed up, 1 failed, 1 without a world.\n" +
|
||||
"Left stopped: bravo. Start them from the panel when you are done.\n" +
|
||||
bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowSkipsRunningServersWithoutStop(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), runningWorld("bravo"))
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1 (bravo was not backed up)", code)
|
||||
}
|
||||
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
if len(g.polls) != 0 {
|
||||
t.Fatalf("polled a server it did not stop: %v", g.polls)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), "Nothing changed")
|
||||
if run != "" {
|
||||
t.Fatalf("-yes printed the plan's closing line")
|
||||
}
|
||||
_, run, _ = strings.Cut(g.out.String(), bnManualKeep)
|
||||
want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 skipped (running).\n" + bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowExitsCleanWhenEverythingIsSaved(t *testing.T) {
|
||||
// -stop with nothing running stops nothing and says nothing about stopping.
|
||||
g := newBNRig(stoppedWorld("alpha"), emptyWorld("charlie"))
|
||||
if code := g.run(nil, true, true); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0; output\n%s", code, g.out.String())
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
|
||||
if want := " alpha: backing up\n alpha: archived in 0s\n1 backed up, 1 without a world.\n" + bnOffsite; run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
|
||||
g = newBNRig(emptyWorld("charlie"))
|
||||
if code := g.run(nil, true, false); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0 for a server with nothing to save", code)
|
||||
}
|
||||
// Nothing to archive: no manual_keep warning, and no off-site hint.
|
||||
if want := "felis backup-now: 1 server(s), backed up one at a time:\n" +
|
||||
" charlie Stopped skip: no world volume (never started, nothing to save)\n" +
|
||||
"0 backed up, 1 without a world.\n"; g.out.String() != want {
|
||||
t.Fatalf("output =\n%s\nwant\n%s", g.out.String(), want)
|
||||
}
|
||||
if len(g.events) != 0 {
|
||||
t.Fatalf("events = %v, want none", g.events)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowStopsAtAnUnreachableAPI(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), runningWorld("charlie"))
|
||||
g.reqErr["alpha"] = fmt.Errorf("%w: dial tcp 10.43.0.9:8081: connect: connection refused", errBackupAPIUnreachable)
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v: nothing after the API proved unreachable, and no server stopped", g.events, want)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
want := " alpha: failed: felis-api unreachable (a backup needs it alive): dial tcp 10.43.0.9:8081: connect: connection refused\n" +
|
||||
"Stopped: nothing more can be backed up until felis-api answers (kubectl -n felis get pods).\n" +
|
||||
"0 backed up, 1 failed.\n"
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
|
||||
// Any other refusal is that world's alone: the run goes on.
|
||||
g = newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
|
||||
g.reqErr["alpha"] = errors.New("felis-api: the world is being restored")
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"request alpha", "request bravo"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowGivesUpOnAServerThatDoesNotStop(t *testing.T) {
|
||||
g := newBNRig(runningWorld("bravo"), runningWorld("echo"))
|
||||
g.stopAfter["bravo"] = -1
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"halt bravo", "halt echo", "request echo"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
// One poll at the start and one per 2s sleep up to the 10-minute mark.
|
||||
if g.polls["bravo"] != 301 {
|
||||
t.Fatalf("bravo polled %d times, want 301", g.polls["bravo"])
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
want := " bravo: stopping\n" +
|
||||
" bravo: failed: did not stop within 10m0s (kubectl -n minecraft describe minecraftserver bravo)\n" +
|
||||
" echo: stopping\n" +
|
||||
" echo: stopped after 0s\n" +
|
||||
" echo: backing up\n" +
|
||||
" echo: archived in 0s\n" +
|
||||
"1 backed up, 1 failed.\n" +
|
||||
"Left stopped: bravo, echo. Start them from the panel when you are done.\n" +
|
||||
bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowReportsAJobThatNeverRuns(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
states []string
|
||||
polls int
|
||||
want string
|
||||
}{
|
||||
{"never appears", []string{""}, 31,
|
||||
"felis-api accepted the backup, but no backup Job appeared within 1m0s (kubectl -n minecraft get jobs)"},
|
||||
{"deleted while running", []string{"running", "running", ""}, 3,
|
||||
"its backup Job backup-alpha-new was deleted before it finished"},
|
||||
{"fails without a message", []string{"failed"}, 1,
|
||||
"the backup Job failed (kubectl -n minecraft logs job/backup-alpha-new)"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
g.states["alpha"] = tc.states
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if !strings.Contains(g.out.String(), " alpha: failed: "+tc.want+"\n") {
|
||||
t.Fatalf("output =\n%s\nwant the line %q", g.out.String(), tc.want)
|
||||
}
|
||||
if g.jobPolls["alpha"] != tc.polls {
|
||||
t.Fatalf("polled the Jobs %d times after the request, want %d", g.jobPolls["alpha"], tc.polls)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowNamedServers(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"), stoppedWorld("delta"))
|
||||
if code := g.run([]string{"delta", "alpha", "delta"}, true, false); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0", code)
|
||||
}
|
||||
if want := []string{"request delta", "request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
if !strings.HasPrefix(g.out.String(), "felis backup-now: 2 server(s), backed up one at a time:\n delta Stopped back up\n alpha Stopped back up\n") {
|
||||
t.Fatalf("plan =\n%s", g.out.String())
|
||||
}
|
||||
|
||||
for _, tc := range []struct{ name, want string }{
|
||||
{"login", "felis backup-now: login is a system server: its world is rebuilt by felis setup and has no backups\n"},
|
||||
{"lobby", "felis backup-now: lobby is a system server: its world is rebuilt by felis setup and has no backups\n"},
|
||||
{"nope", "felis backup-now: no server named \"nope\"\n"},
|
||||
} {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
if code := g.run([]string{"alpha", tc.name}, true, false); code != 2 {
|
||||
t.Fatalf("%s: exit = %d, want 2", tc.name, code)
|
||||
}
|
||||
if g.errw.String() != tc.want {
|
||||
t.Fatalf("%s: stderr = %q, want %q", tc.name, g.errw.String(), tc.want)
|
||||
}
|
||||
if len(g.events) != 0 || g.out.Len() != 0 {
|
||||
t.Fatalf("%s: acted on a bad name: events %v, output %q", tc.name, g.events, g.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
g = newBNRig()
|
||||
if code := g.run(nil, true, false); code != 0 || g.out.String() != "felis backup-now: there are no user servers\n" {
|
||||
t.Fatalf("empty fleet: exit %d, output %q", code, g.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The world list mirrors the backup route's own gate, so the plan says exactly what
|
||||
// the route would refuse.
|
||||
func TestListBackupNowWorlds(t *testing.T) {
|
||||
withStatus := func(ms *v1alpha1.MinecraftServer, ready bool) *v1alpha1.MinecraftServer {
|
||||
ms.Status.Ready = ready
|
||||
return ms
|
||||
}
|
||||
pvc := func(server, ns string) *corev1.PersistentVolumeClaim {
|
||||
return &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: naming.WorldPVCName(server), Namespace: ns}}
|
||||
}
|
||||
pod := func(name, ns string, labels map[string]string) *corev1.Pod {
|
||||
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns, Labels: labels}}
|
||||
}
|
||||
gameLabels := func(server string) map[string]string {
|
||||
return map[string]string{v1alpha1.LabelServer: server, v1alpha1.LabelComponent: "server"}
|
||||
}
|
||||
other := mcServer("zulu", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped)
|
||||
other.Namespace = "elsewhere"
|
||||
hotel := mcServer("hotel", v1alpha1.DesiredStopped, "")
|
||||
objs := []client.Object{
|
||||
mcServer("alpha", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("alpha", haltNS),
|
||||
// A backup Job's pod carries the server label with its own component, and
|
||||
// a game pod of the same name in another namespace is somebody else's.
|
||||
pod("backup-alpha-1-x", haltNS, map[string]string{v1alpha1.LabelServer: "alpha", "app.kubernetes.io/component": "world-backup"}),
|
||||
pod("alpha-0", "elsewhere", gameLabels("alpha")),
|
||||
withStatus(mcServer("bravo", v1alpha1.DesiredRunning, v1alpha1.PhaseRunning), true), pvc("bravo", haltNS), pod("bravo-0", haltNS, gameLabels("bravo")),
|
||||
mcServer("charlie", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped),
|
||||
mcServer("delta", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("delta", haltNS), pod("delta-0", haltNS, gameLabels("delta")),
|
||||
mcServer("echo", v1alpha1.DesiredStopped, v1alpha1.PhaseStopping), pvc("echo", haltNS),
|
||||
mcServer("foxtrot", v1alpha1.DesiredRunning, v1alpha1.PhaseStopped), pvc("foxtrot", haltNS),
|
||||
withStatus(mcServer("golf", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), true), pvc("golf", haltNS),
|
||||
hotel, pvc("hotel", haltNS),
|
||||
mcServer("login", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("login", haltNS),
|
||||
mcServer("lobby", v1alpha1.DesiredStopped, v1alpha1.PhaseStopped), pvc("lobby", haltNS),
|
||||
other, pvc("zulu", "elsewhere"),
|
||||
}
|
||||
got, err := listBackupNowWorlds(context.Background(), haltClient(t, objs...), haltNS)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := []backupNowWorld{
|
||||
{name: "alpha", phase: "Stopped", stopped: true, hasWorld: true},
|
||||
{name: "bravo", phase: "Running", hasWorld: true},
|
||||
{name: "charlie", phase: "Stopped", stopped: true},
|
||||
{name: "delta", phase: "Stopped", hasWorld: true}, // its pod is still going
|
||||
{name: "echo", phase: "Stopping", hasWorld: true}, // still stopping
|
||||
{name: "foxtrot", phase: "Stopped", hasWorld: true}, // asked to start
|
||||
{name: "golf", phase: "Stopped", hasWorld: true}, // still reports ready
|
||||
{name: "hotel", phase: "Stopped", hasWorld: true}, // not reconciled yet: the desired state shows
|
||||
}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("worlds =\n%+v\nwant\n%+v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBackupNowStopsWhenInterrupted(t *testing.T) {
|
||||
t.Run("between worlds", func(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"), stoppedWorld("bravo"))
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
g.ctx = ctx
|
||||
g.onAct = func(e string) {
|
||||
if e == "request alpha" {
|
||||
cancel()
|
||||
}
|
||||
}
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if want := []string{"request alpha"}; !reflect.DeepEqual(g.events, want) {
|
||||
t.Fatalf("events = %v, want %v", g.events, want)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnManualKeep)
|
||||
want := " alpha: backing up\n alpha: archived in 0s\n" +
|
||||
"Interrupted: a backup Job already started runs to its end.\n" +
|
||||
"1 backed up.\n" + bnOffsite
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
})
|
||||
t.Run("while a Job runs", func(t *testing.T) {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
g.states["alpha"] = []string{"running"}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
g.ctx = ctx
|
||||
g.onAct = func(string) { cancel() }
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if g.jobPolls["alpha"] != 1 {
|
||||
t.Fatalf("polled the Job %d times after the interrupt, want 1", g.jobPolls["alpha"])
|
||||
}
|
||||
if !strings.Contains(g.out.String(), " alpha: failed: context canceled\nInterrupted: a backup Job already started runs to its end.\n0 backed up, 1 failed.\n") {
|
||||
t.Fatalf("output =\n%s", g.out.String())
|
||||
}
|
||||
})
|
||||
t.Run("while a server stops", func(t *testing.T) {
|
||||
g := newBNRig(runningWorld("bravo"))
|
||||
g.stopAfter["bravo"] = -1
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
g.ctx = ctx
|
||||
g.onAct = func(string) { cancel() }
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
if g.polls["bravo"] != 1 {
|
||||
t.Fatalf("polled bravo %d times after the interrupt, want 1", g.polls["bravo"])
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
want := " bravo: stopping\n bravo: failed: context canceled\n" +
|
||||
"Interrupted: a backup Job already started runs to its end.\n" +
|
||||
"0 backed up, 1 failed.\n" +
|
||||
"Left stopped: bravo. Start them from the panel when you are done.\n"
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestBackupNowReportsClusterErrors(t *testing.T) {
|
||||
g := newBNRig(runningWorld("bravo"))
|
||||
g.stopErr["bravo"] = errors.New("the apiserver is gone")
|
||||
if code := g.run(nil, true, true); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1", code)
|
||||
}
|
||||
_, run, _ := strings.Cut(g.out.String(), bnStopping)
|
||||
// The stop was asked for, so the server stays stopped whatever the poll said.
|
||||
want := " bravo: stopping\n bravo: failed: the apiserver is gone\n0 backed up, 1 failed.\n" +
|
||||
"Left stopped: bravo. Start them from the panel when you are done.\n"
|
||||
if run != want {
|
||||
t.Fatalf("run =\n%s\nwant\n%s", run, want)
|
||||
}
|
||||
|
||||
for _, tc := range []struct {
|
||||
call int
|
||||
events []string
|
||||
}{
|
||||
{1, nil}, // before the request: nothing is asked for
|
||||
{2, []string{"request alpha"}}, // the first poll after it
|
||||
} {
|
||||
g := newBNRig(stoppedWorld("alpha"))
|
||||
g.jobsFail["alpha"] = tc.call
|
||||
if code := g.run(nil, true, false); code != 1 {
|
||||
t.Fatalf("call %d: exit = %d, want 1", tc.call, code)
|
||||
}
|
||||
if !reflect.DeepEqual(g.events, tc.events) {
|
||||
t.Fatalf("call %d: events = %v, want %v", tc.call, g.events, tc.events)
|
||||
}
|
||||
if !strings.Contains(g.out.String(), " alpha: failed: list its backup Jobs: the apiserver is gone\n") {
|
||||
t.Fatalf("call %d: output =\n%s", tc.call, g.out.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package main
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
@@ -152,6 +153,10 @@ func TestRequestBackup(t *testing.T) {
|
||||
if err == nil || !strings.Contains(err.Error(), "unreachable") {
|
||||
t.Fatalf("err = %v, want an 'unreachable' transport error", err)
|
||||
}
|
||||
// backup-now ends its run on this error alone.
|
||||
if !errors.Is(err, errBackupAPIUnreachable) {
|
||||
t.Fatalf("err = %v, want it to wrap errBackupAPIUnreachable", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
+73
-59
@@ -17,6 +17,7 @@ import (
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
@@ -33,13 +34,15 @@ import (
|
||||
//
|
||||
// Root is necessary but NOT sufficient for accountability: root is machine
|
||||
// authority, not a human identity, so the console additionally captures WHO is
|
||||
// breaking the glass. When a staff account already exists it asks the operator to
|
||||
// authenticate as an existing admin (the verified identity is the accountable
|
||||
// actor); when none exists yet it bootstraps the first Owner from the typed
|
||||
// credential and attributes the act to the OS user. The audit row records the
|
||||
// difference. This attribution is best-effort, not tamper-proof — whoever runs
|
||||
// this is root and can edit Postgres directly — but it produces an honest trail
|
||||
// for an honest operator, which is the point.
|
||||
// breaking the glass. When a staff account already exists the operator names one
|
||||
// and types the one-time code the console mails to its verified address
|
||||
// (breakglass_otp.go); that account is then the accountable actor. When no code can
|
||||
// be sent or proven, the typed OVERRIDE proceeds as the OS user and the audit row
|
||||
// says why. When no staff account exists yet it bootstraps the first Owner and
|
||||
// attributes the act to the OS user. The audit row records which of these
|
||||
// happened. This attribution is best-effort, not tamper-proof — whoever runs this
|
||||
// is root and can edit Postgres directly — but it produces an honest trail for an
|
||||
// honest operator, which is the point.
|
||||
//
|
||||
// When a staff account already exists the console opens on a thin top-level menu
|
||||
// (menuModel) so that operations are peers, not tails of one wizard. Two account
|
||||
@@ -62,7 +65,7 @@ import (
|
||||
// suspension for the interactive `cloudflared tunnel login` browser consent.
|
||||
|
||||
// breakGlassOverrideToken is the literal an operator must type to proceed when no
|
||||
// admin credential could be verified. Requiring an explicit, deliberate word (not a
|
||||
// admin could be verified by a mailed code. Requiring an explicit, deliberate word (not a
|
||||
// bare Enter) keeps the unverified root override from happening by reflex.
|
||||
const breakGlassOverrideToken = "OVERRIDE"
|
||||
|
||||
@@ -142,7 +145,7 @@ func cmdBreakGlass(args []string, stdout, stderr io.Writer) int {
|
||||
repo := api.NewPGRepo(drv.DB())
|
||||
|
||||
// Decide bootstrap (no admin yet → typed credential mints the first Owner) vs
|
||||
// recovery (an admin exists → the operator must authenticate as one) BEFORE the
|
||||
// recovery (an admin exists → the operator proves one with a mailed code) BEFORE the
|
||||
// alt-screen TUI takes over, so a database fault surfaces as a plain error.
|
||||
adminExists, err := repo.AdminExists(ctx)
|
||||
if err != nil {
|
||||
@@ -150,7 +153,12 @@ func cmdBreakGlass(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
res, err := runBreakGlassTUI(ctx, repo, cfg.Database.URL, cfg.Server.RootDomain, cfg.Auth.AdminHostname, cfg.Auth.PanelHostname, cfg.Auth.AccessJWTAud, cfg.K8s.Namespace, accountableOSUser(), adminExists)
|
||||
// Recovery mails its code through [smtp]; the relay is opened only if a code is
|
||||
// asked for.
|
||||
host, _ := os.Hostname()
|
||||
recovery := recoveryConfig{open: hostRecoveryMailer(cfg.SMTP, hostSMTPPasswordPath, platform.DefaultControlNamespace), host: host}
|
||||
|
||||
res, err := runBreakGlassTUI(ctx, repo, cfg.Database, cfg.Server.RootDomain, cfg.Auth.AdminHostname, cfg.Auth.PanelHostname, cfg.Auth.AccessJWTAud, cfg.K8s.Namespace, accountableOSUser(), adminExists, recovery)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis breakGlass: %v\n", err)
|
||||
return 1
|
||||
@@ -174,6 +182,9 @@ func cmdBreakGlass(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stdout, "\nfelis breakGlass: Owner account %q provisioned; local session sign-in is ENABLED.\n", res.username)
|
||||
}
|
||||
fmt.Fprintf(stdout, "Recorded as %q (mode: %s, os user: %s).\n", res.accountable, res.mode, res.osUser)
|
||||
if res.mode == "root_override" {
|
||||
fmt.Fprintln(stdout, "No admin was proven by an email code; the audit row records this run as an unverified root override and why.")
|
||||
}
|
||||
if res.setupTokenURL != "" {
|
||||
fmt.Fprintf(stdout, "One-time setup URL (opens a lockdown session to verify email / enroll passkey):\n\n %s\n\n", res.setupTokenURL)
|
||||
}
|
||||
@@ -258,29 +269,6 @@ func newOwnerID() string {
|
||||
return "usr-" + hex.EncodeToString(b[:])
|
||||
}
|
||||
|
||||
// authenticateAdmin resolves a typed admin username for recovery-mode attribution.
|
||||
// Password verification is gone (passwordless design); Phase 3 replaces this with
|
||||
// email-OTP recovery. For now it confirms the named admin exists.
|
||||
func authenticateAdmin(ctx context.Context, s ownerStore, username string) (matched string, ok bool, err error) {
|
||||
username = strings.TrimSpace(username)
|
||||
if username == "" {
|
||||
return "", false, nil
|
||||
}
|
||||
u, err := s.UserByUsername(ctx, username)
|
||||
if errors.Is(err, api.ErrNotFound) {
|
||||
return "", false, nil
|
||||
}
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
// Staff means admin OR owner: recovery attribution must accept the Owner (the
|
||||
// primary break-glass identity), not just plain admins.
|
||||
if u.Role != "admin" && u.Role != "owner" {
|
||||
return "", false, nil
|
||||
}
|
||||
return u.Username, true, nil
|
||||
}
|
||||
|
||||
// provisionOwner mints or resets the single Owner account direct-to-Postgres,
|
||||
// passwordless. The account is role=owner with no password — the Owner completes
|
||||
// passwordless login setup via the web setup-token flow after `felis setup`.
|
||||
@@ -371,6 +359,10 @@ type breakGlassOp struct {
|
||||
ownerUsername string
|
||||
ownerEmail string
|
||||
attemptedAdmin string // recovery / override: the admin username the operator typed
|
||||
verifiedBy string // recovery: how the admin was proven (verifiedByEmailOTP)
|
||||
codeSentTo string // recovery: the address the proving code went to
|
||||
otpSkipped string // root_override: why no code proved an admin (otpSkip*)
|
||||
otpSkipDetail string // root_override: what failed, when something did
|
||||
}
|
||||
|
||||
// breakGlassOutcome is what performBreakGlass reports back to the TUI.
|
||||
@@ -494,16 +486,7 @@ func auditSetupMCBind(ctx context.Context, s ownerStore, osUser, mcUUID, authSou
|
||||
// does not fail the recovery if this write fails — and intentionally honest: it
|
||||
// records attribution, it does not prove it (a malicious root can edit the row).
|
||||
func auditBreakGlass(ctx context.Context, s ownerStore, op breakGlassOp) error {
|
||||
payload := map[string]any{
|
||||
"mode": op.mode,
|
||||
"owner": op.ownerUsername,
|
||||
"os_user": op.osUser,
|
||||
"verified": op.mode == "recovery",
|
||||
}
|
||||
if op.attemptedAdmin != "" {
|
||||
payload["admin_account"] = op.attemptedAdmin
|
||||
}
|
||||
blob, err := json.Marshal(payload)
|
||||
blob, err := json.Marshal(breakGlassPayload(op, "owner"))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -515,6 +498,33 @@ func auditBreakGlass(ctx context.Context, s ownerStore, op breakGlassOp) error {
|
||||
})
|
||||
}
|
||||
|
||||
// breakGlassPayload is the who/how both account audits carry, with the account the
|
||||
// run wrote under subjectKey. verified is true only for a run a mailed code proved;
|
||||
// such a run names the address the code went to, and an override names why no code
|
||||
// proved anyone.
|
||||
func breakGlassPayload(op breakGlassOp, subjectKey string) map[string]any {
|
||||
payload := map[string]any{
|
||||
"mode": op.mode,
|
||||
subjectKey: op.ownerUsername,
|
||||
"os_user": op.osUser,
|
||||
"verified": op.verifiedBy != "",
|
||||
}
|
||||
if op.attemptedAdmin != "" {
|
||||
payload["admin_account"] = op.attemptedAdmin
|
||||
}
|
||||
if op.verifiedBy != "" {
|
||||
payload["verified_by"] = op.verifiedBy
|
||||
payload["code_sent_to"] = op.codeSentTo
|
||||
}
|
||||
if op.otpSkipped != "" {
|
||||
payload["otp_skipped"] = op.otpSkipped
|
||||
if op.otpSkipDetail != "" {
|
||||
payload["otp_skip_detail"] = op.otpSkipDetail
|
||||
}
|
||||
}
|
||||
return payload
|
||||
}
|
||||
|
||||
// performAddOperator mints a NEW Operator account and records a best-effort
|
||||
// accountability row. It mirrors performBreakGlass — passwordless — with two
|
||||
// deliberate differences. (1) It provisions insert-only (provisionOperator), so it
|
||||
@@ -538,16 +548,7 @@ func performAddOperator(ctx context.Context, s ownerStore, op breakGlassOp) (bre
|
||||
// break_glass.operator_create action, naming the new account under an "operator" key
|
||||
// rather than "owner".
|
||||
func auditAddOperator(ctx context.Context, s ownerStore, op breakGlassOp) error {
|
||||
payload := map[string]any{
|
||||
"mode": op.mode,
|
||||
"operator": op.ownerUsername,
|
||||
"os_user": op.osUser,
|
||||
"verified": op.mode == "recovery",
|
||||
}
|
||||
if op.attemptedAdmin != "" {
|
||||
payload["admin_account"] = op.attemptedAdmin
|
||||
}
|
||||
blob, err := json.Marshal(payload)
|
||||
blob, err := json.Marshal(breakGlassPayload(op, "operator"))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -626,16 +627,29 @@ const (
|
||||
cloudflareAPITokenDocsURL = "https://developers.cloudflare.com/fundamentals/api/how-to/account-owned-token-template/"
|
||||
)
|
||||
|
||||
func runBreakGlassTUI(ctx context.Context, s ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool) (breakGlassResult, error) {
|
||||
return runConsoleTUI(ctx, s, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeBreakGlass)
|
||||
func runBreakGlassTUI(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, recovery recoveryConfig) (breakGlassResult, error) {
|
||||
return runConsoleTUI(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeBreakGlass, recovery)
|
||||
}
|
||||
|
||||
func runSetupTUI(ctx context.Context, s ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool) (breakGlassResult, error) {
|
||||
return runConsoleTUI(ctx, s, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeSetup)
|
||||
// runSetupTUI never reaches recovery: setup with a staff account present lands on
|
||||
// the status screen, so it has no relay to hand over.
|
||||
func runSetupTUI(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool) (breakGlassResult, error) {
|
||||
return runConsoleTUI(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, consoleModeSetup, recoveryConfig{})
|
||||
}
|
||||
|
||||
func runConsoleTUI(ctx context.Context, s ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode) (breakGlassResult, error) {
|
||||
rm := newRootModel(ctx, s, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, mode)
|
||||
// newConsoleRoot is the console's root model as the host runs it: the summary
|
||||
// reads this host's alert route.
|
||||
func newConsoleRoot(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode, recovery recoveryConfig) *rootModel {
|
||||
rm := newRootModel(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, mode)
|
||||
rm.recovery = recovery
|
||||
rm.alertRoute = func(ctx context.Context) alertRoute {
|
||||
return hostAlertRoute(ctx, hostSetupConfigPath, db.URL, defaultHeartbeatFile)
|
||||
}
|
||||
return rm
|
||||
}
|
||||
|
||||
func runConsoleTUI(ctx context.Context, s ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode, recovery recoveryConfig) (breakGlassResult, error) {
|
||||
rm := newConsoleRoot(ctx, s, db, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser, adminExists, mode, recovery)
|
||||
final, err := tea.NewProgram(rm, tea.WithAltScreen()).Run()
|
||||
if err != nil {
|
||||
return breakGlassResult{}, err
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"crypto/subtle"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math/big"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
)
|
||||
|
||||
// Recovery mode proves who is breaking the glass (#13). Naming a staff account is
|
||||
// where it starts: the console then mails a one-time code to that account's verified
|
||||
// address, and only that code, typed within recoveryCodeTTL, makes the run a
|
||||
// recovery attributed to the account. Every other ending — no such account, no
|
||||
// verified address, no relay, a send that fails, a wrong or late code, or the
|
||||
// operator giving up on the mail — leads to the typed OVERRIDE, which the audit row
|
||||
// records as an unverified root_override together with the reason (otp_skipped).
|
||||
// The code goes through the same [smtp] relay as the panel's login codes, so with
|
||||
// that relay down recovery still works, as an override that says why.
|
||||
|
||||
const (
|
||||
recoveryCodeTTL = 10 * time.Minute
|
||||
recoveryCodeAttempts = 5
|
||||
)
|
||||
|
||||
// The reasons a run fell back to the override, recorded as otp_skipped.
|
||||
const (
|
||||
otpSkipUnknownAdmin = "unknown_admin"
|
||||
otpSkipNoVerifiedEmail = "no_verified_email"
|
||||
otpSkipNoRelay = "no_relay"
|
||||
otpSkipSendFailed = "send_failed"
|
||||
otpSkipCodeExpired = "code_expired"
|
||||
otpSkipCodeRejected = "code_rejected"
|
||||
otpSkipByOperator = "operator_skipped"
|
||||
)
|
||||
|
||||
// verifiedByEmailOTP is the audit's verified_by for a recovery the mailed code proved.
|
||||
const verifiedByEmailOTP = "email_otp"
|
||||
|
||||
// recoveryMailer is the one relay call a recovery code needs; *mail.SMTP has it.
|
||||
type recoveryMailer interface {
|
||||
SendNotice(ctx context.Context, email, subject, body string) error
|
||||
}
|
||||
|
||||
// recoveryConfig is what the console needs to mail a recovery code. open resolves
|
||||
// the relay only when a code is about to go out, so a console used to halt a server
|
||||
// never touches [smtp] or the cluster; its error says why no relay is available.
|
||||
// host names this machine in the mail. The zero value has no relay.
|
||||
type recoveryConfig struct {
|
||||
open func(ctx context.Context) (recoveryMailer, error)
|
||||
host string
|
||||
now func() time.Time
|
||||
}
|
||||
|
||||
func (r recoveryConfig) clock() time.Time {
|
||||
if r.now != nil {
|
||||
return r.now()
|
||||
}
|
||||
return time.Now()
|
||||
}
|
||||
|
||||
// recoveryCode is one mailed code: its value, when it stops working, and how many
|
||||
// wrong codes were typed against it.
|
||||
type recoveryCode struct {
|
||||
value string
|
||||
expires time.Time
|
||||
failures int
|
||||
}
|
||||
|
||||
func newRecoveryCode(now time.Time) (*recoveryCode, error) {
|
||||
n, err := rand.Int(rand.Reader, big.NewInt(1_000_000))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("generate recovery code: %w", err)
|
||||
}
|
||||
return &recoveryCode{value: fmt.Sprintf("%06d", n.Int64()), expires: now.Add(recoveryCodeTTL)}, nil
|
||||
}
|
||||
|
||||
type codeVerdict int
|
||||
|
||||
const (
|
||||
codeAccepted codeVerdict = iota
|
||||
codeWrong
|
||||
codeExpired
|
||||
codeExhausted
|
||||
)
|
||||
|
||||
// check compares a typed code in constant time. Each wrong code counts; the one
|
||||
// that reaches recoveryCodeAttempts exhausts the code, which then accepts nothing,
|
||||
// and neither does an expired one.
|
||||
func (c *recoveryCode) check(typed string, now time.Time) codeVerdict {
|
||||
if c.failures >= recoveryCodeAttempts {
|
||||
return codeExhausted
|
||||
}
|
||||
if !now.Before(c.expires) {
|
||||
return codeExpired
|
||||
}
|
||||
if subtle.ConstantTimeCompare([]byte(strings.TrimSpace(typed)), []byte(c.value)) == 1 {
|
||||
return codeAccepted
|
||||
}
|
||||
c.failures++
|
||||
if c.failures >= recoveryCodeAttempts {
|
||||
return codeExhausted
|
||||
}
|
||||
return codeWrong
|
||||
}
|
||||
|
||||
func (c *recoveryCode) attemptsLeft() int { return recoveryCodeAttempts - c.failures }
|
||||
|
||||
// recoveryStart is where naming an admin led: a code on its way to that admin, or
|
||||
// the reason the run has to fall back to the override.
|
||||
type recoveryStart struct {
|
||||
admin *api.StaffUser // the named staff account; nil when none matched
|
||||
code *recoveryCode // set when the code went out
|
||||
skip string // otpSkip* when it did not
|
||||
detail string // what failed, for the override screen and the audit row
|
||||
}
|
||||
|
||||
// resolveAdmin loads the staff account (admin or owner) a typed username names, or
|
||||
// nil when there is none.
|
||||
func resolveAdmin(ctx context.Context, s ownerStore, username string) (*api.StaffUser, error) {
|
||||
username = strings.TrimSpace(username)
|
||||
if username == "" {
|
||||
return nil, nil
|
||||
}
|
||||
u, err := s.UserByUsername(ctx, username)
|
||||
if errors.Is(err, api.ErrNotFound) {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Staff means admin or owner: the Owner is the primary break-glass identity.
|
||||
if u.Role != "admin" && u.Role != "owner" {
|
||||
return nil, nil
|
||||
}
|
||||
return u, nil
|
||||
}
|
||||
|
||||
// beginRecovery resolves the named admin and mails it a recovery code. Only a
|
||||
// datastore or entropy fault is an error; every other way the code cannot go out is
|
||||
// a recoveryStart with skip set.
|
||||
func beginRecovery(ctx context.Context, s ownerStore, rc recoveryConfig, username, osUser string, op bgOperation) (recoveryStart, error) {
|
||||
admin, err := resolveAdmin(ctx, s, username)
|
||||
if err != nil {
|
||||
return recoveryStart{}, err
|
||||
}
|
||||
if admin == nil {
|
||||
return recoveryStart{skip: otpSkipUnknownAdmin}, nil
|
||||
}
|
||||
st := recoveryStart{admin: admin}
|
||||
// An address nobody ever proved vouches for nobody.
|
||||
email := strings.TrimSpace(admin.Email)
|
||||
if email == "" || !admin.EmailVerified {
|
||||
st.skip = otpSkipNoVerifiedEmail
|
||||
return st, nil
|
||||
}
|
||||
if rc.open == nil {
|
||||
st.skip, st.detail = otpSkipNoRelay, "this console has no mail relay"
|
||||
return st, nil
|
||||
}
|
||||
relay, err := rc.open(ctx)
|
||||
if err != nil {
|
||||
st.skip, st.detail = otpSkipNoRelay, err.Error()
|
||||
return st, nil
|
||||
}
|
||||
code, err := newRecoveryCode(rc.clock())
|
||||
if err != nil {
|
||||
return recoveryStart{}, err
|
||||
}
|
||||
sendCtx, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
defer cancel()
|
||||
subject, body := recoveryMail(code.value, rc.host, osUser, admin.Username, op)
|
||||
if err := relay.SendNotice(sendCtx, email, subject, body); err != nil {
|
||||
st.skip, st.detail = otpSkipSendFailed, err.Error()
|
||||
return st, nil
|
||||
}
|
||||
st.code = code
|
||||
return st, nil
|
||||
}
|
||||
|
||||
// recoveryMail words the code mail. It says where, by whom and for what the console
|
||||
// was opened, so an admin who did not ask for it learns that root on that machine is
|
||||
// in someone else's hands.
|
||||
func recoveryMail(code, host, osUser, admin string, op bgOperation) (subject, body string) {
|
||||
what, whatZH := "reset the Owner account", "重置 Owner 账号"
|
||||
if op == bgAddOperator {
|
||||
what, whatZH = "add an Operator account", "添加 Operator 账号"
|
||||
}
|
||||
if host == "" {
|
||||
host = "the Felis host"
|
||||
}
|
||||
minutes := int(recoveryCodeTTL / time.Minute)
|
||||
subject = "Felis break-glass recovery code / 紧急恢复验证码"
|
||||
body = fmt.Sprintf(`Someone with root on %[1]s (OS user %[2]s) opened felis breakGlass and named your staff account %[3]q to %[4]s.
|
||||
|
||||
Recovery code: %[6]s
|
||||
It works for %[7]d minutes.
|
||||
|
||||
If this was not you, root on that machine is in someone else's hands: change its credentials and read the audit log for break_glass entries.
|
||||
|
||||
有人在 %[1]s 上以 root 身份(系统用户 %[2]s)打开了 felis breakGlass,指名你的管理员账号 %[3]q 来%[5]s。
|
||||
|
||||
恢复验证码:%[6]s
|
||||
%[7]d 分钟内有效。
|
||||
|
||||
如果不是你本人,这台机器的 root 已落入他人之手:请更换它的凭据,并查看审计日志中的 break_glass 记录。
|
||||
`, host, osUser, admin, what, whatZH, code, minutes)
|
||||
return subject, body
|
||||
}
|
||||
|
||||
// maskEmail keeps the first character of the local part and the domain, enough for
|
||||
// the operator to recognise the address without putting it on screen whole.
|
||||
func maskEmail(email string) string {
|
||||
at := strings.LastIndex(email, "@")
|
||||
if at <= 0 {
|
||||
return "***"
|
||||
}
|
||||
return email[:1] + strings.Repeat("*", max(at-1, 3)) + email[at:]
|
||||
}
|
||||
|
||||
// hostRecoveryMailer opens the [smtp] relay from the host the way the watchdog does:
|
||||
// the password is the env var password_ref names when that is set, else the host
|
||||
// copy at passwordPath, else the felis-smtp Secret, whose absence means a relay
|
||||
// without AUTH. The cluster is reached only when the host copy is missing, so a
|
||||
// break-glass on a host whose k3s is down still gets its code.
|
||||
func hostRecoveryMailer(c config.SMTPConfig, passwordPath, controlNS string) func(context.Context) (recoveryMailer, error) {
|
||||
return func(ctx context.Context) (recoveryMailer, error) {
|
||||
if strings.TrimSpace(c.Host) == "" {
|
||||
return nil, errors.New("[smtp] is not configured in felis.toml")
|
||||
}
|
||||
if ref := c.PasswordRef; ref != "" && os.Getenv(ref) != "" {
|
||||
return smtpRelay(c, os.Getenv(ref)), nil
|
||||
}
|
||||
if password, ok, err := readHostCredential(passwordPath); err != nil {
|
||||
return nil, fmt.Errorf("read the relay password: %w", err)
|
||||
} else if ok {
|
||||
return smtpRelay(c, password), nil
|
||||
}
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("reach the cluster for the relay password: %w", err)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
|
||||
defer cancel()
|
||||
password, err := smtpSecretPassword(ctx, cl, controlNS)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read the relay password from %s/%s: %w", controlNS, platform.SMTPSecretName, err)
|
||||
}
|
||||
return smtpRelay(c, password), nil
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,498 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime/schema"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
|
||||
)
|
||||
|
||||
// These tests cover the recovery proof (#13): the mailed code's rules, where
|
||||
// naming an admin leads, what the mail says, how the console walks from a name to
|
||||
// a proven (or overridden) run, and what the audit row then records. No mail
|
||||
// leaves the process: the relay is a fake that keeps what it was handed.
|
||||
|
||||
type sentMail struct{ to, subject, body string }
|
||||
|
||||
type fakeRelay struct {
|
||||
sent []sentMail
|
||||
err error
|
||||
}
|
||||
|
||||
func (r *fakeRelay) SendNotice(_ context.Context, to, subject, body string) error {
|
||||
if r.err != nil {
|
||||
return r.err
|
||||
}
|
||||
r.sent = append(r.sent, sentMail{to, subject, body})
|
||||
return nil
|
||||
}
|
||||
|
||||
// sentCode pulls the code out of the one mail the relay carried.
|
||||
func (r *fakeRelay) sentCode(t *testing.T) string {
|
||||
t.Helper()
|
||||
if len(r.sent) != 1 {
|
||||
t.Fatalf("relay carried %d mails, want 1", len(r.sent))
|
||||
}
|
||||
_, after, ok := strings.Cut(r.sent[0].body, "Recovery code: ")
|
||||
if !ok || len(after) < 6 {
|
||||
t.Fatalf("mail carries no recovery code:\n%s", r.sent[0].body)
|
||||
}
|
||||
return after[:6]
|
||||
}
|
||||
|
||||
func relayConfig(r *fakeRelay, now func() time.Time) recoveryConfig {
|
||||
return recoveryConfig{
|
||||
open: func(context.Context) (recoveryMailer, error) { return r, nil },
|
||||
host: "felis-host-1",
|
||||
now: now,
|
||||
}
|
||||
}
|
||||
|
||||
func verifiedAdmin(username, email string) *api.StaffUser {
|
||||
return &api.StaffUser{ID: "usr-" + username, Username: username, Role: "owner", Email: email, EmailVerified: true}
|
||||
}
|
||||
|
||||
func TestRecoveryCodeRules(t *testing.T) {
|
||||
t0 := time.Date(2026, 9, 26, 8, 0, 0, 0, time.UTC)
|
||||
|
||||
t.Run("a fresh code is six digits and lives for the TTL", func(t *testing.T) {
|
||||
seen := map[string]bool{}
|
||||
for i := 0; i < 20; i++ {
|
||||
c, err := newRecoveryCode(t0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !isRecoveryCodeShape(c.value) {
|
||||
t.Fatalf("code %q is not six digits", c.value)
|
||||
}
|
||||
if !c.expires.Equal(t0.Add(recoveryCodeTTL)) {
|
||||
t.Fatalf("expires = %v, want %v", c.expires, t0.Add(recoveryCodeTTL))
|
||||
}
|
||||
seen[c.value] = true
|
||||
}
|
||||
if len(seen) < 2 {
|
||||
t.Error("twenty codes were all the same")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the right code is accepted, surrounding space ignored", func(t *testing.T) {
|
||||
c := &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
|
||||
if v := c.check(" 042917 ", t0); v != codeAccepted {
|
||||
t.Errorf("check = %v, want accepted", v)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("wrong codes count down and the last one exhausts it", func(t *testing.T) {
|
||||
c := &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
|
||||
for i := 1; i < recoveryCodeAttempts; i++ {
|
||||
if v := c.check("000000", t0); v != codeWrong {
|
||||
t.Fatalf("wrong code %d: check = %v, want wrong", i, v)
|
||||
}
|
||||
if left := c.attemptsLeft(); left != recoveryCodeAttempts-i {
|
||||
t.Fatalf("after %d wrong codes attemptsLeft = %d, want %d", i, left, recoveryCodeAttempts-i)
|
||||
}
|
||||
}
|
||||
if v := c.check("000000", t0); v != codeExhausted {
|
||||
t.Fatalf("wrong code %d: check = %v, want exhausted", recoveryCodeAttempts, v)
|
||||
}
|
||||
if v := c.check("042917", t0); v != codeExhausted {
|
||||
t.Errorf("the right code after exhaustion: check = %v, want exhausted", v)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an expired code accepts nothing", func(t *testing.T) {
|
||||
c := &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
|
||||
if v := c.check("042917", t0.Add(recoveryCodeTTL-time.Second)); v != codeAccepted {
|
||||
t.Fatalf("a second before expiry: check = %v, want accepted", v)
|
||||
}
|
||||
c = &recoveryCode{value: "042917", expires: t0.Add(recoveryCodeTTL)}
|
||||
if v := c.check("042917", t0.Add(recoveryCodeTTL)); v != codeExpired {
|
||||
t.Errorf("at expiry: check = %v, want expired", v)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestBeginRecovery(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
t0 := time.Date(2026, 9, 26, 8, 0, 0, 0, time.UTC)
|
||||
clock := func() time.Time { return t0 }
|
||||
|
||||
t.Run("mails a code to the named admin's verified address", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": verifiedAdmin("root", "[email protected]")}}
|
||||
r := &fakeRelay{}
|
||||
st, err := beginRecovery(ctx, f, relayConfig(r, clock), "root", "alice", bgAddOperator)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st.code == nil || st.skip != "" {
|
||||
t.Fatalf("start = %+v, want a code and no skip", st)
|
||||
}
|
||||
if st.admin == nil || st.admin.Username != "root" {
|
||||
t.Fatalf("start.admin = %+v, want root", st.admin)
|
||||
}
|
||||
if got := r.sentCode(t); got != st.code.value {
|
||||
t.Errorf("mailed code %q, want the code the console checks (%q)", got, st.code.value)
|
||||
}
|
||||
m := r.sent[0]
|
||||
if m.to != "[email protected]" {
|
||||
t.Errorf("mail went to %q, want [email protected]", m.to)
|
||||
}
|
||||
for _, want := range []string{"felis-host-1", "OS user alice", `"root"`, "add an Operator account", "添加 Operator 账号", "10 minutes"} {
|
||||
if !strings.Contains(m.body, want) {
|
||||
t.Errorf("mail body lacks %q:\n%s", want, m.body)
|
||||
}
|
||||
}
|
||||
if !st.code.expires.Equal(t0.Add(recoveryCodeTTL)) {
|
||||
t.Errorf("code expires %v, want %v", st.code.expires, t0.Add(recoveryCodeTTL))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the Owner reset is named as such", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": verifiedAdmin("root", "[email protected]")}}
|
||||
r := &fakeRelay{}
|
||||
if _, err := beginRecovery(ctx, f, relayConfig(r, clock), "root", "alice", bgProvisionOwner); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if body := r.sent[0].body; !strings.Contains(body, "reset the Owner account") || !strings.Contains(body, "重置 Owner 账号") {
|
||||
t.Errorf("mail body does not name the Owner reset:\n%s", body)
|
||||
}
|
||||
})
|
||||
|
||||
// Each way the code cannot go out ends in a skip reason and no mail.
|
||||
unverified := verifiedAdmin("root", "[email protected]")
|
||||
unverified.EmailVerified = false
|
||||
noEmail := verifiedAdmin("root", "")
|
||||
cases := []struct {
|
||||
name string
|
||||
user *api.StaffUser
|
||||
rc func(r *fakeRelay) recoveryConfig
|
||||
skip string
|
||||
detail string
|
||||
wantAdmin bool
|
||||
relayError error
|
||||
}{
|
||||
{name: "unknown admin", user: nil, rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) }, skip: otpSkipUnknownAdmin},
|
||||
{name: "unverified address", user: unverified, rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) }, skip: otpSkipNoVerifiedEmail, wantAdmin: true},
|
||||
{name: "no address", user: noEmail, rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) }, skip: otpSkipNoVerifiedEmail, wantAdmin: true},
|
||||
{name: "no relay wired", user: verifiedAdmin("root", "[email protected]"), rc: func(*fakeRelay) recoveryConfig { return recoveryConfig{} }, skip: otpSkipNoRelay, detail: "this console has no mail relay", wantAdmin: true},
|
||||
{name: "relay cannot open", user: verifiedAdmin("root", "[email protected]"), rc: func(*fakeRelay) recoveryConfig {
|
||||
return recoveryConfig{open: func(context.Context) (recoveryMailer, error) {
|
||||
return nil, errors.New("[smtp] is not configured in felis.toml")
|
||||
}}
|
||||
}, skip: otpSkipNoRelay, detail: "[smtp] is not configured in felis.toml", wantAdmin: true},
|
||||
{name: "send fails", user: verifiedAdmin("root", "[email protected]"), rc: func(r *fakeRelay) recoveryConfig { return relayConfig(r, clock) },
|
||||
skip: otpSkipSendFailed, detail: "554 relay refused", wantAdmin: true, relayError: errors.New("554 relay refused")},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{}}
|
||||
if tc.user != nil {
|
||||
f.users["root"] = tc.user
|
||||
}
|
||||
r := &fakeRelay{err: tc.relayError}
|
||||
st, err := beginRecovery(ctx, f, tc.rc(r), "root", "alice", bgProvisionOwner)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if st.code != nil || st.skip != tc.skip || st.detail != tc.detail {
|
||||
t.Errorf("start = {code:%v skip:%q detail:%q}, want no code, skip %q, detail %q", st.code, st.skip, st.detail, tc.skip, tc.detail)
|
||||
}
|
||||
if (st.admin != nil) != tc.wantAdmin {
|
||||
t.Errorf("start.admin = %+v, want present=%v", st.admin, tc.wantAdmin)
|
||||
}
|
||||
if len(r.sent) != 0 {
|
||||
t.Errorf("relay carried %d mails, want none", len(r.sent))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("a datastore fault is an error", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{userErr: errors.New("db down")}
|
||||
if _, err := beginRecovery(ctx, f, relayConfig(&fakeRelay{}, clock), "root", "alice", bgProvisionOwner); err == nil {
|
||||
t.Fatal("want the store fault")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestMaskEmail(t *testing.T) {
|
||||
for in, want := range map[string]string{
|
||||
"[email protected]": "a****@example.com",
|
||||
"[email protected]": "a***@example.com",
|
||||
"@example.com": "***",
|
||||
"nonsense": "***",
|
||||
} {
|
||||
if got := maskEmail(in); got != want {
|
||||
t.Errorf("maskEmail(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHostRecoveryMailer(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
|
||||
t.Run("no [smtp] host is no relay", func(t *testing.T) {
|
||||
_, err := hostRecoveryMailer(config.SMTPConfig{}, hostSMTPPasswordPath, "felis")(ctx)
|
||||
if err == nil || !strings.Contains(err.Error(), "[smtp]") {
|
||||
t.Fatalf("err = %v, want it to name [smtp]", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the password_ref env var supplies the password", func(t *testing.T) {
|
||||
t.Setenv("FELIS_TEST_RELAY_PW", "from-env")
|
||||
off := false
|
||||
c := config.SMTPConfig{Host: "mail.example.com", Port: 2525, From: "[email protected]", Username: "felis", PasswordRef: "FELIS_TEST_RELAY_PW", RequireTLS: &off}
|
||||
got, err := hostRecoveryMailer(c, hostSMTPPasswordPath, "felis")(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
relay, ok := got.(*mail.SMTP)
|
||||
if !ok {
|
||||
t.Fatalf("relay is %T, want *mail.SMTP", got)
|
||||
}
|
||||
if relay.Password != "from-env" || relay.Host != "mail.example.com" || relay.Port != 2525 || relay.From != "[email protected]" || relay.Username != "felis" || relay.RequireTLS {
|
||||
t.Errorf("relay = %+v, want the [smtp] fields with the env password and TLS as configured", *relay)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestSMTPSecretPassword(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
scheme := haltScheme(t)
|
||||
|
||||
t.Run("reads the password key", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(&corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.SMTPSecretName},
|
||||
Data: map[string][]byte{platform.SMTPSecretPasswordKey: []byte("s3cret")},
|
||||
}).Build()
|
||||
if pw, err := smtpSecretPassword(ctx, cl, "felis"); err != nil || pw != "s3cret" {
|
||||
t.Errorf("smtpSecretPassword = (%q, %v), want (s3cret, nil)", pw, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a missing Secret is a relay without AUTH", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).Build()
|
||||
if pw, err := smtpSecretPassword(ctx, cl, "felis"); err != nil || pw != "" {
|
||||
t.Errorf("smtpSecretPassword = (%q, %v), want (\"\", nil)", pw, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("any other read failure is an error", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithInterceptorFuncs(interceptor.Funcs{
|
||||
Get: func(context.Context, client.WithWatch, client.ObjectKey, client.Object, ...client.GetOption) error {
|
||||
return apierrors.NewForbidden(schema.GroupResource{Resource: "secrets"}, platform.SMTPSecretName, errors.New("rbac"))
|
||||
},
|
||||
}).Build()
|
||||
if _, err := smtpSecretPassword(ctx, cl, "felis"); err == nil {
|
||||
t.Fatal("want the read failure")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// recoveryModel is an Owner-reset console for admin "root" (verified address
|
||||
// [email protected]), run by OS user alice, with the fake relay behind it.
|
||||
func recoveryModel(t *testing.T, f *fakeOwnerStore, r *fakeRelay, now func() time.Time) *ownerModel {
|
||||
t.Helper()
|
||||
if f.users == nil {
|
||||
f.users = map[string]*api.StaffUser{"root": verifiedAdmin("root", "[email protected]")}
|
||||
}
|
||||
return newOwnerModel(context.Background(), f, "alice", true).withRecovery(relayConfig(r, now))
|
||||
}
|
||||
|
||||
// nameAdmin submits the admin-name form and feeds the result of the send back in.
|
||||
func nameAdmin(t *testing.T, m *ownerModel, name string) *ownerModel {
|
||||
t.Helper()
|
||||
m.authUser = name
|
||||
_, cmd := m.onFormComplete()
|
||||
if m.step != owWorking {
|
||||
t.Fatalf("after naming the admin step = %v, want owWorking", m.step)
|
||||
}
|
||||
msg := findMsg[owAuthMsg](t, cmd)
|
||||
next, _ := m.Update(msg)
|
||||
return next.(*ownerModel)
|
||||
}
|
||||
|
||||
// findMsg runs a (possibly batched) command and returns the first T it produces.
|
||||
func findMsg[T any](t *testing.T, cmd tea.Cmd) T {
|
||||
t.Helper()
|
||||
var zero T
|
||||
if cmd == nil {
|
||||
t.Fatalf("no command, want one producing %T", zero)
|
||||
}
|
||||
switch msg := cmd().(type) {
|
||||
case T:
|
||||
return msg
|
||||
case tea.BatchMsg:
|
||||
for _, c := range msg {
|
||||
if c == nil {
|
||||
continue
|
||||
}
|
||||
if got, ok := c().(T); ok {
|
||||
return got
|
||||
}
|
||||
}
|
||||
}
|
||||
t.Fatalf("command produced no %T", zero)
|
||||
return zero
|
||||
}
|
||||
|
||||
// typeCode submits the code form with typed.
|
||||
func typeCode(t *testing.T, m *ownerModel, typed string) {
|
||||
t.Helper()
|
||||
if m.step != owCode {
|
||||
t.Fatalf("step = %v, want owCode", m.step)
|
||||
}
|
||||
m.codeInput = typed
|
||||
m.onFormComplete()
|
||||
}
|
||||
|
||||
// provisionAudit finishes the run as an Owner reset and returns its audit payload.
|
||||
func provisionAudit(t *testing.T, m *ownerModel, f *fakeOwnerStore) (api.AuditEntry, map[string]any) {
|
||||
t.Helper()
|
||||
if m.step != owProvision {
|
||||
t.Fatalf("step = %v, want owProvision", m.step)
|
||||
}
|
||||
m.username = "owner"
|
||||
msg := m.provisionCmd()().(owProvisionMsg)
|
||||
if msg.err != nil {
|
||||
t.Fatalf("provision: %v", msg.err)
|
||||
}
|
||||
return auditOf(t, f)
|
||||
}
|
||||
|
||||
func TestRecoveryConsoleFlow(t *testing.T) {
|
||||
t0 := time.Date(2026, 9, 26, 8, 0, 0, 0, time.UTC)
|
||||
fixed := func() time.Time { return t0 }
|
||||
|
||||
t.Run("the mailed code proves the admin and the audit says so", func(t *testing.T) {
|
||||
f, r := &fakeOwnerStore{}, &fakeRelay{}
|
||||
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
|
||||
typeCode(t, m, r.sentCode(t))
|
||||
if m.mode != "recovery" || m.accountable != "root" {
|
||||
t.Fatalf("mode=%q accountable=%q, want recovery attributed to root", m.mode, m.accountable)
|
||||
}
|
||||
e, payload := provisionAudit(t, m, f)
|
||||
if e.Actor != "root" || e.Action != "break_glass.recovery" {
|
||||
t.Errorf("audit = %+v, want actor=root action=break_glass.recovery", e)
|
||||
}
|
||||
if payload["verified"] != true || payload["verified_by"] != verifiedByEmailOTP || payload["code_sent_to"] != "[email protected]" || payload["os_user"] != "alice" {
|
||||
t.Errorf("payload = %v, want verified by email_otp to [email protected], os_user alice", payload)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a wrong code asks again, and the last wrong one leads to the override", func(t *testing.T) {
|
||||
f, r := &fakeOwnerStore{}, &fakeRelay{}
|
||||
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
|
||||
wrong := "000000"
|
||||
if r.sentCode(t) == wrong {
|
||||
wrong = "111111"
|
||||
}
|
||||
for i := 1; i < recoveryCodeAttempts; i++ {
|
||||
typeCode(t, m, wrong)
|
||||
if m.step != owCode || !strings.Contains(m.codeNote, "wrong") {
|
||||
t.Fatalf("wrong code %d: step=%v note=%q, want the code form again with a note", i, m.step, m.codeNote)
|
||||
}
|
||||
}
|
||||
typeCode(t, m, wrong)
|
||||
if m.step != owOverride || m.skip != otpSkipCodeRejected {
|
||||
t.Fatalf("after %d wrong codes step=%v skip=%q, want the override for code_rejected", recoveryCodeAttempts, m.step, m.skip)
|
||||
}
|
||||
m.onFormComplete() // OVERRIDE typed
|
||||
e, payload := provisionAudit(t, m, f)
|
||||
if e.Actor != "alice" || e.Action != "break_glass.root_override" {
|
||||
t.Errorf("audit = %+v, want actor=alice action=break_glass.root_override", e)
|
||||
}
|
||||
if payload["verified"] != false || payload["otp_skipped"] != otpSkipCodeRejected || payload["admin_account"] != "root" {
|
||||
t.Errorf("payload = %v, want unverified, otp_skipped=code_rejected, admin_account=root", payload)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a late code leads to the override", func(t *testing.T) {
|
||||
now := t0
|
||||
f, r := &fakeOwnerStore{}, &fakeRelay{}
|
||||
m := nameAdmin(t, recoveryModel(t, f, r, func() time.Time { return now }), "root")
|
||||
now = t0.Add(recoveryCodeTTL)
|
||||
typeCode(t, m, r.sentCode(t))
|
||||
if m.step != owOverride || m.skip != otpSkipCodeExpired {
|
||||
t.Fatalf("step=%v skip=%q, want the override for code_expired", m.step, m.skip)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("OVERRIDE at the code prompt goes on unverified, saying the operator skipped", func(t *testing.T) {
|
||||
f, r := &fakeOwnerStore{}, &fakeRelay{}
|
||||
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
|
||||
typeCode(t, m, breakGlassOverrideToken)
|
||||
if m.mode != "root_override" || m.accountable != "alice" {
|
||||
t.Fatalf("mode=%q accountable=%q, want root_override as alice", m.mode, m.accountable)
|
||||
}
|
||||
_, payload := provisionAudit(t, m, f)
|
||||
if payload["verified"] != false || payload["otp_skipped"] != otpSkipByOperator {
|
||||
t.Errorf("payload = %v, want unverified, otp_skipped=operator_skipped", payload)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a relay failure leads to the override naming it", func(t *testing.T) {
|
||||
f, r := &fakeOwnerStore{}, &fakeRelay{err: errors.New("dial tcp 10.0.0.9:587: connect: connection refused")}
|
||||
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
|
||||
if m.step != owOverride || m.skip != otpSkipSendFailed {
|
||||
t.Fatalf("step=%v skip=%q, want the override for send_failed", m.step, m.skip)
|
||||
}
|
||||
if reason := m.overrideReason(); !strings.Contains(reason, "connection refused") {
|
||||
t.Errorf("override reason %q does not name the failure", reason)
|
||||
}
|
||||
m.onFormComplete()
|
||||
_, payload := provisionAudit(t, m, f)
|
||||
if payload["otp_skipped"] != otpSkipSendFailed || !strings.Contains(payload["otp_skip_detail"].(string), "connection refused") {
|
||||
t.Errorf("payload = %v, want otp_skipped=send_failed with the relay's error", payload)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("esc at the code prompt starts over and forgets the code", func(t *testing.T) {
|
||||
f, r := &fakeOwnerStore{}, &fakeRelay{}
|
||||
m := nameAdmin(t, recoveryModel(t, f, r, fixed), "root")
|
||||
code := r.sentCode(t)
|
||||
next, _ := m.Update(key(tea.KeyEsc))
|
||||
m = next.(*ownerModel)
|
||||
if m.step != owAuth || m.code != nil || m.admin != nil {
|
||||
t.Fatalf("after esc step=%v code=%v admin=%v, want owAuth with the attempt forgotten", m.step, m.code, m.admin)
|
||||
}
|
||||
// The old code cannot be replayed: the next name mails a new one.
|
||||
m = nameAdmin(t, m, "root")
|
||||
if len(r.sent) != 2 {
|
||||
t.Fatalf("relay carried %d mails, want a second one for the new attempt", len(r.sent))
|
||||
}
|
||||
_, fresh, _ := strings.Cut(r.sent[1].body, "Recovery code: ")
|
||||
if m.code.value != fresh[:6] {
|
||||
t.Errorf("the console checks %q, want the newly mailed %q (old one was %q)", m.code.value, fresh[:6], code)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestRootHandsRecoveryToAccountOperations(t *testing.T) {
|
||||
for _, op := range []bgOperation{bgProvisionOwner, bgAddOperator} {
|
||||
m := newTestRoot(true, consoleModeBreakGlass, "")
|
||||
m.recovery = recoveryConfig{host: "felis-host-1"}
|
||||
m = drive(t, m, menuChoiceMsg{op: op})
|
||||
om, ok := m.screen.(*ownerModel)
|
||||
if !ok {
|
||||
t.Fatalf("op %v: screen = %T, want *ownerModel", op, m.screen)
|
||||
}
|
||||
if om.recovery.host != "felis-host-1" {
|
||||
t.Errorf("op %v: the account screen has no relay config; its codes could never go out", op)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -269,71 +269,57 @@ func TestEnableLocalAuth(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuthenticateAdmin(t *testing.T) {
|
||||
func TestResolveAdmin(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
|
||||
// Password verification is gone (passwordless design): authenticateAdmin now only
|
||||
// resolves the named admin so recovery can attribute the audit to a real identity.
|
||||
// The security boundary is the break-glass root gate, not a typed secret.
|
||||
// resolveAdmin only finds the staff account a typed name points at; proving the
|
||||
// operator holds it is the mailed code's job (beginRecovery).
|
||||
|
||||
t.Run("resolves an existing admin for attribution", func(t *testing.T) {
|
||||
t.Run("resolves an existing admin", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": mkAdmin("root")}}
|
||||
matched, ok, err := authenticateAdmin(ctx, f, "root")
|
||||
u, err := resolveAdmin(ctx, f, " root ")
|
||||
if err != nil {
|
||||
t.Fatalf("authenticateAdmin: %v", err)
|
||||
t.Fatalf("resolveAdmin: %v", err)
|
||||
}
|
||||
if !ok {
|
||||
t.Fatal("ok = false, want true for an existing admin")
|
||||
}
|
||||
if matched != "root" {
|
||||
t.Errorf("matched = %q, want root", matched)
|
||||
if u == nil || u.Username != "root" {
|
||||
t.Fatalf("resolveAdmin = %+v, want the root admin", u)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a non-admin role can never attribute a break-glass", func(t *testing.T) {
|
||||
t.Run("a non-admin role is no staff account", func(t *testing.T) {
|
||||
player := mkAdmin("alice")
|
||||
player.Role = "user" // a player row is not staff
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"alice": player}}
|
||||
_, ok, err := authenticateAdmin(ctx, f, "alice")
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if ok {
|
||||
t.Error("ok = true, want false for a non-admin role")
|
||||
if u, err := resolveAdmin(ctx, f, "alice"); u != nil || err != nil {
|
||||
t.Errorf("resolveAdmin(player) = (%+v, %v), want (nil, nil)", u, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the owner role attributes like an admin", func(t *testing.T) {
|
||||
t.Run("the owner role counts as staff", func(t *testing.T) {
|
||||
owner := mkAdmin("root")
|
||||
owner.Role = "owner" // the platform owner is staff too (migration 0011)
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": owner}}
|
||||
matched, ok, err := authenticateAdmin(ctx, f, "root")
|
||||
if err != nil || !ok || matched != "root" {
|
||||
t.Fatalf("authenticateAdmin(owner) = (%q, %v, %v), want (root, true, nil)", matched, ok, err)
|
||||
if u, err := resolveAdmin(ctx, f, "root"); err != nil || u == nil {
|
||||
t.Fatalf("resolveAdmin(owner) = (%+v, %v), want the owner", u, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an unknown user is a non-match, not an error", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{}
|
||||
_, ok, err := authenticateAdmin(ctx, f, "nobody")
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if ok {
|
||||
t.Error("ok = true, want false for an unknown user")
|
||||
t.Run("an unknown user is nil, not an error", func(t *testing.T) {
|
||||
if u, err := resolveAdmin(ctx, &fakeOwnerStore{}, "nobody"); u != nil || err != nil {
|
||||
t.Errorf("resolveAdmin(unknown) = (%+v, %v), want (nil, nil)", u, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an empty username is a non-match with no store call", func(t *testing.T) {
|
||||
t.Run("an empty username makes no store call", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{userErr: errors.New("must not be called")}
|
||||
if _, ok, err := authenticateAdmin(ctx, f, ""); ok || err != nil {
|
||||
t.Errorf("empty username: ok=%v err=%v, want false,nil", ok, err)
|
||||
if u, err := resolveAdmin(ctx, f, " "); u != nil || err != nil {
|
||||
t.Errorf("empty username: (%+v, %v), want (nil, nil)", u, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a datastore fault is surfaced", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{userErr: errors.New("db down")}
|
||||
if _, _, err := authenticateAdmin(ctx, f, "root"); err == nil {
|
||||
if _, err := resolveAdmin(ctx, f, "root"); err == nil {
|
||||
t.Fatal("want error when the store fails")
|
||||
}
|
||||
})
|
||||
@@ -401,6 +387,8 @@ func TestPerformBreakGlass(t *testing.T) {
|
||||
osUser: "alice",
|
||||
ownerUsername: "owner",
|
||||
attemptedAdmin: "root",
|
||||
verifiedBy: verifiedByEmailOTP,
|
||||
codeSentTo: "[email protected]",
|
||||
}
|
||||
out, err := performBreakGlass(ctx, f, op)
|
||||
if err != nil {
|
||||
@@ -425,6 +413,23 @@ func TestPerformBreakGlass(t *testing.T) {
|
||||
if payload["admin_account"] != "root" {
|
||||
t.Errorf("payload.admin_account = %v, want root", payload["admin_account"])
|
||||
}
|
||||
if payload["verified_by"] != verifiedByEmailOTP || payload["code_sent_to"] != "[email protected]" {
|
||||
t.Errorf("payload = %v, want verified_by=email_otp [email protected]", payload)
|
||||
}
|
||||
if _, present := payload["otp_skipped"]; present {
|
||||
t.Error("a proven recovery carries no otp_skipped")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("recovery without a proof is recorded unverified", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{}
|
||||
op := breakGlassOp{mode: "recovery", accountable: "root", osUser: "alice", ownerUsername: "owner", attemptedAdmin: "root"}
|
||||
if _, err := performBreakGlass(ctx, f, op); err != nil {
|
||||
t.Fatalf("performBreakGlass: %v", err)
|
||||
}
|
||||
if _, payload := auditOf(t, f); payload["verified"] != false {
|
||||
t.Errorf("payload.verified = %v, want false: only a mailed code verifies", payload["verified"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("root override records an unverified row attributed to the OS user", func(t *testing.T) {
|
||||
@@ -435,6 +440,8 @@ func TestPerformBreakGlass(t *testing.T) {
|
||||
osUser: "alice",
|
||||
ownerUsername: "owner",
|
||||
attemptedAdmin: "typo-admin",
|
||||
otpSkipped: otpSkipSendFailed,
|
||||
otpSkipDetail: "dial tcp: connection refused",
|
||||
}
|
||||
if _, err := performBreakGlass(ctx, f, op); err != nil {
|
||||
t.Fatalf("performBreakGlass: %v", err)
|
||||
@@ -450,6 +457,13 @@ func TestPerformBreakGlass(t *testing.T) {
|
||||
if payload["admin_account"] != "typo-admin" {
|
||||
t.Errorf("payload.admin_account = %v, want typo-admin", payload["admin_account"])
|
||||
}
|
||||
// Why no code proved anyone is part of the record.
|
||||
if payload["otp_skipped"] != otpSkipSendFailed || payload["otp_skip_detail"] != "dial tcp: connection refused" {
|
||||
t.Errorf("payload = %v, want otp_skipped=send_failed with its detail", payload)
|
||||
}
|
||||
if _, present := payload["verified_by"]; present {
|
||||
t.Error("an override carries no verified_by")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an audit failure does not fail the recovery", func(t *testing.T) {
|
||||
@@ -616,6 +630,8 @@ func TestPerformAddOperator(t *testing.T) {
|
||||
ownerUsername: "ops-jordan",
|
||||
ownerEmail: "[email protected]",
|
||||
attemptedAdmin: "root",
|
||||
verifiedBy: verifiedByEmailOTP,
|
||||
codeSentTo: "[email protected]",
|
||||
}
|
||||
out, err := performAddOperator(ctx, f, op)
|
||||
if err != nil {
|
||||
@@ -642,14 +658,14 @@ func TestPerformAddOperator(t *testing.T) {
|
||||
if _, present := payload["owner"]; present {
|
||||
t.Error("payload.owner present, want the new account under the operator key")
|
||||
}
|
||||
if payload["verified"] != true || payload["admin_account"] != "root" {
|
||||
t.Errorf("payload = %v, want verified=true admin_account=root", payload)
|
||||
if payload["verified"] != true || payload["admin_account"] != "root" || payload["verified_by"] != verifiedByEmailOTP {
|
||||
t.Errorf("payload = %v, want verified=true admin_account=root verified_by=email_otp", payload)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("root override records an unverified operator row", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{}
|
||||
op := breakGlassOp{mode: "root_override", accountable: "alice", osUser: "alice", ownerUsername: "ops", attemptedAdmin: "typo-admin"}
|
||||
op := breakGlassOp{mode: "root_override", accountable: "alice", osUser: "alice", ownerUsername: "ops", attemptedAdmin: "typo-admin", otpSkipped: otpSkipUnknownAdmin}
|
||||
if _, err := performAddOperator(ctx, f, op); err != nil {
|
||||
t.Fatalf("performAddOperator: %v", err)
|
||||
}
|
||||
@@ -657,8 +673,8 @@ func TestPerformAddOperator(t *testing.T) {
|
||||
t.Fatalf("want 1 insert, got %d", len(f.inserts))
|
||||
}
|
||||
_, payload := auditOf(t, f)
|
||||
if payload["verified"] != false {
|
||||
t.Errorf("payload.verified = %v, want false for root_override", payload["verified"])
|
||||
if payload["verified"] != false || payload["otp_skipped"] != otpSkipUnknownAdmin {
|
||||
t.Errorf("payload = %v, want verified=false otp_skipped=unknown_admin", payload)
|
||||
}
|
||||
})
|
||||
|
||||
|
||||
+53
-1
@@ -11,12 +11,14 @@ import (
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// cmdConverge is the explicit convergence pass over already-installed system
|
||||
// servers (#1), plus the idle-stop default for user servers that predate it. Provisioning is create-if-absent, so a field the desired spec
|
||||
// servers (#1), plus the idle-stop default for user servers that predate it, and
|
||||
// with -user-rcon their RCON block (#3). Provisioning is create-if-absent, so a field the desired spec
|
||||
// gained after an install (spec.rcon, spec.startup.healthHTTPPort, a derived env
|
||||
// key) never reaches the existing CR — and nothing says so. This command fills
|
||||
// exactly those zero-value fields; see convergeSystemServers for the full contract
|
||||
@@ -29,6 +31,7 @@ func cmdConverge(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("converge", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
|
||||
userRcon := fs.Bool("user-rcon", false, "also turn RCON on for user servers created before it was the default")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
@@ -59,6 +62,7 @@ func cmdConverge(args []string, stdout, stderr io.Writer) int {
|
||||
defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
|
||||
|
||||
outcomes = append(outcomes, convergeUserServerIdle(context.Background(), cl, cfg.K8s.Namespace)...)
|
||||
outcomes = append(outcomes, convergeUserServerRcon(context.Background(), cl, cfg.K8s.Namespace, *userRcon)...)
|
||||
|
||||
fmt.Fprintln(stdout, "felis converge: filling fields an installed server predates (operator-set values are never overwritten):")
|
||||
exit := 0
|
||||
@@ -113,3 +117,51 @@ func convergeUserServerIdle(ctx context.Context, cl client.Client, namespace str
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// convergeUserServerRcon handles user servers created before RCON was part of every
|
||||
// new server (694e3cb): spec.rcon entirely unset. Such a server has a dead console,
|
||||
// reports nobody online, and never idles out, because all three ride RCON.
|
||||
//
|
||||
// Only with fill does it turn RCON on, with the same block CreateServer writes
|
||||
// today; without it each such server gets a line saying so. The fill is opt-in
|
||||
// because the operator gates readiness on the RCON probe: a server whose image does
|
||||
// not open the listener RCON_PASSWORD asks for would sit in Starting until it is
|
||||
// marked Failed. Felis's own paper and lobby images open it; an image a user brought
|
||||
// may not, and only the operator running this can tell. A server that already
|
||||
// carries any RCON setting (on or off) is left alone and produces no line.
|
||||
func convergeUserServerRcon(ctx context.Context, cl client.Client, namespace string, fill bool) []systemServerOutcome {
|
||||
var list v1alpha1.MinecraftServerList
|
||||
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
|
||||
return []systemServerOutcome{{name: "user servers", err: fmt.Errorf("list servers: %w", err)}}
|
||||
}
|
||||
var out []systemServerOutcome
|
||||
for i := range list.Items {
|
||||
ms := &list.Items[i]
|
||||
if ms.Labels[v1alpha1.LabelSystemRole] != "" || ms.Spec.Rcon != (v1alpha1.RconSpec{}) {
|
||||
continue
|
||||
}
|
||||
if !fill {
|
||||
out = append(out, systemServerOutcome{name: ms.Name, available: true,
|
||||
skipped: "no RCON (console, online count and idle stop are off); once its image serves RCON, sudo felis converge -user-rcon turns it on"})
|
||||
continue
|
||||
}
|
||||
changed, err := patchOnConflictRetry(ctx, cl, ms, func() bool {
|
||||
if ms.Spec.Rcon != (v1alpha1.RconSpec{}) {
|
||||
return false
|
||||
}
|
||||
ms.Spec.Rcon = v1alpha1.RconSpec{
|
||||
Enabled: true,
|
||||
SecretRef: v1alpha1.SecretKeyRef{Name: naming.RconSecretName(ms.Name), Key: naming.RconSecretKey},
|
||||
}
|
||||
return true
|
||||
})
|
||||
if err != nil {
|
||||
out = append(out, systemServerOutcome{name: ms.Name, err: fmt.Errorf("converge %s: %w", ms.Name, err)})
|
||||
continue
|
||||
}
|
||||
if changed {
|
||||
out = append(out, systemServerOutcome{name: ms.Name, available: true, updated: true, changes: []string{"spec.rcon"}})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -229,3 +229,67 @@ func TestConvergeUserServerIdle(t *testing.T) {
|
||||
t.Fatalf("second pass = %+v, want nothing to do", again)
|
||||
}
|
||||
}
|
||||
|
||||
// TestConvergeUserServerRcon reports a user server with no RCON block at all and
|
||||
// fills it only when asked, with the block CreateServer writes. RCON turned off on
|
||||
// purpose, a server with its own secret, and a system server stay as they are and
|
||||
// produce no line.
|
||||
func TestConvergeUserServerRcon(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
mk := func(name string, rcon v1alpha1.RconSpec, role string) *v1alpha1.MinecraftServer {
|
||||
ms := &v1alpha1.MinecraftServer{}
|
||||
ms.Name, ms.Namespace = name, "minecraft"
|
||||
ms.Spec.Rcon = rcon
|
||||
if role != "" {
|
||||
ms.Labels = map[string]string{v1alpha1.LabelSystemRole: role}
|
||||
}
|
||||
return ms
|
||||
}
|
||||
own := v1alpha1.RconSpec{Enabled: true, Port: 25580, SecretRef: v1alpha1.SecretKeyRef{Name: "own", Key: "pw"}}
|
||||
off := v1alpha1.RconSpec{SecretRef: v1alpha1.SecretKeyRef{Name: "rcon-off", Key: naming.RconSecretKey}}
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||
mk("demo", v1alpha1.RconSpec{}, ""),
|
||||
mk("off", off, ""),
|
||||
mk("own", own, ""),
|
||||
mk(naming.SystemLobbyServer, v1alpha1.RconSpec{}, naming.SystemLobbyServer),
|
||||
).Build()
|
||||
get := func(name string) v1alpha1.RconSpec {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
|
||||
t.Fatalf("get %s: %v", name, err)
|
||||
}
|
||||
return ms.Spec.Rcon
|
||||
}
|
||||
|
||||
report := convergeUserServerRcon(ctx, cl, "minecraft", false)
|
||||
if len(report) != 1 || report[0].name != "demo" || report[0].updated || report[0].err != nil ||
|
||||
!strings.Contains(report[0].skipped, "-user-rcon") {
|
||||
t.Fatalf("report = %+v, want one skipped line for demo naming -user-rcon", report)
|
||||
}
|
||||
if got := get("demo"); got != (v1alpha1.RconSpec{}) {
|
||||
t.Fatalf("the report-only pass wrote demo's rcon: %+v", got)
|
||||
}
|
||||
|
||||
filled := convergeUserServerRcon(ctx, cl, "minecraft", true)
|
||||
if len(filled) != 1 || filled[0].name != "demo" || !filled[0].updated || filled[0].err != nil {
|
||||
t.Fatalf("fill = %+v, want exactly one update for demo", filled)
|
||||
}
|
||||
want := map[string]v1alpha1.RconSpec{
|
||||
"demo": {Enabled: true, SecretRef: v1alpha1.SecretKeyRef{
|
||||
Name: naming.RconSecretName("demo"), Key: naming.RconSecretKey}},
|
||||
"off": off,
|
||||
"own": own,
|
||||
naming.SystemLobbyServer: {},
|
||||
}
|
||||
for name, rcon := range want {
|
||||
if got := get(name); got != rcon {
|
||||
t.Errorf("%s rcon = %+v, want %+v", name, got, rcon)
|
||||
}
|
||||
}
|
||||
for _, fill := range []bool{false, true} {
|
||||
if again := convergeUserServerRcon(ctx, cl, "minecraft", fill); len(again) != 0 {
|
||||
t.Fatalf("second pass (fill=%v) = %+v, want nothing to do", fill, again)
|
||||
}
|
||||
}
|
||||
}
|
||||
+98
-17
@@ -7,6 +7,7 @@ import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
neturl "net/url"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
@@ -15,6 +16,7 @@ import (
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/retention"
|
||||
)
|
||||
|
||||
@@ -34,6 +36,7 @@ var defaultKeep = map[string]int{
|
||||
dbbackup.LabelDaily: 14,
|
||||
dbbackup.LabelPreMigrate: 10,
|
||||
dbbackup.LabelPreRestore: 5,
|
||||
dbbackup.LabelOffsite: 1,
|
||||
}
|
||||
|
||||
// cmdDB implements `felis db`: logical backups of the control-plane database
|
||||
@@ -92,17 +95,49 @@ func parseWithArg(fs *flag.FlagSet, args []string) (string, bool) {
|
||||
}
|
||||
|
||||
func dbDatabaseURL(path string) (string, error) {
|
||||
db, err := dbDatabase(path)
|
||||
return db.URL, err
|
||||
}
|
||||
|
||||
func dbDatabase(path string) (config.DatabaseConfig, error) {
|
||||
cfg, err := config.Load(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
return config.DatabaseConfig{}, err
|
||||
}
|
||||
return cfg.Database.URL, nil
|
||||
return cfg.Database, nil
|
||||
}
|
||||
|
||||
// dbTools places pg_dump, pg_restore and psql. The installer's database runs in
|
||||
// k3s and the host has no PostgreSQL client, so when the host config names the
|
||||
// [database] deployment the tools run in its postgres container over the
|
||||
// container's socket, as the URL's role on the URL's database. Otherwise they
|
||||
// come from PATH and connect with the URL.
|
||||
func dbTools(db config.DatabaseConfig) (dbbackup.Tools, error) {
|
||||
if db.Deployment == "" {
|
||||
return dbbackup.Tools{}, nil
|
||||
}
|
||||
ns, name, _ := strings.Cut(db.Deployment, "/")
|
||||
u, err := neturl.Parse(db.URL)
|
||||
if err != nil || u.User == nil || u.User.Username() == "" || strings.TrimPrefix(u.Path, "/") == "" {
|
||||
return dbbackup.Tools{}, errors.New("[database] url must name the role and the database to run the tools in the database's pod")
|
||||
}
|
||||
return dbbackup.Tools{
|
||||
Exec: []string{"k3s", "kubectl", "exec", "-i", "-n", ns, "deploy/" + name, "-c", platform.PostgresContainer, "--"},
|
||||
Conn: fmt.Sprintf("host=%s port=%d dbname=%s user=%s connect_timeout=15",
|
||||
platform.PostgresSocketDir, platform.PostgresPort,
|
||||
libpqQuote(strings.TrimPrefix(u.Path, "/")), libpqQuote(u.User.Username())),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// libpqQuote renders v as a single-quoted libpq connection-string value.
|
||||
func libpqQuote(v string) string {
|
||||
return "'" + strings.NewReplacer(`\`, `\\`, `'`, `\'`).Replace(v) + "'"
|
||||
}
|
||||
|
||||
func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||
label := fs.String("label", dbbackup.LabelManual, "bundle label; daily/pre-migrate/pre-restore bundles are pruned, manual ones never")
|
||||
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, manual all)")
|
||||
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, offsite 1, manual all)")
|
||||
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory to bundle ("" for none)`)
|
||||
noServers := fs.Bool("no-servers", false, "leave the MinecraftServer objects out of the bundle")
|
||||
metrics := fs.String("metrics-file", "", "node-exporter textfile to rewrite on success (e.g. /var/lib/node_exporter/textfile_collector/felis_db_backup.prom)")
|
||||
@@ -113,7 +148,12 @@ func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
|
||||
fmt.Fprint(stderr, dbUsage)
|
||||
return 2
|
||||
}
|
||||
url, err := dbDatabaseURL(*cfgPath)
|
||||
db, err := dbDatabase(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
tools, err := dbTools(db)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
|
||||
return 1
|
||||
@@ -122,7 +162,7 @@ func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
|
||||
*keep = defaultKeep[*label]
|
||||
}
|
||||
o := dbbackup.BackupOptions{
|
||||
DatabaseURL: url, Dir: *dir, Label: *label, Keep: *keep,
|
||||
DatabaseURL: db.URL, Tools: tools, Dir: *dir, Label: *label, Keep: *keep,
|
||||
StateDir: *stateDir, Version: resolvedVersion(), Log: stderr,
|
||||
MetricsFile: *metrics, Record: true,
|
||||
}
|
||||
@@ -132,6 +172,14 @@ func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
|
||||
defer cancel()
|
||||
path, err := dbbackup.Backup(ctx, o)
|
||||
if errors.Is(err, dbbackup.ErrServersMissing) {
|
||||
// The bundle is on disk and holds the database; the exit status fails
|
||||
// the timer's run so the gap shows in systemctl and the journal, and
|
||||
// the panel and the watchdog read it from the record and the manifest.
|
||||
fmt.Fprintf(stdout, "felis db backup: wrote %s\n", path)
|
||||
fmt.Fprintf(stderr, "felis db backup: %v\n a restore from %s brings back the database but no servers; check `k3s kubectl get minecraftservers -A`, then run `felis db backup` again\n", err, filepath.Base(path))
|
||||
return 1
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
|
||||
return 1
|
||||
@@ -172,12 +220,17 @@ func dbRestore(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.W
|
||||
return 1
|
||||
}
|
||||
if !*yes {
|
||||
fmt.Fprintf(stderr, "felis db restore: this replaces every table in the felis database with %s (%s, taken %s, schema %d).\n",
|
||||
filepath.Base(bundle), m.Label, m.CreatedAt.Format(time.RFC3339), m.SchemaVersion)
|
||||
fmt.Fprintf(stderr, "felis db restore: this replaces every table in the felis database with %s (%s, taken %s, schema %d, holding %s).\n",
|
||||
filepath.Base(bundle), m.Label, m.CreatedAt.Format(time.RFC3339), m.SchemaVersion, m.Counts.String())
|
||||
fmt.Fprintln(stderr, "Scale felis-api and felis-operator to 0 first, then re-run with -yes.")
|
||||
return 2
|
||||
}
|
||||
url, err := dbDatabaseURL(*cfgPath)
|
||||
db, err := dbDatabase(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
tools, err := dbTools(db)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
|
||||
return 1
|
||||
@@ -185,7 +238,7 @@ func dbRestore(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.W
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
|
||||
defer cancel()
|
||||
_, safety, err := dbbackup.Restore(ctx, dbbackup.RestoreOptions{
|
||||
DatabaseURL: url, Bundle: bundle, Dir: *dir, Force: *force, SkipSafetyBackup: *noSafety,
|
||||
DatabaseURL: db.URL, Tools: tools, Bundle: bundle, Dir: *dir, Force: *force, SkipSafetyBackup: *noSafety,
|
||||
Safety: dbbackup.BackupOptions{Keep: defaultKeep[dbbackup.LabelPreRestore], StateDir: *stateDir,
|
||||
Version: resolvedVersion(), ExportServers: exportMinecraftServers},
|
||||
Log: stderr,
|
||||
@@ -224,8 +277,9 @@ func dbVerify(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wr
|
||||
fmt.Fprintf(stderr, "felis db verify: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "%s: ok\n taken %s (%s)\n felis %s\n schema %d\n %s\n",
|
||||
filepath.Base(bundle), m.CreatedAt.Format(time.RFC3339), m.Label, orUnknown(m.FelisVersion), m.SchemaVersion, orUnknown(m.PGDumpVersion))
|
||||
fmt.Fprintf(stdout, "%s: ok\n taken %s (%s)\n felis %s\n schema %d\n holds %s\n %s\n",
|
||||
filepath.Base(bundle), m.CreatedAt.Format(time.RFC3339), m.Label, orUnknown(m.FelisVersion), m.SchemaVersion,
|
||||
m.Counts.String(), orUnknown(m.PGDumpVersion))
|
||||
for _, f := range m.Files {
|
||||
if f.Link != "" {
|
||||
fmt.Fprintf(stdout, " %-40s -> %s\n", f.Name, f.Link)
|
||||
@@ -361,10 +415,11 @@ func humanBytes(n int64) string {
|
||||
return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp])
|
||||
}
|
||||
|
||||
// dbCheck is the freshness probe: exit 1 when the newest bundle is missing or
|
||||
// older than -max-age, for a monitor or the break-glass console to act on.
|
||||
// dbCheck is the freshness probe: exit 1 when the newest daily bundle is
|
||||
// missing or older than -max-age, for a monitor or the break-glass console to
|
||||
// act on.
|
||||
func dbCheck(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||
maxAge := fs.Duration("max-age", dbbackup.StaleAfter, "oldest acceptable newest bundle")
|
||||
maxAge := fs.Duration("max-age", dbbackup.StaleAfter, "oldest acceptable newest daily bundle")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -373,14 +428,40 @@ func dbCheck(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Wri
|
||||
fmt.Fprintf(stderr, "felis db check: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis db check: ok, newest backup %s (%s ago)\n", b.Name, dbbackup.Age(time.Since(b.Created)))
|
||||
fmt.Fprintf(stdout, "felis db check: ok, newest daily backup %s (%s ago)\n", b.Name, dbbackup.Age(time.Since(b.Created)))
|
||||
return 0
|
||||
}
|
||||
|
||||
// exportMinecraftServers reads every MinecraftServer through the host's k3s
|
||||
// serverExportTries and serverExportRetry are how long a backup waits out a
|
||||
// cluster that is briefly away (an apiserver restart) before its bundle goes
|
||||
// without the MinecraftServer objects.
|
||||
const serverExportTries = 3
|
||||
|
||||
var serverExportRetry = 10 * time.Second
|
||||
|
||||
// exportMinecraftServers is dbbackup's ExportServers on the host: the
|
||||
// MinecraftServer objects through k3s kubectl, tried serverExportTries times.
|
||||
func exportMinecraftServers(ctx context.Context) ([]byte, error) {
|
||||
for try := 1; ; try++ {
|
||||
out, err := getMinecraftServers(ctx)
|
||||
if err == nil {
|
||||
return out, nil
|
||||
}
|
||||
if try == serverExportTries {
|
||||
return nil, fmt.Errorf("%w (tried %d times)", err, try)
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil, fmt.Errorf("%w (tried %d times)", err, try)
|
||||
case <-time.After(serverExportRetry):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// getMinecraftServers reads every MinecraftServer through the host's k3s
|
||||
// kubectl and strips what the API server owns, so the result can be fed back
|
||||
// with `kubectl apply -f` on a rebuilt cluster.
|
||||
func exportMinecraftServers(ctx context.Context) ([]byte, error) {
|
||||
func getMinecraftServers(ctx context.Context) ([]byte, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
defer cancel()
|
||||
// Output, not the CombinedOutput kubectlOutput uses: a deprecation warning
|
||||
|
||||
+389
-7
@@ -1,15 +1,20 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"flag"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/store"
|
||||
)
|
||||
|
||||
@@ -25,13 +30,63 @@ func TestDBUsage(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestDBRestoreNeedsYes(t *testing.T) {
|
||||
// A bundle that does not exist fails verification (1) before -yes matters;
|
||||
// the -yes gate itself is exercised against a real bundle in internal/dbbackup
|
||||
// and on the VM. Here: the refusal path never reaches the config or database.
|
||||
func TestDBVerifySaysWhatTheBundleHolds(t *testing.T) {
|
||||
dir := newPodRig(t)
|
||||
cfg := podConfig(t, dir)
|
||||
bundles := filepath.Join(dir, "bundles")
|
||||
var out, errBuf bytes.Buffer
|
||||
if code := run([]string{"db", "restore", "-dir", t.TempDir(), "missing.tar"}, &out, &errBuf); code != 1 {
|
||||
t.Fatalf("exit %d, stderr %q", code, errBuf.String())
|
||||
if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
|
||||
t.Fatalf("backup: exit %d: %s", code, errBuf.String())
|
||||
}
|
||||
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
|
||||
out.Reset()
|
||||
if code := run([]string{"db", "verify", "-dir", bundles, filepath.Base(bundle)}, &out, &errBuf); code != 0 {
|
||||
t.Fatalf("verify: exit %d: %s", code, errBuf.String())
|
||||
}
|
||||
for _, want := range []string{filepath.Base(bundle) + ": ok", "schema 3", "holds 4 accounts, 2 servers", "pg_dump (PostgreSQL) 18.6"} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("verify output lacks %q:\n%s", want, out.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBRestoreNeedsYes: without -yes a restore describes the bundle and stops
|
||||
// before anything reaches the database, even with -force and
|
||||
// -no-safety-backup, which would otherwise let the replay run at once.
|
||||
func TestDBRestoreNeedsYes(t *testing.T) {
|
||||
dir := newPodRig(t)
|
||||
cfg := podConfig(t, dir)
|
||||
bundles := filepath.Join(dir, "bundles")
|
||||
var out, errBuf bytes.Buffer
|
||||
if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
|
||||
t.Fatalf("backup: exit %d: %s", code, errBuf.String())
|
||||
}
|
||||
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
|
||||
podRuns(t, dir)
|
||||
|
||||
out.Reset()
|
||||
errBuf.Reset()
|
||||
code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-force", "-no-safety-backup", filepath.Base(bundle)}, &out, &errBuf)
|
||||
if code != 2 {
|
||||
t.Fatalf("exit %d, want 2; stderr %q", code, errBuf.String())
|
||||
}
|
||||
if want := filepath.Base(bundle) + " (manual, taken "; !strings.Contains(errBuf.String(), want) || !strings.Contains(errBuf.String(), "schema 3, holding 4 accounts, 2 servers).") {
|
||||
t.Errorf("stderr %q does not describe the bundle", errBuf.String())
|
||||
}
|
||||
if !strings.Contains(errBuf.String(), "re-run with -yes") {
|
||||
t.Errorf("stderr %q does not say how to go on", errBuf.String())
|
||||
}
|
||||
if argv, err := os.ReadFile(filepath.Join(dir, "k3s.args")); err == nil {
|
||||
t.Errorf("a restore without -yes ran in the database pod:\n%s", argv)
|
||||
}
|
||||
|
||||
// A bundle that does not verify is refused before -yes is weighed.
|
||||
errBuf.Reset()
|
||||
if code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-yes", "missing.tar"}, &out, &errBuf); code != 1 {
|
||||
t.Errorf("missing bundle: exit %d, want 1; stderr %q", code, errBuf.String())
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(dir, "k3s.args")); err == nil {
|
||||
t.Error("a missing bundle reached the database pod")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -141,7 +196,7 @@ func TestPreMigrateBackupOnlyGuardsAPopulatedDatabase(t *testing.T) {
|
||||
{"up to date", map[int]struct{}{1: {}, 2: {}}, false},
|
||||
{"pending on a populated database", map[int]struct{}{1: {}}, true},
|
||||
} {
|
||||
path, err := preMigrateBackup(context.Background(), appliedDriver{done: tc.done}, ms, badURL, t.TempDir(), io.Discard)
|
||||
path, err := preMigrateBackup(context.Background(), appliedDriver{done: tc.done}, ms, config.DatabaseConfig{URL: badURL}, t.TempDir(), io.Discard)
|
||||
if attempted := err != nil; attempted != tc.attempt {
|
||||
t.Errorf("%s: attempted = %v (err %v), want %v", tc.name, attempted, err, tc.attempt)
|
||||
}
|
||||
@@ -183,3 +238,330 @@ func TestAuditExportBounds(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// podK3s stands in for `k3s kubectl exec ... --`: it logs its argv and runs the
|
||||
// command after -- from the "container" directory, which is the only place the
|
||||
// PostgreSQL tools exist, as on an installed host. `kubectl get` lists one
|
||||
// MinecraftServer, logged to k3s.get, and refuses its first N calls while
|
||||
// servers_fail holds N.
|
||||
const podK3s = `#!/bin/sh
|
||||
if [ "$1" = kubectl ] && [ "$2" = get ]; then
|
||||
printf '%s\n' "$*" >> "$FAKE_DIR/k3s.get"
|
||||
n=$(/usr/bin/wc -l < "$FAKE_DIR/k3s.get")
|
||||
if [ -f "$FAKE_DIR/servers_fail" ] && [ "$n" -le "$(/bin/cat "$FAKE_DIR/servers_fail")" ]; then
|
||||
echo "The connection to the server 127.0.0.1:6443 was refused - did you specify the right host or port?" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo '{"apiVersion":"v1","kind":"List","items":[{"apiVersion":"felis.lolicon.best/v1alpha1","kind":"MinecraftServer","metadata":{"name":"lobby","namespace":"felis-servers","uid":"u-1"},"spec":{"type":"PAPER"},"status":{"phase":"Running"}}]}'
|
||||
exit 0
|
||||
fi
|
||||
printf '%s\n' "$*" >> "$FAKE_DIR/k3s.args"
|
||||
while [ $# -gt 0 ] && [ "$1" != "--" ]; do shift; done
|
||||
shift
|
||||
tool=$1; shift
|
||||
exec /usr/bin/env -i FAKE_DIR="$FAKE_DIR" PATH=/usr/bin:/bin "$FAKE_DIR/container/$tool" "$@"
|
||||
`
|
||||
|
||||
var podTools = map[string]string{
|
||||
"pg_dump": `#!/bin/sh
|
||||
case "$1" in --version) echo "pg_dump (PostgreSQL) 18.6"; exit 0 ;; esac
|
||||
printf 'PGDMP-fake-archive'
|
||||
`,
|
||||
"pg_restore": `#!/bin/sh
|
||||
cat > /dev/null
|
||||
`,
|
||||
"psql": `#!/bin/sh
|
||||
for a in "$@"; do case "$a" in *"FROM users"*) echo "4|2"; exit 0 ;; esac; done
|
||||
for a in "$@"; do [ "$a" = "-c" ] && { echo 3; exit 0; }; done
|
||||
cat > /dev/null
|
||||
`,
|
||||
}
|
||||
|
||||
const podPassword = "pw-must-stay-on-the-host"
|
||||
|
||||
// podDB is the host config's [database] on an installed host.
|
||||
var podDB = config.DatabaseConfig{
|
||||
URL: "postgres://felis:" + podPassword + "@127.0.0.1:15432/felis?sslmode=disable",
|
||||
Deployment: "felis/felis-postgres",
|
||||
}
|
||||
|
||||
const podExecPrefix = "kubectl exec -i -n felis deploy/felis-postgres -c postgres -- "
|
||||
|
||||
// newPodRig puts the fake k3s on PATH, alone, and returns the directory its
|
||||
// k3s.args log lands in.
|
||||
func newPodRig(t *testing.T) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
bin := filepath.Join(dir, "bin")
|
||||
container := filepath.Join(dir, "container")
|
||||
for _, d := range []string{bin, container} {
|
||||
if err := os.Mkdir(d, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
writeTestFile(t, filepath.Join(bin, "k3s"), podK3s, 0o755)
|
||||
for name, body := range podTools {
|
||||
writeTestFile(t, filepath.Join(container, name), body, 0o755)
|
||||
}
|
||||
t.Setenv("PATH", bin)
|
||||
t.Setenv("FAKE_DIR", dir)
|
||||
return dir
|
||||
}
|
||||
|
||||
// podRuns returns what the fake k3s ran since the last call, failing on any
|
||||
// run outside the database container or with the password on its command
|
||||
// line (visible to every local user in ps).
|
||||
func podRuns(t *testing.T, dir string) []string {
|
||||
t.Helper()
|
||||
log := filepath.Join(dir, "k3s.args")
|
||||
argv, err := os.ReadFile(log)
|
||||
if err != nil {
|
||||
t.Fatalf("nothing ran through k3s: %v", err)
|
||||
}
|
||||
os.Remove(log)
|
||||
var runs []string
|
||||
for _, line := range strings.Split(strings.TrimSpace(string(argv)), "\n") {
|
||||
if !strings.HasPrefix(line, podExecPrefix) {
|
||||
t.Errorf("k3s ran %q, want everything under %q", line, podExecPrefix)
|
||||
}
|
||||
if strings.Contains(line, podPassword) {
|
||||
t.Errorf("the password crossed into the pod on a command line: %q", line)
|
||||
}
|
||||
runs = append(runs, strings.TrimPrefix(line, podExecPrefix))
|
||||
}
|
||||
return runs
|
||||
}
|
||||
|
||||
// podConfig writes an installed host's felis.toml, [database] pointing at the
|
||||
// pod, into dir.
|
||||
func podConfig(t *testing.T, dir string) string {
|
||||
t.Helper()
|
||||
toml := strings.Replace(installerTOML("example.com", "127.0.0.1"),
|
||||
`url = "postgres://felis:[email protected]:5432/felis?sslmode=disable"`,
|
||||
`url = "`+podDB.URL+`"
|
||||
deployment = "`+podDB.Deployment+`"`, 1)
|
||||
cfg := filepath.Join(dir, "felis.toml")
|
||||
writeTestFile(t, cfg, toml, 0o600)
|
||||
return cfg
|
||||
}
|
||||
|
||||
func ranIn(runs []string, prefix string) bool {
|
||||
for _, r := range runs {
|
||||
if strings.HasPrefix(r, prefix) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// TestDBBackupAndRestoreRunTheToolsInTheDatabasePod: on an installed host the
|
||||
// database is a k3s Deployment and no PostgreSQL client exists outside it, so
|
||||
// `felis db backup` and `restore` must reach the tools through kubectl exec,
|
||||
// over the pod's socket, and without putting the role's password on a command
|
||||
// line.
|
||||
func TestDBBackupAndRestoreRunTheToolsInTheDatabasePod(t *testing.T) {
|
||||
dir := newPodRig(t)
|
||||
cfg := podConfig(t, dir)
|
||||
|
||||
var out, errBuf bytes.Buffer
|
||||
bundles := filepath.Join(dir, "bundles")
|
||||
if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
|
||||
t.Fatalf("backup: exit %d: %s", code, errBuf.String())
|
||||
}
|
||||
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
|
||||
if _, err := dbbackupVerify(bundle); err != nil {
|
||||
t.Fatalf("the bundle does not verify: %v", err)
|
||||
}
|
||||
runs := podRuns(t, dir)
|
||||
if !ranIn(runs, "pg_dump --format=custom --no-password --dbname=host=/var/run/postgresql port=5432 dbname='felis' user='felis'") {
|
||||
t.Errorf("pg_dump did not dump over the pod's socket as felis on felis: %q", runs)
|
||||
}
|
||||
|
||||
out.Reset()
|
||||
errBuf.Reset()
|
||||
if code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-yes", "-force", "-no-safety-backup", bundle}, &out, &errBuf); code != 0 {
|
||||
t.Fatalf("restore: exit %d: %s", code, errBuf.String())
|
||||
}
|
||||
runs = podRuns(t, dir)
|
||||
if !ranIn(runs, "pg_restore --no-owner --no-privileges --file=-") || !ranIn(runs, "psql -X -q -w -v ON_ERROR_STOP=1 -d host=/var/run/postgresql") {
|
||||
t.Errorf("the replay did not run in the pod: %q", runs)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPreMigrateBackupRunsInTheDatabasePod: the snapshot in front of an upgrade
|
||||
// is the one taken most often, by bootstrap on every rerun.
|
||||
func TestPreMigrateBackupRunsInTheDatabasePod(t *testing.T) {
|
||||
dir := newPodRig(t)
|
||||
ms := []store.Migration{{Version: 1}, {Version: 2}}
|
||||
// The snapshot also bundles /etc/felis, which a test machine may lack; the
|
||||
// dump runs first either way.
|
||||
_, err := preMigrateBackup(context.Background(), appliedDriver{done: map[int]struct{}{1: {}}}, ms, podDB, filepath.Join(dir, "bundles"), io.Discard)
|
||||
if err != nil && !strings.Contains(err.Error(), "read host state") {
|
||||
t.Fatalf("snapshot: %v", err)
|
||||
}
|
||||
if runs := podRuns(t, dir); !ranIn(runs, "pg_dump --format=custom") {
|
||||
t.Errorf("pg_dump did not run in the pod: %q", runs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDBToolsNeedTheRoleAndDatabase(t *testing.T) {
|
||||
if tools, err := dbTools(config.DatabaseConfig{URL: "postgres://felis:pw@db:5432/felis"}); err != nil || len(tools.Exec) != 0 {
|
||||
t.Errorf("no deployment: tools %+v err %v, want the PATH tools", tools, err)
|
||||
}
|
||||
for _, u := range []string{"postgres://db:5432/felis", "postgres://felis:pw@db:5432/"} {
|
||||
if _, err := dbTools(config.DatabaseConfig{URL: u, Deployment: "felis/felis-postgres"}); err == nil {
|
||||
t.Errorf("%s: no error, want a refusal (the pod connection needs the role and the database)", u)
|
||||
}
|
||||
}
|
||||
tools, err := dbTools(config.DatabaseConfig{URL: `postgres://o%27brien@db/my%20db`, Deployment: "felis/felis-postgres"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(tools.Conn, `dbname='my db' user='o\'brien'`) {
|
||||
t.Errorf("Conn = %q, want the values quoted for libpq", tools.Conn)
|
||||
}
|
||||
}
|
||||
|
||||
var dbbackupVerify = dbbackup.Verify
|
||||
|
||||
func noServerExportWait(t *testing.T) {
|
||||
t.Helper()
|
||||
old := serverExportRetry
|
||||
serverExportRetry = 0
|
||||
t.Cleanup(func() { serverExportRetry = old })
|
||||
}
|
||||
|
||||
// serverGets counts the `kubectl get` calls the fake k3s answered or refused.
|
||||
func serverGets(dir string) int {
|
||||
b, _ := os.ReadFile(filepath.Join(dir, "k3s.get"))
|
||||
return strings.Count(string(b), "\n")
|
||||
}
|
||||
|
||||
// bundleServers returns the bundle's k8s/minecraftservers.json, or nil.
|
||||
func bundleServers(t *testing.T, bundle string) []byte {
|
||||
t.Helper()
|
||||
f, err := os.Open(bundle)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
tr := tar.NewReader(f)
|
||||
for {
|
||||
h, err := tr.Next()
|
||||
if err == io.EOF {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if h.Name == "k8s/minecraftservers.json" {
|
||||
data, err := io.ReadAll(tr)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return data
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBBackupWithoutServersFails: when the cluster stays away the daily
|
||||
// bundle is still written, and `felis db backup` exits 1, so the timer's run
|
||||
// shows failed, saying a restore from the bundle brings back no servers.
|
||||
func TestDBBackupWithoutServersFails(t *testing.T) {
|
||||
noServerExportWait(t)
|
||||
dir := newPodRig(t)
|
||||
cfg := podConfig(t, dir)
|
||||
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run([]string{"db", "backup", "-config", cfg, "-dir", filepath.Join(dir, "bundles"), "-state-dir", "", "-label", "daily"}, &out, &errBuf)
|
||||
if code != 1 {
|
||||
t.Fatalf("exit %d, want 1: %s", code, errBuf.String())
|
||||
}
|
||||
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
|
||||
m, err := dbbackupVerify(bundle)
|
||||
if err != nil {
|
||||
t.Fatalf("the database must still be bundled: %v", err)
|
||||
}
|
||||
if !strings.Contains(m.ServersError, "6443 was refused") || !strings.Contains(m.ServersError, "(tried 3 times)") || bundleServers(t, bundle) != nil {
|
||||
t.Errorf("manifest servers error = %q", m.ServersError)
|
||||
}
|
||||
if n := serverGets(dir); n != serverExportTries {
|
||||
t.Errorf("export tried %d times, want %d", n, serverExportTries)
|
||||
}
|
||||
if msg := errBuf.String(); !strings.Contains(msg, "a restore from "+filepath.Base(bundle)+" brings back the database but no servers") {
|
||||
t.Errorf("stderr = %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDBBackupRetriesTheServerExport: a cluster back on the last try costs the
|
||||
// bundle nothing, and what it holds is ready for kubectl apply.
|
||||
func TestDBBackupRetriesTheServerExport(t *testing.T) {
|
||||
noServerExportWait(t)
|
||||
dir := newPodRig(t)
|
||||
cfg := podConfig(t, dir)
|
||||
writeTestFile(t, filepath.Join(dir, "servers_fail"), "2", 0o600)
|
||||
var out, errBuf bytes.Buffer
|
||||
if code := run([]string{"db", "backup", "-config", cfg, "-dir", filepath.Join(dir, "bundles"), "-state-dir", ""}, &out, &errBuf); code != 0 {
|
||||
t.Fatalf("exit %d: %s", code, errBuf.String())
|
||||
}
|
||||
bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
|
||||
servers := string(bundleServers(t, bundle))
|
||||
if !strings.Contains(servers, `"name": "lobby"`) || strings.Contains(servers, "status") || strings.Contains(servers, "u-1") {
|
||||
t.Errorf("k8s/minecraftservers.json = %s", servers)
|
||||
}
|
||||
if n := serverGets(dir); n != 3 {
|
||||
t.Errorf("export tried %d times, want 3", n)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPreMigrateBackupExportsServers: the snapshot every upgrade takes, often
|
||||
// the newest bundle, carries the MinecraftServer objects too; a cluster that
|
||||
// is away does not hold back the migration, whose rollback needs the database
|
||||
// alone.
|
||||
func TestPreMigrateBackupExportsServers(t *testing.T) {
|
||||
noServerExportWait(t)
|
||||
dir := newPodRig(t)
|
||||
old := preMigrateStateDir
|
||||
preMigrateStateDir = ""
|
||||
t.Cleanup(func() { preMigrateStateDir = old })
|
||||
ms := []store.Migration{{Version: 1}, {Version: 2}}
|
||||
bundles := filepath.Join(dir, "bundles")
|
||||
pending := appliedDriver{done: map[int]struct{}{1: {}}}
|
||||
|
||||
path, err := preMigrateBackup(context.Background(), pending, ms, podDB, bundles, io.Discard)
|
||||
if err != nil {
|
||||
t.Fatalf("snapshot: %v", err)
|
||||
}
|
||||
if !strings.Contains(string(bundleServers(t, path)), `"name": "lobby"`) {
|
||||
t.Errorf("%s holds no MinecraftServer objects", path)
|
||||
}
|
||||
|
||||
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
|
||||
path, err = preMigrateBackup(context.Background(), pending, ms, podDB, bundles, io.Discard)
|
||||
if err != nil || path == "" {
|
||||
t.Fatalf("snapshot with the cluster away = %q, %v; want the bundle and no error", path, err)
|
||||
}
|
||||
if m, err := dbbackupVerify(path); err != nil || m.ServersError == "" || bundleServers(t, path) != nil {
|
||||
t.Errorf("snapshot with the cluster away: %+v, %v", m, err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestServerExportStopsWaitingWithTheContext: a backup whose time is up stops
|
||||
// waiting for the cluster between tries.
|
||||
func TestServerExportStopsWaitingWithTheContext(t *testing.T) {
|
||||
dir := newPodRig(t)
|
||||
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
|
||||
// Long enough for the first try to run to its refusal: starting the fake
|
||||
// k3s on a busy machine can take a few hundred ms. Still far below the
|
||||
// 10s retry wait, so waiting it out would fail the check below.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
|
||||
defer cancel()
|
||||
start := time.Now()
|
||||
_, err := exportMinecraftServers(ctx)
|
||||
if err == nil || !strings.Contains(err.Error(), "(tried 1 times)") || serverGets(dir) != 1 {
|
||||
t.Fatalf("err = %v after %d tries, want the first failure alone", err, serverGets(dir))
|
||||
}
|
||||
if took := time.Since(start); took > 5*time.Second {
|
||||
t.Errorf("took %s, want the context's deadline, not the %s retry wait", took, serverExportRetry)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,387 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
)
|
||||
|
||||
// systemdUnitDir is where the installer writes its units.
|
||||
const systemdUnitDir = "/etc/systemd/system"
|
||||
|
||||
// hostCommand runs a host tool (systemctl, journalctl, k3s) and returns its
|
||||
// stdout. A tool that exits non-zero still returns what it printed:
|
||||
// `systemctl is-active` prints "inactive" and exits 3.
|
||||
func hostCommand(ctx context.Context, name string, args ...string) ([]byte, error) {
|
||||
return exec.CommandContext(ctx, name, args...).Output()
|
||||
}
|
||||
|
||||
// hostServices are the long-running units a full install depends on, checked
|
||||
// when their unit file is present: k3s runs the cluster, felis-velocity is the
|
||||
// game proxy, felis-nano the single-binary host that runs without k3s.
|
||||
var hostServices = []string{"k3s.service", "felis-velocity.service", "felis-nano.service"}
|
||||
|
||||
// cmdDoctor runs every check felis watchdog runs, with the settings
|
||||
// felis-watchdog.service gives it, plus what only the host shows (systemd
|
||||
// units that failed or stopped, timers that no longer fire, alerts that reach
|
||||
// no one), and prints them grouped by area with where to look next. It mails
|
||||
// nothing, pings no heartbeat and leaves the watchdog's state alone: it is
|
||||
// safe to run at any time, as often as wanted.
|
||||
func cmdDoctor(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("doctor", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
unitDir := fs.String("systemd-dir", systemdUnitDir, "where the installer's systemd units are")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis doctor: run as root (sudo felis doctor): the checks read root-only state under /etc/felis and /var/lib/felis")
|
||||
return 1
|
||||
}
|
||||
host, _ := os.Hostname()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
return runDoctor(ctx, doctorEnv{unitDir: *unitDir, run: hostCommand, now: time.Now(), host: host}, stdout)
|
||||
}
|
||||
|
||||
// doctorEnv is what one doctor run reads the host through.
|
||||
type doctorEnv struct {
|
||||
unitDir string
|
||||
run func(ctx context.Context, name string, args ...string) ([]byte, error)
|
||||
now time.Time
|
||||
host string
|
||||
}
|
||||
|
||||
// doctorAreas are the report's headings in order, by the area findingArea
|
||||
// puts a finding under.
|
||||
var doctorAreas = []struct{ key, title string }{
|
||||
{key: "config", title: "configuration"},
|
||||
{key: "cluster", title: "Kubernetes cluster"},
|
||||
{key: "postgres", title: "PostgreSQL"},
|
||||
{key: "proxy", title: "game proxy"},
|
||||
{key: "db-backup", title: "database backups"},
|
||||
{key: "offsite", title: "off-site copy"},
|
||||
{key: "scan-db", title: "build scan database"},
|
||||
{key: "disk", title: "disk space"},
|
||||
{key: "memory", title: "memory"},
|
||||
{key: "k3s-certs", title: "k3s certificates"},
|
||||
{key: "host-address", title: "node address"},
|
||||
{key: "clock", title: "clock"},
|
||||
{key: "systemd", title: "systemd units and timers"},
|
||||
{key: "alerts", title: "alerting"},
|
||||
}
|
||||
|
||||
func runDoctor(ctx context.Context, env doctorEnv, stdout io.Writer) int {
|
||||
unit := filepath.Join(env.unitDir, "felis-watchdog.service")
|
||||
w, found, unitErr := watchdogUnitFlags(unit)
|
||||
var report watchdog.Report
|
||||
var notes []string
|
||||
fmt.Fprintf(stdout, "felis doctor on %s at %s\n", env.host, env.now.UTC().Format("2006-01-02 15:04 UTC"))
|
||||
switch {
|
||||
case unitErr != nil:
|
||||
report.Findings = append(report.Findings, watchdog.Finding{
|
||||
Key: "watchdog/unit", Severity: watchdog.Critical,
|
||||
SummaryEN: fmt.Sprintf("cannot read the watchdog's settings: %v", unitErr),
|
||||
Hint: "rerun the installer (deploy/bootstrap.sh) to rewrite felis-watchdog.service",
|
||||
})
|
||||
fmt.Fprintf(stdout, "checks run with the watchdog's defaults (config %s)\n", w.cfgPath)
|
||||
case !found:
|
||||
report.Findings = append(report.Findings, watchdog.Finding{
|
||||
Key: "watchdog/unit", Severity: watchdog.Critical,
|
||||
SummaryEN: fmt.Sprintf("%s is not installed: nothing checks this host or mails anyone when it breaks", unit),
|
||||
Hint: "rerun the installer (deploy/bootstrap.sh), which installs felis-watchdog.timer",
|
||||
})
|
||||
fmt.Fprintf(stdout, "checks run with the watchdog's defaults (config %s)\n", w.cfgPath)
|
||||
default:
|
||||
fmt.Fprintf(stdout, "checks run as %s runs them (config %s)\n", unit, w.cfgPath)
|
||||
}
|
||||
fmt.Fprintln(stdout)
|
||||
|
||||
skip := map[string]string{}
|
||||
cfg, cfgErr := config.Load(w.cfgPath)
|
||||
if cfgErr != nil {
|
||||
report.Findings = append(report.Findings, watchdog.Finding{
|
||||
Key: "config", Severity: watchdog.Critical,
|
||||
SummaryEN: cfgErr.Error(),
|
||||
Hint: "the installer writes it (deploy/bootstrap.sh); felis watchdog fails on every run until it loads",
|
||||
})
|
||||
// The host's units are read without it; whether alerts reach anyone
|
||||
// is not known without its relay.
|
||||
for _, a := range doctorAreas {
|
||||
if a.key != "config" && a.key != "systemd" {
|
||||
skip[a.key] = "the configuration did not load"
|
||||
}
|
||||
}
|
||||
} else {
|
||||
_, owners, ownersErr := watchdogProbes(ctx, w, cfg, env.now, &report)
|
||||
report.Findings = append(report.Findings, alertReachFindings(cfg, owners, ownersErr)...)
|
||||
if w.proxyAddr == "" {
|
||||
skip["proxy"] = "no -proxy-addr"
|
||||
}
|
||||
if w.backupDir == "" {
|
||||
skip["db-backup"] = "no -backup-dir"
|
||||
}
|
||||
if !cfg.Offsite.Enabled() {
|
||||
skip["offsite"] = "not configured"
|
||||
}
|
||||
if !usesMirroredScanDB(cfg) {
|
||||
skip["scan-db"] = "builds do not scan against the registry's copy"
|
||||
}
|
||||
if len(splitList(w.certDirs)) == 0 {
|
||||
skip["k3s-certs"] = "no -k3s-cert-dirs"
|
||||
}
|
||||
if w.nodeIP == "" {
|
||||
skip["host-address"] = "no -node-ip"
|
||||
}
|
||||
}
|
||||
report.Findings = append(report.Findings, unitFindings(ctx, env)...)
|
||||
|
||||
switch url, err := readHeartbeatURL(w.heartbeatFile); {
|
||||
case err != nil:
|
||||
report.Findings = append(report.Findings, watchdog.Finding{
|
||||
Key: "watchdog/heartbeat", Severity: watchdog.Warning,
|
||||
SummaryEN: fmt.Sprintf("the heartbeat URL is unusable, so no run pings it: %v", err),
|
||||
Hint: "rerun the installer with FELIS_WATCHDOG_HEARTBEAT_URL set (docs/troubleshooting.md §14)",
|
||||
})
|
||||
case url == "":
|
||||
notes = append(notes, "no heartbeat URL is set: a host that goes down entirely, or a watchdog that stops running, alerts no one. "+
|
||||
"Rerun the installer with FELIS_WATCHDOG_HEARTBEAT_URL (docs/troubleshooting.md §14)")
|
||||
}
|
||||
if until := watchdog.QuietUntil(w.quietPath); env.now.Before(until) {
|
||||
notes = append(notes, fmt.Sprintf("the watchdog mails nothing until %s (%s): the installer holds it while it restarts things on purpose, "+
|
||||
"and a marker an installer killed mid-run left behind holds it until then",
|
||||
until.UTC().Format("2006-01-02 15:04 UTC"), w.quietPath))
|
||||
}
|
||||
return printDoctorReport(stdout, report.Findings, skip, notes)
|
||||
}
|
||||
|
||||
// printDoctorReport prints each area's findings under its heading, an area
|
||||
// with none as fine or, when skip says why, as not checked, then the notes
|
||||
// and the count. It returns the exit status: 1 when anything was found.
|
||||
func printDoctorReport(stdout io.Writer, findings []watchdog.Finding, skip map[string]string, notes []string) int {
|
||||
byArea := map[string][]watchdog.Finding{}
|
||||
for _, f := range findings {
|
||||
a := findingArea(f.Key)
|
||||
byArea[a] = append(byArea[a], f)
|
||||
}
|
||||
var critical, warning int
|
||||
for _, a := range doctorAreas {
|
||||
fs := byArea[a.key]
|
||||
switch {
|
||||
case len(fs) > 0:
|
||||
case skip[a.key] != "":
|
||||
fmt.Fprintf(stdout, "- %s: not checked, %s\n", a.title, skip[a.key])
|
||||
continue
|
||||
default:
|
||||
fmt.Fprintf(stdout, "✓ %s\n", a.title)
|
||||
continue
|
||||
}
|
||||
mark := "!"
|
||||
if slices.ContainsFunc(fs, func(f watchdog.Finding) bool { return f.Severity == watchdog.Critical }) {
|
||||
mark = "✗"
|
||||
}
|
||||
fmt.Fprintf(stdout, "%s %s\n", mark, a.title)
|
||||
for _, f := range fs {
|
||||
if f.Severity == watchdog.Critical {
|
||||
critical++
|
||||
} else {
|
||||
warning++
|
||||
}
|
||||
fmt.Fprintf(stdout, " %-8s %s: %s\n", f.Severity, f.Key, f.SummaryEN)
|
||||
if f.Hint != "" {
|
||||
fmt.Fprintf(stdout, " → %s\n", f.Hint)
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, n := range notes {
|
||||
fmt.Fprintf(stdout, "\nnote: %s\n", n)
|
||||
}
|
||||
fmt.Fprintln(stdout)
|
||||
if critical+warning == 0 {
|
||||
fmt.Fprintln(stdout, "no problems found")
|
||||
return 0
|
||||
}
|
||||
fmt.Fprintf(stdout, "%d problem(s): %d critical, %d warning(s)\n", critical+warning, critical, warning)
|
||||
return 1
|
||||
}
|
||||
|
||||
// findingArea is the report heading a finding key goes under.
|
||||
func findingArea(key string) string {
|
||||
head, _, _ := strings.Cut(key, "/")
|
||||
switch head {
|
||||
case "kube-api", "deployment", "system-server", "server-failed", "job-failed", "reaper-stale", "node":
|
||||
return "cluster"
|
||||
case "db-backup", "db-backup-servers":
|
||||
return "db-backup"
|
||||
case "unit", "timer":
|
||||
return "systemd"
|
||||
case "watchdog":
|
||||
return "alerts"
|
||||
}
|
||||
return head
|
||||
}
|
||||
|
||||
// alertReachFindings is why the watchdog's alerts would reach no one, which
|
||||
// its own runs only log: no relay, or no owner with a verified address.
|
||||
func alertReachFindings(cfg *config.Config, owners []string, ownersErr error) []watchdog.Finding {
|
||||
var out []watchdog.Finding
|
||||
if cfg.SMTP.Host == "" {
|
||||
out = append(out, watchdog.Finding{
|
||||
Key: "alerts/relay", Severity: watchdog.Warning,
|
||||
SummaryEN: "no [smtp] relay is configured: the watchdog logs its alerts to the journal and mails no one",
|
||||
Hint: "sudo felis setup, step SMTP",
|
||||
})
|
||||
}
|
||||
if ownersErr == nil && len(owners) == 0 {
|
||||
out = append(out, watchdog.Finding{
|
||||
Key: "alerts/recipients", Severity: watchdog.Warning,
|
||||
SummaryEN: "no owner account has a verified email: the watchdog's alerts reach no one",
|
||||
Hint: "an owner verifies an address in the panel's account settings",
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// unitFindings reports the installer's systemd units that failed, the
|
||||
// long-running ones that are not running, and timers that no longer fire.
|
||||
func unitFindings(ctx context.Context, env doctorEnv) []watchdog.Finding {
|
||||
var out []watchdog.Finding
|
||||
seen := map[string]bool{}
|
||||
failed, err := env.run(ctx, "systemctl", "list-units", "--all", "--plain", "--no-legend", "--no-pager", "--state=failed", "felis-*", "k3s.service")
|
||||
if err != nil && len(failed) == 0 {
|
||||
return []watchdog.Finding{{
|
||||
Key: "unit/systemctl", Severity: watchdog.Warning,
|
||||
SummaryEN: fmt.Sprintf("systemctl list-units failed, so no unit was checked: %v", err),
|
||||
}}
|
||||
}
|
||||
for _, line := range strings.Split(string(failed), "\n") {
|
||||
fields := strings.Fields(line)
|
||||
if len(fields) == 0 {
|
||||
continue
|
||||
}
|
||||
name := fields[0]
|
||||
seen[name] = true
|
||||
out = append(out, watchdog.Finding{
|
||||
Key: "unit/" + name, Severity: watchdog.Critical,
|
||||
SummaryEN: name + " failed",
|
||||
Hint: fmt.Sprintf("journalctl -u %s -n 100 --no-pager; once fixed, sudo systemctl reset-failed %s (a timer's job clears on its next good run)", name, name),
|
||||
})
|
||||
}
|
||||
for _, name := range hostServices {
|
||||
if seen[name] {
|
||||
continue
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(env.unitDir, name)); err != nil {
|
||||
continue
|
||||
}
|
||||
if state := unitActiveState(ctx, env, name); state != "active" {
|
||||
out = append(out, watchdog.Finding{
|
||||
Key: "unit/" + name, Severity: watchdog.Critical,
|
||||
SummaryEN: fmt.Sprintf("%s is %s", name, state),
|
||||
Hint: fmt.Sprintf("sudo systemctl start %s; journalctl -u %s -n 100 --no-pager", name, name),
|
||||
})
|
||||
}
|
||||
}
|
||||
timers, _ := filepath.Glob(filepath.Join(env.unitDir, "felis-*.timer"))
|
||||
for _, path := range timers {
|
||||
name := filepath.Base(path)
|
||||
if state := unitActiveState(ctx, env, name); state != "active" {
|
||||
out = append(out, watchdog.Finding{
|
||||
Key: "timer/" + name, Severity: watchdog.Warning,
|
||||
SummaryEN: fmt.Sprintf("%s is %s: the job it starts no longer runs", name, state),
|
||||
Hint: fmt.Sprintf("sudo systemctl enable --now %s", name),
|
||||
})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// unitActiveState is what `systemctl is-active` says of unit.
|
||||
func unitActiveState(ctx context.Context, env doctorEnv, unit string) string {
|
||||
out, err := env.run(ctx, "systemctl", "is-active", unit)
|
||||
if state := strings.TrimSpace(string(out)); state != "" {
|
||||
return state
|
||||
}
|
||||
return fmt.Sprintf("in an unknown state (systemctl is-active: %v)", err)
|
||||
}
|
||||
|
||||
// watchdogUnitFlags reads the flags felis-watchdog.service runs felis
|
||||
// watchdog with. found is false when there is no such unit; w is then the
|
||||
// watchdog's defaults.
|
||||
func watchdogUnitFlags(path string) (w watchdogFlags, found bool, err error) {
|
||||
fs := flag.NewFlagSet("watchdog", flag.ContinueOnError)
|
||||
fs.SetOutput(io.Discard)
|
||||
w.register(fs)
|
||||
raw, err := os.ReadFile(path)
|
||||
if errors.Is(err, os.ErrNotExist) {
|
||||
return w, false, nil
|
||||
}
|
||||
if err != nil {
|
||||
return w, false, err
|
||||
}
|
||||
for _, line := range strings.Split(string(raw), "\n") {
|
||||
cmd, ok := strings.CutPrefix(strings.TrimSpace(line), "ExecStart=")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
fields := execArgs(cmd)
|
||||
i := slices.Index(fields, "watchdog")
|
||||
if i < 0 {
|
||||
continue
|
||||
}
|
||||
if err := fs.Parse(fields[i+1:]); err != nil {
|
||||
return w, true, fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
return w, true, nil
|
||||
}
|
||||
return w, true, fmt.Errorf("%s runs no `felis watchdog`", path)
|
||||
}
|
||||
|
||||
// execArgs splits an ExecStart= command line into its words. A word may be
|
||||
// quoted with " or ', as systemd allows, which is how an empty value is
|
||||
// written.
|
||||
func execArgs(s string) []string {
|
||||
var out []string
|
||||
var cur strings.Builder
|
||||
inWord := false
|
||||
var quote rune
|
||||
for _, r := range s {
|
||||
switch {
|
||||
case quote != 0:
|
||||
if r == quote {
|
||||
quote = 0
|
||||
} else {
|
||||
cur.WriteRune(r)
|
||||
}
|
||||
case r == '"' || r == '\'':
|
||||
quote, inWord = r, true
|
||||
case r == ' ' || r == '\t':
|
||||
if inWord {
|
||||
out = append(out, cur.String())
|
||||
cur.Reset()
|
||||
inWord = false
|
||||
}
|
||||
default:
|
||||
cur.WriteRune(r)
|
||||
inWord = true
|
||||
}
|
||||
}
|
||||
if inWord {
|
||||
out = append(out, cur.String())
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,423 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
)
|
||||
|
||||
// bootstrapWatchdogExecStart is the ExecStart= line deploy/bootstrap.sh writes
|
||||
// into felis-watchdog.service, with its variables filled in as an install
|
||||
// fills them.
|
||||
func bootstrapWatchdogExecStart(t *testing.T, vars map[string]string) string {
|
||||
t.Helper()
|
||||
raw, err := os.ReadFile("../../deploy/bootstrap.sh")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var line string
|
||||
for _, l := range strings.Split(string(raw), "\n") {
|
||||
if strings.HasPrefix(l, "ExecStart=${HOST_BIN} watchdog -config") {
|
||||
line = l
|
||||
}
|
||||
}
|
||||
if line == "" {
|
||||
t.Fatal("deploy/bootstrap.sh writes no `ExecStart=${HOST_BIN} watchdog -config` line")
|
||||
}
|
||||
line = strings.ReplaceAll(line, "${NODE_IP:+ -node-ip ${NODE_IP}}", " -node-ip "+vars["NODE_IP"])
|
||||
line = regexp.MustCompile(`\$\{([A-Za-z_]+)\}`).ReplaceAllStringFunc(line, func(m string) string {
|
||||
v, ok := vars[m[2:len(m)-1]]
|
||||
if !ok {
|
||||
t.Fatalf("bootstrap's watchdog ExecStart= uses %s, which this test does not fill in", m)
|
||||
}
|
||||
return v
|
||||
})
|
||||
return line
|
||||
}
|
||||
|
||||
// felis doctor reads the watchdog's settings from the unit the installer
|
||||
// writes, so it checks the paths the timer's runs check.
|
||||
func TestWatchdogUnitFlagsReadsTheInstallersUnit(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
exec := bootstrapWatchdogExecStart(t, map[string]string{
|
||||
"HOST_BIN": "/usr/local/bin/felis", "STATE_DIR": "/srv/felis-etc", "WATCHDOG_STATE": "/srv/watchdog/state.json",
|
||||
"WATCHDOG_QUIET_FILE": "/srv/quiet-until", "FELIS_DB_BACKUP_DIR": "/srv/db-backups", "FELIS_GAME_PORT": "25577",
|
||||
"disks": "/,/srv/data", "NODE_IP": "10.0.0.5",
|
||||
})
|
||||
unit := filepath.Join(dir, "felis-watchdog.service")
|
||||
writeTestFile(t, unit, "[Unit]\nDescription=Felis watchdog\n\n[Service]\nType=oneshot\n"+exec+"\nTimeoutStartSec=3min\n", 0o644)
|
||||
|
||||
w, found, err := watchdogUnitFlags(unit)
|
||||
if err != nil || !found {
|
||||
t.Fatalf("found %v, err %v", found, err)
|
||||
}
|
||||
got := []string{w.cfgPath, w.statePath, w.quietPath, w.backupDir, w.proxyAddr, w.diskPaths, w.nodeIP, w.heartbeatFile, w.controlNS}
|
||||
want := []string{"/srv/felis-etc/felis.host.toml", "/srv/watchdog/state.json", "/srv/quiet-until", "/srv/db-backups", "127.0.0.1:25577", "/,/srv/data", "10.0.0.5", defaultHeartbeatFile, "felis"}
|
||||
if strings.Join(got, "|") != strings.Join(want, "|") {
|
||||
t.Errorf("read\n %q\nwant\n %q", got, want)
|
||||
}
|
||||
|
||||
w, found, err = watchdogUnitFlags(filepath.Join(dir, "missing.service"))
|
||||
if err != nil || found || w.cfgPath != "/etc/felis/felis.toml" || w.backupDir != "/var/lib/felis/db-backups" {
|
||||
t.Errorf("no unit: found %v, err %v, config %q, backups %q; want the watchdog's defaults", found, err, w.cfgPath, w.backupDir)
|
||||
}
|
||||
|
||||
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis version\n", 0o644)
|
||||
if _, found, err := watchdogUnitFlags(unit); !found || err == nil || !strings.Contains(err.Error(), "runs no `felis watchdog`") {
|
||||
t.Errorf("a unit that runs something else: found %v, err %v", found, err)
|
||||
}
|
||||
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis watchdog -no-such-flag x\n", 0o644)
|
||||
if _, _, err := watchdogUnitFlags(unit); err == nil {
|
||||
t.Error("a flag this binary does not know was accepted")
|
||||
}
|
||||
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis watchdog -backup-dir \"\"\t-proxy-addr '127.0.0.1:1' -disk-paths \"/a b\"\n", 0o644)
|
||||
if w, _, err := watchdogUnitFlags(unit); err != nil || w.backupDir != "" || w.proxyAddr != "127.0.0.1:1" || w.diskPaths != "/a b" {
|
||||
t.Errorf("quoted words: backups %q, proxy %q, disks %q, err %v", w.backupDir, w.proxyAddr, w.diskPaths, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFindingArea(t *testing.T) {
|
||||
for key, want := range map[string]string{
|
||||
"kube-api": "cluster",
|
||||
"deployment/felis-api": "cluster",
|
||||
"system-server/lobby": "cluster",
|
||||
"server-failed/survival": "cluster",
|
||||
"job-failed/reaper-123": "cluster",
|
||||
"reaper-stale": "cluster",
|
||||
"node/felis-1/NotReady": "cluster",
|
||||
"postgres": "postgres",
|
||||
"proxy": "proxy",
|
||||
"db-backup": "db-backup",
|
||||
"db-backup-servers": "db-backup",
|
||||
"offsite": "offsite",
|
||||
"scan-db": "scan-db",
|
||||
"disk//var/lib/felis": "disk",
|
||||
"memory": "memory",
|
||||
"k3s-certs": "k3s-certs",
|
||||
"host-address": "host-address",
|
||||
"clock": "clock",
|
||||
"config": "config",
|
||||
"unit/felis-offsite.service": "systemd",
|
||||
"timer/felis-db-backup.timer": "systemd",
|
||||
"watchdog/unit": "alerts",
|
||||
"watchdog/heartbeat": "alerts",
|
||||
"alerts/relay": "alerts",
|
||||
"alerts/recipients": "alerts",
|
||||
"something-a-later-release-reported": "something-a-later-release-reported",
|
||||
} {
|
||||
if got := findingArea(key); got != want {
|
||||
t.Errorf("findingArea(%q) = %q, want %q", key, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAlertReachFindings(t *testing.T) {
|
||||
relay := &config.Config{SMTP: config.SMTPConfig{Host: "smtp.example.com"}}
|
||||
keys := func(fs []watchdog.Finding) string {
|
||||
var k []string
|
||||
for _, f := range fs {
|
||||
k = append(k, f.Key)
|
||||
}
|
||||
return strings.Join(k, ",")
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
cfg *config.Config
|
||||
owners []string
|
||||
ownersErr error
|
||||
want string
|
||||
}{
|
||||
{"a relay and an owner", relay, []string{"[email protected]"}, nil, ""},
|
||||
{"no relay", &config.Config{}, []string{"[email protected]"}, nil, "alerts/relay"},
|
||||
{"no owner with an address", relay, nil, nil, "alerts/recipients"},
|
||||
{"PostgreSQL down: its own finding says so", relay, nil, errors.New("refused"), ""},
|
||||
{"neither", &config.Config{}, nil, nil, "alerts/relay,alerts/recipients"},
|
||||
} {
|
||||
if got := keys(alertReachFindings(tc.cfg, tc.owners, tc.ownersErr)); got != tc.want {
|
||||
t.Errorf("%s: %q, want %q", tc.what, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// fakeSystemctl answers list-units with failed and is-active from states;
|
||||
// a unit missing from states is "inactive", as systemctl says, and one whose
|
||||
// state is "" gets no answer.
|
||||
func fakeSystemctl(failed string, states map[string]string) func(ctx context.Context, name string, args ...string) ([]byte, error) {
|
||||
return func(_ context.Context, name string, args ...string) ([]byte, error) {
|
||||
if name != "systemctl" || len(args) == 0 {
|
||||
return nil, errors.New("unexpected command " + name)
|
||||
}
|
||||
switch args[0] {
|
||||
case "list-units":
|
||||
return []byte(failed), nil
|
||||
case "is-active":
|
||||
if s, ok := states[args[1]]; ok && s == "" {
|
||||
return nil, errors.New("signal: killed")
|
||||
} else if ok {
|
||||
return []byte(s + "\n"), nil
|
||||
}
|
||||
return []byte("inactive\n"), errors.New("exit status 3")
|
||||
}
|
||||
return nil, errors.New("unexpected systemctl " + args[0])
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnitFindings(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
for _, f := range []string{"k3s.service", "felis-velocity.service", "felis-db-backup.timer", "felis-offsite.timer", "felis-offsite.service"} {
|
||||
writeTestFile(t, filepath.Join(dir, f), "[Unit]\n", 0o644)
|
||||
}
|
||||
env := doctorEnv{unitDir: dir, run: fakeSystemctl(
|
||||
"felis-offsite.service loaded failed failed Felis off-site copy\nfelis-velocity.service loaded failed failed Velocity\n",
|
||||
map[string]string{"k3s.service": "active", "felis-db-backup.timer": "active", "felis-offsite.timer": "inactive"},
|
||||
)}
|
||||
var got []string
|
||||
for _, f := range unitFindings(context.Background(), env) {
|
||||
got = append(got, string(f.Severity)+" "+f.Key+": "+f.SummaryEN)
|
||||
}
|
||||
want := []string{
|
||||
"critical unit/felis-offsite.service: felis-offsite.service failed",
|
||||
"critical unit/felis-velocity.service: felis-velocity.service failed",
|
||||
"warning timer/felis-offsite.timer: felis-offsite.timer is inactive: the job it starts no longer runs",
|
||||
}
|
||||
if strings.Join(got, "\n") != strings.Join(want, "\n") {
|
||||
t.Errorf("findings:\n%s\nwant (felis-nano.service has no unit file here, and a failed unit is reported once):\n%s", strings.Join(got, "\n"), strings.Join(want, "\n"))
|
||||
}
|
||||
|
||||
env.run = fakeSystemctl("", map[string]string{"k3s.service": "activating", "felis-velocity.service": "active", "felis-db-backup.timer": "active", "felis-offsite.timer": "active"})
|
||||
got = nil
|
||||
for _, f := range unitFindings(context.Background(), env) {
|
||||
got = append(got, f.Key+": "+f.SummaryEN)
|
||||
}
|
||||
if strings.Join(got, "\n") != "unit/k3s.service: k3s.service is activating" {
|
||||
t.Errorf("findings %q, want only k3s.service, which is not active", got)
|
||||
}
|
||||
|
||||
env.run = fakeSystemctl("", map[string]string{"k3s.service": "active", "felis-velocity.service": "active", "felis-db-backup.timer": "", "felis-offsite.timer": "active"})
|
||||
got = nil
|
||||
for _, f := range unitFindings(context.Background(), env) {
|
||||
got = append(got, f.Key+": "+f.SummaryEN)
|
||||
}
|
||||
if want := "timer/felis-db-backup.timer: felis-db-backup.timer is in an unknown state (systemctl is-active: signal: killed): the job it starts no longer runs"; strings.Join(got, "\n") != want {
|
||||
t.Errorf("findings %q, want %q", got, want)
|
||||
}
|
||||
|
||||
env.run = func(context.Context, string, ...string) ([]byte, error) { return nil, errors.New("no systemctl") }
|
||||
if fs := unitFindings(context.Background(), env); len(fs) != 1 || fs[0].Key != "unit/systemctl" || fs[0].Severity != watchdog.Warning {
|
||||
t.Errorf("without systemctl: %+v, want the one unit/systemctl warning", fs)
|
||||
}
|
||||
}
|
||||
|
||||
// quoteArgs writes args as an ExecStart= line does, each word in quotes so an
|
||||
// empty one survives.
|
||||
func quoteArgs(args []string) string {
|
||||
q := make([]string, len(args))
|
||||
for i, a := range args {
|
||||
q[i] = `"` + a + `"`
|
||||
}
|
||||
return strings.Join(q, " ")
|
||||
}
|
||||
|
||||
// doctorHost is a host with the installer's watchdog unit, whose API server
|
||||
// and PostgreSQL are down.
|
||||
func doctorHost(t *testing.T, cfg string, extraFlags string) (env doctorEnv, h *watchdogHost) {
|
||||
t.Helper()
|
||||
h = newWatchdogHost(t, cfg, nil)
|
||||
unitDir := t.TempDir()
|
||||
writeTestFile(t, filepath.Join(unitDir, "felis-watchdog.service"),
|
||||
"[Service]\nType=oneshot\nExecStart=/usr/local/bin/felis watchdog "+quoteArgs(h.args)+
|
||||
// Off this machine's disk, whose free space is not the test's.
|
||||
` -disk-paths "/nonexistent-felis-doctor-test"`+extraFlags+"\n", 0o644)
|
||||
writeTestFile(t, filepath.Join(unitDir, "felis-velocity.service"), "[Unit]\n", 0o644)
|
||||
writeTestFile(t, filepath.Join(unitDir, "felis-offsite.timer"), "[Unit]\n", 0o644)
|
||||
return doctorEnv{
|
||||
unitDir: unitDir, now: time.Now(), host: "felis-test",
|
||||
run: fakeSystemctl("felis-db-backup.service loaded failed failed Felis database backup\n",
|
||||
map[string]string{"felis-velocity.service": "active", "felis-offsite.timer": "active"}),
|
||||
}, h
|
||||
}
|
||||
|
||||
func TestDoctorReportsByArea(t *testing.T) {
|
||||
env, _ := doctorHost(t, testWatchdogConfig, "")
|
||||
var out bytes.Buffer
|
||||
code := runDoctor(context.Background(), env, &out)
|
||||
got := out.String()
|
||||
if code != 1 {
|
||||
t.Errorf("exit %d, want 1 with problems found", code)
|
||||
}
|
||||
for _, want := range []string{
|
||||
"felis doctor on felis-test at ",
|
||||
"checks run as " + filepath.Join(env.unitDir, "felis-watchdog.service") + " runs them (config ",
|
||||
"✓ configuration\n",
|
||||
"✗ Kubernetes cluster\n critical kube-api: ",
|
||||
"✗ PostgreSQL\n critical postgres: ",
|
||||
"- game proxy: not checked, no -proxy-addr\n",
|
||||
"- database backups: not checked, no -backup-dir\n",
|
||||
"- off-site copy: not checked, not configured\n",
|
||||
"- build scan database: not checked, builds do not scan against the registry's copy\n",
|
||||
"- k3s certificates: not checked, no -k3s-cert-dirs\n",
|
||||
"- node address: not checked, no -node-ip\n",
|
||||
"✗ systemd units and timers\n critical unit/felis-db-backup.service: felis-db-backup.service failed\n" +
|
||||
" → journalctl -u felis-db-backup.service -n 100 --no-pager; once fixed, sudo systemctl reset-failed felis-db-backup.service",
|
||||
"! alerting\n warning alerts/relay: no [smtp] relay is configured",
|
||||
"\n4 problem(s): 3 critical, 1 warning(s)\n",
|
||||
} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("report lacks %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
if strings.Contains(got, "alerts/recipients") {
|
||||
t.Errorf("reported no recipients although PostgreSQL, which names them, is down:\n%s", got)
|
||||
}
|
||||
if strings.Contains(got, "note: no heartbeat URL") {
|
||||
t.Errorf("the host has a heartbeat URL:\n%s", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A doctor run is only a look: whatever is due to be mailed stays due, no
|
||||
// heartbeat is pinged, and the watchdog's state is left as it was.
|
||||
func TestDoctorMailsPingsAndSavesNothing(t *testing.T) {
|
||||
cfg := testWatchdogConfig + "[smtp]\nhost = \"smtp.example.com\"\nport = 587\nfrom = \"[email protected]\"\n"
|
||||
env, h := doctorHost(t, cfg, "")
|
||||
if err := watchdog.SaveState(h.statePath, duePostgres("cached-pw")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
before, err := os.ReadFile(h.statePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
watchdogSender = h.rec.sender
|
||||
defer func() { watchdogSender = smtpSender }()
|
||||
var out bytes.Buffer
|
||||
runDoctor(context.Background(), env, &out)
|
||||
if !strings.Contains(out.String(), "critical postgres: ") {
|
||||
t.Fatalf("the due PostgreSQL alert was not seen:\n%s", out.String())
|
||||
}
|
||||
if len(h.rec.sent) != 0 || len(h.rec.relays) != 0 {
|
||||
t.Errorf("mailed %v", h.rec.sent)
|
||||
}
|
||||
if n := len(h.pings.pings); n != 0 {
|
||||
t.Errorf("pinged the heartbeat %d times", n)
|
||||
}
|
||||
after, err := os.ReadFile(h.statePath)
|
||||
if err != nil || !bytes.Equal(before, after) {
|
||||
t.Errorf("the watchdog state changed (err %v)", err)
|
||||
}
|
||||
if _, err := os.Stat(h.fallbackPath); !errors.Is(err, os.ErrNotExist) {
|
||||
t.Errorf("a fallback state was written: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDoctorNotes(t *testing.T) {
|
||||
env, h := doctorHost(t, testWatchdogConfig, "")
|
||||
if err := os.Remove(filepath.Join(h.dir, "watchdog-heartbeat-url")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
until := env.now.Add(10 * time.Minute).Unix()
|
||||
writeTestFile(t, filepath.Join(h.dir, "quiet"), strconv.FormatInt(until, 10)+"\n", 0o644)
|
||||
var out bytes.Buffer
|
||||
runDoctor(context.Background(), env, &out)
|
||||
for _, want := range []string{
|
||||
"\nnote: no heartbeat URL is set: ",
|
||||
"\nnote: the watchdog mails nothing until " + time.Unix(until, 0).UTC().Format("2006-01-02 15:04 UTC") + " (" + filepath.Join(h.dir, "quiet") + ")",
|
||||
} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("report lacks %q:\n%s", want, out.String())
|
||||
}
|
||||
}
|
||||
|
||||
writeTestFile(t, filepath.Join(h.dir, "watchdog-heartbeat-url"), "not a url\n", 0o600)
|
||||
out.Reset()
|
||||
runDoctor(context.Background(), env, &out)
|
||||
if !strings.Contains(out.String(), "warning watchdog/heartbeat: the heartbeat URL is unusable") {
|
||||
t.Errorf("an unusable heartbeat URL is not reported:\n%s", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
// Without the watchdog's unit the doctor still checks, with the watchdog's
|
||||
// defaults, and says the host has no watchdog; a configuration that does not
|
||||
// load leaves the checks that need it unchecked and the host's own checks on.
|
||||
func TestDoctorWithoutUnitOrConfig(t *testing.T) {
|
||||
env, _ := doctorHost(t, testWatchdogConfig, "")
|
||||
if err := os.Remove(filepath.Join(env.unitDir, "felis-watchdog.service")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
writeTestFile(t, filepath.Join(env.unitDir, "felis-watchdog.service"), "[Service]\nExecStart=/usr/local/bin/felis watchdog -config "+filepath.Join(t.TempDir(), "absent.toml")+"\n", 0o644)
|
||||
env.run = fakeSystemctl("", map[string]string{"felis-velocity.service": "active", "felis-offsite.timer": "active"})
|
||||
var out bytes.Buffer
|
||||
if code := runDoctor(context.Background(), env, &out); code != 1 {
|
||||
t.Errorf("exit %d, want 1", code)
|
||||
}
|
||||
for _, want := range []string{
|
||||
"✗ configuration\n critical config: ",
|
||||
"- Kubernetes cluster: not checked, the configuration did not load\n",
|
||||
"- PostgreSQL: not checked, the configuration did not load\n",
|
||||
"- disk space: not checked, the configuration did not load\n",
|
||||
"✓ systemd units and timers\n",
|
||||
"- alerting: not checked, the configuration did not load\n",
|
||||
} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("report lacks %q:\n%s", want, out.String())
|
||||
}
|
||||
}
|
||||
|
||||
if err := os.Remove(filepath.Join(env.unitDir, "felis-watchdog.service")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
out.Reset()
|
||||
runDoctor(context.Background(), env, &out)
|
||||
if !strings.Contains(out.String(), "critical watchdog/unit: "+filepath.Join(env.unitDir, "felis-watchdog.service")+" is not installed") ||
|
||||
!strings.Contains(out.String(), "checks run with the watchdog's defaults (config /etc/felis/felis.toml)") {
|
||||
t.Errorf("a host without the watchdog's unit:\n%s", out.String())
|
||||
}
|
||||
|
||||
unit := filepath.Join(env.unitDir, "felis-watchdog.service")
|
||||
writeTestFile(t, unit, "[Service]\nExecStart=/usr/local/bin/felis watchdog -no-such-flag x\n", 0o644)
|
||||
out.Reset()
|
||||
runDoctor(context.Background(), env, &out)
|
||||
if !strings.Contains(out.String(), "critical watchdog/unit: cannot read the watchdog's settings: "+unit+": flag provided but not defined: -no-such-flag\n") ||
|
||||
!strings.Contains(out.String(), "checks run with the watchdog's defaults (config /etc/felis/felis.toml)") {
|
||||
t.Errorf("a watchdog unit this binary cannot read:\n%s", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrintDoctorReport(t *testing.T) {
|
||||
var out bytes.Buffer
|
||||
if code := printDoctorReport(&out, nil, map[string]string{"proxy": "no -proxy-addr"}, nil); code != 0 {
|
||||
t.Errorf("exit %d with nothing found, want 0", code)
|
||||
}
|
||||
want := "✓ configuration\n✓ Kubernetes cluster\n✓ PostgreSQL\n- game proxy: not checked, no -proxy-addr\n✓ database backups\n✓ off-site copy\n" +
|
||||
"✓ build scan database\n✓ disk space\n✓ memory\n✓ k3s certificates\n✓ node address\n✓ clock\n✓ systemd units and timers\n✓ alerting\n" +
|
||||
"\nno problems found\n"
|
||||
if out.String() != want {
|
||||
t.Errorf("report:\n%s\nwant:\n%s", out.String(), want)
|
||||
}
|
||||
|
||||
out.Reset()
|
||||
code := printDoctorReport(&out, []watchdog.Finding{
|
||||
{Key: "unit/systemctl", Severity: watchdog.Warning, SummaryEN: "systemctl list-units failed"},
|
||||
{Key: "postgres", Severity: watchdog.Critical, SummaryEN: "PostgreSQL is down", Hint: "kubectl -n felis get pods"},
|
||||
}, map[string]string{"postgres": "a skip loses to what was found"}, []string{"a note"})
|
||||
if code != 1 {
|
||||
t.Errorf("exit %d with problems, want 1", code)
|
||||
}
|
||||
for _, want := range []string{
|
||||
"✗ PostgreSQL\n critical postgres: PostgreSQL is down\n → kubectl -n felis get pods\n✓ game proxy\n",
|
||||
"! systemd units and timers\n warning unit/systemctl: systemctl list-units failed\n✓ alerting\n",
|
||||
"✓ alerting\n\nnote: a note\n\n2 problem(s): 1 critical, 1 warning(s)\n",
|
||||
} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("report lacks %q:\n%s", want, out.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
+1359
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,892 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"crypto/rsa"
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"crypto/x509/pkix"
|
||||
"encoding/pem"
|
||||
"errors"
|
||||
"math/big"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
|
||||
)
|
||||
|
||||
// installerTOML is felis.toml as deploy/bootstrap.sh write_felis_toml renders it,
|
||||
// comments included: the domain move has to leave all of it but three values alone.
|
||||
func installerTOML(root, dbHost string) string {
|
||||
return `# Generated by deploy/bootstrap.sh; rerun the installer to regenerate. Hand edits are
|
||||
# overwritten, except [smtp], [[auth_source]], [offsite], and the operator-owned
|
||||
# [registry] / [archive] overrides, which carry forward.
|
||||
[server]
|
||||
listen = "0.0.0.0:8080"
|
||||
root_domain = "` + root + `"
|
||||
|
||||
[database]
|
||||
url = "postgres://felis:pw@` + dbHost + `:5432/felis?sslmode=disable"
|
||||
|
||||
[k8s]
|
||||
namespace = "minecraft"
|
||||
egress_mode = "nodeport"
|
||||
|
||||
[velocity]
|
||||
# The two always-on system servers that felis setup provisions.
|
||||
login_image = "felis/limbo:1"
|
||||
lobby_image = "felis/lobby:1"
|
||||
game_port = 25565
|
||||
|
||||
[registry]
|
||||
url = "registry.felis.svc:5000"
|
||||
build_namespace = "felis-build"
|
||||
|
||||
[archive]
|
||||
store = "tarLocal"
|
||||
local_path = "/var/lib/felis/archives"
|
||||
|
||||
[auth]
|
||||
admin_hostname = "op.console.` + root + `"
|
||||
panel_hostname = "console.` + root + `"
|
||||
access_jwt_aud = "aud123"
|
||||
|
||||
# Third-party Yggdrasil sources federated by the hasJoined multiplexer.
|
||||
[[auth_source]]
|
||||
tag = "littleskin"
|
||||
prefix = "LS"
|
||||
url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined"
|
||||
`
|
||||
}
|
||||
|
||||
const linkPropsBody = `# Generated by deploy/bootstrap.sh — do not edit by hand; rerun the installer.
|
||||
api-base-url=http://10.43.0.9:8081
|
||||
service-token=TOKEN-NOT-TO-TOUCH
|
||||
root-domain=old.example
|
||||
panel-hostname=console.old.example
|
||||
admin-hostname=op.console.old.example
|
||||
login-server=login
|
||||
lobby-server=lobby
|
||||
`
|
||||
|
||||
var oldNames = domainNames{root: "old.example", panel: "console.old.example", admin: "op.console.old.example"}
|
||||
var newNames = domainNames{root: "new.example", panel: "console.new.example", admin: "op.console.new.example"}
|
||||
|
||||
// domainRig models the host: the files, the cluster, and a felis-api, proxy and
|
||||
// operator that pick up config the way the real ones do — the api serves what it
|
||||
// read at its last restart, the proxy runs since its last restart, and the login
|
||||
// pod carries the env its MinecraftServer had when it was last rolled.
|
||||
type domainRig struct {
|
||||
h domainHost
|
||||
cl client.Client
|
||||
out *bytes.Buffer
|
||||
dir string
|
||||
events []string
|
||||
|
||||
served domainNames
|
||||
servedCert *x509.Certificate
|
||||
proxySince time.Time
|
||||
proxyLoaded bool
|
||||
unresolved map[string]bool
|
||||
// The fake operator: the CR env the login pod was last rolled to, and how
|
||||
// many looks at the pod since the CR moved on.
|
||||
rolledTo string
|
||||
pending int
|
||||
}
|
||||
|
||||
func (rig *domainRig) path(name string) string { return filepath.Join(rig.dir, name) }
|
||||
|
||||
func newDomainRig(t *testing.T) *domainRig {
|
||||
t.Helper()
|
||||
rig := &domainRig{out: &bytes.Buffer{}, dir: t.TempDir(), proxyLoaded: true, unresolved: map[string]bool{}}
|
||||
writeTestFile(t, rig.path("felis.host.toml"), installerTOML("old.example", "127.0.0.1"), 0o600)
|
||||
writeTestFile(t, rig.path("felis.pod.toml"), installerTOML("old.example", "10.211.55.6"), 0o600)
|
||||
if err := os.Symlink(rig.path("felis.host.toml"), rig.path("felis.toml")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
certPEM, keyPEM, err := issuePanelCert(oldNames, []net.IP{net.ParseIP("10.211.55.6")}, time.Now())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
|
||||
writeTestFile(t, rig.path("panel-tls.key"), string(keyPEM), 0o600)
|
||||
writeTestFile(t, rig.path("felis-link.properties"), linkPropsBody, 0o640)
|
||||
// The proxy started before its config was last written, which is how it
|
||||
// stands after an install.
|
||||
rig.proxySince = time.Now().Add(-time.Hour)
|
||||
|
||||
pod := []byte(installerTOML("old.example", "10.211.55.6"))
|
||||
login, err := loginSystemServer("felis/limbo:1", "minecraft", platform.InternalAPIBaseURL("felis"), oldNames.root, oldNames.panel)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
lobby, err := lobbySystemServer("felis/lobby:1", "minecraft")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
loginPod := &corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: naming.SystemLoginServer + "-0"},
|
||||
Spec: corev1.PodSpec{Containers: []corev1.Container{{Name: "minecraft", Env: podEnv(login.Spec.Env)}}},
|
||||
Status: corev1.PodStatus{Conditions: []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}}},
|
||||
}
|
||||
rig.rolledTo = envKey(login.Spec.Env)
|
||||
rig.cl = fake.NewClientBuilder().WithScheme(haltScheme(t)).WithInterceptorFuncs(interceptor.Funcs{
|
||||
Get: func(ctx context.Context, c client.WithWatch, key client.ObjectKey, obj client.Object, opts ...client.GetOption) error {
|
||||
if key.Name == naming.SystemLoginServer+"-0" {
|
||||
rig.operatorTick(t, c)
|
||||
}
|
||||
return c.Get(ctx, key, obj, opts...)
|
||||
},
|
||||
}).WithObjects(
|
||||
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.ConfigSecretName},
|
||||
Data: map[string][]byte{platform.ConfigSecretKey: pod}},
|
||||
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: platform.ConfigSecretName},
|
||||
Data: map[string][]byte{platform.ConfigSecretKey: pod}},
|
||||
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.APITLSSecretName},
|
||||
Type: corev1.SecretTypeTLS, Data: map[string][]byte{corev1.TLSCertKey: certPEM, corev1.TLSPrivateKeyKey: keyPEM}},
|
||||
login, lobby, loginPod,
|
||||
).Build()
|
||||
rig.served = oldNames
|
||||
rig.servedCert, _ = x509.ParseCertificate(mustCertDER(certPEM))
|
||||
|
||||
rig.h = domainHost{
|
||||
paths: domainPaths{
|
||||
hostTOML: rig.path("felis.host.toml"), podTOML: rig.path("felis.pod.toml"), defaultTOML: rig.path("felis.toml"),
|
||||
cert: rig.path("panel-tls.crt"), key: rig.path("panel-tls.key"),
|
||||
linkProps: rig.path("felis-link.properties"), tunnelConfig: rig.path("cloudflared.yml"),
|
||||
},
|
||||
cl: rig.cl,
|
||||
controlNS: "felis",
|
||||
rollAPI: func(ctx context.Context) error {
|
||||
rig.events = append(rig.events, "roll-api")
|
||||
rig.restartAPI(t)
|
||||
return nil
|
||||
},
|
||||
restartUnit: func(_ context.Context, unit string) error {
|
||||
rig.events = append(rig.events, "restart "+unit)
|
||||
rig.proxySince = time.Now().Add(time.Second)
|
||||
return nil
|
||||
},
|
||||
unitState: func(context.Context, string) (unitStatus, error) {
|
||||
return unitStatus{loaded: rig.proxyLoaded, active: rig.proxyLoaded, since: rig.proxySince}, nil
|
||||
},
|
||||
liveAPI: func(context.Context, string) (liveAPIView, error) {
|
||||
return liveAPIView{names: rig.served, cert: rig.servedCert}, nil
|
||||
},
|
||||
lookupHost: func(_ context.Context, host string) ([]string, error) {
|
||||
if rig.unresolved[host] {
|
||||
return nil, errors.New("no such host")
|
||||
}
|
||||
return []string{"10.211.55.6"}, nil
|
||||
},
|
||||
passkeys: func(context.Context) (int, int, error) { return 3, 2, nil },
|
||||
now: time.Now,
|
||||
out: rig.out,
|
||||
loginWait: 50 * time.Millisecond,
|
||||
pollEvery: time.Millisecond,
|
||||
}
|
||||
return rig
|
||||
}
|
||||
|
||||
func podEnv(env []v1alpha1.EnvVar) []corev1.EnvVar {
|
||||
out := make([]corev1.EnvVar, len(env))
|
||||
for i, e := range env {
|
||||
out[i] = corev1.EnvVar{Name: e.Name, Value: e.Value}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// restartAPI makes the fake felis-api load the config and certificate its
|
||||
// Secrets hold now.
|
||||
func (rig *domainRig) restartAPI(t *testing.T) {
|
||||
t.Helper()
|
||||
var cfg, tlsSec corev1.Secret
|
||||
ctx := context.Background()
|
||||
if err := rig.cl.Get(ctx, client.ObjectKey{Namespace: "felis", Name: platform.ConfigSecretName}, &cfg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := rig.cl.Get(ctx, client.ObjectKey{Namespace: "felis", Name: platform.APITLSSecretName}, &tlsSec); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
names, err := tomlDomainNames(cfg.Data[platform.ConfigSecretKey])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rig.served = names
|
||||
rig.servedCert, err = x509.ParseCertificate(mustCertDER(tlsSec.Data[corev1.TLSCertKey]))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// rollLoginPod is the operator restarting the login pod onto its CR's env.
|
||||
// operatorTick is the operator as the login pod is watched: once the CR's env
|
||||
// changes it takes operatorLag looks at the pod before the restarted pod
|
||||
// carries the new env, the way a real rollout lags the CR.
|
||||
func (rig *domainRig) operatorTick(t *testing.T, c client.Client) {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if envKey(ms.Spec.Env) == rig.rolledTo {
|
||||
return
|
||||
}
|
||||
if rig.pending++; rig.pending < operatorLag {
|
||||
return
|
||||
}
|
||||
var pod corev1.Pod
|
||||
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer + "-0"}, &pod); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
pod.Spec.Containers[0].Env = podEnv(ms.Spec.Env)
|
||||
if err := c.Update(ctx, &pod); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rig.rolledTo, rig.pending = envKey(ms.Spec.Env), 0
|
||||
}
|
||||
|
||||
const operatorLag = 3
|
||||
|
||||
func envKey(env []v1alpha1.EnvVar) string {
|
||||
var b strings.Builder
|
||||
for _, e := range env {
|
||||
b.WriteString(e.Name + "=" + e.Value + "\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func (rig *domainRig) read(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile(rig.path(name))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
func (rig *domainRig) secret(t *testing.T, ns, name string) map[string][]byte {
|
||||
t.Helper()
|
||||
var s corev1.Secret
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return s.Data
|
||||
}
|
||||
|
||||
func (rig *domainRig) crEnv(t *testing.T, name string) map[string]string {
|
||||
t.Helper()
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
env := map[string]string{}
|
||||
for _, e := range ms.Spec.Env {
|
||||
env[e.Name] = e.Value
|
||||
}
|
||||
return env
|
||||
}
|
||||
|
||||
// snapshot is every byte `set` may touch, for proving a refused or dry run
|
||||
// touched none of it.
|
||||
func (rig *domainRig) snapshot(t *testing.T) string {
|
||||
t.Helper()
|
||||
var b strings.Builder
|
||||
entries, _ := os.ReadDir(rig.dir)
|
||||
for _, e := range entries {
|
||||
b.WriteString(e.Name() + "\n" + rig.read(t, e.Name()) + "\n")
|
||||
}
|
||||
var lines []string
|
||||
for _, s := range []struct{ ns, name string }{{"felis", platform.ConfigSecretName}, {"minecraft", platform.ConfigSecretName}, {"felis", platform.APITLSSecretName}} {
|
||||
for k, v := range rig.secret(t, s.ns, s.name) {
|
||||
lines = append(lines, s.ns+"/"+s.name+"/"+k+"\n"+string(v))
|
||||
}
|
||||
}
|
||||
for k, v := range rig.crEnv(t, naming.SystemLoginServer) {
|
||||
lines = append(lines, "env "+k+"="+v)
|
||||
}
|
||||
sort.Strings(lines)
|
||||
b.WriteString(strings.Join(lines, "\n"))
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func TestNormalizeRootDomain(t *testing.T) {
|
||||
for in, want := range map[string]string{
|
||||
"Example.COM.": "example.com",
|
||||
" mc.example.org ": "mc.example.org",
|
||||
"10.211.55.6.nip.io": "10.211.55.6.nip.io",
|
||||
"xn--bcher-kva.example": "xn--bcher-kva.example",
|
||||
} {
|
||||
got, err := normalizeRootDomain(in)
|
||||
if err != nil || got != want {
|
||||
t.Errorf("normalizeRootDomain(%q) = %q, %v; want %q", in, got, err, want)
|
||||
}
|
||||
}
|
||||
for _, in := range []string{"", "https://example.com", "example.com:443", "example.com/x", "10.0.0.1", "::1",
|
||||
"localhost", "a_b.example", "-a.example", "a-.example", strings.Repeat("a", 64) + ".example",
|
||||
strings.Repeat("abcdefghi.", 25) + "example"} {
|
||||
if got, err := normalizeRootDomain(in); err == nil {
|
||||
t.Errorf("normalizeRootDomain(%q) = %q, want an error", in, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPlanDomainChangeMovesDefaultsAndKeepsHandSetNames(t *testing.T) {
|
||||
p := planDomainChange(oldNames, "new.example")
|
||||
if p.to != newNames || p.customPanel || p.customAdmin {
|
||||
t.Fatalf("defaults: %+v", p)
|
||||
}
|
||||
p = planDomainChange(domainNames{root: "old.example", panel: "play.corp.net", admin: "op.console.old.example"}, "new.example")
|
||||
if p.to.panel != "play.corp.net" || !p.customPanel || p.to.admin != "op.console.new.example" || p.customAdmin {
|
||||
t.Fatalf("hand-set panel: %+v", p)
|
||||
}
|
||||
p = planDomainChange(domainNames{root: "old.example", panel: "console.old.example", admin: "admin.corp.net"}, "new.example")
|
||||
if p.to.admin != "admin.corp.net" || !p.customAdmin || p.to.panel != "console.new.example" {
|
||||
t.Fatalf("hand-set admin: %+v", p)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEditTOMLStringsChangesOnlyTheDomainLines(t *testing.T) {
|
||||
orig := installerTOML("old.example", "127.0.0.1")
|
||||
out, err := editTOMLStrings([]byte(orig), domainTOMLEdits(newNames))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
a, b := strings.Split(orig, "\n"), strings.Split(string(out), "\n")
|
||||
if len(a) != len(b) {
|
||||
t.Fatalf("line count %d → %d:\n%s", len(a), len(b), out)
|
||||
}
|
||||
changed := map[string]string{}
|
||||
for i := range a {
|
||||
if a[i] != b[i] {
|
||||
changed[a[i]] = b[i]
|
||||
}
|
||||
}
|
||||
want := map[string]string{
|
||||
`root_domain = "old.example"`: `root_domain = "new.example"`,
|
||||
`admin_hostname = "op.console.old.example"`: `admin_hostname = "op.console.new.example"`,
|
||||
`panel_hostname = "console.old.example"`: `panel_hostname = "console.new.example"`,
|
||||
}
|
||||
if len(changed) != len(want) {
|
||||
t.Fatalf("changed lines %v, want %v", changed, want)
|
||||
}
|
||||
for k, v := range want {
|
||||
if changed[k] != v {
|
||||
t.Errorf("%q → %q, want %q", k, changed[k], v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEditTOMLStringsAddsMissingKeysInTheirTable(t *testing.T) {
|
||||
in := "[server]\nroot_domain = \"old.example\"\n\n[auth]\naccess_jwt_aud = \"x\"\n\n[smtp]\nhost = \"relay\"\n"
|
||||
out, err := editTOMLStrings([]byte(in), domainTOMLEdits(newNames))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := "[server]\nroot_domain = \"new.example\"\n\n[auth]\naccess_jwt_aud = \"x\"\npanel_hostname = \"console.new.example\"\nadmin_hostname = \"op.console.new.example\"\n\n[smtp]\nhost = \"relay\"\n"
|
||||
if string(out) != want {
|
||||
t.Fatalf("got:\n%s\nwant:\n%s", out, want)
|
||||
}
|
||||
|
||||
out, err = editTOMLStrings([]byte("[server]\nroot_domain = \"old.example\"\n\n[[auth_source]]\ntag = \"ls\"\n"), domainTOMLEdits(newNames))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := tomlDomainNames(out)
|
||||
if err != nil || got != newNames || !strings.Contains(string(out), "[[auth_source]]\ntag = \"ls\"\n") {
|
||||
t.Fatalf("no [auth] table: %v %+v\n%s", err, got, out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEditTOMLStringsRefusesWhatItCannotEditExactly(t *testing.T) {
|
||||
for name, in := range map[string]string{
|
||||
"multi-line value": "[server]\nroot_domain = \"\"\"\nold.example\"\"\"\n[auth]\n",
|
||||
// The key's line sits inside another value; the real key is absent.
|
||||
"key inside a string": "[server]\nmotd = \"\"\"\nroot_domain = \"old.example\"\n\"\"\"\n[auth]\n",
|
||||
"quoted header": "[server]\nroot_domain = \"old.example\"\n[\"auth\"]\npanel_hostname = \"console.old.example\"\n",
|
||||
"dotted key": "server.root_domain = \"old.example\"\n",
|
||||
"inline table": "server = { root_domain = \"old.example\" }\n",
|
||||
} {
|
||||
if out, err := editTOMLStrings([]byte(in), domainTOMLEdits(newNames)); err == nil {
|
||||
t.Errorf("%s: edited instead of refusing:\n%s", name, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// caSignedCert is an operator's certificate from their own CA; it names
|
||||
// localhost too, so only the issuer tells it apart from the installer's.
|
||||
func caSignedCert(t *testing.T, hosts ...string) (certPEM, keyPEM []byte) {
|
||||
t.Helper()
|
||||
caKey, _ := rsa.GenerateKey(rand.Reader, 2048)
|
||||
ca := &x509.Certificate{SerialNumber: big.NewInt(1), Subject: pkix.Name{CommonName: "Corp CA"}, IsCA: true,
|
||||
BasicConstraintsValid: true, KeyUsage: x509.KeyUsageCertSign, NotBefore: time.Now().Add(-time.Hour), NotAfter: time.Now().Add(time.Hour)}
|
||||
caDER, err := x509.CreateCertificate(rand.Reader, ca, ca, &caKey.PublicKey, caKey)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
caCert, _ := x509.ParseCertificate(caDER)
|
||||
key, _ := rsa.GenerateKey(rand.Reader, 2048)
|
||||
leaf := &x509.Certificate{SerialNumber: big.NewInt(2), Subject: pkix.Name{CommonName: hosts[0]},
|
||||
DNSNames: append(hosts, "localhost"), IPAddresses: []net.IP{net.IPv4(127, 0, 0, 1)},
|
||||
NotBefore: time.Now().Add(-time.Hour), NotAfter: time.Now().Add(time.Hour)}
|
||||
der, err := x509.CreateCertificate(rand.Reader, leaf, caCert, &key.PublicKey, caKey)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
pk, _ := x509.MarshalPKCS8PrivateKey(key)
|
||||
return pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: der}), pem.EncodeToMemory(&pem.Block{Type: "PRIVATE KEY", Bytes: pk})
|
||||
}
|
||||
|
||||
func TestFelisIssuedCert(t *testing.T) {
|
||||
mine, _, err := issuePanelCert(oldNames, nil, time.Now())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
c, _ := x509.ParseCertificate(mustCertDER(mine))
|
||||
if !felisIssuedCert(c) {
|
||||
t.Error("the installer's kind of certificate is not recognised as Felis-issued")
|
||||
}
|
||||
theirs, _ := caSignedCert(t, "console.old.example")
|
||||
c, _ = x509.ParseCertificate(mustCertDER(theirs))
|
||||
if felisIssuedCert(c) {
|
||||
t.Error("a CA-signed certificate is taken for Felis-issued")
|
||||
}
|
||||
|
||||
// Self-signed but without one of the installer's localhost names: someone
|
||||
// else's.
|
||||
key, _ := rsa.GenerateKey(rand.Reader, 2048)
|
||||
for name, self := range map[string]*x509.Certificate{
|
||||
"no localhost": {DNSNames: []string{"console.old.example"}, IPAddresses: []net.IP{net.IPv4(127, 0, 0, 1)}},
|
||||
"no 127.0.0.1": {DNSNames: []string{"console.old.example", "localhost"}},
|
||||
} {
|
||||
self.SerialNumber, self.Subject = big.NewInt(3), pkix.Name{CommonName: "x"}
|
||||
self.NotBefore, self.NotAfter = time.Now().Add(-time.Hour), time.Now().Add(time.Hour)
|
||||
der, _ := x509.CreateCertificate(rand.Reader, self, self, &key.PublicKey, key)
|
||||
c, _ = x509.ParseCertificate(der)
|
||||
if felisIssuedCert(c) {
|
||||
t.Errorf("%s: a self-signed certificate is taken for Felis-issued", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestIssuePanelCertIsTheInstallersShape(t *testing.T) {
|
||||
now := time.Now()
|
||||
certPEM, keyPEM, err := issuePanelCert(newNames, []net.IP{net.ParseIP("10.211.55.6"), net.IPv4(127, 0, 0, 1)}, now)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := tls.X509KeyPair(certPEM, keyPEM); err != nil {
|
||||
t.Fatalf("key does not match the certificate: %v", err)
|
||||
}
|
||||
if b, _ := pem.Decode(keyPEM); b == nil || b.Type != "PRIVATE KEY" {
|
||||
t.Fatalf("key is not PKCS#8 PEM like openssl writes")
|
||||
}
|
||||
c, _ := x509.ParseCertificate(mustCertDER(certPEM))
|
||||
if !certCovers(c, newNames.panel, newNames.admin, "localhost") || !felisIssuedCert(c) {
|
||||
t.Fatalf("names %v", c.DNSNames)
|
||||
}
|
||||
if len(c.IPAddresses) != 2 || !c.IPAddresses[1].Equal(net.ParseIP("10.211.55.6")) {
|
||||
t.Fatalf("addresses %v, want 127.0.0.1 and the node address once each", c.IPAddresses)
|
||||
}
|
||||
if c.Subject.CommonName != newNames.admin || c.NotAfter.Sub(now) < 824*24*time.Hour || c.NotAfter.Sub(now) > 826*24*time.Hour {
|
||||
t.Fatalf("CN %q, valid until %s", c.Subject.CommonName, c.NotAfter)
|
||||
}
|
||||
if len(c.ExtKeyUsage) != 1 || c.ExtKeyUsage[0] != x509.ExtKeyUsageServerAuth || c.IsCA {
|
||||
t.Fatalf("usage %v, CA %v", c.ExtKeyUsage, c.IsCA)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetMovesEverySurface(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
hostBefore := rig.read(t, "felis.host.toml")
|
||||
code, err := rig.h.set(context.Background(), "New.Example", true)
|
||||
if err != nil || code != 0 {
|
||||
t.Fatalf("set = %d, %v\n%s", code, err, rig.out)
|
||||
}
|
||||
|
||||
// The configs: the three values moved, everything else — comments, the
|
||||
// database host of each copy, the Access audience — is as it was.
|
||||
host := rig.read(t, "felis.host.toml")
|
||||
if want := strings.NewReplacer("old.example", "new.example").Replace(hostBefore); host != want {
|
||||
t.Fatalf("host toml:\n%s", host)
|
||||
}
|
||||
pod := rig.read(t, "felis.pod.toml")
|
||||
if got, _ := tomlDomainNames([]byte(pod)); got != newNames || !strings.Contains(pod, "@10.211.55.6:5432") {
|
||||
t.Fatalf("pod toml:\n%s", pod)
|
||||
}
|
||||
if link, err := os.Readlink(rig.path("felis.toml")); err != nil || link != rig.path("felis.host.toml") {
|
||||
t.Fatalf("felis.toml is no longer the link to the host copy: %q %v", link, err)
|
||||
}
|
||||
if st, _ := os.Stat(rig.path("felis.host.toml")); st.Mode().Perm() != 0o600 {
|
||||
t.Fatalf("host toml mode %v", st.Mode().Perm())
|
||||
}
|
||||
|
||||
// The certificate: reissued for the new names, the node address kept, the old
|
||||
// pair beside it.
|
||||
c, err := readCertFile(rig.path("panel-tls.crt"))
|
||||
if err != nil || !certCovers(c, newNames.panel, newNames.admin) || !felisIssuedCert(c) {
|
||||
t.Fatalf("certificate: %v %v", err, c.DNSNames)
|
||||
}
|
||||
if !c.IPAddresses[len(c.IPAddresses)-1].Equal(net.ParseIP("10.211.55.6")) {
|
||||
t.Fatalf("addresses %v", c.IPAddresses)
|
||||
}
|
||||
if _, err := tls.LoadX509KeyPair(rig.path("panel-tls.crt"), rig.path("panel-tls.key")); err != nil {
|
||||
t.Fatalf("new pair: %v", err)
|
||||
}
|
||||
if st, _ := os.Stat(rig.path("panel-tls.key")); st.Mode().Perm() != 0o600 {
|
||||
t.Fatalf("key mode %v", st.Mode().Perm())
|
||||
}
|
||||
backups, _ := filepath.Glob(rig.path("panel-tls.*.pre-domain-*"))
|
||||
if len(backups) != 2 {
|
||||
t.Fatalf("old pair kept as %v", backups)
|
||||
}
|
||||
for _, b := range backups {
|
||||
if st, _ := os.Stat(b); strings.Contains(b, ".key.") && st.Mode().Perm() != 0o600 {
|
||||
t.Fatalf("kept key %s has mode %v", b, st.Mode().Perm())
|
||||
}
|
||||
old, _ := os.ReadFile(b)
|
||||
if strings.Contains(b, ".crt.") {
|
||||
oc, _ := x509.ParseCertificate(mustCertDER(old))
|
||||
if oc == nil || !certCovers(oc, oldNames.panel) {
|
||||
t.Fatalf("kept certificate is not the old one")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The Secrets carry the files.
|
||||
for _, ns := range []string{"felis", "minecraft"} {
|
||||
if got := rig.secret(t, ns, platform.ConfigSecretName)[platform.ConfigSecretKey]; string(got) != pod {
|
||||
t.Fatalf("%s/felis-config is not felis.pod.toml", ns)
|
||||
}
|
||||
}
|
||||
tlsData := rig.secret(t, "felis", platform.APITLSSecretName)
|
||||
if string(tlsData[corev1.TLSCertKey]) != rig.read(t, "panel-tls.crt") || string(tlsData[corev1.TLSPrivateKeyKey]) != rig.read(t, "panel-tls.key") {
|
||||
t.Fatal("felis-api-tls does not hold the new pair")
|
||||
}
|
||||
|
||||
// The login gate's env moved and nothing else did.
|
||||
env := rig.crEnv(t, naming.SystemLoginServer)
|
||||
if env[envRootDomain] != newNames.root || env[envPanelHostname] != newNames.panel || env[envAPIBaseURL] != platform.InternalAPIBaseURL("felis") {
|
||||
t.Fatalf("login env %v", env)
|
||||
}
|
||||
|
||||
// The proxy's file: the three keys moved, its token and mode did not.
|
||||
props := rig.read(t, "felis-link.properties")
|
||||
if want := strings.NewReplacer("old.example", "new.example").Replace(linkPropsBody); props != want {
|
||||
t.Fatalf("felis-link.properties:\n%s", props)
|
||||
}
|
||||
if st, _ := os.Stat(rig.path("felis-link.properties")); st.Mode().Perm() != 0o640 {
|
||||
t.Fatalf("felis-link.properties mode %v", st.Mode().Perm())
|
||||
}
|
||||
|
||||
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
|
||||
t.Fatalf("events %v", rig.events)
|
||||
}
|
||||
if !strings.Contains(rig.out.String(), "Every surface is on new.example.") {
|
||||
t.Fatalf("the closing check did not pass:\n%s", rig.out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetWithoutYesChangesNothing(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
before := rig.snapshot(t)
|
||||
code, err := rig.h.set(context.Background(), "new.example", false)
|
||||
if err != nil || code != 0 {
|
||||
t.Fatalf("set = %d, %v", code, err)
|
||||
}
|
||||
if rig.snapshot(t) != before || len(rig.events) != 0 {
|
||||
t.Fatalf("a dry run changed something (events %v)", rig.events)
|
||||
}
|
||||
out := rig.out.String()
|
||||
for _, want := range []string{
|
||||
"console.old.example → console.new.example",
|
||||
"3 passkey(s) of 2 user(s) are bound to console.old.example",
|
||||
"does not cover op.console.new.example",
|
||||
"No [smtp] relay is configured",
|
||||
"sudo felis domain set -yes new.example",
|
||||
} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("plan lacks %q:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetKeepsAHandSetPanelHostname(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
for _, f := range []string{"felis.host.toml", "felis.pod.toml"} {
|
||||
writeTestFile(t, rig.path(f), strings.Replace(rig.read(t, f), `panel_hostname = "console.old.example"`, `panel_hostname = "play.corp.net"`, 1), 0o600)
|
||||
}
|
||||
code, err := rig.h.set(context.Background(), "new.example", true)
|
||||
if err != nil {
|
||||
t.Fatalf("set: %v\n%s", err, rig.out)
|
||||
}
|
||||
got, _ := tomlDomainNames([]byte(rig.read(t, "felis.host.toml")))
|
||||
if got != (domainNames{root: "new.example", panel: "play.corp.net", admin: "op.console.new.example"}) {
|
||||
t.Fatalf("names %+v", got)
|
||||
}
|
||||
c, _ := readCertFile(rig.path("panel-tls.crt"))
|
||||
if !certCovers(c, "play.corp.net", "op.console.new.example") {
|
||||
t.Fatalf("certificate names %v", c.DNSNames)
|
||||
}
|
||||
if env := rig.crEnv(t, naming.SystemLoginServer); env[envPanelHostname] != "play.corp.net" {
|
||||
t.Fatalf("login env %v", env)
|
||||
}
|
||||
if code != 0 || !strings.Contains(rig.out.String(), "play.corp.net (set by hand, kept") {
|
||||
t.Fatalf("code %d:\n%s", code, rig.out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetRefusesAnOperatorCertificateForOtherNames(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
certPEM, keyPEM := caSignedCert(t, "console.old.example", "op.console.old.example")
|
||||
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
|
||||
writeTestFile(t, rig.path("panel-tls.key"), string(keyPEM), 0o600)
|
||||
before := rig.snapshot(t)
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err == nil || !strings.Contains(err.Error(), "not issued by Felis") {
|
||||
t.Fatalf("err = %v", err)
|
||||
}
|
||||
if rig.snapshot(t) != before || len(rig.events) != 0 {
|
||||
t.Fatal("a refused move changed something")
|
||||
}
|
||||
|
||||
// The operator's certificate for the new names is kept as it is.
|
||||
certPEM, keyPEM = caSignedCert(t, "console.new.example", "op.console.new.example")
|
||||
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
|
||||
writeTestFile(t, rig.path("panel-tls.key"), string(keyPEM), 0o600)
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
|
||||
t.Fatalf("set: %v", err)
|
||||
}
|
||||
if rig.read(t, "panel-tls.crt") != string(certPEM) {
|
||||
t.Fatal("the operator's certificate was replaced")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetRefusesAConfigItCannotEdit(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
writeTestFile(t, rig.path("felis.pod.toml"), strings.Replace(rig.read(t, "felis.pod.toml"), "[auth]", "[\"auth\"]", 1), 0o600)
|
||||
before := rig.snapshot(t)
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err == nil || !strings.Contains(err.Error(), "by hand") {
|
||||
t.Fatalf("err = %v", err)
|
||||
}
|
||||
if rig.snapshot(t) != before || len(rig.events) != 0 {
|
||||
t.Fatal("a refused move changed something")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetAgainOnlyConvergesWhatIsBehind(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rig.events, rig.out = nil, &bytes.Buffer{}
|
||||
rig.h.out = rig.out
|
||||
before := rig.snapshot(t)
|
||||
code, err := rig.h.set(context.Background(), "new.example", true)
|
||||
if err != nil || code != 0 {
|
||||
t.Fatalf("second set = %d, %v\n%s", code, err, rig.out)
|
||||
}
|
||||
if len(rig.events) != 0 || rig.snapshot(t) != before {
|
||||
t.Fatalf("a converged install was touched again: %v\n%s", rig.events, rig.out)
|
||||
}
|
||||
|
||||
// A proxy that was not restarted after the move is restarted by a re-run, and
|
||||
// an api still on the old config is rolled.
|
||||
rig.proxySince = time.Now().Add(-time.Hour)
|
||||
rig.served = oldNames
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
|
||||
t.Fatalf("events %v", rig.events)
|
||||
}
|
||||
|
||||
// An api on the new names that still presents the old certificate is rolled.
|
||||
rig.events = nil
|
||||
oldCert, _, _ := issuePanelCert(oldNames, nil, time.Now())
|
||||
rig.servedCert, _ = x509.ParseCertificate(mustCertDER(oldCert))
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api" {
|
||||
t.Fatalf("events %v", rig.events)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainSetRefusesALoginServerItDoesNotOwn(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
delete(ms.Labels, v1alpha1.LabelSystemRole)
|
||||
if err := rig.cl.Update(context.Background(), &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err == nil || !strings.Contains(err.Error(), "system role") {
|
||||
t.Fatalf("err = %v", err)
|
||||
}
|
||||
if env := rig.crEnv(t, naming.SystemLoginServer); env[envRootDomain] != oldNames.root {
|
||||
t.Fatalf("a server not marked as the login gate was changed: %v", env)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainCheckNamesTheSurfaceThatIsBehind(t *testing.T) {
|
||||
cases := []struct {
|
||||
surface string
|
||||
breakIt func(t *testing.T, rig *domainRig)
|
||||
}{
|
||||
{"felis.pod.toml", func(t *testing.T, rig *domainRig) {
|
||||
writeTestFile(t, rig.path("felis.pod.toml"), installerTOML("old.example", "10.211.55.6"), 0o600)
|
||||
}},
|
||||
{"Secret minecraft/felis-config", func(t *testing.T, rig *domainRig) {
|
||||
rig.putSecret(t, "minecraft", platform.ConfigSecretName, platform.ConfigSecretKey, installerTOML("old.example", "x"))
|
||||
}},
|
||||
{"Secret felis/felis-config", func(t *testing.T, rig *domainRig) {
|
||||
rig.putSecret(t, "felis", platform.ConfigSecretName, platform.ConfigSecretKey, installerTOML("old.example", "x"))
|
||||
}},
|
||||
{"panel certificate", func(t *testing.T, rig *domainRig) {
|
||||
// The Secret follows the file, so only the certificate's names are wrong.
|
||||
certPEM, _, _ := issuePanelCert(oldNames, nil, time.Now())
|
||||
writeTestFile(t, rig.path("panel-tls.crt"), string(certPEM), 0o644)
|
||||
rig.putSecret(t, "felis", platform.APITLSSecretName, corev1.TLSCertKey, string(certPEM))
|
||||
}},
|
||||
{"Secret felis/felis-api-tls", func(t *testing.T, rig *domainRig) {
|
||||
certPEM, _, _ := issuePanelCert(newNames, nil, time.Now())
|
||||
rig.putSecret(t, "felis", platform.APITLSSecretName, corev1.TLSCertKey, string(certPEM))
|
||||
}},
|
||||
{"felis-api", func(t *testing.T, rig *domainRig) { rig.served = oldNames }},
|
||||
{"felis-api", func(t *testing.T, rig *domainRig) {
|
||||
certPEM, _, _ := issuePanelCert(oldNames, nil, time.Now())
|
||||
rig.servedCert, _ = x509.ParseCertificate(mustCertDER(certPEM))
|
||||
}},
|
||||
{"proxy", func(t *testing.T, rig *domainRig) {
|
||||
writeTestFile(t, rig.path("felis-link.properties"), linkPropsBody, 0o640)
|
||||
rig.proxySince = time.Now().Add(time.Hour)
|
||||
}},
|
||||
{"proxy", func(t *testing.T, rig *domainRig) { rig.proxySince = time.Now().Add(-time.Hour) }},
|
||||
{"login gate", func(t *testing.T, rig *domainRig) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
_ = rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms)
|
||||
for i := range ms.Spec.Env {
|
||||
if ms.Spec.Env[i].Name == envRootDomain {
|
||||
ms.Spec.Env[i].Value = "old.example"
|
||||
}
|
||||
}
|
||||
if err := rig.cl.Update(context.Background(), &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}},
|
||||
{"login gate", func(t *testing.T, rig *domainRig) { rig.setPodEnv(t, envRootDomain, "old.example") }},
|
||||
{"login gate", func(t *testing.T, rig *domainRig) { rig.setPodEnv(t, envPanelHostname, "console.old.example") }},
|
||||
{"Cloudflare tunnel", func(t *testing.T, rig *domainRig) {
|
||||
writeTestFile(t, rig.path("cloudflared.yml"), "tunnel: abc\ningress:\n- hostname: console.new.example\n service: https://127.0.0.1:30443\n- hostname: op.console.old.example\n service: https://127.0.0.1:30443\n- service: http_status:404\n", 0o644)
|
||||
}},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
rig := newDomainRig(t)
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rig.out.Reset()
|
||||
tc.breakIt(t, rig)
|
||||
if code := rig.h.check(context.Background()); code != 1 {
|
||||
t.Errorf("%s behind: check = %d\n%s", tc.surface, code, rig.out)
|
||||
continue
|
||||
}
|
||||
var failed []string
|
||||
for _, ln := range strings.Split(rig.out.String(), "\n") {
|
||||
if strings.HasPrefix(ln, " FAIL ") {
|
||||
failed = append(failed, ln)
|
||||
}
|
||||
}
|
||||
if len(failed) != 1 || !strings.Contains(failed[0], tc.surface) {
|
||||
t.Errorf("%s behind: FAIL lines %q", tc.surface, failed)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDomainCheckPassesAConvergedInstallAndWarnsOnDNS(t *testing.T) {
|
||||
rig := newDomainRig(t)
|
||||
if _, err := rig.h.set(context.Background(), "new.example", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
writeTestFile(t, rig.path("cloudflared.yml"), "tunnel: abc\ningress:\n- hostname: console.new.example\n service: https://127.0.0.1:30443\n- hostname: op.console.new.example\n service: https://127.0.0.1:30443\n- service: http_status:404\n", 0o644)
|
||||
rig.unresolved["op.console.new.example"] = true
|
||||
rig.unresolved[dnsProbeLabel+".new.example"] = true
|
||||
rig.out.Reset()
|
||||
if code := rig.h.check(context.Background()); code != 0 {
|
||||
t.Fatalf("check = %d\n%s", code, rig.out)
|
||||
}
|
||||
out := rig.out.String()
|
||||
for _, want := range []string{" ok Cloudflare tunnel: routes both names", " warn DNS: op.console.new.example, *.new.example do not resolve from this host\n"} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("check lacks %q:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
if strings.Contains(out, "TOKEN-NOT-TO-TOUCH") {
|
||||
t.Fatal("check printed the proxy's service token")
|
||||
}
|
||||
|
||||
// A zone with only the wildcard: the admin name alone is missing, and why is said.
|
||||
delete(rig.unresolved, dnsProbeLabel+".new.example")
|
||||
rig.out.Reset()
|
||||
rig.h.check(context.Background())
|
||||
if want := " warn DNS: op.console.new.example does not resolve from this host: the *.new.example wildcard does not cover op.console.new.example, which needs its own record\n"; !strings.Contains(rig.out.String(), want) {
|
||||
t.Errorf("check lacks %q:\n%s", want, rig.out)
|
||||
}
|
||||
}
|
||||
|
||||
func (rig *domainRig) setPodEnv(t *testing.T, name, value string) {
|
||||
t.Helper()
|
||||
var pod corev1.Pod
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "login-0"}, &pod); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for i, e := range pod.Spec.Containers[0].Env {
|
||||
if e.Name == name {
|
||||
pod.Spec.Containers[0].Env[i].Value = value
|
||||
}
|
||||
}
|
||||
if err := rig.cl.Update(context.Background(), &pod); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func (rig *domainRig) putSecret(t *testing.T, ns, name, key, val string) {
|
||||
t.Helper()
|
||||
var s corev1.Secret
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s.Data[key] = []byte(val)
|
||||
if err := rig.cl.Update(context.Background(), &s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseUnitShow(t *testing.T) {
|
||||
st := parseUnitShow("LoadState=loaded\nActiveState=active\nActiveEnterTimestamp=@1790000000\n")
|
||||
if !st.loaded || !st.active || !st.since.Equal(time.Unix(1790000000, 0)) {
|
||||
t.Fatalf("%+v", st)
|
||||
}
|
||||
st = parseUnitShow("LoadState=not-found\nActiveState=inactive\nActiveEnterTimestamp=\n")
|
||||
if st.loaded || st.active || !st.since.IsZero() {
|
||||
t.Fatalf("%+v", st)
|
||||
}
|
||||
}
|
||||
+22
-11
@@ -17,23 +17,28 @@ var (
|
||||
egressPollInterval = 200 * time.Millisecond
|
||||
)
|
||||
|
||||
// cmdEgressGate is the first initContainer of every build pod. The pod's
|
||||
// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s:
|
||||
// a build-labelled pod reached the internet and the Kubernetes API for its first
|
||||
// ~0.7 s), so the gate dials a destination the policy denies until it stops
|
||||
// answering, and only then lets the pod's next container, eventually the
|
||||
// untrusted Dockerfile, start.
|
||||
// cmdEgressGate is the first initContainer of every build pod and the last of
|
||||
// every game server pod. A pod's NetworkPolicy is programmed asynchronously after
|
||||
// the pod starts (live on k3s: a build-labelled pod reached the internet and the
|
||||
// Kubernetes API for its first ~0.7 s, a server-labelled one felis-api's internal
|
||||
// face on its first request), so the gate dials a destination the policy denies
|
||||
// until it stops answering, and only then lets the pod's next container, the
|
||||
// untrusted Dockerfile or server image, start.
|
||||
//
|
||||
// The default probe is the Kubernetes API Service, which the kubelet names in
|
||||
// every pod's environment and the build policy never admits. A probe that still
|
||||
// answers after --wait means the policy is not enforced at all (a CNI without
|
||||
// every pod's environment and neither policy admits. A probe that still answers
|
||||
// after --wait means the policy is not enforced at all (a CNI without
|
||||
// NetworkPolicy support, or k3s run with --disable-network-policy), and the
|
||||
// build fails closed.
|
||||
// build fails closed. A server passes --fail-open: an operator's
|
||||
// --server-egress-allow-cidr may cover the node the API Service leads to, so a
|
||||
// probe that keeps answering does not prove the fence is missing, and by then
|
||||
// the policy has had --wait to land.
|
||||
func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
|
||||
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused")
|
||||
probe := fs.String("probe", "", "host:port the pod's NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
|
||||
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the gate gives up")
|
||||
failOpen := fs.Bool("fail-open", false, "when --wait runs out, warn and let the pod go on instead of refusing it")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -56,6 +61,12 @@ func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
_ = conn.Close()
|
||||
if time.Since(start) >= *wait {
|
||||
if *failOpen {
|
||||
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s; starting anyway. Either this namespace's "+
|
||||
"NetworkPolicy is not enforced (a CNI without NetworkPolicy support, or k3s started with "+
|
||||
"--disable-network-policy), or an allowed CIDR admits the address behind it\n", *probe, *wait)
|
||||
return 0
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+
|
||||
"(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n",
|
||||
*probe, *wait)
|
||||
|
||||
@@ -75,6 +75,40 @@ func TestEgressGateRefusesAnOpenNetwork(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A server's gate waits out --wait all the same, then lets the pod start with a
|
||||
// warning in its log.
|
||||
func TestEgressGateFailOpenWaitsThenWarns(t *testing.T) {
|
||||
shrinkEgressGate(t)
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer ln.Close()
|
||||
go func() {
|
||||
for {
|
||||
c, err := ln.Accept()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
_ = c.Close()
|
||||
}
|
||||
}()
|
||||
var out, errb bytes.Buffer
|
||||
start := time.Now()
|
||||
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "150ms", "--fail-open"}, &out, &errb); code != 0 {
|
||||
t.Fatalf("exit %d, want 0: %s", code, errb.String())
|
||||
}
|
||||
if waited := time.Since(start); waited < 150*time.Millisecond {
|
||||
t.Errorf("gave up after %s, before --wait ran out", waited)
|
||||
}
|
||||
if !strings.Contains(errb.String(), ln.Addr().String()+" still answers after 150ms; starting anyway") {
|
||||
t.Errorf("stderr = %q", errb.String())
|
||||
}
|
||||
if out.Len() != 0 {
|
||||
t.Errorf("stdout = %q, want nothing: the lock was never seen", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) {
|
||||
shrinkEgressGate(t)
|
||||
t.Setenv("KUBERNETES_SERVICE_HOST", "")
|
||||
|
||||
@@ -0,0 +1,302 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"hash"
|
||||
"io"
|
||||
"io/fs"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/backup"
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
"felis.lolicon.best/internal/worldexport"
|
||||
)
|
||||
|
||||
// cmdExport is the in-Pod entrypoint the export Job runs. internal/worldexport
|
||||
// renders a Pod whose command is `/usr/local/bin/felis export`. It archives the
|
||||
// mounted world, re-streams one archive from the mounted backup store, or sends
|
||||
// one file or folder of the world, PUTs it to felis-api's internal face, and
|
||||
// exits once felis-api says the owner's browser got all of it. It is NOT a
|
||||
// user-facing command and is never invoked by hand.
|
||||
//
|
||||
// Like cmdRestore it holds no database credentials and never calls config.Load:
|
||||
// felis-api made every decision (who may download what, that the server is
|
||||
// stopped, which archive) before the Job existed. Its input is the flags below
|
||||
// plus the one-time upload token in the environment, which opens this one
|
||||
// export and nothing else.
|
||||
//
|
||||
// Whatever leaves goes through the same guards as the file editor
|
||||
// (fileedit.Guard): the proxy forwarding secret, which every server on the
|
||||
// install shares, never leaves, and server.properties leaves with its RCON
|
||||
// password redacted. A backup is stored with both, since a restore must bring
|
||||
// the world back whole, so it is filtered on the way out rather than handed
|
||||
// over as stored.
|
||||
//
|
||||
// Exit status: 0 once felis-api answers 204 (the download completed), 1 when
|
||||
// the export could not be read or handed over, a backup failed its digest
|
||||
// check, or felis-api refused it (the browser never came or left early), 2 on
|
||||
// bad flags. The last stderr line reaches the export's status and the jobs list.
|
||||
func cmdExport(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("export", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
mode := fs.String("mode", "", "what to export: world, backup or files")
|
||||
server := fs.String("server", "", "server name being exported (for logging)")
|
||||
target := fs.String("target-url", "", "felis-api URL to PUT the export to")
|
||||
ref := fs.String("ref", "", "backup only: absolute path to the archive on the backup mount")
|
||||
backupRoot := fs.String("backup-root", "/backups", "backup only: mount path of the backup PVC (the ref must resolve under it)")
|
||||
sum := fs.String("sha256", "", "backup only: the sha256 recorded when the archive was written; a mismatch fails the export before its end is sent")
|
||||
worldsRoot := fs.String("worlds-root", "/world", "world and files: mount path of the world PVC")
|
||||
path := fs.String("path", "", "files only: the file or folder to send, relative to the world root")
|
||||
dir := fs.Bool("dir", false, "files only: the path is a folder, sent as a zip")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
token := os.Getenv(worldexport.TokenEnv)
|
||||
if *target == "" || token == "" {
|
||||
fmt.Fprintf(stderr, "felis export: --target-url and %s are required\n", worldexport.TokenEnv)
|
||||
return 2
|
||||
}
|
||||
limitHeapToCgroup()
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
var err error
|
||||
switch *mode {
|
||||
case worldexport.ModeBackup:
|
||||
if *ref == "" {
|
||||
fmt.Fprintln(stderr, "felis export: --ref is required for a backup")
|
||||
return 2
|
||||
}
|
||||
err = exportBackup(ctx, *target, token, *ref, *backupRoot, *sum, stdout)
|
||||
case worldexport.ModeWorld:
|
||||
err = exportWorld(ctx, *target, token, *worldsRoot, stdout)
|
||||
case worldexport.ModeFiles:
|
||||
if *path == "" {
|
||||
fmt.Fprintln(stderr, "felis export: --path is required for files")
|
||||
return 2
|
||||
}
|
||||
err = exportFiles(ctx, *target, token, *worldsRoot, *path, *dir, stdout)
|
||||
default:
|
||||
fmt.Fprintf(stderr, "felis export: --mode must be %s, %s or %s\n", worldexport.ModeWorld, worldexport.ModeBackup, worldexport.ModeFiles)
|
||||
return 2
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis export: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis export: server=%s mode=%s downloaded\n", *server, *mode)
|
||||
return 0
|
||||
}
|
||||
|
||||
// archiveType is the content type of a world or backup export.
|
||||
const archiveType = "application/gzip"
|
||||
|
||||
// errBackupDigest fails a backup export whose stored archive no longer hashes
|
||||
// to what was recorded when it was written.
|
||||
var errBackupDigest = errors.New("the backup archive does not match the sha256 recorded when it was written")
|
||||
|
||||
// exportBackup re-streams one stored archive through the export guards
|
||||
// (backup.FilterTarGz with archiveFilter). Its length changes on the way, so it
|
||||
// goes chunked. With want set, the stored bytes are hashed as they are read,
|
||||
// and FilterTarGz reads them to their end before it closes its own archive: a
|
||||
// mismatch aborts the upload while what felis-api has passed on still lacks
|
||||
// its end, so the browser never keeps a complete-looking corrupt file.
|
||||
func exportBackup(ctx context.Context, target, token, ref, root, want string, stdout io.Writer) error {
|
||||
// Defense in depth, as in cmdRestore: the ref comes from felis-api, but this
|
||||
// process opens it, so it confirms the ref stays on the backup mount.
|
||||
if !refWithinRoot(ref, root) {
|
||||
return fmt.Errorf("ref %q is not under backup root %q", ref, root)
|
||||
}
|
||||
f, err := os.Open(ref)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
var src io.Reader = f
|
||||
if want != "" {
|
||||
src = &digestReader{r: f, sum: sha256.New(), want: want}
|
||||
}
|
||||
var withheld []string
|
||||
err = streamExport(ctx, target, token, archiveType, -1, func(w io.Writer) error {
|
||||
var err error
|
||||
withheld, err = backup.FilterTarGz(ctx, w, src, archiveFilter)
|
||||
return err
|
||||
})
|
||||
if errors.Is(err, errBackupDigest) {
|
||||
return errBackupDigest // the jobs list shows it as it is, not wrapped as a read error
|
||||
}
|
||||
reportWithheld(stdout, len(withheld))
|
||||
return err
|
||||
}
|
||||
|
||||
// exportWorld archives the world straight into the request body: nothing is
|
||||
// staged, so a world bigger than the Pod's memory or any scratch disk exports
|
||||
// the same.
|
||||
func exportWorld(ctx context.Context, target, token, root string, stdout io.Writer) error {
|
||||
r, err := os.OpenRoot(root)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
guard := fileedit.NewGuard(r)
|
||||
r.Close()
|
||||
var skipped, withheld []string
|
||||
err = streamExport(ctx, target, token, archiveType, -1, func(w io.Writer) error {
|
||||
var err error
|
||||
skipped, withheld, err = backup.WriteTarGz(ctx, w, root, worldFilter(guard))
|
||||
return err
|
||||
})
|
||||
if len(skipped) > 0 {
|
||||
fmt.Fprintf(stdout, "felis export: left out %d entries a tar cannot hold (symbolic links, devices, sockets)\n", len(skipped))
|
||||
}
|
||||
reportWithheld(stdout, len(withheld))
|
||||
return err
|
||||
}
|
||||
|
||||
// exportFiles sends one file or folder of the world (fileedit.OpenDownload): a
|
||||
// file with its exact length, a folder as a zip made as it streams. dir is
|
||||
// what the owner saw at path when they asked.
|
||||
func exportFiles(ctx context.Context, target, token, root, path string, dir bool, stdout io.Writer) error {
|
||||
d, err := fileedit.OpenDownload(root, path, dir)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer d.Close()
|
||||
err = streamExport(ctx, target, token, d.ContentType, d.Size, func(w io.Writer) error { return d.WriteTo(ctx, w) })
|
||||
if d.Skipped > 0 {
|
||||
fmt.Fprintf(stdout, "felis export: left out %d entries a zip does not carry (symbolic links, devices, sockets)\n", d.Skipped)
|
||||
}
|
||||
reportWithheld(stdout, d.Withheld)
|
||||
return err
|
||||
}
|
||||
|
||||
func reportWithheld(stdout io.Writer, n int) {
|
||||
if n > 0 {
|
||||
fmt.Fprintf(stdout, "felis export: left out %d files that hold platform secrets\n", n)
|
||||
}
|
||||
}
|
||||
|
||||
// worldFilter guards a live world by file identity, so a link to a guarded
|
||||
// file under another name is caught as well.
|
||||
func worldFilter(g fileedit.Guard) backup.Filter {
|
||||
return func(_ string, info fs.FileInfo) (bool, func([]byte) []byte) {
|
||||
return guardAction(g.Rule(info))
|
||||
}
|
||||
}
|
||||
|
||||
// archiveFilter guards a stored archive, which has only names.
|
||||
func archiveFilter(name string, _ fs.FileInfo) (bool, func([]byte) []byte) {
|
||||
return guardAction(fileedit.ArchiveRule(name))
|
||||
}
|
||||
|
||||
func guardAction(withhold, redact bool) (bool, func([]byte) []byte) {
|
||||
if redact {
|
||||
return withhold, fileedit.RedactProps
|
||||
}
|
||||
return withhold, nil
|
||||
}
|
||||
|
||||
// digestReader passes r through, hashing it, and turns r's EOF into
|
||||
// errBackupDigest when the bytes do not hash to want.
|
||||
type digestReader struct {
|
||||
r io.Reader
|
||||
sum hash.Hash
|
||||
want string
|
||||
}
|
||||
|
||||
func (d *digestReader) Read(p []byte) (int, error) {
|
||||
n, err := d.r.Read(p)
|
||||
d.sum.Write(p[:n])
|
||||
if err == io.EOF && !strings.EqualFold(hex.EncodeToString(d.sum.Sum(nil)), d.want) {
|
||||
return n, errBackupDigest
|
||||
}
|
||||
return n, err
|
||||
}
|
||||
|
||||
// streamExport runs write straight into the body of the PUT, hashing it as it
|
||||
// goes. Once write has finished, the SHA-256 of all it wrote rides the
|
||||
// request's trailer (worldexport.DigestTrailer), and felis-api holds back the
|
||||
// last bytes from the browser until what it received hashes the same. An error
|
||||
// from write aborts the chunked body before the trailer, and felis-api then
|
||||
// cuts the browser's download off rather than end it; that error is the one
|
||||
// reported, since the PUT's own error only wraps it. When the PUT ends first,
|
||||
// write is stopped.
|
||||
func streamExport(ctx context.Context, target, token, contentType string, size int64, write func(io.Writer) error) error {
|
||||
pr, pw := io.Pipe()
|
||||
trailer := http.Header{worldexport.DigestTrailer: nil}
|
||||
werr := make(chan error, 1)
|
||||
go func() {
|
||||
sum := sha256.New()
|
||||
err := write(io.MultiWriter(pw, sum))
|
||||
if err == nil {
|
||||
// Set before the body ends: the transport reads the trailer once it
|
||||
// has read the body to its end.
|
||||
trailer.Set(worldexport.DigestTrailer, "sha-256=:"+base64.StdEncoding.EncodeToString(sum.Sum(nil))+":")
|
||||
}
|
||||
pw.CloseWithError(err)
|
||||
werr <- err
|
||||
}()
|
||||
err := putExport(ctx, target, token, contentType, pr, size, trailer)
|
||||
pr.CloseWithError(io.ErrClosedPipe)
|
||||
if w := <-werr; w != nil && !errors.Is(w, io.ErrClosedPipe) {
|
||||
return w
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// putExport PUTs the export to felis-api. There is no retry: the token opens
|
||||
// the export once, so a second attempt could only be refused. Redirects are
|
||||
// refused because the request carries the token and the internal face never
|
||||
// redirects. felis-api answers only after the whole download, which the Job's
|
||||
// activeDeadlineSeconds bounds, so the header timeout is a backstop for a
|
||||
// wedged endpoint and not the real limit.
|
||||
//
|
||||
// The body always goes chunked, which is what lets it end with a trailer; a
|
||||
// size the Job knows (-1 when it does not) goes as worldexport.LengthHeader in
|
||||
// place of Content-Length.
|
||||
func putExport(ctx context.Context, target, token, contentType string, body io.Reader, size int64, trailer http.Header) error {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPut, target, body)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
req.ContentLength = -1
|
||||
req.Trailer = trailer
|
||||
if size >= 0 {
|
||||
req.Header.Set(worldexport.LengthHeader, strconv.FormatInt(size, 10))
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
req.Header.Set("Content-Type", contentType)
|
||||
client := &http.Client{
|
||||
Transport: &http.Transport{ResponseHeaderTimeout: 10 * time.Minute},
|
||||
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode == http.StatusNoContent {
|
||||
return nil
|
||||
}
|
||||
var e struct {
|
||||
Error struct {
|
||||
Message string `json:"message"`
|
||||
} `json:"error"`
|
||||
}
|
||||
if json.NewDecoder(io.LimitReader(resp.Body, 4<<10)).Decode(&e) == nil && e.Error.Message != "" {
|
||||
return fmt.Errorf("felis-api answered %s: %s", resp.Status, e.Error.Message)
|
||||
}
|
||||
return fmt.Errorf("felis-api answered %s", resp.Status)
|
||||
}
|
||||
@@ -0,0 +1,532 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"archive/zip"
|
||||
"bytes"
|
||||
"compress/gzip"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/worldexport"
|
||||
)
|
||||
|
||||
// exportReceiver stands in for felis-api's internal upload route: it records
|
||||
// the PUT it gets (or the error reading it ended on) and answers with reply.
|
||||
type exportReceiver struct {
|
||||
srv *httptest.Server
|
||||
hits atomic.Int32
|
||||
req *http.Request
|
||||
body []byte
|
||||
readErr error
|
||||
served chan struct{} // one send per request, once it is answered
|
||||
}
|
||||
|
||||
func receiveExport(t *testing.T, reply func(w http.ResponseWriter)) *exportReceiver {
|
||||
t.Helper()
|
||||
rcv := &exportReceiver{served: make(chan struct{}, 1)}
|
||||
rcv.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
rcv.hits.Add(1)
|
||||
rcv.req = r
|
||||
rcv.body, rcv.readErr = io.ReadAll(r.Body)
|
||||
reply(w)
|
||||
rcv.served <- struct{}{}
|
||||
}))
|
||||
t.Cleanup(rcv.srv.Close)
|
||||
return rcv
|
||||
}
|
||||
|
||||
func noContent(w http.ResponseWriter) { w.WriteHeader(http.StatusNoContent) }
|
||||
|
||||
// sentWhole fails unless the upload rcv got ended with the Content-Digest
|
||||
// trailer of its own bytes, and declared length as its size (-1: none).
|
||||
func sentWhole(t *testing.T, rcv *exportReceiver, length int64) {
|
||||
t.Helper()
|
||||
sum := sha256.Sum256(rcv.body)
|
||||
want := "sha-256=:" + base64.StdEncoding.EncodeToString(sum[:]) + ":"
|
||||
wantLength := ""
|
||||
if length >= 0 {
|
||||
wantLength = strconv.FormatInt(length, 10)
|
||||
}
|
||||
r := rcv.req
|
||||
if rcv.readErr != nil || r.Trailer.Get(worldexport.DigestTrailer) != want || r.Header.Get(worldexport.LengthHeader) != wantLength ||
|
||||
r.ContentLength != -1 || strings.Join(r.TransferEncoding, ",") != "chunked" {
|
||||
t.Fatalf("upload read %v, trailer %v, %s %q, length %d, encoding %v; want trailer %q and %s %q, chunked",
|
||||
rcv.readErr, r.Trailer, worldexport.LengthHeader, r.Header.Get(worldexport.LengthHeader), r.ContentLength, r.TransferEncoding,
|
||||
want, worldexport.LengthHeader, wantLength)
|
||||
}
|
||||
}
|
||||
|
||||
func tarEntries(t *testing.T, archive []byte) map[string]string {
|
||||
t.Helper()
|
||||
gz, err := gzip.NewReader(bytes.NewReader(archive))
|
||||
if err != nil {
|
||||
t.Fatalf("not gzip: %v", err)
|
||||
}
|
||||
out := map[string]string{}
|
||||
tr := tar.NewReader(gz)
|
||||
for {
|
||||
h, err := tr.Next()
|
||||
if err == io.EOF {
|
||||
return out
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("tar: %v", err)
|
||||
}
|
||||
b, _ := io.ReadAll(tr)
|
||||
out[h.Name] = string(b)
|
||||
}
|
||||
}
|
||||
|
||||
// writeTree writes name → body under root, making the folders on the way.
|
||||
func writeTree(t *testing.T, root string, files map[string]string) {
|
||||
t.Helper()
|
||||
for name, body := range files {
|
||||
p := filepath.Join(root, name)
|
||||
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(p, []byte(body), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The two secrets a world holds, and what server.properties reads as once
|
||||
// redacted.
|
||||
const (
|
||||
secretProps = "motd=hi\nrcon.password=hunter2\n"
|
||||
redactedProps = "motd=hi\nrcon.password=<redacted by felis>\n"
|
||||
forwardingKey = "secret: aVeryRealForwardingKey\n"
|
||||
)
|
||||
|
||||
// secretWorld is a world holding both secrets, with a hard link to the
|
||||
// forwarding secret under a name nothing would guard by.
|
||||
func secretWorld(t *testing.T) string {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
writeTree(t, root, map[string]string{
|
||||
"server.properties": secretProps,
|
||||
"config/paper-global.yml": forwardingKey,
|
||||
"world/region/r.0.0.mca": "chunks",
|
||||
})
|
||||
if err := os.MkdirAll(filepath.Join(root, "plugins"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.Link(filepath.Join(root, "config/paper-global.yml"), filepath.Join(root, "plugins/copy.yml")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return root
|
||||
}
|
||||
|
||||
func TestCmdExportWorld(t *testing.T) {
|
||||
root := secretWorld(t)
|
||||
if err := os.Symlink("server.properties", filepath.Join(root, "props-link")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rcv := receiveExport(t, noContent)
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := cmdExport([]string{"--mode", "world", "--server", "survival", "--target-url", rcv.srv.URL + "/api/v1/internal/exports/ab",
|
||||
"--worlds-root", root}, &stdout, &stderr)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
r := rcv.req
|
||||
if r.Method != http.MethodPut || r.URL.Path != "/api/v1/internal/exports/ab" || r.Header.Get("Authorization") != "Bearer tok" ||
|
||||
r.Header.Get("Content-Type") != "application/gzip" || r.ContentLength != -1 || strings.Join(r.TransferEncoding, ",") != "chunked" {
|
||||
t.Fatalf("request = %s %s, headers %v, length %d, encoding %v", r.Method, r.URL.Path, r.Header, r.ContentLength, r.TransferEncoding)
|
||||
}
|
||||
want := map[string]string{
|
||||
"server.properties": redactedProps, "config/": "", "plugins/": "",
|
||||
"world/": "", "world/region/": "", "world/region/r.0.0.mca": "chunks",
|
||||
}
|
||||
if got := tarEntries(t, rcv.body); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("archive holds %v\nwant %v", got, want)
|
||||
}
|
||||
sentWhole(t, rcv, -1)
|
||||
want2 := "felis export: left out 1 entries a tar cannot hold (symbolic links, devices, sockets)\n" +
|
||||
"felis export: left out 2 files that hold platform secrets\n" +
|
||||
"felis export: server=survival mode=world downloaded\n"
|
||||
if stdout.String() != want2 {
|
||||
t.Errorf("stdout = %q, want %q", stdout.String(), want2)
|
||||
}
|
||||
}
|
||||
|
||||
// A world root that cannot be opened fails before anything reaches felis-api.
|
||||
func TestCmdExportWorldUnreadable(t *testing.T) {
|
||||
rcv := receiveExport(t, noContent)
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := cmdExport([]string{"--mode", "world", "--target-url", rcv.srv.URL, "--worlds-root", filepath.Join(t.TempDir(), "missing")}, &stdout, &stderr)
|
||||
if code != 1 || rcv.hits.Load() != 0 {
|
||||
t.Fatalf("exit %d with %d requests, want 1 and none", code, rcv.hits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// An export that fails part-way must never reach felis-api as a complete body:
|
||||
// the chunked upload is cut off, so felis-api aborts the browser's download,
|
||||
// and the failure itself is what the Job reports.
|
||||
func TestStreamExportWriteErrorAbortsTheUpload(t *testing.T) {
|
||||
broken := errors.New("disk read failed")
|
||||
rcv := receiveExport(t, noContent)
|
||||
err := streamExport(context.Background(), rcv.srv.URL, "tok", "application/gzip", -1, func(w io.Writer) error {
|
||||
if _, err := w.Write(bytes.Repeat([]byte("x"), 100_000)); err != nil {
|
||||
return err
|
||||
}
|
||||
return broken
|
||||
})
|
||||
if err != broken {
|
||||
t.Fatalf("err = %v, want the write's own error, unwrapped", err)
|
||||
}
|
||||
select {
|
||||
case <-rcv.served:
|
||||
if rcv.readErr == nil {
|
||||
t.Fatalf("felis-api read a complete %d-byte body from a failed export", len(rcv.body))
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("the request never reached felis-api")
|
||||
}
|
||||
}
|
||||
|
||||
// When felis-api refuses first, its reason is reported, not the closed pipe
|
||||
// that then stops the writer.
|
||||
func TestStreamExportRefusalStopsTheWriter(t *testing.T) {
|
||||
refusing := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
}))
|
||||
defer refusing.Close()
|
||||
stopped := make(chan error, 1)
|
||||
err := streamExport(context.Background(), refusing.URL, "tok", "application/gzip", -1, func(w io.Writer) error {
|
||||
for {
|
||||
if _, err := w.Write(make([]byte, 32<<10)); err != nil {
|
||||
stopped <- err
|
||||
return err
|
||||
}
|
||||
}
|
||||
})
|
||||
if err == nil || err.Error() != "felis-api answered 404 Not Found" {
|
||||
t.Fatalf("err = %v, want felis-api's answer", err)
|
||||
}
|
||||
if werr := <-stopped; !errors.Is(werr, io.ErrClosedPipe) {
|
||||
t.Fatalf("the writer stopped on %v, want the closed pipe", werr)
|
||||
}
|
||||
|
||||
// A PUT that never starts leaves no transport to close the body: the writer
|
||||
// is still stopped, and the export fails rather than hangs.
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- streamExport(context.Background(), "http://[::1", "tok", "application/gzip", -1, func(w io.Writer) error {
|
||||
_, err := w.Write([]byte("x"))
|
||||
return err
|
||||
})
|
||||
}()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err == nil || !strings.Contains(err.Error(), "missing ']'") {
|
||||
t.Fatalf("err = %v, want the bad URL", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("an export whose PUT never started hung")
|
||||
}
|
||||
}
|
||||
|
||||
// storedBackup writes, at path, a gzip+tar like one the backup store holds:
|
||||
// the world whole, both secrets included, and a region file that does not
|
||||
// compress. It returns the archive's sha256.
|
||||
func storedBackup(t *testing.T, path string) string {
|
||||
t.Helper()
|
||||
region := make([]byte, 64<<10)
|
||||
if _, err := rand.Read(region); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var buf bytes.Buffer
|
||||
zw := gzip.NewWriter(&buf)
|
||||
tw := tar.NewWriter(zw)
|
||||
for _, e := range []struct{ name, body string }{
|
||||
{"server.properties", secretProps},
|
||||
{"config/paper-global.yml", forwardingKey},
|
||||
{"world/level.dat", "level"},
|
||||
{"world/region/r.0.0.mca", string(region)},
|
||||
} {
|
||||
if err := tw.WriteHeader(&tar.Header{Name: e.name, Typeflag: tar.TypeReg, Mode: 0o600, Size: int64(len(e.body))}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := io.WriteString(tw, e.body); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := tw.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := zw.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(path, buf.Bytes(), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sum := sha256.Sum256(buf.Bytes())
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
func TestCmdExportBackup(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
ref := filepath.Join(root, "survival-1.tar.gz")
|
||||
sum := storedBackup(t, ref)
|
||||
stored, err := os.ReadFile(ref)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
region := tarEntries(t, stored)["world/region/r.0.0.mca"]
|
||||
args := func(url, ref string) []string {
|
||||
return []string{"--mode", "backup", "--server", "survival", "--target-url", url, "--ref", ref, "--backup-root", root}
|
||||
}
|
||||
|
||||
for name, extra := range map[string][]string{
|
||||
"no digest recorded": nil,
|
||||
"recorded digest matches": {"--sha256", sum},
|
||||
} {
|
||||
t.Run(name+": re-streamed through the guards", func(t *testing.T) {
|
||||
rcv := receiveExport(t, noContent)
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdExport(append(args(rcv.srv.URL, ref), extra...), &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
r := rcv.req
|
||||
if r.ContentLength != -1 || r.Header.Get("Content-Type") != "application/gzip" || r.Header.Get("Authorization") != "Bearer tok" {
|
||||
t.Fatalf("length %d, headers %v", r.ContentLength, r.Header)
|
||||
}
|
||||
want := map[string]string{"server.properties": redactedProps, "world/level.dat": "level", "world/region/r.0.0.mca": region}
|
||||
if got := tarEntries(t, rcv.body); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("archive holds %d entries, want exactly the redacted properties, level.dat and the region file", len(got))
|
||||
}
|
||||
sentWhole(t, rcv, -1)
|
||||
if want := "felis export: left out 1 files that hold platform secrets\nfelis export: server=survival mode=backup downloaded\n"; stdout.String() != want {
|
||||
t.Errorf("stdout = %q, want %q", stdout.String(), want)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The stored bytes are checked as they stream, and the archive the Job sends
|
||||
// is only closed once they are all read: a mismatch cuts the upload off
|
||||
// short of its end, so felis-api never passes on a complete-looking copy.
|
||||
t.Run("a digest mismatch cuts the upload off before its end", func(t *testing.T) {
|
||||
rcv := receiveExport(t, noContent)
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdExport(append(args(rcv.srv.URL, ref), "--sha256", strings.Repeat("ab", 32)), &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("exit %d, want 1", code)
|
||||
}
|
||||
if want := "felis export: the backup archive does not match the sha256 recorded when it was written\n"; stderr.String() != want {
|
||||
t.Fatalf("stderr = %q, want %q", stderr.String(), want)
|
||||
}
|
||||
select {
|
||||
case <-rcv.served:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("the upload never reached felis-api")
|
||||
}
|
||||
if rcv.readErr == nil {
|
||||
t.Fatalf("felis-api read a complete %d-byte body", len(rcv.body))
|
||||
}
|
||||
zr, err := gzip.NewReader(bytes.NewReader(rcv.body))
|
||||
if err == nil {
|
||||
_, err = io.ReadAll(zr)
|
||||
}
|
||||
if err == nil {
|
||||
t.Fatal("what felis-api got is a complete archive")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a ref outside the backup root is refused before any request", func(t *testing.T) {
|
||||
outside := filepath.Join(t.TempDir(), "secret.tar.gz")
|
||||
if err := os.WriteFile(outside, []byte("secret"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rcv := receiveExport(t, noContent)
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdExport(args(rcv.srv.URL, outside), &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("exit %d, want 1", code)
|
||||
}
|
||||
if n := rcv.hits.Load(); n != 0 {
|
||||
t.Fatalf("felis-api got %d requests", n)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a refusal exits 1 with felis-api's reason", func(t *testing.T) {
|
||||
rcv := receiveExport(t, func(w http.ResponseWriter) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.WriteHeader(http.StatusGone)
|
||||
io.WriteString(w, `{"error":{"code":"export_expired","message":"nobody opened the download"}}`)
|
||||
})
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdExport(args(rcv.srv.URL, ref), &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("exit %d, want 1", code)
|
||||
}
|
||||
if got := stderr.String(); got != "felis export: felis-api answered 410 Gone: nobody opened the download\n" {
|
||||
t.Fatalf("stderr = %q", got)
|
||||
}
|
||||
})
|
||||
|
||||
// The request carries the token and the internal face never redirects, so a
|
||||
// redirect is refused rather than followed with the token attached. A 302 or
|
||||
// 303 is the one net/http would follow on its own (as a GET, and to the same
|
||||
// host with the Authorization header still on it).
|
||||
for _, status := range []int{http.StatusFound, http.StatusSeeOther, http.StatusTemporaryRedirect, http.StatusPermanentRedirect} {
|
||||
t.Run("a redirect is not followed: "+strconv.Itoa(status), func(t *testing.T) {
|
||||
elsewhere := receiveExport(t, noContent)
|
||||
redirecting := httptest.NewServer(http.RedirectHandler(elsewhere.srv.URL, status))
|
||||
defer redirecting.Close()
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdExport(args(redirecting.URL, ref), &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("exit %d, want 1", code)
|
||||
}
|
||||
if n := elsewhere.hits.Load(); n != 0 {
|
||||
t.Fatalf("the redirect target got %d requests", n)
|
||||
}
|
||||
if got, want := stderr.String(), "felis export: felis-api answered "+strconv.Itoa(status)+" "+http.StatusText(status)+"\n"; got != want {
|
||||
t.Fatalf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdExportFiles(t *testing.T) {
|
||||
root := secretWorld(t)
|
||||
writeTree(t, root, map[string]string{"plugins/Essentials/config.yml": "x: 1"})
|
||||
if err := os.Symlink("config.yml", filepath.Join(root, "plugins/Essentials/link.yml")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
export := func(t *testing.T, path string, dir bool) (*exportReceiver, int, string, string) {
|
||||
t.Helper()
|
||||
rcv := receiveExport(t, noContent)
|
||||
t.Setenv(worldexport.TokenEnv, "tok")
|
||||
args := []string{"--mode", "files", "--server", "survival", "--target-url", rcv.srv.URL, "--worlds-root", root, "--path", path}
|
||||
if dir {
|
||||
args = append(args, "--dir")
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := cmdExport(args, &stdout, &stderr)
|
||||
return rcv, code, stdout.String(), stderr.String()
|
||||
}
|
||||
|
||||
for path, want := range map[string]string{
|
||||
"world/region/r.0.0.mca": "chunks",
|
||||
"server.properties": redactedProps,
|
||||
} {
|
||||
t.Run("a file goes with its exact length: "+path, func(t *testing.T) {
|
||||
rcv, code, stdout, stderr := export(t, path, false)
|
||||
if code != 0 || stdout != "felis export: server=survival mode=files downloaded\n" {
|
||||
t.Fatalf("exit %d, stdout %q, stderr %q", code, stdout, stderr)
|
||||
}
|
||||
if string(rcv.body) != want || rcv.req.Header.Get("Content-Type") != "application/octet-stream" {
|
||||
t.Fatalf("body %q, type %q; want %q", rcv.body, rcv.req.Header.Get("Content-Type"), want)
|
||||
}
|
||||
sentWhole(t, rcv, int64(len(want)))
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("a folder goes as a zip, guarded", func(t *testing.T) {
|
||||
rcv, code, stdout, stderr := export(t, "plugins", true)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr)
|
||||
}
|
||||
if rcv.req.ContentLength != -1 || rcv.req.Header.Get("Content-Type") != "application/zip" {
|
||||
t.Fatalf("length %d, type %q", rcv.req.ContentLength, rcv.req.Header.Get("Content-Type"))
|
||||
}
|
||||
zr, err := zip.NewReader(bytes.NewReader(rcv.body), int64(len(rcv.body)))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var names []string
|
||||
for _, f := range zr.File {
|
||||
names = append(names, f.Name)
|
||||
}
|
||||
if want := []string{"plugins/", "plugins/Essentials/", "plugins/Essentials/config.yml"}; !slices.Equal(names, want) {
|
||||
t.Fatalf("zip holds %v, want %v", names, want)
|
||||
}
|
||||
want := "felis export: left out 1 entries a zip does not carry (symbolic links, devices, sockets)\n" +
|
||||
"felis export: left out 1 files that hold platform secrets\n" +
|
||||
"felis export: server=survival mode=files downloaded\n"
|
||||
if stdout != want {
|
||||
t.Errorf("stdout = %q, want %q", stdout, want)
|
||||
}
|
||||
})
|
||||
|
||||
for _, c := range []struct {
|
||||
path string
|
||||
dir bool
|
||||
want string
|
||||
}{
|
||||
{"config/paper-global.yml", false, "forwarding secret"},
|
||||
{"plugins/copy.yml", false, "forwarding secret"},
|
||||
{"plugins", false, "is a folder now"},
|
||||
{"server.properties", true, "is not a folder now"},
|
||||
} {
|
||||
t.Run("refused before any request: "+c.path, func(t *testing.T) {
|
||||
rcv, code, _, stderr := export(t, c.path, c.dir)
|
||||
if code != 1 || rcv.hits.Load() != 0 || !strings.Contains(stderr, c.want) {
|
||||
t.Fatalf("exit %d, %d requests, stderr %q; want 1, none, and %q", code, rcv.hits.Load(), stderr, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdExportUsage(t *testing.T) {
|
||||
for name, tc := range map[string]struct {
|
||||
token string
|
||||
args []string
|
||||
}{
|
||||
"no token": {"", []string{"--mode", "world", "--target-url", "http://api/x"}},
|
||||
"no target": {"tok", []string{"--mode", "world"}},
|
||||
"unknown mode": {"tok", []string{"--mode", "both", "--target-url", "http://api/x"}},
|
||||
"backup without ref": {"tok", []string{"--mode", "backup", "--target-url", "http://api/x"}},
|
||||
"files without path": {"tok", []string{"--mode", "files", "--target-url", "http://api/x"}},
|
||||
} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(worldexport.TokenEnv, tc.token)
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdExport(tc.args, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("exit %d, want 2; stderr %q", code, stderr.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestCmdExportWiring: the Job's `felis export` reaches cmdExport, and
|
||||
// felis-api's executor mounts the backup store at the path the archives were
|
||||
// written under, since a backup's ref is an absolute path there.
|
||||
func TestCmdExportWiring(t *testing.T) {
|
||||
t.Setenv(worldexport.TokenEnv, "")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := run([]string{"export"}, &stdout, &stderr); code != 2 ||
|
||||
stderr.String() != "felis export: --target-url and "+worldexport.TokenEnv+" are required\n" {
|
||||
t.Fatalf("felis export = %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
cfg := &config.Config{}
|
||||
cfg.K8s.Namespace, cfg.Archive.LocalPath = "games", "/srv/felis-backups"
|
||||
want := worldexport.Config{Namespace: "games", Image: "felis:1", BackupPVC: "felis-backups", BackupRoot: "/srv/felis-backups"}
|
||||
if got := exportConfig(cfg, "felis:1", "felis-backups"); got != want {
|
||||
t.Fatalf("exportConfig = %+v, want %+v", got, want)
|
||||
}
|
||||
}
|
||||
+111
-18
@@ -1,11 +1,15 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"context"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/signal"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
)
|
||||
@@ -20,32 +24,43 @@ import (
|
||||
// config.Load: felis-api made the authorization decision (the caller owns this
|
||||
// server, and the server is stopped so the RWO world volume is free); this process
|
||||
// is the unprivileged hands that touch bytes. Its entire input is the flags
|
||||
// below plus, for a write, one environment variable. Every isolation guarantee
|
||||
// lives in the Pod spec (internal/fileedit/jobspec.go), and the path-containment
|
||||
// guarantee lives in fileedit.Execute, which resolves the path through os.Root and
|
||||
// therefore cannot be walked out of the world mount.
|
||||
// below plus, for a write, the content variables and, for an upload, one token.
|
||||
// Every isolation guarantee lives in the Pod spec (internal/fileedit/jobspec.go),
|
||||
// and the path-containment guarantee lives in fileedit.Execute, which resolves
|
||||
// every path through os.Root and therefore cannot be walked out of the world
|
||||
// mount.
|
||||
//
|
||||
// Exit status carries a specific meaning that felis-api depends on: a CALLER-fault
|
||||
// outcome — a path that escapes the root, a file that is missing or too large — is
|
||||
// a SUCCESSFUL run that prints a Result carrying an error code, so the API can map
|
||||
// it to a precise 4xx. A non-zero exit means the operation could not be attempted
|
||||
// at all (the world mount is unreadable, the result unprintable), which the API
|
||||
// reports as a 500.
|
||||
// at all (the world mount is unreadable, an upload's bytes could not be fetched
|
||||
// intact, the result unprintable), which the API reports as a 500.
|
||||
func cmdFiles(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("files", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
op := fs.String("op", "", "operation: list, read, or write")
|
||||
op := fs.String("op", "", "operation: list, read, write, mkdir, delete, rename, upload or unzip")
|
||||
path := fs.String("path", "", "path to operate on, relative to the world root (empty = the root itself)")
|
||||
worldsRoot := fs.String("worlds-root", "/data", "mount path of the world PVC; every path resolves under it")
|
||||
expect := fs.String("expect-sha256", "", "write only: refuse unless the file's current SHA-256 (hex) is this")
|
||||
createOnly := fs.Bool("create-only", false, "write only: refuse a path that already exists")
|
||||
to := fs.String("to", "", "rename only: the destination path")
|
||||
sourceURL := fs.String("source-url", "", "upload only: felis-api URL to fetch the bytes from")
|
||||
size := fs.Int64("size", -1, "upload only: the byte count the fetched file must have")
|
||||
sum := fs.String("sha256", "", "write and upload: the SHA-256 (hex) the content or the fetched file must have")
|
||||
overwrite := fs.Bool("overwrite", false, "upload and unzip: replace files already there")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
|
||||
if *op == "" {
|
||||
fmt.Fprintln(stderr, "felis files: --op is required (list, read, or write)")
|
||||
fmt.Fprintln(stderr, "felis files: --op is required")
|
||||
return 2
|
||||
}
|
||||
limitHeapToCgroup()
|
||||
req := fileedit.Request{
|
||||
Op: *op, Path: *path, To: *to, Expect: *expect, CreateOnly: *createOnly, Overwrite: *overwrite,
|
||||
}
|
||||
|
||||
// New content arrives base64-encoded in the environment rather than in argv:
|
||||
// a process's arguments are world-readable on the node (/proc/<pid>/cmdline),
|
||||
@@ -53,22 +68,46 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
|
||||
// secrets — an RCON password in server.properties is the obvious case. The
|
||||
// encoding is what lets arbitrary bytes (CRLF endings, a BOM, a NUL) survive a
|
||||
// channel that must be a valid string.
|
||||
var content []byte
|
||||
if *op == fileedit.OpWrite {
|
||||
raw, ok := os.LookupEnv(fileedit.ContentEnv)
|
||||
if !ok {
|
||||
fmt.Fprintf(stderr, "felis files: a write needs %s in the environment\n", fileedit.ContentEnv)
|
||||
switch *op {
|
||||
case fileedit.OpWrite:
|
||||
// The content's SHA-256 comes with it, so bytes that changed on the way
|
||||
// to this Job are refused rather than written (Request.ContentSHA256).
|
||||
if *sum == "" {
|
||||
fmt.Fprintln(stderr, "felis files: a write needs --sha256")
|
||||
return 2
|
||||
}
|
||||
decoded, err := base64.StdEncoding.DecodeString(raw)
|
||||
content, err := fileedit.ContentFromEnv(os.LookupEnv)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis files: %s is not valid base64: %v\n", fileedit.ContentEnv, err)
|
||||
fmt.Fprintf(stderr, "felis files: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
content = decoded
|
||||
req.Content, req.ContentSHA256 = content, *sum
|
||||
case fileedit.OpUpload:
|
||||
token := os.Getenv(fileedit.UploadTokenEnv)
|
||||
if *sourceURL == "" || token == "" {
|
||||
fmt.Fprintf(stderr, "felis files: an upload needs --source-url and %s\n", fileedit.UploadTokenEnv)
|
||||
return 2
|
||||
}
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
req.Upload = &fileedit.Upload{
|
||||
Size: *size, SHA256: *sum,
|
||||
Open: func() (io.ReadCloser, error) { return fetchUpload(ctx, *sourceURL, token) },
|
||||
Landed: func() {
|
||||
if err := reportLanded(ctx, *sourceURL, token); err != nil {
|
||||
// The file is in place; felis-api drops its copy when it
|
||||
// has sat idle long enough, and the panel cancels it too.
|
||||
fmt.Fprintf(stderr, "felis files: tell felis-api the upload landed: %v\n", err)
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
// An upload or an unzip (the only ops that report progress) can run long
|
||||
// enough that felis-api does not wait on its Job, and the panel shows how far
|
||||
// it has got from the latest of these lines (fileedit.K8sRunner.Ops).
|
||||
req.Progress = fileedit.ThrottledProgress(stdout, time.Second, time.Now)
|
||||
|
||||
res, err := fileedit.Execute(*worldsRoot, *op, *path, content, *expect)
|
||||
res, err := fileedit.Execute(*worldsRoot, req)
|
||||
if err != nil {
|
||||
// The operation could not be attempted — infrastructure, not caller fault.
|
||||
fmt.Fprintf(stderr, "felis files: %v\n", err)
|
||||
@@ -83,3 +122,57 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// fetchUpload opens the staged upload on felis-api's internal face. There is no
|
||||
// retry: the token opens the upload once (fileedit.Stage), so a second attempt
|
||||
// could only be refused, and the caller retries the failed Job whole (a file
|
||||
// sent in parts stays staged until its Job reports it landed, so that retry
|
||||
// does not send it again). Redirects are refused because the request carries the
|
||||
// token and the internal face never redirects; the header timeout catches a
|
||||
// wedged endpoint, and the Job's activeDeadlineSeconds bounds the body.
|
||||
func fetchUpload(ctx context.Context, url, token string) (io.ReadCloser, error) {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
client := &http.Client{
|
||||
Transport: &http.Transport{ResponseHeaderTimeout: 30 * time.Second},
|
||||
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
resp.Body.Close()
|
||||
return nil, fmt.Errorf("GET returned %s", resp.Status)
|
||||
}
|
||||
return resp.Body, nil
|
||||
}
|
||||
|
||||
// reportLanded tells felis-api the upload's file is in place (DELETE on the URL
|
||||
// it was fetched from, with the same token), so it deletes the copy it staged.
|
||||
// One try: the file has landed whatever the answer, and a copy nobody deletes
|
||||
// is dropped once it has sat idle for fileedit.SessionIdle.
|
||||
func reportLanded(ctx context.Context, url, token string) error {
|
||||
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
defer cancel()
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodDelete, url, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
client := &http.Client{
|
||||
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
|
||||
}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
return fmt.Errorf("DELETE returned %s", resp.Status)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,358 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/zip"
|
||||
"bytes"
|
||||
"cmp"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
)
|
||||
|
||||
// filesResult is the Result a `felis files` run printed on its marked line,
|
||||
// the last it prints.
|
||||
func filesResult(t *testing.T, stdout string) fileedit.Result {
|
||||
t.Helper()
|
||||
lines := strings.Split(strings.TrimSpace(stdout), "\n")
|
||||
line, ok := strings.CutPrefix(lines[len(lines)-1], fileedit.ResultPrefix)
|
||||
if !ok {
|
||||
t.Fatalf("stdout has no result line: %q", stdout)
|
||||
}
|
||||
var res fileedit.Result
|
||||
if err := json.Unmarshal([]byte(line), &res); err != nil {
|
||||
t.Fatalf("result line %q: %v", line, err)
|
||||
}
|
||||
return res
|
||||
}
|
||||
|
||||
// stagedSource is felis-api's internal face for one staged upload. It serves
|
||||
// body to a GET carrying Bearer token and 404 to any other, and answers the
|
||||
// DELETE that reports the file landed with landedCode (204 when unset),
|
||||
// redirecting to landedTo when that is a redirect. reports counts those
|
||||
// DELETEs, each with the token and at the path the bytes came from.
|
||||
type stagedSource struct {
|
||||
*httptest.Server
|
||||
reports, strays atomic.Int32
|
||||
landedCode int
|
||||
landedTo string
|
||||
}
|
||||
|
||||
func stagedUpload(t *testing.T, token string, body []byte) *stagedSource {
|
||||
t.Helper()
|
||||
s := &stagedSource{}
|
||||
s.Server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Header.Get("Authorization") != "Bearer "+token || r.URL.Path != "/u" {
|
||||
s.strays.Add(1)
|
||||
http.Error(w, "no such upload", http.StatusNotFound)
|
||||
return
|
||||
}
|
||||
switch r.Method {
|
||||
case http.MethodGet:
|
||||
w.Write(body)
|
||||
case http.MethodDelete:
|
||||
s.reports.Add(1)
|
||||
if s.landedTo != "" {
|
||||
w.Header().Set("Location", s.landedTo)
|
||||
}
|
||||
w.WriteHeader(cmp.Or(s.landedCode, http.StatusNoContent))
|
||||
default:
|
||||
s.strays.Add(1)
|
||||
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
|
||||
}
|
||||
}))
|
||||
t.Cleanup(s.Close)
|
||||
return s
|
||||
}
|
||||
|
||||
func uploadArgs(root, sourceURL string, body []byte) []string {
|
||||
sum := sha256.Sum256(body)
|
||||
return []string{
|
||||
"--op", "upload", "--path", "plugins/a.jar", "--worlds-root", root,
|
||||
"--source-url", sourceURL, "--size", "4", "--sha256", hex.EncodeToString(sum[:]),
|
||||
}
|
||||
}
|
||||
|
||||
// uploadRoot is a world with the plugins folder an upload lands in.
|
||||
func uploadRoot(t *testing.T) string {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
if err := os.Mkdir(filepath.Join(root, "plugins"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return root
|
||||
}
|
||||
|
||||
func TestCmdFilesUpload(t *testing.T) {
|
||||
body := []byte("PK\x03\x04")
|
||||
|
||||
t.Run("fetches the staged bytes with its token and lands them", func(t *testing.T) {
|
||||
root := uploadRoot(t)
|
||||
srv := stagedUpload(t, "tok", body)
|
||||
t.Setenv(fileedit.UploadTokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if !strings.HasPrefix(stdout.String(), fileedit.ProgressPrefix+`{"done":4,"total":4}`+"\n") {
|
||||
t.Fatalf("stdout %q does not start with the progress to the last byte", stdout.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != "" {
|
||||
t.Fatalf("result = %+v", res)
|
||||
}
|
||||
got, err := os.ReadFile(filepath.Join(root, "plugins", "a.jar"))
|
||||
if err != nil || !bytes.Equal(got, body) {
|
||||
t.Fatalf("landed %q, %v", got, err)
|
||||
}
|
||||
if n, strays := srv.reports.Load(), srv.strays.Load(); n != 1 || strays != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("reported landed %d times, %d stray requests, stderr %q; want once", n, strays, stderr.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a file already there is a result, and nothing is reported landed", func(t *testing.T) {
|
||||
root := uploadRoot(t)
|
||||
if err := os.WriteFile(filepath.Join(root, "plugins", "a.jar"), []byte("old!"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
srv := stagedUpload(t, "tok", body)
|
||||
t.Setenv(fileedit.UploadTokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeExists || srv.reports.Load() != 0 {
|
||||
t.Fatalf("result = %+v, reported landed %d times", res, srv.reports.Load())
|
||||
}
|
||||
})
|
||||
|
||||
// The file is in place whatever felis-api answers, so the Job still succeeds
|
||||
// and says why the staged copy may linger. A redirect is not followed, since
|
||||
// the request carries the token.
|
||||
for name, tc := range map[string]struct {
|
||||
code int
|
||||
stderr string
|
||||
}{
|
||||
"refused": {http.StatusNotFound, "felis files: tell felis-api the upload landed: DELETE returned 404 Not Found\n"},
|
||||
"redirected": {http.StatusFound, "felis files: tell felis-api the upload landed: DELETE returned 302 Found\n"},
|
||||
} {
|
||||
t.Run("a landed report "+name+" still lands the file", func(t *testing.T) {
|
||||
var elsewhere atomic.Int32
|
||||
away := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) { elsewhere.Add(1) }))
|
||||
defer away.Close()
|
||||
root := uploadRoot(t)
|
||||
srv := stagedUpload(t, "tok", body)
|
||||
srv.landedCode, srv.landedTo = tc.code, away.URL+"/u"
|
||||
t.Setenv(fileedit.UploadTokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != "" || stderr.String() != tc.stderr || elsewhere.Load() != 0 {
|
||||
t.Fatalf("result = %+v, stderr %q, redirect followed %d times", res, stderr.String(), elsewhere.Load())
|
||||
}
|
||||
if got, err := os.ReadFile(filepath.Join(root, "plugins", "a.jar")); err != nil || !bytes.Equal(got, body) {
|
||||
t.Fatalf("landed %q, %v", got, err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// A refused fetch is the Job failing, never a Result: the API answers it with a
|
||||
// 500 the caller retries whole.
|
||||
t.Run("a refused fetch exits 1 and lands nothing", func(t *testing.T) {
|
||||
root := uploadRoot(t)
|
||||
srv := stagedUpload(t, "tok", body)
|
||||
t.Setenv(fileedit.UploadTokenEnv, "wrong")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(uploadArgs(root, srv.URL+"/u", body), &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("exit %d, want 1; stdout %q", code, stdout.String())
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "404") {
|
||||
t.Fatalf("stderr %q does not name the status", stderr.String())
|
||||
}
|
||||
if srv.reports.Load() != 0 {
|
||||
t.Fatal("a refused fetch was reported landed")
|
||||
}
|
||||
if _, err := os.Lstat(filepath.Join(root, "plugins", "a.jar")); !os.IsNotExist(err) {
|
||||
t.Fatalf("a refused fetch left a file: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
// The request carries the token, and the internal face never redirects, so a
|
||||
// redirect is refused rather than followed with the token attached.
|
||||
t.Run("a redirect is not followed", func(t *testing.T) {
|
||||
root := uploadRoot(t)
|
||||
var hits atomic.Int32
|
||||
elsewhere := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hits.Add(1)
|
||||
w.Write(body)
|
||||
}))
|
||||
defer elsewhere.Close()
|
||||
redirecting := httptest.NewServer(http.RedirectHandler(elsewhere.URL+"/u", http.StatusFound))
|
||||
defer redirecting.Close()
|
||||
t.Setenv(fileedit.UploadTokenEnv, "tok")
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(uploadArgs(root, redirecting.URL+"/u", body), &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("exit %d, want 1", code)
|
||||
}
|
||||
if n := hits.Load(); n != 0 {
|
||||
t.Fatalf("the redirect target was fetched %d times", n)
|
||||
}
|
||||
})
|
||||
|
||||
for name, tc := range map[string]struct {
|
||||
token string
|
||||
drop string
|
||||
}{
|
||||
"no token": {"", ""},
|
||||
"no source URL": {"tok", "--source-url"},
|
||||
} {
|
||||
t.Run(name+" exits 2", func(t *testing.T) {
|
||||
srv := stagedUpload(t, "tok", body)
|
||||
t.Setenv(fileedit.UploadTokenEnv, tc.token)
|
||||
args := uploadArgs(uploadRoot(t), srv.URL+"/u", body)
|
||||
if tc.drop != "" {
|
||||
for i, a := range args {
|
||||
if a == tc.drop {
|
||||
args = append(args[:i:i], args[i+2:]...)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(args, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("exit %d, want 2", code)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestCmdFilesWrite(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
content := []byte("[]\r\n")
|
||||
sum := sha256.Sum256(content)
|
||||
args := []string{"--op", "write", "--path", "ops.json", "--worlds-root", root, "--sha256", hex.EncodeToString(sum[:])}
|
||||
|
||||
t.Run("reassembles the content parts", func(t *testing.T) {
|
||||
t.Setenv(fileedit.ContentPartsEnv, "1")
|
||||
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString(content))
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles(args, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != "" {
|
||||
t.Fatalf("result = %+v", res)
|
||||
}
|
||||
if got, err := os.ReadFile(filepath.Join(root, "ops.json")); err != nil || !bytes.Equal(got, content) {
|
||||
t.Fatalf("wrote %q, %v", got, err)
|
||||
}
|
||||
})
|
||||
|
||||
// Writing what did arrive of an incomplete spec would truncate the file.
|
||||
t.Run("an incomplete content spec exits 2 and writes nothing", func(t *testing.T) {
|
||||
t.Setenv(fileedit.ContentPartsEnv, "2")
|
||||
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString([]byte("x")))
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles([]string{"--op", "write", "--path", "new.txt", "--worlds-root", root, "--sha256", hex.EncodeToString(sum[:])}, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("exit %d, want 2", code)
|
||||
}
|
||||
if _, err := os.Lstat(filepath.Join(root, "new.txt")); !os.IsNotExist(err) {
|
||||
t.Fatalf("an incomplete spec wrote a file: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
// Without the content's SHA-256 the Job could not tell bytes changed on the
|
||||
// way from the bytes felis-api sent.
|
||||
t.Run("a write without its SHA-256 exits 2 and writes nothing", func(t *testing.T) {
|
||||
t.Setenv(fileedit.ContentPartsEnv, "1")
|
||||
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString(content))
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles([]string{"--op", "write", "--path", "new.txt", "--worlds-root", root}, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("exit %d, want 2", code)
|
||||
}
|
||||
if _, err := os.Lstat(filepath.Join(root, "new.txt")); !os.IsNotExist(err) {
|
||||
t.Fatalf("a write without its SHA-256 wrote a file: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("content that changed on the way is a result and writes nothing", func(t *testing.T) {
|
||||
t.Setenv(fileedit.ContentPartsEnv, "1")
|
||||
t.Setenv(fileedit.ContentEnv+"_0", base64.StdEncoding.EncodeToString([]byte("[]\n")))
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles([]string{"--op", "write", "--path", "new.txt", "--worlds-root", root, "--sha256", hex.EncodeToString(sum[:])}, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeDigestMismatch {
|
||||
t.Fatalf("result = %+v, want %s", res, fileedit.CodeDigestMismatch)
|
||||
}
|
||||
if _, err := os.Lstat(filepath.Join(root, "new.txt")); !os.IsNotExist(err) {
|
||||
t.Fatalf("changed content wrote a file: %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// A caller-fault outcome is a successful run carrying a code, so felis-api can
|
||||
// answer the precise 4xx instead of a 500.
|
||||
func TestCmdFilesCallerFaultIsAResult(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles([]string{"--op", "mkdir", "--path", "../out", "--worlds-root", root}, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != fileedit.CodeBadPath {
|
||||
t.Fatalf("result = %+v, want code %s", res, fileedit.CodeBadPath)
|
||||
}
|
||||
stdout.Reset()
|
||||
if code := cmdFiles([]string{"--worlds-root", root}, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("no --op: exit %d, want 2", code)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCmdFilesUnzip checks an unzip extracts next to the archive and reports its
|
||||
// progress before its result, the same way an upload does.
|
||||
func TestCmdFilesUnzip(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
if err := os.Mkdir(filepath.Join(root, "maps"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var zb bytes.Buffer
|
||||
zw := zip.NewWriter(&zb)
|
||||
for name, body := range map[string]string{"world/level.dat": "level", "world/region/r.0.0.mca": "region!"} {
|
||||
w, err := zw.Create(name)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
io.WriteString(w, body)
|
||||
}
|
||||
if err := zw.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(root, "maps", "a.zip"), zb.Bytes(), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdFiles([]string{"--op", "unzip", "--path", "maps/a.zip", "--worlds-root", root}, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
if res := filesResult(t, stdout.String()); res.Code != "" || res.Files != 2 || res.Bytes != 12 {
|
||||
t.Fatalf("result = %+v", res)
|
||||
}
|
||||
if !strings.HasPrefix(stdout.String(), fileedit.ProgressPrefix) ||
|
||||
!strings.Contains(stdout.String(), fileedit.ProgressPrefix+`{"done":12,"total":12}`+"\n") {
|
||||
t.Fatalf("stdout %q does not report the progress to the last byte", stdout.String())
|
||||
}
|
||||
got, err := os.ReadFile(filepath.Join(root, "maps", "world", "region", "r.0.0.mca"))
|
||||
if err != nil || string(got) != "region!" {
|
||||
t.Fatalf("extracted %q, %v", got, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// Host copies of the credentials `felis setup` takes at the keyboard: the [smtp]
|
||||
// relay password and the uploads bucket's keys. The cluster reads them from the
|
||||
// felis-smtp and felis-uploads-s3 Secrets, and a Secret lives in k3s's datastore,
|
||||
// which a reinstall (uninstall.sh keeps /etc/felis) or a host rebuilt from a
|
||||
// database bundle's state/ starts empty. Each file holds the bare value, mode
|
||||
// 0600, directly in /etc/felis beside secrets.env: every installer run applies
|
||||
// the Secrets from these files, and every database bundle, so the off-site copy
|
||||
// too, carries them.
|
||||
const (
|
||||
hostSMTPPasswordPath = "/etc/felis/smtp-password"
|
||||
hostUploadsS3AccessKeyPath = "/etc/felis/uploads-s3-access-key"
|
||||
hostUploadsS3SecretKeyPath = "/etc/felis/uploads-s3-secret-key"
|
||||
)
|
||||
|
||||
// writeHostCredential replaces the file at path with value, mode 0600, through a
|
||||
// temporary file in the same directory, so a crash leaves the old value or the
|
||||
// new one and never a partial one.
|
||||
func writeHostCredential(path, value string) error {
|
||||
tmp, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
defer os.Remove(tmpPath)
|
||||
// CreateTemp already makes the file 0600; the Chmod states it rather than
|
||||
// leaning on that.
|
||||
if err := tmp.Chmod(0o600); err != nil {
|
||||
_ = tmp.Close()
|
||||
return err
|
||||
}
|
||||
if _, err := tmp.WriteString(value); err != nil {
|
||||
_ = tmp.Close()
|
||||
return err
|
||||
}
|
||||
if err := tmp.Sync(); err != nil {
|
||||
_ = tmp.Close()
|
||||
return err
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmpPath, path)
|
||||
}
|
||||
|
||||
// readHostCredential returns the value in path; ok is false when there is no
|
||||
// such file. An empty file is a value: the relay password of a relay without AUTH.
|
||||
func readHostCredential(path string) (value string, ok bool, err error) {
|
||||
b, err := os.ReadFile(path)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return "", false, nil
|
||||
}
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
return string(b), true, nil
|
||||
}
|
||||
|
||||
// relayPassword is the [smtp] relay password as the host holds it: the copy
|
||||
// `felis setup` keeps at path, else, on an install from before that copy, the
|
||||
// felis-smtp Secret in ns, whose absence means a relay without AUTH. cl is only
|
||||
// used when the file is missing; a nil cl then reports errClusterUnreachable.
|
||||
func relayPassword(ctx context.Context, path string, cl client.Client, ns string) (string, error) {
|
||||
if pw, ok, err := readHostCredential(path); err != nil || ok {
|
||||
return pw, err
|
||||
}
|
||||
if cl == nil {
|
||||
return "", errClusterUnreachable
|
||||
}
|
||||
return smtpSecretPassword(ctx, cl, ns)
|
||||
}
|
||||
|
||||
var errClusterUnreachable = errors.New("the cluster did not answer")
|
||||
@@ -0,0 +1,198 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
func TestHostCredentialFile(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "smtp-password")
|
||||
|
||||
if _, ok, err := readHostCredential(path); ok || err != nil {
|
||||
t.Fatalf("a missing file read as ok=%v err=%v; want not there", ok, err)
|
||||
}
|
||||
// A copy an operator put there by hand, readable by everyone, is tightened.
|
||||
if err := os.WriteFile(path, []byte("by hand"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"first secret", "a \"quoted\" $second\nsecret", ""} {
|
||||
if err := writeHostCredential(path, v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, ok, err := readHostCredential(path)
|
||||
if err != nil || !ok || got != v {
|
||||
t.Fatalf("read back (%q, %v, %v); want (%q, true, nil)", got, ok, err, v)
|
||||
}
|
||||
fi, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if mode := fi.Mode().Perm(); mode != 0o600 {
|
||||
t.Fatalf("mode %v; want 0600", mode)
|
||||
}
|
||||
}
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(entries) != 1 {
|
||||
t.Fatalf("the directory holds %d entries; want the credential alone, no leftover temporary file", len(entries))
|
||||
}
|
||||
|
||||
if _, _, err := readHostCredential(dir); err == nil {
|
||||
t.Fatal("an unreadable credential read as fine")
|
||||
}
|
||||
}
|
||||
|
||||
func smtpSecretClient(t *testing.T, password string) client.Client {
|
||||
t.Helper()
|
||||
return fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(&corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: platform.SMTPSecretName},
|
||||
Data: map[string][]byte{platform.SMTPSecretPasswordKey: []byte(password)},
|
||||
}).Build()
|
||||
}
|
||||
|
||||
func TestRelayPassword(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
dir := t.TempDir()
|
||||
host := filepath.Join(dir, "smtp-password")
|
||||
cl := smtpSecretClient(t, "from-secret")
|
||||
|
||||
if pw, err := relayPassword(ctx, host, cl, "felis"); err != nil || pw != "from-secret" {
|
||||
t.Fatalf("without a host copy = (%q, %v); want the Secret's", pw, err)
|
||||
}
|
||||
if _, err := relayPassword(ctx, host, nil, "felis"); !errors.Is(err, errClusterUnreachable) {
|
||||
t.Fatalf("without a host copy or a cluster err = %v; want errClusterUnreachable", err)
|
||||
}
|
||||
if err := writeHostCredential(host, "from-host"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, c := range []client.Client{cl, nil} {
|
||||
if pw, err := relayPassword(ctx, host, c, "felis"); err != nil || pw != "from-host" {
|
||||
t.Fatalf("with a host copy (cluster %v) = (%q, %v); want the host copy", c != nil, pw, err)
|
||||
}
|
||||
}
|
||||
if _, err := relayPassword(ctx, dir, cl, "felis"); err == nil {
|
||||
t.Fatal("an unreadable host copy fell through to the Secret")
|
||||
}
|
||||
}
|
||||
|
||||
// The watchdog mails the most while the cluster is down: the host copy must
|
||||
// reach it then, and without one the password the last good run cached stays.
|
||||
func TestRefreshSMTPPassword(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
dir := t.TempDir()
|
||||
host := filepath.Join(dir, "smtp-password")
|
||||
var stderr bytes.Buffer
|
||||
|
||||
state := &watchdog.State{SMTPPassword: "cached"}
|
||||
refreshSMTPPassword(ctx, host, nil, "felis", state, &stderr)
|
||||
if state.SMTPPassword != "cached" || stderr.Len() != 0 {
|
||||
t.Fatalf("cluster down, no host copy: password %q, stderr %q; want the cached one kept quietly", state.SMTPPassword, stderr.String())
|
||||
}
|
||||
refreshSMTPPassword(ctx, host, smtpSecretClient(t, "from-secret"), "felis", state, &stderr)
|
||||
if state.SMTPPassword != "from-secret" {
|
||||
t.Fatalf("cluster up, no host copy: password %q; want the Secret's", state.SMTPPassword)
|
||||
}
|
||||
if err := writeHostCredential(host, "from-host"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
refreshSMTPPassword(ctx, host, nil, "felis", state, &stderr)
|
||||
if state.SMTPPassword != "from-host" {
|
||||
t.Fatalf("cluster down, host copy: password %q; want the host copy", state.SMTPPassword)
|
||||
}
|
||||
refreshSMTPPassword(ctx, dir, nil, "felis", state, &stderr)
|
||||
if state.SMTPPassword != "from-host" || !strings.Contains(stderr.String(), "keeping the cached one") {
|
||||
t.Fatalf("unreadable host copy: password %q, stderr %q; want the cached one kept and the failure said", state.SMTPPassword, stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestHostRecoveryMailerHostCopy(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
// No kubeconfig anywhere: reaching for the cluster fails, so a pass proves
|
||||
// the host copy was enough.
|
||||
t.Setenv("KUBECONFIG", filepath.Join(t.TempDir(), "no-kubeconfig"))
|
||||
dir := t.TempDir()
|
||||
host := filepath.Join(dir, "smtp-password")
|
||||
if err := writeHostCredential(host, "from-host"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
off := false
|
||||
c := config.SMTPConfig{Host: "mail.example.com", Port: 2525, From: "[email protected]", Username: "felis", PasswordRef: "FELIS_TEST_UNSET_RELAY_PW", RequireTLS: &off}
|
||||
|
||||
got, err := hostRecoveryMailer(c, host, "felis")(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("with the host copy: %v", err)
|
||||
}
|
||||
if relay, ok := got.(*mail.SMTP); !ok || relay.Password != "from-host" {
|
||||
t.Fatalf("relay = %#v; want the host copy's password", got)
|
||||
}
|
||||
if _, err := hostRecoveryMailer(c, dir, "felis")(ctx); err == nil || !strings.Contains(err.Error(), "read the relay password") {
|
||||
t.Fatalf("unreadable host copy: err = %v; want it named", err)
|
||||
}
|
||||
if _, err := hostRecoveryMailer(c, filepath.Join(dir, "none"), "felis")(ctx); err == nil || !strings.Contains(err.Error(), "reach the cluster") {
|
||||
t.Fatalf("no host copy and no cluster: err = %v; want the cluster named", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A whole watchdog run with the API server and PostgreSQL both down still
|
||||
// takes the relay password from the host copy, so the outage mail can
|
||||
// authenticate even when no earlier run cached it.
|
||||
func TestWatchdogReadsHostCopyWhileClusterDown(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("KUBECONFIG", filepath.Join(dir, "no-kubeconfig"))
|
||||
cfgPath := filepath.Join(dir, "felis.toml")
|
||||
if err := os.WriteFile(cfgPath, []byte(`[database]
|
||||
url = "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"
|
||||
[server]
|
||||
root_domain = "example.com"
|
||||
[archive]
|
||||
store = "tarLocal"
|
||||
[k8s]
|
||||
egress_mode = "nodeport"
|
||||
[smtp]
|
||||
host = "127.0.0.1"
|
||||
port = 1
|
||||
from = "[email protected]"
|
||||
username = "felis"
|
||||
password_ref = "FELIS_TEST_UNSET_RELAY_PW"
|
||||
`), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
pwPath := filepath.Join(dir, "smtp-password")
|
||||
if err := writeHostCredential(pwPath, "from-host"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
statePath := filepath.Join(dir, "state.json")
|
||||
var stdout, stderr bytes.Buffer
|
||||
cmdWatchdog([]string{
|
||||
"-config", cfgPath, "-state", statePath, "-quiet-file", filepath.Join(dir, "quiet"),
|
||||
"-backup-dir", "", "-disk-paths", dir, "-smtp-password-file", pwPath, "-heartbeat-file", filepath.Join(dir, "no-heartbeat"),
|
||||
}, &stdout, &stderr)
|
||||
if !strings.Contains(stdout.String(), "kube-api") {
|
||||
t.Fatalf("the run found the API server up; the test needs it down (stdout %s)", stdout.String())
|
||||
}
|
||||
state, err := watchdog.LoadState(statePath)
|
||||
if err != nil {
|
||||
t.Fatalf("load state: %v (stderr %s)", err, stderr.String())
|
||||
}
|
||||
if state.SMTPPassword != "from-host" {
|
||||
t.Fatalf("cached relay password %q; want the host copy (stdout %s, stderr %s)", state.SMTPPassword, stdout.String(), stderr.String())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strings"
|
||||
"syscall"
|
||||
|
||||
"felis.lolicon.best/internal/imagepush"
|
||||
)
|
||||
|
||||
// cmdImageBundle writes a release's image bundle: one OCI layout tar holding every
|
||||
// image an install runs, for one platform, plus its listing (one "role name
|
||||
// manifest-digest config-digest" line per image). deploy/build-release-artifacts.sh
|
||||
// runs it in CI; deploy/bootstrap.sh imports the tar into k3s's containerd and
|
||||
// pushes it into the platform registry with push-image --image.
|
||||
//
|
||||
// --layout role=name=path an image buildx wrote with --output type=oci
|
||||
// --pull role=ref a digest-pinned public image, named repository@digest
|
||||
//
|
||||
// The tar and the listing are written beside their final paths and renamed into
|
||||
// place, so a failed run leaves neither behind.
|
||||
func cmdImageBundle(args []string, _, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("image-bundle", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
platform := fs.String("platform", "", "os/arch the bundle is for, e.g. linux/arm64")
|
||||
out := fs.String("out", "", "path of the bundle tar to write")
|
||||
list := fs.String("list", "", "path of the listing to write")
|
||||
var images []imagepush.BundleImage
|
||||
fs.Func("layout", "role=name=path of an OCI layout tar (repeatable)", func(v string) error {
|
||||
role, rest, ok := strings.Cut(v, "=")
|
||||
name, path, ok2 := strings.Cut(rest, "=")
|
||||
if !ok || !ok2 || role == "" || name == "" || path == "" {
|
||||
return fmt.Errorf("want role=name=path, got %q", v)
|
||||
}
|
||||
images = append(images, imagepush.BundleImage{Role: role, Name: name, Layout: path})
|
||||
return nil
|
||||
})
|
||||
fs.Func("pull", "role=ref of a digest-pinned public image (repeatable)", func(v string) error {
|
||||
role, ref, ok := strings.Cut(v, "=")
|
||||
if !ok || role == "" || !strings.Contains(ref, "@sha256:") {
|
||||
return fmt.Errorf("want role=ref with ref pinned by digest, got %q", v)
|
||||
}
|
||||
images = append(images, imagepush.BundleImage{Role: role, Name: imagepush.PinnedName(ref), Source: ref})
|
||||
return nil
|
||||
})
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if *platform == "" || *out == "" || *list == "" || len(images) == 0 {
|
||||
fmt.Fprintln(stderr, "felis image-bundle: --platform, --out, --list and at least one --layout or --pull are required")
|
||||
return 2
|
||||
}
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
if err := writeImageBundle(ctx, &imagepush.Source{Platform: *platform}, images, *out, *list); err != nil {
|
||||
fmt.Fprintf(stderr, "felis image-bundle: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func writeImageBundle(ctx context.Context, s *imagepush.Source, images []imagepush.BundleImage, out, list string) (err error) {
|
||||
tmpOut, tmpList := out+".tmp", list+".tmp"
|
||||
defer func() {
|
||||
if err != nil {
|
||||
os.Remove(tmpOut)
|
||||
os.Remove(tmpList)
|
||||
}
|
||||
}()
|
||||
f, err := os.Create(tmpOut)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
w := bufio.NewWriterSize(f, 1<<20)
|
||||
entries, err := imagepush.WriteBundle(ctx, s, images, w)
|
||||
if err == nil {
|
||||
err = w.Flush()
|
||||
}
|
||||
if err == nil {
|
||||
err = f.Sync()
|
||||
}
|
||||
if cerr := f.Close(); err == nil {
|
||||
err = cerr
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var b strings.Builder
|
||||
for _, e := range entries {
|
||||
fmt.Fprintf(&b, "%s %s %s %s\n", e.Role, e.Name, e.Digest, e.Config)
|
||||
}
|
||||
if err := os.WriteFile(tmpList, []byte(b.String()), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.Rename(tmpOut, out); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmpList, list)
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// writeTestLayout writes an OCI layout tar holding one linux/arch image with one
|
||||
// layer, the way buildx --output type=oci leaves a single-platform build, and
|
||||
// returns its path with the manifest and config digests.
|
||||
func writeTestLayout(t *testing.T, dir, arch, layer string) (path, manifestDigest, configDigest string) {
|
||||
t.Helper()
|
||||
blobs := map[string][]byte{}
|
||||
add := func(b []byte) (string, int) {
|
||||
sum := sha256.Sum256(b)
|
||||
d := "sha256:" + hex.EncodeToString(sum[:])
|
||||
blobs[d] = b
|
||||
return d, len(b)
|
||||
}
|
||||
cfgDigest, cfgSize := add([]byte(`{"architecture":"` + arch + `","os":"linux","rootfs":{"type":"layers"}}`))
|
||||
layerDigest, layerSize := add([]byte(layer))
|
||||
manifest := fmt.Sprintf(`{"schemaVersion":2,"mediaType":"application/vnd.oci.image.manifest.v1+json",`+
|
||||
`"config":{"mediaType":"application/vnd.oci.image.config.v1+json","digest":%q,"size":%d},`+
|
||||
`"layers":[{"mediaType":"application/vnd.oci.image.layer.v1.tar+gzip","digest":%q,"size":%d}]}`,
|
||||
cfgDigest, cfgSize, layerDigest, layerSize)
|
||||
mDigest, mSize := add([]byte(manifest))
|
||||
index, _ := json.Marshal(map[string]any{
|
||||
"schemaVersion": 2,
|
||||
"manifests": []map[string]any{{
|
||||
"mediaType": "application/vnd.oci.image.manifest.v1+json", "digest": mDigest, "size": mSize,
|
||||
}},
|
||||
})
|
||||
var buf bytes.Buffer
|
||||
tw := tar.NewWriter(&buf)
|
||||
put := func(name string, b []byte) {
|
||||
tw.WriteHeader(&tar.Header{Name: name, Mode: 0o644, Size: int64(len(b)), Typeflag: tar.TypeReg})
|
||||
tw.Write(b)
|
||||
}
|
||||
put("oci-layout", []byte(`{"imageLayoutVersion":"1.0.0"}`))
|
||||
put("index.json", index)
|
||||
for d, b := range blobs {
|
||||
put("blobs/sha256/"+strings.TrimPrefix(d, "sha256:"), b)
|
||||
}
|
||||
tw.Close()
|
||||
path = filepath.Join(dir, arch+"-"+layer+".tar")
|
||||
if err := os.WriteFile(path, buf.Bytes(), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path, mDigest, cfgDigest
|
||||
}
|
||||
|
||||
func TestImageBundleWritesTheListingTheInstallerReads(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
limbo, limboDigest, limboConfig := writeTestLayout(t, dir, "arm64", "limbo")
|
||||
lobby, lobbyDigest, lobbyConfig := writeTestLayout(t, dir, "arm64", "lobby")
|
||||
out, list := filepath.Join(dir, "images.tar"), filepath.Join(dir, "images.txt")
|
||||
var stderr bytes.Buffer
|
||||
code := cmdImageBundle([]string{"--platform", "linux/arm64", "--out", out, "--list", list,
|
||||
"--layout", "limbo=registry.felis.svc:5000/felis/limbo:demo=" + limbo,
|
||||
"--layout", "lobby=registry.felis.svc:5000/felis/lobby:demo=" + lobby,
|
||||
}, nil, &stderr)
|
||||
if code != 0 {
|
||||
t.Fatalf("image-bundle = %d: %s", code, stderr.String())
|
||||
}
|
||||
// deploy/bootstrap.sh reads this with `read -r role name digest config`.
|
||||
got, _ := os.ReadFile(list)
|
||||
want := "limbo registry.felis.svc:5000/felis/limbo:demo " + limboDigest + " " + limboConfig + "\n" +
|
||||
"lobby registry.felis.svc:5000/felis/lobby:demo " + lobbyDigest + " " + lobbyConfig + "\n"
|
||||
if string(got) != want {
|
||||
t.Errorf("listing:\n%s\nwant:\n%s", got, want)
|
||||
}
|
||||
if fi, err := os.Stat(out); err != nil || fi.Size() == 0 {
|
||||
t.Errorf("bundle: %v", err)
|
||||
}
|
||||
if leftovers, _ := filepath.Glob(filepath.Join(dir, "*.tmp")); len(leftovers) != 0 {
|
||||
t.Errorf("left behind %v", leftovers)
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageBundleLeavesNothingWhenAnImageIsRefused(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
amd, _, _ := writeTestLayout(t, dir, "amd64", "limbo")
|
||||
out, list := filepath.Join(dir, "images.tar"), filepath.Join(dir, "images.txt")
|
||||
// The pair an earlier run wrote stays as it was: a listing beside a bundle it
|
||||
// does not describe would have the installer look for images that are not there.
|
||||
os.WriteFile(out, []byte("old bundle"), 0o644)
|
||||
os.WriteFile(list, []byte("old listing\n"), 0o644)
|
||||
var stderr bytes.Buffer
|
||||
code := cmdImageBundle([]string{"--platform", "linux/arm64", "--out", out, "--list", list,
|
||||
"--layout", "limbo=registry.felis.svc:5000/felis/limbo:demo=" + amd}, nil, &stderr)
|
||||
if code != 1 || !strings.Contains(stderr.String(), "is a linux/amd64 image") {
|
||||
t.Fatalf("image-bundle = %d: %s", code, stderr.String())
|
||||
}
|
||||
if got, _ := os.ReadFile(out); string(got) != "old bundle" {
|
||||
t.Errorf("bundle was replaced with %d bytes", len(got))
|
||||
}
|
||||
if got, _ := os.ReadFile(list); string(got) != "old listing\n" {
|
||||
t.Errorf("listing was replaced with %q", got)
|
||||
}
|
||||
if leftovers, _ := filepath.Glob(filepath.Join(dir, "*.tmp")); len(leftovers) != 0 {
|
||||
t.Errorf("left behind %v", leftovers)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPushImageReadsABundleByImageName(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
limbo, _, _ := writeTestLayout(t, dir, "arm64", "limbo")
|
||||
out, list := filepath.Join(dir, "images.tar"), filepath.Join(dir, "images.txt")
|
||||
if code := cmdImageBundle([]string{"--platform", "linux/arm64", "--out", out, "--list", list,
|
||||
"--layout", "limbo=registry.felis.svc:5000/felis/limbo:demo=" + limbo}, nil, &bytes.Buffer{}); code != 0 {
|
||||
t.Fatal("image-bundle failed")
|
||||
}
|
||||
t.Setenv("FELIS_REGISTRY_USERNAME", "platform")
|
||||
t.Setenv("FELIS_REGISTRY_PASSWORD", "x")
|
||||
// The name is looked up in the bundle's index before the registry is contacted,
|
||||
// so a name the bundle lacks fails here with what it does hold.
|
||||
var stderr bytes.Buffer
|
||||
code := cmdPushImage([]string{"--tar", out, "--image", "registry.felis.svc:5000/felis/lobby:demo",
|
||||
"--ref", "127.0.0.1:1/felis/lobby:demo"}, &bytes.Buffer{}, &stderr)
|
||||
if code != 1 || !strings.Contains(stderr.String(), "holds no image named registry.felis.svc:5000/felis/lobby:demo (it holds: registry.felis.svc:5000/felis/limbo:demo") {
|
||||
t.Fatalf("push-image = %d: %s", code, stderr.String())
|
||||
}
|
||||
}
|
||||
@@ -6,8 +6,32 @@ package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
)
|
||||
|
||||
func main() {
|
||||
ensureHostBinDirOnPath()
|
||||
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
|
||||
}
|
||||
|
||||
// hostBinDir is where deploy/bootstrap.sh installs felis, k3s and cloudflared.
|
||||
const hostBinDir = "/usr/local/bin"
|
||||
|
||||
// ensureHostBinDirOnPath appends hostBinDir to PATH when it is missing, so the
|
||||
// k3s and cloudflared this binary execs are found beside it. sudo's secure_path
|
||||
// on EL leaves /usr/local/bin out: `sudo /usr/local/bin/felis db backup` would
|
||||
// otherwise run with no k3s to reach the database's pod through. Appended, so a
|
||||
// PATH that names another k3s first keeps it.
|
||||
func ensureHostBinDirOnPath() {
|
||||
path := os.Getenv("PATH")
|
||||
for _, dir := range filepath.SplitList(path) {
|
||||
if dir == hostBinDir {
|
||||
return
|
||||
}
|
||||
}
|
||||
if path == "" {
|
||||
os.Setenv("PATH", hostBinDir)
|
||||
return
|
||||
}
|
||||
os.Setenv("PATH", path+string(os.PathListSeparator)+hostBinDir)
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"regexp"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestEnsureHostBinDirOnPath(t *testing.T) {
|
||||
for _, c := range []struct{ in, want string }{
|
||||
{"/usr/sbin:/usr/bin", "/usr/sbin:/usr/bin:/usr/local/bin"},
|
||||
{"/usr/local/bin:/usr/bin", "/usr/local/bin:/usr/bin"},
|
||||
{"/opt/k3s:/usr/bin:/usr/local/bin", "/opt/k3s:/usr/bin:/usr/local/bin"},
|
||||
{"", "/usr/local/bin"},
|
||||
} {
|
||||
t.Setenv("PATH", c.in)
|
||||
ensureHostBinDirOnPath()
|
||||
if got := os.Getenv("PATH"); got != c.want {
|
||||
t.Errorf("PATH %q became %q, want %q", c.in, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// run is what the tests drive, so the PATH fix must sit in main, before it.
|
||||
func TestMainFixesPathBeforeRunning(t *testing.T) {
|
||||
b, err := os.ReadFile("main.go")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !regexp.MustCompile(`func main\(\) \{\n\tensureHostBinDirOnPath\(\)\n\tos\.Exit\(run\(`).Match(b) {
|
||||
t.Fatal("main does not call ensureHostBinDirOnPath before run")
|
||||
}
|
||||
}
|
||||
+25
-1
@@ -35,6 +35,11 @@ func (m *multiFlag) Set(v string) error {
|
||||
// --velocity-cidr records the proxy host addresses allowed by the game NetworkPolicy.
|
||||
// Kubernetes permits resident-node traffic regardless, but remote proxy deployments
|
||||
// need an explicit CIDR, so the renderer refuses to guess.
|
||||
//
|
||||
// --only postgres renders just the control-plane database (platform.PostgresObjects),
|
||||
// which the installer brings up before migrations, before it has anything else
|
||||
// to render the full bundle with; it needs neither --felis-image nor
|
||||
// --velocity-cidr.
|
||||
func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("manifests", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
@@ -46,6 +51,8 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
panelNodePort := fs.Int("panel-node-port", int(platform.DefaultPanelNodePort), "NodePort that exposes the built-in HTTPS panel/API origin")
|
||||
felisImage := fs.String("felis-image", "", "container image the felis-api/operator Deployments run, also passed through as FELIS_IMAGE (REQUIRED)")
|
||||
registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry 2.8.3, pinned by digest)")
|
||||
postgresImage := fs.String("postgres-image", "", "control-plane database image (default: PostgreSQL 18.6, pinned by digest)")
|
||||
only := fs.String("only", "", `render one part of the bundle instead of all of it; "postgres" is the control-plane database`)
|
||||
backupPVC := fs.String("backup-pvc", "felis-backups", "name of the world-archive PVC this bundle renders in the Minecraft namespace and advertises to the backup/restore executors via FELIS_BACKUP_PVC (default: felis-backups; pass an empty value to render none, leaving backup/restore answering 503)")
|
||||
worldsHostPath := fs.String("worlds-host-path", "", "node directory the reaper reads worlds from: each world PVC resolves as <path>/<pvc>, or as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (k3s storage root: /var/lib/rancher/k3s/storage); enables the reaper CronJob (requires --archive-local-path and a non-empty --backup-pvc)")
|
||||
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path. With the backup PVC alone it renders the retention-only CronJob, which deletes backups past their expiry and never touches a world")
|
||||
@@ -64,6 +71,18 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
switch *only {
|
||||
case "":
|
||||
case "postgres":
|
||||
return renderManifests(stdout, stderr, platform.PostgresObjects(platform.Params{
|
||||
ControlNamespace: *controlNS,
|
||||
MinecraftNamespace: *minecraftNS,
|
||||
PostgresImage: *postgresImage,
|
||||
}))
|
||||
default:
|
||||
fmt.Fprintf(stderr, "felis manifests: --only %q: the one part that renders alone is \"postgres\"\n", *only)
|
||||
return 2
|
||||
}
|
||||
|
||||
// Keep proxy placement explicit. This matters for remote proxies and documents
|
||||
// the expected source even when Velocity runs on the resident node.
|
||||
@@ -165,6 +184,7 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
PanelNodePort: int32(*panelNodePort),
|
||||
FelisImage: *felisImage,
|
||||
RegistryImage: *registryImage,
|
||||
PostgresImage: *postgresImage,
|
||||
BackupPVC: *backupPVC,
|
||||
WorldsHostPath: *worldsHostPath,
|
||||
ReaperNode: *reaperNode,
|
||||
@@ -183,7 +203,11 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis manifests: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
out, err := platform.RenderYAML(params)
|
||||
return renderManifests(stdout, stderr, platform.Objects(params))
|
||||
}
|
||||
|
||||
func renderManifests(stdout, stderr io.Writer, objs []platform.Object) int {
|
||||
out, err := platform.RenderObjects(objs)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis manifests: render: %v\n", err)
|
||||
return 1
|
||||
|
||||
@@ -278,3 +278,37 @@ func TestManifestsStorageSizes(t *testing.T) {
|
||||
t.Errorf("--registry-storage lots: exit %d, stderr %q; want a refusal naming the flag", code, errBuf.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsOnlyPostgres: the installer renders the database before it has
|
||||
// an image or a proxy address for the rest of the bundle, so --only postgres
|
||||
// must render without them, and render the database and nothing else (a stray
|
||||
// Deployment in that apply would start without its identities).
|
||||
func TestManifestsOnlyPostgres(t *testing.T) {
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run([]string{"manifests", "--only", "postgres", "--control-namespace", "ctl", "--postgres-image", "example/pg:18@sha256:abc"}, &out, &errBuf)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String())
|
||||
}
|
||||
var kinds []string
|
||||
for _, doc := range strings.Split(out.String(), "\n---\n") {
|
||||
for _, line := range strings.Split(doc, "\n") {
|
||||
if strings.HasPrefix(line, "kind: ") {
|
||||
kinds = append(kinds, strings.TrimPrefix(line, "kind: "))
|
||||
}
|
||||
}
|
||||
}
|
||||
if got, want := strings.Join(kinds, ","), "Namespace,ConfigMap,NetworkPolicy,Deployment,Service"; got != want {
|
||||
t.Errorf("rendered kinds %s, want %s", got, want)
|
||||
}
|
||||
for _, want := range []string{"name: felis-postgres", "namespace: ctl", "image: example/pg:18@sha256:abc"} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("rendered database missing %q", want)
|
||||
}
|
||||
}
|
||||
|
||||
out.Reset()
|
||||
errBuf.Reset()
|
||||
if code := run([]string{"manifests", "--only", "registry"}, &out, &errBuf); code != 2 || out.Len() != 0 {
|
||||
t.Errorf("--only registry: exit %d with %d bytes of YAML, want 2 and none", code, out.Len())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"runtime/debug"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// cgroupMemoryFiles are where a container reads the memory it is allowed:
|
||||
// cgroup v2 first, then v1.
|
||||
var cgroupMemoryFiles = []string{"/sys/fs/cgroup/memory.max", "/sys/fs/cgroup/memory/memory.limit_in_bytes"}
|
||||
|
||||
// limitHeapToCgroup sets the Go heap's soft limit from the container's memory
|
||||
// limit, so the collector works harder as a Job nears it and the kernel does not
|
||||
// kill the Job first. An extraction or a folder zipped for download keeps a few
|
||||
// hundred bytes per entry for as long as it runs; without the limit the heap
|
||||
// grows to twice that before a collection, and a 256 MiB Job was killed at
|
||||
// 400,000 entries whose live heap was 115 MB. GOMEMLIMIT set by hand wins.
|
||||
func limitHeapToCgroup() {
|
||||
if os.Getenv("GOMEMLIMIT") != "" {
|
||||
return
|
||||
}
|
||||
for _, f := range cgroupMemoryFiles {
|
||||
b, err := os.ReadFile(f)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if n, ok := softMemoryLimit(string(b)); ok {
|
||||
debug.SetMemoryLimit(n)
|
||||
}
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// softMemoryLimit answers three fifths of the limit a cgroup memory file holds,
|
||||
// or false for "max" (no limit) and anything unreadable. The rest is left for
|
||||
// what the kernel charges the container beyond the Go heap: the page cache of
|
||||
// the files it reads and writes, and the inodes it creates. Under a 256 MiB
|
||||
// limit, 400,000 extracted entries peaked at 184 MB resident with the heap held
|
||||
// to 150 MiB.
|
||||
func softMemoryLimit(content string) (int64, bool) {
|
||||
n, err := strconv.ParseInt(strings.TrimSpace(content), 10, 64)
|
||||
if err != nil || n <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
return n / 5 * 3, true
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime/debug"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestSoftMemoryLimit(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
in string
|
||||
want int64
|
||||
ok bool
|
||||
}{
|
||||
{"268435456\n", 161061273, true}, // 256 MiB, as memory.max holds it
|
||||
{"max\n", 0, false},
|
||||
{"0\n", 0, false},
|
||||
{"-1", 0, false},
|
||||
{"", 0, false},
|
||||
} {
|
||||
got, ok := softMemoryLimit(c.in)
|
||||
if got != c.want || ok != c.ok {
|
||||
t.Errorf("softMemoryLimit(%q) = %d %v, want %d %v", c.in, got, ok, c.want, c.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLimitHeapToCgroup checks the limit comes from the first cgroup file there
|
||||
// is, and that GOMEMLIMIT set by hand leaves the heap alone.
|
||||
func TestLimitHeapToCgroup(t *testing.T) {
|
||||
prevFiles, prevLimit := cgroupMemoryFiles, debug.SetMemoryLimit(-1)
|
||||
t.Cleanup(func() { cgroupMemoryFiles = prevFiles; debug.SetMemoryLimit(prevLimit) })
|
||||
dir := t.TempDir()
|
||||
v1, v1b := filepath.Join(dir, "v1"), filepath.Join(dir, "v1b")
|
||||
if err := os.WriteFile(v1, []byte("268435456\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(v1b, []byte("536870912\n"), 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
cgroupMemoryFiles = []string{filepath.Join(dir, "missing"), v1, v1b}
|
||||
|
||||
t.Setenv("GOMEMLIMIT", "")
|
||||
debug.SetMemoryLimit(math.MaxInt64)
|
||||
limitHeapToCgroup()
|
||||
if got := debug.SetMemoryLimit(-1); got != 161061273 {
|
||||
t.Fatalf("limit = %d, want three fifths of 256 MiB", got)
|
||||
}
|
||||
|
||||
t.Setenv("GOMEMLIMIT", "1GiB")
|
||||
debug.SetMemoryLimit(math.MaxInt64)
|
||||
limitHeapToCgroup()
|
||||
if got := debug.SetMemoryLimit(-1); got != math.MaxInt64 {
|
||||
t.Fatalf("limit = %d with GOMEMLIMIT set, want it left alone", got)
|
||||
}
|
||||
}
|
||||
+52
-7
@@ -2,9 +2,11 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
@@ -18,7 +20,8 @@ import (
|
||||
// database that already holds a schema and has migrations pending is bundled
|
||||
// first (internal/dbbackup, label pre-migrate). A failed snapshot stops the
|
||||
// upgrade; -no-backup is the explicit way past it, e.g. for an external
|
||||
// database whose server is newer than the host's pg_dump.
|
||||
// database (no [database] deployment, so the host's own pg_dump runs) whose
|
||||
// server is newer than that pg_dump.
|
||||
func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("migrate", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
@@ -58,7 +61,7 @@ func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
if !*noBackup {
|
||||
path, err := preMigrateBackup(ctx, drv, migrations, cfg.Database.URL, *backupDir, stderr)
|
||||
path, err := preMigrateBackup(ctx, drv, migrations, cfg.Database, *backupDir, stderr)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis migrate: pre-migration backup failed, nothing applied: %v\n", err)
|
||||
fmt.Fprintln(stderr, " fix the backup, or re-run with -no-backup to migrate without one")
|
||||
@@ -82,10 +85,14 @@ func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// preMigrateStateDir is the host state a pre-migrate bundle carries; tests
|
||||
// point it at a directory of their own.
|
||||
var preMigrateStateDir = dbbackup.DefaultStateDir
|
||||
|
||||
// preMigrateBackup bundles the database when it already carries a schema and
|
||||
// some of migrations are not applied yet, and returns the bundle's path ("" when
|
||||
// there was nothing to protect: a fresh database, or nothing pending).
|
||||
func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.Migration, dbURL, dir string, log io.Writer) (string, error) {
|
||||
func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.Migration, db config.DatabaseConfig, dir string, log io.Writer) (string, error) {
|
||||
if err := drv.EnsureVersionTable(ctx); err != nil {
|
||||
return "", fmt.Errorf("ensure version table: %w", err)
|
||||
}
|
||||
@@ -96,11 +103,22 @@ func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.
|
||||
if len(done) == 0 || !hasPending(done, migrations) {
|
||||
return "", nil
|
||||
}
|
||||
return dbbackup.Backup(ctx, dbbackup.BackupOptions{
|
||||
DatabaseURL: dbURL, Dir: dir, Label: dbbackup.LabelPreMigrate,
|
||||
Keep: defaultKeep[dbbackup.LabelPreMigrate], StateDir: dbbackup.DefaultStateDir,
|
||||
Version: resolvedVersion(), Log: log, Record: true,
|
||||
tools, err := dbTools(db)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
path, err := dbbackup.Backup(ctx, dbbackup.BackupOptions{
|
||||
DatabaseURL: db.URL, Tools: tools, Dir: dir, Label: dbbackup.LabelPreMigrate,
|
||||
Keep: defaultKeep[dbbackup.LabelPreMigrate], StateDir: preMigrateStateDir,
|
||||
Version: resolvedVersion(), ExportServers: exportMinecraftServers, Log: log, Record: true,
|
||||
})
|
||||
if errors.Is(err, dbbackup.ErrServersMissing) {
|
||||
// Rolling the migration back needs the database alone. Backup logged
|
||||
// the gap, and the panel and the watchdog show it while this is the
|
||||
// newest bundle.
|
||||
return path, nil
|
||||
}
|
||||
return path, err
|
||||
}
|
||||
|
||||
func hasPending(done map[int]struct{}, migrations []store.Migration) bool {
|
||||
@@ -121,6 +139,33 @@ func openStore(ctx context.Context, url string, allowPending bool) (*store.Postg
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return checkSchema(ctx, drv, allowPending)
|
||||
}
|
||||
|
||||
// podDBWindow and podDBInterval bound how long a pod that has just started retries its
|
||||
// first database dial while the network policy has yet to admit it
|
||||
// (store.OpenRetrying). A minute is far past the sync lag and far inside every Job's
|
||||
// deadline. Vars so a test can shrink them.
|
||||
var (
|
||||
podDBWindow = time.Minute
|
||||
podDBInterval = time.Second
|
||||
)
|
||||
|
||||
// openPodStore is openStore for felis-api and the reaper and backup Jobs, whose first
|
||||
// dial comes milliseconds after their pod starts.
|
||||
func openPodStore(ctx context.Context, url, prog string, stderr io.Writer) (*store.PostgresDriver, error) {
|
||||
drv, err := store.OpenRetrying(ctx, url, podDBWindow, podDBInterval, func(err error) {
|
||||
fmt.Fprintf(stderr, "felis %s: %v; retrying (a pod that has just started waits for the network policy to admit it)\n", prog, err)
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return checkSchema(ctx, drv, false)
|
||||
}
|
||||
|
||||
// checkSchema closes drv and fails when its schema is not the one this build was
|
||||
// written against (see openStore).
|
||||
func checkSchema(ctx context.Context, drv *store.PostgresDriver, allowPending bool) (*store.PostgresDriver, error) {
|
||||
s, err := store.ReadSchema(ctx, drv)
|
||||
if err == nil {
|
||||
err = s.Err()
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// openPodStore retries a refused first dial for podDBWindow, saying so on stderr
|
||||
// under the calling command's name each time.
|
||||
func TestOpenPodStoreRetriesARefusedDial(t *testing.T) {
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
addr := ln.Addr().String()
|
||||
ln.Close()
|
||||
|
||||
window, interval := podDBWindow, podDBInterval
|
||||
podDBWindow, podDBInterval = 200*time.Millisecond, 20*time.Millisecond
|
||||
t.Cleanup(func() { podDBWindow, podDBInterval = window, interval })
|
||||
|
||||
var stderr bytes.Buffer
|
||||
start := time.Now()
|
||||
_, err = openPodStore(context.Background(), fmt.Sprintf("postgres://felis@%s/felis?sslmode=disable", addr), "reaper", &stderr)
|
||||
if !errors.Is(err, syscall.ECONNREFUSED) {
|
||||
t.Fatalf("err = %v, want a refused dial", err)
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed < podDBWindow {
|
||||
t.Fatalf("gave up after %s, inside the %s window", elapsed, podDBWindow)
|
||||
}
|
||||
lines := strings.Split(strings.TrimSuffix(stderr.String(), "\n"), "\n")
|
||||
if len(lines) < 2 || len(lines) > 11 {
|
||||
t.Fatalf("%d retry lines in a 200ms window at 20ms:\n%s", len(lines), stderr.String())
|
||||
}
|
||||
for _, l := range lines {
|
||||
if !strings.HasPrefix(l, "felis reaper: failed to connect to `user=felis database=felis`: ") ||
|
||||
!strings.HasSuffix(l, "; retrying (a pod that has just started waits for the network policy to admit it)") {
|
||||
t.Fatalf("retry line %q", l)
|
||||
}
|
||||
}
|
||||
}
|
||||
+341
-48
@@ -33,12 +33,25 @@ const offsiteUsage = `usage:
|
||||
felis offsite fetch-worlds [-config path] [-archive-dir dir]
|
||||
felis offsite fetch-images [-config path] [-registry host:port] [-at version]
|
||||
felis offsite fetch-uploads [-config path] [-uploads-dir dir] [-at version]
|
||||
felis offsite check-key [-config path]
|
||||
felis offsite take-over [-config path] [-status-file path] [-yes]
|
||||
felis offsite keygen
|
||||
|
||||
Every verb but keygen reads the bucket credentials and the encryption key from
|
||||
the variables [offsite] names (default FELIS_OFFSITE_ACCESS_KEY,
|
||||
FELIS_OFFSITE_SECRET_KEY, FELIS_OFFSITE_KEY), taking any that are unset from
|
||||
-env-file (default /etc/felis/offsite.env).
|
||||
|
||||
check-key tells whether the key is the one the bucket's objects are sealed
|
||||
with, writing nothing; it exits 3 when they are sealed with another key, and
|
||||
sync then refuses to write or prune anything in the bucket.
|
||||
|
||||
take-over names the host that writes the bucket, writing nothing; it exits 4
|
||||
when that is another host and this one never wrote it, and 5 when another
|
||||
host took the bucket over from this one. A host built from another host's
|
||||
backup (a rehearsal, or a rebuild) copies nothing into that host's bucket
|
||||
until -yes makes it the writer; the host it replaces then stops copying and
|
||||
says so.
|
||||
`
|
||||
|
||||
// defaultOffsiteEnvFile is where bootstrap keeps the [offsite] secrets; the
|
||||
@@ -74,6 +87,10 @@ func cmdOffsite(args []string, stdout, stderr io.Writer) int {
|
||||
return offsiteFetchImages(fs, rest, stdout, stderr)
|
||||
case "fetch-uploads":
|
||||
return offsiteFetchUploads(fs, rest, stdout, stderr)
|
||||
case "check-key":
|
||||
return offsiteCheckKey(fs, rest, stdout, stderr)
|
||||
case "take-over":
|
||||
return offsiteTakeOver(fs, rest, stdout, stderr)
|
||||
case "keygen":
|
||||
k, err := offsite.NewKey()
|
||||
if err != nil {
|
||||
@@ -194,6 +211,7 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
|
||||
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
|
||||
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
|
||||
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory bundled into the database bundle taken after archives are copied ("" for none)`)
|
||||
registry := fs.String("registry", "", `host[:port] of the registry whose user images are copied (default: the in-cluster registry's loopback hostPort; "off" copies none)`)
|
||||
uploadsDir := fs.String("uploads-dir", "", "host directory of the submission uploads volume (default: resolved from the uploads PVC through the cluster)")
|
||||
uploadsPVC := fs.String("uploads-pvc", platform.UploadsPVCName, `the submission uploads PVC, in the control-plane namespace ("" copies no uploads)`)
|
||||
@@ -206,33 +224,32 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
st := offsite.Status{
|
||||
LastAttempt: time.Now().UTC(), Endpoint: env.cfg.Endpoint, Bucket: env.cfg.Bucket,
|
||||
Prefix: env.cfg.Prefix, KeyID: offsite.KeyID(env.key),
|
||||
}
|
||||
if prev, _ := offsite.ReadStatus(*statusFile); prev != nil {
|
||||
st.LastSuccess = prev.LastSuccess
|
||||
}
|
||||
st, lease := startRun(env.cfg, env.key, *statusFile, time.Now())
|
||||
res, err := runOffsiteSync(cfg, env, offsiteSources{
|
||||
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir,
|
||||
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir, stateDir: *stateDir,
|
||||
registry: offsiteRegistryEndpoint(*registry, cfg.Registry),
|
||||
uploadsDir: *uploadsDir, uploadsPVC: *uploadsPVC,
|
||||
}, stderr)
|
||||
st.Result = res
|
||||
if err != nil {
|
||||
st.LastError = err.Error()
|
||||
} else {
|
||||
st.LastSuccess = st.LastAttempt
|
||||
}
|
||||
}, &lease, stderr)
|
||||
recordRun(&st, res, err, lease)
|
||||
if werr := offsite.WriteStatus(*statusFile, st); werr != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite sync: record status: %v\n", werr)
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; images copied=%d blobs=%d pruned=%d; uploads copied=%d pruned=%d; bucket holds %d worlds (%s), %d bundles, %d images in %d repositories (%s), %d uploads (%s)\n",
|
||||
res.WorldsUploaded, res.WorldsPending, len(res.WorldsMissing), res.WorldsExpired,
|
||||
res.DBUploaded, res.DBPruned, res.ImagesUploaded, res.ImageBlobsUploaded, res.ImageObjectsPruned,
|
||||
res.UploadsUploaded, res.UploadObjectsPruned,
|
||||
res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB, res.Images, res.ImageRepos, offsite.HumanBytes(res.RemoteImageBytes),
|
||||
res.Uploads, offsite.HumanBytes(res.RemoteUploadBytes))
|
||||
return reportRun(res, err, stdout, stderr)
|
||||
}
|
||||
|
||||
// reportRun prints one pass's outcome and its exit code. A run stopped before
|
||||
// it copied anything (a bucket that did not answer, another key's objects,
|
||||
// another host writing the bucket) prints no counts: its zeros would read as
|
||||
// an empty bucket.
|
||||
func reportRun(res offsite.Result, err error, stdout, stderr io.Writer) int {
|
||||
if err == nil || len(res.Errors) > 0 {
|
||||
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; images copied=%d blobs=%d pruned=%d; uploads copied=%d pruned=%d; bucket holds %d worlds (%s), %d bundles, %d images in %d repositories (%s), %d uploads (%s)\n",
|
||||
res.WorldsUploaded, res.WorldsPending, len(res.WorldsMissing), res.WorldsExpired,
|
||||
res.DBUploaded, res.DBPruned, res.ImagesUploaded, res.ImageBlobsUploaded, res.ImageObjectsPruned,
|
||||
res.UploadsUploaded, res.UploadObjectsPruned,
|
||||
res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB, res.Images, res.ImageRepos, offsite.HumanBytes(res.RemoteImageBytes),
|
||||
res.Uploads, offsite.HumanBytes(res.RemoteUploadBytes))
|
||||
}
|
||||
for _, m := range res.WorldsMissing {
|
||||
fmt.Fprintf(stderr, "felis offsite sync: recorded archive not on the volume, nothing to copy: %s\n", m)
|
||||
}
|
||||
@@ -246,19 +263,62 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
return 0
|
||||
}
|
||||
|
||||
// startRun begins a pass: its status record, in this release's format and
|
||||
// carrying the last success over, and this host's lease, read from the record
|
||||
// the last pass left.
|
||||
func startRun(cfg config.OffsiteConfig, key []byte, statusFile string, now time.Time) (offsite.Status, offsite.Lease) {
|
||||
st := offsite.Status{
|
||||
LastAttempt: now.UTC(), Endpoint: cfg.Endpoint, Bucket: cfg.Bucket,
|
||||
Prefix: cfg.Prefix, KeyID: offsite.KeyID(key), Format: offsite.StatusFormat,
|
||||
}
|
||||
if prev, _ := offsite.ReadStatus(statusFile); prev != nil {
|
||||
st.LastSuccess = prev.LastSuccess
|
||||
}
|
||||
return st, offsite.HostLease(statusFile)
|
||||
}
|
||||
|
||||
// recordRun puts one pass's outcome into its status record. A host that
|
||||
// inherited the bucket from an older release keeps that until it has an id.
|
||||
func recordRun(st *offsite.Status, res offsite.Result, err error, lease offsite.Lease) {
|
||||
st.Result = res
|
||||
if err != nil {
|
||||
st.LastError = err.Error()
|
||||
st.KeyMismatch = errors.Is(err, offsite.ErrKeyMismatch)
|
||||
var we *offsite.WriterError
|
||||
if errors.As(err, &we) {
|
||||
st.Standby = errors.Is(err, offsite.ErrStandby)
|
||||
st.Displaced = errors.Is(err, offsite.ErrDisplaced)
|
||||
st.Writer = we.Writer
|
||||
}
|
||||
} else {
|
||||
st.LastSuccess = st.LastAttempt
|
||||
}
|
||||
if lease.Inherited {
|
||||
id, _ := lease.ID()
|
||||
st.Inherited = id == ""
|
||||
}
|
||||
}
|
||||
|
||||
// offsiteSources is where one sync pass reads from: the world archive volume
|
||||
// (archiveDir, or the backupPVC's directory), the bundle directory, the
|
||||
// registry's loopback endpoint and the uploads volume (uploadsDir, or the
|
||||
// uploadsPVC's directory). An empty source is skipped.
|
||||
// (archiveDir, or the backupPVC's directory), the bundle directory (with the
|
||||
// host state the pass bundles, stateDir), the registry's loopback endpoint and
|
||||
// the uploads volume (uploadsDir, or the uploadsPVC's directory). An empty
|
||||
// source is skipped.
|
||||
type offsiteSources struct {
|
||||
archiveDir, backupPVC string
|
||||
dbDir string
|
||||
dbDir, stateDir string
|
||||
registry string
|
||||
uploadsDir, uploadsPVC string
|
||||
}
|
||||
|
||||
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, log io.Writer) (offsite.Result, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 50*time.Minute)
|
||||
// offsiteRunLimit backstops one sync pass. Each upload has its own deadline,
|
||||
// scaled to its size (internal/offsite), so a pass over a big archive may run
|
||||
// for hours; the timer starts no second pass while one runs, and the unit's
|
||||
// TimeoutStartSec sits above this.
|
||||
const offsiteRunLimit = 23 * time.Hour
|
||||
|
||||
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, lease *offsite.Lease, log io.Writer) (offsite.Result, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), offsiteRunLimit)
|
||||
defer cancel()
|
||||
checkCtx, checkCancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
err := env.bucket.Check(checkCtx)
|
||||
@@ -288,10 +348,8 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, log
|
||||
return offsite.Result{}, fmt.Errorf("open database: %w", err)
|
||||
}
|
||||
defer drv.Close()
|
||||
s := &offsite.Syncer{
|
||||
Bucket: env.bucket, Catalog: offsite.PGCatalog{DB: drv.DB()}, Key: env.key,
|
||||
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Log: log,
|
||||
}
|
||||
s := offsiteSyncer(cfg, env, src, archiveDir, uploadsDir, lease, log)
|
||||
s.Catalog = offsite.PGCatalog{DB: drv.DB()}
|
||||
if src.registry != "" {
|
||||
s.Images = newRegistryImages(src.registry)
|
||||
s.ImagePins = imagePins(drv.DB(), cfg.Registry.URL)
|
||||
@@ -299,6 +357,55 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, log
|
||||
return s.Run(ctx)
|
||||
}
|
||||
|
||||
// offsiteSyncer is the pass runOffsiteSync runs over the resolved archive and
|
||||
// uploads directories, before its catalog and registry are attached. It
|
||||
// snapshots the database into the bundle directory after copying archives, and
|
||||
// sweeps world objects no backup records once they outlive every retention in
|
||||
// [archive]; a retention that does not parse sweeps none.
|
||||
func offsiteSyncer(cfg *config.Config, env *offsiteEnv, src offsiteSources, archiveDir, uploadsDir string, lease *offsite.Lease, log io.Writer) *offsite.Syncer {
|
||||
s := &offsite.Syncer{
|
||||
Bucket: env.bucket, Key: env.key,
|
||||
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Lease: lease, Log: log,
|
||||
}
|
||||
if src.dbDir != "" {
|
||||
s.Snapshot = offsiteSnapshot(cfg.Database, src.dbDir, src.stateDir, log)
|
||||
}
|
||||
if rc, err := reaperConfig(cfg); err != nil {
|
||||
fmt.Fprintf(log, "felis offsite: world objects no backup records are kept: %v\n", err)
|
||||
} else {
|
||||
s.OrphanAfter = max(rc.Retention, rc.ManualRetention, rc.ScheduledRetention)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// offsiteSnapshot takes the bundle a pass sends after copying world archives
|
||||
// (offsite.Syncer.Snapshot): what `felis db backup` takes, labelled offsite,
|
||||
// with the newest one kept in dir. It is not recorded for the panel, whose
|
||||
// backup card watches felis-db-backup.timer: snapshots come only when archives
|
||||
// are copied, and would hide a daily timer that stopped. It requires the
|
||||
// MinecraftServer objects: it becomes the newest bundle in the bucket, which a
|
||||
// lost host restores from, and a pass that cannot take a whole one fails and
|
||||
// tries again next hour.
|
||||
func offsiteSnapshot(db config.DatabaseConfig, dir, stateDir string, log io.Writer) func(context.Context) error {
|
||||
return func(ctx context.Context) error {
|
||||
tools, err := dbTools(db)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(ctx, 30*time.Minute)
|
||||
defer cancel()
|
||||
path, err := dbbackup.Backup(ctx, dbbackup.BackupOptions{
|
||||
DatabaseURL: db.URL, Tools: tools, Dir: dir, Label: dbbackup.LabelOffsite,
|
||||
Keep: defaultKeep[dbbackup.LabelOffsite], StateDir: stateDir, Version: resolvedVersion(),
|
||||
ExportServers: exportMinecraftServers, RequireServers: true, Log: log,
|
||||
})
|
||||
if err == nil {
|
||||
fmt.Fprintf(log, "felis offsite: took database bundle %s, which lists the archives just copied\n", filepath.Base(path))
|
||||
}
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
// volumeKind names a PVC the off-site copy reads or restores, for messages,
|
||||
// with the flag that bypasses finding it through the cluster.
|
||||
type volumeKind struct{ what, dirFlag, empty string }
|
||||
@@ -432,6 +539,27 @@ func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) in
|
||||
if st.LastError != "" {
|
||||
fmt.Fprintf(stdout, "last error: %s\n", st.LastError)
|
||||
}
|
||||
if st.KeyMismatch {
|
||||
fmt.Fprintf(stdout, "\nThe last run was refused: the bucket's objects are sealed with another key than this host's (key id %s). No sync copies or prunes anything there until FELIS_OFFSITE_KEY in %s is theirs (sudo felis offsite check-key).\n", st.KeyID, defaultOffsiteEnvFile)
|
||||
return 1
|
||||
}
|
||||
if st.Displaced && st.Writer != nil {
|
||||
fmt.Fprintf(stdout, "\nThe last run was refused: %s took the bucket over (it last wrote it at %s), and this host copies nothing there any more. If that host is a rehearsal machine, take the bucket back: sudo felis offsite take-over -yes\n",
|
||||
st.Writer, st.Writer.At.Local().Format(time.DateTime))
|
||||
return 1
|
||||
}
|
||||
if st.Standby {
|
||||
switch w := st.StandsBy(now); {
|
||||
case w != nil:
|
||||
fmt.Fprintf(stdout, "\nThis host stands by: %s writes the bucket (last at %s). This host was built from its backup, copies nothing into the bucket and, while that host keeps writing, mails no watchdog alert.", w, w.At.Local().Format(time.DateTime))
|
||||
case st.Writer != nil:
|
||||
fmt.Fprintf(stdout, "\nThis host copies nothing into the bucket: %s wrote it, last at %s, and this host was built from its backup.", st.Writer, st.Writer.At.Local().Format(time.DateTime))
|
||||
default:
|
||||
fmt.Fprint(stdout, "\nThis host copies nothing into the bucket: it holds copies this host did not write, and names no host writing it.")
|
||||
}
|
||||
fmt.Fprintln(stdout, " Once this host replaces that one for good: sudo felis offsite take-over -yes")
|
||||
return 1
|
||||
}
|
||||
r := st.Result
|
||||
fmt.Fprintf(stdout, "bucket holds: %d world archives (%s), %d database bundles, newest %s\n",
|
||||
r.RemoteWorlds, offsite.HumanBytes(r.RemoteBytes), r.RemoteDB, orNone(r.NewestDB))
|
||||
@@ -495,10 +623,7 @@ func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "database bundles (%d, newest first):\n", len(bundles))
|
||||
for _, b := range bundles {
|
||||
fmt.Fprintf(stdout, " %s %s\n", b.Key, offsite.HumanBytes(b.Size))
|
||||
}
|
||||
printDBBundles(ctx, env.bucket, env.key, bundles, stdout)
|
||||
var total int64
|
||||
for _, w := range worlds {
|
||||
total += w.Size
|
||||
@@ -535,6 +660,164 @@ func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// printDBBundles lists the database bundles with what each one's database
|
||||
// held, read off the front of each, so a restore can pick one by its contents.
|
||||
func printDBBundles(ctx context.Context, b offsite.Bucket, key []byte, bundles []offsite.Object, stdout io.Writer) {
|
||||
fmt.Fprintf(stdout, "database bundles (%d, newest first; restore one with fetch-db):\n", len(bundles))
|
||||
for _, o := range bundles {
|
||||
m, err := offsite.PeekDB(ctx, b, key, o.Key)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stdout, " %s %s unreadable: %v\n", o.Key, offsite.HumanBytes(o.Size), err)
|
||||
continue
|
||||
}
|
||||
gap := ""
|
||||
if m.ServersError != "" {
|
||||
gap = ", no MinecraftServer objects"
|
||||
}
|
||||
fmt.Fprintf(stdout, " %s %s %s%s\n", o.Key, offsite.HumanBytes(o.Size), m.Counts.String(), gap)
|
||||
}
|
||||
}
|
||||
|
||||
func offsiteCheckKey(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
_, env, err := loadOffsite(*cfgPath, *envFile)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite check-key: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
if err := env.bucket.Check(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite check-key: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return checkKey(ctx, env.bucket, env.key, stdout, stderr)
|
||||
}
|
||||
|
||||
// checkKey is check-key once the bucket is open: 0 when the key fits, 3 when
|
||||
// the bucket's objects are sealed with another one, 1 when it cannot tell.
|
||||
func checkKey(ctx context.Context, b offsite.Bucket, key []byte, stdout, stderr io.Writer) int {
|
||||
fit, err := offsite.CheckKey(ctx, b, key)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite check-key: %v\n", err)
|
||||
if errors.Is(err, offsite.ErrKeyMismatch) {
|
||||
return 3
|
||||
}
|
||||
return 1
|
||||
}
|
||||
id := offsite.KeyID(key)
|
||||
switch fit {
|
||||
case offsite.KeyRecorded:
|
||||
fmt.Fprintf(stdout, "felis offsite check-key: the bucket records key id %s, this key's\n", id)
|
||||
case offsite.KeyOpens:
|
||||
fmt.Fprintf(stdout, "felis offsite check-key: the bucket's newest objects open with this key (key id %s); the next sync records it\n", id)
|
||||
case offsite.KeyUnused:
|
||||
fmt.Fprintf(stdout, "felis offsite check-key: the bucket holds no sealed object yet; the first sync records key id %s\n", id)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func offsiteTakeOver(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "the record `sync` writes; this host's id is kept next to it")
|
||||
yes := fs.Bool("yes", false, "make this host the one that writes the bucket")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
_, env, err := loadOffsite(*cfgPath, *envFile)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
if err := env.bucket.Check(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return takeOver(ctx, env.bucket, env.key, offsite.HostLease(*statusFile), *statusFile, *yes, time.Now(), stdout, stderr)
|
||||
}
|
||||
|
||||
// takeOver is take-over once the bucket is open: without yes it says which
|
||||
// host writes the bucket, 0 for this one (or none yet), 4 for another and 5
|
||||
// for one that took the bucket over from this host; with
|
||||
// yes it records this host as the writer. A key the bucket's objects refuse
|
||||
// is 3, as in check-key: taking over a bucket this host cannot copy into
|
||||
// would only stop the host that can.
|
||||
func takeOver(ctx context.Context, b offsite.Bucket, key []byte, lease offsite.Lease, statusFile string, yes bool, now time.Time, stdout, stderr io.Writer) int {
|
||||
fit, err := offsite.CheckKey(ctx, b, key)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
|
||||
if errors.Is(err, offsite.ErrKeyMismatch) {
|
||||
return 3
|
||||
}
|
||||
return 1
|
||||
}
|
||||
role, w, err := lease.Plan(ctx, b, fit == offsite.KeyUnused)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
switch role {
|
||||
case offsite.RoleWrites:
|
||||
id, _ := lease.ID()
|
||||
fmt.Fprintf(stdout, "felis offsite take-over: this host (id %s) writes the bucket; nothing to take over\n", id)
|
||||
return 0
|
||||
case offsite.RoleClaims:
|
||||
fmt.Fprintln(stdout, "felis offsite take-over: the bucket names no host writing it; this host's next sync records itself")
|
||||
return 0
|
||||
}
|
||||
who := "another host"
|
||||
if w != nil {
|
||||
who = w.String()
|
||||
fmt.Fprintf(stdout, "felis offsite take-over: %s writes the bucket, last at %s (%s ago)\n", w, w.At.Local().Format(time.DateTime), dbbackup.Age(now.Sub(w.At)))
|
||||
} else {
|
||||
fmt.Fprintln(stdout, "felis offsite take-over: the bucket holds copies this host did not write, and names no host writing it")
|
||||
}
|
||||
if !yes {
|
||||
if role == offsite.RoleDisplaced {
|
||||
fmt.Fprintf(stdout, "It took the bucket over from this host: this host copies nothing there any more, and its watchdog mails the owners about it. If that host is a rehearsal machine, take the bucket back:\n sudo felis offsite take-over -yes\n")
|
||||
return 5
|
||||
}
|
||||
fmt.Fprintf(stdout, "This host was built from its backup and copies nothing into the bucket.\n")
|
||||
fmt.Fprintf(stdout, "Taking it over makes this host the one that copies into the bucket and prunes it; %s stops at its next copy and mails its owners. Do it once that host is gone for good, or is a rehearsal machine you are done with:\n sudo felis offsite take-over -yes\n", who)
|
||||
return 4
|
||||
}
|
||||
if _, err := lease.TakeOver(ctx, b, now); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite take-over: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
// The refusal the last sync recorded is over: the watchdog mails again
|
||||
// from now on, and status shows the next run's outcome.
|
||||
if st, err := offsite.ReadStatus(statusFile); err == nil && st != nil && (st.Standby || st.Displaced) {
|
||||
st.Standby, st.Displaced, st.Writer, st.LastError, st.Inherited = false, false, nil, "", false
|
||||
if err := offsite.WriteStatus(statusFile, *st); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite take-over: record status: %v\n", err)
|
||||
}
|
||||
}
|
||||
id, _ := lease.ID()
|
||||
fmt.Fprintf(stdout, "felis offsite take-over: this host (id %s) writes the bucket now; %s stops at its next copy.\nStart the first copy: sudo systemctl start felis-offsite.service\n", id, who)
|
||||
return 0
|
||||
}
|
||||
|
||||
// keyHint explains an object the key cannot open when the bucket records
|
||||
// another key's id, "" otherwise.
|
||||
func keyHint(ctx context.Context, b offsite.Bucket, key []byte, err error) string {
|
||||
if !errors.Is(err, offsite.ErrAuth) {
|
||||
return ""
|
||||
}
|
||||
id, ierr := offsite.BucketKeyID(ctx, b)
|
||||
if ierr != nil || id == "" || id == offsite.KeyID(key) {
|
||||
return ""
|
||||
}
|
||||
return fmt.Sprintf("\n the bucket records key id %s, and this key is %s: set FELIS_OFFSITE_KEY to the key the bucket was written with", id, offsite.KeyID(key))
|
||||
}
|
||||
|
||||
func offsiteFetchDB(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml; on a host with no install yet, give -endpoint and -bucket instead")
|
||||
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||
@@ -577,37 +860,47 @@ func offsiteFetchDB(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) i
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
|
||||
defer cancel()
|
||||
return fetchDB(ctx, env.bucket, env.key, arg, *dir, time.Now(), stdout, stderr)
|
||||
}
|
||||
|
||||
// fetchDB is fetch-db once the bucket is open: arg is a bundle name or latest.
|
||||
func fetchDB(ctx context.Context, b offsite.Bucket, key []byte, arg, dir string, now time.Time, stdout, stderr io.Writer) int {
|
||||
name := arg
|
||||
if name == "latest" {
|
||||
bundles, err := offsite.ListDB(ctx, env.bucket)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||
var err error
|
||||
if name, _, err = offsite.ChooseDB(ctx, b, key); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: %v%s\n", err, keyHint(ctx, b, key, err))
|
||||
return 1
|
||||
}
|
||||
if len(bundles) == 0 {
|
||||
fmt.Fprintln(stderr, "felis offsite fetch-db: the bucket holds no database bundle")
|
||||
return 1
|
||||
}
|
||||
name = bundles[0].Key
|
||||
}
|
||||
if _, _, ok := dbbackup.ParseBundleName(name); !ok {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: %q is not a bundle name (felis-db-<stamp>-<label>.tar); see `felis offsite list`\n", name)
|
||||
return 2
|
||||
}
|
||||
if err := os.MkdirAll(*dir, 0o700); err != nil {
|
||||
if err := os.MkdirAll(dir, 0o700); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
dst := filepath.Join(*dir, name)
|
||||
if err := offsite.FetchObject(ctx, env.bucket, env.key, offsite.DBKey(name), dst, 0o600); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||
dst := filepath.Join(dir, name)
|
||||
if err := offsite.FetchObject(ctx, b, key, offsite.DBKey(name), dst, 0o600); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: %v%s\n", err, keyHint(ctx, b, key, err))
|
||||
return 1
|
||||
}
|
||||
if _, err := dbbackup.Verify(dst); err != nil {
|
||||
m, err := dbbackup.Verify(dst)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-db: fetched %s but it does not verify: %v\n", dst, err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis offsite fetch-db: wrote %s (verified)\n", dst)
|
||||
fmt.Fprintf(stdout, " taken %s (%s, %s ago)\n felis %s, schema %d\n holds %s\n",
|
||||
m.CreatedAt.Format(time.RFC3339), m.Label, dbbackup.Age(now.Sub(m.CreatedAt)),
|
||||
orUnknown(m.FelisVersion), m.SchemaVersion, m.Counts.String())
|
||||
if m.Counts.Fresh() {
|
||||
fmt.Fprintln(stdout, " This database holds no servers and at most one account, like a new install's. Check it is the state to restore before `felis db restore`.")
|
||||
}
|
||||
if m.ServersError != "" {
|
||||
fmt.Fprintf(stdout, " This bundle lacks the MinecraftServer objects (%s): `felis db restore` brings back the database, and the servers come from k8s/minecraftservers.json in the newest bundle `felis offsite list` shows without that gap.\n", m.ServersError)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
|
||||
@@ -1,14 +1,24 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/imagepush"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
)
|
||||
@@ -120,3 +130,646 @@ func TestRegistryGoneMarksNotFound(t *testing.T) {
|
||||
t.Fatalf("503 = %v, want it kept an ordinary failure", err)
|
||||
}
|
||||
}
|
||||
|
||||
// mapBucket is an in-memory offsite.Bucket.
|
||||
type mapBucket map[string][]byte
|
||||
|
||||
func (b mapBucket) Put(_ context.Context, key string, r io.Reader, _ int64) error {
|
||||
data, err := io.ReadAll(r)
|
||||
b[key] = data
|
||||
return err
|
||||
}
|
||||
|
||||
func (b mapBucket) Get(_ context.Context, key string) (io.ReadCloser, error) {
|
||||
data, ok := b[key]
|
||||
if !ok {
|
||||
return nil, offsite.ErrNotFound
|
||||
}
|
||||
return io.NopCloser(bytes.NewReader(data)), nil
|
||||
}
|
||||
|
||||
func (b mapBucket) List(_ context.Context, prefix string) ([]offsite.Object, error) {
|
||||
var out []offsite.Object
|
||||
for k, v := range b {
|
||||
if strings.HasPrefix(k, prefix) {
|
||||
out = append(out, offsite.Object{Key: k, Size: int64(len(v))})
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (b mapBucket) Remove(_ context.Context, key string) error {
|
||||
delete(b, key)
|
||||
return nil
|
||||
}
|
||||
|
||||
var fetchT0 = time.Date(2026, 9, 20, 3, 30, 0, 0, time.UTC)
|
||||
|
||||
// putBundle seals a bundle that verifies, taken daysAgo days before fetchT0,
|
||||
// into b and returns its name.
|
||||
func putBundle(t *testing.T, b mapBucket, key []byte, daysAgo int, counts *dbbackup.Counts) string {
|
||||
t.Helper()
|
||||
return putBundleWith(t, b, key, daysAgo, counts, "")
|
||||
}
|
||||
|
||||
// putBundleWith is putBundle for a bundle whose server export failed with
|
||||
// serversError, when that is not empty.
|
||||
func putBundleWith(t *testing.T, b mapBucket, key []byte, daysAgo int, counts *dbbackup.Counts, serversError string) string {
|
||||
t.Helper()
|
||||
created := fetchT0.AddDate(0, 0, -daysAgo)
|
||||
dump := []byte("PGDMP " + created.String())
|
||||
sum := sha256.Sum256(dump)
|
||||
manifest, err := json.Marshal(dbbackup.Manifest{
|
||||
Format: 1, CreatedAt: created, Label: dbbackup.LabelDaily, FelisVersion: "v1.2.3", SchemaVersion: 21, Counts: counts, ServersError: serversError,
|
||||
Files: []dbbackup.ManifestEntry{{Name: "db.dump", Size: int64(len(dump)), SHA256: hex.EncodeToString(sum[:]), Mode: 0o600}},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var plain bytes.Buffer
|
||||
tw := tar.NewWriter(&plain)
|
||||
for _, f := range []struct {
|
||||
name string
|
||||
data []byte
|
||||
}{{"MANIFEST.json", manifest}, {"db.dump", dump}} {
|
||||
if err := tw.WriteHeader(&tar.Header{Name: f.name, Mode: 0o600, Size: int64(len(f.data)), Typeflag: tar.TypeReg}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := tw.Write(f.data); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := tw.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var sealed bytes.Buffer
|
||||
if err := offsite.Encrypt(&sealed, &plain, key); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name := dbbackup.BundleName(created, dbbackup.LabelDaily)
|
||||
b[offsite.DBKey(name)] = sealed.Bytes()
|
||||
return name
|
||||
}
|
||||
|
||||
func TestOffsiteFetchDB(t *testing.T) {
|
||||
rawKey, _ := offsite.NewKey()
|
||||
key, _ := offsite.ParseKey(rawKey)
|
||||
now := fetchT0.Add(2 * time.Hour)
|
||||
fetch := func(b mapBucket, arg string) (dir string, code int, stdout, stderr string) {
|
||||
dir = t.TempDir()
|
||||
var out, errb bytes.Buffer
|
||||
code = fetchDB(context.Background(), b, key, arg, dir, now, &out, &errb)
|
||||
return dir, code, out.String(), errb.String()
|
||||
}
|
||||
fetched := func(t *testing.T, dir string) []string {
|
||||
t.Helper()
|
||||
var names []string
|
||||
entries, _ := os.ReadDir(dir)
|
||||
for _, e := range entries {
|
||||
names = append(names, e.Name())
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
t.Run("latest skips a rebuilt host's empty bundle", func(t *testing.T) {
|
||||
b := mapBucket{}
|
||||
full := putBundle(t, b, key, 3, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
empty := putBundle(t, b, key, 0, &dbbackup.Counts{})
|
||||
dir, code, out, errb := fetch(b, "latest")
|
||||
if code != 1 || !strings.Contains(errb, empty) || !strings.Contains(errb, full+" (5 accounts, 3 servers)") {
|
||||
t.Fatalf("exit %d, stdout %q, stderr %q; want a refusal naming %s", code, out, errb, full)
|
||||
}
|
||||
if got := fetched(t, dir); len(got) != 0 {
|
||||
t.Errorf("a refused fetch wrote %v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("latest takes the newest bundle and says what it holds", func(t *testing.T) {
|
||||
b := mapBucket{}
|
||||
putBundle(t, b, key, 3, &dbbackup.Counts{Users: 5, Servers: 2})
|
||||
newest := putBundle(t, b, key, 1, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
dir, code, out, errb := fetch(b, "latest")
|
||||
if code != 0 {
|
||||
t.Fatalf("exit %d: %s", code, errb)
|
||||
}
|
||||
if got := fetched(t, dir); !slices.Equal(got, []string{newest}) {
|
||||
t.Errorf("wrote %v, want %s", got, newest)
|
||||
}
|
||||
for _, want := range []string{
|
||||
"wrote " + filepath.Join(dir, newest) + " (verified)",
|
||||
"taken 2026-09-19T03:30:00Z (daily, 26h0m ago)",
|
||||
"felis v1.2.3, schema 21",
|
||||
"holds 5 accounts, 3 servers",
|
||||
} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("stdout lacks %q:\n%s", want, out)
|
||||
}
|
||||
}
|
||||
if strings.Contains(out, "new install") {
|
||||
t.Errorf("a bundle with servers flagged as a new install's:\n%s", out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an empty bundle named outright is fetched with a warning", func(t *testing.T) {
|
||||
b := mapBucket{}
|
||||
putBundle(t, b, key, 3, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
empty := putBundle(t, b, key, 0, &dbbackup.Counts{Users: 1})
|
||||
dir, code, out, errb := fetch(b, empty)
|
||||
if code != 0 || !slices.Equal(fetched(t, dir), []string{empty}) {
|
||||
t.Fatalf("exit %d, wrote %v: %s", code, fetched(t, dir), errb)
|
||||
}
|
||||
if !strings.Contains(out, "holds 1 account, 0 servers") || !strings.Contains(out, "like a new install's") {
|
||||
t.Errorf("stdout = %s", out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a bundle without the servers says where they come from", func(t *testing.T) {
|
||||
b := mapBucket{}
|
||||
gapped := putBundleWith(t, b, key, 1, &dbbackup.Counts{Users: 5, Servers: 3}, "connection refused (tried 3 times)")
|
||||
_, code, out, errb := fetch(b, gapped)
|
||||
if code != 0 || !strings.Contains(out, "This bundle lacks the MinecraftServer objects (connection refused (tried 3 times))") ||
|
||||
!strings.Contains(out, "k8s/minecraftservers.json in the newest bundle `felis offsite list` shows without that gap") {
|
||||
t.Errorf("exit %d, stdout %q, stderr %q", code, out, errb)
|
||||
}
|
||||
whole := putBundle(t, b, key, 0, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
if _, _, out, _ := fetch(b, whole); strings.Contains(out, "lacks the MinecraftServer objects") {
|
||||
t.Errorf("a whole bundle flagged:\n%s", out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a bundle from before counts says so", func(t *testing.T) {
|
||||
b := mapBucket{}
|
||||
old := putBundle(t, b, key, 0, nil)
|
||||
_, code, out, errb := fetch(b, "latest")
|
||||
if code != 0 || !strings.Contains(out, old) || !strings.Contains(out, "holds not recorded") || strings.Contains(out, "new install") {
|
||||
t.Errorf("exit %d, stdout %q, stderr %q", code, out, errb)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestOffsiteCheckKey(t *testing.T) {
|
||||
newKey := func() []byte {
|
||||
raw, _ := offsite.NewKey()
|
||||
k, _ := offsite.ParseKey(raw)
|
||||
return k
|
||||
}
|
||||
key, other := newKey(), newKey()
|
||||
marked := func(k []byte) mapBucket { return mapBucket{"felis-key-id": []byte(offsite.KeyID(k) + "\n")} }
|
||||
unmarked := func(k []byte) mapBucket {
|
||||
b := mapBucket{}
|
||||
putBundle(t, b, k, 0, nil)
|
||||
return b
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
bucket mapBucket
|
||||
code int
|
||||
says []string
|
||||
}{
|
||||
{"the recorded key", marked(key), 0, []string{"records key id " + offsite.KeyID(key)}},
|
||||
{"an unmarked bucket the key opens", unmarked(key), 0, []string{"open with this key", "the next sync records it"}},
|
||||
{"an empty bucket", mapBucket{}, 0, []string{"no sealed object yet", "records key id " + offsite.KeyID(key)}},
|
||||
{"another recorded key", marked(other), 3, []string{offsite.KeyID(other), offsite.KeyID(key), "FELIS_OFFSITE_KEY"}},
|
||||
{"an unmarked bucket under another key", unmarked(other), 3, []string{"opens none of db/felis-db-"}},
|
||||
{"a marker Felis did not write", mapBucket{"felis-key-id": []byte("hello")}, 1, []string{"not a key id"}},
|
||||
} {
|
||||
t.Run(tc.what, func(t *testing.T) {
|
||||
before := len(tc.bucket)
|
||||
var out, errb bytes.Buffer
|
||||
code := checkKey(context.Background(), tc.bucket, key, &out, &errb)
|
||||
if code != tc.code {
|
||||
t.Fatalf("exit %d, want %d; stdout %q, stderr %q", code, tc.code, out.String(), errb.String())
|
||||
}
|
||||
said := out.String() + errb.String()
|
||||
for _, s := range tc.says {
|
||||
if !strings.Contains(said, s) {
|
||||
t.Errorf("output lacks %q: %s", s, said)
|
||||
}
|
||||
}
|
||||
if len(tc.bucket) != before {
|
||||
t.Errorf("check-key wrote to the bucket: %d objects, had %d", len(tc.bucket), before)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// fetch-db names both ids when the bucket records another key.
|
||||
b := marked(other)
|
||||
name := putBundle(t, b, other, 0, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
for _, arg := range []string{"latest", name} {
|
||||
var out, errb bytes.Buffer
|
||||
if code := fetchDB(context.Background(), b, key, arg, t.TempDir(), fetchT0, &out, &errb); code != 1 ||
|
||||
!strings.Contains(errb.String(), "the bucket records key id "+offsite.KeyID(other)+", and this key is "+offsite.KeyID(key)) {
|
||||
t.Errorf("fetch-db %s under another key: exit %d, stderr %q", arg, code, errb.String())
|
||||
}
|
||||
}
|
||||
// A bundle the bucket lacks is not the key's fault.
|
||||
var missOut, missErr bytes.Buffer
|
||||
if code := fetchDB(context.Background(), b, key, "felis-db-20200101T000000Z-daily.tar", t.TempDir(), fetchT0, &missOut, &missErr); code != 1 || strings.Contains(missErr.String(), "records key id") {
|
||||
t.Errorf("missing bundle: exit %d, stderr %q", code, missErr.String())
|
||||
}
|
||||
// A bundle damaged under the recorded key gets no such hint.
|
||||
b = marked(key)
|
||||
name = putBundle(t, b, key, 0, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
b[offsite.DBKey(name)][60] ^= 1
|
||||
var out, errb bytes.Buffer
|
||||
if code := fetchDB(context.Background(), b, key, name, t.TempDir(), fetchT0, &out, &errb); code != 1 || strings.Contains(errb.String(), "records key id") {
|
||||
t.Errorf("damaged bundle: exit %d, stderr %q", code, errb.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecordRun(t *testing.T) {
|
||||
t0 := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
|
||||
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: t0}
|
||||
for _, tc := range []struct {
|
||||
err error
|
||||
mismatch, standby, displaced, success bool
|
||||
}{
|
||||
{nil, false, false, false, true},
|
||||
{errors.New("list worlds/ in the bucket: connection reset"), false, false, false, false},
|
||||
{fmt.Errorf("%w: the bucket records key id 0123456789abcdef", offsite.ErrKeyMismatch), true, false, false, false},
|
||||
{&offsite.WriterError{Kind: offsite.ErrStandby, Writer: w}, false, true, false, false},
|
||||
{&offsite.WriterError{Kind: offsite.ErrDisplaced, Writer: w}, false, false, true, false},
|
||||
} {
|
||||
st := offsite.Status{LastAttempt: t0}
|
||||
recordRun(&st, offsite.Result{RemoteDB: 2}, tc.err, offsite.Lease{})
|
||||
if st.KeyMismatch != tc.mismatch || st.Standby != tc.standby || st.Displaced != tc.displaced ||
|
||||
(st.Writer != nil) != (tc.standby || tc.displaced) || st.LastSuccess.Equal(t0) != tc.success || st.Result.RemoteDB != 2 {
|
||||
t.Errorf("err %v: status %+v", tc.err, st)
|
||||
}
|
||||
}
|
||||
|
||||
// A host that copied before writers were recorded keeps that claim over
|
||||
// failed runs until it has an id.
|
||||
l := offsite.Lease{IDFile: filepath.Join(t.TempDir(), offsite.HostIDFile), Inherited: true}
|
||||
st := offsite.Status{}
|
||||
recordRun(&st, offsite.Result{}, errors.New("cannot reach bucket"), l)
|
||||
if !st.Inherited {
|
||||
t.Error("a failed run on a host with no id dropped the older release's claim")
|
||||
}
|
||||
writeTestFile(t, l.IDFile, "aaaaaaaaaaaaaaaa\n", 0o600)
|
||||
st = offsite.Status{Inherited: true}
|
||||
recordRun(&st, offsite.Result{}, nil, l)
|
||||
if st.Inherited {
|
||||
t.Error("a host with an id still carries the older release's claim")
|
||||
}
|
||||
}
|
||||
|
||||
// A refused run has copied nothing and listed nothing: its zero counts would
|
||||
// tell the journal the bucket is empty. A pass that ran and failed a step
|
||||
// shows what it did get to.
|
||||
func TestReportRun(t *testing.T) {
|
||||
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)}
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
res offsite.Result
|
||||
err error
|
||||
code int
|
||||
counts bool
|
||||
}{
|
||||
{"a pass", offsite.Result{DBUploaded: 1, RemoteDB: 3}, nil, 0, true},
|
||||
{"a pass with a failed step", offsite.Result{RemoteDB: 3, Errors: []string{"copy db/x: timeout"}}, errors.New("1 of this run's steps failed; first: copy db/x: timeout"), 1, true},
|
||||
{"standing by", offsite.Result{}, &offsite.WriterError{Kind: offsite.ErrStandby, Writer: w}, 1, false},
|
||||
{"another key", offsite.Result{}, fmt.Errorf("%w: the bucket records key id 1111111111111111", offsite.ErrKeyMismatch), 1, false},
|
||||
{"no bucket", offsite.Result{}, errors.New("bucket: access denied"), 1, false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
var out, errOut bytes.Buffer
|
||||
code := reportRun(tc.res, tc.err, &out, &errOut)
|
||||
if code != tc.code {
|
||||
t.Errorf("exit %d, want %d", code, tc.code)
|
||||
}
|
||||
if got := strings.Contains(out.String(), "bucket holds 0 worlds (0 B), 3 bundles"); got != tc.counts {
|
||||
t.Errorf("counts shown = %v, want %v: %q", got, tc.counts, out.String())
|
||||
}
|
||||
if !tc.counts && out.Len() > 0 {
|
||||
t.Errorf("a refused run printed %q", out.String())
|
||||
}
|
||||
if tc.err != nil && !strings.Contains(errOut.String(), "felis offsite sync: "+tc.err.Error()) {
|
||||
t.Errorf("stderr %q lacks the error", errOut.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsiteTakeOver(t *testing.T) {
|
||||
newKey := func() []byte {
|
||||
raw, _ := offsite.NewKey()
|
||||
k, _ := offsite.ParseKey(raw)
|
||||
return k
|
||||
}
|
||||
key, other := newKey(), newKey()
|
||||
const mine, theirs = "aaaaaaaaaaaaaaaa", "bbbbbbbbbbbbbbbb"
|
||||
t0 := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
|
||||
sealed := func(k []byte, writer string) mapBucket {
|
||||
b := mapBucket{}
|
||||
putBundle(t, b, k, 0, nil)
|
||||
b["felis-key-id"] = []byte(offsite.KeyID(k) + "\n")
|
||||
if writer != "" {
|
||||
raw, _ := json.Marshal(offsite.Writer{HostID: writer, Host: "prod-1", At: t0.Add(-20 * time.Minute)})
|
||||
b["felis-writer"] = raw
|
||||
}
|
||||
return b
|
||||
}
|
||||
lease := func(id string) offsite.Lease {
|
||||
l := offsite.Lease{IDFile: filepath.Join(t.TempDir(), offsite.HostIDFile), Host: "spare-1"}
|
||||
if id != "" {
|
||||
writeTestFile(t, l.IDFile, id+"\n", 0o600)
|
||||
}
|
||||
return l
|
||||
}
|
||||
run := func(b mapBucket, l offsite.Lease, statusFile string, yes bool) (int, string, string) {
|
||||
t.Helper()
|
||||
var out, errb bytes.Buffer
|
||||
code := takeOver(context.Background(), b, key, l, statusFile, yes, t0, &out, &errb)
|
||||
return code, out.String(), errb.String()
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
bucket mapBucket
|
||||
id string
|
||||
code int
|
||||
says []string
|
||||
}{
|
||||
{"this host writes the bucket", sealed(key, mine), mine, 0, []string{"this host (id " + mine + ") writes the bucket; nothing to take over"}},
|
||||
{"an empty bucket", mapBucket{}, "", 0, []string{"names no host writing it; this host's next sync records itself"}},
|
||||
{"a host built from the writer's backup", sealed(key, theirs), "", 4, []string{"host prod-1 (id " + theirs + ") writes the bucket, last at", "(20m ago)", "built from its backup", "sudo felis offsite take-over -yes"}},
|
||||
{"another host's copies, no writer named", sealed(key, ""), "", 4, []string{"holds copies this host did not write, and names no host writing it", "take-over -yes"}},
|
||||
{"a host another one took over from", sealed(key, theirs), mine, 5, []string{"took the bucket over from this host"}},
|
||||
{"a key the bucket refuses", sealed(other, theirs), "", 3, []string{"sealed with another key"}},
|
||||
} {
|
||||
t.Run(tc.what, func(t *testing.T) {
|
||||
before := string(tc.bucket["felis-writer"])
|
||||
l := lease(tc.id)
|
||||
code, out, errb := run(tc.bucket, l, filepath.Join(t.TempDir(), "status.json"), false)
|
||||
if code != tc.code {
|
||||
t.Fatalf("exit %d, want %d\n%s%s", code, tc.code, out, errb)
|
||||
}
|
||||
for _, s := range tc.says {
|
||||
if !strings.Contains(out+errb, s) {
|
||||
t.Errorf("output lacks %q:\n%s%s", s, out, errb)
|
||||
}
|
||||
}
|
||||
if string(tc.bucket["felis-writer"]) != before {
|
||||
t.Error("take-over without -yes wrote the bucket's writer")
|
||||
}
|
||||
if tc.id == "" {
|
||||
if _, err := os.Stat(l.IDFile); !errors.Is(err, os.ErrNotExist) {
|
||||
t.Errorf("take-over without -yes made this host an id: %v", err)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// -yes on a standby host: the bucket names it, and the refusal the last
|
||||
// sync recorded is cleared, so the watchdog mails again at once.
|
||||
b := sealed(key, theirs)
|
||||
l := lease("")
|
||||
statusFile := filepath.Join(t.TempDir(), "status.json")
|
||||
lastSuccess := t0.Add(-48 * time.Hour)
|
||||
if err := offsite.WriteStatus(statusFile, offsite.Status{
|
||||
LastAttempt: t0.Add(-time.Hour), LastSuccess: lastSuccess, LastError: "offsite: another host writes this bucket", Format: offsite.StatusFormat,
|
||||
Standby: true, Writer: &offsite.Writer{HostID: theirs, Host: "prod-1", At: t0}, Inherited: true,
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
code, out, errb := run(b, l, statusFile, true)
|
||||
id, _ := l.ID()
|
||||
if code != 0 || id == "" || !strings.Contains(out, "this host (id "+id+") writes the bucket now; host prod-1 (id "+theirs+") stops at its next copy") || !strings.Contains(out, "systemctl start felis-offsite.service") {
|
||||
t.Fatalf("take-over -yes: exit %d, id %q\n%s%s", code, id, out, errb)
|
||||
}
|
||||
if w, err := offsite.BucketWriter(context.Background(), b); err != nil || w.HostID != id || w.Host != "spare-1" || !w.At.Equal(t0) {
|
||||
t.Errorf("writer after take-over -yes = %+v, %v", w, err)
|
||||
}
|
||||
st, err := offsite.ReadStatus(statusFile)
|
||||
if err != nil || st.Standby || st.Writer != nil || st.LastError != "" || st.Inherited || !st.LastSuccess.Equal(lastSuccess) {
|
||||
t.Errorf("status after take-over -yes = %+v, %v; want the refusal cleared and the last success kept", st, err)
|
||||
}
|
||||
|
||||
// -yes with a key the bucket refuses writes nothing.
|
||||
b = sealed(other, theirs)
|
||||
before := string(b["felis-writer"])
|
||||
if code, _, _ := run(b, lease(""), filepath.Join(t.TempDir(), "status.json"), true); code != 3 || string(b["felis-writer"]) != before {
|
||||
t.Errorf("take-over -yes under another key: exit %d, writer %s", code, b["felis-writer"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsiteStatusSaysWhoWritesTheBucket(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cfg := filepath.Join(dir, "felis.toml")
|
||||
writeTestFile(t, cfg, installerTOML("example.com", "127.0.0.1")+"\n[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis-backups\"\n", 0o600)
|
||||
statusFile := filepath.Join(dir, "status.json")
|
||||
now := time.Now()
|
||||
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: now.Add(-30 * time.Minute)}
|
||||
stale := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: now.Add(-offsite.WriterLive - time.Hour)}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
st offsite.Status
|
||||
says []string
|
||||
not string
|
||||
}{
|
||||
{"standing by for a live writer", offsite.Status{Standby: true, Writer: w}, []string{"This host stands by: host prod-1 (id bbbbbbbbbbbbbbbb) writes the bucket", "mails no watchdog alert", "take-over -yes"}, "wrote it, last at"},
|
||||
{"standing by for a writer gone quiet", offsite.Status{Standby: true, Writer: stale}, []string{"copies nothing into the bucket: host prod-1 (id bbbbbbbbbbbbbbbb) wrote it, last at", "take-over -yes"}, "mails no watchdog alert"},
|
||||
{"standing by, no writer named", offsite.Status{Standby: true}, []string{"names no host writing it", "take-over -yes"}, "stands by:"},
|
||||
{"displaced", offsite.Status{Displaced: true, Writer: w}, []string{"host prod-1 (id bbbbbbbbbbbbbbbb) took the bucket over", "rehearsal machine", "take-over -yes"}, "stands by"},
|
||||
} {
|
||||
t.Run(tc.what, func(t *testing.T) {
|
||||
tc.st.LastAttempt, tc.st.LastSuccess, tc.st.LastError = now.Add(-time.Minute), now.Add(-time.Hour), "offsite: another host writes this bucket"
|
||||
if err := offsite.WriteStatus(statusFile, tc.st); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var out, errb bytes.Buffer
|
||||
code := cmdOffsite([]string{"status", "-config", cfg, "-status-file", statusFile}, &out, &errb)
|
||||
if code != 1 || strings.Contains(out.String(), "bucket holds:") || strings.Contains(out.String(), tc.not) {
|
||||
t.Errorf("exit %d\n%s%s", code, out.String(), errb.String())
|
||||
}
|
||||
for _, s := range tc.says {
|
||||
if !strings.Contains(out.String(), s) {
|
||||
t.Errorf("output lacks %q:\n%s", s, out.String())
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsiteStatusSaysTheKeyWasRefused(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cfg := filepath.Join(dir, "felis.toml")
|
||||
writeTestFile(t, cfg, installerTOML("example.com", "127.0.0.1")+"\n[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis-backups\"\n", 0o600)
|
||||
statusFile := filepath.Join(dir, "status.json")
|
||||
st := offsite.Status{LastAttempt: time.Now().Add(-time.Minute), LastSuccess: time.Now().Add(-time.Hour), KeyID: "0123456789abcdef"}
|
||||
status := func() (int, string) {
|
||||
t.Helper()
|
||||
if err := offsite.WriteStatus(statusFile, st); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var out, errb bytes.Buffer
|
||||
code := cmdOffsite([]string{"status", "-config", cfg, "-status-file", statusFile}, &out, &errb)
|
||||
return code, out.String() + errb.String()
|
||||
}
|
||||
if code, out := status(); code != 0 || strings.Contains(out, "refused") {
|
||||
t.Fatalf("a recent success: exit %d\n%s", code, out)
|
||||
}
|
||||
st.LastError, st.KeyMismatch = "offsite: the bucket's objects are sealed with another key", true
|
||||
code, out := status()
|
||||
if code != 1 || !strings.Contains(out, "The last run was refused") || !strings.Contains(out, "key id 0123456789abcdef") || strings.Contains(out, "bucket holds:") {
|
||||
t.Fatalf("a refused run: exit %d\n%s", code, out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrintDBBundlesSaysWhatEachHolds(t *testing.T) {
|
||||
rawKey, _ := offsite.NewKey()
|
||||
key, _ := offsite.ParseKey(rawKey)
|
||||
otherRaw, _ := offsite.NewKey()
|
||||
other, _ := offsite.ParseKey(otherRaw)
|
||||
b := mapBucket{}
|
||||
gapped := putBundleWith(t, b, key, 4, &dbbackup.Counts{Users: 5, Servers: 3}, "connection refused")
|
||||
old := putBundle(t, b, key, 3, nil)
|
||||
full := putBundle(t, b, key, 2, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
sealedElsewhere := putBundle(t, b, other, 1, &dbbackup.Counts{Users: 5, Servers: 3})
|
||||
empty := putBundle(t, b, key, 0, &dbbackup.Counts{})
|
||||
bundles, err := offsite.ListDB(context.Background(), b)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var out bytes.Buffer
|
||||
printDBBundles(context.Background(), b, key, bundles, &out)
|
||||
lines := strings.Split(strings.TrimSpace(out.String()), "\n")
|
||||
want := []struct{ name, holds string }{
|
||||
{empty, "0 accounts, 0 servers"},
|
||||
{sealedElsewhere, "unreadable: offsite: object does not decrypt with this key"},
|
||||
{full, "5 accounts, 3 servers"},
|
||||
{old, "not recorded"},
|
||||
{gapped, "5 accounts, 3 servers, no MinecraftServer objects"},
|
||||
}
|
||||
if len(lines) != len(want)+1 || !strings.HasPrefix(lines[0], "database bundles (5, newest first") {
|
||||
t.Fatalf("output:\n%s", out.String())
|
||||
}
|
||||
for i, w := range want {
|
||||
if l := lines[i+1]; !strings.HasPrefix(l, " "+w.name+" ") || !strings.Contains(l, w.holds) || strings.Contains(l, "no MinecraftServer") != (w.name == gapped) {
|
||||
t.Errorf("line %d = %q, want %s with %q", i+1, l, w.name, w.holds)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestoredHostKeepsStandingBy walks the status file across runs: a host
|
||||
// restored from the writer's backup stands by on its first run and on every
|
||||
// run after it, a host an older release left copying claims the bucket once,
|
||||
// and a failed first run after the upgrade keeps that claim.
|
||||
func TestRestoredHostKeepsStandingBy(t *testing.T) {
|
||||
rawKey, _ := offsite.NewKey()
|
||||
key, _ := offsite.ParseKey(rawKey)
|
||||
cfg := config.OffsiteConfig{Endpoint: "https://s3.example.com", Bucket: "felis-backups"}
|
||||
t0 := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
|
||||
standby := &offsite.WriterError{Kind: offsite.ErrStandby, Writer: &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: t0}}
|
||||
pass := func(statusFile string, at time.Time, err error) offsite.Lease {
|
||||
t.Helper()
|
||||
st, lease := startRun(cfg, key, statusFile, at)
|
||||
recordRun(&st, offsite.Result{}, err, lease)
|
||||
if werr := offsite.WriteStatus(statusFile, st); werr != nil {
|
||||
t.Fatal(werr)
|
||||
}
|
||||
return lease
|
||||
}
|
||||
|
||||
restored := filepath.Join(t.TempDir(), "status.json")
|
||||
for i := range 3 {
|
||||
if l := pass(restored, t0.Add(time.Duration(i)*time.Hour), standby); l.Inherited {
|
||||
t.Fatalf("run %d of a restored host claims the bucket", i+1)
|
||||
}
|
||||
}
|
||||
if st, _ := offsite.ReadStatus(restored); !st.Standby || st.Format != offsite.StatusFormat || st.KeyID != offsite.KeyID(key) || st.Bucket != "felis-backups" {
|
||||
t.Errorf("restored host's status = %+v", st)
|
||||
}
|
||||
|
||||
upgraded := filepath.Join(t.TempDir(), "status.json")
|
||||
if err := offsite.WriteStatus(upgraded, offsite.Status{LastAttempt: t0.Add(-time.Hour), LastSuccess: t0.Add(-time.Hour)}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if l := pass(upgraded, t0, errors.New("cannot reach bucket")); !l.Inherited {
|
||||
t.Fatal("the first run after the upgrade does not claim the bucket")
|
||||
}
|
||||
l := pass(upgraded, t0.Add(time.Hour), nil)
|
||||
if !l.Inherited {
|
||||
t.Fatal("a failed first run after the upgrade lost the claim")
|
||||
}
|
||||
if st, _ := offsite.ReadStatus(upgraded); !st.LastSuccess.Equal(t0.Add(time.Hour)) {
|
||||
t.Errorf("upgraded host's status = %+v", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestOffsiteSyncerSnapshotsAndSweeps: the pass `offsite sync` runs takes its
|
||||
// snapshot the way `felis db backup` does, in the database pod, into the
|
||||
// bundle directory, keeping the newest one there; and it sweeps unrecorded
|
||||
// world objects only past the longest retention [archive] gives any backup.
|
||||
func TestOffsiteSyncerSnapshotsAndSweeps(t *testing.T) {
|
||||
dir := newPodRig(t)
|
||||
bundles := filepath.Join(dir, "bundles")
|
||||
env := &offsiteEnv{cfg: config.OffsiteConfig{DBKeep: 5}}
|
||||
var log bytes.Buffer
|
||||
cfg := &config.Config{Database: podDB, Archive: config.ArchiveConfig{Retention: "120d"}}
|
||||
s := offsiteSyncer(cfg, env, offsiteSources{dbDir: bundles}, "/archives", "/uploads", nil, &log)
|
||||
if s.DBDir != bundles || s.DBKeep != 5 || s.ArchiveDir != "/archives" || s.UploadsDir != "/uploads" {
|
||||
t.Fatalf("syncer = %+v", s)
|
||||
}
|
||||
if s.OrphanAfter != 120*24*time.Hour {
|
||||
t.Errorf("OrphanAfter = %s, want the 120d retention", s.OrphanAfter)
|
||||
}
|
||||
if s.Snapshot == nil {
|
||||
t.Fatal("the pass takes no snapshot after copying archives")
|
||||
}
|
||||
for i := 0; i < 2; i++ {
|
||||
if err := s.Snapshot(context.Background()); err != nil {
|
||||
t.Fatalf("snapshot %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
got, err := dbbackup.List(bundles)
|
||||
if err != nil || len(got) != 1 || got[0].Label != dbbackup.LabelOffsite {
|
||||
t.Fatalf("bundle directory = %+v, %v; want the newest offsite bundle alone", got, err)
|
||||
}
|
||||
if _, err := dbbackupVerify(got[0].Path); err != nil {
|
||||
t.Fatalf("the snapshot does not verify: %v", err)
|
||||
}
|
||||
// The MinecraftServer objects are exported alongside, as in the daily bundle.
|
||||
argv, _ := os.ReadFile(filepath.Join(dir, "k3s.args"))
|
||||
if ran := string(argv); !strings.Contains(ran, podExecPrefix+"pg_dump --format=custom") {
|
||||
t.Errorf("k3s ran %q, want pg_dump in the pod", ran)
|
||||
}
|
||||
if !strings.Contains(string(bundleServers(t, got[0].Path)), `"name": "lobby"`) {
|
||||
t.Errorf("the snapshot holds no MinecraftServer objects")
|
||||
}
|
||||
if !strings.Contains(log.String(), "took database bundle "+got[0].Name) {
|
||||
t.Errorf("the snapshot is not logged:\n%s", log.String())
|
||||
}
|
||||
// It becomes the newest bundle in the bucket, so with the cluster away it
|
||||
// fails, leaves no bundle, and the next pass tries again.
|
||||
noServerExportWait(t)
|
||||
writeTestFile(t, filepath.Join(dir, "servers_fail"), "99", 0o600)
|
||||
if err := s.Snapshot(context.Background()); err == nil || !strings.Contains(err.Error(), "export the MinecraftServer objects") {
|
||||
t.Errorf("snapshot with the cluster away: %v, want a failure", err)
|
||||
}
|
||||
if again, _ := dbbackup.List(bundles); len(again) != 1 || again[0].Name != got[0].Name {
|
||||
t.Errorf("bundle directory after the failed snapshot = %+v, want %s alone", again, got[0].Name)
|
||||
}
|
||||
|
||||
for _, c := range []struct {
|
||||
archive config.ArchiveConfig
|
||||
want time.Duration
|
||||
}{
|
||||
{config.ArchiveConfig{}, 90 * 24 * time.Hour},
|
||||
{config.ArchiveConfig{ScheduledRetention: "200d"}, 200 * 24 * time.Hour},
|
||||
{config.ArchiveConfig{ManualRetention: "150d", Retention: "30d", ScheduledRetention: "60d"}, 150 * 24 * time.Hour},
|
||||
} {
|
||||
s := offsiteSyncer(&config.Config{Database: podDB, Archive: c.archive}, env, offsiteSources{dbDir: bundles}, "", "", nil, io.Discard)
|
||||
if s.OrphanAfter != c.want {
|
||||
t.Errorf("%+v: OrphanAfter = %s, want %s", c.archive, s.OrphanAfter, c.want)
|
||||
}
|
||||
}
|
||||
log.Reset()
|
||||
s = offsiteSyncer(&config.Config{Database: podDB, Archive: config.ArchiveConfig{Retention: "soon"}}, env, offsiteSources{}, "", "", nil, &log)
|
||||
if s.OrphanAfter != 0 || !strings.Contains(log.String(), "world objects no backup records are kept") {
|
||||
t.Errorf("a retention that does not parse: OrphanAfter %s, log %q; want no sweep, said", s.OrphanAfter, log.String())
|
||||
}
|
||||
if s.Snapshot != nil {
|
||||
t.Error("a pass that copies no bundles takes a snapshot")
|
||||
}
|
||||
}
|
||||
@@ -120,7 +120,10 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
Jobs: mgr.GetAPIReader(),
|
||||
// Uncached too: RCON Secrets are read by name, so the Role grants
|
||||
// secrets:get without the list/watch an informer would need.
|
||||
Secrets: mgr.GetAPIReader(),
|
||||
Secrets: mgr.GetAPIReader(),
|
||||
// And pods: pod-0 is read by name, so the Role grants pods:get and no
|
||||
// namespace-wide pod informer runs.
|
||||
Pods: mgr.GetAPIReader(),
|
||||
Recorder: mgr.GetEventRecorderFor("felis-operator"),
|
||||
Watch: watch,
|
||||
}
|
||||
|
||||
+26
-5
@@ -78,7 +78,7 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
drv, err := openStore(ctx, cfg.Database.URL, false)
|
||||
drv, err := openPodStore(ctx, cfg.Database.URL, "reaper", stderr)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: open database: %v\n", err)
|
||||
return 1
|
||||
@@ -144,8 +144,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
// FelisWorldJobFailed rule) reach the operator: a world that cannot be archived
|
||||
// is kept, and without this nobody would learn that it is never reaped.
|
||||
func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
|
||||
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d released=%d deleted=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
|
||||
sum.Evaluated, sum.WorldsReaped, sum.Released, sum.ServersDeleted, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed,
|
||||
sum.Verified, sum.Corrupt, sum.VerifyFailed, sum.Swept, sum.OrphanArchives)
|
||||
if !sum.Failed() {
|
||||
@@ -199,8 +199,9 @@ func (w *mailWarner) Warn(ctx context.Context, ownerID, server, remaining string
|
||||
|
||||
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
|
||||
// idle deadline is fixed by §18; only the warning offsets, retention, the
|
||||
// store soft-cap and the on-demand backup bounds are configurable (§24). The
|
||||
// backup Job and felis-api read the manual_* bounds through it too.
|
||||
// store soft-cap, the on-demand backup bounds and the scheduled restore points
|
||||
// are configurable (§24). The backup Job and felis-api read the manual_* and
|
||||
// scheduled_* keys through it too.
|
||||
func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
||||
rc := reaper.DefaultConfig()
|
||||
if v := cfg.Archive.Retention; v != "" {
|
||||
@@ -248,6 +249,26 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
||||
}
|
||||
rc.ManualCooldown = d
|
||||
}
|
||||
if v := cfg.Archive.ScheduledEvery; v != "" {
|
||||
d, err := parseSpanDuration(v)
|
||||
if err != nil || d < 0 {
|
||||
return rc, fmt.Errorf("[archive] scheduled_every %q: want a span such as 1d (0s for none)", v)
|
||||
}
|
||||
rc.ScheduledEvery = d
|
||||
}
|
||||
switch n := cfg.Archive.ScheduledKeep; {
|
||||
case n < 0:
|
||||
return rc, fmt.Errorf("[archive] scheduled_keep %d: want 1 or more", n)
|
||||
case n > 0:
|
||||
rc.ScheduledKeep = n
|
||||
}
|
||||
if v := cfg.Archive.ScheduledRetention; v != "" {
|
||||
d, err := parseSpanDuration(v)
|
||||
if err != nil || d <= 0 {
|
||||
return rc, fmt.Errorf("[archive] scheduled_retention %q: want a positive span such as 90d", v)
|
||||
}
|
||||
rc.ScheduledRetention = d
|
||||
}
|
||||
rc.RequireOffsite = cfg.Offsite.Enabled()
|
||||
return rc, nil
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
@@ -27,6 +28,7 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
|
||||
want int
|
||||
}{
|
||||
{"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0},
|
||||
{"retirements", reaper.Summary{Evaluated: 5, WorldsReaped: 3, Released: 2, ServersDeleted: 1}, 0},
|
||||
{"waiting for a stop", reaper.Summary{Evaluated: 3, AwaitingStop: 1}, 0},
|
||||
{"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1},
|
||||
{"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1},
|
||||
@@ -40,7 +42,8 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
|
||||
if got := reportReaperRun(tc.sum, &out, &errb); got != tc.want {
|
||||
t.Errorf("%s: exit %d, want %d", tc.name, got, tc.want)
|
||||
}
|
||||
if !strings.Contains(out.String(), "skipped=") || !strings.Contains(out.String(), "expire_failed=") {
|
||||
if !strings.Contains(out.String(), "skipped=") || !strings.Contains(out.String(), "expire_failed=") ||
|
||||
!strings.Contains(out.String(), fmt.Sprintf(" released=%d deleted=%d ", tc.sum.Released, tc.sum.ServersDeleted)) {
|
||||
t.Errorf("%s: summary line = %q", tc.name, out.String())
|
||||
}
|
||||
if (tc.want == 1) != (errb.Len() > 0) {
|
||||
@@ -85,6 +88,42 @@ func TestReaperConfigManualKeys(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestReaperConfigScheduledKeys: the scheduled restore points default to one a
|
||||
// day, seven per server and the reaper's 90 days, accept overrides ("0s" turns
|
||||
// them off), and refuse values that would keep nothing or run backwards.
|
||||
func TestReaperConfigScheduledKeys(t *testing.T) {
|
||||
rc, err := reaperConfig(&config.Config{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if rc.ScheduledEvery != reaper.Day || rc.ScheduledKeep != 7 || rc.ScheduledRetention != 90*reaper.Day {
|
||||
t.Fatalf("defaults = %v / %d / %v", rc.ScheduledEvery, rc.ScheduledKeep, rc.ScheduledRetention)
|
||||
}
|
||||
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{
|
||||
ScheduledEvery: "12h", ScheduledKeep: 3, ScheduledRetention: "14d"}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if rc.ScheduledEvery != 12*time.Hour || rc.ScheduledKeep != 3 || rc.ScheduledRetention != 14*reaper.Day {
|
||||
t.Fatalf("overrides = %v / %d / %v", rc.ScheduledEvery, rc.ScheduledKeep, rc.ScheduledRetention)
|
||||
}
|
||||
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{ScheduledEvery: "0s"}})
|
||||
if err != nil || rc.ScheduledEvery != 0 {
|
||||
t.Fatalf("scheduled_every 0s = %v, %v; want off", rc.ScheduledEvery, err)
|
||||
}
|
||||
for _, bad := range []config.ArchiveConfig{
|
||||
{ScheduledEvery: "-1h"},
|
||||
{ScheduledEvery: "daily"},
|
||||
{ScheduledKeep: -1},
|
||||
{ScheduledRetention: "0d"},
|
||||
{ScheduledRetention: "forever"},
|
||||
} {
|
||||
if _, err := reaperConfig(&config.Config{Archive: bad}); err == nil {
|
||||
t.Errorf("%+v was accepted", bad)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolveWorldDir(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
root := t.TempDir()
|
||||
|
||||
@@ -131,7 +131,8 @@ func loopbackAddr(addr string) bool {
|
||||
func cmdPushImage(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("push-image", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
tarPath := fs.String("tar", "", "image tarball Kaniko wrote with --tar-path")
|
||||
tarPath := fs.String("tar", "", "image tarball Kaniko wrote with --tar-path, or with --image an OCI layout tar")
|
||||
image := fs.String("image", "", "push the image this name (io.containerd.image.name) marks in the OCI layout tar --tar, e.g. a release's image bundle")
|
||||
ref := fs.String("ref", "", "host/repository:tag to publish it as")
|
||||
scheme := fs.String("scheme", "http", "registry scheme: http for the in-cluster registry, https otherwise")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
@@ -154,7 +155,13 @@ func cmdPushImage(args []string, stdout, stderr io.Writer) int {
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
p := &imagepush.Pusher{Scheme: *scheme, Username: user, Password: pass, Log: stderr}
|
||||
digest, err := p.Push(ctx, *tarPath, *ref)
|
||||
var digest string
|
||||
var err error
|
||||
if *image != "" {
|
||||
digest, err = p.PushLayout(ctx, *tarPath, *image, *ref)
|
||||
} else {
|
||||
digest, err = p.Push(ctx, *tarPath, *ref)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis push-image: %v\n", err)
|
||||
return 1
|
||||
|
||||
+645
-100
@@ -2,40 +2,70 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/hmac"
|
||||
"crypto/pbkdf2"
|
||||
"crypto/rand"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
neturl "net/url"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/registrygate"
|
||||
|
||||
"github.com/BurntSushi/toml"
|
||||
"github.com/jackc/pgx/v5"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// rotate-token replaces one internal caller's token (naming.CallerTokens): a new
|
||||
// value goes into the installer's record, the control-namespace Secret and the
|
||||
// replica the caller's pods mount, felis-api rolls so it accepts only the new
|
||||
// value, and then the caller restarts so it presents it. Between the api's
|
||||
// rollout and the caller's restart the caller is turned away with 401; for the
|
||||
// login gate and the proxy that is the few seconds of a pod or unit restart.
|
||||
// rotate-token replaces one credential the installer generated: an internal
|
||||
// caller's token (naming.CallerTokens), the registry's write tokens, the
|
||||
// Velocity forwarding secret or the database password. The new value goes into
|
||||
// the installer's record first, so that whatever fails later a re-run of the
|
||||
// installer puts it everywhere; then into the Secrets and files that carry it;
|
||||
// then whatever read the old value at start restarts. Without -yes it prints
|
||||
// what it would change and what that interrupts, and changes nothing.
|
||||
|
||||
const (
|
||||
defaultSecretsEnvPath = "/etc/felis/secrets.env"
|
||||
defaultLinkPropsPath = "/opt/felis/velocity/plugins/felis-link/felis-link.properties"
|
||||
defaultForwardingPath = "/opt/felis/velocity/forwarding.secret"
|
||||
velocityUnit = "felis-velocity"
|
||||
apiDeployment = "felis-api"
|
||||
registryDeployment = "registry"
|
||||
|
||||
kindRegistry = "registry"
|
||||
kindForwarding = "forwarding"
|
||||
kindDB = "db"
|
||||
|
||||
// proxyReloadWait is how long the host proxy gets, once felis-api has rolled,
|
||||
// to show it re-read its token: it reads the file again on its next call to
|
||||
// felis-api, and it calls every 15 seconds.
|
||||
proxyReloadWait = 60 * time.Second
|
||||
|
||||
// apiRestartNote is in every plan: felis-api runs as one replica replaced in
|
||||
// place, so its restart is a short outage of everything that talks to it.
|
||||
apiRestartNote = "felis-api restarts (a single replica): the panel, sign-in and the proxy's calls are unavailable for the few seconds that takes"
|
||||
)
|
||||
|
||||
// installerTokenKeys names each caller's token in the installer's secrets.env
|
||||
@@ -49,21 +79,55 @@ var installerTokenKeys = map[string]string{
|
||||
"ops": "OPS_TOKEN",
|
||||
}
|
||||
|
||||
// installerRegistryKeys names the registry principals' tokens in secrets.env,
|
||||
// in registrygate.Principals order.
|
||||
var installerRegistryKeys = map[string]string{
|
||||
registrygate.PrincipalPlatform: "REGISTRY_PLATFORM_TOKEN",
|
||||
registrygate.PrincipalBuild: "REGISTRY_BUILD_TOKEN",
|
||||
registrygate.PrincipalPrune: "REGISTRY_PRUNE_TOKEN",
|
||||
}
|
||||
|
||||
// installerForwardingKey and installerDBKey name the forwarding secret and the
|
||||
// database password in secrets.env.
|
||||
const (
|
||||
installerForwardingKey = "FORWARDING_SECRET"
|
||||
installerDBKey = "DB_PASSWORD"
|
||||
)
|
||||
|
||||
type tokenRotator struct {
|
||||
cl client.Client
|
||||
controlNS string
|
||||
minecraftNS string
|
||||
buildNS string
|
||||
// secretsEnv and linkProps are the installer's record and the proxy's
|
||||
// felis-link.properties; a missing file is reported and skipped.
|
||||
secretsEnv string
|
||||
linkProps string
|
||||
newToken func() (string, error)
|
||||
// rollAPI restarts felis-api and waits for the rollout.
|
||||
rollAPI func(ctx context.Context) error
|
||||
// secretsEnv, linkProps and forwardingFile are the installer's record and the
|
||||
// host proxy's felis-link.properties and forwarding.secret; a missing file is
|
||||
// reported and skipped.
|
||||
secretsEnv string
|
||||
linkProps string
|
||||
forwardingFile string
|
||||
// hostTOML, podTOML and defaultTOML are the config copies that carry the
|
||||
// database URL (defaultTOML only when it is a file of its own).
|
||||
hostTOML string
|
||||
podTOML string
|
||||
defaultTOML string
|
||||
newToken func() (string, error)
|
||||
// rollout restarts a control-namespace Deployment and waits for it.
|
||||
rollout func(ctx context.Context, deployment string) error
|
||||
// restartUnit restarts a systemd unit on this host.
|
||||
restartUnit func(ctx context.Context, unit string) error
|
||||
out io.Writer
|
||||
// proxyLog is what the host proxy has logged since a moment, as far as it
|
||||
// can be read.
|
||||
proxyLog func(ctx context.Context, since time.Time) string
|
||||
// alterRole stores a password verifier for a role of the database that runs
|
||||
// as deployment ("namespace/name"); verifyDB connects with a URL.
|
||||
alterRole func(ctx context.Context, deployment, role, verifier string) error
|
||||
verifyDB func(ctx context.Context, url string) error
|
||||
now func() time.Time
|
||||
// reloadWait bounds the wait for the proxy to re-read its token, polled
|
||||
// every pollEvery.
|
||||
reloadWait time.Duration
|
||||
pollEvery time.Duration
|
||||
out io.Writer
|
||||
}
|
||||
|
||||
func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
|
||||
@@ -72,9 +136,12 @@ func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
|
||||
secretsEnv := fs.String("secrets-env", defaultSecretsEnvPath, "the installer's secrets file, updated so a re-run keeps the new value")
|
||||
linkProps := fs.String("link-properties", defaultLinkPropsPath, "the host proxy's felis-link.properties (velocity only)")
|
||||
forwarding := fs.String("forwarding-secret", defaultForwardingPath, "the host proxy's forwarding secret file (forwarding only)")
|
||||
yes := fs.Bool("yes", false, "rotate; without it the plan is printed and nothing changes")
|
||||
fs.Usage = func() {
|
||||
fmt.Fprintf(stderr, "Usage: felis rotate-token [flags] <%s>\n\n", strings.Join(callerNames(), "|"))
|
||||
fmt.Fprintln(stderr, "Replaces one internal caller's token: the Secrets, felis-api, then the caller itself.")
|
||||
fmt.Fprintf(stderr, "Usage: felis rotate-token [-yes] [flags] <%s>\n\n", strings.Join(rotationKinds(), "|"))
|
||||
fmt.Fprintln(stderr, "Replaces one generated credential: the installer's record, the Secrets and files that carry it, then what reads it.")
|
||||
fmt.Fprintln(stderr, "Without -yes it prints what would change and what that interrupts.")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
if err := fs.Parse(args); err != nil {
|
||||
@@ -87,12 +154,13 @@ func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
|
||||
fs.Usage()
|
||||
return 2
|
||||
}
|
||||
if _, ok := callerToken(fs.Arg(0)); !ok {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: unknown caller %q (one of %s)\n", fs.Arg(0), strings.Join(callerNames(), ", "))
|
||||
kind := fs.Arg(0)
|
||||
if !knownRotation(kind) {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: unknown credential %q (one of %s)\n", kind, strings.Join(rotationKinds(), ", "))
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis rotate-token: refused — rotating writes the cluster Secrets and the installer's secrets file, so it must run as root (try: sudo felis rotate-token "+fs.Arg(0)+")")
|
||||
fmt.Fprintln(stderr, "felis rotate-token: refused — rotating writes the cluster Secrets and the installer's secrets file, so it must run as root (try: sudo felis rotate-token "+kind+")")
|
||||
return 1
|
||||
}
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
@@ -109,24 +177,43 @@ func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
|
||||
if buildNS == "" {
|
||||
buildNS = platform.DefaultBuildNamespace
|
||||
}
|
||||
controlNS := platform.DefaultControlNamespace
|
||||
r := tokenRotator{
|
||||
cl: cl,
|
||||
controlNS: platform.DefaultControlNamespace,
|
||||
minecraftNS: cfg.K8s.Namespace,
|
||||
buildNS: buildNS,
|
||||
secretsEnv: *secretsEnv,
|
||||
linkProps: *linkProps,
|
||||
newToken: randomToken,
|
||||
rollAPI: func(ctx context.Context) error {
|
||||
if err := kubectl(ctx, "-n", platform.DefaultControlNamespace, "rollout", "restart", "deployment/felis-api"); err != nil {
|
||||
cl: cl,
|
||||
controlNS: controlNS,
|
||||
minecraftNS: cfg.K8s.Namespace,
|
||||
buildNS: buildNS,
|
||||
secretsEnv: *secretsEnv,
|
||||
linkProps: *linkProps,
|
||||
forwardingFile: *forwarding,
|
||||
hostTOML: hostSetupConfigPath,
|
||||
podTOML: podSetupConfigPath,
|
||||
defaultTOML: defaultSetupConfigPath,
|
||||
newToken: randomToken,
|
||||
rollout: func(ctx context.Context, deployment string) error {
|
||||
if err := kubectl(ctx, "-n", controlNS, "rollout", "restart", "deployment/"+deployment); err != nil {
|
||||
return err
|
||||
}
|
||||
return kubectl(ctx, "-n", platform.DefaultControlNamespace, "rollout", "status", "deployment/felis-api", "--timeout=180s")
|
||||
return kubectl(ctx, "-n", controlNS, "rollout", "status", "deployment/"+deployment, "--timeout=180s")
|
||||
},
|
||||
restartUnit: func(ctx context.Context, unit string) error { return systemctl(ctx, "restart", unit) },
|
||||
out: stdout,
|
||||
proxyLog: journalSince,
|
||||
alterRole: alterRoleInPod,
|
||||
verifyDB: func(ctx context.Context, url string) error {
|
||||
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
defer cancel()
|
||||
conn, err := pgx.Connect(ctx, url)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return conn.Close(ctx)
|
||||
},
|
||||
now: time.Now,
|
||||
reloadWait: proxyReloadWait,
|
||||
pollEvery: 3 * time.Second,
|
||||
out: stdout,
|
||||
}
|
||||
if err := r.rotate(context.Background(), fs.Arg(0)); err != nil {
|
||||
if err := r.rotate(context.Background(), kind, *yes); err != nil {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
@@ -150,6 +237,21 @@ func callerToken(name string) (naming.CallerToken, bool) {
|
||||
return naming.CallerToken{}, false
|
||||
}
|
||||
|
||||
// rotationKinds is every credential rotate-token replaces: the callers' tokens
|
||||
// and the installer's other generated secrets.
|
||||
func rotationKinds() []string {
|
||||
return append(callerNames(), kindRegistry, kindForwarding, kindDB)
|
||||
}
|
||||
|
||||
func knownRotation(kind string) bool {
|
||||
for _, k := range rotationKinds() {
|
||||
if k == kind {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// randomToken is 32 random bytes in hex, the shape the installer generates.
|
||||
func randomToken() (string, error) {
|
||||
b := make([]byte, 32)
|
||||
@@ -159,99 +261,527 @@ func randomToken() (string, error) {
|
||||
return hex.EncodeToString(b), nil
|
||||
}
|
||||
|
||||
func (r tokenRotator) rotate(ctx context.Context, caller string) error {
|
||||
ct, ok := callerToken(caller)
|
||||
if !ok {
|
||||
return fmt.Errorf("unknown caller %q", caller)
|
||||
}
|
||||
tok, err := r.newToken()
|
||||
if err != nil {
|
||||
return fmt.Errorf("generate a token: %w", err)
|
||||
}
|
||||
// tokenFingerprint is how the proxy names the token it reloaded in its log
|
||||
// (plugins/shared FileToken.fingerprint): the first twelve hex digits of its
|
||||
// SHA-256, enough to tell tokens apart and useless for finding one.
|
||||
func tokenFingerprint(token string) string {
|
||||
sum := sha256.Sum256([]byte(token))
|
||||
return hex.EncodeToString(sum[:])[:12]
|
||||
}
|
||||
|
||||
// The installer's record first: from here on, whatever fails, a re-run of the
|
||||
// installer puts the new value everywhere.
|
||||
switch err := setKeyValueLine(r.secretsEnv, installerTokenKeys[ct.Caller], "=", tok); {
|
||||
func (r tokenRotator) rotate(ctx context.Context, kind string, apply bool) error {
|
||||
switch kind {
|
||||
case kindRegistry:
|
||||
return r.rotateRegistry(ctx, apply)
|
||||
case kindForwarding:
|
||||
return r.rotateForwarding(ctx, apply)
|
||||
case kindDB:
|
||||
return r.rotateDB(ctx, apply)
|
||||
}
|
||||
ct, ok := callerToken(kind)
|
||||
if !ok {
|
||||
return fmt.Errorf("unknown credential %q (one of %s)", kind, strings.Join(rotationKinds(), ", "))
|
||||
}
|
||||
return r.rotateCaller(ctx, ct, apply)
|
||||
}
|
||||
|
||||
// confirm ends the plan: without apply it says nothing changed and how to go
|
||||
// ahead, and reports false.
|
||||
func (r tokenRotator) confirm(kind string, apply bool) bool {
|
||||
if !apply {
|
||||
fmt.Fprintf(r.out, "\nNothing was changed. To rotate: sudo felis rotate-token -yes %s\n", kind)
|
||||
return false
|
||||
}
|
||||
fmt.Fprintln(r.out, "\nRotating:")
|
||||
return true
|
||||
}
|
||||
|
||||
// record writes new values into the installer's secrets.env. It comes first in
|
||||
// every rotation: from then on, whatever fails, a re-run of the installer puts
|
||||
// the new values everywhere.
|
||||
func (r tokenRotator) record(kv ...[2]string) error {
|
||||
keys := make([]string, len(kv))
|
||||
for i, p := range kv {
|
||||
keys[i] = p[0]
|
||||
}
|
||||
switch err := setKeyValueLines(r.secretsEnv, "=", kv); {
|
||||
case errors.Is(err, fs.ErrNotExist):
|
||||
fmt.Fprintf(r.out, " - %s: not found, skipped (this host was not installed by deploy/bootstrap.sh)\n", r.secretsEnv)
|
||||
case err != nil:
|
||||
return fmt.Errorf("record the new token in %s: %w", r.secretsEnv, err)
|
||||
return fmt.Errorf("record the new value in %s: %w", r.secretsEnv, err)
|
||||
default:
|
||||
fmt.Fprintf(r.out, " - %s: %s updated\n", r.secretsEnv, installerTokenKeys[ct.Caller])
|
||||
fmt.Fprintf(r.out, " - %s: %s updated\n", r.secretsEnv, strings.Join(keys, ", "))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (r tokenRotator) putSecret(ctx context.Context, ns, name string, data map[string][]byte) error {
|
||||
if _, err := putSecretKeys(ctx, r.cl, ns, name, corev1.SecretTypeOpaque, data); err != nil {
|
||||
return fmt.Errorf("write Secret %s/%s: %w", ns, name, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - Secret %s/%s: updated\n", ns, name)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (r tokenRotator) rollAPI(ctx context.Context) error {
|
||||
if err := r.rollout(ctx, apiDeployment); err != nil {
|
||||
return fmt.Errorf("roll felis-api: %w", err)
|
||||
}
|
||||
fmt.Fprintln(r.out, " - felis-api: rolled out on the new value")
|
||||
return nil
|
||||
}
|
||||
|
||||
// withMinecraft is the control namespace plus the minecraft namespace when that
|
||||
// is a different one: where the Secrets game pods and Jobs mount are mirrored.
|
||||
func (r tokenRotator) withMinecraft() []string {
|
||||
if r.minecraftNS == "" || r.minecraftNS == r.controlNS {
|
||||
return []string{r.controlNS}
|
||||
}
|
||||
return []string{r.controlNS, r.minecraftNS}
|
||||
}
|
||||
|
||||
func (r tokenRotator) rotateCaller(ctx context.Context, ct naming.CallerToken, apply bool) error {
|
||||
namespaces := []string{r.controlNS}
|
||||
replica := map[string]string{"minecraft": r.minecraftNS, "build": r.buildNS}[ct.Replica]
|
||||
if replica != "" && replica != r.controlNS {
|
||||
namespaces = append(namespaces, replica)
|
||||
}
|
||||
for _, ns := range namespaces {
|
||||
if err := writeTokenSecret(ctx, r.cl, ns, ct.Secret, tok); err != nil {
|
||||
return fmt.Errorf("write Secret %s/%s: %w", ns, ct.Secret, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - Secret %s/%s: updated\n", ns, ct.Secret)
|
||||
}
|
||||
key := installerTokenKeys[ct.Caller]
|
||||
|
||||
hostProxy := false
|
||||
if ct.Caller == "velocity" {
|
||||
switch err := setKeyValueLine(r.linkProps, "service-token", "=", tok); {
|
||||
case errors.Is(err, fs.ErrNotExist):
|
||||
fmt.Fprintf(r.out, " - %s: not found; set service-token in your proxy's felis-link.properties to the value in Secret %s/%s and restart it\n",
|
||||
r.linkProps, r.controlNS, ct.Secret)
|
||||
case err != nil:
|
||||
return fmt.Errorf("write the proxy's token into %s: %w", r.linkProps, err)
|
||||
default:
|
||||
hostProxy = true
|
||||
fmt.Fprintf(r.out, " - %s: service-token updated\n", r.linkProps)
|
||||
}
|
||||
fmt.Fprintf(r.out, "felis rotate-token %s: a new internal token for the %s caller\n", ct.Caller, ct.Caller)
|
||||
secrets := make([]string, len(namespaces))
|
||||
for i, ns := range namespaces {
|
||||
secrets[i] = "Secret " + ns + "/" + ct.Secret
|
||||
}
|
||||
|
||||
if err := r.rollAPI(ctx); err != nil {
|
||||
return fmt.Errorf("roll felis-api: %w", err)
|
||||
writes := append([]string{r.secretsEnv + " (" + key + ")"}, secrets...)
|
||||
hostProxy := ct.Caller == "velocity" && fileExists(r.linkProps)
|
||||
if hostProxy {
|
||||
writes = append(writes, r.linkProps+" (service-token)")
|
||||
}
|
||||
fmt.Fprintln(r.out, " - felis-api: rolled out, accepting only the new token")
|
||||
|
||||
fmt.Fprintf(r.out, " - writes %s\n", strings.Join(writes, ", "))
|
||||
fmt.Fprintf(r.out, " - %s\n", apiRestartNote)
|
||||
switch ct.Caller {
|
||||
case "velocity":
|
||||
if hostProxy {
|
||||
if err := r.restartUnit(ctx, velocityUnit); err != nil {
|
||||
return fmt.Errorf("restart %s: %w", velocityUnit, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: restarted (players on the proxy were disconnected and can rejoin)\n", velocityUnit)
|
||||
fmt.Fprintf(r.out, " - the proxy re-reads its token from the file and keeps its players; if it has not within %s of felis-api's restart, it is restarted, which disconnects everyone online\n", r.reloadWait)
|
||||
} else {
|
||||
fmt.Fprintf(r.out, " - no proxy on this host (%s): set service-token in your proxy's felis-link.properties to the value in Secret %s/%s afterwards; it re-reads the file without a restart\n",
|
||||
r.linkProps, r.controlNS, ct.Secret)
|
||||
}
|
||||
case "limbo":
|
||||
if err := r.cl.DeleteAllOf(ctx, &corev1.Pod{}, client.InNamespace(r.minecraftNS),
|
||||
client.MatchingLabels{v1alpha1.LabelServer: naming.SystemLoginServer}); err != nil {
|
||||
fmt.Fprintln(r.out, " - the login gate's pod restarts: a player signing in at that moment reconnects")
|
||||
case "build":
|
||||
fmt.Fprintln(r.out, " - a build fetching its context at that moment fails and can be submitted again")
|
||||
case "ops":
|
||||
fmt.Fprintln(r.out, " - felis backup-now presents the new token on its next run")
|
||||
}
|
||||
if !r.confirm(ct.Caller, apply) {
|
||||
return nil
|
||||
}
|
||||
|
||||
tok, err := r.newToken()
|
||||
if err != nil {
|
||||
return fmt.Errorf("generate a token: %w", err)
|
||||
}
|
||||
if err := r.record([2]string{key, tok}); err != nil {
|
||||
return err
|
||||
}
|
||||
for _, ns := range namespaces {
|
||||
if err := r.putSecret(ctx, ns, ct.Secret, map[string][]byte{naming.ServiceTokenSecretKey: []byte(tok)}); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
// The proxy's file changes before felis-api rolls, so the file never falls
|
||||
// behind the Secret; the proxy's log is read from just before the write,
|
||||
// since it may pick the new token up before the rollout ends.
|
||||
since := r.now()
|
||||
if hostProxy {
|
||||
if err := setProxyToken(r.linkProps, tok); err != nil {
|
||||
return fmt.Errorf("write the proxy's token into %s: %w", r.linkProps, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: service-token updated\n", r.linkProps)
|
||||
}
|
||||
|
||||
if err := r.rollAPI(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
switch ct.Caller {
|
||||
case "velocity":
|
||||
if !hostProxy {
|
||||
break
|
||||
}
|
||||
if r.proxyReloaded(ctx, since, tokenFingerprint(tok)) {
|
||||
fmt.Fprintf(r.out, " - %s: took the new token from its properties; players stayed connected\n", velocityUnit)
|
||||
break
|
||||
}
|
||||
if err := r.restartUnit(ctx, velocityUnit); err != nil {
|
||||
return fmt.Errorf("restart %s: %w", velocityUnit, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: had not taken the new token within %s, restarted (players on the proxy were disconnected and can rejoin)\n", velocityUnit, r.reloadWait)
|
||||
case "limbo":
|
||||
// Only the operator's server pods carry its managed-by label; a backup or
|
||||
// restore Job's pod carries the server label too, and app.kubernetes.io ones.
|
||||
if err := r.cl.DeleteAllOf(ctx, &corev1.Pod{}, client.InNamespace(r.minecraftNS), client.MatchingLabels{
|
||||
v1alpha1.LabelServer: naming.SystemLoginServer,
|
||||
v1alpha1.LabelManagedBy: operator.ManagedByValue,
|
||||
}); err != nil {
|
||||
return fmt.Errorf("restart the login gate: %w", err)
|
||||
}
|
||||
fmt.Fprintln(r.out, " - login gate: pod restarted to read the new token")
|
||||
case "build":
|
||||
fmt.Fprintln(r.out, " - builds: the next build Job reads the new token; one fetching its context right now fails and can be submitted again")
|
||||
fmt.Fprintln(r.out, " - builds: the next build Job reads the new token")
|
||||
case "ops":
|
||||
fmt.Fprintln(r.out, " - felis backup-now reads the new token on its next run")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// writeTokenSecret sets the token in a Secret, creating it when absent.
|
||||
func writeTokenSecret(ctx context.Context, cl client.Client, namespace, name, token string) error {
|
||||
var sec corev1.Secret
|
||||
err := cl.Get(ctx, client.ObjectKey{Namespace: namespace, Name: name}, &sec)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return cl.Create(ctx, &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: namespace, Name: name},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
Data: map[string][]byte{naming.ServiceTokenSecretKey: []byte(token)},
|
||||
})
|
||||
// proxyReloaded waits up to reloadWait for the host proxy to log that it
|
||||
// reloaded the token with this fingerprint (plugins/shared FileToken).
|
||||
func (r tokenRotator) proxyReloaded(ctx context.Context, since time.Time, fingerprint string) bool {
|
||||
want := "(fingerprint " + fingerprint + ")"
|
||||
deadline := r.now().Add(r.reloadWait)
|
||||
for {
|
||||
if strings.Contains(r.proxyLog(ctx, since), want) {
|
||||
return true
|
||||
}
|
||||
if !r.now().Before(deadline) {
|
||||
return false
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return false
|
||||
case <-time.After(r.pollEvery):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// journalSince is the proxy unit's journal from the second since falls in. A
|
||||
// journal that cannot be read reads as one without the line, which ends in the
|
||||
// restart a rotation made before the proxy could reload.
|
||||
func journalSince(ctx context.Context, since time.Time) string {
|
||||
out, _ := exec.CommandContext(ctx, "journalctl", "-u", velocityUnit, "--since", "@"+strconv.FormatInt(since.Unix(), 10),
|
||||
"-o", "cat", "--no-pager", "-q").Output()
|
||||
return string(out)
|
||||
}
|
||||
|
||||
// setProxyToken writes the proxy's token into felis-link.properties and puts
|
||||
// the file's modification time back. The proxy re-reads its token by itself,
|
||||
// while `felis domain check` and `felis domain set` read a file newer than the
|
||||
// proxy's start as config it has not loaded (the installer's
|
||||
// install_if_changed keeps the time for the same reason).
|
||||
func setProxyToken(path, token string) error {
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if sec.Data == nil {
|
||||
sec.Data = map[string][]byte{}
|
||||
if err := setKeyValueLine(path, "service-token", "=", token); err != nil {
|
||||
return err
|
||||
}
|
||||
sec.Data[naming.ServiceTokenSecretKey] = []byte(token)
|
||||
return cl.Update(ctx, &sec)
|
||||
return os.Chtimes(path, time.Time{}, info.ModTime())
|
||||
}
|
||||
|
||||
func (r tokenRotator) rotateRegistry(ctx context.Context, apply bool) error {
|
||||
keys := make([]string, len(registrygate.Principals))
|
||||
for i, p := range registrygate.Principals {
|
||||
keys[i] = installerRegistryKeys[p]
|
||||
}
|
||||
fmt.Fprintf(r.out, "felis rotate-token %s: new write tokens for the image registry's principals (%s)\n", kindRegistry, strings.Join(registrygate.Principals, ", "))
|
||||
fmt.Fprintf(r.out, " - writes %s (%s), Secret %s/%s, Secret %s/%s\n", r.secretsEnv, strings.Join(keys, ", "),
|
||||
r.controlNS, naming.RegistryAuthSecretName, r.buildNS, naming.RegistryPushSecretName)
|
||||
fmt.Fprintln(r.out, " - the registry restarts to load them: a build pushing its image at that moment fails and can be submitted again, and an image pull in that moment retries")
|
||||
fmt.Fprintf(r.out, " - %s (it presents the prune token)\n", apiRestartNote)
|
||||
if !r.confirm(kindRegistry, apply) {
|
||||
return nil
|
||||
}
|
||||
|
||||
tokens := map[string][]byte{}
|
||||
kv := make([][2]string, 0, len(registrygate.Principals))
|
||||
for _, p := range registrygate.Principals {
|
||||
tok, err := r.newToken()
|
||||
if err != nil {
|
||||
return fmt.Errorf("generate a token: %w", err)
|
||||
}
|
||||
tokens[p] = []byte(tok)
|
||||
kv = append(kv, [2]string{installerRegistryKeys[p], tok})
|
||||
}
|
||||
if err := r.record(kv...); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := r.putSecret(ctx, r.controlNS, naming.RegistryAuthSecretName, tokens); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := r.putSecret(ctx, r.buildNS, naming.RegistryPushSecretName, map[string][]byte{
|
||||
naming.RegistryPushUsernameKey: []byte(registrygate.PrincipalBuild),
|
||||
naming.RegistryPushPasswordKey: tokens[registrygate.PrincipalBuild],
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := r.rollout(ctx, registryDeployment); err != nil {
|
||||
return fmt.Errorf("restart the registry: %w", err)
|
||||
}
|
||||
fmt.Fprintln(r.out, " - registry: restarted on the new tokens")
|
||||
return r.rollAPI(ctx)
|
||||
}
|
||||
|
||||
func (r tokenRotator) rotateForwarding(ctx context.Context, apply bool) error {
|
||||
restart, held, err := r.gamePods(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
hostProxy := fileExists(r.forwardingFile)
|
||||
fmt.Fprintf(r.out, "felis rotate-token %s: a new Velocity forwarding secret, the key a server checks each player's identity with\n", kindForwarding)
|
||||
secrets := []string{}
|
||||
for _, ns := range r.withMinecraft() {
|
||||
secrets = append(secrets, "Secret "+ns+"/"+naming.ForwardingSecretName)
|
||||
}
|
||||
writes := append([]string{r.secretsEnv + " (" + installerForwardingKey + ")"}, secrets...)
|
||||
if hostProxy {
|
||||
writes = append(writes, r.forwardingFile)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - writes %s\n", strings.Join(writes, ", "))
|
||||
fmt.Fprintf(r.out, " - every running server restarts to read it (%d now), saving its world on the way down, and the proxy restarts: everyone online is disconnected and can rejoin once their server is back\n", len(restart))
|
||||
if len(held) > 0 {
|
||||
fmt.Fprintf(r.out, " - left running, because a backup, restore or file write holds its world: %s. Players cannot join it until it restarts: stop and start it from the panel once that finishes\n", strings.Join(held, ", "))
|
||||
}
|
||||
if !hostProxy {
|
||||
fmt.Fprintf(r.out, " - no proxy on this host (%s): put the value in Secret %s/%s into your proxy's forwarding secret file and restart it\n",
|
||||
r.forwardingFile, r.controlNS, naming.ForwardingSecretName)
|
||||
}
|
||||
if !r.confirm(kindForwarding, apply) {
|
||||
return nil
|
||||
}
|
||||
|
||||
tok, err := r.newToken()
|
||||
if err != nil {
|
||||
return fmt.Errorf("generate a secret: %w", err)
|
||||
}
|
||||
if err := r.record([2]string{installerForwardingKey, tok}); err != nil {
|
||||
return err
|
||||
}
|
||||
if hostProxy {
|
||||
info, err := os.Stat(r.forwardingFile)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := replaceFileKeepingMode(r.forwardingFile, info, []byte(tok)); err != nil {
|
||||
return fmt.Errorf("write %s: %w", r.forwardingFile, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: updated\n", r.forwardingFile)
|
||||
}
|
||||
for _, ns := range r.withMinecraft() {
|
||||
if err := r.putSecret(ctx, ns, naming.ForwardingSecretName, map[string][]byte{naming.ForwardingSecretKey: []byte(tok)}); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
// The servers go first: their pods read the Secret as they are recreated,
|
||||
// and a player who rejoins through the restarted proxy meets a server on the
|
||||
// new secret, or one still starting.
|
||||
for i := range restart {
|
||||
p := &restart[i]
|
||||
if err := r.cl.Delete(ctx, p, client.Preconditions{UID: &p.UID}); err != nil && !apierrors.IsNotFound(err) && !apierrors.IsConflict(err) {
|
||||
return fmt.Errorf("restart %s: %w", p.Labels[v1alpha1.LabelServer], err)
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(r.out, " - servers: %d restarting on the new secret\n", len(restart))
|
||||
if hostProxy {
|
||||
if err := r.restartUnit(ctx, velocityUnit); err != nil {
|
||||
return fmt.Errorf("restart %s: %w", velocityUnit, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: restarted on the new secret\n", velocityUnit)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// gamePods lists the running game server pods: those a rotation may restart,
|
||||
// and the servers left alone because a backup, restore or file write holds
|
||||
// their world (internal/maintenance), which a restart in the middle of would
|
||||
// break.
|
||||
func (r tokenRotator) gamePods(ctx context.Context) ([]corev1.Pod, []string, error) {
|
||||
var pods corev1.PodList
|
||||
// The operator's managed-by label is on its server pods alone (see rotateCaller).
|
||||
if err := r.cl.List(ctx, &pods, client.InNamespace(r.minecraftNS), client.MatchingLabels{v1alpha1.LabelManagedBy: operator.ManagedByValue}); err != nil {
|
||||
return nil, nil, fmt.Errorf("list the servers' pods: %w", err)
|
||||
}
|
||||
var jobs batchv1.JobList
|
||||
if err := r.cl.List(ctx, &jobs, client.InNamespace(r.minecraftNS)); err != nil {
|
||||
return nil, nil, fmt.Errorf("list the maintenance Jobs: %w", err)
|
||||
}
|
||||
var restart []corev1.Pod
|
||||
var held []string
|
||||
for _, p := range pods.Items {
|
||||
if p.DeletionTimestamp != nil {
|
||||
continue
|
||||
}
|
||||
server := p.Labels[v1alpha1.LabelServer]
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := r.cl.Get(ctx, client.ObjectKey{Namespace: r.minecraftNS, Name: server}, &ms); client.IgnoreNotFound(err) != nil {
|
||||
return nil, nil, fmt.Errorf("read server %s: %w", server, err)
|
||||
}
|
||||
if kind, ok := maintenance.Holder(server, ms.Annotations, jobs.Items, r.now()); ok {
|
||||
held = append(held, server+" ("+kind+")")
|
||||
continue
|
||||
}
|
||||
restart = append(restart, p)
|
||||
}
|
||||
return restart, held, nil
|
||||
}
|
||||
|
||||
func (r tokenRotator) rotateDB(ctx context.Context, apply bool) error {
|
||||
cfg, err := config.Load(r.hostTOML)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if cfg.Database.Deployment == "" {
|
||||
return fmt.Errorf("[database] deployment is unset in %s, so the database is not the installer's felis-postgres: change the role's password where it runs, then in [database] url of each config copy", r.hostTOML)
|
||||
}
|
||||
u, err := neturl.Parse(cfg.Database.URL)
|
||||
if err != nil || u.User == nil || u.User.Username() == "" {
|
||||
return fmt.Errorf("[database] url in %s names no role", r.hostTOML)
|
||||
}
|
||||
role := u.User.Username()
|
||||
targets, err := tomlTargetsOf(r.hostTOML, r.podTOML, r.defaultTOML)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
paths := make([]string, len(targets))
|
||||
for i, t := range targets {
|
||||
paths[i] = t.path
|
||||
}
|
||||
secrets := []string{}
|
||||
for _, ns := range r.withMinecraft() {
|
||||
secrets = append(secrets, ns+"/"+platform.ConfigSecretName)
|
||||
}
|
||||
fmt.Fprintf(r.out, "felis rotate-token %s: a new password for the database role %q\n", kindDB, role)
|
||||
fmt.Fprintf(r.out, " - writes %s (%s), the role in %s, [database] url in %s, Secret %s\n",
|
||||
r.secretsEnv, installerDBKey, cfg.Database.Deployment, strings.Join(paths, " and "), strings.Join(secrets, " and "))
|
||||
fmt.Fprintf(r.out, " - %s\n", apiRestartNote)
|
||||
fmt.Fprintln(r.out, " - a backup, restore or file Job that connects in the seconds between the password change and felis-api's restart fails and can be run again; the host's timers read the new config on their next run")
|
||||
if !r.confirm(kindDB, apply) {
|
||||
return nil
|
||||
}
|
||||
|
||||
password, err := r.newToken()
|
||||
if err != nil {
|
||||
return fmt.Errorf("generate a password: %w", err)
|
||||
}
|
||||
// Every config copy is edited in memory first, so one this cannot edit stops
|
||||
// the rotation before the role's password changes.
|
||||
edited := make([][]byte, len(targets))
|
||||
var hostURL string
|
||||
var podConfig []byte
|
||||
for i, t := range targets {
|
||||
raw, err := os.ReadFile(t.real)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var doc struct {
|
||||
Database struct {
|
||||
URL string `toml:"url"`
|
||||
} `toml:"database"`
|
||||
}
|
||||
if _, err := toml.Decode(string(raw), &doc); err != nil {
|
||||
return fmt.Errorf("%s: %w", t.path, err)
|
||||
}
|
||||
next, err := withPassword(doc.Database.URL, password)
|
||||
if err != nil {
|
||||
return fmt.Errorf("%s: %w", t.path, err)
|
||||
}
|
||||
if edited[i], err = editTOMLStrings(raw, []tomlStringEdit{{"database", "url", next}}); err != nil {
|
||||
return fmt.Errorf("%s: %w; set the password in its [database] url by hand", t.path, err)
|
||||
}
|
||||
switch t.path {
|
||||
case r.hostTOML:
|
||||
hostURL = next
|
||||
case r.podTOML:
|
||||
podConfig = edited[i]
|
||||
}
|
||||
}
|
||||
if podConfig == nil {
|
||||
return fmt.Errorf("%s resolves to the same file as %s; the pods reach the database at another address and need a copy of their own", r.podTOML, r.hostTOML)
|
||||
}
|
||||
|
||||
if err := r.record([2]string{installerDBKey, password}); err != nil {
|
||||
return err
|
||||
}
|
||||
salt := make([]byte, 16)
|
||||
if _, err := rand.Read(salt); err != nil {
|
||||
return err
|
||||
}
|
||||
verifier, err := scramVerifier(password, salt, scramIterations)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := r.alterRole(ctx, cfg.Database.Deployment, role, verifier); err != nil {
|
||||
return fmt.Errorf("set the role's password: %w", err)
|
||||
}
|
||||
if err := r.verifyDB(ctx, hostURL); err != nil {
|
||||
return fmt.Errorf("the database does not accept the new password (%v); the config copies still hold the old one: run the installer again (sudo bash deploy/bootstrap.sh), which sets the password in %s everywhere", err, r.secretsEnv)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - role %s: password changed, and the database accepts it\n", role)
|
||||
for i, t := range targets {
|
||||
info, err := os.Stat(t.real)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := replaceFileKeepingMode(t.real, info, edited[i]); err != nil {
|
||||
return fmt.Errorf("write %s: %w", t.path, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: [database] url updated\n", t.path)
|
||||
}
|
||||
for _, ns := range r.withMinecraft() {
|
||||
if err := r.putSecret(ctx, ns, platform.ConfigSecretName, map[string][]byte{platform.ConfigSecretKey: podConfig}); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return r.rollAPI(ctx)
|
||||
}
|
||||
|
||||
// withPassword is a database URL with its password replaced.
|
||||
func withPassword(raw, password string) (string, error) {
|
||||
u, err := neturl.Parse(raw)
|
||||
if err != nil || u.User == nil || u.User.Username() == "" {
|
||||
return "", errors.New("its [database] url names no role")
|
||||
}
|
||||
u.User = neturl.UserPassword(u.User.Username(), password)
|
||||
return u.String(), nil
|
||||
}
|
||||
|
||||
// scramIterations is PostgreSQL's default scram_iterations.
|
||||
const scramIterations = 4096
|
||||
|
||||
// scramVerifier is the SCRAM-SHA-256 verifier PostgreSQL stores for a password
|
||||
// (RFC 5802 and RFC 7677, in the form libpq's PQencryptPasswordConn makes).
|
||||
// ALTER ROLE stores a verifier as it is given, so the password itself never
|
||||
// reaches the server, where a failing statement is logged with its text.
|
||||
func scramVerifier(password string, salt []byte, iterations int) (string, error) {
|
||||
salted, err := pbkdf2.Key(sha256.New, password, salt, iterations, sha256.Size)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
mac := func(msg string) []byte {
|
||||
h := hmac.New(sha256.New, salted)
|
||||
h.Write([]byte(msg))
|
||||
return h.Sum(nil)
|
||||
}
|
||||
stored := sha256.Sum256(mac("Client Key"))
|
||||
b64 := base64.StdEncoding.EncodeToString
|
||||
return fmt.Sprintf("SCRAM-SHA-256$%d:%s$%s:%s", iterations, b64(salt), b64(stored[:]), b64(mac("Server Key"))), nil
|
||||
}
|
||||
|
||||
// alterRoleInPod runs ALTER ROLE as the superuser inside the database's
|
||||
// container, over its socket, with the statement on stdin.
|
||||
func alterRoleInPod(ctx context.Context, deployment, role, verifier string) error {
|
||||
ns, name, _ := strings.Cut(deployment, "/")
|
||||
return kubectlWithInput(ctx, []byte(alterRoleSQL(role, verifier)), "-n", ns, "exec", "-i", "deploy/"+name, "-c", platform.PostgresContainer, "--",
|
||||
"psql", "-X", "-q", "-v", "ON_ERROR_STOP=1", "-U", "postgres", "-d", "postgres")
|
||||
}
|
||||
|
||||
// alterRoleSQL sets role's password to a SCRAM verifier, which holds no quote.
|
||||
func alterRoleSQL(role, verifier string) string {
|
||||
return "ALTER ROLE \"" + strings.ReplaceAll(role, `"`, `""`) + "\" WITH PASSWORD '" + verifier + "';\n"
|
||||
}
|
||||
|
||||
// setKeyValueLine rewrites the `key<sep>value` line of a flat key/value file
|
||||
@@ -259,6 +789,11 @@ func writeTokenSecret(ctx context.Context, cl client.Client, namespace, name, to
|
||||
// file is replaced atomically and keeps its mode and owner: felis-link.properties
|
||||
// is root:felis-velocity 0640, and the proxy must still be able to read it.
|
||||
func setKeyValueLine(path, key, sep, value string) error {
|
||||
return setKeyValueLines(path, sep, [][2]string{{key, value}})
|
||||
}
|
||||
|
||||
// setKeyValueLines is setKeyValueLine for several keys in one rewrite.
|
||||
func setKeyValueLines(path, sep string, kv [][2]string) error {
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -268,17 +803,27 @@ func setKeyValueLine(path, key, sep, value string) error {
|
||||
return err
|
||||
}
|
||||
lines := strings.Split(strings.TrimRight(string(raw), "\n"), "\n")
|
||||
found := false
|
||||
for i, ln := range lines {
|
||||
k, _, ok := strings.Cut(ln, sep)
|
||||
if ok && strings.TrimSpace(k) == key {
|
||||
lines[i] = key + sep + value
|
||||
found = true
|
||||
for _, p := range kv {
|
||||
key, value := p[0], p[1]
|
||||
found := false
|
||||
for i, ln := range lines {
|
||||
k, _, ok := strings.Cut(ln, sep)
|
||||
if ok && strings.TrimSpace(k) == key {
|
||||
lines[i] = key + sep + value
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
lines = append(lines, key+sep+value)
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
lines = append(lines, key+sep+value)
|
||||
}
|
||||
return replaceFileKeepingMode(path, info, []byte(strings.Join(lines, "\n")+"\n"))
|
||||
}
|
||||
|
||||
// replaceFileKeepingMode atomically replaces path with data, keeping the mode and
|
||||
// owner info describes: these files are read by other users (the proxy's) and
|
||||
// some hold credentials, so a rewrite must not widen or narrow who can read them.
|
||||
func replaceFileKeepingMode(path string, info os.FileInfo, data []byte) error {
|
||||
tmp, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".*")
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -294,7 +839,7 @@ func setKeyValueLine(path, key, sep, value string) error {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if _, err := tmp.WriteString(strings.Join(lines, "\n") + "\n"); err != nil {
|
||||
if _, err := tmp.Write(data); err != nil {
|
||||
tmp.Close()
|
||||
return err
|
||||
}
|
||||
|
||||
+659
-59
@@ -3,13 +3,20 @@ package main
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/base64"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
@@ -22,52 +29,111 @@ type rotationRig struct {
|
||||
out *bytes.Buffer
|
||||
events []string
|
||||
dir string
|
||||
clock time.Time
|
||||
}
|
||||
|
||||
func tokenSecret(ns, name, val string) *corev1.Secret {
|
||||
return keySecret(ns, name, "token", val)
|
||||
}
|
||||
|
||||
func keySecret(ns, name, key, val string) *corev1.Secret {
|
||||
return &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name},
|
||||
Data: map[string][]byte{"token": []byte(val)},
|
||||
Data: map[string][]byte{key: []byte(val)},
|
||||
}
|
||||
}
|
||||
|
||||
// serverPod is a game server's pod as the operator labels it.
|
||||
func serverPod(ns, name, server string) *corev1.Pod {
|
||||
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name,
|
||||
Labels: map[string]string{"felis.lolicon.best/server": server}}}
|
||||
Labels: map[string]string{
|
||||
"felis.lolicon.best/server": server,
|
||||
"felis.lolicon.best/managed-by": "felis-operator",
|
||||
"felis.lolicon.best/component": "server",
|
||||
}}}
|
||||
}
|
||||
|
||||
// backupPod is a backup Job's pod as internal/backupjob labels it: it carries
|
||||
// the server's label too, and no rotation may take it for the server's own.
|
||||
func backupPod(ns, name, server string) *corev1.Pod {
|
||||
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name,
|
||||
Labels: map[string]string{
|
||||
"felis.lolicon.best/server": server,
|
||||
"app.kubernetes.io/managed-by": "felis-backup",
|
||||
"app.kubernetes.io/component": "world-backup",
|
||||
}}}
|
||||
}
|
||||
|
||||
func newRotationRig(t *testing.T, objs ...client.Object) *rotationRig {
|
||||
t.Helper()
|
||||
rig := &rotationRig{out: &bytes.Buffer{}, dir: t.TempDir()}
|
||||
rig := &rotationRig{out: &bytes.Buffer{}, dir: t.TempDir(), clock: time.Unix(1_800_000_000, 0)}
|
||||
rig.cl = fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(objs...).Build()
|
||||
rig.r = tokenRotator{
|
||||
cl: rig.cl,
|
||||
controlNS: "felis",
|
||||
minecraftNS: "minecraft",
|
||||
buildNS: "felis-build",
|
||||
secretsEnv: filepath.Join(rig.dir, "secrets.env"),
|
||||
linkProps: filepath.Join(rig.dir, "felis-link.properties"),
|
||||
newToken: func() (string, error) { return "NEWTOKEN", nil },
|
||||
rollAPI: func(context.Context) error {
|
||||
rig.events = append(rig.events, "roll-api")
|
||||
cl: rig.cl,
|
||||
controlNS: "felis",
|
||||
minecraftNS: "minecraft",
|
||||
buildNS: "felis-build",
|
||||
secretsEnv: filepath.Join(rig.dir, "secrets.env"),
|
||||
linkProps: filepath.Join(rig.dir, "felis-link.properties"),
|
||||
forwardingFile: filepath.Join(rig.dir, "forwarding.secret"),
|
||||
hostTOML: filepath.Join(rig.dir, "felis.host.toml"),
|
||||
podTOML: filepath.Join(rig.dir, "felis.pod.toml"),
|
||||
defaultTOML: filepath.Join(rig.dir, "felis.toml"),
|
||||
newToken: func() (string, error) { return "NEWTOKEN", nil },
|
||||
rollout: func(_ context.Context, deployment string) error {
|
||||
rig.events = append(rig.events, "roll "+deployment)
|
||||
return nil
|
||||
},
|
||||
restartUnit: func(_ context.Context, unit string) error {
|
||||
rig.events = append(rig.events, "restart "+unit)
|
||||
return nil
|
||||
},
|
||||
out: rig.out,
|
||||
proxyLog: func(context.Context, time.Time) string { return "" },
|
||||
alterRole: func(context.Context, string, string, string) error {
|
||||
rig.events = append(rig.events, "alter-role")
|
||||
return nil
|
||||
},
|
||||
verifyDB: func(context.Context, string) error {
|
||||
rig.events = append(rig.events, "verify-db")
|
||||
return nil
|
||||
},
|
||||
// Every look at the clock moves it a second on, so a wait measured with it
|
||||
// ends after a known number of looks.
|
||||
now: func() time.Time {
|
||||
rig.clock = rig.clock.Add(time.Second)
|
||||
return rig.clock
|
||||
},
|
||||
reloadWait: 5 * time.Second,
|
||||
out: rig.out,
|
||||
}
|
||||
return rig
|
||||
}
|
||||
|
||||
func (rig *rotationRig) secret(t *testing.T, ns, name string) string {
|
||||
func (rig *rotationRig) secretKey(t *testing.T, ns, name, key string) string {
|
||||
t.Helper()
|
||||
var s corev1.Secret
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
|
||||
return "<missing>"
|
||||
}
|
||||
return string(s.Data["token"])
|
||||
return string(s.Data[key])
|
||||
}
|
||||
|
||||
func (rig *rotationRig) secret(t *testing.T, ns, name string) string {
|
||||
t.Helper()
|
||||
return rig.secretKey(t, ns, name, "token")
|
||||
}
|
||||
|
||||
func (rig *rotationRig) pods(t *testing.T) string {
|
||||
t.Helper()
|
||||
var pods corev1.PodList
|
||||
if err := rig.cl.List(context.Background(), &pods, client.InNamespace("minecraft")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
names := make([]string, len(pods.Items))
|
||||
for i, p := range pods.Items {
|
||||
names[i] = p.Name
|
||||
}
|
||||
return strings.Join(names, ",")
|
||||
}
|
||||
|
||||
func writeTestFile(t *testing.T, path, body string, mode os.FileMode) {
|
||||
@@ -80,6 +146,15 @@ func writeTestFile(t *testing.T, path, body string, mode os.FileMode) {
|
||||
}
|
||||
}
|
||||
|
||||
func readTestFile(t *testing.T, path string) string {
|
||||
t.Helper()
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(raw)
|
||||
}
|
||||
|
||||
func TestRotateLimboToken(t *testing.T) {
|
||||
rig := newRotationRig(t,
|
||||
tokenSecret("felis", "felis-limbo-token", "old"),
|
||||
@@ -87,14 +162,15 @@ func TestRotateLimboToken(t *testing.T) {
|
||||
tokenSecret("felis", "felis-service-token", "proxy"),
|
||||
serverPod("minecraft", "login-0", "login"),
|
||||
serverPod("minecraft", "survival-0", "survival"),
|
||||
backupPod("minecraft", "login-backup-x", "login"),
|
||||
)
|
||||
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=old\nOPS_TOKEN=ops\n", 0o600)
|
||||
|
||||
// felis-api must roll only after both copies hold the new value, and the login
|
||||
// pod must still be there then: restarting it earlier would have it present
|
||||
// the new token to an api that does not know it yet.
|
||||
rig.r.rollAPI = func(context.Context) error {
|
||||
rig.events = append(rig.events, "roll-api")
|
||||
rig.r.rollout = func(_ context.Context, deployment string) error {
|
||||
rig.events = append(rig.events, "roll "+deployment)
|
||||
if got := rig.secret(t, "felis", "felis-limbo-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("api rolled while the control Secret held %q", got)
|
||||
}
|
||||
@@ -107,13 +183,12 @@ func TestRotateLimboToken(t *testing.T) {
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), "limbo"); err != nil {
|
||||
if err := rig.r.rotate(context.Background(), "limbo", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
raw, _ := os.ReadFile(rig.r.secretsEnv)
|
||||
if string(raw) != "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=NEWTOKEN\nOPS_TOKEN=ops\n" {
|
||||
t.Errorf("secrets.env = %q", raw)
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=NEWTOKEN\nOPS_TOKEN=ops\n" {
|
||||
t.Errorf("secrets.env = %q", got)
|
||||
}
|
||||
if info, _ := os.Stat(rig.r.secretsEnv); info.Mode().Perm() != 0o600 {
|
||||
t.Errorf("secrets.env mode = %v, want 0600", info.Mode().Perm())
|
||||
@@ -121,14 +196,12 @@ func TestRotateLimboToken(t *testing.T) {
|
||||
if got := rig.secret(t, "felis", "felis-service-token"); got != "proxy" {
|
||||
t.Errorf("the proxy's token changed to %q", got)
|
||||
}
|
||||
var pods corev1.PodList
|
||||
if err := rig.cl.List(context.Background(), &pods, client.InNamespace("minecraft")); err != nil {
|
||||
t.Fatal(err)
|
||||
// The login pod restarted; a user server and the backup Job's pod, which
|
||||
// carries the login server's label too, are untouched.
|
||||
if got := rig.pods(t); got != "login-backup-x,survival-0" {
|
||||
t.Errorf("pods left = %s, want login-backup-x,survival-0", got)
|
||||
}
|
||||
if len(pods.Items) != 1 || pods.Items[0].Name != "survival-0" {
|
||||
t.Errorf("pods left = %v, want only survival-0 (the login pod restarted, user servers untouched)", pods.Items)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api" {
|
||||
if strings.Join(rig.events, ",") != "roll felis-api" {
|
||||
t.Errorf("events = %v, want only the api roll (no unit restart for limbo)", rig.events)
|
||||
}
|
||||
if strings.Contains(rig.out.String(), "NEWTOKEN") {
|
||||
@@ -136,21 +209,56 @@ func TestRotateLimboToken(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateVelocityTokenOnTheHostProxy(t *testing.T) {
|
||||
const testLinkProps = "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=old\nroot-domain=example.com\n"
|
||||
|
||||
func reloadLine(token string) string {
|
||||
return "[12:00:00 INFO] [felis-link]: Felis: service-token reloaded from /x/felis-link.properties (fingerprint " + tokenFingerprint(token) + ")\n"
|
||||
}
|
||||
|
||||
// The host proxy re-reads its token: the rotation waits for it to say so and
|
||||
// leaves it running.
|
||||
func TestRotateVelocityTokenReloadsTheHostProxy(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=old\n", 0o600)
|
||||
writeTestFile(t, rig.r.linkProps, "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=old\nroot-domain=example.com\n", 0o640)
|
||||
|
||||
if err := rig.r.rotate(context.Background(), "velocity"); err != nil {
|
||||
writeTestFile(t, rig.r.linkProps, testLinkProps, 0o640)
|
||||
written := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)
|
||||
if err := os.Chtimes(rig.r.linkProps, written, written); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, _ := os.ReadFile(rig.r.linkProps)
|
||||
if string(raw) != "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=NEWTOKEN\nroot-domain=example.com\n" {
|
||||
t.Errorf("felis-link.properties = %q", raw)
|
||||
|
||||
// The proxy logs its reload on its next call to felis-api, which may be
|
||||
// while the api rolls: the log must be read from before that.
|
||||
var reloadedAt time.Time
|
||||
rig.r.rollout = func(_ context.Context, deployment string) error {
|
||||
rig.events = append(rig.events, "roll "+deployment)
|
||||
if got := readTestFile(t, rig.r.linkProps); !strings.Contains(got, "service-token=NEWTOKEN\n") {
|
||||
t.Errorf("api rolled before the proxy's file held the new token: %q", got)
|
||||
}
|
||||
reloadedAt = rig.clock
|
||||
return nil
|
||||
}
|
||||
if info, _ := os.Stat(rig.r.linkProps); info.Mode().Perm() != 0o640 {
|
||||
looks := 0
|
||||
rig.r.proxyLog = func(_ context.Context, since time.Time) string {
|
||||
looks++
|
||||
log := reloadLine("old") // an earlier rotation's
|
||||
if looks >= 3 && !since.After(reloadedAt) {
|
||||
log += reloadLine("NEWTOKEN")
|
||||
}
|
||||
return log
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), "velocity", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.linkProps); got != "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=NEWTOKEN\nroot-domain=example.com\n" {
|
||||
t.Errorf("felis-link.properties = %q", got)
|
||||
}
|
||||
info, _ := os.Stat(rig.r.linkProps)
|
||||
if info.Mode().Perm() != 0o640 {
|
||||
t.Errorf("properties mode = %v, want 0640 (the proxy's group must still read it)", info.Mode().Perm())
|
||||
}
|
||||
if !info.ModTime().Equal(written) {
|
||||
t.Errorf("properties mtime = %v, want it kept at %v (felis domain check would read the proxy as stale)", info.ModTime(), written)
|
||||
}
|
||||
if got := rig.secret(t, "felis", "felis-service-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
@@ -158,9 +266,49 @@ func TestRotateVelocityTokenOnTheHostProxy(t *testing.T) {
|
||||
if got := rig.secret(t, "minecraft", "felis-service-token"); got != "<missing>" {
|
||||
t.Errorf("rotation copied the proxy token into minecraft (%q)", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
|
||||
if strings.Join(rig.events, ",") != "roll felis-api" {
|
||||
t.Errorf("events = %v, want the api roll and no proxy restart", rig.events)
|
||||
}
|
||||
if looks != 3 {
|
||||
t.Errorf("the log was read %d times, want 3 (until the line appeared)", looks)
|
||||
}
|
||||
if out := rig.out.String(); !strings.Contains(out, "felis-velocity: took the new token from its properties; players stayed connected") {
|
||||
t.Errorf("output does not say the players stayed: %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// A proxy that never logs the new fingerprint (an older plugin, a stuck proxy)
|
||||
// is restarted once the wait is over.
|
||||
func TestRotateVelocityTokenRestartsAProxyThatDoesNotReload(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=old\n", 0o600)
|
||||
writeTestFile(t, rig.r.linkProps, testLinkProps, 0o640)
|
||||
looks := 0
|
||||
rig.r.proxyLog = func(context.Context, time.Time) string {
|
||||
looks++
|
||||
return reloadLine("old")
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), "velocity", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll felis-api,restart felis-velocity" {
|
||||
t.Errorf("events = %v, want the api roll then the proxy restart", rig.events)
|
||||
}
|
||||
// reloadWait is 5s and each look moves the clock a second.
|
||||
if looks != 5 {
|
||||
t.Errorf("the log was read %d times, want 5", looks)
|
||||
}
|
||||
if out := rig.out.String(); !strings.Contains(out, "felis-velocity: had not taken the new token within 5s, restarted") {
|
||||
t.Errorf("output does not explain the restart: %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// The proxy and the Java plugin name a token by the same fingerprint
|
||||
// (plugins/shared LinkConfigLoaderTest fileTokenFollowsTheFile).
|
||||
func TestTokenFingerprintMatchesThePlugin(t *testing.T) {
|
||||
if got := tokenFingerprint("new-token"); got != "348e9df2a42b" {
|
||||
t.Errorf("tokenFingerprint(new-token) = %q, want 348e9df2a42b", got)
|
||||
}
|
||||
}
|
||||
|
||||
// An external proxy has no felis-link.properties here: the Secret still rotates,
|
||||
@@ -168,13 +316,13 @@ func TestRotateVelocityTokenOnTheHostProxy(t *testing.T) {
|
||||
// without printing it.
|
||||
func TestRotateVelocityTokenForAnExternalProxy(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
|
||||
if err := rig.r.rotate(context.Background(), "velocity"); err != nil {
|
||||
if err := rig.r.rotate(context.Background(), "velocity", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := rig.secret(t, "felis", "felis-service-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api" {
|
||||
if strings.Join(rig.events, ",") != "roll felis-api" {
|
||||
t.Errorf("events = %v, want no proxy restart", rig.events)
|
||||
}
|
||||
out := rig.out.String()
|
||||
@@ -187,7 +335,7 @@ func TestRotateBuildTokenReachesTheBuildNamespace(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-build-token", "old"))
|
||||
// An install from before per-caller tokens has no BUILD_TOKEN line yet.
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=proxy", 0o600)
|
||||
if err := rig.r.rotate(context.Background(), "build"); err != nil {
|
||||
if err := rig.r.rotate(context.Background(), "build", true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := rig.secret(t, "felis-build", "felis-build-token"); got != "NEWTOKEN" {
|
||||
@@ -196,9 +344,8 @@ func TestRotateBuildTokenReachesTheBuildNamespace(t *testing.T) {
|
||||
if got := rig.secret(t, "felis", "felis-build-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
raw, _ := os.ReadFile(rig.r.secretsEnv)
|
||||
if string(raw) != "SERVICE_TOKEN=proxy\nBUILD_TOKEN=NEWTOKEN\n" {
|
||||
t.Errorf("secrets.env = %q", raw)
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "SERVICE_TOKEN=proxy\nBUILD_TOKEN=NEWTOKEN\n" {
|
||||
t.Errorf("secrets.env = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -209,31 +356,453 @@ func TestRotateStopsWhenTheAPIDoesNotRoll(t *testing.T) {
|
||||
tokenSecret("felis", "felis-limbo-token", "old"),
|
||||
serverPod("minecraft", "login-0", "login"),
|
||||
)
|
||||
rig.r.rollAPI = func(context.Context) error { return errors.New("rollout timed out") }
|
||||
err := rig.r.rotate(context.Background(), "limbo")
|
||||
rig.r.rollout = func(context.Context, string) error { return errors.New("rollout timed out") }
|
||||
err := rig.r.rotate(context.Background(), "limbo", true)
|
||||
if err == nil || !strings.Contains(err.Error(), "rollout timed out") {
|
||||
t.Fatalf("err = %v, want the rollout failure", err)
|
||||
}
|
||||
var pod corev1.Pod
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "login-0"}, &pod); err != nil {
|
||||
t.Error("the login pod was restarted although the api never rolled")
|
||||
if got := rig.pods(t); got != "login-0" {
|
||||
t.Errorf("pods left = %s: the login pod was restarted although the api never rolled", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateRefusesAnUnknownCaller(t *testing.T) {
|
||||
func TestRotateRefusesAnUnknownCredential(t *testing.T) {
|
||||
rig := newRotationRig(t)
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=proxy\n", 0o600)
|
||||
if err := rig.r.rotate(context.Background(), "admin"); err == nil {
|
||||
t.Fatal("rotated a token for an unknown caller")
|
||||
if err := rig.r.rotate(context.Background(), "admin", true); err == nil {
|
||||
t.Fatal("rotated an unknown credential")
|
||||
}
|
||||
if raw, _ := os.ReadFile(rig.r.secretsEnv); string(raw) != "SERVICE_TOKEN=proxy\n" {
|
||||
t.Errorf("secrets.env = %q, want it untouched", raw)
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "SERVICE_TOKEN=proxy\n" {
|
||||
t.Errorf("secrets.env = %q, want it untouched", got)
|
||||
}
|
||||
if len(rig.events) != 0 {
|
||||
t.Errorf("events = %v, want nothing touched", rig.events)
|
||||
}
|
||||
}
|
||||
|
||||
const (
|
||||
testHostTOML = "# Written by the installer.\n[database]\nurl = \"postgres://felis:[email protected]:30432/felis?sslmode=disable\"\ndeployment = \"felis/felis-postgres\"\n\n[server]\nroot_domain = \"example.com\"\n"
|
||||
testPodTOML = "[database]\n# The Service address.\nurl = \"postgres://felis:[email protected]:5432/felis?sslmode=disable\"\n\n[server]\nroot_domain = \"example.com\"\n"
|
||||
)
|
||||
|
||||
// installedRig is a host as the installer leaves it, with every credential in
|
||||
// place, for the plan test.
|
||||
func installedRig(t *testing.T) *rotationRig {
|
||||
t.Helper()
|
||||
rig := newRotationRig(t,
|
||||
tokenSecret("felis", "felis-service-token", "old"),
|
||||
tokenSecret("felis", "felis-limbo-token", "old"),
|
||||
tokenSecret("minecraft", "felis-limbo-token", "old"),
|
||||
tokenSecret("felis", "felis-build-token", "old"),
|
||||
tokenSecret("felis-build", "felis-build-token", "old"),
|
||||
tokenSecret("felis", "felis-ops-token", "old"),
|
||||
keySecret("felis", "felis-registry-auth", "platform", "old"),
|
||||
keySecret("felis-build", "felis-registry-push", "password", "old"),
|
||||
keySecret("felis", "felis-forwarding-secret", "secret", "old"),
|
||||
keySecret("minecraft", "felis-forwarding-secret", "secret", "old"),
|
||||
keySecret("felis", "felis-config", "felis.toml", testPodTOML),
|
||||
keySecret("minecraft", "felis-config", "felis.toml", testPodTOML),
|
||||
serverPod("minecraft", "login-0", "login"),
|
||||
serverPod("minecraft", "survival-0", "survival"),
|
||||
)
|
||||
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=oldpw\nSERVICE_TOKEN=old\n", 0o600)
|
||||
writeTestFile(t, rig.r.linkProps, testLinkProps, 0o640)
|
||||
writeTestFile(t, rig.r.forwardingFile, "old", 0o640)
|
||||
writeTestFile(t, rig.r.hostTOML, testHostTOML, 0o600)
|
||||
writeTestFile(t, rig.r.podTOML, testPodTOML, 0o600)
|
||||
return rig
|
||||
}
|
||||
|
||||
// Without -yes every rotation prints its plan and changes nothing.
|
||||
func TestRotatePlanChangesNothing(t *testing.T) {
|
||||
for _, kind := range rotationKinds() {
|
||||
t.Run(kind, func(t *testing.T) {
|
||||
rig := installedRig(t)
|
||||
files := map[string]string{}
|
||||
for _, p := range []string{rig.r.secretsEnv, rig.r.linkProps, rig.r.forwardingFile, rig.r.hostTOML, rig.r.podTOML} {
|
||||
files[p] = readTestFile(t, p)
|
||||
}
|
||||
rig.r.newToken = func() (string, error) {
|
||||
t.Error("a plan generated a token")
|
||||
return "NEWTOKEN", nil
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), kind, false); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for p, before := range files {
|
||||
if got := readTestFile(t, p); got != before {
|
||||
t.Errorf("%s changed to %q", filepath.Base(p), got)
|
||||
}
|
||||
}
|
||||
for _, s := range [][3]string{
|
||||
{"felis", "felis-service-token", "token"}, {"felis", "felis-limbo-token", "token"},
|
||||
{"felis-build", "felis-build-token", "token"}, {"felis", "felis-ops-token", "token"},
|
||||
{"felis", "felis-registry-auth", "platform"}, {"felis-build", "felis-registry-push", "password"},
|
||||
{"minecraft", "felis-forwarding-secret", "secret"},
|
||||
} {
|
||||
if got := rig.secretKey(t, s[0], s[1], s[2]); got != "old" {
|
||||
t.Errorf("Secret %s/%s changed to %q", s[0], s[1], got)
|
||||
}
|
||||
}
|
||||
if got := rig.secretKey(t, "minecraft", "felis-config", "felis.toml"); got != testPodTOML {
|
||||
t.Errorf("felis-config changed to %q", got)
|
||||
}
|
||||
if len(rig.events) != 0 {
|
||||
t.Errorf("events = %v, want none", rig.events)
|
||||
}
|
||||
if got := rig.pods(t); got != "login-0,survival-0" {
|
||||
t.Errorf("pods left = %s, want all", got)
|
||||
}
|
||||
out := rig.out.String()
|
||||
if !strings.HasSuffix(out, "\nNothing was changed. To rotate: sudo felis rotate-token -yes "+kind+"\n") {
|
||||
t.Errorf("the plan does not end with how to go ahead: %s", out)
|
||||
}
|
||||
// Each plan names the interruption it causes.
|
||||
want := apiRestartNote
|
||||
if kind == kindForwarding {
|
||||
want = "everyone online is disconnected"
|
||||
}
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("the plan does not say %q: %s", want, out)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateRegistryTokens(t *testing.T) {
|
||||
rig := newRotationRig(t,
|
||||
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis", Name: "felis-registry-auth"},
|
||||
Data: map[string][]byte{"platform": []byte("a"), "build": []byte("b"), "prune": []byte("c")}},
|
||||
&corev1.Secret{ObjectMeta: metav1.ObjectMeta{Namespace: "felis-build", Name: "felis-registry-push"},
|
||||
Data: map[string][]byte{"username": []byte("build"), "password": []byte("b")}},
|
||||
)
|
||||
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nREGISTRY_PLATFORM_TOKEN=a\nREGISTRY_BUILD_TOKEN=b\nREGISTRY_PRUNE_TOKEN=c\n", 0o600)
|
||||
n := 0
|
||||
rig.r.newToken = func() (string, error) {
|
||||
n++
|
||||
return fmt.Sprintf("NEWTOKEN%d", n), nil
|
||||
}
|
||||
// The registry reads its tokens as it starts: it restarts after the Secret
|
||||
// holds them, and felis-api (the prune token) after that.
|
||||
rig.r.rollout = func(_ context.Context, deployment string) error {
|
||||
rig.events = append(rig.events, "roll "+deployment)
|
||||
if got := rig.secretKey(t, "felis", "felis-registry-auth", "prune"); got != "NEWTOKEN3" {
|
||||
t.Errorf("%s rolled while the prune token was %q", deployment, got)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), kindRegistry, true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=db\nREGISTRY_PLATFORM_TOKEN=NEWTOKEN1\nREGISTRY_BUILD_TOKEN=NEWTOKEN2\nREGISTRY_PRUNE_TOKEN=NEWTOKEN3\n" {
|
||||
t.Errorf("secrets.env = %q", got)
|
||||
}
|
||||
for key, want := range map[string]string{"platform": "NEWTOKEN1", "build": "NEWTOKEN2", "prune": "NEWTOKEN3"} {
|
||||
if got := rig.secretKey(t, "felis", "felis-registry-auth", key); got != want {
|
||||
t.Errorf("felis-registry-auth %s = %q, want %s", key, got, want)
|
||||
}
|
||||
}
|
||||
if got := rig.secretKey(t, "felis-build", "felis-registry-push", "password"); got != "NEWTOKEN2" {
|
||||
t.Errorf("the build Jobs' push password = %q, want the build token", got)
|
||||
}
|
||||
if got := rig.secretKey(t, "felis-build", "felis-registry-push", "username"); got != "build" {
|
||||
t.Errorf("the build Jobs' push username = %q, want build", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll registry,roll felis-api" {
|
||||
t.Errorf("events = %v, want the registry then felis-api", rig.events)
|
||||
}
|
||||
if strings.Contains(rig.out.String(), "NEWTOKEN") {
|
||||
t.Errorf("a new token was printed: %s", rig.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateForwardingSecret(t *testing.T) {
|
||||
lock := time.Unix(1_800_000_000, 0)
|
||||
rig := newRotationRig(t,
|
||||
keySecret("felis", "felis-forwarding-secret", "secret", "old"),
|
||||
keySecret("minecraft", "felis-forwarding-secret", "secret", "old"),
|
||||
serverPod("minecraft", "survival-0", "survival"),
|
||||
serverPod("minecraft", "lobby-0", "lobby"),
|
||||
// creative is being backed up: a running Job holds its world.
|
||||
serverPod("minecraft", "creative-0", "creative"),
|
||||
&batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "creative-backup",
|
||||
Labels: map[string]string{maintenance.LabelServer: "creative", maintenance.LabelManagedBy: "felis-backup"}}},
|
||||
// survival's last backup is over; its finished pod stays until the Job's TTL.
|
||||
&batchv1.Job{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "survival-backup",
|
||||
Labels: map[string]string{maintenance.LabelServer: "survival", maintenance.LabelManagedBy: "felis-backup"}},
|
||||
Status: batchv1.JobStatus{Conditions: []batchv1.JobCondition{{Type: batchv1.JobComplete, Status: corev1.ConditionTrue}}}},
|
||||
backupPod("minecraft", "survival-backup-x", "survival"),
|
||||
// A pod already on its way out is neither counted nor deleted again.
|
||||
func() *corev1.Pod {
|
||||
p := serverPod("minecraft", "lobby-old", "lobby")
|
||||
p.DeletionTimestamp = &metav1.Time{Time: lock}
|
||||
p.Finalizers = []string{"felis.lolicon.best/test"}
|
||||
return p
|
||||
}(),
|
||||
// skyblock's restore has just been admitted, its Job not created yet.
|
||||
serverPod("minecraft", "skyblock-0", "skyblock"),
|
||||
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "skyblock",
|
||||
Annotations: map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindRestore, lock)}}},
|
||||
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Namespace: "minecraft", Name: "survival"}},
|
||||
)
|
||||
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nFORWARDING_SECRET=old\n", 0o600)
|
||||
writeTestFile(t, rig.r.forwardingFile, "old", 0o640)
|
||||
// The proxy restarts after the servers were told to: a player who rejoins
|
||||
// must not meet a server still on the old secret.
|
||||
rig.r.restartUnit = func(_ context.Context, unit string) error {
|
||||
rig.events = append(rig.events, "restart "+unit)
|
||||
if got := rig.pods(t); strings.Contains(got, "survival-0") {
|
||||
t.Errorf("the proxy restarted before the servers (pods %s)", got)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.forwardingFile); got != "NEWTOKEN" {
|
||||
t.Errorf("the proxy restarted on forwarding.secret %q", got)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), kindForwarding, true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=db\nFORWARDING_SECRET=NEWTOKEN\n" {
|
||||
t.Errorf("secrets.env = %q", got)
|
||||
}
|
||||
if info, _ := os.Stat(rig.r.forwardingFile); info.Mode().Perm() != 0o640 {
|
||||
t.Errorf("forwarding.secret mode = %v, want 0640", info.Mode().Perm())
|
||||
}
|
||||
for _, ns := range []string{"felis", "minecraft"} {
|
||||
if got := rig.secretKey(t, ns, "felis-forwarding-secret", "secret"); got != "NEWTOKEN" {
|
||||
t.Errorf("Secret %s/felis-forwarding-secret = %q, want NEWTOKEN", ns, got)
|
||||
}
|
||||
}
|
||||
if got := rig.pods(t); got != "creative-0,lobby-old,skyblock-0,survival-backup-x" {
|
||||
t.Errorf("pods left = %s, want the held servers, the terminating pod and the backup's pod", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "restart felis-velocity" {
|
||||
t.Errorf("events = %v, want only the proxy restart (felis-api does not hold this secret)", rig.events)
|
||||
}
|
||||
out := rig.out.String()
|
||||
for _, want := range []string{
|
||||
"every running server restarts to read it (2 now)",
|
||||
"left running, because a backup, restore or file write holds its world: creative (backup), skyblock (restore).",
|
||||
" - servers: 2 restarting on the new secret\n",
|
||||
} {
|
||||
if !strings.Contains(out, want) {
|
||||
t.Errorf("output lacks %q: %s", want, out)
|
||||
}
|
||||
}
|
||||
if strings.Contains(out, "NEWTOKEN") {
|
||||
t.Errorf("the new secret was printed: %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateForwardingSecretForAnExternalProxy(t *testing.T) {
|
||||
rig := newRotationRig(t, keySecret("felis", "felis-forwarding-secret", "secret", "old"))
|
||||
if err := rig.r.rotate(context.Background(), kindForwarding, true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := rig.secretKey(t, "felis", "felis-forwarding-secret", "secret"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
if len(rig.events) != 0 {
|
||||
t.Errorf("events = %v, want no proxy restart on this host", rig.events)
|
||||
}
|
||||
if out := rig.out.String(); !strings.Contains(out, "put the value in Secret felis/felis-forwarding-secret into your proxy's forwarding secret file") {
|
||||
t.Errorf("output does not say where the value is: %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// dbRig is an installed host for the database rotation: felis.toml links to the
|
||||
// host copy, as the installer makes it.
|
||||
func dbRig(t *testing.T) *rotationRig {
|
||||
t.Helper()
|
||||
rig := newRotationRig(t,
|
||||
keySecret("felis", "felis-config", "felis.toml", testPodTOML),
|
||||
keySecret("minecraft", "felis-config", "felis.toml", testPodTOML),
|
||||
)
|
||||
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=oldpw\nSERVICE_TOKEN=s\n", 0o600)
|
||||
writeTestFile(t, rig.r.hostTOML, testHostTOML, 0o600)
|
||||
writeTestFile(t, rig.r.podTOML, testPodTOML, 0o600)
|
||||
if err := os.Symlink("felis.host.toml", rig.r.defaultTOML); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return rig
|
||||
}
|
||||
|
||||
const (
|
||||
wantHostTOML = "# Written by the installer.\n[database]\nurl = \"postgres://felis:[email protected]:30432/felis?sslmode=disable\"\ndeployment = \"felis/felis-postgres\"\n\n[server]\nroot_domain = \"example.com\"\n"
|
||||
wantPodTOML = "[database]\n# The Service address.\nurl = \"postgres://felis:[email protected]:5432/felis?sslmode=disable\"\n\n[server]\nroot_domain = \"example.com\"\n"
|
||||
)
|
||||
|
||||
func TestRotateDatabasePassword(t *testing.T) {
|
||||
rig := dbRig(t)
|
||||
rig.r.alterRole = func(_ context.Context, deployment, role, verifier string) error {
|
||||
rig.events = append(rig.events, "alter-role")
|
||||
if deployment != "felis/felis-postgres" || role != "felis" {
|
||||
t.Errorf("alterRole(%q, %q), want felis/felis-postgres and felis", deployment, role)
|
||||
}
|
||||
if !strings.Contains(readTestFile(t, rig.r.secretsEnv), "DB_PASSWORD=NEWTOKEN\n") {
|
||||
t.Error("the role changed before secrets.env recorded the new password")
|
||||
}
|
||||
// The verifier is the new password's, under the salt it carries.
|
||||
m := regexp.MustCompile(`^SCRAM-SHA-256\$4096:([^$]+)\$`).FindStringSubmatch(verifier)
|
||||
if m == nil {
|
||||
t.Fatalf("verifier %q is not a SCRAM-SHA-256 verifier", verifier)
|
||||
}
|
||||
salt, err := base64.StdEncoding.DecodeString(m[1])
|
||||
if err != nil || len(salt) != 16 {
|
||||
t.Fatalf("verifier salt %q: %v", m[1], err)
|
||||
}
|
||||
if want, _ := scramVerifier("NEWTOKEN", salt, 4096); verifier != want {
|
||||
t.Errorf("verifier = %q, want %q", verifier, want)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
// The configs change only once the database accepts the new password.
|
||||
rig.r.verifyDB = func(_ context.Context, url string) error {
|
||||
rig.events = append(rig.events, "verify-db")
|
||||
if url != "postgres://felis:[email protected]:30432/felis?sslmode=disable" {
|
||||
t.Errorf("verified with %q, want the host URL with the new password", url)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.hostTOML); got != testHostTOML {
|
||||
t.Errorf("the host config changed before the database accepted the password: %q", got)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
rig.r.rollout = func(_ context.Context, deployment string) error {
|
||||
rig.events = append(rig.events, "roll "+deployment)
|
||||
if got := rig.secretKey(t, "felis", "felis-config", "felis.toml"); got != wantPodTOML {
|
||||
t.Errorf("felis-api rolled on felis-config %q", got)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), kindDB, true); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=NEWTOKEN\nSERVICE_TOKEN=s\n" {
|
||||
t.Errorf("secrets.env = %q", got)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.hostTOML); got != wantHostTOML {
|
||||
t.Errorf("host config = %q", got)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.podTOML); got != wantPodTOML {
|
||||
t.Errorf("pod config = %q", got)
|
||||
}
|
||||
if info, err := os.Lstat(rig.r.defaultTOML); err != nil || info.Mode()&os.ModeSymlink == 0 {
|
||||
t.Errorf("felis.toml is no longer the link to the host copy (%v)", err)
|
||||
}
|
||||
for _, ns := range []string{"felis", "minecraft"} {
|
||||
if got := rig.secretKey(t, ns, "felis-config", "felis.toml"); got != wantPodTOML {
|
||||
t.Errorf("Secret %s/felis-config = %q", ns, got)
|
||||
}
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "alter-role,verify-db,roll felis-api" {
|
||||
t.Errorf("events = %v", rig.events)
|
||||
}
|
||||
if strings.Contains(rig.out.String(), "NEWTOKEN") {
|
||||
t.Errorf("the new password was printed: %s", rig.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
// A password the database does not accept leaves the configs on the old one
|
||||
// and felis-api running: the installer's record, already new, is the way back.
|
||||
func TestRotateDatabasePasswordStopsWhenTheDatabaseRefuses(t *testing.T) {
|
||||
rig := dbRig(t)
|
||||
rig.r.verifyDB = func(context.Context, string) error {
|
||||
rig.events = append(rig.events, "verify-db")
|
||||
return errors.New("password authentication failed")
|
||||
}
|
||||
err := rig.r.rotate(context.Background(), kindDB, true)
|
||||
if err == nil || !strings.Contains(err.Error(), "password authentication failed") || !strings.Contains(err.Error(), "run the installer again") {
|
||||
t.Fatalf("err = %v, want the refusal and the way back", err)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.hostTOML); got != testHostTOML {
|
||||
t.Errorf("host config = %q, want it unchanged", got)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.podTOML); got != testPodTOML {
|
||||
t.Errorf("pod config = %q, want it unchanged", got)
|
||||
}
|
||||
if got := rig.secretKey(t, "felis", "felis-config", "felis.toml"); got != testPodTOML {
|
||||
t.Errorf("felis-config = %q, want it unchanged", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "alter-role,verify-db" {
|
||||
t.Errorf("events = %v, want no api roll", rig.events)
|
||||
}
|
||||
}
|
||||
|
||||
// Everything that can refuse does so before the installer's record or the
|
||||
// role change; what the host config alone shows is refused by the plan too.
|
||||
func TestRotateDatabasePasswordRefusesEarly(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
apply bool
|
||||
setup func(t *testing.T, rig *rotationRig)
|
||||
want string
|
||||
}{
|
||||
{"a database the installer does not run", false, func(t *testing.T, rig *rotationRig) {
|
||||
writeTestFile(t, rig.r.hostTOML, strings.Replace(testHostTOML, "deployment = \"felis/felis-postgres\"\n", "", 1), 0o600)
|
||||
}, "[database] deployment is unset"},
|
||||
{"a database url without a role", false, func(t *testing.T, rig *rotationRig) {
|
||||
writeTestFile(t, rig.r.hostTOML, strings.Replace(testHostTOML, "felis:oldpw@", "", 1), 0o600)
|
||||
}, "names no role"},
|
||||
{"a config copy the line editor cannot edit", true, func(t *testing.T, rig *rotationRig) {
|
||||
writeTestFile(t, rig.r.podTOML, "[database]\nurl = \"\"\"\npostgres://felis:[email protected]:5432/felis\"\"\"\n", 0o600)
|
||||
}, "set the password in its [database] url by hand"},
|
||||
{"a pod copy that is the host copy", true, func(t *testing.T, rig *rotationRig) {
|
||||
if err := os.Remove(rig.r.podTOML); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.Symlink("felis.host.toml", rig.r.podTOML); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}, "resolves to the same file as"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
rig := dbRig(t)
|
||||
tc.setup(t, rig)
|
||||
err := rig.r.rotate(context.Background(), kindDB, tc.apply)
|
||||
if err == nil || !strings.Contains(err.Error(), tc.want) {
|
||||
t.Fatalf("err = %v, want %q", err, tc.want)
|
||||
}
|
||||
if got := readTestFile(t, rig.r.secretsEnv); got != "DB_PASSWORD=oldpw\nSERVICE_TOKEN=s\n" {
|
||||
t.Errorf("secrets.env = %q, want it untouched", got)
|
||||
}
|
||||
if len(rig.events) != 0 {
|
||||
t.Errorf("events = %v, want nothing done", rig.events)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The RFC 7677 example (user "user", password "pencil"), with the stored and
|
||||
// server keys computed by openssl's PBKDF2 and HMAC rather than this code.
|
||||
func TestScramVerifier(t *testing.T) {
|
||||
salt, _ := base64.StdEncoding.DecodeString("W22ZaJ0SNY7soEsUEjb6gQ==")
|
||||
got, err := scramVerifier("pencil", salt, 4096)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if want := "SCRAM-SHA-256$4096:W22ZaJ0SNY7soEsUEjb6gQ==$WG5d8oPm3OtcPnkdi4Uo7BkeZkBFzpcXkuLmtbsT4qY=:wfPLwcE6nTWhTAmQ7tl2KeoiWGPlZqQxSrmfPwDl2dU="; got != want {
|
||||
t.Errorf("scramVerifier = %q\nwant %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAlterRoleSQL(t *testing.T) {
|
||||
if got := alterRoleSQL(`fe"lis`, "SCRAM-SHA-256$4096:c2FsdA==$a:b"); got != "ALTER ROLE \"fe\"\"lis\" WITH PASSWORD 'SCRAM-SHA-256$4096:c2FsdA==$a:b';\n" {
|
||||
t.Errorf("alterRoleSQL = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// installerSecretKeys is every secrets.env key a rotation writes.
|
||||
func installerSecretKeys() map[string]bool {
|
||||
keys := map[string]bool{installerForwardingKey: true, installerDBKey: true}
|
||||
for _, k := range installerTokenKeys {
|
||||
keys[k] = true
|
||||
}
|
||||
for _, k := range installerRegistryKeys {
|
||||
keys[k] = true
|
||||
}
|
||||
return keys
|
||||
}
|
||||
|
||||
// The installer and rotate-token must agree on where each caller's token lives:
|
||||
// rotate-token writes installerTokenKeys into secrets.env, and the installer
|
||||
// applies those same keys to the Secrets on its next run. A key the installer
|
||||
@@ -249,12 +818,6 @@ func TestInstallerProvisionsEveryCallerToken(t *testing.T) {
|
||||
if !ok {
|
||||
t.Fatalf("installerTokenKeys names unknown caller %q", caller)
|
||||
}
|
||||
if !strings.Contains(script, key+`="${`+key+`:-$(openssl rand -hex 32)}"`) {
|
||||
t.Errorf("bootstrap.sh does not generate %s", key)
|
||||
}
|
||||
if !strings.Contains(script, key+"=${"+key+"}\n") {
|
||||
t.Errorf("bootstrap.sh does not persist %s to secrets.env", key)
|
||||
}
|
||||
apply := regexp.MustCompile(`apply_literal_secret "\$CONTROL_NS" ` + regexp.QuoteMeta(ct.Secret) + ` token "\$` + key + `"`)
|
||||
if !apply.MatchString(script) {
|
||||
t.Errorf("bootstrap.sh does not apply %s from %s in the control namespace", ct.Secret, key)
|
||||
@@ -281,3 +844,40 @@ func TestInstallerProvisionsEveryCallerToken(t *testing.T) {
|
||||
t.Errorf("installerTokenKeys covers %d callers, naming.CallerTokens lists %d", len(installerTokenKeys), len(callerNames()))
|
||||
}
|
||||
}
|
||||
|
||||
// Every value the installer keeps in secrets.env has a rotation that rewrites
|
||||
// that very key, and every key a rotation writes is one the installer
|
||||
// generates, keeps and so re-applies on its next run.
|
||||
func TestInstallerSecretsAreAllRotatable(t *testing.T) {
|
||||
raw, err := os.ReadFile(filepath.Join("..", "..", "deploy", "bootstrap.sh"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
script := string(raw)
|
||||
m := regexp.MustCompile(`(?s)write_file_atomic "\$SECRETS_ENV" 0600 <<EOF\n(.*?)\nEOF\n`).FindStringSubmatch(script)
|
||||
if m == nil {
|
||||
t.Fatal("bootstrap.sh: no secrets.env heredoc found")
|
||||
}
|
||||
persisted := map[string]bool{}
|
||||
for _, line := range strings.Split(m[1], "\n") {
|
||||
key, value, _ := strings.Cut(line, "=")
|
||||
if value != "${"+key+"}" {
|
||||
t.Errorf("secrets.env line %q is not KEY=${KEY}", line)
|
||||
}
|
||||
persisted[key] = true
|
||||
}
|
||||
rotated := installerSecretKeys()
|
||||
for key := range persisted {
|
||||
if !rotated[key] {
|
||||
t.Errorf("secrets.env keeps %s, which no rotation replaces", key)
|
||||
}
|
||||
}
|
||||
for key := range rotated {
|
||||
if !persisted[key] {
|
||||
t.Errorf("a rotation writes %s, which the installer does not keep in secrets.env", key)
|
||||
}
|
||||
if !regexp.MustCompile(`\n ` + key + `="\$\{` + key + `:-\$\(openssl rand -hex [0-9]+\)\}"\n`).MatchString(script) {
|
||||
t.Errorf("bootstrap.sh does not generate %s when secrets.env lacks it", key)
|
||||
}
|
||||
}
|
||||
}
|
||||
+16
-3
@@ -20,8 +20,10 @@ Commands:
|
||||
reaper Run the world reaper / backup batch
|
||||
restore Extract a world archive into a world volume (internal Job entrypoint)
|
||||
backup Archive a world into the backup store and record it (internal Job entrypoint)
|
||||
files List/read/write one file in a stopped server's world (internal Job entrypoint)
|
||||
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
|
||||
backup-now Archive every user server's world now, one at a time (or the named ones; -stop stops running ones first; prints the plan, -yes applies; requires root/sudo)
|
||||
files List, read, write, mkdir, delete, rename, upload or unzip one path in a stopped server's world (internal Job entrypoint)
|
||||
export Archive a stopped server's world, or read one of its backups, and hand it to felis-api for download (internal Job entrypoint)
|
||||
egress-gate Hold a build or game server pod until its egress NetworkPolicy is enforced (internal init container entrypoint)
|
||||
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
|
||||
scan-gate Apply the scan policy to a build's Trivy report and hand felis-api the report and SBOM (internal Job entrypoint)
|
||||
push-image Push a scanned image tarball to the registry (internal Job entrypoint)
|
||||
@@ -31,7 +33,11 @@ Commands:
|
||||
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
||||
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
||||
converge Fill in fields a newer desired spec added to already-installed system servers
|
||||
rotate-token Replace one internal caller's token and restart what holds it (velocity|limbo|build|ops; requires root/sudo)
|
||||
rotate-token Replace a generated credential and restart what reads it (velocity|limbo|build|ops|registry|forwarding|db; prints the plan, -yes applies; requires root/sudo)
|
||||
domain Move the install to a new root domain on every surface that carries it, or check each one (set|check; requires root/sudo)
|
||||
status Print the platform at a glance: node, control plane, proxy, servers, backups, host, open alerts (requires root/sudo)
|
||||
doctor Run every health check once, grouped by area, with where to look next; mails nothing (requires root/sudo)
|
||||
support-bundle Collect status, doctor, logs and cluster state into one redacted tar.gz to share when asking for help (requires root/sudo)
|
||||
watchdog Check the platform once and mail the owners what has gone wrong (run by felis-watchdog.timer)
|
||||
version Print the build stamp of this binary
|
||||
update Report which platform components have updates available
|
||||
@@ -59,7 +65,9 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||
"reaper": cmdReaper,
|
||||
"restore": cmdRestore,
|
||||
"backup": cmdBackup,
|
||||
"backup-now": cmdBackupNow,
|
||||
"files": cmdFiles,
|
||||
"export": cmdExport,
|
||||
"egress-gate": cmdEgressGate,
|
||||
"fetch-context": cmdFetchContext,
|
||||
"scan-gate": cmdScanGate,
|
||||
@@ -71,14 +79,19 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||
"setup": cmdSetup,
|
||||
"converge": cmdConverge,
|
||||
"rotate-token": cmdRotateToken,
|
||||
"domain": cmdDomain,
|
||||
"breakGlass": cmdBreakGlass,
|
||||
"bootstrap-assets": cmdBootstrapAssets,
|
||||
"init-forwarding": cmdInitForwarding,
|
||||
"init-volume": cmdInitVolume,
|
||||
"pin-images": cmdPinImages,
|
||||
"image-bundle": cmdImageBundle,
|
||||
"version": cmdVersion,
|
||||
"update": cmdUpdate,
|
||||
"watchdog": cmdWatchdog,
|
||||
"status": cmdStatus,
|
||||
"doctor": cmdDoctor,
|
||||
"support-bundle": cmdSupportBundle,
|
||||
}
|
||||
|
||||
// run dispatches a subcommand. It is separate from main so the router is
|
||||
|
||||
@@ -43,7 +43,7 @@ func TestRunUnknownCommand(t *testing.T) {
|
||||
// decision rather than an oversight.
|
||||
var undocumentedCommands = map[string]bool{
|
||||
"bootstrap-assets": true, "init-forwarding": true, "init-volume": true,
|
||||
"pin-images": true,
|
||||
"pin-images": true, "image-bundle": true,
|
||||
}
|
||||
|
||||
// The usage text and the dispatch table must describe the same set of commands.
|
||||
|
||||
+1
-1
@@ -112,7 +112,7 @@ func cmdSetup(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
res, err := runSetupTUI(ctx, setup.repo, setup.cfg.Database.URL, setup.cfg.Server.RootDomain, setup.cfg.Auth.AdminHostname, setup.cfg.Auth.PanelHostname, setup.cfg.Auth.AccessJWTAud, setup.cfg.K8s.Namespace, accountableOSUser(), setup.adminExists)
|
||||
res, err := runSetupTUI(ctx, setup.repo, setup.cfg.Database, setup.cfg.Server.RootDomain, setup.cfg.Auth.AdminHostname, setup.cfg.Auth.PanelHostname, setup.cfg.Auth.AccessJWTAud, setup.cfg.K8s.Namespace, accountableOSUser(), setup.adminExists)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis setup: %v\n", err)
|
||||
return 1
|
||||
|
||||
@@ -0,0 +1,438 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
"text/tabwriter"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/labels"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// cmdStatus prints the platform at a glance: the release and the node, the
|
||||
// control plane's workloads, the game proxy, every server with its players
|
||||
// and newest world backup, the database backups and the off-site copy, the
|
||||
// host's disks and memory, and what the watchdog has open. It changes
|
||||
// nothing, and a part that is down (the cluster, PostgreSQL) reads as such
|
||||
// while the rest still prints. felis doctor says what is wrong and where to
|
||||
// look.
|
||||
func cmdStatus(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("status", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
unitDir := fs.String("systemd-dir", systemdUnitDir, "where the installer's systemd units are")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis status: run as root (sudo felis status): it reads root-only state under /etc/felis and /var/lib/felis")
|
||||
return 1
|
||||
}
|
||||
env, err := hostStatusEnv(*unitDir)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis status: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), time.Minute)
|
||||
defer cancel()
|
||||
printStatus(ctx, env, stdout)
|
||||
return 0
|
||||
}
|
||||
|
||||
// statusEnv is what one status report reads the host through.
|
||||
type statusEnv struct {
|
||||
cfg *config.Config
|
||||
// w is how felis-watchdog.service runs the watchdog: where the backups,
|
||||
// the off-site record and the proxy are.
|
||||
w watchdogFlags
|
||||
cl client.Client // nil while the API server is unreachable
|
||||
clErr error
|
||||
backups func(ctx context.Context) (map[string]time.Time, error)
|
||||
run func(ctx context.Context, name string, args ...string) ([]byte, error)
|
||||
unitDir string
|
||||
meminfo string
|
||||
host string
|
||||
now time.Time
|
||||
}
|
||||
|
||||
// hostStatusEnv reads this host: the watchdog's settings, the configuration
|
||||
// they name, and the cluster.
|
||||
func hostStatusEnv(unitDir string) (statusEnv, error) {
|
||||
w, _, err := watchdogUnitFlags(filepath.Join(unitDir, "felis-watchdog.service"))
|
||||
if err != nil {
|
||||
return statusEnv{}, err
|
||||
}
|
||||
cfg, err := config.Load(w.cfgPath)
|
||||
if err != nil {
|
||||
return statusEnv{}, err
|
||||
}
|
||||
cl, clErr := buildSystemServerClient()
|
||||
if clErr != nil {
|
||||
cl = nil
|
||||
}
|
||||
host, _ := os.Hostname()
|
||||
return statusEnv{
|
||||
cfg: cfg, w: w, cl: cl, clErr: clErr,
|
||||
backups: func(ctx context.Context) (map[string]time.Time, error) {
|
||||
return newestWorldBackups(ctx, cfg.Database.URL)
|
||||
},
|
||||
run: hostCommand, unitDir: unitDir, meminfo: "/proc/meminfo", host: host, now: time.Now(),
|
||||
}, nil
|
||||
}
|
||||
|
||||
func printStatus(ctx context.Context, env statusEnv, out io.Writer) {
|
||||
fmt.Fprintf(out, "felis %s on %s at %s\n", resolvedVersion(), env.host, env.now.UTC().Format("2006-01-02 15:04 UTC"))
|
||||
if env.cl == nil {
|
||||
fmt.Fprintf(out, "cluster: unreachable (%v)\n", env.clErr)
|
||||
} else {
|
||||
statusCluster(ctx, env, out)
|
||||
}
|
||||
statusProxy(ctx, env, out)
|
||||
statusServers(ctx, env, out)
|
||||
statusBackups(env, out)
|
||||
statusHost(env, out)
|
||||
statusWatchdog(ctx, env, out)
|
||||
}
|
||||
|
||||
func statusCluster(ctx context.Context, env statusEnv, out io.Writer) {
|
||||
var nodes corev1.NodeList
|
||||
if err := env.cl.List(ctx, &nodes); err != nil {
|
||||
fmt.Fprintf(out, "cluster: unreachable (%v)\n", err)
|
||||
return
|
||||
}
|
||||
for _, n := range nodes.Items {
|
||||
ready := "NotReady"
|
||||
for _, c := range n.Status.Conditions {
|
||||
if c.Type == corev1.NodeReady && c.Status == corev1.ConditionTrue {
|
||||
ready = "Ready"
|
||||
}
|
||||
}
|
||||
info := n.Status.NodeInfo
|
||||
fmt.Fprintf(out, "node: %s %s, k3s %s, %s, kernel %s\n", n.Name, ready, info.KubeletVersion, info.OSImage, info.KernelVersion)
|
||||
}
|
||||
|
||||
ns := env.w.controlNS
|
||||
var deps appsv1.DeploymentList
|
||||
var pods corev1.PodList
|
||||
err := env.cl.List(ctx, &deps, client.InNamespace(ns))
|
||||
if err == nil {
|
||||
err = env.cl.List(ctx, &pods, client.InNamespace(ns))
|
||||
}
|
||||
fmt.Fprintf(out, "\ncontrol plane (namespace %s):\n", ns)
|
||||
if err != nil {
|
||||
fmt.Fprintf(out, " cannot list it: %v\n", err)
|
||||
return
|
||||
}
|
||||
sort.Slice(deps.Items, func(i, j int) bool { return deps.Items[i].Name < deps.Items[j].Name })
|
||||
tw := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0)
|
||||
for _, d := range deps.Items {
|
||||
want := int32(1)
|
||||
if d.Spec.Replicas != nil {
|
||||
want = *d.Spec.Replicas
|
||||
}
|
||||
image := "-"
|
||||
if cs := d.Spec.Template.Spec.Containers; len(cs) > 0 {
|
||||
image = shortImage(cs[0].Image)
|
||||
}
|
||||
fmt.Fprintf(tw, " %s\t%d/%d ready\t%s\trestarts %d\n", d.Name, d.Status.ReadyReplicas, want, image, podRestarts(d.Spec.Selector, pods.Items))
|
||||
}
|
||||
tw.Flush()
|
||||
}
|
||||
|
||||
// podRestarts adds up the container restarts of the pods selector picks (the
|
||||
// API server refuses a Deployment whose selector is empty).
|
||||
func podRestarts(selector *metav1.LabelSelector, pods []corev1.Pod) int32 {
|
||||
sel, err := metav1.LabelSelectorAsSelector(selector)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
var n int32
|
||||
for _, p := range pods {
|
||||
if !sel.Matches(labels.Set(p.Labels)) {
|
||||
continue
|
||||
}
|
||||
for _, cs := range p.Status.ContainerStatuses {
|
||||
n += cs.RestartCount
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// shortImage is an image reference without its registry and repository path,
|
||||
// and with its digest cut to 12 characters.
|
||||
func shortImage(ref string) string {
|
||||
if i := strings.LastIndex(ref, "/"); i >= 0 {
|
||||
ref = ref[i+1:]
|
||||
}
|
||||
if name, digest, ok := strings.Cut(ref, "@sha256:"); ok && len(digest) > 12 {
|
||||
ref = name + "@" + digest[:12]
|
||||
}
|
||||
return ref
|
||||
}
|
||||
|
||||
func statusProxy(ctx context.Context, env statusEnv, out io.Writer) {
|
||||
var parts []string
|
||||
if _, err := os.Stat(filepath.Join(env.unitDir, "felis-velocity.service")); err == nil {
|
||||
parts = append(parts, "felis-velocity "+unitActiveState(ctx, doctorEnv{run: env.run}, "felis-velocity.service"))
|
||||
}
|
||||
if env.w.proxyAddr != "" {
|
||||
d := net.Dialer{Timeout: 3 * time.Second}
|
||||
if conn, err := d.DialContext(ctx, "tcp", env.w.proxyAddr); err != nil {
|
||||
parts = append(parts, fmt.Sprintf("%s refuses connections (%v)", env.w.proxyAddr, err))
|
||||
} else {
|
||||
conn.Close()
|
||||
parts = append(parts, env.w.proxyAddr+" accepts connections")
|
||||
}
|
||||
}
|
||||
if len(parts) == 0 {
|
||||
parts = append(parts, "not on this host")
|
||||
}
|
||||
fmt.Fprintf(out, "\nproxy: %s\n", strings.Join(parts, ", "))
|
||||
}
|
||||
|
||||
func statusServers(ctx context.Context, env statusEnv, out io.Writer) {
|
||||
ns := env.cfg.K8s.Namespace
|
||||
if ns == "" {
|
||||
ns = platform.DefaultMinecraftNamespace
|
||||
}
|
||||
if env.cl == nil {
|
||||
fmt.Fprintf(out, "\nservers (namespace %s): unknown while the cluster is unreachable\n", ns)
|
||||
return
|
||||
}
|
||||
var list v1alpha1.MinecraftServerList
|
||||
if err := env.cl.List(ctx, &list, client.InNamespace(ns)); err != nil {
|
||||
fmt.Fprintf(out, "\nservers (namespace %s): cannot list them: %v\n", ns, err)
|
||||
return
|
||||
}
|
||||
newest, backupErr := env.backups(ctx)
|
||||
running, online := 0, int32(0)
|
||||
for _, ms := range list.Items {
|
||||
if ms.Status.Phase == v1alpha1.PhaseRunning {
|
||||
running++
|
||||
online += ms.Status.Players.Online
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(out, "\nservers (namespace %s): %d, %d running, %d players online\n", ns, len(list.Items), running, online)
|
||||
if len(list.Items) == 0 {
|
||||
return
|
||||
}
|
||||
sort.Slice(list.Items, func(i, j int) bool { return list.Items[i].Name < list.Items[j].Name })
|
||||
tw := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0)
|
||||
fmt.Fprintln(tw, " NAME\tROLE\tDESIRED\tPHASE\tPLAYERS\tNEWEST WORLD BACKUP")
|
||||
for _, ms := range list.Items {
|
||||
role := ms.Labels[v1alpha1.LabelSystemRole]
|
||||
if role == "" {
|
||||
role = "-"
|
||||
}
|
||||
phase := string(ms.Status.Phase)
|
||||
if phase == "" {
|
||||
phase = "-"
|
||||
}
|
||||
players := "-"
|
||||
if ms.Status.Phase == v1alpha1.PhaseRunning {
|
||||
players = fmt.Sprintf("%d/%d", ms.Status.Players.Online, ms.Status.Players.Max)
|
||||
}
|
||||
backup := "none"
|
||||
switch at, ok := newest[ms.Name]; {
|
||||
case backupErr != nil:
|
||||
backup = "?"
|
||||
case ok:
|
||||
backup = dbbackup.Age(env.now.Sub(at)) + " ago"
|
||||
}
|
||||
fmt.Fprintf(tw, " %s\t%s\t%s\t%s\t%s\t%s\n", ms.Name, role, orDash(string(ms.Spec.DesiredState)), phase, players, backup)
|
||||
}
|
||||
tw.Flush()
|
||||
if backupErr != nil {
|
||||
fmt.Fprintf(out, " world backups unknown: %v\n", backupErr)
|
||||
}
|
||||
}
|
||||
|
||||
func orDash(s string) string {
|
||||
if s == "" {
|
||||
return "-"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// newestWorldBackups is when each server's newest world backup that a restore
|
||||
// can use was taken.
|
||||
func newestWorldBackups(ctx context.Context, url string) (map[string]time.Time, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
|
||||
defer cancel()
|
||||
drv, err := store.Open(ctx, url)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer drv.Close()
|
||||
rows, err := drv.DB().QueryContext(ctx, `SELECT server_name, max(created_at) FROM world_backups
|
||||
WHERE status = 'present' AND corrupt_at IS NULL GROUP BY server_name`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
out := map[string]time.Time{}
|
||||
for rows.Next() {
|
||||
var name string
|
||||
var at time.Time
|
||||
if err := rows.Scan(&name, &at); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out[name] = at
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
func statusBackups(env statusEnv, out io.Writer) {
|
||||
fmt.Fprintln(out, "\nbackups:")
|
||||
switch bundles, err := dbbackup.List(env.w.backupDir); {
|
||||
case env.w.backupDir == "":
|
||||
fmt.Fprintln(out, " database: not checked (felis-watchdog.service names no -backup-dir)")
|
||||
case err != nil:
|
||||
fmt.Fprintf(out, " database: cannot read %s: %v\n", env.w.backupDir, err)
|
||||
case len(bundles) == 0:
|
||||
fmt.Fprintf(out, " database: none in %s\n", env.w.backupDir)
|
||||
default:
|
||||
fmt.Fprintf(out, " database: newest %s, %s ago; %d bundles in %s\n",
|
||||
bundles[0].Name, dbbackup.Age(env.now.Sub(bundles[0].Created)), len(bundles), env.w.backupDir)
|
||||
}
|
||||
if !env.cfg.Offsite.Enabled() {
|
||||
fmt.Fprintln(out, " off-site: not configured, every backup is on this machine only")
|
||||
return
|
||||
}
|
||||
switch st, err := offsite.ReadStatus(env.w.offsiteStatus); {
|
||||
case err != nil:
|
||||
fmt.Fprintf(out, " off-site: %v\n", err)
|
||||
case st == nil:
|
||||
fmt.Fprintln(out, " off-site: never synced")
|
||||
case st.LastSuccess.IsZero():
|
||||
fmt.Fprintf(out, " off-site: never succeeded; last attempt %s ago: %s\n", dbbackup.Age(env.now.Sub(st.LastAttempt)), st.LastError)
|
||||
default:
|
||||
line := fmt.Sprintf(" off-site: last good sync %s ago to %s", dbbackup.Age(env.now.Sub(st.LastSuccess)), st.Bucket)
|
||||
if st.LastAttempt.After(st.LastSuccess) { // a failed run records no success
|
||||
line += fmt.Sprintf("; the last attempt, %s ago, failed: %s", dbbackup.Age(env.now.Sub(st.LastAttempt)), st.LastError)
|
||||
}
|
||||
fmt.Fprintln(out, line)
|
||||
}
|
||||
}
|
||||
|
||||
func statusHost(env statusEnv, out io.Writer) {
|
||||
fmt.Fprintln(out, "\nhost:")
|
||||
seen := map[uint64]bool{}
|
||||
for _, p := range splitList(env.w.diskPaths) {
|
||||
var st syscall.Stat_t
|
||||
if err := syscall.Stat(p, &st); err != nil {
|
||||
continue
|
||||
}
|
||||
dev := uint64(st.Dev) // int32 on darwin
|
||||
if seen[dev] {
|
||||
continue
|
||||
}
|
||||
seen[dev] = true
|
||||
var fs syscall.Statfs_t
|
||||
if err := syscall.Statfs(p, &fs); err != nil || fs.Blocks == 0 {
|
||||
continue
|
||||
}
|
||||
bsize := uint64(fs.Bsize) // uint32 on darwin
|
||||
total, avail := uint64(fs.Blocks)*bsize, uint64(fs.Bavail)*bsize
|
||||
fmt.Fprintf(out, " disk %s: %s free of %s (%.0f%% free)\n", p, offsite.HumanBytes(int64(avail)), offsite.HumanBytes(int64(total)), float64(fs.Bavail)/float64(fs.Blocks)*100)
|
||||
}
|
||||
if total, avail, ok := readMeminfo(env.meminfo); ok {
|
||||
fmt.Fprintf(out, " memory: %s available of %s\n", offsite.HumanBytes(int64(avail)), offsite.HumanBytes(int64(total)))
|
||||
}
|
||||
}
|
||||
|
||||
// readMeminfo reads MemTotal and MemAvailable, in bytes, from a /proc/meminfo
|
||||
// style file.
|
||||
func readMeminfo(path string) (total, avail uint64, ok bool) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return 0, 0, false
|
||||
}
|
||||
defer f.Close()
|
||||
sc := bufio.NewScanner(f)
|
||||
for sc.Scan() {
|
||||
fields := strings.Fields(sc.Text())
|
||||
if len(fields) < 2 {
|
||||
continue
|
||||
}
|
||||
v, _ := strconv.ParseUint(fields[1], 10, 64) // the kernel writes numbers
|
||||
switch fields[0] {
|
||||
case "MemTotal:":
|
||||
total = v * 1024
|
||||
case "MemAvailable:":
|
||||
avail = v * 1024
|
||||
}
|
||||
}
|
||||
return total, avail, total > 0
|
||||
}
|
||||
|
||||
func statusWatchdog(ctx context.Context, env statusEnv, out io.Writer) {
|
||||
fmt.Fprintln(out, "\nwatchdog:")
|
||||
if _, err := os.Stat(filepath.Join(env.unitDir, "felis-watchdog.timer")); err != nil {
|
||||
fmt.Fprintln(out, " not installed: nothing checks this host")
|
||||
return
|
||||
}
|
||||
line := " timer " + unitActiveState(ctx, doctorEnv{run: env.run}, "felis-watchdog.timer")
|
||||
show, _ := env.run(ctx, "systemctl", "show", "--timestamp=unix", "-p", "Result", "-p", "ExecMainExitTimestamp", "felis-watchdog.service")
|
||||
props := map[string]string{}
|
||||
for _, ln := range strings.Split(string(show), "\n") {
|
||||
if k, v, ok := strings.Cut(strings.TrimSpace(ln), "="); ok {
|
||||
props[k] = v
|
||||
}
|
||||
}
|
||||
// ExecMainExitTimestamp is empty until the service has run.
|
||||
if sec, _ := strconv.ParseInt(strings.TrimPrefix(props["ExecMainExitTimestamp"], "@"), 10, 64); sec > 0 {
|
||||
line += fmt.Sprintf(", last run %s ago (%s)", dbbackup.Age(env.now.Sub(time.Unix(sec, 0))), orDash(props["Result"]))
|
||||
}
|
||||
fmt.Fprintln(out, line)
|
||||
|
||||
state, err := watchdog.LoadState(watchdog.NewestState(env.w.statePath, env.w.fallbackState))
|
||||
if err != nil {
|
||||
fmt.Fprintf(out, " alerts: unknown (%v)\n", err)
|
||||
return
|
||||
}
|
||||
var open []string
|
||||
for key, a := range state.Alerts {
|
||||
if !a.ClearedAt.IsZero() {
|
||||
continue
|
||||
}
|
||||
if a.Notified.IsZero() {
|
||||
open = append(open, fmt.Sprintf("%s (%s, seen %s ago, not mailed yet)", key, a.Severity, dbbackup.Age(env.now.Sub(a.FirstSeen))))
|
||||
} else {
|
||||
open = append(open, fmt.Sprintf("%s (%s, mailed %s ago)", key, a.Severity, dbbackup.Age(env.now.Sub(a.Notified))))
|
||||
}
|
||||
}
|
||||
if len(open) == 0 {
|
||||
fmt.Fprintln(out, " alerts: none open")
|
||||
return
|
||||
}
|
||||
sort.Strings(open)
|
||||
fmt.Fprintf(out, " alerts: %d open (sudo felis doctor says where to look)\n", len(open))
|
||||
for _, o := range open {
|
||||
fmt.Fprintf(out, " %s\n", o)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,490 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
|
||||
)
|
||||
|
||||
// statusCluster is a one-node install: felis-api with a restarted pod, three
|
||||
// servers (one of them the lobby), and a pod of another app whose restarts
|
||||
// are not felis-api's.
|
||||
func statusClusterObjects() []client.Object {
|
||||
replicas := int32(1)
|
||||
apiLabels := map[string]string{"app": "felis-api"}
|
||||
return []client.Object{
|
||||
&corev1.Node{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-1"},
|
||||
Status: corev1.NodeStatus{
|
||||
Conditions: []corev1.NodeCondition{{Type: corev1.NodeReady, Status: corev1.ConditionTrue}},
|
||||
NodeInfo: corev1.NodeSystemInfo{KubeletVersion: "v1.36.4+k3s1", OSImage: "CentOS Stream 9", KernelVersion: "5.14.0-630.el9.aarch64"},
|
||||
},
|
||||
},
|
||||
&appsv1.Deployment{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-api", Namespace: "felis"},
|
||||
Spec: appsv1.DeploymentSpec{
|
||||
Replicas: &replicas,
|
||||
Selector: &metav1.LabelSelector{MatchLabels: apiLabels},
|
||||
Template: corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{{
|
||||
Name: "api", Image: "registry.felis.svc:5000/felis/felis-api:v1.4.0@sha256:0123456789abcdef0123456789abcdef",
|
||||
}}}},
|
||||
},
|
||||
Status: appsv1.DeploymentStatus{ReadyReplicas: 1},
|
||||
},
|
||||
&corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-api-7d9", Namespace: "felis", Labels: apiLabels},
|
||||
Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{Name: "api", RestartCount: 2}}},
|
||||
},
|
||||
&corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "other-1", Namespace: "felis", Labels: map[string]string{"app": "other"}},
|
||||
Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{Name: "x", RestartCount: 5}}},
|
||||
},
|
||||
&v1alpha1.MinecraftServer{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"},
|
||||
Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredRunning},
|
||||
Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseRunning, Players: v1alpha1.PlayersStatus{Online: 2, Max: 20}},
|
||||
},
|
||||
&v1alpha1.MinecraftServer{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "creative", Namespace: "minecraft"},
|
||||
Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredStopped},
|
||||
Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStopped},
|
||||
},
|
||||
&v1alpha1.MinecraftServer{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "lobby", Namespace: "minecraft", Labels: map[string]string{v1alpha1.LabelSystemRole: "lobby"}},
|
||||
Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredRunning},
|
||||
Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStarting},
|
||||
},
|
||||
// Another namespace's server is not this install's.
|
||||
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Name: "elsewhere", Namespace: "other"}},
|
||||
}
|
||||
}
|
||||
|
||||
// statusTestEnv is a host with the cluster above, a database backup nine hours
|
||||
// old, an off-site copy last good two hours ago, and one alert open.
|
||||
func statusTestEnv(t *testing.T) statusEnv {
|
||||
t.Helper()
|
||||
now := time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC)
|
||||
dir := t.TempDir()
|
||||
var w watchdogFlags
|
||||
w.controlNS = "felis"
|
||||
w.backupDir = filepath.Join(dir, "db-backups")
|
||||
w.offsiteStatus = filepath.Join(dir, "offsite-status.json")
|
||||
w.statePath = filepath.Join(dir, "state.json")
|
||||
w.fallbackState = filepath.Join(dir, "fallback.json")
|
||||
w.diskPaths = "/nonexistent-felis-status-test"
|
||||
unitDir := filepath.Join(dir, "systemd")
|
||||
for _, d := range []string{w.backupDir, unitDir} {
|
||||
if err := os.MkdirAll(d, 0o700); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
writeTestFile(t, filepath.Join(w.backupDir, dbbackup.BundleName(now.Add(-9*time.Hour), dbbackup.LabelDaily)), "x", 0o600)
|
||||
writeTestFile(t, filepath.Join(w.backupDir, dbbackup.BundleName(now.Add(-33*time.Hour), dbbackup.LabelDaily)), "x", 0o600)
|
||||
if err := offsite.WriteStatus(w.offsiteStatus, offsite.Status{
|
||||
LastAttempt: now.Add(-2 * time.Hour), LastSuccess: now.Add(-2 * time.Hour), Bucket: "felis-dr", Endpoint: "https://s3.example.com",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
state := &watchdog.State{Alerts: map[string]*watchdog.Alert{
|
||||
"db-backup-servers": {Finding: watchdog.Finding{Key: "db-backup-servers", Severity: watchdog.Warning}, FirstSeen: now.Add(-3 * time.Hour), Notified: now.Add(-90 * time.Minute)},
|
||||
"proxy": {Finding: watchdog.Finding{Key: "proxy", Severity: watchdog.Critical}, FirstSeen: now.Add(-time.Hour), Notified: now.Add(-time.Hour), ClearedAt: now.Add(-10 * time.Minute)},
|
||||
"disk//var": {Finding: watchdog.Finding{Key: "disk//var", Severity: watchdog.Critical}, FirstSeen: now.Add(-4 * time.Minute)},
|
||||
}}
|
||||
if err := watchdog.SaveState(w.statePath, state); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
meminfo := filepath.Join(dir, "meminfo")
|
||||
writeTestFile(t, meminfo, "MemTotal: 8000000 kB\nMemFree: 100000 kB\nMemAvailable: 2000000 kB\n", 0o644)
|
||||
writeTestFile(t, filepath.Join(unitDir, "felis-watchdog.timer"), "[Unit]\n", 0o644)
|
||||
cfg := &config.Config{Offsite: config.OffsiteConfig{Endpoint: "https://s3.example.com", Bucket: "felis-dr"}}
|
||||
return statusEnv{
|
||||
cfg: cfg, w: w, cl: fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(statusClusterObjects()...).Build(),
|
||||
backups: func(context.Context) (map[string]time.Time, error) {
|
||||
return map[string]time.Time{"survival": now.Add(-3 * time.Hour)}, nil
|
||||
},
|
||||
run: func(_ context.Context, name string, args ...string) ([]byte, error) {
|
||||
switch strings.Join(append([]string{name}, args...), " ") {
|
||||
case "systemctl is-active felis-watchdog.timer":
|
||||
return []byte("active\n"), nil
|
||||
case "systemctl show --timestamp=unix -p Result -p ExecMainExitTimestamp felis-watchdog.service":
|
||||
return []byte("Result=success\nExecMainExitTimestamp=@" + strconv.FormatInt(now.Add(-70*time.Second).Unix(), 10) + "\n"), nil
|
||||
}
|
||||
return nil, errors.New("unexpected")
|
||||
},
|
||||
unitDir: unitDir, meminfo: meminfo, host: "felis-test", now: now,
|
||||
}
|
||||
}
|
||||
|
||||
// lineFields is the words of the first line of out that starts with prefix
|
||||
// once trimmed.
|
||||
func lineFields(out, prefix string) []string {
|
||||
for _, l := range strings.Split(out, "\n") {
|
||||
if strings.HasPrefix(strings.TrimSpace(l), prefix) {
|
||||
return strings.Fields(l)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestStatusReport(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
var out bytes.Buffer
|
||||
printStatus(context.Background(), env, &out)
|
||||
got := out.String()
|
||||
for _, want := range []string{
|
||||
"on felis-test at 2026-09-27 12:00 UTC\n",
|
||||
"node: felis-1 Ready, k3s v1.36.4+k3s1, CentOS Stream 9, kernel 5.14.0-630.el9.aarch64\n",
|
||||
"\ncontrol plane (namespace felis):\n",
|
||||
"\nproxy: not on this host\n",
|
||||
"\nservers (namespace minecraft): 3, 1 running, 2 players online\n",
|
||||
" database: newest " + dbbackup.BundleName(env.now.Add(-9*time.Hour), dbbackup.LabelDaily) + ", 9h0m ago; 2 bundles in " + env.w.backupDir + "\n",
|
||||
" off-site: last good sync 2h0m ago to felis-dr\n",
|
||||
" memory: 1.9 GiB available of 7.6 GiB\n",
|
||||
" timer active, last run 1m ago (success)\n",
|
||||
" alerts: 2 open (sudo felis doctor says where to look)\n" +
|
||||
" db-backup-servers (warning, mailed 1h30m ago)\n" +
|
||||
" disk//var (critical, seen 4m ago, not mailed yet)\n",
|
||||
} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("status lacks %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
for prefix, want := range map[string]string{
|
||||
"felis-api": "felis-api 1/1 ready felis-api:v1.4.0@0123456789ab restarts 2",
|
||||
"NAME": "NAME ROLE DESIRED PHASE PLAYERS NEWEST WORLD BACKUP",
|
||||
"creative": "creative - Stopped Stopped - none",
|
||||
"lobby": "lobby lobby Running Starting - none",
|
||||
"survival": "survival - Running Running 2/20 3h0m ago",
|
||||
} {
|
||||
if f := strings.Join(lineFields(got, prefix), " "); f != want {
|
||||
t.Errorf("row %q = %q, want %q\n%s", prefix, f, want, got)
|
||||
}
|
||||
}
|
||||
if strings.Contains(got, "elsewhere") || strings.Contains(got, "other-1") {
|
||||
t.Errorf("status shows what is not this install's:\n%s", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Whatever is down reads as down, and the rest of the report still prints.
|
||||
func TestStatusWithPartsDown(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
env.cl, env.clErr = nil, errors.New("connection refused")
|
||||
env.cfg = &config.Config{}
|
||||
var out bytes.Buffer
|
||||
printStatus(context.Background(), env, &out)
|
||||
for _, want := range []string{
|
||||
"cluster: unreachable (connection refused)\n",
|
||||
"\nservers (namespace minecraft): unknown while the cluster is unreachable\n",
|
||||
" off-site: not configured, every backup is on this machine only\n",
|
||||
" timer active, last run",
|
||||
} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("status lacks %q:\n%s", want, out.String())
|
||||
}
|
||||
}
|
||||
|
||||
env = statusTestEnv(t)
|
||||
env.backups = func(context.Context) (map[string]time.Time, error) { return nil, errors.New("postgres is down") }
|
||||
out.Reset()
|
||||
printStatus(context.Background(), env, &out)
|
||||
if f := strings.Join(lineFields(out.String(), "survival"), " "); f != "survival - Running Running 2/20 ?" {
|
||||
t.Errorf("survival with PostgreSQL down = %q", f)
|
||||
}
|
||||
if !strings.Contains(out.String(), " world backups unknown: postgres is down\n") {
|
||||
t.Errorf("status does not say why the backups are unknown:\n%s", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestShortImage(t *testing.T) {
|
||||
for in, want := range map[string]string{
|
||||
"registry.felis.svc:5000/felis/felis-api:v1.4.0": "felis-api:v1.4.0",
|
||||
"docker.io/library/postgres:17@sha256:0123456789abcdef0123": "postgres:17@0123456789ab",
|
||||
"felis-operator@sha256:fedcba9876543210fedcba9876543210fedcba9876543210fedcba98765432": "felis-operator@fedcba987654",
|
||||
"busybox": "busybox",
|
||||
} {
|
||||
if got := shortImage(in); got != want {
|
||||
t.Errorf("shortImage(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A node that is not ready, a Deployment without replicas or containers, and
|
||||
// lists the API server refuses each read as such.
|
||||
func TestStatusClusterEdges(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
env.cfg = &config.Config{K8s: config.K8sConfig{Namespace: "games"}}
|
||||
env.cl = fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(
|
||||
&corev1.Node{ObjectMeta: metav1.ObjectMeta{Name: "felis-1"}, Status: corev1.NodeStatus{
|
||||
Conditions: []corev1.NodeCondition{{Type: corev1.NodeReady, Status: corev1.ConditionFalse}, {Type: corev1.NodeMemoryPressure, Status: corev1.ConditionTrue}}}},
|
||||
&appsv1.Deployment{ObjectMeta: metav1.ObjectMeta{Name: "felis-bare", Namespace: "felis"}},
|
||||
&corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "p", Namespace: "felis"}, Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{RestartCount: 4}}}},
|
||||
).Build()
|
||||
var out bytes.Buffer
|
||||
printStatus(context.Background(), env, &out)
|
||||
got := out.String()
|
||||
for _, want := range []string{
|
||||
"node: felis-1 NotReady, k3s , , kernel \n",
|
||||
"\nservers (namespace games): 0, 0 running, 0 players online\n\nbackups:\n",
|
||||
} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("status lacks %q:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
if f := strings.Join(lineFields(got, "felis-bare"), " "); f != "felis-bare 0/1 ready - restarts 0" {
|
||||
t.Errorf("row felis-bare = %q, want its one replica wanted, no image, and no pod of its own:\n%s", f, got)
|
||||
}
|
||||
|
||||
failing := func(what client.ObjectList) {
|
||||
env.cl = fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(statusClusterObjects()...).
|
||||
WithInterceptorFuncs(interceptor.Funcs{List: func(ctx context.Context, c client.WithWatch, list client.ObjectList, opts ...client.ListOption) error {
|
||||
if fmt.Sprintf("%T", list) == fmt.Sprintf("%T", what) {
|
||||
return errors.New("forbidden")
|
||||
}
|
||||
return c.List(ctx, list, opts...)
|
||||
}}).Build()
|
||||
out.Reset()
|
||||
printStatus(context.Background(), env, &out)
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
list client.ObjectList
|
||||
want string
|
||||
}{
|
||||
{&corev1.NodeList{}, "cluster: unreachable (forbidden)\n\nproxy:"},
|
||||
{&appsv1.DeploymentList{}, "\ncontrol plane (namespace felis):\n cannot list it: forbidden\n\nproxy:"},
|
||||
{&corev1.PodList{}, "\ncontrol plane (namespace felis):\n cannot list it: forbidden\n\nproxy:"},
|
||||
{&v1alpha1.MinecraftServerList{}, "\nservers (namespace games): cannot list them: forbidden\n\nbackups:"},
|
||||
} {
|
||||
failing(tc.list)
|
||||
if !strings.Contains(out.String(), tc.want) {
|
||||
t.Errorf("listing %T refused: status lacks %q:\n%s", tc.list, tc.want, out.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusProxy(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
writeTestFile(t, filepath.Join(env.unitDir, "felis-velocity.service"), "[Unit]\n", 0o644)
|
||||
env.run = fakeSystemctl("", map[string]string{"felis-velocity.service": "active"})
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
env.w.proxyAddr = ln.Addr().String()
|
||||
var out bytes.Buffer
|
||||
statusProxy(context.Background(), env, &out)
|
||||
if want := "\nproxy: felis-velocity active, " + env.w.proxyAddr + " accepts connections\n"; out.String() != want {
|
||||
t.Errorf("proxy listening: %q, want %q", out.String(), want)
|
||||
}
|
||||
ln.Close()
|
||||
out.Reset()
|
||||
statusProxy(context.Background(), env, &out)
|
||||
if want := "\nproxy: felis-velocity active, " + env.w.proxyAddr + " refuses connections (dial tcp " + env.w.proxyAddr + ": "; !strings.HasPrefix(out.String(), want) {
|
||||
t.Errorf("proxy gone: %q, want it to start %q", out.String(), want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusBackups(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
status := func() string {
|
||||
var out bytes.Buffer
|
||||
statusBackups(env, &out)
|
||||
return out.String()
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
what, dir, want string
|
||||
}{
|
||||
{"no -backup-dir", "", " database: not checked (felis-watchdog.service names no -backup-dir)\n"},
|
||||
{"an empty directory", t.TempDir(), " database: none in %s\n"},
|
||||
{"a file where the directory should be", filepath.Join(env.w.backupDir, dbbackup.BundleName(env.now.Add(-9*time.Hour), dbbackup.LabelDaily)), " database: cannot read %s: "},
|
||||
} {
|
||||
env.w.backupDir = tc.dir
|
||||
want := tc.want
|
||||
if strings.Contains(want, "%s") {
|
||||
want = fmt.Sprintf(want, tc.dir)
|
||||
}
|
||||
if got := status(); !strings.Contains(got, want) {
|
||||
t.Errorf("%s: %q, want %q", tc.what, got, want)
|
||||
}
|
||||
}
|
||||
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
status *offsite.Status
|
||||
raw string
|
||||
want string
|
||||
}{
|
||||
{"never synced", nil, "", " off-site: never synced\n"},
|
||||
{"never succeeded", &offsite.Status{LastAttempt: env.now.Add(-time.Hour), LastError: "403 Forbidden"}, "", " off-site: never succeeded; last attempt 1h0m ago: 403 Forbidden\n"},
|
||||
{"the last attempt failed", &offsite.Status{LastAttempt: env.now.Add(-30 * time.Minute), LastSuccess: env.now.Add(-26 * time.Hour), LastError: "timeout", Bucket: "felis-dr"}, "",
|
||||
" off-site: last good sync 26h0m ago to felis-dr; the last attempt, 30m ago, failed: timeout\n"},
|
||||
{"an error an earlier attempt left", &offsite.Status{LastAttempt: env.now.Add(-2 * time.Hour), LastSuccess: env.now.Add(-time.Hour), LastError: "timeout", Bucket: "felis-dr"}, "",
|
||||
" off-site: last good sync 1h0m ago to felis-dr\n"},
|
||||
{"an unreadable record", nil, "{", " off-site: unexpected end of JSON input\n"},
|
||||
} {
|
||||
os.Remove(env.w.offsiteStatus)
|
||||
switch {
|
||||
case tc.status != nil:
|
||||
if err := offsite.WriteStatus(env.w.offsiteStatus, *tc.status); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
case tc.raw != "":
|
||||
writeTestFile(t, env.w.offsiteStatus, tc.raw, 0o600)
|
||||
}
|
||||
if got := status(); !strings.HasSuffix(got, tc.want) {
|
||||
t.Errorf("%s: %q, want it to end %q", tc.what, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusHost(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
dir := t.TempDir()
|
||||
env.w.diskPaths = "/nonexistent-felis-status-test," + dir + "," + dir
|
||||
env.meminfo = filepath.Join(dir, "meminfo")
|
||||
var out bytes.Buffer
|
||||
statusHost(env, &out)
|
||||
lines := strings.Split(strings.TrimSuffix(out.String(), "\n"), "\n")
|
||||
if len(lines) != 3 || lines[0] != "" || lines[1] != "host:" || !strings.HasPrefix(lines[2], " disk "+dir+": ") || !strings.HasSuffix(lines[2], "% free)") {
|
||||
t.Errorf("host: %q, want one line for %s (a path on a disk already shown, and one that is not there, print none) and no memory line", lines, dir)
|
||||
}
|
||||
|
||||
writeTestFile(t, env.meminfo, "MemTotal: 4096 kB\ngarbage\nMemAvailable: 1024 kB\n", 0o644)
|
||||
if total, avail, ok := readMeminfo(env.meminfo); !ok || total != 4096*1024 || avail != 1024*1024 {
|
||||
t.Errorf("readMeminfo = %d, %d, %v", total, avail, ok)
|
||||
}
|
||||
writeTestFile(t, env.meminfo, "MemAvailable: 1024 kB\n", 0o644)
|
||||
if _, _, ok := readMeminfo(env.meminfo); ok {
|
||||
t.Error("readMeminfo without MemTotal: ok")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusWatchdog(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
status := func() string {
|
||||
var out bytes.Buffer
|
||||
statusWatchdog(context.Background(), env, &out)
|
||||
return out.String()
|
||||
}
|
||||
run := env.run
|
||||
env.run = func(ctx context.Context, name string, args ...string) ([]byte, error) {
|
||||
if len(args) > 0 && args[0] == "show" {
|
||||
return []byte("Result=success\nExecMainExitTimestamp=\n"), nil
|
||||
}
|
||||
return run(ctx, name, args...)
|
||||
}
|
||||
if err := watchdog.SaveState(env.w.statePath, &watchdog.State{Alerts: map[string]*watchdog.Alert{
|
||||
"proxy": {Finding: watchdog.Finding{Key: "proxy", Severity: watchdog.Critical}, FirstSeen: env.now.Add(-time.Hour), ClearedAt: env.now.Add(-time.Minute)},
|
||||
}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, want := status(), "\nwatchdog:\n timer active\n alerts: none open\n"; got != want {
|
||||
t.Errorf("a watchdog that has not run and has nothing open: %q, want %q", got, want)
|
||||
}
|
||||
|
||||
writeTestFile(t, env.w.statePath, "{", 0o600)
|
||||
if got := status(); !strings.Contains(got, "\n alerts: unknown (") {
|
||||
t.Errorf("an unreadable state: %q", got)
|
||||
}
|
||||
|
||||
if err := os.Remove(filepath.Join(env.unitDir, "felis-watchdog.timer")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, want := status(), "\nwatchdog:\n not installed: nothing checks this host\n"; got != want {
|
||||
t.Errorf("no watchdog timer: %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// Rows print by name in whatever order the API server lists them, a server
|
||||
// the operator has not reconciled yet reads as dashes, and open alerts print
|
||||
// by key in whatever order the state's map yields them.
|
||||
func TestStatusOrder(t *testing.T) {
|
||||
env := statusTestEnv(t)
|
||||
objs := append(statusClusterObjects(),
|
||||
&appsv1.Deployment{ObjectMeta: metav1.ObjectMeta{Name: "felis-operator", Namespace: "felis"}},
|
||||
&v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Name: "fresh", Namespace: "minecraft"}})
|
||||
env.cl = fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(objs...).
|
||||
WithInterceptorFuncs(interceptor.Funcs{List: func(ctx context.Context, c client.WithWatch, list client.ObjectList, opts ...client.ListOption) error {
|
||||
// The fake lists by name; the other way round, then.
|
||||
if err := c.List(ctx, list, opts...); err != nil {
|
||||
return err
|
||||
}
|
||||
items, err := meta.ExtractList(list)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
slices.Reverse(items)
|
||||
return meta.SetList(list, items)
|
||||
}}).Build()
|
||||
var out bytes.Buffer
|
||||
printStatus(context.Background(), env, &out)
|
||||
var rows []string
|
||||
for _, l := range strings.Split(out.String(), "\n") {
|
||||
if f := strings.Fields(l); len(f) > 0 && (strings.HasPrefix(f[0], "felis-") || slices.Contains([]string{"creative", "fresh", "lobby", "survival"}, f[0])) {
|
||||
rows = append(rows, strings.Join(f, " "))
|
||||
}
|
||||
}
|
||||
want := []string{
|
||||
"felis-api 1/1 ready felis-api:v1.4.0@0123456789ab restarts 2",
|
||||
"felis-operator 0/1 ready - restarts 0",
|
||||
"creative - Stopped Stopped - none",
|
||||
"fresh - - - - none",
|
||||
"lobby lobby Running Starting - none",
|
||||
"survival - Running Running 2/20 3h0m ago",
|
||||
}
|
||||
if !slices.Equal(rows, want) {
|
||||
t.Errorf("rows %q, want %q:\n%s", rows, want, out.String())
|
||||
}
|
||||
|
||||
alerts := map[string]*watchdog.Alert{}
|
||||
for i, key := range []string{"a", "b", "c", "d"} {
|
||||
alerts[key] = &watchdog.Alert{Finding: watchdog.Finding{Key: key, Severity: watchdog.Warning}, FirstSeen: env.now.Add(-time.Duration(i+1) * time.Minute)}
|
||||
}
|
||||
if err := watchdog.SaveState(env.w.statePath, &watchdog.State{Alerts: alerts}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
wantAlerts := " alerts: 4 open (sudo felis doctor says where to look)\n" +
|
||||
" a (warning, seen 1m ago, not mailed yet)\n b (warning, seen 2m ago, not mailed yet)\n" +
|
||||
" c (warning, seen 3m ago, not mailed yet)\n d (warning, seen 4m ago, not mailed yet)\n"
|
||||
// A map starts its walk at random: fifty walks all in order by chance
|
||||
// is out of the question.
|
||||
for range 50 {
|
||||
out.Reset()
|
||||
statusWatchdog(context.Background(), env, &out)
|
||||
if !strings.HasSuffix(out.String(), wantAlerts) {
|
||||
t.Fatalf("alerts %q, want them to end %q", out.String(), wantAlerts)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A selector the API server would have refused picks no pods.
|
||||
func TestPodRestartsBadSelector(t *testing.T) {
|
||||
pods := []corev1.Pod{{Status: corev1.PodStatus{ContainerStatuses: []corev1.ContainerStatus{{RestartCount: 3}}}}}
|
||||
if n := podRestarts(&metav1.LabelSelector{MatchExpressions: []metav1.LabelSelectorRequirement{{Key: "app", Operator: "Near"}}}, pods); n != 0 {
|
||||
t.Errorf("podRestarts with a bad selector = %d, want 0", n)
|
||||
}
|
||||
if n := podRestarts(&metav1.LabelSelector{}, pods); n != 3 {
|
||||
t.Errorf("podRestarts with a selector that picks all = %d, want 3", n)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,862 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"bytes"
|
||||
"compress/gzip"
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"net/url"
|
||||
"os"
|
||||
"path"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"text/tabwriter"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
"k8s.io/apimachinery/pkg/api/meta"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
// supportBundleDir is where felis support-bundle writes unless told otherwise.
|
||||
const supportBundleDir = "/var/lib/felis/support"
|
||||
|
||||
// redacted stands in for every value the bundle leaves out.
|
||||
const redacted = "<redacted>"
|
||||
|
||||
// minScrubLen is the shortest known secret value the bundle searches for: a
|
||||
// shorter one would blank ordinary words, and every secret the installer
|
||||
// generates is far longer.
|
||||
const minScrubLen = 8
|
||||
|
||||
// hostSecretSources are the files on a Felis host that hold secrets; the
|
||||
// bundle reads them only to take each value out of what it collects.
|
||||
var hostSecretSources = secretSources{
|
||||
envFiles: []string{"/etc/felis/secrets.env", defaultOffsiteEnvFile},
|
||||
valueFiles: []string{hostSMTPPasswordPath, hostUploadsS3AccessKeyPath, hostUploadsS3SecretKeyPath, defaultHeartbeatFile, "/opt/felis/velocity/forwarding.secret"},
|
||||
propsFiles: []string{"/opt/felis/velocity/plugins/felis-link/felis-link.properties"},
|
||||
tokenFiles: []string{"/var/lib/rancher/k3s/server/token", "/var/lib/rancher/k3s/server/agent-token"},
|
||||
}
|
||||
|
||||
// cmdSupportBundle collects what someone helping with this host needs into
|
||||
// one tar.gz: felis status and felis doctor, the logs of the control plane,
|
||||
// the builds and the systemd units, the cluster's workloads and events, and a
|
||||
// summary of the configuration. It never collects a Secret, a ConfigMap, the
|
||||
// contents of a configuration file, the database or a world, and takes every
|
||||
// value of the host's secret files out of what it does collect. Game server
|
||||
// logs, which carry player names, addresses and chat, only with -server-logs.
|
||||
func cmdSupportBundle(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("support-bundle", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
outDir := fs.String("o", supportBundleDir, "directory to write the bundle to (created mode 0700 when missing)")
|
||||
logLines := fs.Int64("log-lines", 2000, "lines kept from the end of each pod log and each unit's journal")
|
||||
since := fs.Duration("since", 48*time.Hour, "how far back each unit's journal is read")
|
||||
serverLogs := fs.Bool("server-logs", false, "also collect the game servers' own logs, which carry player names, IP addresses and chat")
|
||||
unitDir := fs.String("systemd-dir", systemdUnitDir, "where the installer's systemd units are")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis support-bundle: run as root (sudo felis support-bundle): it reads root-only logs and state")
|
||||
return 1
|
||||
}
|
||||
b := hostSupportBundle(*unitDir, *logLines, *since, *serverLogs)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
path, err := b.write(ctx, *outDir)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis support-bundle: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
size := int64(0)
|
||||
if st, err := os.Stat(path); err == nil {
|
||||
size = st.Size()
|
||||
}
|
||||
fmt.Fprintf(stdout, "wrote %s (%s, mode 0600)\n", path, humanSize(size))
|
||||
fmt.Fprintln(stdout, "MANIFEST.txt inside says what it holds and what was taken out. Read it through before you send it anywhere:")
|
||||
fmt.Fprintln(stdout, "redaction finds this host's known secrets and the common ways a secret is logged, and a secret logged another way stays in.")
|
||||
if !*serverLogs {
|
||||
fmt.Fprintln(stdout, "Game server logs are left out; -server-logs adds them (player names, IP addresses, chat).")
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func humanSize(n int64) string {
|
||||
switch {
|
||||
case n >= 1<<20:
|
||||
return fmt.Sprintf("%.1f MiB", float64(n)/(1<<20))
|
||||
case n >= 1<<10:
|
||||
return fmt.Sprintf("%.1f KiB", float64(n)/(1<<10))
|
||||
}
|
||||
return fmt.Sprintf("%d B", n)
|
||||
}
|
||||
|
||||
// shortDuration is d as Duration.String writes it, less the zero minutes and
|
||||
// seconds after whole hours or minutes: 48h, and 90m as 1h30m.
|
||||
func shortDuration(d time.Duration) string {
|
||||
s := d.String()
|
||||
if strings.HasSuffix(s, "m0s") {
|
||||
s = strings.TrimSuffix(s, "0s")
|
||||
}
|
||||
if strings.HasSuffix(s, "h0m") {
|
||||
s = strings.TrimSuffix(s, "0m")
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// supportBundle is one collection and what it reads the host through.
|
||||
type supportBundle struct {
|
||||
host string
|
||||
now time.Time
|
||||
unitDir string
|
||||
// w is how felis-watchdog.service runs the watchdog, wErr why it could
|
||||
// not be read (w then holds the defaults).
|
||||
w watchdogFlags
|
||||
wErr error
|
||||
cfg *config.Config // nil when cfgErr
|
||||
cfgErr error
|
||||
cl client.Client // nil when clErr
|
||||
clErr error
|
||||
logs func(ctx context.Context, ns, pod, container string, previous bool) ([]byte, error)
|
||||
run func(ctx context.Context, name string, args ...string) ([]byte, error)
|
||||
backups func(ctx context.Context) (map[string]time.Time, error)
|
||||
doctor func(ctx context.Context, out io.Writer)
|
||||
secrets secretSources
|
||||
logLines int64
|
||||
since time.Duration
|
||||
serverLogs bool
|
||||
// Host files read whole, and the directory whose listing is kept.
|
||||
meminfo, osRelease, procVersion, stateDir string
|
||||
}
|
||||
|
||||
func hostSupportBundle(unitDir string, logLines int64, since time.Duration, serverLogs bool) *supportBundle {
|
||||
host, _ := os.Hostname()
|
||||
b := &supportBundle{
|
||||
host: host, now: time.Now(), unitDir: unitDir, run: hostCommand, secrets: hostSecretSources,
|
||||
logLines: logLines, since: since, serverLogs: serverLogs,
|
||||
meminfo: "/proc/meminfo", osRelease: "/etc/os-release", procVersion: "/proc/version", stateDir: "/etc/felis",
|
||||
}
|
||||
var found bool
|
||||
b.w, found, b.wErr = watchdogUnitFlags(filepath.Join(unitDir, "felis-watchdog.service"))
|
||||
if b.wErr == nil && !found {
|
||||
b.wErr = fmt.Errorf("%s is not installed; read the watchdog's defaults", filepath.Join(unitDir, "felis-watchdog.service"))
|
||||
}
|
||||
if b.cfg, b.cfgErr = config.Load(b.w.cfgPath); b.cfgErr == nil {
|
||||
cfg := b.cfg
|
||||
b.backups = func(ctx context.Context) (map[string]time.Time, error) {
|
||||
return newestWorldBackups(ctx, cfg.Database.URL)
|
||||
}
|
||||
} else {
|
||||
b.cfg = nil
|
||||
}
|
||||
if b.cl, b.clErr = buildSystemServerClient(); b.clErr != nil {
|
||||
b.cl = nil
|
||||
} else if rc, err := hostRESTConfig(); err != nil {
|
||||
b.clErr = err
|
||||
b.cl = nil
|
||||
} else if cs, err := kubernetes.NewForConfig(rc); err != nil {
|
||||
b.clErr = err
|
||||
b.cl = nil
|
||||
} else {
|
||||
limit := int64(8 << 20)
|
||||
b.logs = func(ctx context.Context, ns, pod, container string, previous bool) ([]byte, error) {
|
||||
return cs.CoreV1().Pods(ns).GetLogs(pod, &corev1.PodLogOptions{
|
||||
Container: container, Previous: previous, Timestamps: true, TailLines: &logLines, LimitBytes: &limit,
|
||||
}).DoRaw(ctx)
|
||||
}
|
||||
}
|
||||
b.doctor = func(ctx context.Context, out io.Writer) {
|
||||
runDoctor(ctx, doctorEnv{unitDir: unitDir, run: hostCommand, now: b.now, host: host}, out)
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// bundleWriter streams scrubbed files into the archive and keeps what went
|
||||
// wrong while collecting, for MANIFEST.txt.
|
||||
type bundleWriter struct {
|
||||
tw *tar.Writer
|
||||
prefix string
|
||||
now time.Time
|
||||
scrub *scrubber
|
||||
errs []string
|
||||
}
|
||||
|
||||
// logText adds a log, or a report that quotes errors: known secrets and
|
||||
// anything logged as one come out.
|
||||
func (bw *bundleWriter) logText(name string, data []byte) error {
|
||||
return bw.add(name, bw.scrub.text(data))
|
||||
}
|
||||
|
||||
// plain adds a file whose secrets were already taken out by structure (the
|
||||
// cluster's objects, the configuration summary) or that names none (the
|
||||
// release, disk use, addresses): known secret values still come out, and
|
||||
// nothing else is rewritten.
|
||||
func (bw *bundleWriter) plain(name string, data []byte) error {
|
||||
return bw.add(name, bw.scrub.values(data))
|
||||
}
|
||||
|
||||
func (bw *bundleWriter) add(name string, data []byte) error {
|
||||
hdr := &tar.Header{Name: path.Join(bw.prefix, name), Mode: 0o600, Size: int64(len(data)), ModTime: bw.now, Typeflag: tar.TypeReg}
|
||||
if err := bw.tw.WriteHeader(hdr); err != nil {
|
||||
return err
|
||||
}
|
||||
_, err := bw.tw.Write(data)
|
||||
return err
|
||||
}
|
||||
|
||||
func (bw *bundleWriter) failed(what string, err error) {
|
||||
bw.errs = append(bw.errs, string(bw.scrub.text([]byte(fmt.Sprintf("%s: %v", what, err)))))
|
||||
}
|
||||
|
||||
// write collects the bundle into dir and returns its path.
|
||||
func (b *supportBundle) write(ctx context.Context, dir string) (string, error) {
|
||||
if _, err := os.Stat(dir); errors.Is(err, fs.ErrNotExist) {
|
||||
if err := os.MkdirAll(dir, 0o700); err != nil {
|
||||
return "", err
|
||||
}
|
||||
}
|
||||
stamp := b.now.UTC().Format("20060102T150405Z")
|
||||
base := fmt.Sprintf("felis-support-%s-%s", safeName(b.host), stamp)
|
||||
final := filepath.Join(dir, base+".tar.gz")
|
||||
f, err := os.CreateTemp(dir, "."+base+".*.partial") // mode 0600
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
keep := false
|
||||
defer func() {
|
||||
if !keep {
|
||||
f.Close()
|
||||
os.Remove(f.Name())
|
||||
}
|
||||
}()
|
||||
gz := gzip.NewWriter(f)
|
||||
bw := &bundleWriter{tw: tar.NewWriter(gz), prefix: base, now: b.now, scrub: b.scrubber()}
|
||||
if err := b.collect(ctx, bw); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := bw.tw.Close(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := gz.Close(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := f.Sync(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := os.Rename(f.Name(), final); err != nil {
|
||||
return "", err
|
||||
}
|
||||
keep = true
|
||||
return final, nil
|
||||
}
|
||||
|
||||
// safeName keeps a host name usable in a file name.
|
||||
func safeName(s string) string {
|
||||
s = strings.Map(func(r rune) rune {
|
||||
if r == '-' || r == '.' || r == '_' || (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z') || (r >= '0' && r <= '9') {
|
||||
return r
|
||||
}
|
||||
return '_'
|
||||
}, s)
|
||||
if s == "" {
|
||||
return "host"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func (b *supportBundle) collect(ctx context.Context, bw *bundleWriter) error {
|
||||
var buf bytes.Buffer
|
||||
cmdVersion(nil, &buf, io.Discard)
|
||||
for _, p := range []string{b.osRelease, b.procVersion} {
|
||||
if raw, err := os.ReadFile(p); err == nil {
|
||||
fmt.Fprintf(&buf, "\n# %s\n%s", p, raw)
|
||||
}
|
||||
}
|
||||
if err := bw.plain("version.txt", buf.Bytes()); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
buf.Reset()
|
||||
switch {
|
||||
case b.cfgErr != nil:
|
||||
fmt.Fprintf(&buf, "felis status: the configuration did not load: %v\n", b.cfgErr)
|
||||
default:
|
||||
printStatus(ctx, statusEnv{
|
||||
cfg: b.cfg, w: b.w, cl: b.cl, clErr: b.clErr, backups: b.backups, run: b.run,
|
||||
unitDir: b.unitDir, meminfo: b.meminfo, host: b.host, now: b.now,
|
||||
}, &buf)
|
||||
}
|
||||
if err := bw.logText("status.txt", buf.Bytes()); err != nil {
|
||||
return err
|
||||
}
|
||||
buf.Reset()
|
||||
b.doctor(ctx, &buf)
|
||||
if err := bw.logText("doctor.txt", buf.Bytes()); err != nil {
|
||||
return err
|
||||
}
|
||||
if b.cfg != nil {
|
||||
if err := bw.plain("config.txt", configSummary(b.cfg, b.w.cfgPath)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := b.collectHost(ctx, bw); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := b.collectJournal(ctx, bw); err != nil {
|
||||
return err
|
||||
}
|
||||
if b.cl == nil {
|
||||
bw.failed("cluster", b.clErr)
|
||||
} else if err := b.collectCluster(ctx, bw); err != nil {
|
||||
return err
|
||||
}
|
||||
return bw.add("MANIFEST.txt", b.manifest(bw))
|
||||
}
|
||||
|
||||
func (b *supportBundle) collectHost(ctx context.Context, bw *bundleWriter) error {
|
||||
commands := []struct {
|
||||
name string
|
||||
argv []string
|
||||
}{
|
||||
{"host/systemd-units.txt", []string{"systemctl", "list-units", "--all", "--no-pager", "--plain", "felis-*", "k3s.service"}},
|
||||
{"host/systemd-timers.txt", []string{"systemctl", "list-timers", "--all", "--no-pager", "felis-*"}},
|
||||
{"host/df.txt", []string{"df", "-h"}},
|
||||
{"host/addresses.txt", []string{"ip", "-brief", "address"}},
|
||||
}
|
||||
for _, c := range commands {
|
||||
out, err := b.run(ctx, c.argv[0], c.argv[1:]...)
|
||||
if err != nil && len(out) == 0 {
|
||||
bw.failed(strings.Join(c.argv, " "), err)
|
||||
continue
|
||||
}
|
||||
if err := bw.plain(c.name, out); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if raw, err := os.ReadFile(b.meminfo); err == nil {
|
||||
if err := bw.plain("host/meminfo.txt", raw); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
listing, err := dirListing(b.stateDir)
|
||||
if err != nil {
|
||||
bw.failed("list "+b.stateDir, err)
|
||||
return nil
|
||||
}
|
||||
return bw.plain("host/etc-felis.txt", listing)
|
||||
}
|
||||
|
||||
// dirListing names every file under dir with its mode, size and time, and
|
||||
// holds nothing of what is in them.
|
||||
func dirListing(dir string) ([]byte, error) {
|
||||
var buf bytes.Buffer
|
||||
tw := tabwriter.NewWriter(&buf, 0, 0, 2, ' ', 0)
|
||||
fmt.Fprintf(tw, "# %s: names, modes, sizes and times only; no contents\n", dir)
|
||||
err := filepath.WalkDir(dir, func(p string, d fs.DirEntry, err error) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
info, err := d.Info()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Fprintf(tw, "%s\t%d\t%s\t%s\n", info.Mode(), info.Size(), info.ModTime().UTC().Format(time.RFC3339), p)
|
||||
return nil
|
||||
})
|
||||
tw.Flush()
|
||||
return buf.Bytes(), err
|
||||
}
|
||||
|
||||
func (b *supportBundle) collectJournal(ctx context.Context, bw *bundleWriter) error {
|
||||
units, _ := filepath.Glob(filepath.Join(b.unitDir, "felis-*.service"))
|
||||
if _, err := os.Stat(filepath.Join(b.unitDir, "k3s.service")); err == nil {
|
||||
units = append(units, filepath.Join(b.unitDir, "k3s.service"))
|
||||
}
|
||||
since := "@" + strconv.FormatInt(b.now.Add(-b.since).Unix(), 10)
|
||||
for _, u := range units {
|
||||
unit := filepath.Base(u)
|
||||
out, err := b.run(ctx, "journalctl", "-u", unit, "--since", since, "-n", strconv.FormatInt(b.logLines, 10), "--no-pager", "-o", "short-iso")
|
||||
if err != nil && len(out) == 0 {
|
||||
bw.failed("journalctl -u "+unit, err)
|
||||
continue
|
||||
}
|
||||
if err := bw.logText("journal/"+strings.TrimSuffix(unit, ".service")+".log", out); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// bundleNamespaces are the namespaces whose objects and logs the bundle
|
||||
// collects: the control plane, the builds and the game servers.
|
||||
func (b *supportBundle) bundleNamespaces() (control, build, minecraft string) {
|
||||
control, build, minecraft = b.w.controlNS, platform.DefaultBuildNamespace, platform.DefaultMinecraftNamespace
|
||||
if b.cfg != nil {
|
||||
if b.cfg.Registry.BuildNamespace != "" {
|
||||
build = b.cfg.Registry.BuildNamespace
|
||||
}
|
||||
if b.cfg.K8s.Namespace != "" {
|
||||
minecraft = b.cfg.K8s.Namespace
|
||||
}
|
||||
}
|
||||
return control, build, minecraft
|
||||
}
|
||||
|
||||
func (b *supportBundle) collectCluster(ctx context.Context, bw *bundleWriter) error {
|
||||
control, build, minecraft := b.bundleNamespaces()
|
||||
dump := func(name string, list client.ObjectList, opts ...client.ListOption) error {
|
||||
if err := b.cl.List(ctx, list, opts...); err != nil {
|
||||
bw.failed("list "+name, err)
|
||||
return nil
|
||||
}
|
||||
redactList(list)
|
||||
out, err := yaml.Marshal(list)
|
||||
if err != nil {
|
||||
bw.failed("encode "+name, err)
|
||||
return nil
|
||||
}
|
||||
return bw.plain(name, out)
|
||||
}
|
||||
if err := dump("cluster/nodes.yaml", &corev1.NodeList{}); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := dump("cluster/persistentvolumes.yaml", &corev1.PersistentVolumeList{}); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := dump("cluster/minecraftservers.yaml", &v1alpha1.MinecraftServerList{}, client.InNamespace(minecraft)); err != nil {
|
||||
return err
|
||||
}
|
||||
for _, ns := range []string{control, build, minecraft} {
|
||||
for _, k := range []struct {
|
||||
name string
|
||||
list client.ObjectList
|
||||
}{
|
||||
{"pods", &corev1.PodList{}},
|
||||
{"deployments", &appsv1.DeploymentList{}},
|
||||
{"statefulsets", &appsv1.StatefulSetList{}},
|
||||
{"jobs", &batchv1.JobList{}},
|
||||
{"cronjobs", &batchv1.CronJobList{}},
|
||||
{"services", &corev1.ServiceList{}},
|
||||
{"persistentvolumeclaims", &corev1.PersistentVolumeClaimList{}},
|
||||
{"networkpolicies", &networkingv1.NetworkPolicyList{}},
|
||||
{"events", &corev1.EventList{}},
|
||||
} {
|
||||
if err := dump("cluster/"+ns+"/"+k.name+".yaml", k.list, client.InNamespace(ns)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var all corev1.PodList
|
||||
if err := b.cl.List(ctx, &all); err != nil {
|
||||
bw.failed("list every pod", err)
|
||||
} else if err := bw.plain("cluster/pods-all-namespaces.txt", podTable(all.Items, b.now)); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
for _, ns := range []string{control, build, minecraft} {
|
||||
var pods corev1.PodList
|
||||
if err := b.cl.List(ctx, &pods, client.InNamespace(ns)); err != nil {
|
||||
continue // already recorded by the dump above
|
||||
}
|
||||
for _, p := range pods.Items {
|
||||
if err := b.collectPodLogs(ctx, bw, p, ns == minecraft && !b.serverLogs); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// collectPodLogs keeps the tail of each container's log, and of its previous
|
||||
// run when it restarted. initOnly keeps only the init containers: a game
|
||||
// server's own log carries player names, addresses and chat.
|
||||
func (b *supportBundle) collectPodLogs(ctx context.Context, bw *bundleWriter, p corev1.Pod, initOnly bool) error {
|
||||
type ctr struct {
|
||||
name string
|
||||
restarts int32
|
||||
}
|
||||
var ctrs []ctr
|
||||
restarts := map[string]int32{}
|
||||
for _, cs := range append(append([]corev1.ContainerStatus(nil), p.Status.InitContainerStatuses...), p.Status.ContainerStatuses...) {
|
||||
restarts[cs.Name] = cs.RestartCount
|
||||
}
|
||||
for _, c := range p.Spec.InitContainers {
|
||||
ctrs = append(ctrs, ctr{c.Name, restarts[c.Name]})
|
||||
}
|
||||
if !initOnly {
|
||||
for _, c := range p.Spec.Containers {
|
||||
ctrs = append(ctrs, ctr{c.Name, restarts[c.Name]})
|
||||
}
|
||||
}
|
||||
for _, c := range ctrs {
|
||||
for _, previous := range []bool{false, true} {
|
||||
if previous && c.restarts == 0 {
|
||||
continue
|
||||
}
|
||||
name := fmt.Sprintf("logs/%s/%s/%s.log", p.Namespace, p.Name, c.name)
|
||||
if previous {
|
||||
name = fmt.Sprintf("logs/%s/%s/%s.previous.log", p.Namespace, p.Name, c.name)
|
||||
}
|
||||
out, err := b.logs(ctx, p.Namespace, p.Name, c.name, previous)
|
||||
if err != nil {
|
||||
bw.failed(name, err)
|
||||
continue
|
||||
}
|
||||
if err := bw.logText(name, out); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// podTable is every pod on the node, one line each.
|
||||
func podTable(pods []corev1.Pod, now time.Time) []byte {
|
||||
sort.Slice(pods, func(i, j int) bool {
|
||||
if pods[i].Namespace != pods[j].Namespace {
|
||||
return pods[i].Namespace < pods[j].Namespace
|
||||
}
|
||||
return pods[i].Name < pods[j].Name
|
||||
})
|
||||
var buf bytes.Buffer
|
||||
tw := tabwriter.NewWriter(&buf, 0, 0, 2, ' ', 0)
|
||||
fmt.Fprintln(tw, "NAMESPACE\tNAME\tPHASE\tREADY\tRESTARTS\tAGE")
|
||||
for _, p := range pods {
|
||||
var ready, restarts int32
|
||||
for _, cs := range p.Status.ContainerStatuses {
|
||||
if cs.Ready {
|
||||
ready++
|
||||
}
|
||||
restarts += cs.RestartCount
|
||||
}
|
||||
age := "-"
|
||||
if !p.CreationTimestamp.IsZero() {
|
||||
age = now.Sub(p.CreationTimestamp.Time).Round(time.Minute).String()
|
||||
}
|
||||
fmt.Fprintf(tw, "%s\t%s\t%s\t%d/%d\t%d\t%s\n", p.Namespace, p.Name, p.Status.Phase, ready, len(p.Spec.Containers), restarts, age)
|
||||
}
|
||||
tw.Flush()
|
||||
return buf.Bytes()
|
||||
}
|
||||
|
||||
// redactList takes out of every object what the bundle must not carry: the
|
||||
// literal env values of pod specs and of MinecraftServers (valueFrom
|
||||
// references stay, naming the Secret without its contents), the
|
||||
// last-applied-configuration annotation that repeats them, and managedFields.
|
||||
func redactList(list client.ObjectList) {
|
||||
items, err := meta.ExtractList(list)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
for _, it := range items {
|
||||
if acc, err := meta.Accessor(it); err == nil {
|
||||
acc.SetManagedFields(nil)
|
||||
if ann := acc.GetAnnotations(); ann != nil {
|
||||
delete(ann, corev1.LastAppliedConfigAnnotation)
|
||||
acc.SetAnnotations(ann)
|
||||
}
|
||||
}
|
||||
switch o := it.(type) {
|
||||
case *corev1.Pod:
|
||||
redactPodSpec(&o.Spec)
|
||||
case *appsv1.Deployment:
|
||||
redactPodSpec(&o.Spec.Template.Spec)
|
||||
case *appsv1.StatefulSet:
|
||||
redactPodSpec(&o.Spec.Template.Spec)
|
||||
case *batchv1.Job:
|
||||
redactPodSpec(&o.Spec.Template.Spec)
|
||||
case *batchv1.CronJob:
|
||||
redactPodSpec(&o.Spec.JobTemplate.Spec.Template.Spec)
|
||||
case *v1alpha1.MinecraftServer:
|
||||
for i := range o.Spec.Env {
|
||||
if o.Spec.Env[i].Value != "" {
|
||||
o.Spec.Env[i].Value = redacted
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func redactPodSpec(s *corev1.PodSpec) {
|
||||
blank := func(cs []corev1.Container) {
|
||||
for i := range cs {
|
||||
for j := range cs[i].Env {
|
||||
if cs[i].Env[j].Value != "" {
|
||||
cs[i].Env[j].Value = redacted
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
blank(s.InitContainers)
|
||||
blank(s.Containers)
|
||||
for i := range s.EphemeralContainers {
|
||||
for j := range s.EphemeralContainers[i].Env {
|
||||
if s.EphemeralContainers[i].Env[j].Value != "" {
|
||||
s.EphemeralContainers[i].Env[j].Value = redacted
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// configSummary is felis.toml without a single credential: the hostnames,
|
||||
// namespaces and which features are on. Fields are picked one by one, so a
|
||||
// field added later stays out until someone decides it is safe.
|
||||
func configSummary(c *config.Config, path string) []byte {
|
||||
var buf bytes.Buffer
|
||||
line := func(k string, v any) { fmt.Fprintf(&buf, "%-34s %v\n", k, v) }
|
||||
fmt.Fprintf(&buf, "# a summary of %s; no password, key or token is in it\n", path)
|
||||
line("server.root_domain", c.Server.RootDomain)
|
||||
line("server.listen", c.Server.Listen)
|
||||
db := "(unparsable)"
|
||||
if u, err := url.Parse(c.Database.URL); err == nil {
|
||||
db = u.Scheme + "://" + u.User.Username() + "@" + u.Host + u.Path
|
||||
}
|
||||
line("database.url (no password)", db)
|
||||
line("database.deployment", c.Database.Deployment)
|
||||
line("velocity.public_ip", c.Velocity.PublicIP)
|
||||
line("velocity.game_port", c.Velocity.GamePort)
|
||||
line("velocity.login_image", c.Velocity.LoginImage)
|
||||
line("velocity.lobby_image", c.Velocity.LobbyImage)
|
||||
line("auth.panel_hostname", c.Auth.PanelHostname)
|
||||
line("auth.admin_hostname", c.Auth.AdminHostname)
|
||||
line("auth.client_ip_header", c.Auth.ClientIPHeader)
|
||||
line("k8s.namespace", c.K8s.Namespace)
|
||||
line("k8s.egress_mode", c.K8s.EgressMode)
|
||||
line("k8s.metallb_pool", c.K8s.MetalLBPool)
|
||||
line("registry.url", c.Registry.URL)
|
||||
line("registry.build_namespace", c.Registry.BuildNamespace)
|
||||
line("registry.trivy_db_repository", c.Registry.TrivyDBRepository)
|
||||
line("registry.build_user_namespaces", c.Registry.BuildUserNamespaces)
|
||||
line("registry.build_runtime_class", c.Registry.BuildRuntimeClass)
|
||||
line("registry.max_concurrent_builds", c.Registry.MaxConcurrentBuilds)
|
||||
line("registry.user_uploads_context", c.Registry.UserUploadsContext)
|
||||
line("archive.store", c.Archive.Store)
|
||||
line("archive.local_path", c.Archive.LocalPath)
|
||||
line("archive.retention", c.Archive.Retention)
|
||||
line("archive.scheduled_every", c.Archive.ScheduledEvery)
|
||||
line("archive.scheduled_keep", c.Archive.ScheduledKeep)
|
||||
line("offsite (configured)", c.Offsite.Enabled())
|
||||
if c.Offsite.Enabled() {
|
||||
line("offsite.endpoint", c.Offsite.Endpoint)
|
||||
line("offsite.bucket", c.Offsite.Bucket)
|
||||
line("offsite.prefix", c.Offsite.Prefix)
|
||||
}
|
||||
line("smtp.host", c.SMTP.Host)
|
||||
if c.SMTP.Host != "" {
|
||||
line("smtp.port", c.SMTP.Port)
|
||||
line("smtp.require_tls", c.SMTP.TLSRequired())
|
||||
line("smtp.max_per_hour", c.SMTP.MaxPerHour)
|
||||
}
|
||||
for i, s := range c.AuthSources {
|
||||
line(fmt.Sprintf("auth_source[%d]", i), s.Tag+" "+s.Prefix+" "+s.URL)
|
||||
}
|
||||
return buf.Bytes()
|
||||
}
|
||||
|
||||
func (b *supportBundle) manifest(bw *bundleWriter) []byte {
|
||||
control, build, minecraft := b.bundleNamespaces()
|
||||
var buf bytes.Buffer
|
||||
fmt.Fprintf(&buf, "Felis support bundle\nhost %s, collected %s, felis %s\n\n", b.host, b.now.UTC().Format(time.RFC3339), resolvedVersion())
|
||||
fmt.Fprintf(&buf, `What it holds:
|
||||
status.txt, doctor.txt felis status and felis doctor at collection time
|
||||
version.txt the release, the OS and the kernel
|
||||
config.txt a summary of felis.toml: hostnames, namespaces, which features are on
|
||||
host/ systemd units and timers, disk use, memory, addresses, and the names,
|
||||
modes, sizes and times of the files under %s (not what is in them)
|
||||
journal/ up to %d lines per Felis unit and k3s, from the last %s
|
||||
logs/ up to %d lines of each container of the pods in %s and %s, and of the
|
||||
init containers of the game server pods in %s; the previous run too
|
||||
where a container restarted
|
||||
cluster/ nodes, volumes, the MinecraftServers, and the pods, workloads, services,
|
||||
volume claims, network policies and events of %s, %s and %s
|
||||
`, b.stateDir, b.logLines, shortDuration(b.since), b.logLines, control, build, minecraft, control, build, minecraft)
|
||||
if b.serverLogs {
|
||||
fmt.Fprintln(&buf, "\nThe game servers' own logs are in logs/ (-server-logs): they carry player names, IP addresses and chat.")
|
||||
} else {
|
||||
fmt.Fprintln(&buf, "\nThe game servers' own logs are left out (they carry player names, IP addresses and chat; -server-logs adds them).")
|
||||
}
|
||||
fmt.Fprint(&buf, `
|
||||
Never collected: Kubernetes Secrets and ConfigMaps, what is in /etc/felis or any configuration
|
||||
file, the database, worlds, uploads.
|
||||
|
||||
Taken out:
|
||||
`)
|
||||
if len(bw.scrub.sources) > 0 {
|
||||
fmt.Fprintf(&buf, " - %d secret values, wherever they appear, read from:\n", len(bw.scrub.vals))
|
||||
for _, s := range bw.scrub.sources {
|
||||
fmt.Fprintf(&buf, " %s\n", s)
|
||||
}
|
||||
} else {
|
||||
fmt.Fprintln(&buf, " - no secret file was found on this host to take values from")
|
||||
}
|
||||
fmt.Fprint(&buf, ` - passwords in URLs, private keys, Bearer and Basic credentials
|
||||
- in logs and command output, whatever follows password=, secret=, token=, api_key=,
|
||||
access_key=, private_key= or credentials= (and the same with a colon)
|
||||
- every literal env value in pod specs and MinecraftServers (valueFrom references stay)
|
||||
|
||||
Read it through before you send it anywhere: redaction finds this host's known secrets and the
|
||||
common ways a secret is logged, and a secret logged another way stays in.
|
||||
`)
|
||||
if b.wErr != nil {
|
||||
fmt.Fprintf(&buf, "\nThe watchdog's settings: %v\n", b.wErr)
|
||||
}
|
||||
if len(bw.errs) > 0 {
|
||||
fmt.Fprintln(&buf, "\nNot collected:")
|
||||
for _, e := range bw.errs {
|
||||
fmt.Fprintf(&buf, " - %s\n", e)
|
||||
}
|
||||
}
|
||||
// Its own words name what is redacted, and would be redacted themselves;
|
||||
// what it quotes was scrubbed as it was recorded.
|
||||
return buf.Bytes()
|
||||
}
|
||||
|
||||
// secretSources are files that hold secrets, by how each is laid out.
|
||||
type secretSources struct {
|
||||
envFiles []string // KEY=VALUE lines, every value a secret (a commented-out one too)
|
||||
valueFiles []string // one secret, the whole file
|
||||
propsFiles []string // key=value lines; the keys naming a token, secret, password or key hold one
|
||||
tokenFiles []string // k3s join tokens: the whole token and its secret part after the last ':'
|
||||
}
|
||||
|
||||
// scrubber takes secrets out of what the bundle collects.
|
||||
type scrubber struct {
|
||||
vals []string // longest first, so a secret that contains another goes whole
|
||||
sources []string // the files vals came from
|
||||
}
|
||||
|
||||
var (
|
||||
pemPrivateKey = regexp.MustCompile(`(?s)-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----.*?-----END [A-Z0-9 ]*PRIVATE KEY-----`)
|
||||
urlUserinfo = regexp.MustCompile(`([A-Za-z][A-Za-z0-9+.-]*://[^/\s:@]*:)[^/\s@]+@`)
|
||||
authScheme = regexp.MustCompile(`(?i)\b(bearer|basic)\s+[A-Za-z0-9._~+/=-]{8,}`)
|
||||
secretAssign = regexp.MustCompile(`(?i)((?:password|passwd|secret|token|api[_-]?key|access[_-]?key|private[_-]?key|credentials?)[A-Za-z0-9_.-]*"?[ \t]*[=:][ \t]*"?)([^\s"',;&]{4,})`)
|
||||
secretPropKey = regexp.MustCompile(`(?i)token|secret|password|key`)
|
||||
)
|
||||
|
||||
func (b *supportBundle) scrubber() *scrubber {
|
||||
s := &scrubber{}
|
||||
s.readSources(b.secrets)
|
||||
if b.cfg != nil {
|
||||
if u, err := url.Parse(b.cfg.Database.URL); err == nil {
|
||||
if pw, ok := u.User.Password(); ok {
|
||||
s.add(pw)
|
||||
}
|
||||
}
|
||||
for _, ref := range []string{
|
||||
b.cfg.Velocity.ServiceTokenRef, b.cfg.SMTP.PasswordRef,
|
||||
b.cfg.Offsite.AccessKeyRef, b.cfg.Offsite.SecretKeyRef, b.cfg.Offsite.KeyRef,
|
||||
b.cfg.Registry.S3.AccessKeyRef, b.cfg.Registry.S3.SecretKeyRef,
|
||||
b.cfg.Archive.S3.AccessKeyRef, b.cfg.Archive.S3.SecretKeyRef,
|
||||
} {
|
||||
if ref != "" {
|
||||
s.add(os.Getenv(ref))
|
||||
}
|
||||
}
|
||||
}
|
||||
if st, err := watchdog.LoadState(watchdog.NewestState(b.w.statePath, b.w.fallbackState)); err == nil {
|
||||
s.add(st.SMTPPassword)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func (s *scrubber) add(v string) bool {
|
||||
v = strings.TrimSpace(v)
|
||||
if len(v) < minScrubLen {
|
||||
return false
|
||||
}
|
||||
for _, have := range s.vals {
|
||||
if have == v {
|
||||
return true
|
||||
}
|
||||
}
|
||||
s.vals = append(s.vals, v)
|
||||
sort.SliceStable(s.vals, func(i, j int) bool { return len(s.vals[i]) > len(s.vals[j]) })
|
||||
return true
|
||||
}
|
||||
|
||||
func (s *scrubber) readSources(src secretSources) {
|
||||
read := func(p string, take func(content string) bool) {
|
||||
raw, err := os.ReadFile(p)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
if take(string(raw)) {
|
||||
s.sources = append(s.sources, p)
|
||||
}
|
||||
}
|
||||
keyValues := func(content string, keep func(key string) bool) bool {
|
||||
found := false
|
||||
for _, line := range strings.Split(content, "\n") {
|
||||
k, v, ok := strings.Cut(line, "=")
|
||||
if !ok || !keep(strings.TrimSpace(k)) {
|
||||
continue
|
||||
}
|
||||
v = strings.TrimSpace(v)
|
||||
if len(v) >= 2 && (v[0] == '\'' || v[0] == '"') && v[len(v)-1] == v[0] {
|
||||
v = v[1 : len(v)-1]
|
||||
}
|
||||
found = s.add(v) || found
|
||||
}
|
||||
return found
|
||||
}
|
||||
for _, p := range src.envFiles {
|
||||
read(p, func(c string) bool { return keyValues(c, func(string) bool { return true }) })
|
||||
}
|
||||
for _, p := range src.propsFiles {
|
||||
read(p, func(c string) bool { return keyValues(c, secretPropKey.MatchString) })
|
||||
}
|
||||
for _, p := range src.valueFiles {
|
||||
read(p, func(c string) bool { return s.add(c) })
|
||||
}
|
||||
for _, p := range src.tokenFiles {
|
||||
read(p, func(c string) bool {
|
||||
c = strings.TrimSpace(c)
|
||||
whole := s.add(c)
|
||||
part := false
|
||||
if i := strings.LastIndex(c, ":"); i >= 0 {
|
||||
part = s.add(c[i+1:])
|
||||
}
|
||||
return whole || part
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// values takes every known secret value out of data.
|
||||
func (s *scrubber) values(data []byte) []byte {
|
||||
t := pemPrivateKey.ReplaceAllString(string(data), "<redacted private key>")
|
||||
for _, v := range s.vals {
|
||||
t = strings.ReplaceAll(t, v, redacted)
|
||||
}
|
||||
t = urlUserinfo.ReplaceAllString(t, "${1}"+redacted+"@")
|
||||
t = authScheme.ReplaceAllString(t, "${1} "+redacted)
|
||||
return []byte(t)
|
||||
}
|
||||
|
||||
// text is values, and whatever a log names as a secret as well.
|
||||
func (s *scrubber) text(data []byte) []byte {
|
||||
return secretAssign.ReplaceAll(s.values(data), []byte("${1}"+redacted))
|
||||
}
|
||||
@@ -0,0 +1,433 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"compress/gzip"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
// The secrets a bundle host holds, each of which must be nowhere in the bundle.
|
||||
var plantedSecrets = map[string]string{
|
||||
"secrets.env SERVICE_TOKEN": "svc-token-planted-0a1b2c3d",
|
||||
"secrets.env DB_PASSWORD": "db-pass-planted-4e5f6a7b",
|
||||
"offsite.env FELIS_OFFSITE_KEY": "offsite-key-planted+8c9d/0e1f=",
|
||||
"smtp-password": "smtp-pass-planted-2a3b",
|
||||
"felis-link.properties token": "props-token-planted-4c5d",
|
||||
"k3s token secret part": "k3s-secret-part-planted-6e7f",
|
||||
"watchdog state relay password": "cached-relay-pw-planted-8a9b",
|
||||
"pod env literal": "env-literal-planted-0c1d",
|
||||
"deployment env literal": "deploy-env-planted-2e3f",
|
||||
"MinecraftServer env literal": "cr-env-planted-4a5b",
|
||||
"last-applied annotation": "annotation-planted-6c7d",
|
||||
"logged password": "hunter2-planted-8e9f",
|
||||
"logged bearer": "bearerplanted0a1b2c3d",
|
||||
"URL password": "urlpass-planted-4e5f",
|
||||
"token in an error": "errtoken-planted-0f1e",
|
||||
"felis.host.toml DB password": "cfg-db-pass-planted-1c2d",
|
||||
"service token by its env ref": "env-ref-token-planted-3e4f",
|
||||
"init container env literal": "init-env-planted-5a6b",
|
||||
"ephemeral container env": "ephemeral-env-planted-7c8d",
|
||||
"statefulset env literal": "sts-env-planted-9e0f",
|
||||
"job env literal": "job-env-planted-1a2b",
|
||||
"cronjob env literal": "cronjob-env-planted-3c4d",
|
||||
"commented-out offsite key": "old-offsite-key-planted-5e6f",
|
||||
"doctor quoting a password": "doctor-pw-planted-7a8b",
|
||||
"status quoting a password": "status-pw-planted-9c0d",
|
||||
"journal quoting a password": "journal-pw-planted-1e2f",
|
||||
}
|
||||
|
||||
// bundleHost is a Felis host for felis support-bundle: its secret files, its
|
||||
// watchdog unit and state, a cluster with a control-plane pod that restarted
|
||||
// and a game server, and logs and a journal that name secrets.
|
||||
func bundleHost(t *testing.T) *supportBundle {
|
||||
t.Helper()
|
||||
p := plantedSecrets
|
||||
dir := t.TempDir()
|
||||
etc := filepath.Join(dir, "etc-felis")
|
||||
for _, d := range []string{etc, filepath.Join(dir, "systemd"), filepath.Join(dir, "k3s")} {
|
||||
if err := os.MkdirAll(d, 0o700); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
writeTestFile(t, filepath.Join(etc, "secrets.env"), "SERVICE_TOKEN="+p["secrets.env SERVICE_TOKEN"]+"\nDB_PASSWORD="+p["secrets.env DB_PASSWORD"]+"\nSHORT=abc\n", 0o600)
|
||||
writeTestFile(t, filepath.Join(etc, "offsite.env"), "# before the rotation\n# FELIS_OFFSITE_KEY="+p["commented-out offsite key"]+"\nFELIS_OFFSITE_KEY='"+p["offsite.env FELIS_OFFSITE_KEY"]+"'\n", 0o600)
|
||||
writeTestFile(t, filepath.Join(etc, "smtp-password"), p["smtp-password"]+"\n", 0o600)
|
||||
writeTestFile(t, filepath.Join(etc, "felis-link.properties"), "api-base-url=http://10.43.0.10:8081\nservice-token="+p["felis-link.properties token"]+"\nroot-domain=games.example.org\n", 0o640)
|
||||
writeTestFile(t, filepath.Join(dir, "k3s", "token"), "K10deadbeefcafe::server:"+p["k3s token secret part"]+"\n", 0o600)
|
||||
cfgPath := filepath.Join(etc, "felis.host.toml")
|
||||
writeTestFile(t, cfgPath, "[database]\nurl = \"postgres://felis:"+p["felis.host.toml DB password"]+"@127.0.0.1:1/felis?sslmode=disable&connect_timeout=1\"\n"+
|
||||
"[server]\nroot_domain = \"games.example.org\"\n[archive]\nstore = \"tarLocal\"\n[k8s]\negress_mode = \"nodeport\"\n"+
|
||||
"[velocity]\nservice_token_ref = \"FELIS_BUNDLE_TEST_SERVICE_TOKEN\"\n", 0o600)
|
||||
t.Setenv("FELIS_BUNDLE_TEST_SERVICE_TOKEN", p["service token by its env ref"])
|
||||
statePath := filepath.Join(dir, "state.json")
|
||||
if err := watchdog.SaveState(statePath, &watchdog.State{SMTPPassword: p["watchdog state relay password"]}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
unitDir := filepath.Join(dir, "systemd")
|
||||
writeTestFile(t, filepath.Join(unitDir, "felis-watchdog.service"), "[Service]\nExecStart=/usr/local/bin/felis watchdog -config "+cfgPath+" -state "+statePath+"\n", 0o644)
|
||||
writeTestFile(t, filepath.Join(unitDir, "felis-offsite.service"), "[Unit]\n", 0o644)
|
||||
writeTestFile(t, filepath.Join(unitDir, "k3s.service"), "[Unit]\n", 0o644)
|
||||
meminfo := filepath.Join(dir, "meminfo")
|
||||
writeTestFile(t, meminfo, "MemTotal: 8000000 kB\nMemAvailable: 2000000 kB\n", 0o644)
|
||||
|
||||
cfg, err := config.Load(cfgPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var w watchdogFlags
|
||||
w, _, err = watchdogUnitFlags(filepath.Join(unitDir, "felis-watchdog.service"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
w.diskPaths = "/nonexistent-felis-bundle-test"
|
||||
|
||||
secretEnv := corev1.EnvVar{Name: "FELIS_SMTP_PASSWORD", ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
|
||||
LocalObjectReference: corev1.LocalObjectReference{Name: "felis-smtp"}, Key: "password"}}}
|
||||
replicas := int32(1)
|
||||
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(
|
||||
&corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "felis-api-7d9", Namespace: "felis",
|
||||
Annotations: map[string]string{corev1.LastAppliedConfigAnnotation: `{"env":"` + p["last-applied annotation"] + `"}`, "felis.lolicon.best/kept": "yes"},
|
||||
ManagedFields: []metav1.ManagedFieldsEntry{{Manager: "kubectl-client-side-apply"}},
|
||||
},
|
||||
Spec: corev1.PodSpec{
|
||||
InitContainers: []corev1.Container{{Name: "wait-db", Image: "busybox", Env: []corev1.EnvVar{{Name: "FELIS_INIT_SETTING", Value: p["init container env literal"]}}}},
|
||||
Containers: []corev1.Container{{Name: "api", Image: "felis-api:v1", Env: []corev1.EnvVar{
|
||||
{Name: "FELIS_PLAIN_SETTING", Value: p["pod env literal"]}, secretEnv,
|
||||
}}},
|
||||
EphemeralContainers: []corev1.EphemeralContainer{{EphemeralContainerCommon: corev1.EphemeralContainerCommon{
|
||||
Name: "debug", Env: []corev1.EnvVar{{Name: "FELIS_DEBUG_SETTING", Value: p["ephemeral container env"]}}}}},
|
||||
},
|
||||
Status: corev1.PodStatus{Phase: corev1.PodRunning, ContainerStatuses: []corev1.ContainerStatus{{Name: "api", RestartCount: 1, Ready: true}}},
|
||||
},
|
||||
&appsv1.Deployment{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-api", Namespace: "felis"},
|
||||
Spec: appsv1.DeploymentSpec{Replicas: &replicas, Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app": "felis-api"}},
|
||||
Template: corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{{Name: "api", Env: []corev1.EnvVar{
|
||||
{Name: "FELIS_DEPLOY_SETTING", Value: p["deployment env literal"]},
|
||||
}}}}}},
|
||||
},
|
||||
&corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "survival-0", Namespace: "minecraft"},
|
||||
Spec: corev1.PodSpec{
|
||||
InitContainers: []corev1.Container{{Name: "prepare-data"}, {Name: "egress-gate"}},
|
||||
Containers: []corev1.Container{{Name: "server"}},
|
||||
},
|
||||
Status: corev1.PodStatus{Phase: corev1.PodRunning},
|
||||
},
|
||||
&v1alpha1.MinecraftServer{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"},
|
||||
Spec: v1alpha1.MinecraftServerSpec{Env: []v1alpha1.EnvVar{{Name: "DISCORD_WEBHOOK", Value: p["MinecraftServer env literal"]}}},
|
||||
},
|
||||
&appsv1.StatefulSet{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-postgres", Namespace: "felis"},
|
||||
Spec: appsv1.StatefulSetSpec{Template: envTemplate("FELIS_STS_SETTING", p["statefulset env literal"])},
|
||||
},
|
||||
&batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "build-1", Namespace: "felis-build"},
|
||||
Spec: batchv1.JobSpec{Template: envTemplate("FELIS_JOB_SETTING", p["job env literal"])},
|
||||
},
|
||||
&batchv1.CronJob{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-reaper", Namespace: "felis"},
|
||||
Spec: batchv1.CronJobSpec{JobTemplate: batchv1.JobTemplateSpec{Spec: batchv1.JobSpec{Template: envTemplate("FELIS_CRON_SETTING", p["cronjob env literal"])}}},
|
||||
},
|
||||
&corev1.Node{ObjectMeta: metav1.ObjectMeta{Name: "felis-1"}},
|
||||
).Build()
|
||||
|
||||
secretLog := fmt.Sprintf("started with SERVICE_TOKEN=%s\nlogin password=%s ok\nAuthorization: Bearer %s\ndial postgres://felis:%s@db:5432/felis\n",
|
||||
p["secrets.env SERVICE_TOKEN"], p["logged password"], p["logged bearer"], p["URL password"])
|
||||
return &supportBundle{
|
||||
host: "felis-test", now: time.Date(2026, 9, 27, 12, 0, 0, 0, time.UTC), unitDir: unitDir,
|
||||
w: w, cfg: cfg, cl: cl,
|
||||
logs: func(_ context.Context, ns, pod, container string, previous bool) ([]byte, error) {
|
||||
if container == "wait-db" {
|
||||
return nil, errors.New("container \"wait-db\" is waiting to start; token=" + p["token in an error"])
|
||||
}
|
||||
return []byte(fmt.Sprintf("log of %s/%s/%s previous=%v\n%s", ns, pod, container, previous, secretLog)), nil
|
||||
},
|
||||
run: func(_ context.Context, name string, args ...string) ([]byte, error) {
|
||||
switch name {
|
||||
case "journalctl":
|
||||
return []byte("journal of " + args[1] + "\nFELIS_OFFSITE_KEY=" + p["offsite.env FELIS_OFFSITE_KEY"] + "\nrelay " + p["smtp-password"] + " refused\n" +
|
||||
"relay login with the cached " + p["watchdog state relay password"] + "\ndatabase auth failed for " + p["felis.host.toml DB password"] +
|
||||
"\nservice token " + p["service token by its env ref"] + " rejected\nnode joined with " + p["k3s token secret part"] + "\n" +
|
||||
"old copies sealed with " + p["commented-out offsite key"] + "\nretrying with password=" + p["journal quoting a password"] + "\n"), nil
|
||||
case "systemctl", "df", "ip":
|
||||
return []byte(name + " output\n"), nil
|
||||
}
|
||||
return nil, errors.New("unexpected " + name)
|
||||
},
|
||||
backups: func(context.Context) (map[string]time.Time, error) {
|
||||
return nil, errors.New("connect: password=" + p["status quoting a password"])
|
||||
},
|
||||
doctor: func(_ context.Context, out io.Writer) {
|
||||
fmt.Fprintf(out, "doctor report; the k3s token K10deadbeefcafe::server:%s leaked here\nprobe said password=%s\n", p["k3s token secret part"], p["doctor quoting a password"])
|
||||
},
|
||||
secrets: secretSources{
|
||||
envFiles: []string{filepath.Join(etc, "secrets.env"), filepath.Join(etc, "offsite.env"), filepath.Join(etc, "absent.env")},
|
||||
valueFiles: []string{filepath.Join(etc, "smtp-password"), filepath.Join(etc, "uploads-s3-secret-key")},
|
||||
propsFiles: []string{filepath.Join(etc, "felis-link.properties")},
|
||||
tokenFiles: []string{filepath.Join(dir, "k3s", "token")},
|
||||
},
|
||||
logLines: 500, since: 48 * time.Hour,
|
||||
meminfo: meminfo, osRelease: filepath.Join(dir, "os-release"), procVersion: filepath.Join(dir, "version"), stateDir: etc,
|
||||
}
|
||||
}
|
||||
|
||||
// envTemplate is a pod template whose one container sets name to a literal.
|
||||
func envTemplate(name, value string) corev1.PodTemplateSpec {
|
||||
return corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{{Name: "main", Env: []corev1.EnvVar{{Name: name, Value: value}}}}}}
|
||||
}
|
||||
|
||||
// readBundle is every file in the bundle at path, by its name inside the
|
||||
// bundle's directory, and each one's mode.
|
||||
func readBundle(t *testing.T, path string) (map[string]string, map[string]int64) {
|
||||
t.Helper()
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
gz, err := gzip.NewReader(f)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
tr := tar.NewReader(gz)
|
||||
files, modes := map[string]string{}, map[string]int64{}
|
||||
prefix := strings.TrimSuffix(filepath.Base(path), ".tar.gz") + "/"
|
||||
for {
|
||||
hdr, err := tr.Next()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
name, ok := strings.CutPrefix(hdr.Name, prefix)
|
||||
if !ok {
|
||||
t.Errorf("%s is outside the bundle's directory %s", hdr.Name, prefix)
|
||||
}
|
||||
raw, err := io.ReadAll(tr)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
files[name], modes[name] = string(raw), hdr.Mode
|
||||
}
|
||||
return files, modes
|
||||
}
|
||||
|
||||
func TestSupportBundle(t *testing.T) {
|
||||
b := bundleHost(t)
|
||||
out := filepath.Join(t.TempDir(), "support")
|
||||
path, err := b.write(context.Background(), out)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if want := filepath.Join(out, "felis-support-felis-test-20260927T120000Z.tar.gz"); path != want {
|
||||
t.Errorf("path %s, want %s", path, want)
|
||||
}
|
||||
for p, want := range map[string]os.FileMode{out: 0o700 | os.ModeDir, path: 0o600} {
|
||||
if st, err := os.Stat(p); err != nil || st.Mode() != want {
|
||||
t.Errorf("%s: mode %v (err %v), want %v", p, st.Mode(), err, want)
|
||||
}
|
||||
}
|
||||
if left, _ := filepath.Glob(filepath.Join(out, ".*partial")); len(left) != 0 {
|
||||
t.Errorf("left behind %v", left)
|
||||
}
|
||||
|
||||
files, modes := readBundle(t, path)
|
||||
var names []string
|
||||
for n := range files {
|
||||
names = append(names, n)
|
||||
if modes[n] != 0o600 {
|
||||
t.Errorf("%s: mode %o in the archive, want 600", n, modes[n])
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
want := []string{
|
||||
"MANIFEST.txt",
|
||||
"cluster/felis-build/cronjobs.yaml", "cluster/felis-build/deployments.yaml", "cluster/felis-build/events.yaml", "cluster/felis-build/jobs.yaml",
|
||||
"cluster/felis-build/networkpolicies.yaml", "cluster/felis-build/persistentvolumeclaims.yaml", "cluster/felis-build/pods.yaml",
|
||||
"cluster/felis-build/services.yaml", "cluster/felis-build/statefulsets.yaml",
|
||||
"cluster/felis/cronjobs.yaml", "cluster/felis/deployments.yaml", "cluster/felis/events.yaml", "cluster/felis/jobs.yaml",
|
||||
"cluster/felis/networkpolicies.yaml", "cluster/felis/persistentvolumeclaims.yaml", "cluster/felis/pods.yaml",
|
||||
"cluster/felis/services.yaml", "cluster/felis/statefulsets.yaml",
|
||||
"cluster/minecraft/cronjobs.yaml", "cluster/minecraft/deployments.yaml", "cluster/minecraft/events.yaml", "cluster/minecraft/jobs.yaml",
|
||||
"cluster/minecraft/networkpolicies.yaml", "cluster/minecraft/persistentvolumeclaims.yaml", "cluster/minecraft/pods.yaml",
|
||||
"cluster/minecraft/services.yaml", "cluster/minecraft/statefulsets.yaml",
|
||||
"cluster/minecraftservers.yaml", "cluster/nodes.yaml", "cluster/persistentvolumes.yaml", "cluster/pods-all-namespaces.txt",
|
||||
"config.txt", "doctor.txt",
|
||||
"host/addresses.txt", "host/df.txt", "host/etc-felis.txt", "host/meminfo.txt", "host/systemd-timers.txt", "host/systemd-units.txt",
|
||||
"journal/felis-offsite.log", "journal/felis-watchdog.log", "journal/k3s.log",
|
||||
"logs/felis/felis-api-7d9/api.log", "logs/felis/felis-api-7d9/api.previous.log",
|
||||
"logs/minecraft/survival-0/egress-gate.log", "logs/minecraft/survival-0/prepare-data.log",
|
||||
"status.txt", "version.txt",
|
||||
}
|
||||
if strings.Join(names, "\n") != strings.Join(want, "\n") {
|
||||
t.Errorf("bundle holds\n %s\nwant\n %s", strings.Join(names, "\n "), strings.Join(want, "\n "))
|
||||
}
|
||||
|
||||
for what, secret := range plantedSecrets {
|
||||
for n, body := range files {
|
||||
if strings.Contains(body, secret) {
|
||||
t.Errorf("%s (%s) is in %s:\n%s", what, secret, n, body)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// What was taken out leaves what a reader needs around it.
|
||||
for name, wants := range map[string][]string{
|
||||
"logs/felis/felis-api-7d9/api.previous.log": {"log of felis/felis-api-7d9/api previous=true\n", "SERVICE_TOKEN=<redacted>\n", "login password=<redacted> ok\n", "Authorization: Bearer <redacted>\n", "postgres://felis:<redacted>@db:5432/felis\n"},
|
||||
"journal/k3s.log": {"journal of k3s.service\n", "FELIS_OFFSITE_KEY=<redacted>\n", "relay <redacted> refused\n",
|
||||
"relay login with the cached <redacted>\n", "database auth failed for <redacted>\n", "service token <redacted> rejected\n", "node joined with <redacted>\n"},
|
||||
"cluster/felis/statefulsets.yaml": {"name: FELIS_STS_SETTING\n"},
|
||||
"cluster/felis/cronjobs.yaml": {"name: FELIS_CRON_SETTING\n"},
|
||||
"cluster/felis-build/jobs.yaml": {"name: FELIS_JOB_SETTING\n"},
|
||||
"cluster/felis/pods.yaml": {"name: FELIS_PLAIN_SETTING\n value: <redacted>\n", "name: FELIS_SMTP_PASSWORD\n valueFrom:\n secretKeyRef:\n key: password\n name: felis-smtp\n", "felis.lolicon.best/kept: \"yes\"", "name: FELIS_INIT_SETTING\n", "name: FELIS_DEBUG_SETTING\n"},
|
||||
"cluster/felis/deployments.yaml": {"name: FELIS_DEPLOY_SETTING\n value: <redacted>\n"},
|
||||
"cluster/minecraftservers.yaml": {"name: DISCORD_WEBHOOK\n value: <redacted>\n"},
|
||||
"config.txt": {fmt.Sprintf("%-34s %s\n", "server.root_domain", "games.example.org"), fmt.Sprintf("%-34s %s\n", "database.url (no password)", "postgres://[email protected]:1/felis")},
|
||||
"doctor.txt": {"the k3s token <redacted> leaked here\nprobe said password=<redacted>\n"},
|
||||
"status.txt": {" world backups unknown: connect: password=<redacted>\n"},
|
||||
"host/etc-felis.txt": {"secrets.env\n", "smtp-password\n", "felis-link.properties\n"},
|
||||
"MANIFEST.txt": {
|
||||
"The game servers' own logs are left out",
|
||||
" journal/ up to 500 lines per Felis unit and k3s, from the last 48h\n",
|
||||
" - 11 secret values, wherever they appear, read from:\n",
|
||||
"secrets.env\n", "offsite.env\n", "smtp-password\n", "felis-link.properties\n", "token\n",
|
||||
"logs/felis/felis-api-7d9/wait-db.log: container \"wait-db\" is waiting to start; token=<redacted>\n",
|
||||
// Its own words about what is redacted come through whole.
|
||||
" - passwords in URLs, private keys, Bearer and Basic credentials\n",
|
||||
"access_key=, private_key= or credentials= (and the same with a colon)\n",
|
||||
"Read it through before you send it anywhere",
|
||||
},
|
||||
} {
|
||||
for _, w := range wants {
|
||||
if !strings.Contains(files[name], w) {
|
||||
t.Errorf("%s lacks %q:\n%s", name, w, files[name])
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, gone := range []string{"managedFields", "last-applied-configuration"} {
|
||||
if strings.Contains(files["cluster/felis/pods.yaml"], gone) {
|
||||
t.Errorf("pods.yaml keeps %s:\n%s", gone, files["cluster/felis/pods.yaml"])
|
||||
}
|
||||
}
|
||||
if strings.Contains(files["MANIFEST.txt"], "absent.env") || strings.Contains(files["MANIFEST.txt"], "uploads-s3-secret-key") {
|
||||
t.Errorf("MANIFEST names files this host does not have:\n%s", files["MANIFEST.txt"])
|
||||
}
|
||||
}
|
||||
|
||||
// -server-logs adds the game servers' own logs, and says so.
|
||||
func TestSupportBundleServerLogs(t *testing.T) {
|
||||
b := bundleHost(t)
|
||||
b.serverLogs = true
|
||||
path, err := b.write(context.Background(), t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
files, _ := readBundle(t, path)
|
||||
if _, ok := files["logs/minecraft/survival-0/server.log"]; !ok {
|
||||
t.Error("no server.log with -server-logs")
|
||||
}
|
||||
if !strings.Contains(files["MANIFEST.txt"], "The game servers' own logs are in logs/ (-server-logs)") {
|
||||
t.Errorf("MANIFEST does not say server logs are in:\n%s", files["MANIFEST.txt"])
|
||||
}
|
||||
}
|
||||
|
||||
// With the cluster and the configuration gone the bundle still holds what the
|
||||
// host shows, and MANIFEST says what it could not collect.
|
||||
func TestSupportBundleWithTheClusterDown(t *testing.T) {
|
||||
b := bundleHost(t)
|
||||
b.cl, b.clErr = nil, errors.New("connection refused")
|
||||
b.cfg, b.cfgErr = nil, errors.New("felis.host.toml: no such file")
|
||||
b.wErr = errors.New("felis-watchdog.service is not installed; read the watchdog's defaults")
|
||||
path, err := b.write(context.Background(), t.TempDir())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
files, _ := readBundle(t, path)
|
||||
for _, n := range []string{"status.txt", "doctor.txt", "host/etc-felis.txt", "journal/k3s.log"} {
|
||||
if _, ok := files[n]; !ok {
|
||||
t.Errorf("no %s", n)
|
||||
}
|
||||
}
|
||||
for n := range files {
|
||||
if strings.HasPrefix(n, "cluster/") || strings.HasPrefix(n, "logs/") || n == "config.txt" {
|
||||
t.Errorf("%s with no cluster and no configuration", n)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(files["status.txt"], "felis status: the configuration did not load: felis.host.toml: no such file") {
|
||||
t.Errorf("status.txt:\n%s", files["status.txt"])
|
||||
}
|
||||
if !strings.Contains(files["MANIFEST.txt"], "\nThe watchdog's settings: felis-watchdog.service is not installed; read the watchdog's defaults\n") ||
|
||||
!strings.Contains(files["MANIFEST.txt"], " - cluster: connection refused\n") {
|
||||
t.Errorf("MANIFEST.txt:\n%s", files["MANIFEST.txt"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestShortDuration(t *testing.T) {
|
||||
for d, want := range map[time.Duration]string{
|
||||
48 * time.Hour: "48h",
|
||||
90 * time.Minute: "1h30m",
|
||||
45 * time.Minute: "45m",
|
||||
30 * time.Second: "30s",
|
||||
time.Hour + 30*time.Second: "1h0m30s",
|
||||
} {
|
||||
if got := shortDuration(d); got != want {
|
||||
t.Errorf("shortDuration(%v) = %q, want %q", d, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestScrubber(t *testing.T) {
|
||||
s := &scrubber{}
|
||||
for _, v := range []string{"known-secret-value", "known-secret-value-longer", "short", "known-secret-value"} {
|
||||
s.add(v)
|
||||
}
|
||||
if got := strings.Join(s.vals, ","); got != "known-secret-value-longer,known-secret-value" {
|
||||
t.Errorf("kept %q; want each value once, longest first, none under %d characters", s.vals, minScrubLen)
|
||||
}
|
||||
for in, want := range map[string]string{
|
||||
"a known-secret-value-longer b": "a <redacted> b",
|
||||
"a known-secret-value b": "a <redacted> b",
|
||||
"password=abcd1234 next": "password=<redacted> next",
|
||||
`{"token": "abcd1234"}`: `{"token": "<redacted>"}`,
|
||||
"SMTP_PASSWORD: s3cr3t!x": "SMTP_PASSWORD: <redacted>",
|
||||
"x-api-key=zzzz9999&q=1": "x-api-key=<redacted>&q=1",
|
||||
"Authorization: Basic dXNlcjpwYXNzd29yZA==": "Authorization: Basic <redacted>",
|
||||
"s3://AKIA:secretpart123@bucket/key": "s3://AKIA:<redacted>@bucket/key",
|
||||
"https://user@host/path": "https://user@host/path",
|
||||
"-----BEGIN EC PRIVATE KEY-----\nMHc\n-----END EC PRIVATE KEY-----": "<redacted private key>",
|
||||
"tokens: 3": "tokens: 3",
|
||||
"read the relay password (keeping the cached one)": "read the relay password (keeping the cached one)",
|
||||
`secrets "felis-smtp" not found`: `secrets "felis-smtp" not found`,
|
||||
} {
|
||||
if got := string(s.text([]byte(in))); got != want {
|
||||
t.Errorf("text(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
// Structured files keep what only looks like a secret.
|
||||
if got := string(s.values([]byte("secretName: felis-forwarding-secret and known-secret-value"))); got != "secretName: felis-forwarding-secret and <redacted>" {
|
||||
t.Errorf("values() = %q", got)
|
||||
}
|
||||
}
|
||||
@@ -15,6 +15,7 @@ import (
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
"k8s.io/client-go/rest"
|
||||
"k8s.io/client-go/tools/clientcmd"
|
||||
"k8s.io/client-go/util/retry"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
@@ -243,6 +244,15 @@ func buildSystemServerClient() (client.Client, error) {
|
||||
if err := v1alpha1.AddToScheme(scheme); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
cfg, err := hostRESTConfig()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return client.New(cfg, client.Options{Scheme: scheme})
|
||||
}
|
||||
|
||||
// hostRESTConfig is the cluster connection buildSystemServerClient describes.
|
||||
func hostRESTConfig() (*rest.Config, error) {
|
||||
cfg, err := ctrl.GetConfig()
|
||||
if err != nil {
|
||||
cfg, err = clientcmd.BuildConfigFromFlags("", hostBootstrapKubeconfigPath)
|
||||
@@ -250,7 +260,7 @@ func buildSystemServerClient() (client.Client, error) {
|
||||
return nil, fmt.Errorf("no reachable kubeconfig (tried in-cluster/$KUBECONFIG/~/.kube and %s): %w", hostBootstrapKubeconfigPath, err)
|
||||
}
|
||||
}
|
||||
return client.New(cfg, client.Options{Scheme: scheme})
|
||||
return cfg, nil
|
||||
}
|
||||
|
||||
// systemServerOutcome records what ensureSystemServers did with one service so
|
||||
|
||||
@@ -0,0 +1,202 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
"github.com/charmbracelet/lipgloss"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
)
|
||||
|
||||
// TestAlertRouteLines: the summary's alert rows say where the watchdog's
|
||||
// alerts go, and flag every route that reaches no one.
|
||||
func TestAlertRouteLines(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
r alertRoute
|
||||
alerts string
|
||||
alertsOK bool
|
||||
beat string
|
||||
beatOK bool
|
||||
}{
|
||||
{
|
||||
name: "fresh install",
|
||||
alerts: "only logged: no email relay (press e)",
|
||||
beat: "none: no outside check (troubleshooting.md §14)",
|
||||
},
|
||||
{
|
||||
name: "relay, no verified owner",
|
||||
r: alertRoute{relay: "smtp.example.com:587", heartbeat: "https://hc-ping.com/..."},
|
||||
alerts: "only logged: no verified Owner email (panel → Account)",
|
||||
beat: "https://hc-ping.com/... every 2 minutes", beatOK: true,
|
||||
},
|
||||
{
|
||||
name: "relay, owners unreadable",
|
||||
r: alertRoute{relay: "smtp.example.com:587", lookupErr: errors.New("connection refused")},
|
||||
alerts: "via smtp.example.com:587; could not read the Owner addresses",
|
||||
beat: "none: no outside check (troubleshooting.md §14)",
|
||||
},
|
||||
{
|
||||
name: "owners but no relay",
|
||||
r: alertRoute{recipients: []string{"[email protected]"}},
|
||||
alerts: "only logged: no email relay (press e)",
|
||||
beat: "none: no outside check (troubleshooting.md §14)",
|
||||
},
|
||||
{
|
||||
name: "mailed",
|
||||
r: alertRoute{relay: "smtp.example.com:587", recipients: []string{"[email protected]", "[email protected]"}, heartbeatErr: errors.New("/etc/felis/watchdog-heartbeat-url: the heartbeat URL is not an http:// or https:// URL")},
|
||||
alerts: "mailed to [email protected], [email protected] via smtp.example.com:587", alertsOK: true,
|
||||
beat: "unreadable: /etc/felis/watchdog-heartbeat-url: the heartbeat URL is not an http:// or https:// URL",
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if line, ok := c.r.alertsLine(); line != c.alerts || ok != c.alertsOK {
|
||||
t.Errorf("alerts = %q, %v; want %q, %v", line, ok, c.alerts, c.alertsOK)
|
||||
}
|
||||
if line, ok := c.r.heartbeatLine(); line != c.beat || ok != c.beatOK {
|
||||
t.Errorf("heartbeat = %q, %v; want %q, %v", line, ok, c.beat, c.beatOK)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestSummaryAlertRows: the rows show on the card, a route that needs action
|
||||
// marked so it reads without colour, and stay off a summary with no route.
|
||||
func TestSummaryAlertRows(t *testing.T) {
|
||||
m := &summaryModel{ownerUsername: "owner", alerts: &alertRoute{relay: "smtp.example.com:587", recipients: []string{"[email protected]"}}}
|
||||
v := m.View()
|
||||
for _, want := range []string{"alerts mailed to [email protected] via smtp.example.com:587", "heartbeat ⚠ none: no outside check"} {
|
||||
if !strings.Contains(v, want) {
|
||||
t.Errorf("summary lacks %q:\n%s", want, v)
|
||||
}
|
||||
}
|
||||
if strings.Contains(v, "⚠ mailed") {
|
||||
t.Errorf("a route that reaches the owners is marked:\n%s", v)
|
||||
}
|
||||
m.alerts = &alertRoute{heartbeat: "https://hc-ping.com/..."}
|
||||
v = m.View()
|
||||
for _, want := range []string{"alerts ⚠ only logged: no email relay (press e)", "heartbeat https://hc-ping.com/... every 2 minutes"} {
|
||||
if !strings.Contains(v, want) {
|
||||
t.Errorf("summary lacks %q:\n%s", want, v)
|
||||
}
|
||||
}
|
||||
m.alerts = nil
|
||||
if v = m.View(); strings.Contains(v, "alerts") || strings.Contains(v, "heartbeat") {
|
||||
t.Errorf("a summary with no route shows alert rows:\n%s", v)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRootSummaryReadsAlertRoute: the summary and the re-run status screen
|
||||
// read the route each time they open, so a relay configured with e shows at
|
||||
// once.
|
||||
func TestRootSummaryReadsAlertRoute(t *testing.T) {
|
||||
m := newTestRoot(false, consoleModeSetup, "")
|
||||
reads := 0
|
||||
route := alertRoute{}
|
||||
m.alertRoute = func(context.Context) alertRoute {
|
||||
reads++
|
||||
return route
|
||||
}
|
||||
m = drive(t, m, storageResultMsg{method: storageLocal, detail: "local disk"})
|
||||
sum, ok := m.screen.(*summaryModel)
|
||||
if !ok || sum.alerts == nil || sum.alerts.relay != "" {
|
||||
t.Fatalf("summary after storage: %T %+v", m.screen, sum)
|
||||
}
|
||||
route = alertRoute{relay: "smtp.example.com:587", recipients: []string{"[email protected]"}}
|
||||
m = drive(t, m, smtpResultMsg{configured: true})
|
||||
sum, ok = m.screen.(*summaryModel)
|
||||
if !ok || sum.alerts == nil || sum.alerts.relay != "smtp.example.com:587" {
|
||||
t.Fatalf("summary after configuring email: %T %+v", m.screen, sum)
|
||||
}
|
||||
if reads != 2 {
|
||||
t.Errorf("route read %d times, want 2", reads)
|
||||
}
|
||||
|
||||
status := newTestRoot(true, consoleModeSetup, "")
|
||||
status.alertRoute = m.alertRoute
|
||||
status.showStatus()
|
||||
if sum, ok := status.screen.(*summaryModel); !ok || sum.alerts == nil || sum.alerts.relay != "smtp.example.com:587" {
|
||||
t.Fatalf("status screen: %T %+v", status.screen, status.screen)
|
||||
}
|
||||
}
|
||||
|
||||
// TestHostAlertRoute reads the relay from the host config and the heartbeat
|
||||
// file, shows the heartbeat by its host only, and reports a database it cannot
|
||||
// reach.
|
||||
func TestHostAlertRoute(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cfg := filepath.Join(dir, "felis.host.toml")
|
||||
writeTestFile(t, cfg, testWatchdogConfig+testWatchdogSMTP, 0o600)
|
||||
beat := filepath.Join(dir, "watchdog-heartbeat-url")
|
||||
writeTestFile(t, beat, "https://hc-ping.com/check-key\n", 0o600)
|
||||
dbURL := "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"
|
||||
|
||||
r := hostAlertRoute(context.Background(), cfg, dbURL, beat)
|
||||
if r.relay != "smtp.config.example:2525" {
|
||||
t.Errorf("relay = %q", r.relay)
|
||||
}
|
||||
if r.heartbeat != "https://hc-ping.com/..." || r.heartbeatErr != nil {
|
||||
t.Errorf("heartbeat = %q, %v", r.heartbeat, r.heartbeatErr)
|
||||
}
|
||||
if r.lookupErr == nil {
|
||||
t.Errorf("an unreachable database reads as %v", r.recipients)
|
||||
}
|
||||
|
||||
noRelay := filepath.Join(dir, "no-relay.toml")
|
||||
writeTestFile(t, noRelay, testWatchdogConfig, 0o600)
|
||||
writeTestFile(t, beat, "hc-ping.com/check-key\n", 0o600)
|
||||
r = hostAlertRoute(context.Background(), noRelay, dbURL, beat)
|
||||
if r.relay != "" {
|
||||
t.Errorf("no [smtp]: relay = %q", r.relay)
|
||||
}
|
||||
if r.heartbeatErr == nil || strings.Contains(r.heartbeatErr.Error(), "check-key") {
|
||||
t.Errorf("a bad heartbeat file: %v", r.heartbeatErr)
|
||||
}
|
||||
r = hostAlertRoute(context.Background(), noRelay, dbURL, filepath.Join(dir, "none"))
|
||||
if r.heartbeat != "" || r.heartbeatErr != nil {
|
||||
t.Errorf("no heartbeat file: %q, %v", r.heartbeat, r.heartbeatErr)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSummaryWithAlertsFitsTerminal: the two rows keep the summary inside the
|
||||
// terminal, with the longest route lines.
|
||||
func TestSummaryWithAlertsFitsTerminal(t *testing.T) {
|
||||
for _, w := range []int{60, 80, 90} {
|
||||
for _, h := range []int{24, 30, 45} {
|
||||
m := newTestRoot(false, consoleModeSetup, "")
|
||||
m.alertRoute = func(context.Context) alertRoute {
|
||||
return alertRoute{relay: "smtp.example.com:587", lookupErr: errors.New("dial tcp 127.0.0.1:5432: connect: connection refused")}
|
||||
}
|
||||
m = drive(t, m, tea.WindowSizeMsg{Width: w, Height: h})
|
||||
m = drive(t, m, preflightDoneMsg{})
|
||||
m = drive(t, m, ownerResultMsg{username: "owner", setupTokenURL: "https://op.console.example.com/setup?token=t0ken"})
|
||||
m = drive(t, m, connectResultMsg{method: connectLocal, panelHostname: "panel.example.com"})
|
||||
m = drive(t, m, storageResultMsg{method: storageLocal, detail: "local disk · /var/lib/felis/uploads"})
|
||||
if _, ok := m.screen.(*summaryModel); !ok {
|
||||
t.Fatalf("screen = %T, want the summary", m.screen)
|
||||
}
|
||||
if got := lipgloss.Height(m.View()); got > h {
|
||||
t.Errorf("terminal %dx%d: summary with alert rows = %d rows (exceeds height)", w, h, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestConsoleRootReadsHostAlertRoute: the console the host runs gives its
|
||||
// summary this host's route, read against its database.
|
||||
func TestConsoleRootReadsHostAlertRoute(t *testing.T) {
|
||||
db := config.DatabaseConfig{URL: "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"}
|
||||
rm := newConsoleRoot(context.Background(), &fakeOwnerStore{}, db, "felis.example.com", "admin.felis.example.com", "panel.felis.example.com", "", "minecraft", "root", false, consoleModeSetup, recoveryConfig{})
|
||||
if rm.alertRoute == nil {
|
||||
t.Fatal("the console's summary reads no alert route")
|
||||
}
|
||||
if r := rm.alertRoute(context.Background()); r.lookupErr == nil {
|
||||
t.Errorf("the route did not query the console's database: %+v", r)
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"io"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/store"
|
||||
)
|
||||
@@ -13,10 +14,10 @@ import (
|
||||
// applyMigrations runs any pending schema migrations and returns the resulting
|
||||
// applied count. Used by the preflight stage to self-heal a freshly bootstrapped
|
||||
// (or upgraded) database.
|
||||
func applyMigrations(dbURL string) (int, error) {
|
||||
func applyMigrations(db config.DatabaseConfig) (int, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
drv, err := store.Open(ctx, dbURL)
|
||||
drv, err := store.Open(ctx, db.URL)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("open db: %w", err)
|
||||
}
|
||||
@@ -27,7 +28,7 @@ func applyMigrations(dbURL string) (int, error) {
|
||||
}
|
||||
// Same guard as `felis migrate up`: never roll a populated database forward
|
||||
// without a snapshot to roll back to.
|
||||
if _, err := preMigrateBackup(ctx, drv, migrations, dbURL, dbbackup.DefaultDir, io.Discard); err != nil {
|
||||
if _, err := preMigrateBackup(ctx, drv, migrations, db, dbbackup.DefaultDir, io.Discard); err != nil {
|
||||
return 0, fmt.Errorf("pre-migration backup: %w", err)
|
||||
}
|
||||
if _, err := store.Up(ctx, drv, migrations); err != nil {
|
||||
|
||||
+147
-28
@@ -5,6 +5,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
|
||||
@@ -14,9 +15,8 @@ import (
|
||||
)
|
||||
|
||||
type owAuthMsg struct {
|
||||
matched string
|
||||
ok bool
|
||||
err error
|
||||
start recoveryStart
|
||||
err error
|
||||
}
|
||||
|
||||
type owProvisionMsg struct {
|
||||
@@ -29,14 +29,15 @@ type owStep int
|
||||
const (
|
||||
owAuth owStep = iota
|
||||
owOverride
|
||||
owCode // typing the recovery code mailed to the named admin
|
||||
owProvision
|
||||
owWorking
|
||||
owDone
|
||||
owError
|
||||
)
|
||||
|
||||
// ownerModel collects the owner account. The input phases (admin auth, root
|
||||
// override, owner details) are huh forms; the async phases (verifying,
|
||||
// ownerModel collects the owner account. The input phases (admin name, recovery
|
||||
// code, root override, owner details) are huh forms; the async phases (verifying,
|
||||
// provisioning) show a spinner; the done phase shows the credential card. The
|
||||
// outward contract is unchanged: it emits an ownerResultMsg when finished.
|
||||
//
|
||||
@@ -55,6 +56,18 @@ type ownerModel struct {
|
||||
mode string // "bootstrap", "recovery", "root_override"
|
||||
accountable string
|
||||
attempt string
|
||||
recovery recoveryConfig
|
||||
|
||||
// Recovery proof: the admin the typed name resolved to, the code mailed to it,
|
||||
// and once settled either how it was proven or why the run fell back to the
|
||||
// override.
|
||||
admin *api.StaffUser
|
||||
code *recoveryCode
|
||||
codeNote string // "wrong code" line shown above a rebuilt code form
|
||||
verifiedBy string
|
||||
codeSentTo string
|
||||
skip string // otpSkip*
|
||||
skipDetail string
|
||||
|
||||
step owStep
|
||||
form *huh.Form
|
||||
@@ -66,6 +79,7 @@ type ownerModel struct {
|
||||
|
||||
// huh-bound form values
|
||||
authUser string
|
||||
codeInput string
|
||||
overrideTok string
|
||||
ownerUser string
|
||||
ownerEmail string
|
||||
@@ -124,6 +138,12 @@ func newOperatorModel(ctx context.Context, store ownerStore, osUser string) *own
|
||||
return m
|
||||
}
|
||||
|
||||
// withRecovery hands the model the relay its recovery codes go through.
|
||||
func (m *ownerModel) withRecovery(r recoveryConfig) *ownerModel {
|
||||
m.recovery = r
|
||||
return m
|
||||
}
|
||||
|
||||
func (m *ownerModel) Init() tea.Cmd { return m.form.Init() }
|
||||
|
||||
// subject is the human label for the account being provisioned, branching every
|
||||
@@ -152,7 +172,7 @@ func (m *ownerModel) sized(f *huh.Form) *huh.Form {
|
||||
}
|
||||
|
||||
func (m *ownerModel) isFormStep() bool {
|
||||
return m.step == owAuth || m.step == owOverride || m.step == owProvision
|
||||
return m.step == owAuth || m.step == owOverride || m.step == owCode || m.step == owProvision
|
||||
}
|
||||
|
||||
func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
@@ -161,16 +181,15 @@ func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
if msg.err != nil {
|
||||
return m, m.failCmd(msg.err)
|
||||
}
|
||||
if msg.ok {
|
||||
m.mode = "recovery"
|
||||
m.accountable = msg.matched
|
||||
m.step = owProvision
|
||||
m.form = m.sized(m.buildProvisionForm())
|
||||
m.admin = msg.start.admin
|
||||
if msg.start.code != nil {
|
||||
m.code = msg.start.code
|
||||
m.codeNote = ""
|
||||
m.step = owCode
|
||||
m.form = m.sized(m.buildCodeForm())
|
||||
return m, m.form.Init()
|
||||
}
|
||||
m.step = owOverride
|
||||
m.form = m.sized(m.buildOverrideForm())
|
||||
return m, m.form.Init()
|
||||
return m.toOverride(msg.start.skip, msg.start.detail)
|
||||
|
||||
case owProvisionMsg:
|
||||
if msg.err != nil {
|
||||
@@ -221,7 +240,9 @@ func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
case "ctrl+c":
|
||||
return m, tea.Quit
|
||||
case "esc":
|
||||
if m.step == owOverride {
|
||||
if m.step == owOverride || m.step == owCode {
|
||||
// Start over: a code mailed for the old attempt dies with it.
|
||||
m.admin, m.code, m.skip, m.skipDetail = nil, nil, "", ""
|
||||
m.step = owAuth
|
||||
m.form = m.sized(m.buildAuthForm())
|
||||
return m, m.form.Init()
|
||||
@@ -253,18 +274,41 @@ func (m *ownerModel) onFormComplete() (tea.Model, tea.Cmd) {
|
||||
case owAuth:
|
||||
m.attempt = strings.TrimSpace(m.authUser)
|
||||
m.step = owWorking
|
||||
m.working = "Verifying admin…"
|
||||
user := m.authUser
|
||||
m.working = "Sending a recovery code…"
|
||||
user, rc, osUser, op := m.authUser, m.recovery, m.osUser, m.operation
|
||||
return m, tea.Batch(m.sp.Tick, func() tea.Msg {
|
||||
matched, ok, err := authenticateAdmin(m.ctx, m.store, user)
|
||||
return owAuthMsg{matched: matched, ok: ok, err: err}
|
||||
start, err := beginRecovery(m.ctx, m.store, rc, user, osUser, op)
|
||||
return owAuthMsg{start: start, err: err}
|
||||
})
|
||||
case owCode:
|
||||
typed := strings.TrimSpace(m.codeInput)
|
||||
m.codeInput = ""
|
||||
if typed == breakGlassOverrideToken {
|
||||
m.skip, m.skipDetail = otpSkipByOperator, ""
|
||||
m.code = nil
|
||||
return m.proceedAsRoot()
|
||||
}
|
||||
switch m.code.check(typed, m.recovery.clock()) {
|
||||
case codeAccepted:
|
||||
m.mode = "recovery"
|
||||
m.accountable = m.admin.Username
|
||||
m.verifiedBy = verifiedByEmailOTP
|
||||
m.codeSentTo = m.admin.Email
|
||||
m.code = nil
|
||||
m.step = owProvision
|
||||
m.form = m.sized(m.buildProvisionForm())
|
||||
return m, m.form.Init()
|
||||
case codeWrong:
|
||||
m.codeNote = fmt.Sprintf("That code is wrong. %d attempts left.", m.code.attemptsLeft())
|
||||
m.form = m.sized(m.buildCodeForm())
|
||||
return m, m.form.Init()
|
||||
case codeExpired:
|
||||
return m.toOverride(otpSkipCodeExpired, "")
|
||||
default:
|
||||
return m.toOverride(otpSkipCodeRejected, fmt.Sprintf("%d wrong codes", recoveryCodeAttempts))
|
||||
}
|
||||
case owOverride:
|
||||
m.mode = "root_override"
|
||||
m.accountable = m.osUser
|
||||
m.step = owProvision
|
||||
m.form = m.sized(m.buildProvisionForm())
|
||||
return m, m.form.Init()
|
||||
return m.proceedAsRoot()
|
||||
case owProvision:
|
||||
m.username = strings.TrimSpace(m.ownerUser)
|
||||
m.step = owWorking
|
||||
@@ -274,6 +318,24 @@ func (m *ownerModel) onFormComplete() (tea.Model, tea.Cmd) {
|
||||
return m, nil
|
||||
}
|
||||
|
||||
// toOverride records why no code proved an admin and asks for the typed OVERRIDE.
|
||||
func (m *ownerModel) toOverride(skip, detail string) (tea.Model, tea.Cmd) {
|
||||
m.skip, m.skipDetail = skip, detail
|
||||
m.code = nil
|
||||
m.step = owOverride
|
||||
m.form = m.sized(m.buildOverrideForm())
|
||||
return m, m.form.Init()
|
||||
}
|
||||
|
||||
// proceedAsRoot is the typed OVERRIDE: the run goes on as the OS user, unverified.
|
||||
func (m *ownerModel) proceedAsRoot() (tea.Model, tea.Cmd) {
|
||||
m.mode = "root_override"
|
||||
m.accountable = m.osUser
|
||||
m.step = owProvision
|
||||
m.form = m.sized(m.buildProvisionForm())
|
||||
return m, m.form.Init()
|
||||
}
|
||||
|
||||
func (m *ownerModel) provisionCmd() tea.Cmd {
|
||||
op := breakGlassOp{
|
||||
mode: m.mode,
|
||||
@@ -282,6 +344,10 @@ func (m *ownerModel) provisionCmd() tea.Cmd {
|
||||
ownerUsername: m.username,
|
||||
ownerEmail: m.ownerEmail,
|
||||
attemptedAdmin: m.attempt,
|
||||
verifiedBy: m.verifiedBy,
|
||||
codeSentTo: m.codeSentTo,
|
||||
otpSkipped: m.skip,
|
||||
otpSkipDetail: m.skipDetail,
|
||||
}
|
||||
// performAddOperator and performBreakGlass share a signature; the operation
|
||||
// discriminator selects which one runs. The operator path is insert-only and
|
||||
@@ -320,7 +386,7 @@ func (m *ownerModel) buildAuthForm() *huh.Form {
|
||||
return m.sized(newFelisForm(huh.NewGroup(
|
||||
huh.NewNote().
|
||||
Title("Admin authentication").
|
||||
Description("A staff account already exists. Identify yourself to continue."),
|
||||
Description("A staff account already exists. Name yours: a one-time code goes to its verified email address."),
|
||||
huh.NewInput().
|
||||
Title("Admin username").
|
||||
Value(&m.authUser).
|
||||
@@ -328,11 +394,64 @@ func (m *ownerModel) buildAuthForm() *huh.Form {
|
||||
)))
|
||||
}
|
||||
|
||||
func (m *ownerModel) buildCodeForm() *huh.Form {
|
||||
desc := fmt.Sprintf("A recovery code went to %s, the verified address of %q. It works for %d minutes.\n\n"+
|
||||
"No mail? Type %s to go on as OS user %q with root authority; the audit log records that as an unverified override. Esc starts over.",
|
||||
maskEmail(m.admin.Email), m.admin.Username, int(recoveryCodeTTL/time.Minute), breakGlassOverrideToken, m.osUser)
|
||||
if m.codeNote != "" {
|
||||
desc = m.codeNote + "\n\n" + desc
|
||||
}
|
||||
return m.sized(newFelisForm(huh.NewGroup(
|
||||
huh.NewNote().Title("Email verification").Description(desc),
|
||||
huh.NewInput().
|
||||
Title("Recovery code").
|
||||
Value(&m.codeInput).
|
||||
Validate(func(s string) error {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == breakGlassOverrideToken || isRecoveryCodeShape(s) {
|
||||
return nil
|
||||
}
|
||||
return errors.New("enter the 6-digit code, or " + breakGlassOverrideToken)
|
||||
}),
|
||||
)))
|
||||
}
|
||||
|
||||
func isRecoveryCodeShape(s string) bool {
|
||||
if len(s) != 6 {
|
||||
return false
|
||||
}
|
||||
for _, c := range s {
|
||||
if c < '0' || c > '9' {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// overrideReason says why no code proved an admin, first line of the override form.
|
||||
func (m *ownerModel) overrideReason() string {
|
||||
switch m.skip {
|
||||
case otpSkipUnknownAdmin:
|
||||
return fmt.Sprintf("No staff account is named %q.", m.attempt)
|
||||
case otpSkipNoVerifiedEmail:
|
||||
return fmt.Sprintf("%q has no verified email address, so no recovery code can reach it.", m.admin.Username)
|
||||
case otpSkipNoRelay:
|
||||
return "No mail relay can send a recovery code: " + m.skipDetail + "."
|
||||
case otpSkipSendFailed:
|
||||
return "The recovery code could not be sent: " + m.skipDetail + "."
|
||||
case otpSkipCodeExpired:
|
||||
return "The recovery code expired."
|
||||
case otpSkipCodeRejected:
|
||||
return fmt.Sprintf("%d wrong codes; that code no longer works.", recoveryCodeAttempts)
|
||||
}
|
||||
return "No admin was verified."
|
||||
}
|
||||
|
||||
func (m *ownerModel) buildOverrideForm() *huh.Form {
|
||||
return m.sized(newFelisForm(huh.NewGroup(
|
||||
huh.NewNote().
|
||||
Title("Root override").
|
||||
Description(fmt.Sprintf("That credential did not match. Proceed as OS user %q with root authority by typing the confirmation token.", m.osUser)),
|
||||
Description(m.overrideReason()+"\n\n"+fmt.Sprintf("Proceed as OS user %q with root authority by typing the confirmation token. The audit log records this run as an unverified root override and the reason above. Esc starts over.", m.osUser)),
|
||||
huh.NewInput().
|
||||
Title("Type "+breakGlassOverrideToken+" to confirm").
|
||||
Value(&m.overrideTok).
|
||||
@@ -352,14 +471,14 @@ func (m *ownerModel) buildProvisionForm() *huh.Form {
|
||||
desc := fmt.Sprintf("Create the first Owner — recorded as OS user %q.", m.osUser)
|
||||
switch m.mode {
|
||||
case "recovery":
|
||||
desc = fmt.Sprintf("Authenticated as %q.", m.accountable)
|
||||
desc = fmt.Sprintf("Verified as %q by an email code.", m.accountable)
|
||||
case "root_override":
|
||||
desc = "Root override — the Owner will be reset."
|
||||
}
|
||||
if m.operation == bgAddOperator {
|
||||
switch m.mode {
|
||||
case "recovery":
|
||||
desc = fmt.Sprintf("Add an Operator — authenticated as %q.", m.accountable)
|
||||
desc = fmt.Sprintf("Add an Operator — verified as %q by an email code.", m.accountable)
|
||||
case "root_override":
|
||||
desc = "Add an Operator (root override)."
|
||||
}
|
||||
|
||||
@@ -3,6 +3,8 @@ package main
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
|
||||
"github.com/charmbracelet/bubbles/spinner"
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
)
|
||||
@@ -14,7 +16,7 @@ import (
|
||||
// of the wizard can assume a healthy backend. It is non-interactive — it runs
|
||||
// to completion and reports back to the root via preflightDoneMsg.
|
||||
type preflightModel struct {
|
||||
dbURL string
|
||||
db config.DatabaseConfig
|
||||
rootDomain string
|
||||
adminHost string
|
||||
|
||||
@@ -57,11 +59,11 @@ type pfMigApplyMsg struct {
|
||||
|
||||
type pfPanelMsg struct{ err error }
|
||||
|
||||
func newPreflightModel(dbURL, rootDomain, adminHostname string) *preflightModel {
|
||||
func newPreflightModel(db config.DatabaseConfig, rootDomain, adminHostname string) *preflightModel {
|
||||
sp := spinner.New()
|
||||
sp.Spinner = spinner.Dot
|
||||
sp.Style = tuiLabel
|
||||
return &preflightModel{dbURL: dbURL, rootDomain: rootDomain, adminHost: adminHostname, sp: sp, state: pfCheckDB}
|
||||
return &preflightModel{db: db, rootDomain: rootDomain, adminHost: adminHostname, sp: sp, state: pfCheckDB}
|
||||
}
|
||||
|
||||
func (m *preflightModel) Init() tea.Cmd {
|
||||
@@ -200,7 +202,7 @@ func (m *preflightModel) errText() string {
|
||||
}
|
||||
|
||||
func (m *preflightModel) checkDB() tea.Cmd {
|
||||
return func() tea.Msg { return pfDBMsg{err: checkPostgres(m.dbURL)} }
|
||||
return func() tea.Msg { return pfDBMsg{err: checkPostgres(m.db.URL)} }
|
||||
}
|
||||
|
||||
func (m *preflightModel) checkMigrations() tea.Cmd {
|
||||
@@ -208,7 +210,7 @@ func (m *preflightModel) checkMigrations() tea.Cmd {
|
||||
// The sets, not their sizes: a database a newer release migrated can hold as
|
||||
// many rows as this build has migrations, and must stop here rather than be
|
||||
// "healed" by an older binary.
|
||||
s, err := readSchema(m.dbURL)
|
||||
s, err := readSchema(m.db.URL)
|
||||
if err == nil {
|
||||
err = s.Newer()
|
||||
}
|
||||
@@ -218,7 +220,7 @@ func (m *preflightModel) checkMigrations() tea.Cmd {
|
||||
|
||||
func (m *preflightModel) applyMigrations() tea.Cmd {
|
||||
return func() tea.Msg {
|
||||
applied, err := applyMigrations(m.dbURL)
|
||||
applied, err := applyMigrations(m.db)
|
||||
return pfMigApplyMsg{applied: applied, err: err}
|
||||
}
|
||||
}
|
||||
|
||||
+23
-6
@@ -5,6 +5,7 @@ import (
|
||||
"strings"
|
||||
|
||||
"felis.lolicon.best/internal/cfsetup"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
@@ -133,7 +134,7 @@ type rootModel struct {
|
||||
err error
|
||||
|
||||
mode consoleMode
|
||||
dbURL string
|
||||
db config.DatabaseConfig
|
||||
store ownerStore
|
||||
osUser string
|
||||
rootDomain string
|
||||
@@ -142,13 +143,17 @@ type rootModel struct {
|
||||
accessAud string
|
||||
namespace string // minecraft workload namespace (cfg.K8s.Namespace); target of the halt op
|
||||
adminExists bool
|
||||
recovery recoveryConfig // how the account operations mail a recovery code
|
||||
// alertRoute reads where the watchdog's alerts go for the summary; nil
|
||||
// leaves those rows out.
|
||||
alertRoute func(context.Context) alertRoute
|
||||
}
|
||||
|
||||
func newRootModel(ctx context.Context, store ownerStore, dbURL, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode) *rootModel {
|
||||
func newRootModel(ctx context.Context, store ownerStore, db config.DatabaseConfig, rootDomain, adminHostname, panelHostname, accessAud, namespace, osUser string, adminExists bool, mode consoleMode) *rootModel {
|
||||
rm := &rootModel{
|
||||
ctx: ctx,
|
||||
reviewing: -1,
|
||||
dbURL: dbURL,
|
||||
db: db,
|
||||
store: store,
|
||||
osUser: osUser,
|
||||
rootDomain: rootDomain,
|
||||
@@ -179,7 +184,7 @@ func newRootModel(ctx context.Context, store ownerStore, dbURL, rootDomain, admi
|
||||
}
|
||||
} else {
|
||||
rm.stage = stagePreflight
|
||||
rm.screen = newPreflightModel(dbURL, rootDomain, adminHostname)
|
||||
rm.screen = newPreflightModel(db, rootDomain, adminHostname)
|
||||
}
|
||||
return rm
|
||||
}
|
||||
@@ -221,7 +226,7 @@ func (m *rootModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
m.stage = stageOwner
|
||||
switch msg.op {
|
||||
case bgAddOperator:
|
||||
return m.adopt(newOperatorModel(m.ctx, m.store, m.osUser))
|
||||
return m.adopt(newOperatorModel(m.ctx, m.store, m.osUser).withRecovery(m.recovery))
|
||||
case bgHaltServer:
|
||||
return m.adopt(newHaltModel(m.ctx, m.store, m.namespace, m.osUser))
|
||||
case bgSyncBackup:
|
||||
@@ -229,7 +234,7 @@ func (m *rootModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
// Service + service token the peer dials live in the control namespace.
|
||||
return m.adopt(newBackupModel(m.ctx, m.namespace, platform.DefaultControlNamespace, m.osUser))
|
||||
default:
|
||||
return m.adopt(newOwnerModel(m.ctx, m.store, m.osUser, m.adminExists))
|
||||
return m.adopt(newOwnerModel(m.ctx, m.store, m.osUser, m.adminExists).withRecovery(m.recovery))
|
||||
}
|
||||
|
||||
case haltResultMsg:
|
||||
@@ -554,6 +559,7 @@ func (m *rootModel) showSummary() (tea.Model, tea.Cmd) {
|
||||
routedHosts: routed,
|
||||
localHint: m.result.connectMethod == connectLocal,
|
||||
alreadySetUp: m.result.alreadySetUp,
|
||||
alerts: m.readAlertRoute(),
|
||||
})
|
||||
}
|
||||
|
||||
@@ -574,9 +580,20 @@ func (m *rootModel) showStatus() (tea.Model, tea.Cmd) {
|
||||
accessLabel: accessLabel,
|
||||
alreadySetUp: true,
|
||||
localHint: m.accessAud == "" && rootDomainEmbeddedIP(m.rootDomain) != "",
|
||||
alerts: m.readAlertRoute(),
|
||||
})
|
||||
}
|
||||
|
||||
// readAlertRoute reads the alert route afresh, so the summary shows a relay the
|
||||
// Owner just configured with e.
|
||||
func (m *rootModel) readAlertRoute() *alertRoute {
|
||||
if m.alertRoute == nil {
|
||||
return nil
|
||||
}
|
||||
r := m.alertRoute(m.ctx)
|
||||
return &r
|
||||
}
|
||||
|
||||
func panelURLFor(method connectMethod, panelHostname, rootDomain, adminHostname string) string {
|
||||
if method != connectLocal && panelHostname != "" {
|
||||
return "https://" + panelHostname
|
||||
|
||||
@@ -5,6 +5,8 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
)
|
||||
|
||||
@@ -30,7 +32,7 @@ func newTestRoot(adminExists bool, mode consoleMode, accessAud string) *rootMode
|
||||
return newRootModel(
|
||||
context.Background(),
|
||||
&fakeOwnerStore{},
|
||||
"postgres://localhost/felis",
|
||||
config.DatabaseConfig{URL: "postgres://localhost/felis"},
|
||||
"felis.example.com",
|
||||
"admin.felis.example.com",
|
||||
"panel.felis.example.com",
|
||||
|
||||
+14
-5
@@ -276,11 +276,14 @@ func validateSMTPFrom(s string) error {
|
||||
}
|
||||
|
||||
// currentSMTPInputs reads the relay already recorded in felis.toml so the form
|
||||
// pre-fills the non-secret fields. The password lives only in the felis-smtp
|
||||
// Secret and is deliberately never read back — it must be re-entered to change.
|
||||
// pre-fills the non-secret fields. The password (the felis-smtp Secret and its
|
||||
// host copy) is deliberately never read back — it must be re-entered to change.
|
||||
// Any read error falls back to a blank form rather than blocking reconfig.
|
||||
func currentSMTPInputs() smtpInputs {
|
||||
cfg, err := config.Load(hostSetupConfigPath)
|
||||
func currentSMTPInputs() smtpInputs { return smtpInputsFrom(hostSetupConfigPath) }
|
||||
|
||||
// smtpInputsFrom is currentSMTPInputs for the config at path.
|
||||
func smtpInputsFrom(path string) smtpInputs {
|
||||
cfg, err := config.Load(path)
|
||||
if err != nil || cfg.SMTP.Host == "" {
|
||||
return smtpInputs{}
|
||||
}
|
||||
@@ -295,7 +298,8 @@ func currentSMTPInputs() smtpInputs {
|
||||
// applySMTPConfig proves the relay works, then persists it and rolls felis-api:
|
||||
// Ping (a full transaction — connect/STARTTLS/AUTH/MAIL FROM/RCPT/DATA, which
|
||||
// delivers one self-test message to the From address) → [smtp] into both config
|
||||
// files → the felis-smtp Secret → the config Secret → rollout. A failed Ping
|
||||
// files → the password into /etc/felis/smtp-password (hostcreds.go) → the
|
||||
// felis-smtp Secret → the config Secret → rollout. A failed Ping
|
||||
// leaves the install untouched, so a bad relay dies at the keyboard, not at a
|
||||
// player's OTP.
|
||||
//
|
||||
@@ -330,6 +334,11 @@ func applySMTPConfig(ctx context.Context, in smtpInputs) error {
|
||||
return err
|
||||
}
|
||||
}
|
||||
// The host copy first: should the Secret fail, the next installer run applies
|
||||
// it from this file.
|
||||
if err := writeHostCredential(hostSMTPPasswordPath, in.password); err != nil {
|
||||
return fmt.Errorf("keep the relay password in %s: %w", hostSMTPPasswordPath, err)
|
||||
}
|
||||
if err := applySMTPSecret(ctx, in.password); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -53,8 +53,17 @@ func applyStorageConfig(ctx context.Context, method storageMethod, in s3Inputs)
|
||||
return err
|
||||
}
|
||||
// S3: land the credentials in their own Secret BEFORE the roll, so the optional
|
||||
// env refs resolve on the fresh pod. Local needs no Secret.
|
||||
// env refs resolve on the fresh pod, and on the host before that, so the next
|
||||
// installer run can apply the Secret again (hostcreds.go). Local needs no Secret.
|
||||
if method == storageS3 {
|
||||
for _, c := range []struct{ path, value string }{
|
||||
{hostUploadsS3AccessKeyPath, in.accessKey},
|
||||
{hostUploadsS3SecretKeyPath, in.secretKey},
|
||||
} {
|
||||
if err := writeHostCredential(c.path, c.value); err != nil {
|
||||
return fmt.Errorf("keep the bucket credentials in %s: %w", c.path, err)
|
||||
}
|
||||
}
|
||||
if err := applyUploadsS3Secret(ctx, in.accessKey, in.secretKey); err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -70,8 +79,9 @@ func applyStorageConfig(ctx context.Context, method storageMethod, in s3Inputs)
|
||||
|
||||
// currentStorageInputs reads the storage backend already recorded in felis.toml so
|
||||
// the reconfigure flow can pre-select the method and pre-fill the non-secret S3
|
||||
// fields (endpoint/bucket/region). Credentials live only in the felis-uploads-s3
|
||||
// Secret and are deliberately never read back — they must be re-entered to change.
|
||||
// fields (endpoint/bucket/region). The credentials (the felis-uploads-s3 Secret
|
||||
// and its host copies) are deliberately never read back — they must be re-entered
|
||||
// to change.
|
||||
// Any read error falls back to a blank local default rather than blocking reconfig.
|
||||
func currentStorageInputs() (storageMethod, s3Inputs) {
|
||||
cfg, err := config.Load(hostSetupConfigPath)
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
tea "github.com/charmbracelet/bubbletea"
|
||||
)
|
||||
@@ -20,6 +22,8 @@ type summaryModel struct {
|
||||
routedHosts []string
|
||||
alreadySetUp bool // re-run: Owner pre-existed
|
||||
localHint bool // show the self-signed-cert note
|
||||
// alerts is where the watchdog's alerts go; nil leaves the rows out.
|
||||
alerts *alertRoute
|
||||
}
|
||||
|
||||
func (m *summaryModel) Init() tea.Cmd { return nil }
|
||||
@@ -73,6 +77,12 @@ func (m *summaryModel) View() string {
|
||||
if m.panelURL != "" {
|
||||
card.WriteString(tuiLabel.Render("panel ") + m.panelURL + "\n")
|
||||
}
|
||||
if m.alerts != nil {
|
||||
line, ok := m.alerts.alertsLine()
|
||||
card.WriteString(routeRow("alerts ", line, ok))
|
||||
line, ok = m.alerts.heartbeatLine()
|
||||
card.WriteString(routeRow("heartbeat ", line, ok))
|
||||
}
|
||||
b.WriteString(tuiCardStyle.Render(strings.TrimRight(card.String(), "\n")) + "\n\n")
|
||||
|
||||
b.WriteString(tuiHint.Render("ℹ Everything else — servers, users, plugins — is configured in the panel. You won't need this console again.") + "\n")
|
||||
@@ -83,3 +93,70 @@ func (m *summaryModel) View() string {
|
||||
b.WriteString("\n" + tuiAction("c", "change connection", "s", "change storage", "e", "configure email", "enter/esc", "exit"))
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// alertRoute is where this host's watchdog alerts go, as the summary shows it:
|
||||
// by mail through the [smtp] relay to the Owners' verified addresses, and the
|
||||
// heartbeat that notices the host itself going down (docs/troubleshooting.md
|
||||
// §14). Setup runs mail-less by design, so a fresh install has neither; the
|
||||
// summary says so where the Owner can press e.
|
||||
type alertRoute struct {
|
||||
relay string // "host:port", "" with no [smtp] relay
|
||||
recipients []string // enabled Owners' verified addresses
|
||||
lookupErr error // the recipients could not be read
|
||||
heartbeat string // the heartbeat URL's scheme and host, "" with none
|
||||
heartbeatErr error // the heartbeat file does not read
|
||||
}
|
||||
|
||||
// hostAlertRoute reads the route from the host config at cfgPath, the database
|
||||
// at dbURL and the heartbeat file at heartbeatPath: what the next watchdog run
|
||||
// uses.
|
||||
func hostAlertRoute(ctx context.Context, cfgPath, dbURL, heartbeatPath string) alertRoute {
|
||||
var r alertRoute
|
||||
if in := smtpInputsFrom(cfgPath); in.host != "" {
|
||||
r.relay = in.host + ":" + in.port
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(ctx, 5*time.Second)
|
||||
defer cancel()
|
||||
r.recipients, r.lookupErr = ownerEmails(ctx, dbURL)
|
||||
u, err := readHeartbeatURL(heartbeatPath)
|
||||
switch {
|
||||
case err != nil:
|
||||
r.heartbeatErr = err
|
||||
case u != "":
|
||||
r.heartbeat = redactURL(u)
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// alertsLine is the summary's alerts row; ok is false when the alerts reach no one.
|
||||
func (r alertRoute) alertsLine() (line string, ok bool) {
|
||||
switch {
|
||||
case r.relay == "":
|
||||
return "only logged: no email relay (press e)", false
|
||||
case r.lookupErr != nil:
|
||||
return "via " + r.relay + "; could not read the Owner addresses", false
|
||||
case len(r.recipients) == 0:
|
||||
return "only logged: no verified Owner email (panel → Account)", false
|
||||
}
|
||||
return "mailed to " + strings.Join(r.recipients, ", ") + " via " + r.relay, true
|
||||
}
|
||||
|
||||
// heartbeatLine is the summary's heartbeat row; ok is false with no heartbeat.
|
||||
func (r alertRoute) heartbeatLine() (line string, ok bool) {
|
||||
switch {
|
||||
case r.heartbeatErr != nil:
|
||||
return "unreadable: " + r.heartbeatErr.Error(), false
|
||||
case r.heartbeat == "":
|
||||
return "none: no outside check (troubleshooting.md §14)", false
|
||||
}
|
||||
return r.heartbeat + " every 2 minutes", true
|
||||
}
|
||||
|
||||
// routeRow renders one alert row, marked and in the warning style when it
|
||||
// needs action: the mark reads without colour too.
|
||||
func routeRow(label, line string, ok bool) string {
|
||||
if !ok {
|
||||
line = tuiWarn.Render("⚠ " + line)
|
||||
}
|
||||
return tuiLabel.Render(label) + line + "\n"
|
||||
}
|
||||
+6
-4
@@ -128,8 +128,9 @@ var updateTargets = []updateTarget{
|
||||
selector: "postgres",
|
||||
help: "select PostgreSQL",
|
||||
component: "postgresql",
|
||||
note: "PostgreSQL comes from the distribution's packages, so a minor release is a package update followed by a restart (a few seconds without the API). A new major needs pg_upgrade first: docs/operations.md §4",
|
||||
command: "sudo dnf upgrade 'postgresql*' || sudo apt-get install --only-upgrade 'postgresql*'; sudo systemctl restart postgresql",
|
||||
note: "PostgreSQL runs as the felis-postgres Deployment from the image the Felis release pins by digest; a newer minor reaches the host with a release that moves the pin, and the installer re-run restarts the database on it (a few seconds without the API). A new major is a dump and restore: docs/operations.md §4",
|
||||
command: installerRerun,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "mc",
|
||||
@@ -149,8 +150,9 @@ var updateTargets = []updateTarget{
|
||||
// reinstall/repair case.
|
||||
//
|
||||
// It never applies anything and never mutates the node, so unlike setup/breakGlass
|
||||
// it needs no root. The versions it reads come from this host: k3s, cloudflared and
|
||||
// PostgreSQL answer `--version`, Velocity's version is read out of the installed jar's
|
||||
// it needs no root, apart from PostgreSQL's version, which the felis-postgres container
|
||||
// answers through the cluster's admin kubeconfig. The versions it reads come from this
|
||||
// host: k3s, cloudflared and PostgreSQL answer `--version`, Velocity's version is read out of the installed jar's
|
||||
// manifest, the JRE's out of its release file, and felis-api's is this binary's own
|
||||
// build stamp — the same value `felis version` prints, which is what the user asked
|
||||
// to be the source of truth.
|
||||
|
||||
@@ -215,8 +215,8 @@ func TestUpdateSelectorsAreFlags(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// k3s and cloudflared move only when the re-run is told to; PostgreSQL is the package
|
||||
// manager's, so its guidance carries no installer trailer.
|
||||
// k3s and cloudflared move only when the re-run is told to; the release pins the JRE
|
||||
// build and the PostgreSQL image, so their guidance is the plain re-run.
|
||||
func TestApplyGuidanceForHostDependencies(t *testing.T) {
|
||||
notify := func(c string) updater.Result {
|
||||
return planResult([]updates.Action{{Component: c, Kind: updates.ActionNotify, LatestKnown: true}})
|
||||
@@ -232,8 +232,8 @@ func TestApplyGuidanceForHostDependencies(t *testing.T) {
|
||||
t.Errorf("--jre guidance is the plain installer re-run:\n%s", jre)
|
||||
}
|
||||
pg := renderApplyGuidance(notify("postgresql"), map[string]bool{"postgres": true}, false)
|
||||
if !strings.Contains(pg, "apt-get install --only-upgrade") || strings.Contains(pg, "Re-running the installer") {
|
||||
t.Errorf("--postgres guidance is the package manager, without the installer trailer:\n%s", pg)
|
||||
if !strings.Contains(pg, "| sudo bash") || strings.Contains(pg, "FELIS_UPGRADE_DEPS") || strings.Contains(pg, "apt-get") || strings.Contains(pg, "systemctl") {
|
||||
t.Errorf("--postgres guidance is the plain installer re-run that moves the image pin:\n%s", pg)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+350
-108
@@ -2,17 +2,18 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
@@ -28,42 +29,134 @@ const proxyFor = 3 * time.Minute
|
||||
|
||||
// cmdWatchdog runs one pass of the platform watchdog (internal/watchdog): it
|
||||
// checks the cluster, PostgreSQL, the game proxy, the database backups and the
|
||||
// host, prints every finding, and mails the platform owners what came due.
|
||||
// host (disks, memory, its address and clock), prints every finding, and mails
|
||||
// the platform owners what came due.
|
||||
// deploy/bootstrap.sh runs it every two minutes from felis-watchdog.timer.
|
||||
func cmdWatchdog(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("watchdog", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
|
||||
statePath := fs.String("state", "/var/lib/felis/watchdog/state.json", "state kept between runs (root only: it caches the relay password)")
|
||||
quietPath := fs.String("quiet-file", "/run/felis/watchdog-quiet-until", "Unix time before which nothing is mailed; the installer writes it while it restarts things on purpose")
|
||||
backupDir := fs.String("backup-dir", "/var/lib/felis/db-backups", `control-plane database backups to check for freshness ("" skips the check)`)
|
||||
diskPaths := fs.String("disk-paths", "/,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis", "comma-separated paths whose filesystems must keep free space")
|
||||
proxyAddr := fs.String("proxy-addr", "", `game proxy address to dial, e.g. 127.0.0.1:25565 ("" skips the check)`)
|
||||
controlNS := fs.String("control-namespace", platform.DefaultControlNamespace, "namespace of the control plane")
|
||||
offsiteStatus := fs.String("offsite-status", offsite.DefaultStatusFile, "the record `felis offsite sync` leaves, checked when [offsite] is configured")
|
||||
toolsStatus := fs.String("build-tools-status", defaultBuildToolsStatus, "the record `felis mirror-build-tools` leaves, checked when builds scan against the registry's DB copy")
|
||||
dryRun := fs.Bool("dry-run", false, "print every finding and the mail that is due; send nothing and keep the state as it was")
|
||||
var w watchdogFlags
|
||||
w.register(fs)
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
now := time.Now()
|
||||
if w.unitFailed {
|
||||
return watchdogUnitFailed(unitFailedRun{
|
||||
cfgPath: w.cfgPath, statePath: w.statePath, fallbackPath: w.fallbackState, quietPath: w.quietPath,
|
||||
offsiteStatus: w.offsiteStatus, heartbeatFile: w.heartbeatFile,
|
||||
result: os.Getenv("MONITOR_SERVICE_RESULT"), exitStatus: os.Getenv("MONITOR_EXIT_STATUS"),
|
||||
send: watchdogSender, client: http.DefaultClient, now: now,
|
||||
}, stdout, stderr)
|
||||
}
|
||||
cfg, err := config.Load(w.cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
state, err := watchdog.LoadState(*statePath)
|
||||
loadPath := watchdog.NewestState(w.statePath, w.fallbackState)
|
||||
state, aside, err := watchdog.RecoverState(loadPath, now)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
|
||||
defer cancel()
|
||||
now := time.Now()
|
||||
|
||||
var report watchdog.Report
|
||||
if aside != "" {
|
||||
fmt.Fprintf(stderr, "felis watchdog: %s was unreadable; moved it to %s and started over\n", loadPath, aside)
|
||||
report.Findings = append(report.Findings, watchdog.StateSetAside(aside))
|
||||
}
|
||||
cl, owners, ownersErr := watchdogProbes(ctx, w, cfg, now, &report)
|
||||
if cfg.SMTP.Host != "" {
|
||||
refreshSMTPPassword(ctx, w.smtpPasswordFile, cl, w.controlNS, state, stderr)
|
||||
}
|
||||
state.Relay = cachedRelay(cfg.SMTP)
|
||||
if ownersErr == nil {
|
||||
state.Recipients = owners
|
||||
}
|
||||
|
||||
if len(report.Findings) == 0 {
|
||||
fmt.Fprintln(stdout, "felis watchdog: every check passed")
|
||||
}
|
||||
for _, f := range report.Findings {
|
||||
fmt.Fprintf(stdout, "felis watchdog: [%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
|
||||
}
|
||||
|
||||
plan := state.Observe(report, now)
|
||||
host, _ := os.Hostname()
|
||||
subject, body := plan.Message(host, now)
|
||||
beat := heartbeat{
|
||||
standby: standsBy(cfg.Offsite.Enabled(), w.offsiteStatus),
|
||||
quiet: now.Before(watchdog.QuietUntil(w.quietPath)),
|
||||
}
|
||||
if beat.url, err = readHeartbeatURL(w.heartbeatFile); err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: %v; pinging no heartbeat\n", err)
|
||||
}
|
||||
if w.dryRun {
|
||||
if plan.Empty() {
|
||||
fmt.Fprintln(stdout, "felis watchdog: nothing is due to be mailed")
|
||||
} else {
|
||||
fmt.Fprintf(stdout, "felis watchdog: due to be mailed to %s:\nSubject: %s\n\n%s", strings.Join(state.Recipients, ", "), subject, strings.ReplaceAll(body, "\r\n", "\n"))
|
||||
}
|
||||
if beat.url != "" {
|
||||
fmt.Fprintf(stdout, "felis watchdog: a run pings the heartbeat at %s\n", redactURL(beat.url))
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
m := configMailer(cfg.SMTP, state.SMTPPassword, watchdogSender)
|
||||
unheard, mailFailed := m.deliver(ctx, state, plan, subject, body, mailHold(w.quietPath, cfg.Offsite.Enabled(), w.offsiteStatus, now), now, stdout, stderr)
|
||||
saveErr := watchdog.SaveStateOr(w.statePath, w.fallbackState, state)
|
||||
if saveErr != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", saveErr)
|
||||
}
|
||||
beat.report = failureReport(unheard, mailFailed, saveErr, state.Open(), report)
|
||||
beat.fail = beat.report != ""
|
||||
beat.send(http.DefaultClient, stdout, stderr)
|
||||
if mailFailed || saveErr != nil {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// watchdogFlags are felis watchdog's flags. felis doctor reads them back from
|
||||
// the ExecStart= line of felis-watchdog.service, so it checks what the timer's
|
||||
// runs check, with the same paths.
|
||||
type watchdogFlags struct {
|
||||
cfgPath, statePath, fallbackState, smtpPasswordFile, quietPath string
|
||||
backupDir, diskPaths, certDirs, proxyAddr, nodeIP, controlNS string
|
||||
offsiteStatus, toolsStatus, heartbeatFile string
|
||||
dryRun, unitFailed bool
|
||||
}
|
||||
|
||||
func (w *watchdogFlags) register(fs *flag.FlagSet) {
|
||||
fs.StringVar(&w.cfgPath, "config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
|
||||
fs.StringVar(&w.statePath, "state", "/var/lib/felis/watchdog/state.json", "state kept between runs (root only: it caches the relay password)")
|
||||
fs.StringVar(&w.fallbackState, "fallback-state", watchdog.FallbackStatePath, "where a run keeps its state while -state cannot be written, so what it mailed is not mailed again (tmpfs: until the host restarts; \"\" keeps none)")
|
||||
fs.StringVar(&w.smtpPasswordFile, "smtp-password-file", hostSMTPPasswordPath, "the relay password `felis setup` keeps on the host; the felis-smtp Secret stands in while it is missing")
|
||||
fs.StringVar(&w.quietPath, "quiet-file", "/run/felis/watchdog-quiet-until", "Unix time before which nothing is mailed; the installer writes it while it restarts things on purpose")
|
||||
fs.StringVar(&w.backupDir, "backup-dir", "/var/lib/felis/db-backups", `control-plane database backups to check for freshness ("" skips the check)`)
|
||||
fs.StringVar(&w.diskPaths, "disk-paths", "/,/var/lib/rancher/k3s,/var/lib/felis", "comma-separated paths whose filesystems must keep free space")
|
||||
fs.StringVar(&w.certDirs, "k3s-cert-dirs", strings.Join(watchdog.K3sCertDirs, ","), `k3s certificate directories whose *.crt files must not be near expiry ("" skips the check)`)
|
||||
fs.StringVar(&w.proxyAddr, "proxy-addr", "", `game proxy address to dial, e.g. 127.0.0.1:25565 ("" skips the check)`)
|
||||
fs.StringVar(&w.nodeIP, "node-ip", "", `the node address the install was made on, which must stay on this host ("" skips the check)`)
|
||||
fs.StringVar(&w.controlNS, "control-namespace", platform.DefaultControlNamespace, "namespace of the control plane")
|
||||
fs.StringVar(&w.offsiteStatus, "offsite-status", offsite.DefaultStatusFile, "the record `felis offsite sync` leaves, checked when [offsite] is configured")
|
||||
fs.StringVar(&w.toolsStatus, "build-tools-status", defaultBuildToolsStatus, "the record `felis mirror-build-tools` leaves, checked when builds scan against the registry's DB copy")
|
||||
fs.BoolVar(&w.dryRun, "dry-run", false, "print every finding and the mail that is due; send nothing and keep the state as it was")
|
||||
fs.StringVar(&w.heartbeatFile, "heartbeat-file", defaultHeartbeatFile, "file holding the heartbeat URL each run pings, a dead man's switch at a monitoring service that alerts when the pings stop (no file pings nothing)")
|
||||
fs.BoolVar(&w.unitFailed, "unit-failed", false, "report a failed run of felis-watchdog.service instead of checking; felis-watchdog-failed.service runs this through OnFailure=")
|
||||
}
|
||||
|
||||
// watchdogProbes is one pass of every check, appended to report. cl is the
|
||||
// cluster client, nil while the API server is unreachable; owners is who the
|
||||
// alerts go to, and ownersErr why PostgreSQL did not say.
|
||||
func watchdogProbes(ctx context.Context, w watchdogFlags, cfg *config.Config, now time.Time, report *watchdog.Report) (cl client.Client, owners []string, ownersErr error) {
|
||||
add := func(f *watchdog.Finding) {
|
||||
if f != nil {
|
||||
report.Findings = append(report.Findings, *f)
|
||||
@@ -78,90 +171,237 @@ func cmdWatchdog(args []string, stdout, stderr io.Writer) int {
|
||||
cl, err := buildSystemServerClient()
|
||||
var found []watchdog.Finding
|
||||
if err == nil {
|
||||
found, err = watchdog.Cluster{Client: cl, ControlNamespace: *controlNS, MinecraftNamespace: minecraftNS}.Check(ctx, now)
|
||||
found, err = watchdog.Cluster{Client: cl, ControlNamespace: w.controlNS, MinecraftNamespace: minecraftNS}.Check(ctx, now)
|
||||
}
|
||||
if err != nil {
|
||||
cl = nil
|
||||
f := watchdog.KubeAPIDown(err)
|
||||
add(&f)
|
||||
report.Unknown = append(report.Unknown, watchdog.ClusterPrefixes...)
|
||||
} else {
|
||||
report.Findings = append(report.Findings, found...)
|
||||
if cfg.SMTP.Host != "" {
|
||||
refreshSMTPPassword(ctx, cl, *controlNS, state, stderr)
|
||||
}
|
||||
}
|
||||
|
||||
if recipients, err := ownerEmails(ctx, cfg.Database.URL); err != nil {
|
||||
f := watchdog.PostgresDown(err)
|
||||
if owners, ownersErr = ownerEmails(ctx, cfg.Database.URL); ownersErr != nil {
|
||||
f := watchdog.PostgresDown(ownersErr)
|
||||
add(&f)
|
||||
} else {
|
||||
state.Recipients = recipients
|
||||
}
|
||||
|
||||
if *proxyAddr != "" {
|
||||
add(proxyFinding(ctx, *proxyAddr))
|
||||
if w.proxyAddr != "" {
|
||||
add(proxyFinding(ctx, w.proxyAddr))
|
||||
}
|
||||
if *backupDir != "" {
|
||||
add(watchdog.BackupFinding(*backupDir, now))
|
||||
if w.backupDir != "" {
|
||||
add(watchdog.BackupFinding(w.backupDir, now))
|
||||
}
|
||||
if cfg.Offsite.Enabled() {
|
||||
add(watchdog.OffsiteFinding(*offsiteStatus, now))
|
||||
add(watchdog.OffsiteFinding(w.offsiteStatus, now))
|
||||
}
|
||||
if usesMirroredScanDB(cfg) {
|
||||
add(watchdog.ScanDBFinding(*toolsStatus, now))
|
||||
add(watchdog.ScanDBFinding(w.toolsStatus, now))
|
||||
}
|
||||
report.Findings = append(report.Findings, watchdog.DiskFindings(splitList(*diskPaths))...)
|
||||
report.Findings = append(report.Findings, watchdog.DiskFindings(splitList(w.diskPaths))...)
|
||||
add(watchdog.MemoryFinding("/proc/meminfo"))
|
||||
|
||||
if len(report.Findings) == 0 {
|
||||
fmt.Fprintln(stdout, "felis watchdog: every check passed")
|
||||
}
|
||||
for _, f := range report.Findings {
|
||||
fmt.Fprintf(stdout, "felis watchdog: [%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
|
||||
}
|
||||
|
||||
plan := state.Observe(report, now)
|
||||
host, _ := os.Hostname()
|
||||
subject, body := plan.Message(host, now)
|
||||
if *dryRun {
|
||||
if plan.Empty() {
|
||||
fmt.Fprintln(stdout, "felis watchdog: nothing is due to be mailed")
|
||||
} else {
|
||||
fmt.Fprintf(stdout, "felis watchdog: due to be mailed to %s:\nSubject: %s\n\n%s", strings.Join(state.Recipients, ", "), subject, strings.ReplaceAll(body, "\r\n", "\n"))
|
||||
add(watchdog.CertFinding(splitList(w.certDirs), now))
|
||||
if w.nodeIP != "" {
|
||||
if held, err := watchdog.HostAddresses(); err == nil {
|
||||
add(watchdog.AddressFinding(w.nodeIP, held))
|
||||
}
|
||||
return 0
|
||||
}
|
||||
add(watchdog.ClockFinding(watchdog.ClockStatus()))
|
||||
return cl, owners, ownersErr
|
||||
}
|
||||
|
||||
save := func() int {
|
||||
if err := watchdog.SaveState(*statePath, state); err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
// failureReport is what the heartbeat's failure ping carries, "" when the run
|
||||
// pings success: the alerts this run knows of reach no one (a mail that
|
||||
// failed, or no relay or recipient while something is open), or the state did
|
||||
// not save to its file: kept on tmpfs it holds until the host restarts, which
|
||||
// forgets what was mailed, and with nowhere to keep it the next run mails the
|
||||
// same alerts again.
|
||||
func failureReport(unheard string, mailFailed bool, saveErr error, open bool, r watchdog.Report) string {
|
||||
var why []string
|
||||
if unheard != "" && (mailFailed || open) {
|
||||
why = append(why, "the alerts reach no one: "+unheard)
|
||||
}
|
||||
if plan.Empty() {
|
||||
return save()
|
||||
if saveErr != nil {
|
||||
why = append(why, "the watchdog state did not save: "+saveErr.Error())
|
||||
}
|
||||
if until := watchdog.QuietUntil(*quietPath); now.Before(until) {
|
||||
fmt.Fprintf(stdout, "felis watchdog: quiet until %s (installer running); holding this mail: %s\n", until.UTC().Format(time.RFC3339), subject)
|
||||
return save()
|
||||
if len(why) == 0 {
|
||||
return ""
|
||||
}
|
||||
return strings.Join(why, "\n") + "\n\n" + findingLines(r)
|
||||
}
|
||||
|
||||
// findingLines is the report as the journal shows it.
|
||||
func findingLines(r watchdog.Report) string {
|
||||
var b strings.Builder
|
||||
for _, f := range r.Findings {
|
||||
fmt.Fprintf(&b, "[%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// mailer is how a run reaches the owners.
|
||||
type mailer struct {
|
||||
relay *watchdog.Relay // nil: no [smtp] relay
|
||||
password string
|
||||
send func(*mail.SMTP) alertSender
|
||||
}
|
||||
|
||||
// deliver mails plan to the owners unless hold says why it waits, and commits
|
||||
// it once it reached them, or once it is logged because nothing can reach them.
|
||||
// unheard is why the owners hear nothing of this run's alerts, "" when they do;
|
||||
// mailFailed is a mail that did not go out, left uncommitted so the same
|
||||
// alerts come due again next run.
|
||||
func (m mailer) deliver(ctx context.Context, state *watchdog.State, plan watchdog.Plan, subject, body, hold string, now time.Time, stdout, stderr io.Writer) (unheard string, mailFailed bool) {
|
||||
switch {
|
||||
case m.relay == nil:
|
||||
unheard = "no [smtp] relay is configured"
|
||||
case len(state.Recipients) == 0:
|
||||
unheard = "no owner account has a verified email"
|
||||
}
|
||||
switch {
|
||||
case cfg.SMTP.Host == "":
|
||||
fmt.Fprintf(stdout, "felis watchdog: no [smtp] relay configured, so this is logged only: %s\n", subject)
|
||||
case len(state.Recipients) == 0:
|
||||
fmt.Fprintf(stdout, "felis watchdog: no owner account has a verified email, so this is logged only: %s\n", subject)
|
||||
case plan.Empty():
|
||||
case hold != "":
|
||||
fmt.Fprintf(stdout, "felis watchdog: %s; holding this mail: %s\n", hold, subject)
|
||||
case unheard != "":
|
||||
fmt.Fprintf(stdout, "felis watchdog: %s, so this is logged only: %s\n", unheard, subject)
|
||||
state.Commit(plan, now)
|
||||
default:
|
||||
if err := sendAlert(ctx, cfg, state, subject, body); err != nil {
|
||||
// Not committed: the same alerts come due again next run.
|
||||
relay := &mail.SMTP{Host: m.relay.Host, Port: m.relay.Port, From: m.relay.From, Username: m.relay.Username, Password: m.password, RequireTLS: m.relay.RequireTLS}
|
||||
if err := sendAlert(ctx, m.send(relay), state.Recipients, subject, body); err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: mail %q: %v\n", subject, err)
|
||||
save()
|
||||
return 1
|
||||
return "the alert mail failed: " + err.Error(), true
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis watchdog: mailed %s: %s\n", strings.Join(state.Recipients, ", "), subject)
|
||||
state.Commit(plan, now)
|
||||
}
|
||||
state.Commit(plan, now)
|
||||
return save()
|
||||
return unheard, false
|
||||
}
|
||||
|
||||
// configMailer reaches the owners through the relay felis.toml configures.
|
||||
func configMailer(c config.SMTPConfig, cachedPassword string, send func(*mail.SMTP) alertSender) mailer {
|
||||
return mailer{relay: cachedRelay(c), password: smtpPassword(c, cachedPassword), send: send}
|
||||
}
|
||||
|
||||
// cachedRelay is what State.Relay keeps of c, nil when no relay is configured.
|
||||
func cachedRelay(c config.SMTPConfig) *watchdog.Relay {
|
||||
if c.Host == "" {
|
||||
return nil
|
||||
}
|
||||
return &watchdog.Relay{Host: c.Host, Port: c.Port, From: c.From, Username: c.Username, RequireTLS: c.TLSRequired()}
|
||||
}
|
||||
|
||||
// smtpPassword is the relay password a mail signs in with: the env var [smtp]
|
||||
// password_ref names when it is set, else the one the state caches.
|
||||
func smtpPassword(c config.SMTPConfig, cached string) string {
|
||||
if ref := c.PasswordRef; ref != "" && os.Getenv(ref) != "" {
|
||||
return os.Getenv(ref)
|
||||
}
|
||||
return cached
|
||||
}
|
||||
|
||||
// unitFailedRun is one run of felis-watchdog-failed.service.
|
||||
type unitFailedRun struct {
|
||||
cfgPath, statePath, fallbackPath, quietPath, offsiteStatus, heartbeatFile string
|
||||
// result and exitStatus are what systemd hands an OnFailure= unit
|
||||
// (MONITOR_SERVICE_RESULT, MONITOR_EXIT_STATUS; systemd 251 and later).
|
||||
result, exitStatus string
|
||||
send func(*mail.SMTP) alertSender
|
||||
client *http.Client
|
||||
now time.Time
|
||||
}
|
||||
|
||||
// watchdogUnitFailed is felis-watchdog-failed.service, which systemd starts
|
||||
// through OnFailure= when a run of felis-watchdog.service fails: a crash, a
|
||||
// felis.toml that no longer loads, a hang past the unit's timeout, a mail
|
||||
// that did not go out. Such a run checks and mails nothing, so this records
|
||||
// the failure as the alert watchdog/run, due after five failed runs in a row
|
||||
// and cleared by the next run that succeeds; mails it through the relay the
|
||||
// last good run cached when felis.toml does not load; and pings the
|
||||
// heartbeat's failure endpoint. Every other alert keeps its state.
|
||||
func watchdogUnitFailed(r unitFailedRun, stdout, stderr io.Writer) int {
|
||||
detail := failureDetail(r.result, r.exitStatus)
|
||||
cfg, cfgErr := config.Load(r.cfgPath)
|
||||
if cfgErr != nil {
|
||||
detail += "; " + cfgErr.Error()
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis watchdog: felis-watchdog.service failed: %s\n", detail)
|
||||
offsiteOn := cfgErr != nil || cfg.Offsite.Enabled()
|
||||
beat := heartbeat{
|
||||
fail: true,
|
||||
report: "felis-watchdog.service failed: " + detail,
|
||||
standby: standsBy(offsiteOn, r.offsiteStatus),
|
||||
quiet: r.now.Before(watchdog.QuietUntil(r.quietPath)),
|
||||
}
|
||||
var err error
|
||||
if beat.url, err = readHeartbeatURL(r.heartbeatFile); err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: %v; pinging no heartbeat\n", err)
|
||||
}
|
||||
state, err := watchdog.LoadState(watchdog.NewestState(r.statePath, r.fallbackPath))
|
||||
if err != nil {
|
||||
// The next run that gets that far moves a state that does not parse
|
||||
// aside (watchdog.RecoverState).
|
||||
fmt.Fprintf(stderr, "felis watchdog: %v; mailing nothing\n", err)
|
||||
beat.report += "\nThe watchdog state does not load either, so nothing was mailed: " + err.Error()
|
||||
beat.send(r.client, stdout, stderr)
|
||||
return 1
|
||||
}
|
||||
plan := state.Observe(watchdog.Report{Findings: []watchdog.Finding{watchdog.WatchdogFailed(detail)}, Unknown: []string{""}}, r.now)
|
||||
host, _ := os.Hostname()
|
||||
subject, body := plan.Message(host, r.now)
|
||||
m := mailer{relay: state.Relay, password: state.SMTPPassword, send: r.send}
|
||||
if cfgErr == nil {
|
||||
m = configMailer(cfg.SMTP, state.SMTPPassword, r.send)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), time.Minute)
|
||||
defer cancel()
|
||||
unheard, mailFailed := m.deliver(ctx, state, plan, subject, body, mailHold(r.quietPath, offsiteOn, r.offsiteStatus, r.now), r.now, stdout, stderr)
|
||||
code := 0
|
||||
if mailFailed {
|
||||
code = 1
|
||||
}
|
||||
if err := watchdog.SaveStateOr(r.statePath, r.fallbackPath, state); err != nil {
|
||||
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", err)
|
||||
code = 1
|
||||
}
|
||||
if unheard != "" {
|
||||
beat.report += "\nThe alerts reach no one: " + unheard
|
||||
}
|
||||
beat.send(r.client, stdout, stderr)
|
||||
return code
|
||||
}
|
||||
|
||||
// failureDetail names how felis-watchdog.service failed.
|
||||
func failureDetail(result, exitStatus string) string {
|
||||
switch {
|
||||
case result == "":
|
||||
return "systemd named no cause (journalctl -u felis-watchdog -n 50)"
|
||||
case exitStatus == "":
|
||||
return "result " + result
|
||||
}
|
||||
return "result " + result + ", exit status " + exitStatus
|
||||
}
|
||||
|
||||
// mailHold is why this run's mail waits, "" when it goes out: the installer's
|
||||
// quiet window, or this host standing by for another host's off-site bucket
|
||||
// (offsite.Status.StandsBy). A standby host is a rehearsal, or a rebuild not
|
||||
// yet taken over, and the owners its restored database names are that host's,
|
||||
// which mails them itself.
|
||||
func mailHold(quietPath string, offsiteOn bool, offsiteStatus string, now time.Time) string {
|
||||
if until := watchdog.QuietUntil(quietPath); now.Before(until) {
|
||||
return fmt.Sprintf("quiet until %s (installer running)", until.UTC().Format(time.RFC3339))
|
||||
}
|
||||
if !offsiteOn {
|
||||
return ""
|
||||
}
|
||||
st, err := offsite.ReadStatus(offsiteStatus)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
if w := st.StandsBy(now); w != nil {
|
||||
return fmt.Sprintf("this host stands by for %s, which wrote the off-site bucket at %s and mails its owners itself", w, w.At.UTC().Format(time.RFC3339))
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// usesMirroredScanDB reports whether build scans read the vulnerability DB copy
|
||||
@@ -175,24 +415,40 @@ func usesMirroredScanDB(cfg *config.Config) bool {
|
||||
return repo == "" || strings.HasPrefix(repo, cfg.Registry.URL+"/mirror/")
|
||||
}
|
||||
|
||||
// refreshSMTPPassword caches the relay password from the felis-smtp Secret, or
|
||||
// forgets it when the Secret is gone (a relay without AUTH). An env var named by
|
||||
// [smtp] password_ref, when set, wins at send time instead.
|
||||
func refreshSMTPPassword(ctx context.Context, cl client.Client, ns string, state *watchdog.State, stderr io.Writer) {
|
||||
var sec corev1.Secret
|
||||
err := cl.Get(ctx, client.ObjectKey{Namespace: ns, Name: platform.SMTPSecretName}, &sec)
|
||||
// refreshSMTPPassword caches the relay password (relayPassword: the host copy at
|
||||
// path, else the felis-smtp Secret), or forgets it when the Secret is gone (a
|
||||
// relay without AUTH). The host copy is read even while the cluster is down, the
|
||||
// time an alert matters most; without one, a down cluster keeps the cached
|
||||
// password. An env var named by [smtp] password_ref, when set, wins at send time
|
||||
// instead.
|
||||
func refreshSMTPPassword(ctx context.Context, path string, cl client.Client, ns string, state *watchdog.State, stderr io.Writer) {
|
||||
password, err := relayPassword(ctx, path, cl, ns)
|
||||
switch {
|
||||
case apierrors.IsNotFound(err):
|
||||
state.SMTPPassword = ""
|
||||
case errors.Is(err, errClusterUnreachable):
|
||||
return
|
||||
case err != nil:
|
||||
fmt.Fprintf(stderr, "felis watchdog: read %s/%s (keeping the cached relay password): %v\n", ns, platform.SMTPSecretName, err)
|
||||
default:
|
||||
state.SMTPPassword = string(sec.Data[platform.SMTPSecretPasswordKey])
|
||||
fmt.Fprintf(stderr, "felis watchdog: read the relay password (keeping the cached one): %v\n", err)
|
||||
return
|
||||
}
|
||||
state.SMTPPassword = password
|
||||
}
|
||||
|
||||
// ownerEmails pings PostgreSQL and returns the verified addresses of the
|
||||
// enabled owner accounts, the people who can act on an alert.
|
||||
// smtpSecretPassword reads the relay password from the felis-smtp Secret. A missing
|
||||
// Secret is a relay without AUTH and reads as "".
|
||||
func smtpSecretPassword(ctx context.Context, cl client.Client, ns string) (string, error) {
|
||||
var sec corev1.Secret
|
||||
err := cl.Get(ctx, client.ObjectKey{Namespace: ns, Name: platform.SMTPSecretName}, &sec)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return "", nil
|
||||
}
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(sec.Data[platform.SMTPSecretPasswordKey]), nil
|
||||
}
|
||||
|
||||
// ownerEmails pings PostgreSQL and returns the owners an alert goes to
|
||||
// (watchdog.OwnerEmails).
|
||||
func ownerEmails(ctx context.Context, url string) ([]string, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
|
||||
defer cancel()
|
||||
@@ -201,24 +457,7 @@ func ownerEmails(ctx context.Context, url string) ([]string, error) {
|
||||
return nil, err
|
||||
}
|
||||
defer drv.Close()
|
||||
rows, err := drv.DB().QueryContext(ctx,
|
||||
`SELECT email FROM users
|
||||
WHERE role = 'owner' AND email_verified AND COALESCE(email, '') <> ''
|
||||
AND NOT disabled AND deleted_at IS NULL
|
||||
ORDER BY email`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []string
|
||||
for rows.Next() {
|
||||
var email sql.NullString
|
||||
if err := rows.Scan(&email); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, email.String)
|
||||
}
|
||||
return out, rows.Err()
|
||||
return watchdog.OwnerEmails(ctx, drv.DB())
|
||||
}
|
||||
|
||||
// proxyFinding dials the game proxy; players reach every server through it.
|
||||
@@ -237,21 +476,24 @@ func proxyFinding(ctx context.Context, addr string) *watchdog.Finding {
|
||||
}
|
||||
}
|
||||
|
||||
// alertSender mails one alert; smtpSender is the real one.
|
||||
type alertSender func(ctx context.Context, to, subject, body string) error
|
||||
|
||||
func smtpSender(relay *mail.SMTP) alertSender { return relay.SendNotice }
|
||||
|
||||
// watchdogSender is what a run mails through; tests stand a recorder in.
|
||||
var watchdogSender = smtpSender
|
||||
|
||||
// sendAlert mails subject/body to every recipient; it fails only when no
|
||||
// recipient got it.
|
||||
func sendAlert(ctx context.Context, cfg *config.Config, state *watchdog.State, subject, body string) error {
|
||||
password := state.SMTPPassword
|
||||
if ref := cfg.SMTP.PasswordRef; ref != "" && os.Getenv(ref) != "" {
|
||||
password = os.Getenv(ref)
|
||||
}
|
||||
relay := smtpRelay(cfg.SMTP, password)
|
||||
func sendAlert(ctx context.Context, send alertSender, recipients []string, subject, body string) error {
|
||||
var errs []error
|
||||
for _, to := range state.Recipients {
|
||||
if err := relay.SendNotice(ctx, to, subject, body); err != nil {
|
||||
for _, to := range recipients {
|
||||
if err := send(ctx, to, subject, body); err != nil {
|
||||
errs = append(errs, fmt.Errorf("%s: %w", to, err))
|
||||
}
|
||||
}
|
||||
if len(errs) == len(state.Recipients) {
|
||||
if len(errs) == len(recipients) {
|
||||
return errors.Join(errs...)
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -0,0 +1,164 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
)
|
||||
|
||||
// defaultHeartbeatFile holds the heartbeat URL deploy/bootstrap.sh writes from
|
||||
// FELIS_WATCHDOG_HEARTBEAT_URL. It lives in /etc/felis, so a host rebuilt from
|
||||
// a bundle pings the same check once it takes the off-site bucket over.
|
||||
const defaultHeartbeatFile = "/etc/felis/watchdog-heartbeat-url"
|
||||
|
||||
const (
|
||||
// heartbeatTimeout bounds one ping; a monitoring service slower than that
|
||||
// is as good as down for the run.
|
||||
heartbeatTimeout = 10 * time.Second
|
||||
// heartbeatReportMax caps the report a failure ping carries: monitoring
|
||||
// services keep about the first 10 KB of a ping's body.
|
||||
heartbeatReportMax = 10000
|
||||
)
|
||||
|
||||
// heartbeat is the ping one run sends to a dead man's switch at a monitoring
|
||||
// service (Healthchecks.io, and the services that copy its API), which alerts
|
||||
// its own users when the pings stop: the host down, the timer gone, the
|
||||
// watchdog failing before it can mail. No check on the host can report those.
|
||||
// A run whose alerts reach the owners GETs url; one whose alerts reach no one
|
||||
// POSTs report to url/fail, or withholds the ping when url has a query, where
|
||||
// no /fail can be added and the missing ping trips the check instead.
|
||||
type heartbeat struct {
|
||||
url string // "" pings nothing
|
||||
fail bool
|
||||
report string
|
||||
// standby is a host standing by for another host's off-site bucket: a
|
||||
// rehearsal, or a rebuild not taken over. It pings nothing, or it would
|
||||
// keep the check of the host that writes the bucket green after that host
|
||||
// died.
|
||||
standby bool
|
||||
// quiet is the installer's quiet window, which restarts things on
|
||||
// purpose: no failure is pinged.
|
||||
quiet bool
|
||||
}
|
||||
|
||||
// send pings the heartbeat and logs the outcome.
|
||||
func (b heartbeat) send(cl *http.Client, stdout, stderr io.Writer) {
|
||||
switch {
|
||||
case b.url == "":
|
||||
return
|
||||
case b.standby:
|
||||
fmt.Fprintln(stdout, "felis watchdog: this host stands by for the off-site bucket; pinging no heartbeat (the host that writes the bucket pings it)")
|
||||
return
|
||||
case b.fail && b.quiet:
|
||||
fmt.Fprintln(stdout, "felis watchdog: quiet while the installer runs; withholding the failure ping")
|
||||
return
|
||||
}
|
||||
sent, err := b.ping(cl)
|
||||
switch {
|
||||
case err != nil:
|
||||
fmt.Fprintf(stderr, "felis watchdog: heartbeat: %v\n", err)
|
||||
case !sent:
|
||||
fmt.Fprintf(stdout, "felis watchdog: withholding the heartbeat ping (the URL has a query, so it has no /fail endpoint)\n")
|
||||
case b.fail:
|
||||
fmt.Fprintf(stdout, "felis watchdog: pinged the heartbeat's failure endpoint at %s\n", redactURL(b.url))
|
||||
}
|
||||
}
|
||||
|
||||
// ping sends the request; sent is false when a failure withholds it.
|
||||
func (b heartbeat) ping(cl *http.Client) (sent bool, err error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), heartbeatTimeout)
|
||||
defer cancel()
|
||||
method, target, body := http.MethodGet, b.url, io.Reader(nil)
|
||||
if b.fail {
|
||||
if strings.Contains(b.url, "?") {
|
||||
return false, nil
|
||||
}
|
||||
method, target = http.MethodPost, strings.TrimSuffix(b.url, "/")+"/fail"
|
||||
body = strings.NewReader(clipUTF8(b.report, heartbeatReportMax))
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, method, target, body)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("%s %s: bad URL", method, redactURL(target))
|
||||
}
|
||||
resp, err := cl.Do(req)
|
||||
if err != nil {
|
||||
// A url.Error quotes the whole URL, the check's key among it.
|
||||
var ue *url.Error
|
||||
if errors.As(err, &ue) {
|
||||
err = ue.Err
|
||||
}
|
||||
return false, fmt.Errorf("%s %s: %w", method, redactURL(target), err)
|
||||
}
|
||||
io.Copy(io.Discard, io.LimitReader(resp.Body, 4096))
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode/100 != 2 {
|
||||
return false, fmt.Errorf("%s %s: %s", method, redactURL(target), resp.Status)
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// clipUTF8 cuts s to at most n bytes without splitting a character.
|
||||
func clipUTF8(s string, n int) string {
|
||||
if len(s) <= n {
|
||||
return s
|
||||
}
|
||||
return strings.ToValidUTF8(s[:n], "")
|
||||
}
|
||||
|
||||
// readHeartbeatURL reads the heartbeat URL from path; no file is no heartbeat.
|
||||
func readHeartbeatURL(path string) (string, error) {
|
||||
if path == "" {
|
||||
return "", nil
|
||||
}
|
||||
raw, err := os.ReadFile(path)
|
||||
if errors.Is(err, os.ErrNotExist) {
|
||||
return "", nil
|
||||
}
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
s := strings.TrimSpace(string(raw))
|
||||
if err := checkHeartbeatURL(s); err != nil {
|
||||
return "", fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
return s, nil
|
||||
}
|
||||
|
||||
// checkHeartbeatURL accepts an http(s) URL with a host. Its error leaves the
|
||||
// URL out: the path is the check's key, which anyone who reads it can ping in
|
||||
// the host's name.
|
||||
func checkHeartbeatURL(s string) error {
|
||||
u, err := url.Parse(s)
|
||||
if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" || strings.ContainsAny(s, " \t\r\n") {
|
||||
return errors.New("the heartbeat URL is not an http:// or https:// URL")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// redactURL is a heartbeat URL as logs show it: its scheme and host.
|
||||
func redactURL(s string) string {
|
||||
u, err := url.Parse(s)
|
||||
if err != nil || u.Host == "" {
|
||||
return "(the heartbeat URL)"
|
||||
}
|
||||
return u.Scheme + "://" + u.Host + "/..."
|
||||
}
|
||||
|
||||
// standsBy reports whether this host stands by for another host's off-site
|
||||
// bucket (offsite.Status.Standby), however old that record is: the
|
||||
// installer's take-over or the host's first write ends it.
|
||||
func standsBy(offsiteOn bool, statusPath string) bool {
|
||||
if !offsiteOn {
|
||||
return false
|
||||
}
|
||||
st, err := offsite.ReadStatus(statusPath)
|
||||
return err == nil && st != nil && st.Standby
|
||||
}
|
||||
@@ -0,0 +1,615 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
)
|
||||
|
||||
// pingLog is a monitoring service that records every ping it gets.
|
||||
type pingLog struct {
|
||||
mu sync.Mutex
|
||||
pings []string // "METHOD path?query"
|
||||
bodies []string
|
||||
status int
|
||||
}
|
||||
|
||||
func newPingServer(t *testing.T) (*pingLog, *httptest.Server) {
|
||||
t.Helper()
|
||||
l := &pingLog{status: http.StatusOK}
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
body, _ := io.ReadAll(r.Body)
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
p := r.Method + " " + r.URL.Path
|
||||
if r.URL.RawQuery != "" {
|
||||
p += "?" + r.URL.RawQuery
|
||||
}
|
||||
l.pings = append(l.pings, p)
|
||||
l.bodies = append(l.bodies, string(body))
|
||||
w.WriteHeader(l.status)
|
||||
}))
|
||||
t.Cleanup(srv.Close)
|
||||
return l, srv
|
||||
}
|
||||
|
||||
func (l *pingLog) got() ([]string, []string) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return append([]string(nil), l.pings...), append([]string(nil), l.bodies...)
|
||||
}
|
||||
|
||||
// TestHeartbeatSend: a run whose alerts reach the owners GETs the URL, one
|
||||
// whose alerts reach no one POSTs its report to /fail, and a standby host, a
|
||||
// failure in the installer's quiet window and a query URL that has no /fail
|
||||
// send nothing.
|
||||
func TestHeartbeatSend(t *testing.T) {
|
||||
const key = "/ping/5f1e0c2a-check-key"
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
beat heartbeat
|
||||
path string // appended to the server URL
|
||||
want string // the ping, "" for none
|
||||
wantBody string
|
||||
wantOut string
|
||||
}{
|
||||
{what: "success", beat: heartbeat{}, path: key, want: "GET " + key},
|
||||
{what: "failure", beat: heartbeat{fail: true, report: "the alerts reach no one: x"}, path: key, want: "POST " + key + "/fail", wantBody: "the alerts reach no one: x", wantOut: "pinged the heartbeat's failure endpoint"},
|
||||
{what: "failure, URL with a trailing slash", beat: heartbeat{fail: true, report: "r"}, path: key + "/", want: "POST " + key + "/fail", wantBody: "r"},
|
||||
{what: "success, URL with a query", beat: heartbeat{}, path: key + "?rid=7", want: "GET " + key + "?rid=7"},
|
||||
{what: "failure, URL with a query", beat: heartbeat{fail: true, report: "r"}, path: key + "?rid=7", wantOut: "withholding the heartbeat ping"},
|
||||
{what: "standby, success", beat: heartbeat{standby: true}, path: key, wantOut: "stands by for the off-site bucket"},
|
||||
{what: "standby, failure", beat: heartbeat{standby: true, fail: true}, path: key, wantOut: "stands by for the off-site bucket"},
|
||||
{what: "quiet, failure", beat: heartbeat{quiet: true, fail: true}, path: key, wantOut: "withholding the failure ping"},
|
||||
{what: "quiet, success", beat: heartbeat{quiet: true}, path: key, want: "GET " + key},
|
||||
} {
|
||||
log, srv := newPingServer(t)
|
||||
b := tc.beat
|
||||
b.url = srv.URL + tc.path
|
||||
var stdout, stderr bytes.Buffer
|
||||
b.send(srv.Client(), &stdout, &stderr)
|
||||
pings, bodies := log.got()
|
||||
switch {
|
||||
case tc.want == "" && len(pings) != 0:
|
||||
t.Errorf("%s: pinged %v, want nothing", tc.what, pings)
|
||||
case tc.want != "" && (len(pings) != 1 || pings[0] != tc.want):
|
||||
t.Errorf("%s: pinged %v, want %q", tc.what, pings, tc.want)
|
||||
case tc.want != "" && bodies[0] != tc.wantBody:
|
||||
t.Errorf("%s: body %q, want %q", tc.what, bodies[0], tc.wantBody)
|
||||
}
|
||||
if !strings.Contains(stdout.String(), tc.wantOut) || stderr.Len() != 0 {
|
||||
t.Errorf("%s: stdout %q (want %q), stderr %q", tc.what, stdout.String(), tc.wantOut, stderr.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestHeartbeatErrors: a ping the service refuses or that cannot connect is
|
||||
// logged without the URL's path, the check's key.
|
||||
func TestHeartbeatErrors(t *testing.T) {
|
||||
const key = "5f1e0c2a-check-key"
|
||||
log, srv := newPingServer(t)
|
||||
log.status = http.StatusNotFound
|
||||
var stdout, stderr bytes.Buffer
|
||||
heartbeat{url: srv.URL + "/" + key}.send(srv.Client(), &stdout, &stderr)
|
||||
if got := stderr.String(); !strings.Contains(got, "heartbeat: GET http://127.0.0.1") || !strings.Contains(got, "404") || strings.Contains(got, key) {
|
||||
t.Errorf("refused ping: stderr %q; want the host and status, not the key", got)
|
||||
}
|
||||
|
||||
closed := httptest.NewServer(http.NotFoundHandler())
|
||||
dead := closed.URL + "/" + key
|
||||
closed.Close()
|
||||
stderr.Reset()
|
||||
heartbeat{url: dead, fail: true, report: "r"}.send(http.DefaultClient, &stdout, &stderr)
|
||||
if got := stderr.String(); !strings.Contains(got, "heartbeat: POST http://127.0.0.1") || strings.Contains(got, key) {
|
||||
t.Errorf("unreachable service: stderr %q; want the host, not the key", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestHeartbeatReportClipped: a long report is cut to what the service keeps,
|
||||
// on a character boundary.
|
||||
func TestHeartbeatReportClipped(t *testing.T) {
|
||||
log, srv := newPingServer(t)
|
||||
report := strings.Repeat("磁盘", heartbeatReportMax)
|
||||
heartbeat{url: srv.URL + "/k", fail: true, report: report}.send(srv.Client(), io.Discard, io.Discard)
|
||||
_, bodies := log.got()
|
||||
if len(bodies) != 1 || len(bodies[0]) > heartbeatReportMax || len(bodies[0]) < heartbeatReportMax-3 || !utf8.ValidString(bodies[0]) {
|
||||
t.Fatalf("clipped body: %d pings, %d bytes, valid UTF-8 %v", len(bodies), len(bodies[0]), utf8.ValidString(bodies[0]))
|
||||
}
|
||||
if got := clipUTF8("short", heartbeatReportMax); got != "short" {
|
||||
t.Errorf("clipUTF8(short) = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadHeartbeatURL(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "watchdog-heartbeat-url")
|
||||
if u, err := readHeartbeatURL(path); u != "" || err != nil {
|
||||
t.Fatalf("no file: %q, %v", u, err)
|
||||
}
|
||||
if u, err := readHeartbeatURL(""); u != "" || err != nil {
|
||||
t.Fatalf("no path: %q, %v", u, err)
|
||||
}
|
||||
writeTestFile(t, path, "https://hc-ping.com/5f1e0c2a\n", 0o600)
|
||||
if u, err := readHeartbeatURL(path); u != "https://hc-ping.com/5f1e0c2a" || err != nil {
|
||||
t.Fatalf("good file: %q, %v", u, err)
|
||||
}
|
||||
for _, bad := range []string{"", "hc-ping.com/5f1e0c2a", "ftp://hc-ping.com/5f1e0c2a", "https:///5f1e0c2a", "https://hc-ping.com/5f1e 0c2a"} {
|
||||
writeTestFile(t, path, bad, 0o600)
|
||||
u, err := readHeartbeatURL(path)
|
||||
if u != "" || err == nil || !strings.Contains(err.Error(), path) || (bad != "" && strings.Contains(err.Error(), "5f1e")) {
|
||||
t.Errorf("%q: %q, %v; want an error naming the file and not the URL", bad, u, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRedactURL(t *testing.T) {
|
||||
for in, want := range map[string]string{
|
||||
"https://hc-ping.com/5f1e0c2a": "https://hc-ping.com/...",
|
||||
"http://user:[email protected]:8080/k?x=1": "http://status.example:8080/...",
|
||||
"not a url": "(the heartbeat URL)",
|
||||
} {
|
||||
if got := redactURL(in); got != want {
|
||||
t.Errorf("redactURL(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestStandsBy: a standby record keeps the host from pinging however old it
|
||||
// is; a displaced host, the writer, or a host without [offsite] pings.
|
||||
func TestStandsBy(t *testing.T) {
|
||||
status := filepath.Join(t.TempDir(), "status.json")
|
||||
if standsBy(true, status) {
|
||||
t.Fatal("no status file stands by")
|
||||
}
|
||||
old := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: time.Now().Add(-30 * 24 * time.Hour)}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
st offsite.Status
|
||||
offsiteOn bool
|
||||
want bool
|
||||
}{
|
||||
{"standing by for a writer gone quiet a month ago", offsite.Status{Standby: true, Writer: old}, true, true},
|
||||
{"standing by, [offsite] since removed", offsite.Status{Standby: true, Writer: old}, false, false},
|
||||
{"displaced", offsite.Status{Displaced: true, Writer: old}, true, false},
|
||||
{"the writer itself", offsite.Status{LastSuccess: time.Now()}, true, false},
|
||||
} {
|
||||
if err := offsite.WriteStatus(status, tc.st); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := standsBy(tc.offsiteOn, status); got != tc.want {
|
||||
t.Errorf("%s: standsBy = %v, want %v", tc.what, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFailureReport(t *testing.T) {
|
||||
r := watchdog.Report{Findings: []watchdog.Finding{{Key: "postgres", Severity: watchdog.Critical, SummaryEN: "PostgreSQL is down"}}}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
unheard string
|
||||
mailFailed bool
|
||||
saveErr error
|
||||
open bool
|
||||
want string // "" for a success ping
|
||||
}{
|
||||
{what: "all good", open: true},
|
||||
{what: "no relay, nothing mailed yet", unheard: "no [smtp] relay is configured"},
|
||||
{what: "no relay, an alert open", unheard: "no [smtp] relay is configured", open: true, want: "the alerts reach no one: no [smtp] relay is configured"},
|
||||
{what: "the mail failed", unheard: "the alert mail failed: refused", mailFailed: true, want: "the alerts reach no one: the alert mail failed: refused"},
|
||||
{what: "the state did not save", saveErr: errors.New("disk full"), want: "the watchdog state did not save: disk full"},
|
||||
} {
|
||||
got := failureReport(tc.unheard, tc.mailFailed, tc.saveErr, tc.open, r)
|
||||
if (got == "") != (tc.want == "") || !strings.Contains(got, tc.want) || (got != "" && !strings.Contains(got, "[critical] postgres: PostgreSQL is down")) {
|
||||
t.Errorf("%s: report %q, want %q and the findings", tc.what, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// alertRecorder records the mail a run hands the relay.
|
||||
type alertRecorder struct {
|
||||
relays []*mail.SMTP
|
||||
sent []string // "to: subject"
|
||||
fail map[string]bool
|
||||
}
|
||||
|
||||
func (f *alertRecorder) sender(relay *mail.SMTP) alertSender {
|
||||
f.relays = append(f.relays, relay)
|
||||
return func(_ context.Context, to, subject, _ string) error {
|
||||
if f.fail[to] {
|
||||
return errors.New("550 refused")
|
||||
}
|
||||
f.sent = append(f.sent, to+": "+subject)
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// TestMailerDeliver: a plan goes out and is committed; held, it waits
|
||||
// uncommitted; with no relay or recipient it is logged and committed; a mail
|
||||
// no recipient got is left uncommitted to come due again.
|
||||
func TestMailerDeliver(t *testing.T) {
|
||||
now := time.Now()
|
||||
relay := &watchdog.Relay{Host: "smtp.example.com", Port: 587, From: "[email protected]", Username: "felis"}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
relay *watchdog.Relay
|
||||
recipients []string
|
||||
hold string
|
||||
fail map[string]bool
|
||||
wantUnheard string
|
||||
wantFailed bool
|
||||
committed bool
|
||||
sent int
|
||||
}{
|
||||
{what: "mailed", relay: relay, recipients: []string{"[email protected]", "[email protected]"}, committed: true, sent: 2},
|
||||
{what: "one recipient refused", relay: relay, recipients: []string{"[email protected]", "[email protected]"}, fail: map[string]bool{"[email protected]": true}, committed: true, sent: 1},
|
||||
{what: "every recipient refused", relay: relay, recipients: []string{"[email protected]"}, fail: map[string]bool{"[email protected]": true}, wantUnheard: "the alert mail failed", wantFailed: true},
|
||||
{what: "held", relay: relay, recipients: []string{"[email protected]"}, hold: "quiet until later"},
|
||||
{what: "no relay", recipients: []string{"[email protected]"}, wantUnheard: "no [smtp] relay is configured", committed: true},
|
||||
{what: "no recipient", relay: relay, wantUnheard: "no owner account has a verified email", committed: true},
|
||||
} {
|
||||
state := &watchdog.State{Recipients: tc.recipients}
|
||||
plan := state.Observe(watchdog.Report{Findings: []watchdog.Finding{{Key: "postgres", Severity: watchdog.Critical, SummaryEN: "down"}}}, now)
|
||||
f := &alertRecorder{fail: tc.fail}
|
||||
m := mailer{relay: tc.relay, password: "relay-pw", send: f.sender}
|
||||
var stdout, stderr bytes.Buffer
|
||||
unheard, failed := m.deliver(context.Background(), state, plan, "subject", "body", tc.hold, now, &stdout, &stderr)
|
||||
if !strings.HasPrefix(unheard, tc.wantUnheard) || (tc.wantUnheard == "") != (unheard == "") || failed != tc.wantFailed {
|
||||
t.Errorf("%s: unheard %q, failed %v; want %q, %v", tc.what, unheard, failed, tc.wantUnheard, tc.wantFailed)
|
||||
}
|
||||
if got := !state.Alerts["postgres"].Notified.IsZero(); got != tc.committed {
|
||||
t.Errorf("%s: committed %v, want %v", tc.what, got, tc.committed)
|
||||
}
|
||||
if len(f.sent) != tc.sent {
|
||||
t.Errorf("%s: sent %v, want %d mails", tc.what, f.sent, tc.sent)
|
||||
}
|
||||
if len(f.relays) > 0 {
|
||||
if got := *f.relays[0]; got != (mail.SMTP{Host: "smtp.example.com", Port: 587, From: "[email protected]", Username: "felis", Password: "relay-pw"}) {
|
||||
t.Errorf("%s: relay %+v", tc.what, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestCachedRelay: the cache keeps the relay's coordinates and its TLS rule,
|
||||
// and nothing when no relay is configured.
|
||||
func TestCachedRelay(t *testing.T) {
|
||||
if r := cachedRelay(testSMTPConfig("")); r != nil {
|
||||
t.Fatalf("no relay cached %+v", r)
|
||||
}
|
||||
got := cachedRelay(testSMTPConfig("smtp.example.com"))
|
||||
if got == nil || *got != (watchdog.Relay{Host: "smtp.example.com", Port: 2525, From: "[email protected]", Username: "felis", RequireTLS: true}) {
|
||||
t.Fatalf("cachedRelay = %+v", got)
|
||||
}
|
||||
if got := cachedRelay(testSMTPConfig("127.0.0.1")); got == nil || got.RequireTLS {
|
||||
t.Fatalf("a relay on this host: %+v, want TLS not required", got)
|
||||
}
|
||||
}
|
||||
|
||||
func testSMTPConfig(host string) config.SMTPConfig {
|
||||
if host == "" {
|
||||
return config.SMTPConfig{}
|
||||
}
|
||||
return config.SMTPConfig{Host: host, Port: 2525, From: "[email protected]", Username: "felis"}
|
||||
}
|
||||
|
||||
// TestSMTPPassword: the env var password_ref names wins over the cache once
|
||||
// it is set.
|
||||
func TestSMTPPassword(t *testing.T) {
|
||||
c := testSMTPConfig("smtp.example.com")
|
||||
c.PasswordRef = "FELIS_TEST_WATCHDOG_RELAY_PW"
|
||||
t.Setenv(c.PasswordRef, "")
|
||||
if got := smtpPassword(c, "cached"); got != "cached" {
|
||||
t.Errorf("env var unset: %q", got)
|
||||
}
|
||||
t.Setenv(c.PasswordRef, "from-env")
|
||||
if got := smtpPassword(c, "cached"); got != "from-env" {
|
||||
t.Errorf("env var set: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
const testWatchdogConfig = `[database]
|
||||
url = "postgres://felis:[email protected]:1/felis?sslmode=disable&connect_timeout=2"
|
||||
[server]
|
||||
root_domain = "example.com"
|
||||
[archive]
|
||||
store = "tarLocal"
|
||||
[k8s]
|
||||
egress_mode = "nodeport"
|
||||
`
|
||||
|
||||
const testWatchdogSMTP = `[smtp]
|
||||
host = "smtp.config.example"
|
||||
port = 2525
|
||||
from = "[email protected]"
|
||||
username = "felis"
|
||||
password_ref = "FELIS_TEST_WATCHDOG_RELAY_PW"
|
||||
`
|
||||
|
||||
// unitFailedFixture is a host whose watchdog last ran well: one alert open,
|
||||
// one owner and the relay cached.
|
||||
func unitFailedFixture(t *testing.T, cfg string) (unitFailedRun, *alertRecorder, *pingLog) {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
log, srv := newPingServer(t)
|
||||
beatFile := filepath.Join(dir, "watchdog-heartbeat-url")
|
||||
writeTestFile(t, beatFile, srv.URL+"/check-key\n", 0o600)
|
||||
cfgPath := filepath.Join(dir, "felis.toml")
|
||||
writeTestFile(t, cfgPath, cfg, 0o600)
|
||||
statePath := filepath.Join(dir, "state.json")
|
||||
state := &watchdog.State{
|
||||
Recipients: []string{"[email protected]"},
|
||||
SMTPPassword: "cached-pw",
|
||||
Relay: &watchdog.Relay{Host: "smtp.cached.example", Port: 587, From: "[email protected]", RequireTLS: true},
|
||||
}
|
||||
mem := watchdog.Finding{Key: "memory", Severity: watchdog.Warning, SummaryEN: "memory low"}
|
||||
state.Commit(state.Observe(watchdog.Report{Findings: []watchdog.Finding{mem}}, time.Now().Add(-time.Hour)), time.Now().Add(-time.Hour))
|
||||
if err := watchdog.SaveState(statePath, state); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f := &alertRecorder{}
|
||||
return unitFailedRun{
|
||||
cfgPath: cfgPath, statePath: statePath, fallbackPath: filepath.Join(t.TempDir(), "watchdog-state.json"), quietPath: filepath.Join(dir, "quiet"),
|
||||
offsiteStatus: filepath.Join(dir, "offsite-status.json"), heartbeatFile: beatFile,
|
||||
result: "exit-code", exitStatus: "1", send: f.sender, client: srv.Client(), now: time.Now(),
|
||||
}, f, log
|
||||
}
|
||||
|
||||
// TestWatchdogUnitFailedBrokenConfig: failed runs of a watchdog whose
|
||||
// felis.toml no longer loads are mailed through the relay the last good run
|
||||
// cached, after five in a row; every one pings /fail; the open alert keeps
|
||||
// its state.
|
||||
func TestWatchdogUnitFailedBrokenConfig(t *testing.T) {
|
||||
r, f, log := unitFailedFixture(t, "[database\n")
|
||||
start := r.now
|
||||
for i := 0; i <= 5; i++ {
|
||||
r.now = start.Add(time.Duration(i) * 2 * time.Minute)
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := watchdogUnitFailed(r, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("run %d: exit %d (stdout %s, stderr %s)", i, code, stdout.String(), stderr.String())
|
||||
}
|
||||
if i < 5 && len(f.sent) != 0 {
|
||||
t.Fatalf("run %d, %v after the first failure: mailed %v, want nothing yet", i, r.now.Sub(start), f.sent)
|
||||
}
|
||||
}
|
||||
if len(f.sent) != 1 || !strings.Contains(f.sent[0], "[email protected]: ") {
|
||||
t.Fatalf("mailed %v, want one alert to the cached owner", f.sent)
|
||||
}
|
||||
if got := *f.relays[0]; got.Host != "smtp.cached.example" || got.Password != "cached-pw" || !got.RequireTLS {
|
||||
t.Fatalf("relay %+v, want the cached one", got)
|
||||
}
|
||||
pings, bodies := log.got()
|
||||
if len(pings) != 6 || pings[0] != "POST /check-key/fail" || !strings.Contains(bodies[0], "felis-watchdog.service failed: result exit-code, exit status 1; ") {
|
||||
t.Fatalf("pings %v, bodies %q", pings, bodies)
|
||||
}
|
||||
state, err := watchdog.LoadState(r.statePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if a := state.Alerts["watchdog/run"]; a == nil || a.Notified.IsZero() || !strings.Contains(a.SummaryEN, "result exit-code, exit status 1") {
|
||||
t.Fatalf("watchdog/run alert = %+v", a)
|
||||
}
|
||||
if a := state.Alerts["memory"]; a == nil || !a.ClearedAt.IsZero() || !a.Notified.Before(start) {
|
||||
t.Fatalf("memory alert = %+v, want it untouched", a)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogUnitFailedStateThatDoesNotSave: with the state file read-only,
|
||||
// each report keeps the state in the fallback and the next one reads it, so
|
||||
// the failure is mailed once, after five in a row, and never again.
|
||||
func TestWatchdogUnitFailedStateThatDoesNotSave(t *testing.T) {
|
||||
if os.Geteuid() == 0 {
|
||||
t.Skip("root writes into a read-only directory")
|
||||
}
|
||||
r, f, _ := unitFailedFixture(t, "[database\n")
|
||||
dir := filepath.Dir(r.statePath)
|
||||
if err := os.Chmod(dir, 0o500); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { os.Chmod(dir, 0o700) })
|
||||
start := r.now
|
||||
for i := 0; i <= 8; i++ {
|
||||
r.now = start.Add(time.Duration(i) * 2 * time.Minute)
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := watchdogUnitFailed(r, &stdout, &stderr); code != 1 || !strings.Contains(stderr.String(), "; kept in "+r.fallbackPath) {
|
||||
t.Fatalf("run %d: exit %d, stderr %s; want exit 1 and the state kept in the fallback", i, code, stderr.String())
|
||||
}
|
||||
if want := min(max(i-4, 0), 1); len(f.sent) != want {
|
||||
t.Fatalf("run %d, %v after the first failure: mailed %d, want %d", i, r.now.Sub(start), len(f.sent), want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogUnitFailedConfigRelay: with felis.toml loading, the relay is
|
||||
// the configured one and signs in with the env var password_ref names.
|
||||
func TestWatchdogUnitFailedConfigRelay(t *testing.T) {
|
||||
r, f, _ := unitFailedFixture(t, testWatchdogConfig+testWatchdogSMTP)
|
||||
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
|
||||
start := r.now
|
||||
for _, at := range []time.Duration{0, 10 * time.Minute} {
|
||||
r.now = start.Add(at)
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := watchdogUnitFailed(r, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit %d (stdout %s, stderr %s)", code, stdout.String(), stderr.String())
|
||||
}
|
||||
}
|
||||
if len(f.relays) != 1 || f.relays[0].Host != "smtp.config.example" || f.relays[0].Port != 2525 || f.relays[0].Password != "env-pw" {
|
||||
t.Fatalf("relays %+v, want the configured one with the env password", f.relays)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogUnitFailedHeld: in the installer's quiet window the alert waits
|
||||
// and no failure is pinged; a host standing by for the off-site bucket pings
|
||||
// nothing either.
|
||||
func TestWatchdogUnitFailedHeld(t *testing.T) {
|
||||
r, f, log := unitFailedFixture(t, "[database\n")
|
||||
writeTestFile(t, r.quietPath, fmt.Sprintf("%d\n", r.now.Add(time.Hour).Unix()), 0o644)
|
||||
start := r.now
|
||||
for _, at := range []time.Duration{0, 10 * time.Minute} {
|
||||
r.now = start.Add(at)
|
||||
watchdogUnitFailed(r, io.Discard, io.Discard)
|
||||
}
|
||||
if pings, _ := log.got(); len(f.sent) != 0 || len(pings) != 0 {
|
||||
t.Fatalf("quiet window: mailed %v, pinged %v", f.sent, pings)
|
||||
}
|
||||
|
||||
os.Remove(r.quietPath)
|
||||
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: r.now.Add(-time.Hour)}
|
||||
if err := offsite.WriteStatus(r.offsiteStatus, offsite.Status{Standby: true, Writer: w}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var stdout bytes.Buffer
|
||||
watchdogUnitFailed(r, &stdout, io.Discard)
|
||||
if pings, _ := log.got(); len(f.sent) != 0 || len(pings) != 0 || !strings.Contains(stdout.String(), "stands by for host prod-1") {
|
||||
t.Fatalf("standby: mailed %v, pinged %v, stdout %s", f.sent, pings, stdout.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogUnitFailedStateUnreadable: with no state to mail from, the
|
||||
// failure still reaches the heartbeat.
|
||||
func TestWatchdogUnitFailedStateUnreadable(t *testing.T) {
|
||||
r, f, log := unitFailedFixture(t, "[database\n")
|
||||
writeTestFile(t, r.statePath, "{", 0o600)
|
||||
if code := watchdogUnitFailed(r, io.Discard, io.Discard); code != 1 {
|
||||
t.Fatalf("exit %d, want 1", code)
|
||||
}
|
||||
pings, bodies := log.got()
|
||||
if len(f.sent) != 0 || len(pings) != 1 || pings[0] != "POST /check-key/fail" || !strings.Contains(bodies[0], "state does not load") {
|
||||
t.Fatalf("mailed %v, pinged %v %q", f.sent, pings, bodies)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFailureDetail(t *testing.T) {
|
||||
for _, tc := range [][3]string{
|
||||
{"exit-code", "1", "result exit-code, exit status 1"},
|
||||
{"timeout", "", "result timeout"},
|
||||
{"", "", "systemd named no cause (journalctl -u felis-watchdog -n 50)"},
|
||||
} {
|
||||
if got := failureDetail(tc[0], tc[1]); got != tc[2] {
|
||||
t.Errorf("failureDetail(%q, %q) = %q, want %q", tc[0], tc[1], got, tc[2])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogRunHeartbeat runs whole watchdog passes with the API server and
|
||||
// PostgreSQL down and the relay a recorder: a pass whose alerts reach the
|
||||
// owners pings success; one whose mail fails, or that has no relay while an
|
||||
// alert is open, posts its report to /fail; a host standing by for the
|
||||
// off-site bucket pings nothing; a state file that does not parse is moved
|
||||
// aside and reported.
|
||||
func TestWatchdogRunHeartbeat(t *testing.T) {
|
||||
due := func(statePath string) {
|
||||
// PostgreSQL has been down for an hour and nobody was told yet.
|
||||
s := &watchdog.State{Recipients: []string{"[email protected]"}, SMTPPassword: "cached-pw"}
|
||||
s.Observe(watchdog.Report{Findings: []watchdog.Finding{watchdog.PostgresDown(errors.New("refused"))}}, time.Now().Add(-time.Hour))
|
||||
if err := watchdog.SaveState(statePath, s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
open := func(statePath string) {
|
||||
// The owners were told of PostgreSQL half an hour ago.
|
||||
s := &watchdog.State{Recipients: []string{"[email protected]"}}
|
||||
at := time.Now().Add(-30 * time.Minute)
|
||||
r := watchdog.Report{Findings: []watchdog.Finding{watchdog.PostgresDown(errors.New("refused"))}}
|
||||
s.Observe(r, at.Add(-time.Hour))
|
||||
s.Commit(s.Observe(r, at), at)
|
||||
if err := watchdog.SaveState(statePath, s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
const offsiteTable = "[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis\"\n"
|
||||
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
cfg string
|
||||
state func(path string)
|
||||
standby bool
|
||||
quiet bool
|
||||
refused bool
|
||||
wantCode int
|
||||
want string // the ping, "" for none
|
||||
wantBody string
|
||||
wantOut string
|
||||
wantSent int
|
||||
}{
|
||||
{what: "nothing due", cfg: testWatchdogConfig + testWatchdogSMTP, want: "GET /check-key"},
|
||||
{what: "mailed", cfg: testWatchdogConfig + testWatchdogSMTP, state: due, want: "GET /check-key", wantSent: 1},
|
||||
{what: "the mail fails", cfg: testWatchdogConfig + testWatchdogSMTP, state: due, refused: true, wantCode: 1, want: "POST /check-key/fail", wantBody: "the alerts reach no one: the alert mail failed: [email protected]: 550 refused"},
|
||||
{what: "no relay, an alert open", cfg: testWatchdogConfig, state: due, want: "POST /check-key/fail", wantBody: "the alerts reach no one: no [smtp] relay is configured"},
|
||||
{what: "no relay, an alert open, the installer running", cfg: testWatchdogConfig, state: open, quiet: true, wantOut: "withholding the failure ping"},
|
||||
{what: "standing by for the off-site bucket", cfg: testWatchdogConfig + testWatchdogSMTP + offsiteTable, state: due, standby: true, wantOut: "stands by for the off-site bucket"},
|
||||
{what: "a state that does not parse", cfg: testWatchdogConfig, state: func(p string) { writeTestFile(t, p, "{", 0o600) }, want: "GET /check-key", wantOut: "[warning] watchdog/state: the watchdog's state file was unreadable and was moved to "},
|
||||
} {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("KUBECONFIG", filepath.Join(dir, "no-kubeconfig"))
|
||||
log, srv := newPingServer(t)
|
||||
beatFile := filepath.Join(dir, "watchdog-heartbeat-url")
|
||||
writeTestFile(t, beatFile, srv.URL+"/check-key\n", 0o600)
|
||||
cfgPath := filepath.Join(dir, "felis.toml")
|
||||
writeTestFile(t, cfgPath, tc.cfg, 0o600)
|
||||
statePath := filepath.Join(dir, "state.json")
|
||||
if tc.state != nil {
|
||||
tc.state(statePath)
|
||||
}
|
||||
statusPath := filepath.Join(dir, "offsite-status.json")
|
||||
if tc.standby {
|
||||
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: time.Now().Add(-time.Hour)}
|
||||
if err := offsite.WriteStatus(statusPath, offsite.Status{Standby: true, Writer: w}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if tc.quiet {
|
||||
writeTestFile(t, filepath.Join(dir, "quiet"), fmt.Sprintf("%d\n", time.Now().Add(time.Hour).Unix()), 0o644)
|
||||
}
|
||||
rec := &alertRecorder{}
|
||||
if tc.refused {
|
||||
rec.fail = map[string]bool{"[email protected]": true}
|
||||
}
|
||||
watchdogSender = rec.sender
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := cmdWatchdog([]string{
|
||||
"-config", cfgPath, "-state", statePath, "-quiet-file", filepath.Join(dir, "quiet"),
|
||||
"-backup-dir", "", "-disk-paths", dir, "-smtp-password-file", filepath.Join(dir, "smtp-password"),
|
||||
"-offsite-status", statusPath, "-heartbeat-file", beatFile,
|
||||
}, &stdout, &stderr)
|
||||
watchdogSender = smtpSender
|
||||
pings, bodies := log.got()
|
||||
fail := code != tc.wantCode || len(rec.sent) != tc.wantSent || !strings.Contains(stdout.String(), tc.wantOut)
|
||||
if tc.want == "" {
|
||||
fail = fail || len(pings) != 0
|
||||
} else {
|
||||
fail = fail || len(pings) != 1 || pings[0] != tc.want || !strings.Contains(bodies[0], tc.wantBody)
|
||||
}
|
||||
if fail {
|
||||
t.Errorf("%s: exit %d, mailed %v, pings %v %q; want exit %d, %d mails, %q with %q\nstdout %s\nstderr %s", tc.what, code, rec.sent, pings, bodies, tc.wantCode, tc.wantSent, tc.want, tc.wantBody, stdout.String(), stderr.String())
|
||||
}
|
||||
if tc.wantSent > 0 {
|
||||
if got := *rec.relays[0]; got.Host != "smtp.config.example" || got.Port != 2525 || got.Password != "env-pw" {
|
||||
t.Errorf("%s: relay %+v, want the configured one with the env password", tc.what, got)
|
||||
}
|
||||
if s, err := watchdog.LoadState(statePath); err != nil || s.Relay == nil || s.Relay.Host != "smtp.config.example" {
|
||||
t.Errorf("%s: the state caches relay %+v (%v), want the configured one", tc.what, s.Relay, err)
|
||||
}
|
||||
}
|
||||
if strings.Contains(tc.wantOut, "watchdog/state") {
|
||||
if aside, _ := filepath.Glob(statePath + ".unreadable-*"); len(aside) != 1 {
|
||||
t.Errorf("%s: moved aside %v, want one file", tc.what, aside)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,10 +1,20 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/watchdog"
|
||||
)
|
||||
|
||||
// TestProxyFinding: a listening proxy is healthy; a closed port is the critical
|
||||
@@ -40,3 +50,282 @@ func TestSplitList(t *testing.T) {
|
||||
t.Fatalf("splitList = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMailHold: mail waits through the installer's quiet window, and on a host
|
||||
// standing by for another host's off-site bucket while that host writes it.
|
||||
func TestMailHold(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
quiet, status := filepath.Join(dir, "quiet"), filepath.Join(dir, "status.json")
|
||||
now := time.Now()
|
||||
if got := mailHold(quiet, true, status, now); got != "" {
|
||||
t.Errorf("no quiet file, no status: %q", got)
|
||||
}
|
||||
writeTestFile(t, quiet, fmt.Sprintf("%d\n", now.Add(time.Hour).Unix()), 0o644)
|
||||
if got := mailHold(quiet, false, status, now); !strings.Contains(got, "quiet until") {
|
||||
t.Errorf("inside the quiet window: %q", got)
|
||||
}
|
||||
os.Remove(quiet)
|
||||
|
||||
w := &offsite.Writer{HostID: "bbbbbbbbbbbbbbbb", Host: "prod-1", At: now.Add(-40 * time.Minute)}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
st offsite.Status
|
||||
offsiteOn bool
|
||||
held bool
|
||||
}{
|
||||
{"standing by for a live writer", offsite.Status{Standby: true, Writer: w}, true, true},
|
||||
{"standing by, [offsite] since removed", offsite.Status{Standby: true, Writer: w}, false, false},
|
||||
{"standing by for a writer gone quiet", offsite.Status{Standby: true, Writer: &offsite.Writer{HostID: w.HostID, Host: w.Host, At: now.Add(-offsite.WriterLive - time.Minute)}}, true, false},
|
||||
{"standing by, no writer named", offsite.Status{Standby: true}, true, false},
|
||||
{"displaced", offsite.Status{Displaced: true, Writer: w}, true, false},
|
||||
{"the writer itself", offsite.Status{LastSuccess: now}, true, false},
|
||||
} {
|
||||
if err := offsite.WriteStatus(status, tc.st); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got := mailHold(quiet, tc.offsiteOn, status, now)
|
||||
if (got != "") != tc.held || (tc.held && !strings.Contains(got, "stands by for host prod-1 (id bbbbbbbbbbbbbbbb)")) {
|
||||
t.Errorf("%s: hold = %q, want held %v", tc.what, got, tc.held)
|
||||
}
|
||||
}
|
||||
writeTestFile(t, status, "{", 0o600)
|
||||
if got := mailHold(quiet, true, status, now); got != "" {
|
||||
t.Errorf("an unreadable status held the mail: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// watchdogHost is a host whole watchdog passes run on: the API server and PostgreSQL
|
||||
// are down (no kubeconfig, nothing on the database port), the relay is a recorder and
|
||||
// the heartbeat URL points at a ping log.
|
||||
type watchdogHost struct {
|
||||
dir, statePath string
|
||||
// fallbackPath stands in for /run/felis, in a directory of its own.
|
||||
fallbackPath string
|
||||
args []string
|
||||
rec *alertRecorder
|
||||
pings *pingLog
|
||||
url string
|
||||
}
|
||||
|
||||
func newWatchdogHost(t *testing.T, cfg string, state *watchdog.State) *watchdogHost {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
t.Setenv("KUBECONFIG", filepath.Join(dir, "no-kubeconfig"))
|
||||
pings, srv := newPingServer(t)
|
||||
h := &watchdogHost{dir: dir, statePath: filepath.Join(dir, "state.json"), fallbackPath: filepath.Join(t.TempDir(), "watchdog-state.json"),
|
||||
rec: &alertRecorder{}, pings: pings, url: srv.URL}
|
||||
writeTestFile(t, filepath.Join(dir, "felis.toml"), cfg, 0o600)
|
||||
writeTestFile(t, filepath.Join(dir, "watchdog-heartbeat-url"), srv.URL+"/check-key\n", 0o600)
|
||||
if state != nil {
|
||||
if err := watchdog.SaveState(h.statePath, state); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
h.args = []string{
|
||||
"-config", filepath.Join(dir, "felis.toml"), "-state", h.statePath, "-fallback-state", h.fallbackPath, "-quiet-file", filepath.Join(dir, "quiet"),
|
||||
"-backup-dir", "", "-disk-paths", dir, "-k3s-cert-dirs", "", "-smtp-password-file", filepath.Join(dir, "smtp-password"),
|
||||
"-offsite-status", filepath.Join(dir, "offsite-status.json"), "-build-tools-status", filepath.Join(dir, "build-tools.json"),
|
||||
"-heartbeat-file", filepath.Join(dir, "watchdog-heartbeat-url"),
|
||||
}
|
||||
return h
|
||||
}
|
||||
|
||||
// run is one pass; flags given here override the host's own.
|
||||
func (h *watchdogHost) run(flags ...string) (code int, stdout, stderr string) {
|
||||
watchdogSender = h.rec.sender
|
||||
defer func() { watchdogSender = smtpSender }()
|
||||
var out, errOut bytes.Buffer
|
||||
code = cmdWatchdog(append(append([]string(nil), h.args...), flags...), &out, &errOut)
|
||||
return code, out.String(), errOut.String()
|
||||
}
|
||||
|
||||
// duePostgres is a state whose owner has not yet been told of PostgreSQL, down
|
||||
// for an hour.
|
||||
func duePostgres(cachedPassword string) *watchdog.State {
|
||||
s := &watchdog.State{Recipients: []string{"[email protected]"}, SMTPPassword: cachedPassword}
|
||||
s.Observe(watchdog.Report{Findings: []watchdog.Finding{watchdog.PostgresDown(errors.New("refused"))}}, time.Now().Add(-time.Hour))
|
||||
return s
|
||||
}
|
||||
|
||||
// TestWatchdogRunKeepsClusterAlertsWhileTheAPIIsDown: with the API server down,
|
||||
// the alerts under the cluster checks keep their state, since nothing looked
|
||||
// at them; an alert of a check that did run and found nothing reads as cleared.
|
||||
func TestWatchdogRunKeepsClusterAlertsWhileTheAPIIsDown(t *testing.T) {
|
||||
told := time.Now().Add(-30 * time.Minute)
|
||||
cluster := []string{"deployment/felis-api", "node/felis-1/NotReady", "server-failed/lobby"}
|
||||
var r watchdog.Report
|
||||
for _, key := range append([]string{"proxy"}, cluster...) {
|
||||
r.Findings = append(r.Findings, watchdog.Finding{Key: key, Severity: watchdog.Critical, SummaryEN: key + " is down"})
|
||||
}
|
||||
s := &watchdog.State{Recipients: []string{"[email protected]"}}
|
||||
s.Commit(s.Observe(r, told), told)
|
||||
h := newWatchdogHost(t, testWatchdogConfig, s)
|
||||
|
||||
code, stdout, stderr := h.run()
|
||||
if code != 0 || len(h.rec.relays) != 0 {
|
||||
t.Fatalf("exit %d, mailed %v\nstdout %s\nstderr %s", code, h.rec.sent, stdout, stderr)
|
||||
}
|
||||
for _, want := range []string{"felis watchdog: [critical] kube-api: ", "felis watchdog: [critical] postgres: "} {
|
||||
if !strings.Contains(stdout, want) {
|
||||
t.Errorf("stdout lacks %q:\n%s", want, stdout)
|
||||
}
|
||||
}
|
||||
got, err := watchdog.LoadState(h.statePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, key := range cluster {
|
||||
if a := got.Alerts[key]; a == nil || !a.ClearedAt.IsZero() || !a.Notified.Equal(told) {
|
||||
t.Errorf("%s with the API server down: %+v, want it kept open as mailed at %s", key, a, told)
|
||||
}
|
||||
}
|
||||
if a := got.Alerts["proxy"]; a == nil || a.ClearedAt.IsZero() {
|
||||
t.Errorf("proxy, whose check ran and found nothing: %+v, want it cleared", a)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogDryRunMailsAndSavesNothing: a dry run prints the mail that is due
|
||||
// and the heartbeat it would ping, and sends, pings and saves nothing.
|
||||
func TestWatchdogDryRunMailsAndSavesNothing(t *testing.T) {
|
||||
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
|
||||
h := newWatchdogHost(t, testWatchdogConfig+testWatchdogSMTP, duePostgres(""))
|
||||
before, err := os.ReadFile(h.statePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
code, stdout, stderr := h.run("-dry-run")
|
||||
after, err := os.ReadFile(h.statePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
pings, _ := h.pings.got()
|
||||
if code != 0 || len(h.rec.relays) != 0 || len(pings) != 0 || !bytes.Equal(before, after) {
|
||||
t.Errorf("dry run: exit %d, relays %d, pings %v, state changed %v\nstdout %s\nstderr %s", code, len(h.rec.relays), pings, !bytes.Equal(before, after), stdout, stderr)
|
||||
}
|
||||
host, _ := os.Hostname()
|
||||
for _, want := range []string{
|
||||
"felis watchdog: due to be mailed to [email protected]:\nSubject: Felis 严重告警(" + host + "):1 项异常 · 1 firing\n\n",
|
||||
"felis watchdog: a run pings the heartbeat at " + h.url + "/...\n",
|
||||
} {
|
||||
if !strings.Contains(stdout, want) {
|
||||
t.Errorf("stdout lacks %q:\n%s", want, stdout)
|
||||
}
|
||||
}
|
||||
if strings.Contains(stdout, "\r") {
|
||||
t.Errorf("the mail body printed with CRLF:\n%q", stdout)
|
||||
}
|
||||
|
||||
h = newWatchdogHost(t, testWatchdogConfig, nil)
|
||||
code, stdout, stderr = h.run("-dry-run")
|
||||
if code != 0 || !strings.Contains(stdout, "felis watchdog: nothing is due to be mailed\n") {
|
||||
t.Errorf("dry run with nothing due: exit %d\nstdout %s\nstderr %s", code, stdout, stderr)
|
||||
}
|
||||
if _, err := os.Stat(h.statePath); !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Errorf("a dry run wrote the state (%v)", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogRunSignsInWithTheHostRelayPassword: a pass caches the relay
|
||||
// password from the host copy even while the cluster is down, and signs in with
|
||||
// it; with no host copy and no cluster it keeps the one it had.
|
||||
func TestWatchdogRunSignsInWithTheHostRelayPassword(t *testing.T) {
|
||||
const smtpNoRef = "[smtp]\nhost = \"smtp.config.example\"\nport = 2525\nfrom = \"[email protected]\"\nusername = \"felis\"\n"
|
||||
for _, tc := range []struct{ what, hostCopy, want string }{
|
||||
{"the host copy", "host-pw", "host-pw"},
|
||||
{"no host copy, the cluster down", "", "cached-pw"},
|
||||
} {
|
||||
h := newWatchdogHost(t, testWatchdogConfig+smtpNoRef, duePostgres("cached-pw"))
|
||||
if tc.hostCopy != "" {
|
||||
writeTestFile(t, filepath.Join(h.dir, "smtp-password"), tc.hostCopy, 0o600)
|
||||
}
|
||||
code, stdout, stderr := h.run()
|
||||
if code != 0 || len(h.rec.relays) != 1 || h.rec.relays[0].Password != tc.want {
|
||||
t.Errorf("%s: exit %d, relays %+v; want one mail signed in with %q\nstdout %s\nstderr %s", tc.what, code, h.rec.relays, tc.want, stdout, stderr)
|
||||
continue
|
||||
}
|
||||
if s, err := watchdog.LoadState(h.statePath); err != nil || s.SMTPPassword != tc.want {
|
||||
t.Errorf("%s: the state caches another password (%v)", tc.what, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogRunChecksWhatIsConfigured: the proxy, the database backups, the
|
||||
// off-site copy and the build lane's scan DB are each checked when the host
|
||||
// has them, and only then.
|
||||
func TestWatchdogRunChecksWhatIsConfigured(t *testing.T) {
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
closed := ln.Addr().String()
|
||||
ln.Close()
|
||||
const offsiteTable = "[offsite]\nendpoint = \"https://s3.example.com\"\nbucket = \"felis\"\n"
|
||||
const registry = "[registry]\nurl = \"registry.felis.svc:5000\"\n"
|
||||
optional := []string{"proxy", "db-backup", "offsite", "scan-db"}
|
||||
for _, tc := range []struct {
|
||||
what string
|
||||
cfg string
|
||||
flags func(dir string) []string
|
||||
want string // the one optional check that reports, "" for none
|
||||
}{
|
||||
{what: "none of them", cfg: testWatchdogConfig},
|
||||
{what: "a proxy address", cfg: testWatchdogConfig, flags: func(string) []string { return []string{"-proxy-addr", closed} }, want: "proxy"},
|
||||
{what: "a backup directory", cfg: testWatchdogConfig, flags: func(dir string) []string { return []string{"-backup-dir", dir} }, want: "db-backup"},
|
||||
{what: "[offsite]", cfg: testWatchdogConfig + offsiteTable, want: "offsite"},
|
||||
{what: "a registry with the default scan DB", cfg: testWatchdogConfig + registry, want: "scan-db"},
|
||||
{what: "a registry with its scan DB under mirror/", cfg: testWatchdogConfig + registry + "trivy_db_repository = \"registry.felis.svc:5000/mirror/trivy-db\"\n", want: "scan-db"},
|
||||
{what: "a registry with the scan DB elsewhere", cfg: testWatchdogConfig + registry + "trivy_db_repository = \"ghcr.io/aquasecurity/trivy-db\"\n"},
|
||||
} {
|
||||
h := newWatchdogHost(t, tc.cfg, nil)
|
||||
var flags []string
|
||||
if tc.flags != nil {
|
||||
flags = tc.flags(t.TempDir())
|
||||
}
|
||||
code, stdout, stderr := h.run(append(flags, "-dry-run")...)
|
||||
var reported []string
|
||||
for _, key := range optional {
|
||||
if strings.Contains(stdout, "] "+key+": ") {
|
||||
reported = append(reported, key)
|
||||
}
|
||||
}
|
||||
if code != 0 || strings.Join(reported, ",") != tc.want {
|
||||
t.Errorf("%s: exit %d, reported %v; want %q\nstdout %s\nstderr %s", tc.what, code, reported, tc.want, stdout, stderr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWatchdogRunStateThatDoesNotSave: a pass whose state does not save mails
|
||||
// as usual, keeps its state in the fallback, exits 1 and posts the failure to
|
||||
// the heartbeat. The next pass reads the fallback and mails nothing again; the
|
||||
// first one whose state saves drops the fallback.
|
||||
func TestWatchdogRunStateThatDoesNotSave(t *testing.T) {
|
||||
if os.Geteuid() == 0 {
|
||||
t.Skip("root writes into a read-only directory")
|
||||
}
|
||||
t.Setenv("FELIS_TEST_WATCHDOG_RELAY_PW", "env-pw")
|
||||
h := newWatchdogHost(t, testWatchdogConfig+testWatchdogSMTP, duePostgres(""))
|
||||
if err := os.Chmod(h.dir, 0o500); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { os.Chmod(h.dir, 0o700) })
|
||||
kept := "; kept in " + h.fallbackPath + " until the host restarts"
|
||||
for i := range 2 {
|
||||
code, stdout, stderr := h.run()
|
||||
pings, bodies := h.pings.got()
|
||||
if code != 1 || len(h.rec.sent) != 1 || !strings.Contains(stderr, "felis watchdog: save state: ") || !strings.Contains(stderr, kept) ||
|
||||
len(pings) != i+1 || pings[i] != "POST /check-key/fail" || !strings.HasPrefix(bodies[i], "the watchdog state did not save: ") {
|
||||
t.Fatalf("pass %d: exit %d, mailed %v, pings %v %q; want exit 1, one mail in all and the save failure posted to /fail\nstdout %s\nstderr %s", i, code, h.rec.sent, pings, bodies, stdout, stderr)
|
||||
}
|
||||
}
|
||||
|
||||
if err := os.Chmod(h.dir, 0o700); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
code, stdout, stderr := h.run()
|
||||
if _, err := os.Stat(h.fallbackPath); code != 0 || len(h.rec.sent) != 1 || !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Fatalf("once the state saves: exit %d, mailed %v, fallback %v; want exit 0, no new mail, the fallback gone\nstdout %s\nstderr %s", code, h.rec.sent, err, stdout, stderr)
|
||||
}
|
||||
if s, err := watchdog.LoadState(h.statePath); err != nil || s.Alerts["postgres"] == nil || s.Alerts["postgres"].Notified.IsZero() {
|
||||
t.Fatalf("saved state = %+v, %v; want the mailed PostgreSQL alert", s, err)
|
||||
}
|
||||
}
|
||||
+2214
-466
File diff suppressed because it is too large.
Load diff
+2759
-182
File diff suppressed because it is too large.
Load diff
Executable
+198
@@ -0,0 +1,198 @@
|
||||
#!/usr/bin/env bash
|
||||
# Builds everything a Felis release installs, for every architecture, into one directory:
|
||||
#
|
||||
# felis-linux-<arch> the felis binary, panel included
|
||||
# felis-image-felis-linux-<arch>.tar the control-plane image (OCI layout tars:
|
||||
# felis-image-game-linux-<arch>.tar the limbo, lobby and paper images `ctr images import`
|
||||
# felis-image-base-linux-<arch>.tar the registry and PostgreSQL reads them as-is)
|
||||
# felis-images-linux-<arch>.txt one "bundle role name manifest-digest config-digest"
|
||||
# line per image in the three tars
|
||||
# felis-velocity.jar the proxy plugin (JVM bytecode, one for every arch)
|
||||
# SHA256SUMS the sha256 of every file above
|
||||
#
|
||||
# Every name above is a contract with deploy/bootstrap.sh, which installs from a release's
|
||||
# assets (or from a directory like this one, FELIS_ARTIFACT_DIR) instead of building on the
|
||||
# host: with them a host needs neither Docker, Gradle, Go nor Docker Hub. The images are split
|
||||
# in three because they change at different rates: the control plane with every release, the
|
||||
# game images when game-stack.lock or a plugin changes, the base images almost never. An
|
||||
# upgrade downloads only the tars holding an image the host does not have yet.
|
||||
#
|
||||
# .github/workflows/release.yml runs this on a tag and publishes the directory; e2e.yml runs
|
||||
# it on a branch and installs from the directory.
|
||||
#
|
||||
# Needs docker with buildx, and binfmt/QEMU for the architectures other than the builder's:
|
||||
# the game images' runtime stages run apt-get on the target platform. The builder's own felis
|
||||
# binary writes the image tars, so Go is not needed here either.
|
||||
#
|
||||
# Usage: deploy/build-release-artifacts.sh <version> <out-dir>
|
||||
# FELIS_RELEASE_ARCHES the architectures to build (default "amd64 arm64")
|
||||
set -Eeuo pipefail
|
||||
|
||||
log() { printf '\033[1;36m[release]\033[0m %s\n' "$*"; }
|
||||
die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
[ "$#" -eq 2 ] || die "usage: $0 <version> <out-dir>"
|
||||
VERSION="$1"
|
||||
OUT="$2"
|
||||
ARCHES="${FELIS_RELEASE_ARCHES:-amd64 arm64}"
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
# The Gradle image the plugin builds run in: deploy/{limbo,lobby}/Dockerfile and bootstrap's
|
||||
# Velocity build name the same one (bootstrap_asset_test.go holds them together).
|
||||
PLUGIN_BUILD_IMAGE="gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01"
|
||||
|
||||
# The names and pins bootstrap uses, read from bootstrap itself so the two cannot drift.
|
||||
bootstrap_value() {
|
||||
local v
|
||||
v="$(sed -n "s/^$1=\"\(.*\)\"\$/\1/p" deploy/bootstrap.sh)"
|
||||
[ -n "$v" ] || die "deploy/bootstrap.sh sets no $1"
|
||||
printf '%s' "$v"
|
||||
}
|
||||
REGISTRY_URL="$(bootstrap_value REGISTRY_URL)"
|
||||
REGISTRY_IMAGE="$(bootstrap_value REGISTRY_IMAGE)"
|
||||
POSTGRES_IMAGE="$(bootstrap_value POSTGRES_IMAGE)"
|
||||
# The final stage of the repo Dockerfile, the base every felis image runs on.
|
||||
FELIS_BASE_IMAGE="$(awk '$1 == "FROM" && $2 ~ /^gcr\.io\/distroless\// { print $2 }' Dockerfile)"
|
||||
[ -n "$FELIS_BASE_IMAGE" ] || die "the Dockerfile names no distroless base"
|
||||
|
||||
# lock_value reads one KEY=value line of deploy/game-stack.lock, which bootstrap reads the
|
||||
# same way: never sourced, values limited to URL and version characters.
|
||||
lock_value() {
|
||||
local v
|
||||
v="$(awk -F= -v k="$1" '$1 == k { print substr($0, length(k) + 2); exit }' deploy/game-stack.lock)"
|
||||
case "$v" in
|
||||
''|*[!A-Za-z0-9._:/+%-]*) die "deploy/game-stack.lock: $1 is missing or malformed" ;;
|
||||
esac
|
||||
printf '%s' "$v"
|
||||
}
|
||||
|
||||
# image_tag_for_version is bootstrap's: the tag bootstrap names this release's image with.
|
||||
image_tag_for_version() {
|
||||
local v
|
||||
v="$(printf '%s' "$1" | tr -c 'A-Za-z0-9_.-' '-')"
|
||||
case "$v" in
|
||||
""|dev|[!A-Za-z0-9_]*) printf 'demo' ;;
|
||||
*) printf '%s' "${v:0:128}" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
host_arch() {
|
||||
case "$(uname -m)" in
|
||||
x86_64|amd64) printf 'amd64' ;;
|
||||
aarch64|arm64) printf 'arm64' ;;
|
||||
*) die "unsupported builder architecture $(uname -m)" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
mkdir -p "$OUT"
|
||||
[ -z "$(ls -A "$OUT")" ] || die "${OUT} is not empty; SHA256SUMS must describe exactly what one run built"
|
||||
WORK="$(mktemp -d)"
|
||||
trap 'rm -rf "$WORK"' EXIT
|
||||
platforms=""
|
||||
for arch in $ARCHES; do
|
||||
case "$arch" in amd64|arm64) ;; *) die "unsupported architecture ${arch}" ;; esac
|
||||
platforms="${platforms:+${platforms},}linux/${arch}"
|
||||
done
|
||||
|
||||
# ---- the felis binaries --------------------------------------------------------------
|
||||
# Through the repo Dockerfile, the one recipe that runs the panel build before go build:
|
||||
# a plain `go build` compiles against the placeholder panel and ships it.
|
||||
log "building felis ${VERSION} for ${platforms}"
|
||||
docker buildx build --platform "$platforms" \
|
||||
--build-arg FELIS_VERSION="$VERSION" \
|
||||
--output "type=local,dest=${WORK}/bin" .
|
||||
for arch in $ARCHES; do
|
||||
src="${WORK}/bin/usr/local/bin/felis"
|
||||
# buildx nests the output per platform only when it builds more than one.
|
||||
[ -f "${WORK}/bin/linux_${arch}/usr/local/bin/felis" ] && src="${WORK}/bin/linux_${arch}/usr/local/bin/felis"
|
||||
install -m 0755 "$src" "${OUT}/felis-linux-${arch}"
|
||||
# By ELF machine, never by running it: binfmt would run the wrong architecture happily.
|
||||
case "$arch" in
|
||||
amd64) want='x86-64' ;;
|
||||
arm64) want='ARM aarch64' ;;
|
||||
esac
|
||||
file "${OUT}/felis-linux-${arch}" | grep -q "$want" \
|
||||
|| die "felis-linux-${arch} is not a ${want} ELF: TARGETARCH did not reach the go build"
|
||||
done
|
||||
HOST_FELIS="${OUT}/felis-linux-$(host_arch)"
|
||||
[ -x "$HOST_FELIS" ] || die "no felis binary for the builder's own architecture ($(host_arch)); add it to FELIS_RELEASE_ARCHES"
|
||||
got="$("$HOST_FELIS" version | head -n 1)"
|
||||
[ "$got" = "felis ${VERSION}" ] || die "felis reports '${got}', want 'felis ${VERSION}': the version stamp did not reach the binary"
|
||||
|
||||
# ---- the Velocity plugin -------------------------------------------------------------
|
||||
# The command bootstrap's source path runs, on a copy of the tree so the checkout stays clean.
|
||||
log "building felis-velocity.jar in ${PLUGIN_BUILD_IMAGE%%@*}"
|
||||
mkdir -p "${WORK}/velocity"
|
||||
tar -C . --exclude=build --exclude=.gradle -cf - plugins/velocity plugins/shared | tar -C "${WORK}/velocity" -xf -
|
||||
docker run --rm --user "$(id -u):$(id -g)" -e HOME=/tmp -e GRADLE_USER_HOME=/tmp/gradle \
|
||||
-v "${WORK}/velocity:/src" -w /src/plugins/velocity \
|
||||
"$PLUGIN_BUILD_IMAGE" gradle --no-daemon clean build
|
||||
jars=( "${WORK}"/velocity/plugins/velocity/build/libs/felis-velocity-*.jar )
|
||||
[ "${#jars[@]}" -eq 1 ] && [ -f "${jars[0]}" ] || die "the felis-velocity build must produce exactly one plugin jar"
|
||||
install -m 0644 "${jars[0]}" "${OUT}/felis-velocity.jar"
|
||||
|
||||
# ---- the images ----------------------------------------------------------------------
|
||||
FELIS_IMAGE="${REGISTRY_URL}/felis/felis:$(image_tag_for_version "$VERSION")"
|
||||
game_build_args=(
|
||||
--build-arg "LIMBO_JAR_URL=$(lock_value LIMBO_JAR_URL)"
|
||||
--build-arg "LIMBO_JAR_SHA256=$(lock_value LIMBO_JAR_SHA256)"
|
||||
--build-arg "LIMBO_SCHEM_URL=$(lock_value LIMBO_SCHEM_URL)"
|
||||
--build-arg "LIMBO_SCHEM_SHA256=$(lock_value LIMBO_SCHEM_SHA256)"
|
||||
--build-arg "LIMBO_VERSION=$(lock_value LIMBO_VERSION)"
|
||||
--build-arg "PAPER_JAR_URL=$(lock_value PAPER_JAR_URL)"
|
||||
--build-arg "PAPER_JAR_SHA256=$(lock_value PAPER_JAR_SHA256)"
|
||||
--build-arg "LUCKPERMS_JAR_URL=$(lock_value LUCKPERMS_JAR_URL)"
|
||||
--build-arg "LUCKPERMS_JAR_SHA256=$(lock_value LUCKPERMS_JAR_SHA256)"
|
||||
)
|
||||
|
||||
# oci_image <arch> <dest> <docker build args...> builds one image for one platform into an
|
||||
# OCI layout tar. No attestations: the bundle carries images only.
|
||||
oci_image() {
|
||||
local arch="$1" dest="$2"
|
||||
shift 2
|
||||
docker buildx build --platform "linux/${arch}" --provenance=false --sbom=false \
|
||||
--output "type=oci,dest=${dest}" "$@"
|
||||
}
|
||||
|
||||
# bundle <arch> <group> <image-bundle flags...> writes one image tar and appends its lines
|
||||
# to the architecture's listing.
|
||||
bundle() {
|
||||
local arch="$1" group="$2" name
|
||||
shift 2
|
||||
name="felis-image-${group}-linux-${arch}.tar"
|
||||
"$HOST_FELIS" image-bundle --platform "linux/${arch}" \
|
||||
--out "${OUT}/${name}" --list "${WORK}/${name}.txt" "$@"
|
||||
awk -v b="$name" '{ print b, $0 }' "${WORK}/${name}.txt" >> "${OUT}/felis-images-linux-${arch}.txt"
|
||||
}
|
||||
|
||||
for arch in $ARCHES; do
|
||||
# The control-plane image wraps the released binary itself, byte for byte, the way
|
||||
# bootstrap wraps a downloaded binary.
|
||||
log "building the linux/${arch} images"
|
||||
mkdir -p "${WORK}/felis-${arch}"
|
||||
cp "${OUT}/felis-linux-${arch}" "${WORK}/felis-${arch}/felis"
|
||||
cat > "${WORK}/felis-${arch}/Dockerfile" <<EOF
|
||||
FROM ${FELIS_BASE_IMAGE}
|
||||
ENV PATH=/usr/local/bin:/usr/bin:/bin
|
||||
COPY --chmod=0755 felis /usr/local/bin/felis
|
||||
USER 65532:65532
|
||||
ENTRYPOINT ["/usr/local/bin/felis"]
|
||||
EOF
|
||||
oci_image "$arch" "${WORK}/felis-${arch}.tar" "${WORK}/felis-${arch}"
|
||||
for role in limbo lobby paper; do
|
||||
oci_image "$arch" "${WORK}/${role}-${arch}.tar" -f "deploy/${role}/Dockerfile" "${game_build_args[@]}" .
|
||||
done
|
||||
|
||||
bundle "$arch" felis --layout "felis=${FELIS_IMAGE}=${WORK}/felis-${arch}.tar"
|
||||
bundle "$arch" game \
|
||||
--layout "limbo=${REGISTRY_URL}/felis/limbo:demo=${WORK}/limbo-${arch}.tar" \
|
||||
--layout "lobby=${REGISTRY_URL}/felis/lobby:demo=${WORK}/lobby-${arch}.tar" \
|
||||
--layout "paper=${REGISTRY_URL}/felis/paper:demo=${WORK}/paper-${arch}.tar"
|
||||
bundle "$arch" base --pull "registry=${REGISTRY_IMAGE}" --pull "postgres=${POSTGRES_IMAGE}"
|
||||
rm -f "${WORK}"/*-"${arch}".tar
|
||||
done
|
||||
|
||||
log "writing SHA256SUMS"
|
||||
(cd "$OUT" && sha256sum -- *) > "${WORK}/SHA256SUMS"
|
||||
mv "${WORK}/SHA256SUMS" "${OUT}/SHA256SUMS"
|
||||
ls -l "$OUT"
|
||||
@@ -370,8 +370,10 @@ spec:
|
||||
description: |-
|
||||
EmptySince is when the operator first observed 0 online players during
|
||||
a Running phase (spec §8 idle auto-stop). It is reset when a player joins
|
||||
or the server stops, so the empty-duration counter starts fresh each time
|
||||
the server becomes unoccupied.
|
||||
or the server restarts or stops, so the empty-duration counter starts fresh
|
||||
each time the server becomes unoccupied. A zero sampled within three minutes
|
||||
of a run's first ready probe stamps nothing: the players it came up for may
|
||||
not be in yet.
|
||||
format: date-time
|
||||
type: string
|
||||
endpoint:
|
||||
@@ -420,7 +422,9 @@ spec:
|
||||
status ping is never sufficient — spec §5).
|
||||
type: boolean
|
||||
readySignalAt:
|
||||
description: ReadySignalAt is when the first RCON probe succeeded.
|
||||
description: |-
|
||||
ReadySignalAt is when the first RCON probe of the current run succeeded.
|
||||
Every Starting or Stopping pass clears it, so each start is measured once.
|
||||
format: date-time
|
||||
type: string
|
||||
startRequestedAt:
|
||||
@@ -433,6 +437,14 @@ spec:
|
||||
because the two endpoints fall in different reconcile passes.
|
||||
format: date-time
|
||||
type: string
|
||||
stopNoticeAt:
|
||||
description: |-
|
||||
StopNoticeAt is when the operator told the players on a server that it is
|
||||
about to stop (desiredState flipped to Stopped with players online). The stop
|
||||
itself waits until StopNoticeWindow has passed since then; the stamp is cleared
|
||||
once the server is scaled down, or when desiredState goes back to Running first.
|
||||
format: date-time
|
||||
type: string
|
||||
type: object
|
||||
type: object
|
||||
served: true
|
||||
|
||||
+124
-10
@@ -7,10 +7,12 @@
|
||||
# sudo bash deploy/e2e_check.sh release # after the newest release installed
|
||||
# sudo bash deploy/e2e_check.sh upgrade # after this commit ran over a release
|
||||
#
|
||||
# It asks what an operator's first minutes ask: the binary runs, the control plane is
|
||||
# rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft
|
||||
# status ping, and the host timers are there. A rerun must also leave the proxy running
|
||||
# (it restarts only when what it runs changed) and keep every earlier answer.
|
||||
# It asks what an operator's first minutes ask: the binary runs, the control plane and its
|
||||
# database are rolled out and ready, a database backup can be taken and restored, the panel
|
||||
# answers on its NodePort, the proxy answers a Minecraft status ping, the host timers are
|
||||
# all there and waiting, and the backup and watchdog runs the installer made succeeded.
|
||||
# A rerun must also leave the proxy running (it restarts only when what it runs changed)
|
||||
# and keep every earlier answer.
|
||||
set -euo pipefail
|
||||
|
||||
phase="${1:?usage: e2e_check.sh install|rerun|release|upgrade}"
|
||||
@@ -29,7 +31,12 @@ check() { # label command...
|
||||
|
||||
check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "'
|
||||
|
||||
for d in felis-api felis-operator registry; do
|
||||
deploys=(felis-api felis-operator registry)
|
||||
# A release may still run the database on the host; the upgrade moves it into felis-postgres.
|
||||
if [ "$phase" != release ] || "${KUBECTL[@]}" -n felis get deploy/felis-postgres >/dev/null 2>&1; then
|
||||
deploys=(felis-postgres "${deploys[@]}")
|
||||
fi
|
||||
for d in "${deploys[@]}"; do
|
||||
check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
|
||||
done
|
||||
|
||||
@@ -41,13 +48,120 @@ internal="$("${KUBECTL[@]}" -n felis get svc felis-api-internal -o jsonpath='{.s
|
||||
check "felis-api is ready (database and cluster reachable)" \
|
||||
curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"
|
||||
|
||||
for unit in k3s postgresql felis-velocity; do
|
||||
for unit in k3s felis-velocity; do
|
||||
check "${unit} is active" systemctl is-active --quiet "$unit"
|
||||
done
|
||||
# A release may predate a timer; what this commit installs has them all.
|
||||
# pod_psql runs one statement in the database's pod, over its socket, as the felis role on
|
||||
# the felis database.
|
||||
pod_psql() {
|
||||
"${KUBECTL[@]}" -n felis exec -i deploy/felis-postgres -c postgres -- \
|
||||
psql -X -q -At -v ON_ERROR_STOP=1 -U felis -d felis -c "$1"
|
||||
}
|
||||
|
||||
# restore_drill walks troubleshooting.md's "Restore on the same host": refused while the
|
||||
# control plane is connected; with it scaled to 0 the bundle comes back (a row written after
|
||||
# it is gone) and the database it replaced is kept; migrate up runs; the control plane serves
|
||||
# again.
|
||||
restore_drill() { # dir bundle
|
||||
local dir="$1" bundle="$2" out rc
|
||||
local sel="app.kubernetes.io/part-of=felis-control-plane,app.kubernetes.io/component in (api,operator)"
|
||||
if ! pod_psql "INSERT INTO platform_settings (key, value) VALUES ('e2e_restore_drill', '1')" >/dev/null; then
|
||||
fail "write a row after the bundle"
|
||||
return
|
||||
fi
|
||||
rc=0
|
||||
out="$(/usr/local/bin/felis db restore -dir "$dir" -yes "$bundle" 2>&1)" || rc=$?
|
||||
if [ "$rc" -eq 1 ] && grep -q "other clients are connected to the database" <<<"$out"; then
|
||||
pass "felis db restore refuses while the control plane is connected"
|
||||
else
|
||||
fail "felis db restore refuses while the control plane is connected (exit ${rc}): ${out}"
|
||||
fi
|
||||
|
||||
"${KUBECTL[@]}" -n felis scale deployment felis-api felis-operator --replicas=0 >/dev/null
|
||||
for _ in $(seq 60); do
|
||||
[ -z "$("${KUBECTL[@]}" -n felis get pods -l "$sel" -o name)" ] && break
|
||||
sleep 2
|
||||
done
|
||||
rc=0
|
||||
out="$(/usr/local/bin/felis db restore -dir "$dir" -yes "$bundle" 2>&1)" || rc=$?
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
pass "felis db restore replays the bundle with the control plane scaled to 0"
|
||||
else
|
||||
fail "felis db restore replays the bundle with the control plane scaled to 0 (exit ${rc}): ${out}"
|
||||
fi
|
||||
check "the restore dropped the row written after the bundle" \
|
||||
test "$(pod_psql "SELECT count(*) FROM platform_settings WHERE key = 'e2e_restore_drill'")" = 0
|
||||
check "the restore kept the database it replaced in a pre-restore bundle" \
|
||||
sh -c "ls '${dir}' | grep -q -- '-pre-restore\.tar\$'"
|
||||
check "felis migrate up runs on the restored database" \
|
||||
/usr/local/bin/felis migrate up -config /etc/felis/felis.host.toml
|
||||
"${KUBECTL[@]}" -n felis scale deployment felis-api felis-operator --replicas=1 >/dev/null
|
||||
for d in felis-api felis-operator; do
|
||||
check "deployment ${d} is rolled out again after the restore" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
|
||||
done
|
||||
check "felis-api is ready on the restored database" \
|
||||
curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"
|
||||
}
|
||||
|
||||
# The database runs in k3s; a release may still run it on the host, and the upgrade moved
|
||||
# it. The host has no PostgreSQL client: a bundle that verifies proves felis reaches the
|
||||
# database's pod through kubectl exec, and that pg_dump there reads every table.
|
||||
if [ "$phase" != release ]; then
|
||||
for timer in felis-db-backup.timer felis-watchdog.timer felis-update-check.timer; do
|
||||
check "${timer} is scheduled" systemctl is-enabled --quiet "$timer"
|
||||
check "the host's own postgresql is stopped" sh -c '! systemctl is-active --quiet postgresql'
|
||||
bundle_dir="$(mktemp -d)"
|
||||
if out="$(/usr/local/bin/felis db backup -dir "$bundle_dir" -state-dir "" -no-servers 2>&1)"; then
|
||||
bundle="$(printf '%s\n' "$out" | sed -n 's/^felis db backup: wrote //p' | tail -n 1)"
|
||||
check "felis db backup writes a bundle that verifies" /usr/local/bin/felis db verify "$bundle"
|
||||
# A rerun keeps what the install left; the install and the upgrade restore it.
|
||||
[ "$phase" = rerun ] || restore_drill "$bundle_dir" "$bundle"
|
||||
else
|
||||
fail "felis db backup writes a bundle: ${out}"
|
||||
fi
|
||||
rm -rf "$bundle_dir"
|
||||
fi
|
||||
# The daily reaper is what deletes expired archives. Run it once from its CronJob: the API
|
||||
# accepts a pod template naming a ServiceAccount that does not exist, and only the Job's
|
||||
# pod creation fails, so rendering it proves nothing. A release may carry exactly that bug.
|
||||
if [ "$phase" != release ]; then
|
||||
reaper_job="felis-e2e-reaper-${phase}"
|
||||
"${KUBECTL[@]}" -n minecraft delete job "$reaper_job" --ignore-not-found >/dev/null
|
||||
if "${KUBECTL[@]}" -n minecraft create job --from=cronjob/felis-reaper "$reaper_job" >/dev/null; then
|
||||
check "the reaper CronJob runs to completion" \
|
||||
"${KUBECTL[@]}" -n minecraft wait --for=condition=complete "job/${reaper_job}" --timeout=180s
|
||||
"${KUBECTL[@]}" -n minecraft delete job "$reaper_job" --ignore-not-found >/dev/null
|
||||
else
|
||||
fail "a Job can be created from the reaper CronJob"
|
||||
fi
|
||||
fi
|
||||
# A release may predate a timer. What this commit installs has every timer bootstrap.sh
|
||||
# names, enabled and waiting, and no felis timer it does not name. The off-site copy's
|
||||
# comes only with an [offsite] bucket, which no e2e host has.
|
||||
if [ "$phase" != release ]; then
|
||||
timers="$(sed -n 's|^[A-Z_]*_TIMER="/etc/systemd/system/\(felis-[a-z-]*\.timer\)"$|\1|p' "$(dirname "$0")/bootstrap.sh")"
|
||||
check "bootstrap.sh names the felis timers" test -n "$timers"
|
||||
for timer in $timers; do
|
||||
if [ "$timer" = felis-offsite.timer ]; then
|
||||
check "${timer} is not installed without an [offsite] bucket" test ! -e "/etc/systemd/system/${timer}"
|
||||
continue
|
||||
fi
|
||||
check "${timer} is enabled" systemctl is-enabled --quiet "$timer"
|
||||
check "${timer} is waiting" systemctl is-active --quiet "$timer"
|
||||
done
|
||||
for timer in $(systemctl list-unit-files --no-legend 'felis-*.timer' | awk '{print $1}'); do
|
||||
check "${timer} is a timer bootstrap.sh installs" grep -qxF "$timer" <<<"$timers"
|
||||
done
|
||||
# The installer runs the backup and the watchdog once itself and only warns when that
|
||||
# fails: a unit that cannot run (a flag the binary lacks, a path its sandbox hides) shows
|
||||
# here. Without an [smtp] relay the watchdog has no mail to fail on.
|
||||
for unit in felis-db-backup.service felis-watchdog.service; do
|
||||
check "${unit} has run" test "$(systemctl show -p ExecMainStartTimestampMonotonic --value "$unit")" != 0
|
||||
check "${unit}'s last run succeeded" test "$(systemctl show -p Result --value "$unit")" = success
|
||||
done
|
||||
# A failed watchdog run starts the unit its OnFailure= names, which must be installed.
|
||||
on_failure="$(systemctl show -p OnFailure --value felis-watchdog.service)"
|
||||
check "felis-watchdog.service names an OnFailure= unit" test -n "$on_failure"
|
||||
for unit in $on_failure; do
|
||||
check "${unit} is installed" test "$(systemctl show -p LoadState --value "$unit")" = loaded
|
||||
done
|
||||
fi
|
||||
|
||||
@@ -126,5 +240,5 @@ if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS (${phase})"
|
||||
else
|
||||
echo "${fails} FAILED (${phase})"
|
||||
exit 1
|
||||
fi
|
||||
exit "$fails"
|
||||
@@ -0,0 +1,112 @@
|
||||
#!/bin/bash
|
||||
# The newest published release, for the e2e workflow's readme and upgrade jobs, and what
|
||||
# their installer logs must show about how that release reached the host:
|
||||
#
|
||||
# bash deploy/e2e_release.sh find # tag, binary, sums into $GITHUB_OUTPUT
|
||||
# bash deploy/e2e_release.sh check-own LOG # the release's own installer (upgrade)
|
||||
# bash deploy/e2e_release.sh check-readme LOG # this commit's installer on its default
|
||||
# # channel, the README's command (readme)
|
||||
#
|
||||
# The checks read TAG, BINARY and SUMS from the environment, as find wrote them. The
|
||||
# runners are x86_64, so the binary asset is felis-linux-amd64. deploy/e2e_release_test.sh
|
||||
# holds this script's own checks, against bootstrap.sh's own messages.
|
||||
set -euo pipefail
|
||||
|
||||
ASSET=felis-linux-amd64
|
||||
HOST_BIN=/usr/local/bin/felis
|
||||
fails=0
|
||||
|
||||
pass() { printf 'PASS %s\n' "$*"; }
|
||||
fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); }
|
||||
has() { # label fixed-string log
|
||||
if grep -qF -- "$2" "$3"; then pass "$1"; else fail "$1: no line with <$2> in $3"; fi
|
||||
}
|
||||
has_re() { # label regex log
|
||||
if grep -qE -- "$2" "$3"; then pass "$1"; else fail "$1: no line matching <$2> in $3"; fi
|
||||
}
|
||||
lacks_re() { # label regex log
|
||||
local hit
|
||||
if hit="$(grep -E -m 1 -- "$2" "$3")"; then fail "$1: ${hit}"; else pass "$1"; fi
|
||||
}
|
||||
|
||||
# find_release: `gh release view` with no tag answers with the newest release that is not a
|
||||
# prerelease, the one the installer's release channel resolves.
|
||||
#
|
||||
# An empty tag skips the readme and upgrade jobs, green. Only gh's own `release not found`
|
||||
# means there is no release; any other failure (a token it refused, a rate limit, a network
|
||||
# error) fails the step, or every release's upgrade would go untested without a word.
|
||||
find_release() {
|
||||
local lines err rc=0 tag names binary="" sums=""
|
||||
err="$(mktemp)"
|
||||
lines="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName,assets --jq '.tagName, .assets[].name' 2>"$err")" || rc=$?
|
||||
if [ "$rc" -ne 0 ] && grep -qx 'release not found' "$err"; then
|
||||
rm -f "$err"
|
||||
echo "::notice::no published release yet; the readme and upgrade jobs have nothing to install"
|
||||
printf 'tag=\nbinary=\nsums=\n' >> "${GITHUB_OUTPUT:-/dev/stdout}"
|
||||
return 0
|
||||
fi
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo "::error::gh release view failed (exit ${rc}), so the newest release is unknown: $(tr '\n' ' ' < "$err")"
|
||||
rm -f "$err"
|
||||
fails=$((fails + 1))
|
||||
return
|
||||
fi
|
||||
rm -f "$err"
|
||||
tag="$(printf '%s\n' "$lines" | head -n 1)"
|
||||
names="$(printf '%s\n' "$lines" | tail -n +2)"
|
||||
if [ -z "$tag" ]; then
|
||||
echo "::error::gh release view answered without a tag"
|
||||
fails=$((fails + 1))
|
||||
return
|
||||
fi
|
||||
if printf '%s\n' "$names" | grep -qxF "$ASSET"; then binary=yes; fi
|
||||
if printf '%s\n' "$names" | grep -qxF SHA256SUMS; then sums=yes; fi
|
||||
printf 'tag=%s\nbinary=%s\nsums=%s\n' "$tag" "$binary" "$sums" >> "${GITHUB_OUTPUT:-/dev/stdout}"
|
||||
}
|
||||
|
||||
# check_own: a release's installer, however old, downloads the release's binary when the
|
||||
# release publishes one, and says so in the same words.
|
||||
check_own() {
|
||||
local log="$1"
|
||||
if [ "${BINARY:-}" != yes ]; then
|
||||
echo "::notice::release ${TAG} publishes no ${ASSET}; its installer builds it from source"
|
||||
return 0
|
||||
fi
|
||||
has "the release's installer installed the release's binary" "installed ${ASSET} ${TAG} at ${HOST_BIN}" "$log"
|
||||
}
|
||||
|
||||
# check_readme: this commit's installer takes everything from a release that publishes its
|
||||
# SHA256SUMS and builds nothing on the host; from one without, it builds the tag from source
|
||||
# and says why.
|
||||
check_readme() {
|
||||
local log="$1" role
|
||||
if [ "${BINARY:-}" = yes ] && [ "${SUMS:-}" = yes ]; then
|
||||
has "the binary is the release's" "installed ${ASSET} ${TAG} at ${HOST_BIN}" "$log"
|
||||
has "the images and plugin come from the release" "release ${TAG}'s prebuilt images and Velocity plugin are installed as published" "$log"
|
||||
for role in felis limbo lobby paper; do
|
||||
has_re "felis/${role} is the release's" "felis/${role}:[^ ]* is the release's" "$log"
|
||||
done
|
||||
has_re "the registry image is the release's" "docker.io/library/registry@sha256:[0-9a-f]* is the release's" "$log"
|
||||
has_re "the postgres image is the release's" "docker.io/library/postgres@sha256:[0-9a-f]* is the release's" "$log"
|
||||
has "felis-velocity.jar is the release's" "felis-velocity.jar is the release's" "$log"
|
||||
lacks_re "nothing fell back to a build or a pull" "on this host instead|from Docker Hub instead|building felis from source" "$log"
|
||||
lacks_re "Docker was left alone" "docker already installed|installing docker|docker running" "$log"
|
||||
elif [ "${BINARY:-}" = yes ]; then
|
||||
has "the source build says why" "release ${TAG} publishes no SHA256SUMS, so ${ASSET} cannot be verified; building ${TAG} from source on this host instead" "$log"
|
||||
echo "::warning::release ${TAG} publishes no SHA256SUMS, so the README's install builds ${TAG} from source on the host, Docker included; a release cut by release.yml puts it on the assets"
|
||||
else
|
||||
has "the source build says why" "release ${TAG} publishes no usable ${ASSET}; building ${TAG} from source on this host instead" "$log"
|
||||
echo "::warning::release ${TAG} publishes no ${ASSET}, so the README's install builds ${TAG} from source on the host, Docker included; a release cut by release.yml puts it on the assets"
|
||||
fi
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
find) find_release ;;
|
||||
check-own) check_own "${2:?usage: e2e_release.sh check-own LOG}" ;;
|
||||
check-readme) check_readme "${2:?usage: e2e_release.sh check-readme LOG}" ;;
|
||||
*)
|
||||
echo "usage: e2e_release.sh find | check-own LOG | check-readme LOG" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
[ "$fails" -eq 0 ] || exit 1
|
||||
@@ -0,0 +1,187 @@
|
||||
#!/bin/bash
|
||||
# Checks for deploy/e2e_release.sh. Run it as: bash deploy/e2e_release_test.sh
|
||||
#
|
||||
# The installer logs it reads are built from bootstrap.sh's own ok/warn messages, expanded
|
||||
# with a release's values, so rewording one of them there fails here rather than in a
|
||||
# two-hour e2e run. gh is a stub that prints what a release listing would.
|
||||
set -u
|
||||
|
||||
here="$(dirname "$0")"
|
||||
ER="${1:-${here}/e2e_release.sh}"
|
||||
BS="${2:-${here}/bootstrap.sh}"
|
||||
[ -f "$ER" ] || { echo "no such script: $ER"; exit 1; }
|
||||
[ -f "$BS" ] || { echo "no such script: $BS"; exit 1; }
|
||||
fails=0
|
||||
|
||||
expect() { # label needle haystack
|
||||
case "$3" in
|
||||
*"$2"*) echo "PASS $1" ;;
|
||||
*) echo "FAIL $1: expected <$2> in:"; echo "$3"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
}
|
||||
status() { # label want got
|
||||
if [ "$2" = "$3" ]; then echo "PASS $1"; else echo "FAIL $1: exit $3, want $2"; fails=$((fails + 1)); fi
|
||||
}
|
||||
|
||||
root="$(mktemp -d)"
|
||||
trap 'rm -rf "$root"' EXIT
|
||||
|
||||
# msg <marker> prints bootstrap.sh's first ok/warn message holding <marker>, expanded with
|
||||
# the variables below the way the installer expands it. They are read only through that eval.
|
||||
# shellcheck disable=SC2034
|
||||
{
|
||||
tag=v1.2.3
|
||||
FELIS_REF="$tag"
|
||||
v="$tag"
|
||||
asset=felis-linux-amd64
|
||||
name="$asset"
|
||||
HOST_BIN=/usr/local/bin/felis
|
||||
digest=sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef
|
||||
}
|
||||
msg() {
|
||||
local line body
|
||||
line="$(grep -F -- "$1" "$BS" | grep -E '^[[:space:]]*(ok|warn) "' | head -n 1)"
|
||||
if [ -z "$line" ]; then
|
||||
echo "FAIL bootstrap.sh prints no message holding <$1>" >&2
|
||||
fails=$((fails + 1))
|
||||
return
|
||||
fi
|
||||
body="${line#*\"}"
|
||||
body="${body%\"*}"
|
||||
eval "printf '%s\n' \"${body}\""
|
||||
}
|
||||
# unusable <marker> prints the warning of bootstrap.sh's first artifact_unusable call holding
|
||||
# <marker>; artifact_unusable joins its two arguments with "; ".
|
||||
unusable() {
|
||||
local line
|
||||
line="$(grep -F -- "$1" "$BS" | grep -E '^[[:space:]]*artifact_unusable "' | head -n 1)"
|
||||
if [ -z "$line" ]; then
|
||||
echo "FAIL bootstrap.sh has no artifact_unusable call holding <$1>" >&2
|
||||
fails=$((fails + 1))
|
||||
return
|
||||
fi
|
||||
artifact_unusable() { printf '%s; %s\n' "$1" "$2"; }
|
||||
eval "$line"
|
||||
}
|
||||
image_line() { # target
|
||||
# shellcheck disable=SC2034 # read by msg's eval
|
||||
local target="$1"
|
||||
msg 'is the release'"'"'s (${digest:7:12})'
|
||||
}
|
||||
|
||||
# What an install from a complete release prints, in the installer's order.
|
||||
{
|
||||
msg 'ok "installed ${asset} ${FELIS_REF} at ${HOST_BIN}"'
|
||||
msg 'matches release ${tag}'"'"'s SHA256SUMS'
|
||||
msg 'prebuilt images and Velocity plugin are installed as published'
|
||||
for t in felis/felis:v1.2.3 felis/limbo:v1.2.3 felis/lobby:v1.2.3 felis/paper:v1.2.3 \
|
||||
docker.io/library/registry@sha256:aa docker.io/library/postgres@sha256:bb; do
|
||||
image_line "$t"
|
||||
done
|
||||
msg 'felis-velocity.jar is the release'"'"'s"'
|
||||
} > "$root/complete.log"
|
||||
|
||||
readme() { # log binary sums
|
||||
TAG="$tag" BINARY="$2" SUMS="$3" bash "$ER" check-readme "$1" 2>&1
|
||||
}
|
||||
own() { # log binary
|
||||
TAG="$tag" BINARY="$2" bash "$ER" check-own "$1" 2>&1
|
||||
}
|
||||
|
||||
out="$(readme "$root/complete.log" yes yes)"
|
||||
status "a complete release's install passes" 0 $?
|
||||
expect " and every image is checked" "PASS felis/paper is the release's" "$out"
|
||||
|
||||
grep -v 'felis/limbo:' "$root/complete.log" > "$root/nolimbo.log"
|
||||
out="$(readme "$root/nolimbo.log" yes yes)"
|
||||
status "an image missing from the log fails" 1 $?
|
||||
expect " and names it" "FAIL felis/limbo is the release's" "$out"
|
||||
|
||||
{
|
||||
cat "$root/complete.log"
|
||||
name=felis-images-linux-amd64.txt unusable '} cannot be used" "building its images on this host instead"'
|
||||
} > "$root/fallback.log"
|
||||
out="$(readme "$root/fallback.log" yes yes)"
|
||||
status "an image built on the host fails" 1 $?
|
||||
expect " and quotes the fallback" "building its images on this host instead" "$out"
|
||||
|
||||
{ cat "$root/complete.log"; msg 'ok "docker already installed"'; } > "$root/docker.log"
|
||||
out="$(readme "$root/docker.log" yes yes)"
|
||||
status "Docker touched fails" 1 $?
|
||||
|
||||
sed 's/installed felis-linux-amd64 v1.2.3/installed felis-linux-amd64 v1.2.2/' "$root/complete.log" > "$root/other.log"
|
||||
out="$(readme "$root/other.log" yes yes)"
|
||||
status "another release's binary fails" 1 $?
|
||||
|
||||
# A release that publishes its binary but no SHA256SUMS: this commit's installer builds the
|
||||
# tag from source and says so.
|
||||
msg 'publishes no SHA256SUMS, so ${name} cannot be verified' > "$root/nosums.log"
|
||||
out="$(readme "$root/nosums.log" yes "")"
|
||||
status "a release without SHA256SUMS passes when the fallback is announced" 0 $?
|
||||
expect " and warns in the run" "::warning::release v1.2.3 publishes no SHA256SUMS" "$out"
|
||||
out="$(readme "$root/complete.log" yes "")"
|
||||
status "a release without SHA256SUMS fails when nothing announced the fallback" 1 $?
|
||||
|
||||
msg 'publishes no usable ${asset}' > "$root/nobinary.log"
|
||||
out="$(readme "$root/nobinary.log" "" "")"
|
||||
status "a release without a binary passes when the fallback is announced" 0 $?
|
||||
out="$(readme "$root/nosums.log" "" "")"
|
||||
status "a release without a binary fails when the log says otherwise" 1 $?
|
||||
|
||||
# check-own: the release's own installer, which may predate SHA256SUMS.
|
||||
out="$(own "$root/complete.log" yes)"
|
||||
status "the release's installer downloading its binary passes" 0 $?
|
||||
out="$(own "$root/nobinary.log" yes)"
|
||||
status "the release's installer building from source fails" 1 $?
|
||||
out="$(own "$root/other.log" yes)"
|
||||
status "the release's installer downloading another release fails" 1 $?
|
||||
out="$(own "$root/nobinary.log" "")"
|
||||
status "a release without a binary asks nothing of its installer" 0 $?
|
||||
|
||||
# find: the listing gh answers with, in the step's outputs, exactly. GH_FAIL is what a
|
||||
# failing gh prints on stderr before it exits 1.
|
||||
mkdir -p "$root/bin"
|
||||
cat > "$root/bin/gh" <<'STUB'
|
||||
#!/bin/sh
|
||||
if [ -n "${GH_FAIL:-}" ]; then
|
||||
printf '%s\n' "$GH_FAIL" >&2
|
||||
exit 1
|
||||
fi
|
||||
printf '%s\n' $GH_LISTING
|
||||
STUB
|
||||
chmod +x "$root/bin/gh"
|
||||
find_run() { # listing [gh's error]: prints what find says; its outputs land in $root/out
|
||||
: > "$root/out"
|
||||
GH_LISTING="$1" GH_FAIL="${2:-}" GITHUB_REPOSITORY=FelisMC/Felis GITHUB_OUTPUT="$root/out" PATH="$root/bin:$PATH" bash "$ER" find 2>&1
|
||||
}
|
||||
same() { # label want got
|
||||
if [ "$2" = "$3" ]; then echo "PASS $1"; else printf 'FAIL %s: got\n%s\nwant\n%s\n' "$1" "$3" "$2"; fails=$((fails + 1)); fi
|
||||
}
|
||||
find_run "v1.2.3 felis-linux-amd64 felis-linux-arm64 SHA256SUMS felis-velocity.jar" >/dev/null
|
||||
same "find: a complete release" "$(printf 'tag=v1.2.3\nbinary=yes\nsums=yes')" "$(cat "$root/out")"
|
||||
find_run "v0.1.0 felis-linux-amd64 felis-linux-arm64" >/dev/null
|
||||
same "find: a release without SHA256SUMS" "$(printf 'tag=v0.1.0\nbinary=yes\nsums=')" "$(cat "$root/out")"
|
||||
find_run "v0.1.0 felis-linux-arm64 felis-linux-amd64.cdx.json SHA256SUMS.sig" >/dev/null
|
||||
same "find: only exact asset names count" "$(printf 'tag=v0.1.0\nbinary=\nsums=')" "$(cat "$root/out")"
|
||||
out="$(find_run "" "release not found")"
|
||||
status "find: no release passes" 0 $?
|
||||
same " with an empty tag, which skips the release jobs" "$(printf 'tag=\nbinary=\nsums=')" "$(cat "$root/out")"
|
||||
same " and says so" "::notice::no published release yet; the readme and upgrade jobs have nothing to install" "$out"
|
||||
|
||||
# Anything else gh fails on leaves the newest release unknown: the step fails, and writes no
|
||||
# tag that would skip the jobs.
|
||||
for e in "HTTP 401: Bad credentials (https://api.github.com/graphql)" \
|
||||
"API rate limit exceeded for installation ID 1." \
|
||||
"error connecting to api.github.com"; do
|
||||
out="$(find_run "" "$e")"
|
||||
status "find: gh failing with <$e> fails" 1 $?
|
||||
same " and quotes gh" "::error::gh release view failed (exit 1), so the newest release is unknown: $e " "$out"
|
||||
same " and writes no outputs" "" "$(cat "$root/out")"
|
||||
done
|
||||
out="$(find_run "")"
|
||||
status "find: gh answering nothing fails" 1 $?
|
||||
same " and says so" "::error::gh release view answered without a tag" "$out"
|
||||
same " and writes no outputs" "" "$(cat "$root/out")"
|
||||
|
||||
echo
|
||||
if [ "$fails" -eq 0 ]; then echo "ALL PASS"; else echo "${fails} FAILED"; exit 1; fi
|
||||
@@ -0,0 +1,208 @@
|
||||
#!/bin/bash
|
||||
# Seeds the database of the release the e2e upgrade job installed, and checks after the
|
||||
# upgrade that every seeded row came through unchanged. The job runs it around the upgrade:
|
||||
#
|
||||
# sudo bash deploy/e2e_seed.sh seed # after the release installed
|
||||
# sudo bash deploy/e2e_seed.sh check # after this commit ran over it
|
||||
#
|
||||
# A fresh install's database holds only the rows its migrations write, so without the seed
|
||||
# the upgrade moves and migrates an almost empty database: the move's row-count comparison
|
||||
# compares zeros and the pending migrations never meet an existing row. The seed writes the
|
||||
# columns the oldest release has (v0.1.0), so it applies to every release since, and the
|
||||
# check reads the same columns back: a migration that rewrites one of them on purpose
|
||||
# updates the snapshot query here.
|
||||
#
|
||||
# Every row is inert. Nothing in it is a credential anyone holds: the session and the
|
||||
# challenge carry the hash of bytes nobody kept and are spent already. Nothing is due for
|
||||
# the reaper or the operator, and the retention sweep keeps spent rows for 30 days after
|
||||
# they were spent. servers stays empty: its rows mirror MinecraftServer objects the
|
||||
# operator and the reaper reconcile, and a row without one would not stay as written.
|
||||
set -euo pipefail
|
||||
|
||||
phase="${1:?usage: e2e_seed.sh seed|check}"
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
KUBECTL=(/usr/local/bin/k3s kubectl)
|
||||
STATE_DIR=/var/tmp/felis-e2e-seed
|
||||
PG_MOVED_MARKER=/var/lib/felis/postgres-moved
|
||||
TABLES=(users account_links quotas sessions webauthn_challenges platform_settings audit_logs
|
||||
world_backups image_builds image_submissions account_migrations op_login_requests)
|
||||
fails=0
|
||||
|
||||
pass() { printf 'PASS %s\n' "$*"; }
|
||||
fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); }
|
||||
|
||||
# db_where says where the platform's database lives: in felis-postgres once the install
|
||||
# has it, else in the host PostgreSQL a release before it ran.
|
||||
db_where() {
|
||||
if "${KUBECTL[@]}" -n felis get deploy/felis-postgres >/dev/null 2>&1; then
|
||||
echo pod
|
||||
else
|
||||
echo host
|
||||
fi
|
||||
}
|
||||
|
||||
# db_psql runs the SQL on its stdin in the felis database at <where>.
|
||||
db_psql() { # where
|
||||
case "$1" in
|
||||
pod)
|
||||
"${KUBECTL[@]}" -n felis exec -i deploy/felis-postgres -c postgres -- \
|
||||
psql -X -q -At -v ON_ERROR_STOP=1 -U felis -d felis
|
||||
;;
|
||||
# From /, which the postgres user can always enter: psql warns about any other cwd.
|
||||
host) (cd / && runuser -u postgres -- psql -X -q -At -v ON_ERROR_STOP=1 -d felis) ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# snapshot prints every seeded row, one table after the other, each in key order. The host
|
||||
# server's time zone is the host's and the pod's is UTC, so timestamps print in UTC.
|
||||
snapshot() { # where
|
||||
db_psql "$1" <<'EOF'
|
||||
SET TimeZone = 'UTC';
|
||||
SELECT 'users ' || row(id, username, email, role, email_verified, disabled, deleted_at, created_at, updated_at)::text
|
||||
FROM users WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
SELECT 'account_links ' || row(user_id, mc_uuid, auth_source, verified_at)::text
|
||||
FROM account_links WHERE user_id LIKE 'e2e-seed-%' ORDER BY mc_uuid;
|
||||
SELECT 'quotas ' || row(user_id, max_servers, max_cpu_milli, max_memory_mb, max_storage_gb, updated_by, updated_at)::text
|
||||
FROM quotas WHERE user_id LIKE 'e2e-seed-%' ORDER BY user_id;
|
||||
SELECT 'sessions ' || row(token_hash, user_id, created_at, expires_at, revoked_at)::text
|
||||
FROM sessions WHERE user_id LIKE 'e2e-seed-%' ORDER BY token_hash;
|
||||
SELECT 'webauthn_challenges ' || row(id, user_id, purpose, session_data, expires_at, consumed_at, created_at)::text
|
||||
FROM webauthn_challenges WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
SELECT 'platform_settings ' || row(key, value, updated_at)::text
|
||||
FROM platform_settings WHERE key = 'e2e_seed';
|
||||
SELECT 'audit_logs ' || row(id, actor, source, action, server_name, request_id, payload, created_at)::text
|
||||
FROM audit_logs WHERE actor = 'e2e-seed-staff' ORDER BY id;
|
||||
SELECT 'world_backups ' || row(id, server_name, former_owner, backup_ref, size_bytes, reason, status, created_at, expires_at, deleted_at)::text
|
||||
FROM world_backups WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
SELECT 'image_builds ' || row(id, image_ref, status, dockerfile, context_ref, base_image, requested_by, job_name, log_ref, error, created_at, finished_at)::text
|
||||
FROM image_builds WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
SELECT 'image_submissions ' || row(id, submitted_by, display_name, context_ref, status, image_ref, build_id, reviewed_by, reject_reason, created_at, reviewed_at)::text
|
||||
FROM image_submissions WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
SELECT 'account_migrations ' || row(id, source_user_id, target_user_id, state, confirm_factor, confirmed_at, code_hash, code_expires_at, redeemed_at, created_at, updated_at)::text
|
||||
FROM account_migrations WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
SELECT 'op_login_requests ' || row(id, user_id, email, expires_at, created_at, consumed_at, approved_at, approved_by)::text
|
||||
FROM op_login_requests WHERE id LIKE 'e2e-seed-%' ORDER BY id;
|
||||
EOF
|
||||
}
|
||||
|
||||
# seed writes the rows in one transaction and keeps where they went and what they read as.
|
||||
# Text carries quotes, backslashes, control characters and non-ASCII, and the bytea a NUL,
|
||||
# so a dump or restore that mangles any of them shows in the check. Rows that the retention
|
||||
# sweep would judge by age are dated now.
|
||||
seed() {
|
||||
local where before t
|
||||
where="$(db_where)"
|
||||
if ! db_psql "$where" <<'EOF'; then
|
||||
BEGIN;
|
||||
INSERT INTO users (id, username, email, role, email_verified, disabled, created_at, updated_at) VALUES
|
||||
('e2e-seed-player', 'e2e_seed_player', '[email protected]', 'user', true, false,
|
||||
'2026-01-02 03:04:05.678901+00', '2026-01-02 03:04:05.678901+00'),
|
||||
('e2e-seed-staff', 'e2e_seed_staff', '[email protected]', 'admin', false, true,
|
||||
'2026-01-03 00:00:00+00', '2026-01-03 00:00:00+00');
|
||||
INSERT INTO account_links (user_id, mc_uuid, auth_source, verified_at) VALUES
|
||||
('e2e-seed-player', '5eed0000-e2e0-4000-8000-000000000001', 'thirdparty', '2026-01-02 04:00:00+00');
|
||||
INSERT INTO quotas (user_id, max_servers, max_cpu_milli, max_memory_mb, max_storage_gb, updated_by, updated_at) VALUES
|
||||
('e2e-seed-player', 2, 4000, 8192, NULL, 'e2e-seed-staff', '2026-01-04 00:00:00+00');
|
||||
INSERT INTO sessions (token_hash, user_id, created_at, expires_at, revoked_at) VALUES
|
||||
(encode(sha256(convert_to(gen_random_uuid()::text, 'UTF8')), 'hex'), 'e2e-seed-player',
|
||||
now(), now() + interval '30 days', now());
|
||||
INSERT INTO webauthn_challenges (id, user_id, purpose, session_data, expires_at, consumed_at, created_at) VALUES
|
||||
('e2e-seed-challenge', 'e2e-seed-player', 'passkey_register', '\x00ff0a0d5c27'::bytea,
|
||||
now() + interval '5 minutes', now(), now());
|
||||
INSERT INTO platform_settings (key, value, updated_at) VALUES
|
||||
('e2e_seed', '{"text": "quote '' dq \" backslash \\ tab\t newline\n 猫 🐱", "n": 1.50, "list": [1, null, true, {"k": "v"}]}',
|
||||
'2026-01-05 00:00:00+00');
|
||||
INSERT INTO audit_logs (actor, source, action, server_name, request_id, payload, created_at) VALUES
|
||||
('e2e-seed-staff', 'panel', 'e2e.seed', NULL, 'e2e-seed-req-1', '{"reason": "种子 \"quoted\"", "ids": [1, 2, 3]}', now()),
|
||||
('e2e-seed-staff', 'cli', 'e2e.seed', 'e2e-seed-world', 'e2e-seed-req-2', NULL, now());
|
||||
INSERT INTO world_backups (id, server_name, former_owner, backup_ref, size_bytes, reason, status, created_at, expires_at, deleted_at) VALUES
|
||||
('e2e-seed-backup', 'e2e-seed-world', 'e2e-seed-player', 'e2e-seed/world.tar.zst', 123456789012, 'manual', 'deleted',
|
||||
'2026-01-06 00:00:00+00', '2026-04-06 00:00:00+00', '2026-04-07 00:00:00+00');
|
||||
INSERT INTO image_builds (id, image_ref, status, dockerfile, context_ref, requested_by, error, created_at, finished_at) VALUES
|
||||
('e2e-seed-build', 'registry.felis.svc:5000/user-uploads/e2e-seed:latest', 'failed',
|
||||
E'FROM scratch\nLABEL note="e2e seed"\n', 'e2e-seed/context', 'e2e-seed-staff', 'e2e seed: never built',
|
||||
'2026-01-07 00:00:00+00', '2026-01-07 00:01:00+00');
|
||||
INSERT INTO image_submissions (id, submitted_by, display_name, context_ref, status, reviewed_by, reject_reason, created_at, reviewed_at) VALUES
|
||||
('e2e-seed-submission', 'e2e-seed-player', '种子整合包 e2e', 'e2e-seed/submission', 'rejected', 'e2e-seed-staff', 'e2e seed',
|
||||
'2026-01-08 00:00:00+00', '2026-01-08 01:00:00+00');
|
||||
INSERT INTO account_migrations (id, source_user_id, target_user_id, state, confirm_factor, confirmed_at, redeemed_at, created_at, updated_at) VALUES
|
||||
('e2e-seed-migration', 'e2e-seed-staff', 'e2e-seed-player', 'redeemed', 'email_otp', '2026-01-09 00:00:00+00',
|
||||
'2026-01-09 00:05:00+00', '2026-01-09 00:00:00+00', '2026-01-09 00:05:00+00');
|
||||
INSERT INTO op_login_requests (id, user_id, email, expires_at, created_at, consumed_at, approved_at, approved_by) VALUES
|
||||
('e2e-seed-oplogin', 'e2e-seed-staff', '[email protected]', now() + interval '10 minutes', now(), now(), now(), 'e2e-seed-staff');
|
||||
COMMIT;
|
||||
EOF
|
||||
fail "seed the release's database (${where})"
|
||||
return
|
||||
fi
|
||||
mkdir -p "$STATE_DIR"
|
||||
printf '%s\n' "$where" > "${STATE_DIR}/where"
|
||||
if ! before="$(snapshot "$where")"; then
|
||||
fail "read the seeded rows back"
|
||||
return
|
||||
fi
|
||||
printf '%s\n' "$before" > "${STATE_DIR}/before"
|
||||
for t in "${TABLES[@]}"; do
|
||||
if grep -q "^${t} " <<<"$before"; then
|
||||
pass "seeded ${t} (${where})"
|
||||
else
|
||||
fail "seeded ${t} (${where}): the snapshot has no row of it"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# check_upgrade reads the seeded rows from felis-postgres and compares them with what the
|
||||
# release's database held. A database the upgrade moved off the host also left the marker.
|
||||
check_upgrade() {
|
||||
local where after seq
|
||||
if [ ! -s "${STATE_DIR}/before" ]; then
|
||||
fail "the seed step left its snapshot in ${STATE_DIR}/before"
|
||||
return
|
||||
fi
|
||||
where="$(db_where)"
|
||||
if [ "$where" != pod ]; then
|
||||
fail "the database lives in felis-postgres after the upgrade"
|
||||
return
|
||||
fi
|
||||
if [ "$(cat "${STATE_DIR}/where")" = host ]; then
|
||||
if [ -f "$PG_MOVED_MARKER" ]; then
|
||||
pass "the upgrade moved the seeded host database into felis-postgres"
|
||||
else
|
||||
fail "the upgrade moved the seeded host database into felis-postgres: no ${PG_MOVED_MARKER}"
|
||||
fi
|
||||
fi
|
||||
if ! after="$(snapshot pod)"; then
|
||||
fail "read the seeded rows from felis-postgres"
|
||||
return
|
||||
fi
|
||||
if [ "$after" = "$(cat "${STATE_DIR}/before")" ]; then
|
||||
pass "every seeded row came through the upgrade unchanged"
|
||||
else
|
||||
fail "the seeded rows changed across the upgrade (< release, > felis-postgres):"
|
||||
diff "${STATE_DIR}/before" <(printf '%s\n' "$after") || true
|
||||
fi
|
||||
# A restore that loses a sequence's position hands out ids that are taken: the next
|
||||
# audit row would fail on its primary key.
|
||||
seq="$(db_psql pod <<<"SELECT coalesce(pg_sequence_last_value(pg_get_serial_sequence('audit_logs', 'id')::regclass), 0) >= (SELECT max(id) FROM audit_logs)")" || seq=error
|
||||
if [ "$seq" = t ]; then
|
||||
pass "audit_logs' id sequence is past every seeded id"
|
||||
else
|
||||
fail "audit_logs' id sequence is past every seeded id (${seq})"
|
||||
fi
|
||||
}
|
||||
|
||||
case "$phase" in
|
||||
seed) seed ;;
|
||||
check) check_upgrade ;;
|
||||
*)
|
||||
echo "usage: e2e_seed.sh seed|check" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS (seed ${phase})"
|
||||
exit 0
|
||||
fi
|
||||
echo "${fails} FAILED (seed ${phase})"
|
||||
exit 1
|
||||
@@ -33,7 +33,10 @@
|
||||
# must be >= 21 because current LOOHP/Limbo releases ship Java 21 API classes
|
||||
# (class-file major 65); a JDK 17 fails to read them with "wrong version 65.0, should be
|
||||
# 61.0". build.gradle still targets release 17 bytecode so the plugin loads on Java 17+.
|
||||
FROM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
|
||||
# On the build platform: the jar is plain bytecode, identical for every architecture, so a
|
||||
# release building the arm64 image on an amd64 runner compiles it natively instead of under
|
||||
# QEMU (deploy/build-release-artifacts.sh). The runtime stage below stays on the target.
|
||||
FROM --platform=$BUILDPLATFORM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
|
||||
WORKDIR /src
|
||||
# Copy what the limbo module needs: its own tree plus the shared link core it
|
||||
# srcDir-includes (../shared/src/main/java → /src/plugins/shared/src/main/java), so
|
||||
|
||||
@@ -58,6 +58,8 @@ Configuration (deployment inputs, never compiled in; env wins over a
|
||||
| `FELIS_LOBBY_SERVER` | Velocity server name to transfer to | `lobby` |
|
||||
| `FELIS_LOGIN_TIMEOUT_SECONDS` | login window (clamped 30–3600) | `600` |
|
||||
| `FELIS_HEALTH_PORT` | readiness port | `8080` |
|
||||
| `FELIS_API_CONNECT_TIMEOUT_SECONDS` | felis-api connect timeout (1–120) | `10` |
|
||||
| `FELIS_API_REQUEST_TIMEOUT_SECONDS` | felis-api call timeout (1–120) | `10` |
|
||||
|
||||
If the API config **or** the root domain is absent the login flow stays **OFF** and
|
||||
the plugin runs readiness-only (the same "load un-crippled" fail-safe the other
|
||||
@@ -114,6 +116,10 @@ LOOHP/Limbo would otherwise default to `30000`, unreachable through the Velocity
|
||||
idempotent, so a persisted world volume keeps all its other `server.properties`
|
||||
settings. Do **not** override `FELIS_GAME_PORT` except in lockstep with the operator.
|
||||
|
||||
It also pins `max-players=-1` (no cap, Limbo's own default): unbound players wait at
|
||||
the gate for up to ten minutes and a stopped server's players all fall back here at
|
||||
once, so a cap left on the volume would turn players away at the door.
|
||||
|
||||
## Configure (deployer's responsibility)
|
||||
|
||||
One setting this image does **not** guess (it keeps the release's own default):
|
||||
@@ -138,7 +144,7 @@ set them by hand:
|
||||
the minecraft namespace (and `felis setup` refreshes that replica from the control
|
||||
namespace), and the operator injects it into the `login` pod (only) as
|
||||
`FELIS_SERVICE_TOKEN` via a `secretKeyRef`, keyed off the reserved `login` name.
|
||||
`sudo felis rotate-token limbo` replaces it and restarts the pod. Until the token is
|
||||
`sudo felis rotate-token -yes limbo` replaces it and restarts the pod. Until the token is
|
||||
present the plugin fail-safes to readiness-only, so the gate is never broken — it
|
||||
simply does not authenticate yet.
|
||||
- **Service:** the login pod dials `FELIS_API_BASE_URL`, which resolves to the
|
||||
|
||||
@@ -61,7 +61,11 @@ set_prop() {
|
||||
# The secret is base64/hex-ish, but a '/' or '&' would still break a bare sed s///.
|
||||
# '|' as the delimiter plus escaping it is enough for every value we write.
|
||||
esc=$(printf '%s' "$2" | sed 's/[|\\&]/\\&/g')
|
||||
sed -i "s|^$1=.*|$1=${esc}|" "$PROPS"
|
||||
# Through a temp file rather than sed -i, which BSD sed reads differently, so the
|
||||
# same function runs under the entrypoint tests on any machine.
|
||||
sed "s|^$1=.*|$1=${esc}|" "$PROPS" > "$PROPS.tmp"
|
||||
cat "$PROPS.tmp" > "$PROPS"
|
||||
rm -f "$PROPS.tmp"
|
||||
else
|
||||
printf '%s=%s\n' "$1" "$2" >> "$PROPS"
|
||||
fi
|
||||
@@ -75,6 +79,11 @@ set_prop bungeecord false
|
||||
set_prop bungee-guard false
|
||||
set_prop velocity-modern true
|
||||
set_prop forwarding-secrets "$SECRET"
|
||||
# Unbound players wait here for up to ten minutes, and a stopped server's players all
|
||||
# fall back here at once, so the gate must never be full. -1 (no cap) is Limbo's own
|
||||
# default; it is pinned so a hand-edited properties file on the volume cannot bring
|
||||
# a cap back.
|
||||
set_prop max-players -1
|
||||
|
||||
echo "felis-limbo: server-port=${PORT}, velocity-modern=true (forwarding secret loaded, UUIDs are Mojang-verified)"
|
||||
JAVA_MEMORY_ARG=""
|
||||
|
||||
@@ -32,7 +32,10 @@
|
||||
# build rather than the module's wrapper, which would download the same distribution
|
||||
# again on every image build. The build checks every dependency against
|
||||
# plugins/paper/gradle/verification-metadata.xml and fails on a mismatch.
|
||||
FROM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
|
||||
# On the build platform: the jar is plain bytecode, identical for every architecture, so a
|
||||
# release building the arm64 image on an amd64 runner compiles it natively instead of under
|
||||
# QEMU (deploy/build-release-artifacts.sh). The runtime stage below stays on the target.
|
||||
FROM --platform=$BUILDPLATFORM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
|
||||
WORKDIR /src
|
||||
COPY plugins/paper/ ./plugins/paper/
|
||||
COPY plugins/shared/ ./plugins/shared/
|
||||
|
||||
@@ -46,3 +46,25 @@ sudo felis setup
|
||||
- Align `online-mode` / player forwarding with the off-cluster Velocity proxy.
|
||||
- The lobby speaks only the `felis:control` plugin-message channel; it holds no
|
||||
felis-api token by design (spec §12).
|
||||
|
||||
## What the lobby allows
|
||||
|
||||
felis-paper's `LobbyGuard` keeps the lobby a hub that nobody can hurt, get hurt in,
|
||||
or leave a mark on:
|
||||
|
||||
- every world is peaceful, with natural spawning, PvP, mob griefing and TNT off,
|
||||
time frozen at noon, clear weather and inventories kept;
|
||||
- players take no damage and never go hungry; a fall into the void lands at spawn;
|
||||
- a player without `felis.lobby.build` joins at spawn in adventure mode and cannot
|
||||
break or place blocks, use buckets, trample farmland, light fires, or harm mobs,
|
||||
item frames, paintings, armor stands or vehicles. Buttons, doors, pressure plates
|
||||
and containers keep working;
|
||||
- every join gets a chat line with a click that runs `/menu`.
|
||||
|
||||
`felis.lobby.build` defaults to ops. To let an admin build the lobby, grant it with
|
||||
LuckPerms (`lp user <name> permission set felis.lobby.build true` on the lobby console)
|
||||
or op them.
|
||||
|
||||
The entrypoint pins `max-players=200` on every boot, over Paper's default of 20: every
|
||||
authenticated player passes through here, and a stopped server's players arrive all at
|
||||
once.
|
||||
@@ -57,7 +57,11 @@ set_prop() {
|
||||
# otherwise corrupt this bare sed s||| and silently break the key. Same escaping as
|
||||
# deploy/limbo — without it an unlucky password kills the console/permission channel.
|
||||
esc=$(printf '%s' "$2" | sed 's/[|\\&]/\\&/g')
|
||||
sed -i "s|^$1=.*|$1=${esc}|" "$PROPS"
|
||||
# Through a temp file rather than sed -i, which BSD sed reads differently, so the
|
||||
# same function runs under the entrypoint tests on any machine.
|
||||
sed "s|^$1=.*|$1=${esc}|" "$PROPS" > "$PROPS.tmp"
|
||||
cat "$PROPS.tmp" > "$PROPS"
|
||||
rm -f "$PROPS.tmp"
|
||||
else
|
||||
printf '%s=%s\n' "$1" "$2" >> "$PROPS"
|
||||
fi
|
||||
@@ -66,6 +70,14 @@ set_prop() {
|
||||
set_prop server-port "$PORT"
|
||||
set_prop online-mode false
|
||||
|
||||
# Every authenticated player passes through the lobby, and a stopped server's players
|
||||
# arrive together (they fall back to the login gate, which sends them straight on).
|
||||
# Paper's default cap of 20 would turn the 21st away at the door. 200 is far above
|
||||
# what one node serves at once, and a flood beyond it is refused at the door instead
|
||||
# of running the 1Gi lobby out of memory. What the world itself allows (no damage, no
|
||||
# building, the /menu hint) is felis-paper's LobbyGuard.
|
||||
set_prop max-players 200
|
||||
|
||||
# RCON is the control plane's write channel (spec §8 写=RCON): the operator probes it
|
||||
# for readiness and the player tally, and felis-api runs console/permission commands over
|
||||
# it. Paper only reads these three keys from server.properties, so the operator's injected
|
||||
|
||||
+135
-19
@@ -6,17 +6,20 @@
|
||||
# curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes
|
||||
#
|
||||
# Options:
|
||||
# --purge also drop the felis database and role, and delete /etc/felis and
|
||||
# /var/lib/felis (the database bundles, and anything an earlier keep-data
|
||||
# run set aside). Asks for the word "purge" unless --yes is given.
|
||||
# --purge also delete /etc/felis and /var/lib/felis (the database's cluster, the
|
||||
# database bundles, and anything an earlier keep-data run set aside), and
|
||||
# drop the felis database and role from a host PostgreSQL an earlier
|
||||
# release installed. Asks for the word "purge" unless --yes is given.
|
||||
# --keep-k3s leave k3s installed and remove only Felis's namespaces and CRD.
|
||||
# --remove-k3s run k3s's own uninstaller even when other workloads live in the cluster.
|
||||
# --no-backup skip the final database bundle keep-data mode takes first.
|
||||
# --yes do not ask.
|
||||
#
|
||||
# Keep-data mode (the default) first takes a database bundle (`felis db backup -label
|
||||
# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the
|
||||
# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive,
|
||||
# manual`) and stops if that fails. It leaves /etc/felis (the secrets, felis.toml,
|
||||
# offsite.env) and /var/lib/felis in place, and with it the felis database: felis-postgres
|
||||
# keeps its cluster in /var/lib/felis/postgres, stopped cleanly before k3s goes, and a
|
||||
# host PostgreSQL an earlier release installed keeps its copy. The world, archive,
|
||||
# registry and upload volumes live under k3s's storage directory, which k3s's uninstaller
|
||||
# deletes, so they are moved to /var/lib/felis/retained/k3s-storage-<UTC stamp> first; with
|
||||
# --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go.
|
||||
@@ -41,6 +44,12 @@ CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}"
|
||||
VELOCITY_USER="felis-velocity"
|
||||
DB_NAME="felis"
|
||||
DB_USER="felis"
|
||||
CONTROL_NS="felis"
|
||||
# The database bootstrap runs in k3s, its cluster on a hostPath under DATA_DIR, and the
|
||||
# marker bootstrap leaves once it moved a host PostgreSQL's felis database into it.
|
||||
PG_DEPLOYMENT="felis-postgres"
|
||||
PG_DATA_DIR="${DATA_DIR}/postgres"
|
||||
PG_MOVED_MARKER="${DATA_DIR}/postgres-moved"
|
||||
POD_CIDR="10.42.0.0/16"
|
||||
SERVICE_CIDR="10.43.0.0/16"
|
||||
FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}"
|
||||
@@ -51,7 +60,7 @@ FELIS_CRD="minecraftservers.felis.lolicon.best"
|
||||
# service that is already gone.
|
||||
FELIS_UNITS=(
|
||||
felis-db-backup.timer felis-watchdog.timer felis-offsite.timer felis-build-tools.timer felis-update-check.timer
|
||||
felis-db-backup.service felis-watchdog.service felis-offsite.service felis-build-tools.service felis-update-check.service
|
||||
felis-db-backup.service felis-watchdog.service felis-watchdog-failed.service felis-offsite.service felis-build-tools.service felis-update-check.service
|
||||
felis-velocity.service felis-nano.service cloudflared-felis.service
|
||||
felis-postgres-firewall.service
|
||||
)
|
||||
@@ -124,18 +133,19 @@ confirm() {
|
||||
print_plan() {
|
||||
log "this will remove from $(uname -n):"
|
||||
log " the felis-* systemd units, cloudflared-felis.service, the ${VELOCITY_USER} user,"
|
||||
log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld openings"
|
||||
log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld and ufw openings"
|
||||
case "$K3S_MODE" in
|
||||
remove) log " k3s, with everything in it (${K3S_BIN_DIR}/k3s-uninstall.sh)" ;;
|
||||
keep) log " Felis's namespaces (${FELIS_NAMESPACES[*]}) and the ${FELIS_CRD} CRD; k3s stays" ;;
|
||||
absent) ;;
|
||||
esac
|
||||
if [ "$PURGE" = 1 ]; then
|
||||
log " PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles"
|
||||
log " and anything set aside before), every world and archive, the Felis images and Docker's build cache"
|
||||
log " PURGE: ${STATE_DIR} (secrets), ${DATA_DIR} (the ${DB_NAME} database's cluster, the database"
|
||||
log " bundles and anything set aside before), the ${DB_NAME} database and role in a host PostgreSQL,"
|
||||
log " every world and archive, the Felis images and Docker's build cache"
|
||||
else
|
||||
[ "$BACKUP" = 1 ] && log " after a final database bundle into ${DATA_DIR}/db-backups"
|
||||
log " kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/"
|
||||
log " kept: ${STATE_DIR}, ${DATA_DIR} (the ${DB_NAME} database in ${PG_DATA_DIR}); the volumes move to ${RETAIN_DIR}/"
|
||||
fi
|
||||
}
|
||||
|
||||
@@ -234,6 +244,35 @@ remove_firewalld_rules() { # game-port nano-port
|
||||
fi
|
||||
}
|
||||
|
||||
# remove_ufw_rules takes back the ufw rules the installer added, found by their felis-
|
||||
# comments (configure_k3s_firewall and its neighbours in bootstrap.sh). The k3s ranges stay
|
||||
# when k3s does, and go with it or when it is already gone. ufw lists a rule's IPv6 twin
|
||||
# under its own number, and each delete renumbers the rules after it, so they go from the
|
||||
# highest number down.
|
||||
remove_ufw_rules() {
|
||||
command -v ufw >/dev/null 2>&1 || return 0
|
||||
local status nums n keep_k3s=1
|
||||
status="$(LC_ALL=C ufw status numbered 2>/dev/null)" || return 0
|
||||
[ "$(printf '%s\n' "$status" | head -n 1)" = "Status: active" ] || return 0
|
||||
[ "$K3S_MODE" = keep ] || keep_k3s=""
|
||||
nums="$(printf '%s\n' "$status" | awk -v keep_k3s="$keep_k3s" '
|
||||
match($0, /# felis-[a-z0-9-]+ *$/) {
|
||||
tag = substr($0, RSTART + 2)
|
||||
sub(/ +$/, "", tag)
|
||||
if (keep_k3s != "" && tag ~ /^felis-k3s-/) next
|
||||
if (match($0, /^\[ *[0-9]+\]/)) {
|
||||
n = substr($0, RSTART + 1, RLENGTH - 2)
|
||||
gsub(/ /, "", n)
|
||||
print n
|
||||
}
|
||||
}' | sort -rn)"
|
||||
[ -n "$nums" ] || return 0
|
||||
for n in $nums; do
|
||||
ufw --force delete "$n" >/dev/null
|
||||
done
|
||||
ok "ufw rules removed"
|
||||
}
|
||||
|
||||
# retain_volumes_in_cluster keeps every volume Felis's claims are bound to when the
|
||||
# namespaces go: local-path deletes a Delete-policy volume's directory with its claim.
|
||||
retain_volumes_in_cluster() {
|
||||
@@ -273,8 +312,18 @@ remove_from_cluster() {
|
||||
ok "Felis removed from the cluster; k3s stays"
|
||||
}
|
||||
|
||||
# stop_database_pod shuts felis-postgres down cleanly: k3s-killall.sh SIGKILLs every
|
||||
# container, and the cluster it leaves in PG_DATA_DIR is what a reinstall starts from.
|
||||
stop_database_pod() {
|
||||
kube -n "$CONTROL_NS" scale deployment "$PG_DEPLOYMENT" --replicas=0 >/dev/null 2>&1 || return 0
|
||||
kube -n "$CONTROL_NS" wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres \
|
||||
--timeout=120s >/dev/null 2>&1 \
|
||||
|| warn "${PG_DEPLOYMENT} did not stop within 2 minutes; its cluster recovers from its WAL on the next start"
|
||||
}
|
||||
|
||||
remove_k3s() {
|
||||
local stamp
|
||||
[ "$PURGE" = 1 ] || stop_database_pod
|
||||
if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then
|
||||
# k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a
|
||||
# world while it moves.
|
||||
@@ -313,6 +362,9 @@ remove_host_files() {
|
||||
fi
|
||||
rm -rf "$OPT_DIR"
|
||||
rm -f "$HOST_BIN" "${HOST_BIN}.new" "${HOST_BIN}.prev"
|
||||
# The release assets an install that stopped part way left for its rerun: downloads, not
|
||||
# data, so keep-data mode drops them too.
|
||||
rm -rf "${DATA_DIR}/artifacts"
|
||||
remove_cloudflared_binary
|
||||
if [ "$PURGE" = 0 ]; then
|
||||
# What describes the removed install goes; what a reinstall reuses stays. Without
|
||||
@@ -325,7 +377,8 @@ remove_host_files() {
|
||||
|
||||
as_postgres() { (cd / && runuser -u postgres -- "$@"); }
|
||||
|
||||
# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else.
|
||||
# remove_hba_block drops the block bootstrap heads pg_hba.conf with (the rules of an
|
||||
# install on the host server, or the lockout the move into k3s left), and nothing else.
|
||||
remove_hba_block() { # file
|
||||
local tmp
|
||||
tmp="$(mktemp)"
|
||||
@@ -340,23 +393,80 @@ remove_hba_block() { # file
|
||||
rm -f "$tmp"
|
||||
}
|
||||
|
||||
# check_database_purge runs before anything is removed. DROP ROLE refuses a role that
|
||||
# still owns a database or holds anything in one besides felis (the felis_pgint database
|
||||
# CONTRIBUTING.md has developers make for the PG contract tests, a grant made by hand),
|
||||
# and by the time purge_database runs the units and k3s are already gone. It lists what
|
||||
# holds the role instead, so the purge either runs to the end or not at all.
|
||||
check_database_purge() {
|
||||
[ "$PURGE" = 1 ] || return 0
|
||||
systemctl is-active --quiet postgresql 2>/dev/null || return 0
|
||||
local sql held err
|
||||
read -r -d '' sql <<SQL || true
|
||||
WITH r AS (SELECT oid FROM pg_roles WHERE rolname = '${DB_USER}'),
|
||||
f AS (SELECT oid FROM pg_database WHERE datname = '${DB_NAME}')
|
||||
SELECT DISTINCT CASE
|
||||
WHEN s.classid = 'pg_database'::regclass THEN 'database ' || (SELECT datname FROM pg_database WHERE oid = s.objid)
|
||||
WHEN s.classid = 'pg_tablespace'::regclass THEN 'tablespace ' || (SELECT spcname FROM pg_tablespace WHERE oid = s.objid)
|
||||
ELSE 'objects in database ' || (SELECT datname FROM pg_database WHERE oid = s.dbid)
|
||||
END || CASE s.deptype WHEN 'o' THEN ' (owned)' ELSE ' (privileges)' END
|
||||
FROM pg_shdepend s
|
||||
WHERE s.refclassid = 'pg_authid'::regclass AND s.refobjid = (SELECT oid FROM r)
|
||||
AND s.dbid IS DISTINCT FROM (SELECT oid FROM f)
|
||||
AND NOT (s.classid = 'pg_database'::regclass AND s.objid IS NOT DISTINCT FROM (SELECT oid FROM f))
|
||||
ORDER BY 1;
|
||||
SQL
|
||||
err="$(mktemp)"
|
||||
if ! held="$(as_postgres psql -v ON_ERROR_STOP=1 -tAq 2>"$err" <<<"$sql")"; then
|
||||
held="$(cat "$err")"
|
||||
rm -f "$err"
|
||||
die "could not ask PostgreSQL what the ${DB_USER} role still holds, so nothing was removed: ${held}"
|
||||
fi
|
||||
rm -f "$err"
|
||||
[ -n "$held" ] || return 0
|
||||
die "the ${DB_USER} role still holds $(printf '%s' "$held" | paste -sd ';' - | sed 's/;/; /g'), so DROP ROLE would fail halfway through the purge; nothing was removed. Hand them to postgres first (sudo -u postgres psql -c 'ALTER DATABASE <name> OWNER TO postgres', or REASSIGN OWNED BY ${DB_USER} TO postgres; DROP OWNED BY ${DB_USER}; inside that database), or rerun without --purge"
|
||||
}
|
||||
|
||||
# purge_database drops the felis database and role from a host PostgreSQL an earlier
|
||||
# release installed; the cluster felis-postgres runs goes with DATA_DIR (purge_state). That
|
||||
# host server either still serves the platform, or the move into felis-postgres stopped it
|
||||
# with the pre-move copy left in it for a rollback: a purge takes that copy too and leaves
|
||||
# the server stopped. The units and k3s are gone by now, so a failure here is a warning
|
||||
# with the commands to finish by hand, and the purge goes on.
|
||||
purge_database() {
|
||||
[ "$PURGE" = 1 ] || return 0
|
||||
local hba started=0
|
||||
if ! systemctl is-active --quiet postgresql 2>/dev/null; then
|
||||
warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it"
|
||||
return 0
|
||||
[ -e "$PG_MOVED_MARKER" ] && systemctl cat postgresql >/dev/null 2>&1 || return 0
|
||||
if ! systemctl start postgresql >/dev/null 2>&1; then
|
||||
warn "could not start the host PostgreSQL, so its copy of the ${DB_NAME} database from before the move into k3s stays in it; drop it once it runs: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}"
|
||||
return 0
|
||||
fi
|
||||
started=1
|
||||
fi
|
||||
local hba
|
||||
hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
|
||||
as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
|
||||
if ! as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
|
||||
SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = '${DB_NAME}' AND pid <> pg_backend_pid();
|
||||
DROP DATABASE IF EXISTS ${DB_NAME};
|
||||
DROP ROLE IF EXISTS ${DB_USER};
|
||||
ALTER SYSTEM RESET listen_addresses;
|
||||
SQL
|
||||
[ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba"
|
||||
systemctl restart postgresql
|
||||
ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again"
|
||||
then
|
||||
[ "$started" = 0 ] || systemctl stop postgresql >/dev/null 2>&1 || true
|
||||
warn "could not drop the ${DB_NAME} database and role from the host PostgreSQL; drop them by hand: sudo -u postgres dropdb ${DB_NAME}; sudo -u postgres dropuser ${DB_USER}"
|
||||
return 0
|
||||
fi
|
||||
if [ -n "$hba" ] && [ -f "$hba" ]; then
|
||||
remove_hba_block "$hba"
|
||||
rm -f "${hba}.pre-pg-move"
|
||||
fi
|
||||
if [ "$started" = 1 ]; then
|
||||
systemctl stop postgresql
|
||||
ok "the host PostgreSQL's copy of '${DB_NAME}' from before the move into k3s dropped; the server stays stopped"
|
||||
else
|
||||
systemctl restart postgresql
|
||||
ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again"
|
||||
fi
|
||||
}
|
||||
|
||||
purge_images() {
|
||||
@@ -385,6 +495,10 @@ purge_state() {
|
||||
&& cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)"
|
||||
if [ -n "$cred" ]; then rm -f "$cred"; fi
|
||||
rm -f "$TUNNEL_CONFIG"
|
||||
# The file-context rule bootstrap gave the database's cluster directory.
|
||||
if command -v semanage >/dev/null 2>&1; then
|
||||
semanage fcontext -d "${PG_DATA_DIR}(/.*)?" >/dev/null 2>&1 || true
|
||||
fi
|
||||
rm -rf "$STATE_DIR" "$DATA_DIR"
|
||||
ok "${STATE_DIR} and ${DATA_DIR} removed"
|
||||
}
|
||||
@@ -395,6 +509,7 @@ main() {
|
||||
[ -e "$STATE_DIR" ] || [ -e "$HOST_BIN" ] || [ -e "$OPT_DIR" ] \
|
||||
|| die "no Felis install here (${STATE_DIR}, ${HOST_BIN} and ${OPT_DIR} are all absent)"
|
||||
decide_k3s
|
||||
check_database_purge
|
||||
print_plan
|
||||
confirm
|
||||
final_backup
|
||||
@@ -410,6 +525,7 @@ main() {
|
||||
esac
|
||||
remove_nft_tables
|
||||
remove_firewalld_rules "$game" "$nano"
|
||||
remove_ufw_rules
|
||||
remove_host_files
|
||||
purge_database
|
||||
purge_images
|
||||
@@ -418,7 +534,7 @@ main() {
|
||||
if [ "$PURGE" = 1 ]; then
|
||||
ok "Felis is gone from this host"
|
||||
else
|
||||
ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}"
|
||||
ok "Felis is removed; the data stays in ${STATE_DIR} and ${DATA_DIR}, the ${DB_NAME} database in ${PG_DATA_DIR}"
|
||||
log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\""
|
||||
fi
|
||||
}
|
||||
|
||||
+174
-4
@@ -44,6 +44,10 @@ fresh_host() {
|
||||
printf 'x\n' > "$root/h/etc/$f"
|
||||
done
|
||||
printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat"
|
||||
mkdir -p "$root/h/data/artifacts"
|
||||
printf 'bundle\n' > "$root/h/data/artifacts/felis-image-game-linux-amd64.tar.partial"
|
||||
mkdir -p "$root/h/data/postgres/18/docker"
|
||||
printf '18\n' > "$root/h/data/postgres/18/docker/PG_VERSION"
|
||||
for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do
|
||||
printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b"
|
||||
chmod +x "$root/h/bin/$b"
|
||||
@@ -72,11 +76,19 @@ run_uninstall() {
|
||||
. "$0"
|
||||
calls="$ROOT/calls"
|
||||
id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; }
|
||||
# PG_HOST: the host PostgreSQL is active (the default), stopped, or not installed (none);
|
||||
# PG_START=fail for one that will not start.
|
||||
systemctl() {
|
||||
echo "SYSTEMCTL $*" >> "$calls"
|
||||
case "$*" in "is-active --quiet firewalld") return 1 ;; esac
|
||||
return 0
|
||||
case "$*" in
|
||||
"is-active --quiet firewalld") return 1 ;;
|
||||
"is-active --quiet postgresql") [ "${PG_HOST:-active}" = active ] ;;
|
||||
"cat postgresql") [ "${PG_HOST:-active}" != none ] ;;
|
||||
"start postgresql") [ "${PG_START:-}" != fail ] ;;
|
||||
*) return 0 ;;
|
||||
esac
|
||||
}
|
||||
semanage() { echo "SEMANAGE $*" >> "$calls"; }
|
||||
kube() {
|
||||
echo "KUBE $*" >> "$calls"
|
||||
case "$*" in
|
||||
@@ -89,7 +101,25 @@ run_uninstall() {
|
||||
shift 3
|
||||
case "$*" in
|
||||
*"SHOW hba_file"*) echo "$ROOT/h/hba.conf" ;;
|
||||
*) echo "PSQL $* $(cat)" >> "$calls" ;;
|
||||
*)
|
||||
sql="$(cat)"
|
||||
case "$sql" in
|
||||
# What the felis role still holds: PG_HELD, one line each; PG_CHECK=fail for a
|
||||
# server that refuses the query.
|
||||
*pg_shdepend*)
|
||||
echo "PSQL-CHECK" >> "$calls"
|
||||
[ "${PG_CHECK:-}" = fail ] && { echo "psql: error: connection refused" >&2; return 2; }
|
||||
[ -z "${PG_HELD:-}" ] || printf "%s\n" "$PG_HELD" ;;
|
||||
*) echo "PSQL $* $sql" >> "$calls"; [ "${PG_DROP:-}" != fail ] ;;
|
||||
esac ;;
|
||||
esac
|
||||
}
|
||||
# UFW_STATUS: what `ufw status numbered` answers; ufw translates it outside the C locale.
|
||||
ufw() {
|
||||
case "$*" in
|
||||
"status numbered")
|
||||
if [ "${LC_ALL:-}" = C ]; then printf "%s\n" "${UFW_STATUS:-Status: inactive}"; else echo "状态:激活"; fi ;;
|
||||
*) echo "UFW $*" >> "$calls" ;;
|
||||
esac
|
||||
}
|
||||
docker() { echo "DOCKER $*" >> "$calls"; }
|
||||
@@ -116,6 +146,9 @@ expect "the volumes are set aside before k3s deletes them" "k3s-storage-" "$kept
|
||||
[ ! -e "$root/h/etc/bootstrap.done" ] && [ ! -e "$root/h/etc/velocity.fingerprint" ] \
|
||||
&& echo "PASS the markers of the removed install go, so a reinstall starts fresh" \
|
||||
|| { echo "FAIL bootstrap.done or the proxy fingerprint was left"; fails=$((fails + 1)); }
|
||||
[ ! -e "$root/h/data/artifacts" ] && [ -d "$root/h/data/postgres" ] \
|
||||
&& echo "PASS keep-data drops the downloaded release assets and keeps the database" \
|
||||
|| { echo "FAIL keep-data left the release-asset cache or took the database: $(ls "$root/h/data")"; fails=$((fails + 1)); }
|
||||
refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
|
||||
[ ! -e "$root/h/opt" ] && [ ! -e "$root/h/bin/felis" ] \
|
||||
&& echo "PASS /opt/felis and the host binary are removed" \
|
||||
@@ -125,6 +158,15 @@ refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
|
||||
expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls"
|
||||
expect "the velocity user is removed" "USERDEL felis-velocity" "$calls"
|
||||
expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out"
|
||||
[ -f "$root/h/data/postgres/18/docker/PG_VERSION" ] && echo "PASS the database's cluster stays for the reinstall" \
|
||||
|| { echo "FAIL keep-data removed the database's cluster"; fails=$((fails + 1)); }
|
||||
stop_at="$(grep -n 'KUBE -n felis scale deployment felis-postgres --replicas=0' "$root/calls" | head -n 1 | cut -d: -f1)"
|
||||
kill_at="$(grep -n 'RUN k3s-killall.sh' "$root/calls" | head -n 1 | cut -d: -f1)"
|
||||
[ -n "$stop_at" ] && [ -n "$kill_at" ] && [ "$stop_at" -lt "$kill_at" ] \
|
||||
&& echo "PASS the database shuts down cleanly before k3s-killall.sh kills what is left" \
|
||||
|| { echo "FAIL felis-postgres was not stopped before k3s-killall.sh: $calls"; fails=$((fails + 1)); }
|
||||
expect "and the uninstall waits for it to stop" "KUBE -n felis wait --for=delete pod -l app.kubernetes.io/name=felis,app.kubernetes.io/component=postgres" "$calls"
|
||||
refute "keep-data keeps the cluster directory's SELinux rule" "SEMANAGE" "$calls"
|
||||
|
||||
# --- a failed final bundle stops everything ------------------------------------------------
|
||||
fresh_host
|
||||
@@ -161,6 +203,7 @@ fresh_host
|
||||
out="$(run_uninstall "default felis minecraft" --purge --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
refute "purge takes no bundle" "db backup" "$calls"
|
||||
expect "purge asks what the role holds first" "PSQL-CHECK" "$calls"
|
||||
expect "purge drops the database" "DROP DATABASE IF EXISTS felis;" "$calls"
|
||||
expect "purge drops the role" "DROP ROLE IF EXISTS felis;" "$calls"
|
||||
expect "purge puts listen_addresses back" "ALTER SYSTEM RESET listen_addresses;" "$calls"
|
||||
@@ -178,6 +221,71 @@ case "$hba" in
|
||||
*) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls"
|
||||
expect "purge drops the cluster directory's SELinux rule" "SEMANAGE fcontext -d $root/h/data/postgres(/.*)?" "$calls"
|
||||
refute "purge leaves k3s's pods to k3s-uninstall.sh" "scale deployment felis-postgres" "$calls"
|
||||
|
||||
# --- purge after the move into felis-postgres ---------------------------------------------
|
||||
# The move stopped the host server with the felis database left in it for a rollback.
|
||||
moved_host() { fresh_host; printf 'moved\n' > "$root/h/data/postgres-moved"; printf 'saved\n' > "$root/h/hba.conf.pre-pg-move"; }
|
||||
moved_host
|
||||
out="$(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
expect "purge starts the stopped host server to drop the pre-move copy" "SYSTEMCTL start postgresql" "$calls"
|
||||
expect "and drops it" "DROP DATABASE IF EXISTS felis;" "$calls"
|
||||
expect "the server stays stopped afterwards" "SYSTEMCTL stop postgresql" "$calls"
|
||||
refute "and is not restarted" "SYSTEMCTL restart postgresql" "$calls"
|
||||
refute "the lockout leaves pg_hba.conf" "FELIS MANAGED" "$(cat "$root/h/hba.conf")"
|
||||
[ ! -e "$root/h/hba.conf.pre-pg-move" ] && echo "PASS the pg_hba.conf the move saved goes with the copy" \
|
||||
|| { echo "FAIL purge left pg_hba.conf.pre-pg-move"; fails=$((fails + 1)); }
|
||||
moved_host
|
||||
out="$(PG_HOST=stopped PG_START=fail run_uninstall "default felis minecraft" --purge --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
expect "a host server that will not start is named" "could not start the host PostgreSQL" "$out"
|
||||
refute "so nothing is dropped" "DROP DATABASE" "$calls"
|
||||
[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS and the purge goes on" \
|
||||
|| { echo "FAIL the purge stopped at the host server"; fails=$((fails + 1)); }
|
||||
moved_host
|
||||
out="$(PG_HOST=stopped PG_DROP=fail run_uninstall "default felis minecraft" --purge --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
expect "a drop that fails says how to finish by hand" "sudo -u postgres dropdb felis" "$out"
|
||||
expect "and stops the server it started" "SYSTEMCTL stop postgresql" "$calls"
|
||||
[ ! -e "$root/h/data" ] && echo "PASS a failed drop does not stop the purge halfway" \
|
||||
|| { echo "FAIL the purge stopped at a failed drop"; fails=$((fails + 1)); }
|
||||
fresh_host
|
||||
(PG_HOST=stopped run_uninstall "default felis minecraft" --purge --yes >/dev/null) # a subshell: sh keeps a prefix assignment to a function
|
||||
refute "a stopped host server the move never touched is left alone" "SYSTEMCTL start postgresql" "$(cat "$root/calls")"
|
||||
moved_host
|
||||
(PG_HOST=none run_uninstall "default felis minecraft" --purge --yes >/dev/null) # a subshell: sh keeps a prefix assignment to a function
|
||||
refute "a host server removed since the move is not started" "SYSTEMCTL start postgresql" "$(cat "$root/calls")"
|
||||
|
||||
# --- a purge DROP ROLE would refuse -------------------------------------------------------
|
||||
# The VM drill: the PG contract tests' felis_pgint was owned by felis, the purge removed the
|
||||
# units and k3s, then stopped at DROP ROLE with half the host gone.
|
||||
untouched() { # label
|
||||
[ -d "$root/h/opt" ] && [ -f "$root/h/units/felis-velocity.service" ] && [ -d "$root/h/etc" ] \
|
||||
&& ! grep -q "k3s-uninstall.sh\|DROP DATABASE\|SYSTEMCTL disable" "$root/calls" \
|
||||
&& echo "PASS $1" \
|
||||
|| { echo "FAIL $1: $(cat "$root/calls")"; fails=$((fails + 1)); }
|
||||
}
|
||||
fresh_host
|
||||
out="$(PG_HELD="database felis_pgint (owned)
|
||||
objects in database shop (privileges)" run_uninstall "default felis minecraft" --purge --yes)"
|
||||
expect "a purge the role cannot survive is refused" "the felis role still holds database felis_pgint (owned); objects in database shop (privileges)" "$out"
|
||||
expect "with the way to hand the database over" "ALTER DATABASE <name> OWNER TO postgres" "$out"
|
||||
untouched "nothing is removed when DROP ROLE would fail"
|
||||
fresh_host
|
||||
out="$(PG_HELD="database felis_pgint (owned)" CONFIRM_TTY="$root/no-tty/x" run_uninstall "default felis minecraft" --purge)"
|
||||
expect "the check comes before the plan and the prompt" "the felis role still holds database felis_pgint" "$out"
|
||||
refute "so nobody confirms a purge that cannot finish" "this will remove" "$out"
|
||||
fresh_host
|
||||
out="$(PG_CHECK=fail run_uninstall "default felis minecraft" --purge --yes)"
|
||||
expect "a server that cannot be asked stops the purge" "could not ask PostgreSQL what the felis role still holds, so nothing was removed: psql: error: connection refused" "$out"
|
||||
untouched "nothing is removed when the check cannot run"
|
||||
fresh_host
|
||||
PG_HELD="database felis_pgint (owned)" run_uninstall "default felis minecraft" --yes >/dev/null
|
||||
calls="$(cat "$root/calls")"
|
||||
refute "keep-data drops no role, so it asks nothing" "PSQL-CHECK" "$calls"
|
||||
expect "and goes on" "RUN k3s-uninstall.sh" "$calls"
|
||||
|
||||
# --- the pieces read before they are removed ---------------------------------------------
|
||||
fresh_host
|
||||
@@ -213,9 +321,71 @@ for u in $(sed -n 's|^[A-Z_]*="/etc/systemd/system/\([^"]*\)"$|\1|p' "$(dirname
|
||||
expect "the uninstaller removes $u" " $u" " $(printf '%s' "$units" | tr '\n' ' ')"
|
||||
done
|
||||
|
||||
# --- ufw: the rules bootstrap added, by the comments bootstrap gives them ------------------
|
||||
# The listing is what `ufw status numbered` prints once bootstrap has run: a rule of the
|
||||
# operator's own first (commented, as an operator may), then each felis- rule bootstrap.sh can
|
||||
# add and its IPv6 twin.
|
||||
tags="$(grep -o 'comment felis-[a-z0-9-]*' "$(dirname "$US")/bootstrap.sh" | awk '{ print $2 }' | sort -u)"
|
||||
case " $(printf '%s ' $tags)" in
|
||||
*" felis-k3s-pods "*" felis-proxy "*) echo "PASS bootstrap tags its ufw rules" ;;
|
||||
*) echo "FAIL bootstrap.sh adds no felis-k3s-pods and felis-proxy ufw rules: <$tags>"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
listing="Status: active
|
||||
|
||||
To Action From
|
||||
-- ------ ----
|
||||
[ 1] 22/tcp ALLOW IN Anywhere # ssh"
|
||||
n=1
|
||||
for twin in "" " (v6)"; do
|
||||
for t in $tags; do
|
||||
n=$((n + 1))
|
||||
listing="${listing}
|
||||
$(printf '[%2d] Rule%-22s ALLOW IN Anywhere%-19s # %s' "$n" "$twin" "$twin" "$t")"
|
||||
done
|
||||
done
|
||||
listing="${listing}
|
||||
$(printf '[%2d] 22/tcp (v6) ALLOW IN Anywhere (v6)' "$((n + 1))")"
|
||||
# numbers <listing> <regex>: the rule numbers whose line matches, highest first.
|
||||
numbers() { printf '%s\n' "$1" | grep -E "$2" | sed 's/^\[ *\([0-9]*\)\].*/\1/' | sort -rn | paste -sd ' ' -; }
|
||||
deletes() { grep '^UFW --force delete' "$root/calls" | awk '{ print $4 }' | paste -sd ' ' -; }
|
||||
|
||||
fresh_host
|
||||
out="$(UFW_STATUS="$listing" run_uninstall "default felis minecraft" --yes)"
|
||||
want="$(numbers "$listing" '# felis-')"
|
||||
[ -n "$want" ] && [ "$(deletes)" = "$want" ] \
|
||||
&& echo "PASS a Felis-only cluster's uninstall deletes every felis- ufw rule, highest first" \
|
||||
|| { echo "FAIL a Felis-only cluster: deleted <$(deletes)>, want <$want>"; fails=$((fails + 1)); }
|
||||
case " $(deletes) " in
|
||||
*" 1 "* | *" $((n + 1)) "*) echo "FAIL the operator's own ufw rules were deleted: $(deletes)"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS the operator's own ufw rules stay" ;;
|
||||
esac
|
||||
expect " and says so" "ufw rules removed" "$out"
|
||||
|
||||
fresh_host
|
||||
UFW_STATUS="$listing" run_uninstall "default kube-system felis minecraft felis-build shop" --yes >/dev/null
|
||||
[ "$(deletes)" = "$(numbers "$listing" '# felis-' | tr ' ' '\n' | grep -vxE "$(numbers "$listing" '# felis-k3s-' | tr ' ' '|')" | paste -sd ' ' -)" ] \
|
||||
&& echo "PASS a k3s that stays keeps its pod and service ranges in ufw" \
|
||||
|| { echo "FAIL a kept k3s: deleted <$(deletes)>, listing:"; echo "$listing"; fails=$((fails + 1)); }
|
||||
|
||||
fresh_host
|
||||
UFW_STATUS="Status: inactive" run_uninstall "default felis minecraft" --yes >/dev/null
|
||||
[ -z "$(deletes)" ] && echo "PASS an inactive ufw is left alone" \
|
||||
|| { echo "FAIL an inactive ufw: deleted <$(deletes)>"; fails=$((fails + 1)); }
|
||||
|
||||
# The database's cluster and the move's marker are where bootstrap put them, or keep-data
|
||||
# and purge act on a directory that is not there.
|
||||
paths="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; printf "%s %s\n" "$PG_DATA_DIR" "$PG_MOVED_MARKER"' "$US")"
|
||||
bspaths="$(sed -n 's/^PG_DATA_DIR="\(.*\)"$/\1/p; s/^PG_MOVED_MARKER="\(.*\)"$/\1/p' "$(dirname "$US")/bootstrap.sh" | paste -sd ' ' -)"
|
||||
[ -n "$bspaths" ] && [ "$paths" = "$bspaths" ] && echo "PASS the database paths agree with bootstrap" \
|
||||
|| { echo "FAIL uninstall's database paths <$paths> differ from bootstrap's <$bspaths>"; fails=$((fails + 1)); }
|
||||
bscache="$(sed -n 's/^ARTIFACT_CACHE="\(.*\)"$/\1/p' "$(dirname "$US")/bootstrap.sh")"
|
||||
uscache="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; printf "%s/artifacts\n" "$DATA_DIR"' "$US")"
|
||||
[ -n "$bscache" ] && [ "$uscache" = "$bscache" ] && echo "PASS the release-asset cache is where bootstrap keeps it" \
|
||||
|| { echo "FAIL uninstall removes <$uscache>, bootstrap caches release assets in <$bscache>"; fails=$((fails + 1)); }
|
||||
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS"
|
||||
else
|
||||
echo "$fails FAILED"
|
||||
exit 1
|
||||
fi
|
||||
exit "$fails"
|
||||
+2046
-84
File diff suppressed because it is too large.
Load diff
+522
-39
@@ -12,13 +12,16 @@ Evidence tags follow troubleshooting.md: **[VM-VERIFIED]** was run on a real hos
|
||||
## 1. Supported hosts
|
||||
|
||||
`deploy/bootstrap.sh` provisions a single node. It needs systemd, root, and one of the
|
||||
package managers below; everything else (Docker, k3s, PostgreSQL, the JRE, cloudflared)
|
||||
it installs.
|
||||
package managers below; everything else (k3s, the JRE, cloudflared, and Docker when an image
|
||||
has to be built on the host; see "Where the binary and the images come from" below) it
|
||||
installs.
|
||||
PostgreSQL runs inside k3s as the `felis-postgres` Deployment, from the official image the
|
||||
release pins by digest, with its data on the host in `/var/lib/felis/postgres`.
|
||||
|
||||
| OS family | Package manager | Architectures | Status |
|
||||
|---|---|---|---|
|
||||
| CentOS Stream 9 (firewalld active, PostgreSQL 13) | dnf | aarch64 | **[VM-VERIFIED]** install, same-version rerun, upgrade, uninstall and reinstall |
|
||||
| Ubuntu 24.04 LTS | apt | x86_64 | **[CI]** fresh install, same-commit rerun, and upgrade from the newest release to the pushed commit |
|
||||
| CentOS Stream 9 (firewalld active, SELinux enforcing) | dnf | aarch64 | **[VM-VERIFIED]** fresh install from release assets and its rerun, upgrade from v0.1.0 (moving the database off the host PostgreSQL 13 into felis-postgres), uninstall and reinstall |
|
||||
| Ubuntu 24.04 LTS | apt | x86_64 | **[CI]** fresh install and same-commit rerun from the pushed commit's release assets; the README's one-line install as a new host runs it (the newest release's own assets); upgrade from the newest release, installed from its assets by its own installer and seeded with rows in twelve tables, onto them, every seeded row read back unchanged; the on-host build weekly |
|
||||
| RHEL / Rocky / Alma 9, Fedora | dnf | x86_64, aarch64 | [CODE-ONLY] same code path as CentOS Stream |
|
||||
| Debian 12, other Ubuntu releases | apt | x86_64, aarch64 | [CODE-ONLY] |
|
||||
| openSUSE Leap / Tumbleweed | zypper | x86_64, aarch64 | [CODE-ONLY] |
|
||||
@@ -34,11 +37,66 @@ cloudflared is left as it is, see §4):
|
||||
| Temurin JRE (Velocity) | 25, patch build pinned | `FELIS_JRE_VERSION`, sha256 per architecture |
|
||||
| Go (nano builds) | 1.26.8 | `GO_PINNED_VERSION`, sha256 per architecture |
|
||||
| Minecraft / Limbo / Paper / Velocity / LuckPerms | `deploy/game-stack.lock` | §15b |
|
||||
| PostgreSQL | the distribution's package | 13 and 18 are exercised by the `pgint` CI job |
|
||||
| PostgreSQL | 18.6, the official `postgres` image by digest | `POSTGRES_IMAGE` in `bootstrap.sh`, `defaultPostgresImage` in `internal/platform` |
|
||||
|
||||
32-bit hosts are not supported: there is no k3s, JRE or Go build the installer will fetch
|
||||
for them.
|
||||
|
||||
Before it changes anything the installer checks the host and reports every problem at
|
||||
once, then stops with nothing touched **[SH-TESTED]**:
|
||||
|
||||
- the architecture, systemd as init, and the memory cgroup controller k3s needs;
|
||||
- RAM: under 1.75 GiB is refused (a "2 GB" VPS passes), under 3.5 GiB is a warning;
|
||||
- free disk on each filesystem it writes to, summed when they share one: on a bare host
|
||||
about 17 GiB installing a release, 15 GiB from `FELIS_ARTIFACT_DIR` and 23 GiB when it
|
||||
builds the images itself; 7 GiB for a rerun; a directory that already holds data
|
||||
(Docker's cache, a reused k3s) counting at the rerun size; a filesystem that would end
|
||||
over 85%, where k3s starts deleting cached images, is a warning;
|
||||
- the ports it will listen on: the game port, the panel NodePort, k3s's 6443/6444 and
|
||||
10248–10259 and the registry's loopback 5000. A port held by the installer's own
|
||||
proxy or k3s is a rerun and passes;
|
||||
- another Kubernetes (kubelet, RKE2, k0s, MicroK8s) or a k3s agent on the host;
|
||||
- the node address or a routed network inside k3s's `10.42.0.0/16` and `10.43.0.0/16`
|
||||
(a Docker network there is the usual case); a wider route such as a `10.0.0.0/8` VPN
|
||||
is a warning;
|
||||
- HTTPS to the hosts it downloads from: GitHub and PaperMC's download API always, Docker
|
||||
Hub when it builds images on the host, Rancher's RPM repository where k3s's installer
|
||||
adds it (the list is under "Where the binary and the images come from"). A host counts
|
||||
as reachable once a TLS handshake with it completes, and each gets three tries two
|
||||
seconds apart. Installing a release, an
|
||||
unreachable Docker Hub is a warning (it is needed only if an asset turns out unusable);
|
||||
from `FELIS_ARTIFACT_DIR` it is not checked.
|
||||
|
||||
`FELIS_PREFLIGHT=warn` reports the same problems as warnings and installs anyway, for a
|
||||
host the checks misjudge.
|
||||
|
||||
A host firewall is opened, never turned off **[SH-TESTED]**. With firewalld active the
|
||||
installer adds the panel NodePort, the game port and 6443, and puts k3s's pod and service
|
||||
ranges in the trusted zone. With ufw enabled (common on Ubuntu and Debian, and enabled in
|
||||
the CI install **[CI]**) it admits `10.42.0.0/16` and `10.43.0.0/16`, the panel NodePort and
|
||||
the game port, each rule commented `felis-…`; 6443 stays closed to the network, since pods
|
||||
reach the API server from their own range. Felis-nano opens its port to
|
||||
`FELIS_NANO_PROXY_CIDR` alone in either. `uninstall.sh` removes these again, the k3s ranges
|
||||
only when k3s goes too. Any other firewall in front of the host must admit the same:
|
||||
dropped pod traffic shows up as the first rollout timing out ("control-plane rollout did
|
||||
not complete").
|
||||
|
||||
Two things the host must keep for as long as the install lives:
|
||||
|
||||
- **Its address.** The install is bound to the IPv4 address it was made on (the k3s
|
||||
node, the network policies, the panel certificate and the default nip.io domain all
|
||||
carry it). Give the host a static address or a DHCP reservation before installing;
|
||||
the installer warns when the address is a lease, and the watchdog reports
|
||||
`host-address` when the host loses it (troubleshooting §13c). The k3s node name is
|
||||
pinned at install time, so a hostname change is harmless.
|
||||
- **A synchronized clock.** The installer turns NTP on (chrony where nothing else can)
|
||||
and the watchdog reports a clock that stays unsynchronized. Allow outbound UDP 123,
|
||||
or set `FELIS_MANAGE_TIME_SYNC=0` on a host whose clock is kept another way.
|
||||
|
||||
The installer also makes the system journal persistent (capped at
|
||||
`FELIS_JOURNAL_MAX_USE`, default 1G; `FELIS_MANAGE_JOURNAL=0` skips it) and writes the
|
||||
admin kubeconfig `/etc/rancher/k3s/k3s.yaml` root-only: run `sudo k3s kubectl`.
|
||||
|
||||
One node is the whole supported shape. A world volume is a ReadWriteOnce claim on the
|
||||
node's local-path storage, so a game server's pod is pinned to the node that first
|
||||
scheduled it and cannot move when that node fails; the operator and felis-api each run
|
||||
@@ -48,6 +106,78 @@ and gains no failover. A multi-node shape would need, at least, storage that can
|
||||
pod to another node and leader election in felis-operator (controller-runtime's
|
||||
`LeaderElection`) so a second replica can stand by.
|
||||
|
||||
### Where the binary and the images come from
|
||||
|
||||
A release install (the default channel, and the setup console) takes everything Felis
|
||||
builds from that release's assets, each checked against the release's `SHA256SUMS` before
|
||||
it is used: the `felis` binary, the control-plane image, the limbo, lobby and paper images,
|
||||
the registry and PostgreSQL images (at the digests `bootstrap.sh` pins), and
|
||||
`felis-velocity.jar`. The images go into k3s's containerd with `k3s ctr images import` and
|
||||
from there into the in-cluster registry, so the host needs no Docker, Gradle, Go or Docker
|
||||
Hub for them. k3s's own images come from k3s's GitHub release
|
||||
(`k3s-airgap-images-<arch>.tar.zst`, checked against k3s's sha256 list) before k3s first
|
||||
starts. An upgrade downloads only the image tars holding an image the host lacks; they wait
|
||||
in `/var/lib/felis/artifacts` until the registry has the images, and are deleted then.
|
||||
`deploy/build-release-artifacts.sh` documents every asset. The decisions are **[SH-TESTED]**.
|
||||
Installing from the assets is **[VM-VERIFIED]** on CentOS Stream 9 aarch64 through
|
||||
`FELIS_ARTIFACT_DIR`: a fresh install and an upgrade over a release that built on the host
|
||||
pulled no image and built nothing, and a rerun imported and uploaded nothing. Downloading them from a
|
||||
release is [SH-TESTED] until a release publishes assets.
|
||||
The release is the newest one unless `FELIS_RELEASE=<tag>` names an earlier one, which
|
||||
installs from that release's assets the same way: the way back after a bad upgrade
|
||||
(troubleshooting §16, "Roll back an upgrade that broke the database"), with the installer
|
||||
read at that tag.
|
||||
|
||||
The installer builds on the host instead, installing Docker for it and stopping Docker once
|
||||
the images are in the registry, when:
|
||||
|
||||
- the source is not a release: `FELIS_VERSION_BOOTSTRAP=dev`, a pinned `FELIS_REF`, or
|
||||
`FELIS_SKIP_FETCH`;
|
||||
- `FELIS_GAME_STACK=latest`, for the login, lobby and paper images (the rest still come
|
||||
from the release);
|
||||
- the release publishes no `SHA256SUMS` (one cut before release assets existed, or still
|
||||
uploading), or an asset is missing, fails its checksum or is malformed. Only that image is
|
||||
built (the registry and PostgreSQL images are pulled from Docker Hub instead), and a
|
||||
warning names it; troubleshooting §15c lists the messages. Each download is tried
|
||||
three times first, and a host without the room for the build stops before installing
|
||||
Docker (troubleshooting §15c).
|
||||
|
||||
`FELIS_ARTIFACT_DIR=<absolute path>` installs from a directory instead of the release: a
|
||||
release's assets downloaded there (every `felis-*` file and `SHA256SUMS`), or the directory
|
||||
`deploy/build-release-artifacts.sh <version> <dir>` wrote. Nothing of Felis's own is
|
||||
downloaded or built (except the game images under `FELIS_GAME_STACK=latest`, which no release
|
||||
ships), so an asset the directory lacks, or one failing its checksum, stops the install. It
|
||||
cannot be combined with `FELIS_REF` or `FELIS_SKIP_FETCH`, which name a source too.
|
||||
|
||||
The rest of the host's software still downloads, so the host needs outbound HTTPS to these,
|
||||
directly or through `https_proxy`. A host with no outbound access cannot be installed yet
|
||||
**[SH-TESTED]**:
|
||||
|
||||
| Host | What comes from it |
|
||||
|---|---|
|
||||
| `github.com`, and the githubusercontent.com hosts its release downloads redirect to | k3s and its images (`k3s-airgap-images-<arch>.tar.zst`), cloudflared, the Temurin JRE, ViaVersion, ViaBackwards and ViaRewind |
|
||||
| `raw.githubusercontent.com` | k3s's install script, until k3s is installed |
|
||||
| `rpm.rancher.io` | k3s-selinux, which k3s's install script adds on an SELinux host of the Red Hat or SUSE family (CentOS Stream, RHEL, Rocky, Alma, Fedora, openSUSE Leap), until k3s is installed |
|
||||
| `fill-data.papermc.io` | the Velocity jar, unless `FELIS_VELOCITY_FORK_JAR` supplies one |
|
||||
| the distribution's package mirrors | the base packages (CA certificates, OpenSSL, curl and tar where missing), and container-selinux beside k3s-selinux |
|
||||
|
||||
Preflight probes each named host above before it changes anything and lists every one it
|
||||
cannot reach in one refusal (`cannot reach … over HTTPS`); the package manager reports its
|
||||
own mirrors. An override adds a host the download itself tries: `FELIS_JRE_VERSION` reads
|
||||
`api.adoptium.net`, a `FELIS_VELOCITY_VERSION` other than the pinned one reads
|
||||
`fill.papermc.io`, and `FELIS_GAME_STACK=latest` builds its images on the host from Docker
|
||||
Hub (which preflight probes), PaperMC, Limbo's CI and LuckPerms.
|
||||
|
||||
```
|
||||
# on a machine with access: the release's assets for the host's architecture
|
||||
gh release download v1.4.0 --repo FelisMC/Felis --dir felis-v1.4.0 \
|
||||
--pattern 'felis-*linux-amd64*' --pattern felis-velocity.jar --pattern SHA256SUMS
|
||||
# on the host, after copying the directory over
|
||||
sudo FELIS_ARTIFACT_DIR=/root/felis-v1.4.0 bash bootstrap.sh
|
||||
```
|
||||
|
||||
`SHA256SUMS` lists both architectures; the files of the other one may be left out.
|
||||
|
||||
### While felis-api restarts
|
||||
|
||||
An installer rerun that changes felis-api, a node restart or a crashed pod takes the API
|
||||
@@ -70,6 +200,17 @@ the window:
|
||||
`/opt/felis/velocity/plugins/felis-link/last-servers.json`, the last server list the
|
||||
API answered with, until a refresh succeeds (every 15 s).
|
||||
|
||||
A longer outage reads like this in the logs [VM-VERIFIED]. The drill scaled felis-api
|
||||
to 0 for about 8 minutes on the reference VM.
|
||||
|
||||
- The proxy logged `server list refresh failed ... keeping current registrations` 11 s
|
||||
in, then `still failing: 22 failed attempts over 304 s` at the 5-minute mark.
|
||||
- The watchdog found `deployment/felis-api` critical on its first run after the scale.
|
||||
It raised the alert on the first run past 5 minutes, at about 7 minutes; with no
|
||||
`[smtp]` that is logged only (`journalctl -u felis-watchdog`).
|
||||
- The proxy logged `server list refresh recovered after 32 failed attempts over 469 s`
|
||||
as soon as the new pod was Available.
|
||||
|
||||
## 2. Sizing
|
||||
|
||||
### What the platform itself uses
|
||||
@@ -85,15 +226,16 @@ server running **[VM-VERIFIED]**:
|
||||
| lobby (Paper, pod limit 1 GiB) | ~0.7–0.85 GB |
|
||||
| login (Limbo, pod limit 512 MiB) | ~0.16 GB |
|
||||
| felis-api, felis-operator, registry gate | ~50 MB each |
|
||||
| PostgreSQL | ~30 MB plus page cache |
|
||||
| PostgreSQL (the felis-postgres pod) | ~30 MB plus page cache |
|
||||
| **Total in use** | **~3.4 GB** |
|
||||
|
||||
Every game server adds the memory its owner gave it: the pod's limit equals its request,
|
||||
and the JVM heap is derived from it (§1a). Quotas cap it per user (panel → 管理 → 配额).
|
||||
|
||||
The installer's own peak is the image builds (Docker plus a Gradle container); it stops
|
||||
Docker afterwards so that memory goes back to the servers. On a host under 2 GB of RAM
|
||||
without swap it adds a 2 GiB `/swapfile`.
|
||||
A release install builds nothing (§1). When the installer builds on the host its peak is
|
||||
the image builds (Docker plus a Gradle container); it stops Docker afterwards so that memory
|
||||
goes back to the servers. On a host under 2 GB of RAM without swap it adds a 2 GiB
|
||||
`/swapfile`.
|
||||
|
||||
### Recommendations
|
||||
|
||||
@@ -127,15 +269,107 @@ curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo FELIS_VELOCITY_XMX=2G bash
|
||||
| World archives | the `felis-backups` volume (`FELIS_BACKUP_STORAGE`, default 10Gi requested) | about one compressed world per backup kept |
|
||||
| In-cluster registry | the `registry` volume (default 10Gi requested) | 2–3 GB for the stock images; grows with custom builds, pruned daily (§9) |
|
||||
| k3s's containerd images | `/var/lib/rancher/k3s/agent/containerd` | 6–9 GB |
|
||||
| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store) | 5–10 GB after repeated upgrades |
|
||||
| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store), on a host that built its images (§1) | 5–10 GB after repeated upgrades |
|
||||
| Release assets during an install | `/var/lib/felis/artifacts` | up to ~2 GB, deleted once the images are in the registry |
|
||||
| Toolchains and sources | `/opt/felis` | ~2.5 GB |
|
||||
| Database | `/var/lib/felis/postgres` (felis-postgres's cluster) | tens of MB; the audit log is most of it |
|
||||
| Database bundles | `/var/lib/felis/db-backups` | a few MB each, 14 daily kept |
|
||||
|
||||
k3s's local-path volumes do not enforce the requested sizes (§9), so every volume shares
|
||||
the root filesystem. Give the host at least **40 GB**, and 60 GB or more once worlds and
|
||||
custom images accumulate. The watchdog mails the owners when a watched filesystem passes
|
||||
its threshold, and §13b covers a full disk. `docker builder prune -af` (with Docker
|
||||
started) reclaims the build cache when space is short; the next upgrade rebuilds it.
|
||||
its threshold, and §13b covers a full disk. On a host that built its images,
|
||||
`docker builder prune -af` (with Docker started) reclaims the build cache when space is
|
||||
short; the next upgrade rebuilds it.
|
||||
|
||||
### Growing the disk
|
||||
|
||||
Everything above shares the root filesystem, so more room means a bigger root
|
||||
filesystem. It grows in place, with everything running: enlarge the virtual disk at the
|
||||
provider, then the partition and the filesystem on it.
|
||||
|
||||
```bash
|
||||
sudo felis backup-now -yes # a mistyped partition number is how a resize loses a disk
|
||||
lsblk -f # which disk and partition hold /, and whether LVM sits on it
|
||||
sudo growpart /dev/vda 3 # cloud-utils-growpart (RHEL) / cloud-guest-utils (Debian, Ubuntu)
|
||||
# LVM (the RHEL-family default):
|
||||
sudo pvresize /dev/vda3
|
||||
sudo lvextend -r -l +100%FREE /dev/<vg>/root # -r grows the filesystem with it
|
||||
# no LVM:
|
||||
sudo xfs_growfs / # xfs
|
||||
sudo resize2fs /dev/vda3 # ext4
|
||||
df -h /
|
||||
```
|
||||
|
||||
`felis backup-now` (troubleshooting.md §10) archives every stopped world; add `-stop` to
|
||||
include the running ones.
|
||||
|
||||
### Moving the data to its own disk [VM-VERIFIED]
|
||||
|
||||
The bulk lives under `/var/lib/rancher/k3s`: the worlds, the world archives, the registry
|
||||
and the images. On a disk of its own it grows without touching the system, and a full
|
||||
world store leaves the root filesystem alone. The database and its bundles
|
||||
(`/var/lib/felis`) are small and stay on the root disk. The move takes the platform down
|
||||
for the copy plus a minute or two: the drill copied 4.2 GB in 18 s, and felis-api answered
|
||||
`/readyz` 14 s after k3s started on the new disk.
|
||||
|
||||
1. Attach the disk and put a filesystem on it (the whole disk; `lsblk` shows it empty):
|
||||
|
||||
```bash
|
||||
sudo mkfs.xfs /dev/vdb
|
||||
U=$(sudo blkid -s UUID -o value /dev/vdb)
|
||||
```
|
||||
|
||||
2. Archive every world, stopping the servers so each one saves, and keep the watchdog
|
||||
quiet for the next hour (the marker the installer writes: no mail, no failure pings
|
||||
to the heartbeat, until the time in it):
|
||||
|
||||
```bash
|
||||
sudo felis backup-now -yes -stop
|
||||
sudo install -d -m 0755 /run/felis
|
||||
echo $(( $(date +%s) + 3600 )) | sudo tee /run/felis/watchdog-quiet-until
|
||||
```
|
||||
|
||||
3. Stop k3s and copy:
|
||||
|
||||
```bash
|
||||
sudo systemctl stop k3s
|
||||
sudo /usr/local/bin/k3s-killall.sh # the containers k3s leaves running, and their mounts
|
||||
sudo mkdir -p /mnt/felis-data
|
||||
sudo mount UUID=$U /mnt/felis-data
|
||||
sudo rsync -aHAX --numeric-ids /var/lib/rancher/k3s/ /mnt/felis-data/
|
||||
sudo umount /mnt/felis-data
|
||||
```
|
||||
|
||||
`-X` carries the SELinux labels k3s set itself. Leave `restorecon` out: it would reset
|
||||
runc and the CNI binaries from `container_runtime_exec_t` to the policy default.
|
||||
|
||||
4. Mount it in place, and tie k3s to the mount:
|
||||
|
||||
```bash
|
||||
sudo mv /var/lib/rancher/k3s /var/lib/rancher/k3s.old
|
||||
sudo mkdir /var/lib/rancher/k3s
|
||||
echo "UUID=$U /var/lib/rancher/k3s xfs defaults,nofail 0 0" | sudo tee -a /etc/fstab
|
||||
sudo mkdir -p /etc/systemd/system/k3s.service.d
|
||||
printf '[Unit]\nRequiresMountsFor=/var/lib/rancher/k3s\n' | sudo tee /etc/systemd/system/k3s.service.d/data-disk.conf
|
||||
sudo systemctl daemon-reload
|
||||
sudo mount /var/lib/rancher/k3s
|
||||
sudo systemctl start k3s
|
||||
```
|
||||
|
||||
The drop-in is what keeps the data safe: k3s started on the empty mount point creates
|
||||
a new, empty cluster there. With it, a disk that does not come up fails the start with
|
||||
`A dependency job for k3s.service failed`, and `nofail` keeps the host booting so you
|
||||
can reach it. In the drill a detached disk left k3s inactive and the mount point empty;
|
||||
reattached, `systemctl start k3s` mounted it and started.
|
||||
|
||||
5. Check that `sudo k3s kubectl -n felis get pods` shows every pod ready and
|
||||
`findmnt /var/lib/rancher/k3s` names the new disk, then start the servers from the
|
||||
panel and `sudo rm /run/felis/watchdog-quiet-until`. Once the host has run a day,
|
||||
`sudo rm -rf /var/lib/rancher/k3s.old` frees the root disk.
|
||||
|
||||
The watchdog already watches `/var/lib/rancher/k3s` as a filesystem of its own (its
|
||||
`-disk-paths`), so the new disk's fill level is mailed like the root's.
|
||||
|
||||
## 3. Uninstall
|
||||
|
||||
@@ -151,24 +385,31 @@ curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --purge # remove th
|
||||
With a private repository, fetch it the way the README fetches `bootstrap.sh`.
|
||||
|
||||
Both modes remove the `felis-*` systemd units and `cloudflared-felis.service`, the
|
||||
Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the installer's cloudflared binary
|
||||
(unless another unit runs it), the `felis_postgres` and `felis_edge` nftables tables and
|
||||
the firewalld ports the installer opened. k3s goes with k3s's own `k3s-uninstall.sh` when
|
||||
the cluster holds nothing but Felis's namespaces; when it runs anything else only
|
||||
`felis`, `minecraft`, `felis-build` and the MinecraftServer CRD are deleted.
|
||||
Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the release assets an interrupted
|
||||
install left in `/var/lib/felis/artifacts`, the installer's cloudflared binary (unless
|
||||
another unit runs it), the `felis_edge` nftables table (and `felis_postgres`, which
|
||||
releases before the database moved into k3s loaded), the firewalld ports the installer
|
||||
opened and its `felis-`-commented ufw rules. k3s goes with k3s's own `k3s-uninstall.sh` when the cluster holds nothing but
|
||||
Felis's namespaces; when it runs anything else only `felis`, `minecraft`, `felis-build`
|
||||
and the MinecraftServer CRD are deleted.
|
||||
`--keep-k3s` and `--remove-k3s` override that choice.
|
||||
|
||||
| | keep data (default) | `--purge` |
|
||||
|---|---|---|
|
||||
| Final database bundle | taken first (`felis db backup -label manual`); a failure stops the uninstall before anything is removed. `--no-backup` skips it | none |
|
||||
| `felis` database and role | kept | dropped; `listen_addresses` and `pg_hba.conf` go back to how they were |
|
||||
| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file |
|
||||
| `/var/lib/felis` (database bundles) | kept | deleted |
|
||||
| The database (`/var/lib/felis/postgres`) | kept; felis-postgres is stopped cleanly before k3s goes | deleted with `/var/lib/felis` |
|
||||
| A host PostgreSQL an earlier release ran the database on | kept as it is: stopped after the move into k3s (below, §4), with its old copy of `felis` | its `felis` database and role are dropped (the server is started for that and stopped again), and `listen_addresses` and `pg_hba.conf` go back to how they were. Checked before anything is removed: a role that still owns another database (the `felis_pgint` the PG contract tests use, CONTRIBUTING.md) or holds grants elsewhere stops the purge up front with the list and the `ALTER DATABASE … OWNER TO postgres` to run |
|
||||
| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, the mail relay password and uploads bucket keys `felis setup` took, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file |
|
||||
| `/var/lib/felis` (the database, its bundles) | kept | deleted |
|
||||
| Worlds, archives, registry, uploads | moved to `/var/lib/felis/retained/k3s-storage-<stamp>/` (with `--keep-k3s`: their volumes switch to `Retain` and stay in place) | deleted |
|
||||
| Felis images, Docker build cache | kept | deleted |
|
||||
|
||||
Neither mode removes packages (Docker, PostgreSQL, git, nftables) or the swap file: other
|
||||
software may use them. On a host that should end up bare:
|
||||
The two database rows are [SH-TESTED] (`deploy/uninstall_test.sh`); the VM runs above
|
||||
predate felis-postgres.
|
||||
|
||||
Neither mode removes packages (Docker, git, nftables, and the PostgreSQL server an earlier
|
||||
release installed) or the swap file: other software may use them. On a host that should
|
||||
end up bare:
|
||||
|
||||
```
|
||||
sudo swapoff /swapfile && sudo rm /swapfile && sudo sed -i '\|^/swapfile |d' /etc/fstab
|
||||
@@ -183,8 +424,9 @@ DNS records for the panel hostnames, and the Access application.
|
||||
|
||||
A keep-data uninstall leaves everything a reinstall needs. The installer reuses
|
||||
`/etc/felis/secrets.env`, so the database password and the forwarding and session
|
||||
secrets are unchanged, and it migrates the kept database instead of creating one
|
||||
**[VM-VERIFIED]**.
|
||||
secrets are unchanged, and the installer migrates the kept database instead of creating
|
||||
one **[VM-VERIFIED]** (with the host database of the releases before felis-postgres).
|
||||
felis-postgres starts again on the cluster kept in `/var/lib/felis/postgres` [SH-TESTED].
|
||||
|
||||
Each step below was run on the reference VM after a keep-data uninstall, and the
|
||||
restored worlds matched their kept `level.dat` checksums **[VM-VERIFIED]**. `kept` names
|
||||
@@ -256,7 +498,7 @@ the version they were installed with unless noted:
|
||||
| Temurin JRE | moves to the pinned patch build | rerun |
|
||||
| k3s | left alone | rerun with `FELIS_UPGRADE_DEPS=1`: moves to the pinned release through that tag's install script, one minor version at a time (a bigger jump stops before anything changes and names the release to go through), never backwards |
|
||||
| cloudflared | left alone | rerun with `FELIS_UPGRADE_DEPS=1`: swaps `/usr/local/bin/cloudflared` for the pinned, sha256-checked release and restarts `cloudflared-felis`; a cloudflared the distribution installed stays with its package manager |
|
||||
| PostgreSQL | the distribution's package | the package manager for a minor release; a major version needs `pg_upgrade` first (below) |
|
||||
| PostgreSQL | follows the image the release pins | a minor release comes with a Felis release, and the rerun restarts felis-postgres on it (a few seconds without the API); a major version is a dump and restore (below) |
|
||||
| Docker, git, nftables | distribution packages | the package manager |
|
||||
|
||||
```sh
|
||||
@@ -266,8 +508,9 @@ curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap
|
||||
|
||||
`sudo felis update` reports Felis, Velocity, k3s, cloudflared, the JRE and PostgreSQL
|
||||
against their newest releases; `--k3s`, `--cloudflared`, `--jre` and `--postgres` narrow
|
||||
it to one. PostgreSQL is compared within its major, since a minor release is a package
|
||||
update, and a major past its end of life gets a note naming the current one.
|
||||
it to one. PostgreSQL is read from the felis-postgres container and compared within its
|
||||
major, since a minor release arrives with a Felis release, and a major past its end of life
|
||||
gets a note naming the current one.
|
||||
|
||||
The installer also sets up `felis-update-check.timer`, which runs `felis update --record`
|
||||
once a day around 05:30 (and at boot after a missed run). `--record` stores the result
|
||||
@@ -283,22 +526,115 @@ journalctl -u felis-update-check -n 50 --no-pager
|
||||
sudo felis update --record # record a fresh check now
|
||||
```
|
||||
|
||||
### PostgreSQL major versions [CODE-ONLY]
|
||||
### Bringing an older install up to date [VM-VERIFIED]
|
||||
|
||||
The installer takes the major the distribution ships (13 on EL9) and never moves it. To
|
||||
go to a newer one, stop the writers, keep a dump, then use the distribution's upgrade
|
||||
path:
|
||||
Three pieces of an install keep the shape they had on the day they were created, and
|
||||
neither `felis setup` nor `kubectl rollout restart` reaches them: the felis-api
|
||||
Deployment (an env var added later, such as `FELIS_SMTP_PASSWORD`, is absent until the
|
||||
Deployment is rendered again), the lobby image (built with whatever plugins the recipe
|
||||
had then; LuckPerms came later, and without it every permission change from the panel
|
||||
answers `luckperms_missing`), and the `MinecraftServer` specs (a field added later stays
|
||||
unset). Bring all three forward in this order, images first:
|
||||
|
||||
```sh
|
||||
# 1. Rerun the installer: renders and applies the control-plane bundle, rebuilds and
|
||||
# re-imports the login and lobby images, and recreates those two pods so they run
|
||||
# the new images. The [smtp] relay the setup wizard wrote is carried forward.
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
|
||||
# 2. Fill the spec fields the system servers gained since (troubleshooting §12b), then
|
||||
# RCON for user servers created before it was the default. -user-rcon waits on
|
||||
# each server's image opening RCON; see §12b before running it.
|
||||
sudo felis converge
|
||||
sudo felis converge -user-rcon
|
||||
```
|
||||
|
||||
Check each piece:
|
||||
|
||||
```sh
|
||||
kubectl -n felis get deploy felis-api \
|
||||
-o jsonpath='{.spec.template.spec.containers[0].env[*].name}' | tr ' ' '\n' | grep SMTP
|
||||
kubectl -n minecraft exec lobby-0 -- ls /data/plugins | grep -i luckperms
|
||||
kubectl -n minecraft get minecraftserver \
|
||||
-o custom-columns=NAME:.metadata.name,RCON:.spec.rcon.enabled,IDLE:.spec.idle.autoStopEnabled
|
||||
```
|
||||
|
||||
The env var only carries the password; mail still needs the relay itself, set in
|
||||
`felis setup` → email. A user server picks up its new RCON block at its next start.
|
||||
|
||||
### PostgreSQL major versions [CODE-ONLY]
|
||||
|
||||
felis-postgres keeps its cluster in `/var/lib/felis/postgres/<major>/docker`. A release that
|
||||
moves the image to a new major finds the old major's cluster there and stops before it
|
||||
changes anything: the new server would start an empty cluster beside it. The way across is
|
||||
a bundle, taken on the release you run now, restored into the new major's empty cluster:
|
||||
|
||||
```sh
|
||||
# On the release you run now:
|
||||
b="$(sudo felis db backup -label pre-upgrade | sed -n 's/^felis db backup: wrote //p')"
|
||||
sudo k3s kubectl -n felis scale deploy/felis-postgres --replicas=0
|
||||
sudo mv /var/lib/felis/postgres/18 /var/lib/felis/postgres-18.old # the old major's cluster, for a way back
|
||||
|
||||
# Install the new release: it starts an empty cluster on the new major and creates the schema.
|
||||
curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo bash
|
||||
|
||||
# Put the data back and bring its schema up to the new release.
|
||||
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator --replicas=0
|
||||
sudo -u postgres pg_dumpall > /root/felis-pg-$(date +%F).sql
|
||||
# EL9: sudo systemctl stop postgresql; sudo dnf module switch-to postgresql:16
|
||||
# sudo dnf install postgresql-upgrade; sudo postgresql-setup --upgrade
|
||||
# Debian/Ubuntu: sudo pg_upgradecluster <old-major> main
|
||||
sudo systemctl start postgresql
|
||||
sudo felis db restore -yes -no-safety-backup "$b"
|
||||
sudo felis migrate up -config /etc/felis/felis.host.toml
|
||||
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator --replicas=1
|
||||
```
|
||||
|
||||
Delete `/var/lib/felis/postgres-18.old` once the new release has run for a while. To go back
|
||||
instead, scale felis-postgres to 0, move the new major's directory out of
|
||||
`/var/lib/felis/postgres`, move `postgres-18.old` back as `/var/lib/felis/postgres/18`, and
|
||||
rerun the older release's installer.
|
||||
|
||||
### The database's move into k3s [VM-VERIFIED] [CI]
|
||||
|
||||
Releases before the move ran the database on a PostgreSQL the installer installed on the
|
||||
host. The first rerun of a release with felis-postgres moves it, once:
|
||||
|
||||
1. It stops felis-api, felis-operator and the host timers, and heads the host's
|
||||
`pg_hba.conf` with a block that refuses every connection to `felis` but its own copy
|
||||
(the original is kept beside it as `pg_hba.conf.pre-pg-move`).
|
||||
2. It takes a `pre-pg-move` bundle of the host database (`felis db backup`), restores it
|
||||
into felis-postgres (`felis db restore`) and compares the row count of every table on
|
||||
both servers. Any failure up to here puts `pg_hba.conf` and the control plane back and
|
||||
the platform keeps running on the host database, untouched.
|
||||
3. It stops and disables the host `postgresql` service, which stays installed with its
|
||||
copy of the data, and writes `/var/lib/felis/postgres-moved`. A host server that also
|
||||
holds other databases keeps running; its `felis` copy is then reachable over loopback
|
||||
only.
|
||||
|
||||
The e2e upgrade job seeds the newest release's database with users, links, sessions, audit
|
||||
rows, backups, builds and the rest (`deploy/e2e_seed.sh`), upgrades, and checks that
|
||||
felis-postgres holds every seeded row with the same values after the pending migrations.
|
||||
While that release is v0.1.0, the upgrade is this move.
|
||||
|
||||
From then on the host config points at felis-postgres (`127.0.0.1:15432`, and
|
||||
`deployment = "felis/felis-postgres"`, through which `felis db` runs `pg_dump`, `psql`
|
||||
and `pg_restore` inside the pod) and the pods at `felis-postgres.felis.svc:5432`.
|
||||
|
||||
To go back to the host database, for instance to reinstall the release before the move:
|
||||
|
||||
```sh
|
||||
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator deploy/felis-postgres --replicas=0
|
||||
hba="$(sudo -u postgres psql -XtAc 'SHOW hba_file' 2>/dev/null || echo /var/lib/pgsql/data/pg_hba.conf)"
|
||||
sudo cp -p "${hba}.pre-pg-move" "$hba"
|
||||
sudo systemctl enable --now postgresql
|
||||
sudo rm /var/lib/felis/postgres-moved
|
||||
curl -fsSL <raw-url-of-that-release>/deploy/bootstrap.sh | sudo bash
|
||||
```
|
||||
|
||||
`SHOW hba_file` needs the server running; with it stopped, the fallback path is EL's
|
||||
(Debian and Ubuntu keep it in `/etc/postgresql/<major>/main/`). Whatever the platform wrote
|
||||
after the move lives only in felis-postgres; take a bundle there first
|
||||
(`sudo felis db backup`) and restore it onto the host database afterwards if that matters.
|
||||
Once the move has run for a while, drop the host copy:
|
||||
`sudo systemctl start postgresql; sudo -u postgres dropdb felis; sudo -u postgres dropuser felis`,
|
||||
or remove the server package altogether.
|
||||
|
||||
### The MinecraftServer CRD [VM-VERIFIED]
|
||||
|
||||
Every rerun applies the CRD embedded in the `felis` binary (`felis bootstrap-assets crd`).
|
||||
@@ -330,6 +666,38 @@ version, in this order, each step one release:
|
||||
3. A later release stops serving `v1alpha1`. Felis itself reads through one Go type at a
|
||||
time, so the operator and felis-api switch in the release that moves storage.
|
||||
|
||||
### Legacy-forwarded backends [VM-VERIFIED]
|
||||
|
||||
A 1.8-era backend sits behind ViaVersion, which drops modern forwarding's login plugin
|
||||
message on the way down to protocol 47, so the proxy has to hand that server the
|
||||
player's identity BungeeCord-style, in the handshake address. Only the Felis-Legacy
|
||||
Velocity fork can do that per server. Mark the server's CR and the proxy picks it up at
|
||||
its next server-list refresh (every 15 s):
|
||||
|
||||
```sh
|
||||
kubectl -n minecraft label minecraftserver <name> felis.lolicon.best/forwarding=legacy
|
||||
kubectl -n minecraft label minecraftserver <name> felis.lolicon.best/forwarding- # back to modern
|
||||
journalctl -u felis-velocity | grep 'legacy forwarding list'
|
||||
```
|
||||
|
||||
The installer's `FELIS_LEGACY_FORWARDING_SERVERS` (default `legacy18`) stays in the list
|
||||
whatever the labels say. What a label does depends on the proxy the host runs:
|
||||
|
||||
| Proxy | A label applies |
|
||||
|---|---|
|
||||
| Fork with patch 0004 (`build-velocity.sh` default arm) | from the next connection to that server |
|
||||
| Fork with 0003 alone (`--deployed`) | at the next `systemctl restart felis-velocity` |
|
||||
| Stock Velocity | never; the log line is a warning naming the server |
|
||||
|
||||
On the test VM (fork with 0004) labelling a server logged `legacy forwarding list is now
|
||||
[legacy18,resolvecheck]` 12 s later, and removing the label logged the list back to
|
||||
`[legacy18]`. The fork's own test (`FelisLegacyForwardingTest`) covers the next
|
||||
connection following the rewritten list.
|
||||
|
||||
Legacy forwarding carries no secret. A marked server believes any identity that reaches
|
||||
its game port, which `felis-allow-game-from-velocity` limits to the proxy and the node
|
||||
itself; anything else running on the node can reach it too.
|
||||
|
||||
## 5. Disaster recovery
|
||||
|
||||
The procedures are in §16: what a database bundle holds, restoring one on the same host,
|
||||
@@ -344,7 +712,122 @@ production install:
|
||||
holds only sealed objects.
|
||||
- **Keep one database bundle off the host** as well when there is no bucket. It contains
|
||||
`secrets.env`, which a rebuild needs to read the rest.
|
||||
- **Rehearse the rebuild** once on a spare VM: §16 "Rebuild on a new host", steps 1–5,
|
||||
then log in and restore one world. `felis offsite status` and `felis db check` exit
|
||||
non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
|
||||
- **Rehearse the rebuild** once on a spare VM: troubleshooting.md §16 "Rebuild on a new
|
||||
host", every step but 8 (take-over) and 11 (the tunnel), then its checks: sign in with
|
||||
an email code, restore one world and join it. `felis offsite status` and `felis db check` exit
|
||||
non-zero when the copy or the newest daily bundle is stale; wire them into your monitoring,
|
||||
or rely on the watchdog's mail.
|
||||
|
||||
### Moving to another host (planned)
|
||||
|
||||
A planned move is the rebuild of troubleshooting.md §16, with the old host still there to
|
||||
hand over a copy that misses nothing. It needs the off-site bucket: that is how the world
|
||||
archives reach the new host (§16 step 7). The platform is down from step 1 until the new
|
||||
host serves.
|
||||
|
||||
1. **On the old host**, stop everything that changes a world, then send the last copy:
|
||||
|
||||
```bash
|
||||
sudo install -d -m 0755 /run/felis
|
||||
echo $(( $(date +%s) + 4 * 3600 )) | sudo tee /run/felis/watchdog-quiet-until
|
||||
sudo systemctl stop felis-velocity # no joins, so no server wakes
|
||||
sudo felis backup-now -yes -stop # every world archived; the servers stay stopped
|
||||
sudo k3s kubectl -n felis scale deploy/felis-operator --replicas=0 # nothing starts a server from here on
|
||||
sudo felis db backup # a bundle that lists those archives
|
||||
sudo systemctl start felis-offsite.service
|
||||
sudo felis offsite status # again until nothing waits
|
||||
```
|
||||
|
||||
The order matters. The new host fetches the archives its restored database lists, so
|
||||
the bundle comes after the last archive. The operator goes after `backup-now`, which
|
||||
needs it to stop the servers. The quiet marker keeps the watchdog from mailing the
|
||||
owners about the stopped proxy and operator for the next 4 hours.
|
||||
2. **On the new host**, follow troubleshooting.md §16 "Rebuild on a new host" from step 1;
|
||||
`fetch-db latest` picks the bundle the old host just sent. Step 8 (`felis offsite
|
||||
take-over`) makes the new host the one that writes the bucket, and from then on the
|
||||
old host copies nothing more. Steps 10 and 11 move the names and the tunnel.
|
||||
3. **Check the new host** before announcing it: sign in with an email code, restore one
|
||||
world and join it, and see `sudo felis offsite status` show a recent `last success` and
|
||||
no stand-by notice.
|
||||
4. **Retire the old host.** It holds the last copy of every world outside the bucket, so
|
||||
keep it powered off with its disk for a few days first, disabled so a boot brings
|
||||
nothing up:
|
||||
|
||||
```bash
|
||||
sudo systemctl disable k3s felis-velocity felis-watchdog.timer felis-offsite.timer \
|
||||
felis-db-backup.timer felis-update-check.timer felis-build-tools.timer
|
||||
sudo poweroff
|
||||
```
|
||||
|
||||
Then uninstall it (§3) or wipe it.
|
||||
|
||||
Each step is covered where it is documented (backup-now in troubleshooting.md §10, the
|
||||
rebuild in §16); the sequence as a whole has not been rehearsed as one move.
|
||||
|
||||
## 6. Changing the root domain [VM-VERIFIED] [GO-TESTED] [SH-TESTED]
|
||||
|
||||
The root domain is written into more places than the installer's config: the panel
|
||||
certificate (`/etc/felis/panel-tls.crt`), the `felis-config` Secret in both namespaces,
|
||||
the `felis-api-tls` Secret, the proxy's `felis-link.properties`, the login gate's
|
||||
`MinecraftServer` env (`FELIS_ROOT_DOMAIN`, `FELIS_PANEL_HOSTNAME`), the Cloudflare tunnel
|
||||
and DNS. `felis domain set` moves every one of them that lives on the host, in that
|
||||
order, then restarts what reads them; `felis domain check` reports each surface on its
|
||||
own line. The installer keeps the installed domain: a rerun with a different
|
||||
`FELIS_ROOT_DOMAIN` stops and names this command.
|
||||
|
||||
```sh
|
||||
sudo felis domain set new.example.net # the plan: every surface, what it moves to, what it costs
|
||||
sudo felis domain set -yes new.example.net # do it
|
||||
sudo felis domain check # one line per surface; exits 1 while any is behind
|
||||
```
|
||||
|
||||
What it keeps:
|
||||
|
||||
- A panel or admin-console hostname set by hand in `[auth]` (anything other than
|
||||
`console.<root>` / `op.console.<root>`) stays as it is; change it in
|
||||
`/etc/felis/felis.host.toml` yourself if it should move, then run `set` again.
|
||||
- The other `[auth]` keys (`access_jwt_aud`, `client_ip_header`) and every other line of
|
||||
both config files. The edit refuses a file it cannot change line for line (a multi-line
|
||||
value, a quoted or dotted key) and names what to fix.
|
||||
- An operator's own certificate. The installer's self-signed certificate is reissued for
|
||||
the new names (same shape, the old pair saved beside it as `*.pre-domain-<time>`); a
|
||||
certificate from another issuer that does not cover the new names stops the command
|
||||
before anything changes. Replace it with one that does, then run `set` again.
|
||||
|
||||
What it costs, which the plan prints before `-yes`:
|
||||
|
||||
- **DNS.** `<root>`, `console.<root>`, `op.console.<root>` and `*.<root>` must reach the
|
||||
host. The wildcard does not cover `op.console.<root>`, a third-level name: give it its
|
||||
own record. `check` resolves each name and warns on the ones that do not resolve yet.
|
||||
- **Players.** Servers are reached as `<name>.<new root>`; the old addresses stop routing,
|
||||
and the proxy restart disconnects everyone online. The first installer re-run after a
|
||||
move restarts the proxy once more: the fingerprint it keeps of the proxy's files
|
||||
predates the move.
|
||||
- **Sign-in.** Session cookies belong to the old hostnames, so everyone signs in again.
|
||||
Passkeys are bound to the panel hostname: when it changes, the plan counts the passkeys
|
||||
that stop working, and their users sign in with an email code and register a new one.
|
||||
Without an `[smtp]` relay no code is delivered; an Owner locked out that way recovers
|
||||
with `sudo felis breakGlass`.
|
||||
- **Cloudflare.** The tunnel's ingress and the Access application still carry the old
|
||||
names. Re-run the Cloudflare step of `sudo felis setup` after the move; `check` lists
|
||||
the tunnel's hostnames against the new ones.
|
||||
- **A proxy on another host** (a remote `felis-link.properties`) is outside this host's
|
||||
reach: `set` prints the three keys to put there.
|
||||
|
||||
`set` is safe to repeat: a second run changes only what is still behind, and on an
|
||||
install that is already on the domain it converges whatever `check` reports. The same
|
||||
holds after an interruption.
|
||||
|
||||
On the reference VM the move from `10.211.55.6.nip.io` to `10-211-55-6.nip.io` took 34
|
||||
seconds. The certificate served on 30443, `/config.json` on both hostnames, the proxy's
|
||||
`Felis routing ready: rootDomain=` log line and the login pod's env all carried the new
|
||||
names afterwards. A second `set -yes` changed and restarted nothing; the installer run
|
||||
with the old `FELIS_ROOT_DOMAIN` stopped at its first check; a full installer re-run kept
|
||||
the moved domain and left `check` clean; moving back restored every surface
|
||||
**[VM-VERIFIED]**.
|
||||
|
||||
`check` reads the proxy as behind when `felis-velocity` started before
|
||||
`felis-link.properties` last changed. Installers before this command rewrote that file
|
||||
on every run, so a host upgraded from one can show that line once with the file already
|
||||
on the names; `sudo systemctl restart felis-velocity` clears it. The installer now leaves
|
||||
the file alone when its content is the same.
|
||||
@@ -80,7 +80,8 @@ sequenceDiagram
|
||||
API-->>Panel: 412 not_linked
|
||||
else linked
|
||||
Repo-->>API: true
|
||||
API->>Repo: QuotaAvailable(user_id)
|
||||
API->>Repo: QuotaCheck(user_id, the server's real size)
|
||||
Note over API,Repo: all four caps: servers, CPU, memory, storage
|
||||
alt quota exhausted
|
||||
Repo-->>API: false
|
||||
API-->>Panel: 403 quota_exceeded
|
||||
|
||||
+1261
-98
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,133 @@
|
||||
package felis
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The lobby and login gate take their player cap from server.properties, which the
|
||||
// entrypoint rewrites on every boot over whatever the volume already holds. These run
|
||||
// the shipped entrypoints the way a pod does (image and volume paths pointed into temp
|
||||
// dirs, java replaced by a stub that exits) and read the file the server would start on.
|
||||
|
||||
// runEntrypoint runs the embedded entrypoint with its runtime dir holding the given
|
||||
// files and server.properties seeded with props ("" for a first boot), and returns
|
||||
// server.properties afterwards.
|
||||
func runEntrypoint(t *testing.T, name, runtimeVar string, files []string, props string, env ...string) string {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
runtime, data, bin := filepath.Join(root, "image"), filepath.Join(root, "data"), filepath.Join(root, "bin")
|
||||
for _, f := range files {
|
||||
p := filepath.Join(runtime, f)
|
||||
if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(p, []byte("jar"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
for _, d := range []string{data, bin} {
|
||||
if err := os.MkdirAll(d, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if props != "" {
|
||||
if err := os.WriteFile(filepath.Join(data, "server.properties"), []byte(props), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(bin, "java"), []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
script := readGameStackFile(t, name)
|
||||
for old, repl := range map[string]string{
|
||||
runtimeVar: `RUNTIME_DIR="` + runtime + `"`,
|
||||
`DATA_DIR="/data"`: `DATA_DIR="` + data + `"`,
|
||||
} {
|
||||
if !strings.Contains(script, old) {
|
||||
t.Fatalf("%s no longer sets %s", name, old)
|
||||
}
|
||||
script = strings.Replace(script, old, repl, 1)
|
||||
}
|
||||
path := filepath.Join(root, "entrypoint.sh")
|
||||
if err := os.WriteFile(path, []byte(script), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
cmd := exec.Command("sh", path)
|
||||
cmd.Env = append([]string{"PATH=" + bin + ":" + os.Getenv("PATH"), "FELIS_FORWARDING_SECRET=fwd-test"}, env...)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("%s failed: %v\n%s", name, err, out)
|
||||
}
|
||||
got, err := os.ReadFile(filepath.Join(data, "server.properties"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(got)
|
||||
}
|
||||
|
||||
// propLines lists the key's lines, so a key written twice shows up as two.
|
||||
func propLines(props, key string) []string {
|
||||
var out []string
|
||||
for _, line := range strings.Split(props, "\n") {
|
||||
if strings.HasPrefix(line, key+"=") {
|
||||
out = append(out, line)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func assertProp(t *testing.T, props, key, want string) {
|
||||
t.Helper()
|
||||
got := propLines(props, key)
|
||||
if len(got) != 1 || got[0] != key+"="+want {
|
||||
t.Errorf("%s: got %q, want exactly [%s=%s]\nserver.properties:\n%s", key, got, key, want, props)
|
||||
}
|
||||
}
|
||||
|
||||
var lobbyImage = []string{"paper.jar", "plugins/felis-paper.jar", "plugins/LuckPerms.jar"}
|
||||
|
||||
func TestLobbyEntrypointLiftsThePlayerCap(t *testing.T) {
|
||||
t.Run("over the cap Paper wrote on an earlier boot", func(t *testing.T) {
|
||||
props := runEntrypoint(t, "deploy/lobby/entrypoint.sh", `RUNTIME_DIR="/paper"`, lobbyImage,
|
||||
"#Minecraft server properties\nmax-players=20\nmotd=Kept as it was\n")
|
||||
assertProp(t, props, "max-players", "200")
|
||||
assertProp(t, props, "motd", "Kept as it was")
|
||||
})
|
||||
t.Run("on a first boot", func(t *testing.T) {
|
||||
props := runEntrypoint(t, "deploy/lobby/entrypoint.sh", `RUNTIME_DIR="/paper"`, lobbyImage, "")
|
||||
assertProp(t, props, "max-players", "200")
|
||||
assertProp(t, props, "online-mode", "false")
|
||||
})
|
||||
}
|
||||
|
||||
// The RCON password is arbitrary bytes from a Secret; each of sed's special characters
|
||||
// has to land in the file as itself when an earlier boot's line is replaced.
|
||||
func TestLobbyEntrypointWritesTheRconPasswordVerbatim(t *testing.T) {
|
||||
const password = `a|b\c&d/e`
|
||||
props := runEntrypoint(t, "deploy/lobby/entrypoint.sh", `RUNTIME_DIR="/paper"`, lobbyImage,
|
||||
"enable-rcon=false\nrcon.password=stale\n", "RCON_PASSWORD="+password)
|
||||
assertProp(t, props, "enable-rcon", "true")
|
||||
assertProp(t, props, "rcon.password", password)
|
||||
}
|
||||
|
||||
var limboImage = []string{"Limbo.jar", "plugins/felis-limbo.jar"}
|
||||
|
||||
func TestLimboEntrypointNeverCapsTheGate(t *testing.T) {
|
||||
t.Run("over a cap left on the volume", func(t *testing.T) {
|
||||
props := runEntrypoint(t, "deploy/limbo/entrypoint.sh", `RUNTIME_DIR="/limbo"`, limboImage,
|
||||
"max-players=10\nlevel-name=world;spawn.schem\n")
|
||||
assertProp(t, props, "max-players", "-1")
|
||||
assertProp(t, props, "level-name", "world;spawn.schem")
|
||||
assertProp(t, props, "velocity-modern", "true")
|
||||
})
|
||||
t.Run("on a first boot", func(t *testing.T) {
|
||||
props := runEntrypoint(t, "deploy/limbo/entrypoint.sh", `RUNTIME_DIR="/limbo"`, limboImage, "")
|
||||
assertProp(t, props, "max-players", "-1")
|
||||
assertProp(t, props, "forwarding-secrets", "fwd-test")
|
||||
})
|
||||
}
|
||||
@@ -16,6 +16,7 @@ require (
|
||||
github.com/minio/minio-go/v7 v7.2.1
|
||||
github.com/prometheus/client_golang v1.19.1
|
||||
github.com/prometheus/client_model v0.6.1
|
||||
golang.org/x/text v0.39.0
|
||||
k8s.io/api v0.31.3
|
||||
k8s.io/apimachinery v0.31.3
|
||||
k8s.io/client-go v0.31.0
|
||||
@@ -95,11 +96,10 @@ require (
|
||||
golang.org/x/crypto v0.52.0 // indirect
|
||||
golang.org/x/exp v0.0.0-20231006140011-7918f672742d // indirect
|
||||
golang.org/x/net v0.55.0 // indirect
|
||||
golang.org/x/oauth2 v0.21.0 // indirect
|
||||
golang.org/x/oauth2 v0.27.0 // indirect
|
||||
golang.org/x/sync v0.21.0 // indirect
|
||||
golang.org/x/sys v0.45.0 // indirect
|
||||
golang.org/x/term v0.43.0 // indirect
|
||||
golang.org/x/text v0.39.0 // indirect
|
||||
golang.org/x/time v0.3.0 // indirect
|
||||
gomodules.xyz/jsonpatch/v2 v2.4.0 // indirect
|
||||
google.golang.org/protobuf v1.36.10 // indirect
|
||||
|
||||
@@ -249,8 +249,8 @@ golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLL
|
||||
golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU=
|
||||
golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8=
|
||||
golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww=
|
||||
golang.org/x/oauth2 v0.21.0 h1:tsimM75w1tF/uws5rbeHzIWxEqElMehnc+iW793zsZs=
|
||||
golang.org/x/oauth2 v0.21.0/go.mod h1:XYTD2NtWslqkgxebSiOHnXEap4TF09sJSc7H1sXbhtI=
|
||||
golang.org/x/oauth2 v0.27.0 h1:da9Vo7/tDv5RH/7nZDz1eMGS/q1Vv1N/7FCrBhI9I3M=
|
||||
golang.org/x/oauth2 v0.27.0/go.mod h1:onh5ek6nERTohokkhCD/y2cV4Do3fxFHFuAejCkRWT8=
|
||||
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
|
||||
Loaded 100 of 508 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user