Compare commits
57
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a252f04b0f | ||
|
|
3e61fde06d | ||
|
|
f0ac5b48ce | ||
|
|
1141ebcd49 | ||
|
|
82545548e7 | ||
|
|
aace66e8d3 | ||
|
|
b1627e9ba0 | ||
|
|
a4fa16ae69 | ||
|
|
80fbc779d9 | ||
|
|
07eb682358 | ||
|
|
3ad817eeff | ||
|
|
26dc1722f4 | ||
|
|
c4a4f1f25c | ||
|
|
8296153a38 | ||
|
|
bf18712a6c | ||
|
|
c0643af118 | ||
|
|
80c4ea8966 | ||
|
|
e75448a118 | ||
|
|
720ab81bf8 | ||
|
|
9baa0a10af | ||
|
|
151c9d2e30 | ||
|
|
05e8c64e47 | ||
|
|
7f772bccbb | ||
|
|
c49336ba21 | ||
|
|
5cadbd40a9 | ||
|
|
a27d76ae1d | ||
|
|
e836a73a8a | ||
|
|
e521cf5976 | ||
|
|
13b65e19ec | ||
|
|
9bda3a52fa | ||
|
|
f69cdec9d4 | ||
|
|
22becd6859 | ||
|
|
47890ca913 | ||
|
|
fa8db4c7e8 | ||
|
|
d17524cd67 | ||
|
|
d50492b86f | ||
|
|
0c29ab3a96 | ||
|
|
215bfd78d7 | ||
|
|
857b6da886 | ||
|
|
5521e498a9 | ||
|
|
346a93921e | ||
|
|
c1796bea17 | ||
|
|
bd909d5858 | ||
|
|
857c73a66a | ||
|
|
c4e4953f3d | ||
|
|
15f729ffea | ||
|
|
18e2397272 | ||
|
|
248fa5d100 | ||
|
|
c7db7d4126 | ||
|
|
abfe60d62d | ||
|
|
7819e5de50 | ||
|
|
3424852a39 | ||
|
|
8f684cedc0 | ||
|
|
955433ba81 | ||
|
|
4648175773 | ||
|
|
3247b9e61a | ||
|
|
b9c97ffac9 |
No files matched your search
@@ -0,0 +1,23 @@
|
|||||||
|
# The workflows pin every action to a commit SHA, and every Dockerfile pins its base images
|
||||||
|
# by digest. This keeps those pins moving: Dependabot reads the "# vX.Y.Z" comment next to
|
||||||
|
# each action SHA, and the tag in front of each image digest, and opens a PR that bumps both
|
||||||
|
# together.
|
||||||
|
version: 2
|
||||||
|
updates:
|
||||||
|
- package-ecosystem: github-actions
|
||||||
|
directory: /
|
||||||
|
schedule:
|
||||||
|
interval: weekly
|
||||||
|
- package-ecosystem: docker
|
||||||
|
directories:
|
||||||
|
- /
|
||||||
|
- /deploy/limbo
|
||||||
|
- /deploy/lobby
|
||||||
|
- /deploy/paper
|
||||||
|
schedule:
|
||||||
|
interval: weekly
|
||||||
|
# A new major is a runtime change (Paper 26.x needs Java 25, Limbo's jar is Java 21
|
||||||
|
# bytecode), so only digests and minors are proposed; majors move by hand.
|
||||||
|
ignore:
|
||||||
|
- dependency-name: "*"
|
||||||
|
update-types: ["version-update:semver-major"]
|
||||||
+80
-12
@@ -14,10 +14,14 @@
|
|||||||
# PR is what asks for the answer.
|
# PR is what asks for the answer.
|
||||||
name: ci
|
name: ci
|
||||||
|
|
||||||
|
#
|
||||||
|
# release.yml calls this workflow (workflow_call) before it builds anything, so a tag passes
|
||||||
|
# exactly these gates and there is one list of them.
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
branches: [main]
|
branches: [main]
|
||||||
pull_request:
|
pull_request:
|
||||||
|
workflow_call:
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
@@ -30,9 +34,9 @@ jobs:
|
|||||||
go:
|
go:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
- uses: actions/setup-go@v5
|
- uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0
|
||||||
with:
|
with:
|
||||||
go-version-file: go.mod
|
go-version-file: go.mod
|
||||||
|
|
||||||
@@ -43,12 +47,66 @@ jobs:
|
|||||||
echo "gofmt needed on:"; echo "$unformatted"; exit 1
|
echo "gofmt needed on:"; echo "$unformatted"; exit 1
|
||||||
fi
|
fi
|
||||||
- run: go vet ./...
|
- run: go vet ./...
|
||||||
- run: go test ./...
|
# -race: felis-api and the operator are mostly goroutines (watchers, the
|
||||||
|
# registry pruner, the backup scheduler, the rate limiters).
|
||||||
|
- run: go test -race ./...
|
||||||
|
# The version is pinned here and bumped by hand; Dependabot does not read `go run`.
|
||||||
|
- name: staticcheck
|
||||||
|
run: go run honnef.co/go/tools/cmd/[email protected] ./...
|
||||||
|
|
||||||
|
# Separate from the go job so a newly published advisory reads as what it is. govulncheck
|
||||||
|
# exits non-zero only for vulnerable code this module can actually reach, standard
|
||||||
|
# library included: setup-go installs the newest patch of go.mod's Go line, so a finding
|
||||||
|
# there means the Dockerfile's golang digest (which ships the release) needs a bump too.
|
||||||
|
vuln:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
|
- uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
|
||||||
|
- run: go run golang.org/x/vuln/cmd/[email protected] ./...
|
||||||
|
|
||||||
|
# The business stores' SQL against a real PostgreSQL (internal/pgint): the unit suites run
|
||||||
|
# on fakes, and PGRepo drifted from them three times while those stayed green. 13 is the
|
||||||
|
# oldest server a supported distribution installs (EL9), 18 the newest (Arch).
|
||||||
|
pgint:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
postgres: ['13', '18']
|
||||||
|
services:
|
||||||
|
postgres:
|
||||||
|
image: postgres:${{ matrix.postgres }}
|
||||||
|
env:
|
||||||
|
POSTGRES_USER: felis
|
||||||
|
POSTGRES_PASSWORD: pgint
|
||||||
|
POSTGRES_DB: felis_pgint
|
||||||
|
ports:
|
||||||
|
- 5432:5432
|
||||||
|
options: >-
|
||||||
|
--health-cmd "pg_isready -U felis -d felis_pgint"
|
||||||
|
--health-interval 2s
|
||||||
|
--health-timeout 5s
|
||||||
|
--health-retries 30
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
|
- uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
|
||||||
|
- run: go test -race -tags pgint -count=1 ./internal/pgint/
|
||||||
|
env:
|
||||||
|
FELIS_TEST_PG_URL: postgres://felis:pgint@localhost:5432/felis_pgint?sslmode=disable
|
||||||
|
|
||||||
shell:
|
shell:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
# bootstrap.sh is the only thing that ever runs on a fresh host, and nothing here can
|
# bootstrap.sh is the only thing that ever runs on a fresh host, and nothing here can
|
||||||
# run it — it wants root, a package manager and k3s. Syntax plus the extracted-block
|
# run it — it wants root, a package manager and k3s. Syntax plus the extracted-block
|
||||||
@@ -66,12 +124,22 @@ jobs:
|
|||||||
esac
|
esac
|
||||||
done
|
done
|
||||||
|
|
||||||
|
# A pinned release rather than the runner image's copy, so a runner update cannot
|
||||||
|
# change what fails. Warnings and errors fail the job; style notes (info) do not.
|
||||||
|
- name: shellcheck
|
||||||
|
run: |
|
||||||
|
curl -fsSL -o shellcheck.tar.xz \
|
||||||
|
https://github.com/koalaman/shellcheck/releases/download/v0.11.0/shellcheck-v0.11.0.linux.x86_64.tar.xz
|
||||||
|
echo "8c3be12b05d5c177a04c29e3c78ce89ac86f1595681cab149b65b97c4e227198 shellcheck.tar.xz" | sha256sum -c
|
||||||
|
tar -xJf shellcheck.tar.xz
|
||||||
|
./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh')
|
||||||
|
|
||||||
- run: sh deploy/bootstrap_test.sh
|
- run: sh deploy/bootstrap_test.sh
|
||||||
|
|
||||||
panel:
|
panel:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
# The Dockerfile's `FROM node:<major>` is the only place the panel's Node version is
|
# The Dockerfile's `FROM node:<major>` is the only place the panel's Node version is
|
||||||
# declared — there is no .nvmrc and no engines field. Reading it here rather than
|
# declared — there is no .nvmrc and no engines field. Reading it here rather than
|
||||||
@@ -84,7 +152,7 @@ jobs:
|
|||||||
[ -n "$version" ] || { echo "Dockerfile has no 'FROM ... node:<major>' line"; exit 1; }
|
[ -n "$version" ] || { echo "Dockerfile has no 'FROM ... node:<major>' line"; exit 1; }
|
||||||
echo "version=${version}" >> "$GITHUB_OUTPUT"
|
echo "version=${version}" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
- uses: actions/setup-node@v4
|
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
|
||||||
with:
|
with:
|
||||||
node-version: ${{ steps.node.outputs.version }}
|
node-version: ${{ steps.node.outputs.version }}
|
||||||
cache: npm
|
cache: npm
|
||||||
@@ -102,18 +170,18 @@ jobs:
|
|||||||
plugins:
|
plugins:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
# The other jobs never touch the Java layer: the plugin jars were only ever
|
# The other jobs never touch the Java layer: the plugin jars were only ever
|
||||||
# compiled by bootstrap on a live host, and the three test mains under
|
# compiled by bootstrap on a live host, and the three test mains under
|
||||||
# plugins/*/test were run by hand. JDK 21 plus the Gradle major the plugin
|
# plugins/*/test were run by hand. JDK 21 plus the Gradle major the plugin
|
||||||
# Dockerfiles pin (8.14) is that same toolchain, in CI.
|
# Dockerfiles pin (8.14) is that same toolchain, in CI.
|
||||||
- uses: actions/setup-java@v4
|
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
|
||||||
with:
|
with:
|
||||||
distribution: temurin
|
distribution: temurin
|
||||||
java-version: '21'
|
java-version: '21'
|
||||||
|
|
||||||
- uses: gradle/actions/setup-gradle@v4
|
- uses: gradle/actions/setup-gradle@ed408507eac070d1f99cc633dbcf757c94c7933a # v4.4.3
|
||||||
with:
|
with:
|
||||||
gradle-version: '8.14'
|
gradle-version: '8.14'
|
||||||
|
|
||||||
@@ -122,18 +190,18 @@ jobs:
|
|||||||
mods:
|
mods:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
# The three loader mods (Minecraft 1.20.1 / 1.20.4, Java-17 lines) compile
|
# The three loader mods (Minecraft 1.20.1 / 1.20.4, Java-17 lines) compile
|
||||||
# through their vendored Gradle wrappers, which fetch their own Gradle. Until
|
# through their vendored Gradle wrappers, which fetch their own Gradle. Until
|
||||||
# this job nothing ever built them: no install path touches them, and their
|
# this job nothing ever built them: no install path touches them, and their
|
||||||
# gradlew scripts were committed without the exec bit, so the README's
|
# gradlew scripts were committed without the exec bit, so the README's
|
||||||
# one-liners failed on a fresh clone.
|
# one-liners failed on a fresh clone.
|
||||||
- uses: actions/setup-java@v4
|
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
|
||||||
with:
|
with:
|
||||||
distribution: temurin
|
distribution: temurin
|
||||||
java-version: '17'
|
java-version: '17'
|
||||||
|
|
||||||
- uses: gradle/actions/setup-gradle@v4
|
- uses: gradle/actions/setup-gradle@ed408507eac070d1f99cc633dbcf757c94c7933a # v4.4.3
|
||||||
|
|
||||||
- run: bash plugins/test-mods.sh
|
- run: bash plugins/test-mods.sh
|
||||||
+108
-43
@@ -17,6 +17,16 @@
|
|||||||
# quietly ships a release whose panel is that placeholder. The Dockerfile runs the npm
|
# quietly ships a release whose panel is that placeholder. The Dockerfile runs the npm
|
||||||
# build first, and is the same recipe bootstrap uses, so there is one way to build felis
|
# build first, and is the same recipe bootstrap uses, so there is one way to build felis
|
||||||
# rather than two that can drift.
|
# rather than two that can drift.
|
||||||
|
#
|
||||||
|
# SHA256SUMS is a contract with bootstrap too: download_release_binary refuses a binary whose
|
||||||
|
# hash is not listed there, BEFORE it runs it. A release without the file installs by source
|
||||||
|
# build instead.
|
||||||
|
#
|
||||||
|
# The write token never meets the test suite: `gates` (ci.yml) and `build` run the tests,
|
||||||
|
# Gradle and the Docker build (each of which executes third-party code) with a read-only
|
||||||
|
# token, and `build` hands the binaries over as a workflow artifact; `publish` holds contents:write and runs only
|
||||||
|
# pinned actions and gh. Every action is pinned to a commit SHA (the tag in the trailing
|
||||||
|
# comment is for humans); .github/dependabot.yml proposes the bumps.
|
||||||
name: release
|
name: release
|
||||||
|
|
||||||
on:
|
on:
|
||||||
@@ -24,53 +34,29 @@ on:
|
|||||||
tags: ['v*']
|
tags: ['v*']
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: write # gh release create/upload
|
contents: read
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
release:
|
# A tag that ships red is worse than a tag that fails to ship. These are ci.yml's gates,
|
||||||
|
# called rather than copied: Go (race, vet, staticcheck), govulncheck, the PostgreSQL
|
||||||
|
# contract suite, shellcheck and the bootstrap tests, the panel, and the Java layer the
|
||||||
|
# binary EMBEDS (bootstrap_asset.go ships the plugin sources, so a tag whose plugins do
|
||||||
|
# not compile turns every install of that release into a failed bootstrap).
|
||||||
|
gates:
|
||||||
|
uses: ./.github/workflows/ci.yml
|
||||||
|
|
||||||
|
build:
|
||||||
|
needs: gates
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||||
|
|
||||||
- uses: actions/setup-go@v5
|
|
||||||
with:
|
|
||||||
go-version-file: go.mod
|
|
||||||
|
|
||||||
# A tag that ships red is worse than a tag that fails to ship.
|
|
||||||
- run: go vet ./...
|
|
||||||
- run: go test ./...
|
|
||||||
|
|
||||||
# The same reason, for the Java layer the binary EMBEDS: the release asset is
|
|
||||||
# the tree's plugin sources (bootstrap_asset.go), and a tag whose plugins don't
|
|
||||||
# compile turns every install of that release into a failed bootstrap. JDK 21
|
|
||||||
# gates the install-time plugins + codec/invite tests; JDK 17 gates the loader
|
|
||||||
# mods (their vendored wrappers fetch their own Gradle).
|
|
||||||
- uses: actions/setup-java@v4
|
|
||||||
with:
|
|
||||||
distribution: temurin
|
|
||||||
java-version: '21'
|
|
||||||
|
|
||||||
- uses: gradle/actions/setup-gradle@v4
|
|
||||||
with:
|
|
||||||
gradle-version: '8.14'
|
|
||||||
|
|
||||||
- run: bash plugins/test.sh
|
|
||||||
|
|
||||||
- uses: actions/setup-java@v4
|
|
||||||
with:
|
|
||||||
distribution: temurin
|
|
||||||
java-version: '17'
|
|
||||||
|
|
||||||
- uses: gradle/actions/setup-gradle@v4
|
|
||||||
|
|
||||||
- run: bash plugins/test-mods.sh
|
|
||||||
|
|
||||||
# Both architectures, because bootstrap's default release channel DOWNLOADS these
|
# Both architectures, because bootstrap's default release channel DOWNLOADS these
|
||||||
# rather than compiling on the target host — an arm64 host with no asset silently
|
# rather than compiling on the target host — an arm64 host with no asset silently
|
||||||
# falls back to a slow source build. Neither stage is emulated: the Dockerfile pins
|
# falls back to a slow source build. Neither stage is emulated: the Dockerfile pins
|
||||||
# both build stages to $BUILDPLATFORM and the Go stage cross-compiles via TARGETARCH,
|
# both build stages to $BUILDPLATFORM and the Go stage cross-compiles via TARGETARCH,
|
||||||
# so the second architecture costs about a minute.
|
# so the second architecture costs about a minute.
|
||||||
- uses: docker/setup-buildx-action@v3
|
- uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||||
|
|
||||||
- name: Build the stamped binaries
|
- name: Build the stamped binaries
|
||||||
run: |
|
run: |
|
||||||
@@ -100,8 +86,67 @@ jobs:
|
|||||||
file ./felis-linux-arm64 | grep -q 'ARM aarch64' \
|
file ./felis-linux-arm64 | grep -q 'ARM aarch64' \
|
||||||
|| { echo "felis-linux-arm64 is not an arm64 ELF — TARGETARCH did not reach the go build"; exit 1; }
|
|| { echo "felis-linux-arm64 is not an arm64 ELF — TARGETARCH did not reach the go build"; exit 1; }
|
||||||
|
|
||||||
# --verify-tag refuses to invent a release for a tag that is not pushed. The upload
|
- name: Checksum the binaries
|
||||||
# fallback makes a re-run converge rather than failing on an existing release.
|
run: sha256sum felis-linux-amd64 felis-linux-arm64 | tee SHA256SUMS
|
||||||
|
|
||||||
|
# A CycloneDX SBOM per binary: the Go modules (and versions) linked into it, read
|
||||||
|
# from the build info the linker embeds.
|
||||||
|
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0.24.0
|
||||||
|
with:
|
||||||
|
file: felis-linux-amd64
|
||||||
|
format: cyclonedx-json
|
||||||
|
output-file: felis-linux-amd64.cdx.json
|
||||||
|
upload-artifact: false
|
||||||
|
upload-release-assets: false
|
||||||
|
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0.24.0
|
||||||
|
with:
|
||||||
|
file: felis-linux-arm64
|
||||||
|
format: cyclonedx-json
|
||||||
|
output-file: felis-linux-arm64.cdx.json
|
||||||
|
upload-artifact: false
|
||||||
|
upload-release-assets: false
|
||||||
|
|
||||||
|
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||||
|
with:
|
||||||
|
name: release-assets
|
||||||
|
path: |
|
||||||
|
felis-linux-amd64
|
||||||
|
felis-linux-arm64
|
||||||
|
felis-linux-amd64.cdx.json
|
||||||
|
felis-linux-arm64.cdx.json
|
||||||
|
SHA256SUMS
|
||||||
|
if-no-files-found: error
|
||||||
|
retention-days: 7
|
||||||
|
|
||||||
|
publish:
|
||||||
|
needs: build
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
contents: write # gh release create/upload
|
||||||
|
id-token: write # the Sigstore certificate behind the provenance attestation
|
||||||
|
attestations: write
|
||||||
|
steps:
|
||||||
|
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
|
||||||
|
with:
|
||||||
|
name: release-assets
|
||||||
|
|
||||||
|
# The artifact store sits between the two jobs, so check the handover too.
|
||||||
|
- run: sha256sum -c SHA256SUMS
|
||||||
|
|
||||||
|
# Signed SLSA provenance: which workflow run, commit and repository produced each
|
||||||
|
# binary. Check one with `gh attestation verify felis-linux-amd64 --repo FelisMC/Felis`.
|
||||||
|
# GitHub only stores attestations for private repositories on Enterprise Cloud, and a
|
||||||
|
# failure here would block the release, so a private repository skips the step and
|
||||||
|
# relies on SHA256SUMS alone.
|
||||||
|
- name: Attest build provenance
|
||||||
|
if: ${{ !github.event.repository.private }}
|
||||||
|
uses: actions/attest-build-provenance@977bb373ede98d70efdf65b84cb5f73e068dcc2a # v3.0.0
|
||||||
|
with:
|
||||||
|
subject-path: |
|
||||||
|
felis-linux-amd64
|
||||||
|
felis-linux-arm64
|
||||||
|
|
||||||
|
# --verify-tag refuses to invent a release for a tag that is not pushed.
|
||||||
#
|
#
|
||||||
# The prerelease flag has to be passed explicitly: the trigger glob is v*, so v1.2.3-rc1
|
# The prerelease flag has to be passed explicitly: the trigger glob is v*, so v1.2.3-rc1
|
||||||
# lands here too, and gh does not read semver out of the tag name. Published as a full
|
# lands here too, and gh does not read semver out of the tag name. Published as a full
|
||||||
@@ -109,13 +154,33 @@ jobs:
|
|||||||
# channel installs from and `felis update` polls — so every fresh install would get the
|
# channel installs from and `felis update` polls — so every fresh install would get the
|
||||||
# RC binary and every deployed felis-api would error on the felis component until a
|
# RC binary and every deployed felis-api would error on the felis component until a
|
||||||
# stable tag was cut. Flagged, GitHub keeps latest pointing at the last stable release.
|
# stable tag was cut. Flagged, GitHub keeps latest pointing at the last stable release.
|
||||||
|
#
|
||||||
|
# A re-run (the release already exists) uploads only what is missing and never
|
||||||
|
# replaces a published asset: hosts may already have installed it, and their
|
||||||
|
# SHA256SUMS check would start failing against a swapped file. An asset that is
|
||||||
|
# there with different bytes stops the job; cut a new tag instead.
|
||||||
- name: Publish the release
|
- name: Publish the release
|
||||||
env:
|
env:
|
||||||
GH_TOKEN: ${{ github.token }}
|
GH_TOKEN: ${{ github.token }}
|
||||||
|
GH_REPO: ${{ github.repository }}
|
||||||
run: |
|
run: |
|
||||||
|
assets="felis-linux-amd64 felis-linux-arm64 felis-linux-amd64.cdx.json felis-linux-arm64.cdx.json SHA256SUMS"
|
||||||
flags=""
|
flags=""
|
||||||
case "$GITHUB_REF_NAME" in *-*) flags="--prerelease" ;; esac
|
case "$GITHUB_REF_NAME" in *-*) flags="--prerelease" ;; esac
|
||||||
gh release create "$GITHUB_REF_NAME" --verify-tag --generate-notes $flags \
|
if ! gh release view "$GITHUB_REF_NAME" >/dev/null 2>&1; then
|
||||||
./felis-linux-amd64 ./felis-linux-arm64 \
|
# shellcheck disable=SC2086 # word-splitting the list is the point
|
||||||
|| gh release upload "$GITHUB_REF_NAME" \
|
gh release create "$GITHUB_REF_NAME" --verify-tag --generate-notes $flags $assets
|
||||||
./felis-linux-amd64 ./felis-linux-arm64 --clobber
|
exit 0
|
||||||
|
fi
|
||||||
|
# The REST payload's per-asset "digest" is GitHub's own sha256 of the stored file.
|
||||||
|
published="$(gh api "repos/${GH_REPO}/releases/tags/${GITHUB_REF_NAME}" --jq '.assets[] | "\(.name) \(.digest)"')"
|
||||||
|
for a in $assets; do
|
||||||
|
have="$(printf '%s\n' "$published" | awk -v n="$a" '$1 == n { print $2 }')"
|
||||||
|
want="sha256:$(sha256sum < "$a" | cut -d' ' -f1)"
|
||||||
|
if [ -z "$have" ]; then
|
||||||
|
gh release upload "$GITHUB_REF_NAME" "$a"
|
||||||
|
elif [ "$have" != "$want" ]; then
|
||||||
|
echo "::error::$a is already published with $have; this run built $want. Published assets are never replaced."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
+88
-2
@@ -45,7 +45,7 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
|||||||
| 14 | controller-runtime 日志噪音 | ✅ 修复:SetLogger 接 slog(`a415246`) |
|
| 14 | controller-runtime 日志噪音 | ✅ 修复:SetLogger 接 slog(`a415246`) |
|
||||||
| 15 | reaper 启用未演练 | ✅ 已演练(见"第二日"节;CronJob 仍 `suspend=true` 防误删) |
|
| 15 | reaper 启用未演练 | ✅ 已演练(见"第二日"节;CronJob 仍 `suspend=true` 防误删) |
|
||||||
|
|
||||||
## 可达性分级(#1–#74;#62 立案后剔除)——这些缺陷真实使用中到底谁能踩到
|
## 可达性分级(#1–#79;#62 立案后剔除)——这些缺陷真实使用中到底谁能踩到
|
||||||
|
|
||||||
回应质疑"是不是全在测边界条件 / 只有内部 hook 才能触发":
|
回应质疑"是不是全在测边界条件 / 只有内部 hook 才能触发":
|
||||||
|
|
||||||
@@ -130,6 +130,10 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
|||||||
|
|
||||||
第四十五批追加:#69(README 中英"审批通过后自动构建**并部署**"过度承诺——数据模型无目标服务器、部署实为"选用该镜像";① 轻);#70(kaniko 拉内建底座默认 HTTPS——`--insecure` 只覆盖推,任何 `FROM registry.felis.svc:5000/…` 构建必败 ①);#71(kaniko 以 drop-ALL 解包底座层,chown 必败——任何非 scratch 底座必败 ①);#72(trivy 扫 jar 必拉 Java DB、被 egress 锁拒绝——含 jar 即所有真实模组包的构建必败 ①);#73(bootstrap 测试在跑 k3s 的主机上必假失败 ④ 工具);#74(console 断连测试读写竞态 flaky ④ 工具)。
|
第四十五批追加:#69(README 中英"审批通过后自动构建**并部署**"过度承诺——数据模型无目标服务器、部署实为"选用该镜像";① 轻);#70(kaniko 拉内建底座默认 HTTPS——`--insecure` 只覆盖推,任何 `FROM registry.felis.svc:5000/…` 构建必败 ①);#71(kaniko 以 drop-ALL 解包底座层,chown 必败——任何非 scratch 底座必败 ①);#72(trivy 扫 jar 必拉 Java DB、被 egress 锁拒绝——含 jar 即所有真实模组包的构建必败 ①);#73(bootstrap 测试在跑 k3s 的主机上必假失败 ④ 工具);#74(console 断连测试读写竞态 flaky ④ 工具)。
|
||||||
|
|
||||||
|
第四十六批追加:#75(提交上传面无上限——pending 无个数上限、无存储预算、create/upload 无节流;一个账号可无限堆积上下文刷爆 uploads PVC ①);#76(提交无撤回/管理员删除路径——提交者无法自救、运维无法回收占用 ①);#77(未完成引导的会话触发受保护操作 → 面板显示"无权执行此操作"而非送往 /setup;tracker #8 ①);#78(create-if-absent 使新增 CR 字段在已装机永不落地——lobby 无 RCON 故控制台死、玩家数恒 0;tracker #1 ②);#79(troubleshooting [INERT] 图例指向已不存在字段 ④ 文档)。
|
||||||
|
|
||||||
|
第四十七批追加:无新缺陷——首个稳定 tag `v0.1.0` 的发布链与 release 安装/升级通道全实弹(release 二进制下载 → 无 checkout 全量安装 2m17s 零 fail/零 warn → `felis update` 双向报告;证据见该批节)。
|
||||||
|
|
||||||
## 已验证事实(正向清单)
|
## 已验证事实(正向清单)
|
||||||
|
|
||||||
- 安装→hook 发码→Owner 绑定→passkey(虚拟认证器)→面板管理员全链路 ✅
|
- 安装→hook 发码→Owner 绑定→passkey(虚拟认证器)→面板管理员全链路 ✅
|
||||||
@@ -851,6 +855,80 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
|||||||
- 升级滚动撞 kubelet **ephemeral-storage 压力**:auditfix64 滚动时新 api/operator pod `Pending 4m46s`(事件 `untolerated taint(s)`),02:58:12 kubelet eviction manager 回收后自动调度成功。诱因 = 构建把根盘压到 89%(docker 构建缓存 + kaniko/emptyDir 临时层);`docker builder prune -af` 两清共回收 ~11G,余 11G free。生产清单:根盘 ≥40G + 升级前清构建缓存(与批 44 注记合并)。
|
- 升级滚动撞 kubelet **ephemeral-storage 压力**:auditfix64 滚动时新 api/operator pod `Pending 4m46s`(事件 `untolerated taint(s)`),02:58:12 kubelet eviction manager 回收后自动调度成功。诱因 = 构建把根盘压到 89%(docker 构建缓存 + kaniko/emptyDir 临时层);`docker builder prune -af` 两清共回收 ~11G,余 11G free。生产清单:根盘 ≥40G + 升级前清构建缓存(与批 44 注记合并)。
|
||||||
- 中间失败构建留下的 registry 镜像(`user-uploads/sub-83f6…`、`sub-171e…` = 已推未过门;`sub-c006…` = capstone 产物)留档;测试服 `test-one` 已回 `felis/paper:demo` + Stopped;docker 停。
|
- 中间失败构建留下的 registry 镜像(`user-uploads/sub-83f6…`、`sub-171e…` = 已推未过门;`sub-c006…` = capstone 产物)留档;测试服 `test-one` 已回 `felis/paper:demo` + Stopped;docker 停。
|
||||||
|
|
||||||
|
### 本轮新增真机证据(第四十六批:上传面 #75/#76 真机闭环 + tracker #8/#1 收口 #77/#78 + 文案 #79 + 发布链 rc smoke)
|
||||||
|
|
||||||
|
**零、现场**:代码三条 commit 先落(`43699b4` #77 面板跳转、`c57daaf` #78 converge、`b9ebc87` #79 INERT 文案),随后按 `b9ebc87` 重建镜像 = `registry.felis.svc:5000/felis/felis:auditfix77`(`v0.0.0+fix77`;宿主源 `10.43.182.43:5000`,docker build → push),控制面 api/operator + minecraft ns `felis-reaper` CronJob + api/operator 的 `FELIS_IMAGE` 四处成对齐;`/opt/felis/src` = b9ebc87 快照(上一版 `src.bak46`);宿主 drill 二进制 `/root/felis-fix77.bin`(Mac 侧 `GOOS=linux GOARCH=arm64` 交叉编译,`-X main.version=v0.0.0+fix77`)。
|
||||||
|
|
||||||
|
**一、#75 真机闭环(配额/限流,owner 会话,auditfix77)**——留档 `/root/probe77/create-lane.log`、`lane2.log`:
|
||||||
|
- **create 失败不烧窗口**:坏 JSON → 400,**同一秒**合法 create → 201(`sub-811427aa7dcf6807`)——`release` 路径生效;
|
||||||
|
- **create 冷却**:同用户 30s 内第二条 → 429 `submission_cooldown`;
|
||||||
|
- **pending 上限**:攒到 5 条 pending → 第 6 条 → 403 `submission_quota_exceeded`;
|
||||||
|
- **upload 失败不烧窗口**:对已审核行上传 → 409 `already_reviewed`,**同一秒**对 pending 行上传 → 200;
|
||||||
|
- **upload 冷却**:15s 内第二次 → 429 `submission_cooldown`;等 16s → 200;
|
||||||
|
- **存储预算**:把 S2 的 blob 稀疏 `truncate` 到让 owner 已存字节 = 2GiB−100B(`blocks=8`,不占盘)→ 上传 → 403 `submission_quota_exceeded`;管理员删除 S2(行 + 目录双清)后 → 同一上传 → 200(预算即时释放)。
|
||||||
|
- 口径:预算按 blob 的实际占用聚合(`Blobs.Size` 遍历汇总,正是"用久了就超"的同一条读取路径);稀疏垫付只是把"已经存了 2GiB 的用户"这一状态合成出来。
|
||||||
|
|
||||||
|
**二、#76 真机闭环(撤回 + 管理员删除)**——留档 `/root/probe77/lane3.log`:
|
||||||
|
- 撤回自己 pending → 200;DB 行 = 0、uploads 目录 = gone;重复撤回 → 404;对已审核 → 409 `already_reviewed`;对他人 id → 404 `not_found`(owner 判定先于状态判定,不泄露他人行状态);
|
||||||
|
- 管理员删除(pending)→ 200;重复 → 404;行与 blob 双清(lane2 的 S2 同证:`s2 rows=0 dir=gone`);
|
||||||
|
- 面板侧 CDP:**撤回两步确认**(展开行 → 「撤回提交」→「确认撤回」→ 行消失)与 **admin 队列每行两步删除**(trash →「确认删除」→ 行消失、计数回落)双双走过;截图 `/tmp/withdraw-{1,2}-*.png`、`/tmp/admdelete-{1,2}-*.png`(Mac),脚本 `/tmp/cdp-withdraw76.js`、`/tmp/cdp-admdelete76.js`。
|
||||||
|
- 真机注记(非缺陷,分级 ZT 的对照实证):admin 面只在 `op.console.<root>` 主机可用——同一 owner 会话在玩家面板主机上 `is_admin=false`(admin 路由 403),在 `op.console.<root>` 上 `is_admin=true`(200);面板按要求显示「无权访问」而不是假装能点。
|
||||||
|
|
||||||
|
**三、#77(tracker #8,`43699b4`)真机闭环:锁定会话 → /setup**
|
||||||
|
- 铸一个未完成引导会话(SQL:`sha256('drill-locked-77')` → `sessions`,用户 `3f2c1b0a-…`,`email_verified=f`、无 passkey)→ `GET /me` 200、`GET /me/submissions` → 403 `setup_required`(后端本就正确);
|
||||||
|
- CDP 用该 cookie 打开 `/submissions`:页面请求 `/me/submissions` 收 403(code=`setup_required`)→ **自动跳转 `/setup`**(`location.href` 实测)→ 向导渲染「初始化你的账户 / 第一步 · 填写邮箱」(截图 `/tmp/setup8-redirect.png`;网络事件 = 200/403(/setup/status 200) 序列在脚本输出里)。
|
||||||
|
- 顺带确认:Dashboard 首屏三请求(`/me`、`/me/servers`、`account/link/start`)都是 `SetupAllowed`——所以补丁前用户是在"点受保护操作"时才撞到那句误报「无权执行此操作」。
|
||||||
|
|
||||||
|
**四、#78(tracker #1,`c57daaf`)真机演练:`felis converge`**
|
||||||
|
- 现场先剥后补:`kubectl patch` 清空 lobby `spec.rcon` 与 login `spec.startup.healthHTTPPort`(复刻老装机 CR 缺后加字段的状态)→ `/root/felis-fix77.bin converge` → 两条全部填回(`rcon{enabled:true,secretRef:lobby-rcon/password}`、`healthHTTPPort:8080`);**第二次运行 → 双双 `already converged`**(幂等);operator 侧滚动收尾,login/lobby 回 Running/Ready(日志 `converge-{1,2}.log`、`crs-before.yaml`)。
|
||||||
|
- 语义边界(单测):非零值一律不碰(运维自换的 secretRef 生还)、缺席 CR 只提示(创建仍归 setup)、占名非系统角色 CR 拒绝收敛、未配镜像跳过。
|
||||||
|
|
||||||
|
**五、#79(`b9ebc87`,文档)**:troubleshooting 的 `[INERT]` 图例仍写「§12 列着唯一仍适用的字段」,而唯一候选 `spec.storage.retainOnDelete` 早已移除、§12 自述「每个字段都有 controller 读」。改为「今天没有字段处于该状态」并指向 §13 的移除记录。
|
||||||
|
|
||||||
|
**六、发布链 rc smoke(tag `v0.1.0-rc1` → release run `35949621233`)**:
|
||||||
|
**六、发布链 rc smoke(tag `v0.1.0-rc1` → release run `35949621233`):首次全绿**
|
||||||
|
- run 步骤实况:go vet/test ✅ → `plugins/test.sh` ✅ → `plugins/test-mods.sh` ✅(**release.yml 史上第一次真正跑这三关**)→ buildx 双架构构建 ✅ → **stamp 校验** ✅(`felis v0.1.0-rc1` + arm64 ELF 断言)→ publish ✅;
|
||||||
|
- 产物释出:`felis-linux-amd64`(61,378,722 B)与 `felis-linux-arm64`(57,344,162 B)双资产,release 标记 `prerelease=true`;
|
||||||
|
- **prerelease 语义复核**:`GET /repos/FelisMC/Felis/releases/latest` → **404**(RC 没有顶掉 latest——正是 release.yml 注释里防的那件事);bootstrap 默认通道在无 stable 时按设计给出显式指引后 die、`felis update` 的 404 报错可读——两个行为本批均实测;
|
||||||
|
- 产物级复验(目标架构实机):arm64 资产 scp 到 VM 执行 → `felis v0.1.0-rc1`(`go1.26.8 linux/arm64`),且直接可用:`/root/felis-rc1.bin converge` 打现集群 → 双服 `already converged`;
|
||||||
|
- 留档:`/root/felis-rc1.bin`;asset 副本 Mac `/tmp/rc1b/felis-linux-arm64`。
|
||||||
|
- 剩余(产物决定,未代拍板):仓库仍无 **stable** release → 新装走默认 release 通道会以指引性报错 die(提示改 dev 或等 stable);切首个稳定 tag(如 `v0.1.0`)即让默认通道与 `felis update` 真正上线,建议维护者择时执行。
|
||||||
|
|
||||||
|
**七、运维注记(非缺陷)**
|
||||||
|
- pg_hba 的 `host felis_pgint felis 127.0.0.1/32 scram-sha-256` 行**再次丢失**(批 31 补过一次)——补回并 reload 后 pgint 才能连。重装/动过 PG 后先查这条(已写进速查)。
|
||||||
|
- 升级域注记:#75/#76 的 pgint 新断言(`CountPendingSubmissionsBy`、`DeletePendingSubmission`/`DeleteSubmission` CAS)首次上真 PG:**17/17 全绿**。
|
||||||
|
- 根盘:镜像构建后 83% → `docker builder prune -af` 回收 3.5G,docker 停回 inactive。
|
||||||
|
- CI:`43699b4` success;`c57daaf` 被后一 commit 的并发策略取消(同分支 cancel-in-progress),其树被 `b9ebc87` 的 success 完整覆盖;`b9ebc87` success。
|
||||||
|
- tracker 收编:**关** #10/#20/#21/#22(#20→`d9246dd`、#21→`f5a76cf`、#22→`3f2b28d`、#10→`23792d6`,均附证据评论)+ **关** #1/#8(本轮实现并真机验证);**注记**(保持打开作老装机待办)#2/#3/#4;**留存** #9/#12/#13/#15(真增强,超出生产可用主干,未动)。
|
||||||
|
|
||||||
|
### 本轮新增真机证据(第四十七批:首个稳定版 `v0.1.0` 发布链全实弹 + release 通道无 checkout 全量安装 + `felis update` 双向报告)
|
||||||
|
|
||||||
|
**零、现场**:切首个稳定 tag `v0.1.0` → `b9c97ff`(annotated),release run `35950722509` 全绿,产物 `felis-linux-amd64` 61,378,722 B / `felis-linux-arm64` 57,344,162 B;`GET /releases/latest` 现解析到 `v0.1.0`(`prerelease=false`),`v0.1.0-rc1` 保持 Pre-release。VM 侧事件前快照 `/root/probe77/pre77/`(host toml + deploys + crs + `hostbin.sha`=`106c4c9e…`);引导函数副本 `/root/probe77/bootstrap-funcs.sh`(删 `main "$@"` 行、可 source)与完整版 `/root/probe77/bootstrap-full.sh`(sha256 `89d0181d…` = 仓库 `deploy/bootstrap.sh` 逐字节)。
|
||||||
|
|
||||||
|
**一、Stage 1 — release 二进制下载(真实 github_api + asset + 原子安装)**:
|
||||||
|
- `resolve_install_ref`(release 通道)→ `REF=v0.1.0`;
|
||||||
|
- `download_release_binary`:宿主二进制 `felis v0.0.0+gunknown`(sha `106c4c9e…`)→ `felis v0.1.0`(sha `a64f32c5…`,57,344,162 B 与 arm64 资产一致);
|
||||||
|
- 复跑收敛:`host binary is already v0.1.0; skipping the download`(零 API、零下载)。留档 `stage1.log`、`stage1-release-download.sh`。
|
||||||
|
|
||||||
|
**二、Stage 2 — 无 checkout 全量安装(真实 `curl|bash` 形态;本批主线)**:
|
||||||
|
- 配方:`systemctl stop felis-velocity` → `mv /opt/felis/src src.bak47`(全程无 checkout)→ `bash < bootstrap-full.sh`(stdin 形态),`FELIS_IMAGE=…/felis/felis:v0.1.0`、`FELIS_WORLDS_HOST_PATH=/var/lib/rancher/k3s/storage`(对齐在册 reaper 形态);
|
||||||
|
- 结果:**rc=0、2m17s、`[fail]`=0、零 `[warn]`**(留档 `stage2.log`、`stage2-run.sh`)。
|
||||||
|
- 关键路径逐条为真:`use_release_binary` 命中 → 跳过下载(`HAVE_PREBUILT_BINARY=1`)→ `build_image_from_binary`(宿主二进制裹 distroless 镜像并导入 k3s)→ **`game_stack_source` 走 `unpacking the embedded game-stack sources (no checkout on this host)`**(二进制内嵌 tar 解包)→ limbo/lobby/paper 构建(Paper 26.3-38)→ 四镜像全 mirror 进内部 registry(实测 tags:`felis` 增 `v0.1.0`、limbo/lobby/paper = `demo`)→ api/operator 收敛到 `felis:v0.1.0` 且 rollout 全绿 → **minecraft ns `felis-reaper` CronJob 模板同步为 `felis:v0.1.0`** → 既有系统服 login/lobby 滚到新镜像 Running。
|
||||||
|
- 数据面复验:`users=16`、`servers=2`(resolvecheck/test-one)不变;面板 `https 200`;`/etc/felis/felis.host.toml` 与事件前快照**逐字节一致**(手改保留承诺实测);`bootstrap.done` 更新至 `03:32:18Z`。
|
||||||
|
- 收尾:`src.bak47` 复原为 `/opt/felis/src`;docker 停回 inactive。
|
||||||
|
|
||||||
|
**三、Stage 3 — `felis update` 报告(双向对照)**:
|
||||||
|
- `v0.1.0-rc1` 二进制:`felis-api v0.1.0-rc1 -> v0.1.0 update available (notify)`;velocity `3.5.1 -> 4.2.0 (notify)`;附 apply 命令(重跑安装器 + 私仓 token 指引文案);
|
||||||
|
- 现装 `v0.1.0` 二进制:`felis-api v0.1.0 up to date`——锤实「felis-api 已装版本 = 运行中二进制自身版本」(panel 内嵌,无独立版本可探);
|
||||||
|
- velocity minor(3.5.x→4.2.0)按设计走 notify;installer 只取 pinned minor 的最新 build。
|
||||||
|
|
||||||
|
**四、运维注记(非缺陷)**
|
||||||
|
- pg_hba `felis_pgint` 行又被重装清掉(第三轮)——**根因定位**:前两轮的补回位置落在 `# BEGIN/END FELIS MANAGED HBA` **块内**,重写段的 skip 对块内照删;本轮改为插在 `# END FELIS MANAGED HBA` 之后(块外,清理条件 `$2==felis` 不命中),下轮重装可复检;速查已注明正确插入位。
|
||||||
|
- 控制面镜像引用:`auditfix77` → `felis:v0.1.0`(release 二进制包裹);回退面 = 旧 `auditfix77` 仍在宿主 docker 与 registry;registry 的 `v0.1.0` tag 保留为 kubelet 回拉源。
|
||||||
|
- 演练后 VM 复位:docker inactive、k3s/felis-velocity active、test-one 仍 Stopped。
|
||||||
|
|
||||||
|
**五、结论**:stable release 通道端到端闭合——「tag → CI 双资产 → `/releases/latest` → 无 checkout 全量安装(embedded 资产 + registry mirror)→ 控制面/游戏栈全绿 → `felis update` 报告」全链实弹,无新缺陷。
|
||||||
|
|
||||||
## 结论:离"生产可用"还差什么(按优先级)
|
## 结论:离"生产可用"还差什么(按优先级)
|
||||||
|
|
||||||
1. ~~构建链路的上下文通道~~ ✅ **已修**(`f79e5eb`/`02fd2de`,真机全链路含拉回校验;Trivy DB 需按 §8e 镜像一次)。
|
1. ~~构建链路的上下文通道~~ ✅ **已修**(`f79e5eb`/`02fd2de`,真机全链路含拉回校验;Trivy DB 需按 §8e 镜像一次)。
|
||||||
@@ -858,7 +936,7 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
|||||||
3. ~~PG 级契约测试~~ ✅ **已落地**(`2a55a0d`):`internal/pgint`(`-tags pgint`,需 `FELIS_TEST_PG_URL` 指向名字含 `pgint` 的库,harness 会 drop schema + 重放真实迁移)已覆盖会话/OTP/op-login/绑定码/submission/build/owner 角色;**首跑即抓到 #20**(索引与 ErrEmailTaken 从未存在)。运行方式见 CONTRIBUTING.md。
|
3. ~~PG 级契约测试~~ ✅ **已落地**(`2a55a0d`):`internal/pgint`(`-tags pgint`,需 `FELIS_TEST_PG_URL` 指向名字含 `pgint` 的库,harness 会 drop schema + 重放真实迁移)已覆盖会话/OTP/op-login/绑定码/submission/build/owner 角色;**首跑即抓到 #20**(索引与 ErrEmailTaken 从未存在)。运行方式见 CONTRIBUTING.md。
|
||||||
4. ~~面板把 /jobs 显示出来~~ ✅ **已完成**(`97a64c8`,备份页「最近操作」卡,正是它把 #35 暴露出来的)。
|
4. ~~面板把 /jobs 显示出来~~ ✅ **已完成**(`97a64c8`,备份页「最近操作」卡,正是它把 #35 暴露出来的)。
|
||||||
5. ~~多节点回收~~ ✅ **已修**(`daf7602`:`--reaper-node` → CronJob pod `kubernetes.io/hostname` nodeSelector,真机 render/dry-run/收敛 diff 三连;单节点部署不传即维持原状)。附带核销缺陷 #38(渲染提示里过期的 uid-1000/setfacl 指导)。
|
5. ~~多节点回收~~ ✅ **已修**(`daf7602`:`--reaper-node` → CronJob pod `kubernetes.io/hostname` nodeSelector,真机 render/dry-run/收敛 diff 三连;单节点部署不传即维持原状)。附带核销缺陷 #38(渲染提示里过期的 uid-1000/setfacl 指导)。
|
||||||
6. ~~告警~~ ✅ **已完成**(`94f71ee` + `43df08b` + 第二十七批实弹演练:真实构建失败 → pending → 08:33:14Z firing;规则随 `deploy/alerts/` 交付)。
|
6. ~~告警~~ ✅ **已完成**(`94f71ee` + `43df08b` + 第二十七批实弹演练:真实构建失败 → pending → 08:33:14Z firing;规则随 `deploy/alerts/` 交付)。**2026-09-24 补齐无 Prometheus 的告警面**(`d17524c`):主机 `felis-watchdog.timer` 每 2 分钟巡检控制面/登录门/系统服/失败服/Job/reaper/节点/PG/代理/DB 备份/磁盘/内存,按持续时长门限给平台所有者发邮件;operator 增加 `felis_server_phase`、`felis_build_info`、卡死存活探针,规则新增 `felis.platform.rules`/`felis.jobs.rules`(promtool 22 条全过)。真机:felis-api 缩到 0 → 5 分钟后告警邮件落地("Felis 严重告警…1 项异常"),恢复 10 分钟后收到恢复邮件。
|
||||||
7. ~~玩家可见的构建结果~~ ✅ **已修**(`72c4aa3`,缺陷 #36:列表路由附 `build_status`/`build_error`,面板「我的提交」展开行呈现,真机双例验证)。
|
7. ~~玩家可见的构建结果~~ ✅ **已修**(`72c4aa3`,缺陷 #36:列表路由附 `build_status`/`build_error`,面板「我的提交」展开行呈现,真机双例验证)。
|
||||||
8. ~~面板文件编辑器入口~~ ✅ **已补**(`0a36b3f`,缺口补齐,真机 CDP 全链)。
|
8. ~~面板文件编辑器入口~~ ✅ **已补**(`0a36b3f`,缺口补齐,真机 CDP 全链)。
|
||||||
9. ~~fleet 系统服务死操作~~ ✅ **已修**(`2f90851`,缺陷 #37)。
|
9. ~~fleet 系统服务死操作~~ ✅ **已修**(`2f90851`,缺陷 #37)。
|
||||||
@@ -880,6 +958,12 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
|||||||
25. ~~内建底座构建(`FROM registry.felis.svc:5000/…`)~~ ✅ **已修并真机闭环**(第四十五批三连修:#70 拉取缺 `--insecure-pull`、#71 drop-ALL 卡解包、#72 trivy Java DB 未镜像——"真实模组包"形态的三个必踩点;修复后 paper 底座构建 27s Complete、`paper.jar` 扫描 0 漏洞)。
|
25. ~~内建底座构建(`FROM registry.felis.svc:5000/…`)~~ ✅ **已修并真机闭环**(第四十五批三连修:#70 拉取缺 `--insecure-pull`、#71 drop-ALL 卡解包、#72 trivy Java DB 未镜像——"真实模组包"形态的三个必踩点;修复后 paper 底座构建 27s Complete、`paper.jar` 扫描 0 漏洞)。
|
||||||
26. ~~提交 → 装服 → 起服 闭环~~ ✅ **已演练**(第四十五批 capstone:构建产物 PATCH 为服务器镜像〔白名单门通过〕→ wake → Running 1/1 → 用户 marker 可读 + `Done (4.678s)!`;test-one 已复原为 paper:demo/Stopped)。
|
26. ~~提交 → 装服 → 起服 闭环~~ ✅ **已演练**(第四十五批 capstone:构建产物 PATCH 为服务器镜像〔白名单门通过〕→ wake → Running 1/1 → 用户 marker 可读 + `Done (4.678s)!`;test-one 已复原为 paper:demo/Stopped)。
|
||||||
27. ~~文档与测试工具面~~ ✅ **已修**(第四十五批:#69 README 部署承诺对齐实现;#73 bootstrap 测试密闭化〔VM 140 PASS〕;#74 console 断连测试消抖;CI 全绿)。
|
27. ~~文档与测试工具面~~ ✅ **已修**(第四十五批:#69 README 部署承诺对齐实现;#73 bootstrap 测试密闭化〔VM 140 PASS〕;#74 console 断连测试消抖;CI 全绿)。
|
||||||
|
28. ~~提交上传面的上限与生命周期~~ ✅ **已修并真机闭环**(第四十六批:#75 per-user create/upload 冷却〔429〕、pending ≤5〔403〕、2GiB 存储预算〔403〕;#76 撤回〔owner+pending CAS〕与管理员删除〔行+blob 双清〕;面板两步确认 CDP 全绿)。
|
||||||
|
29. ~~未完成引导的导航(tracker #8)~~ ✅ **已修并真机闭环**(第四十六批 #77:`403 setup_required` → 自动 `/setup`;CDP 复验)。
|
||||||
|
30. ~~已装机系统服的新增 CR 字段(tracker #1)~~ ✅ **已修并真机闭环**(第四十六批 #78:`sudo felis converge`——零值才填、非零不覆写;剥字段→填回→幂等三连真机过)。
|
||||||
|
31. ~~troubleshooting `[INERT]` 图例~~ ✅ **已修**(第四十六批 #79,文档级)。
|
||||||
|
32. ~~稳定版发布与 release 安装/升级通道~~ ✅ **已实证**(第四十七批:tag `v0.1.0`(`b9c97ff`)+ release run `35950722509` 双资产、`/releases/latest` 解析到 v0.1.0;无 checkout 全量安装 2m17s / 零 fail / 零 warn、registry mirror 四镜像、reaper CronJob 对齐 `felis:v0.1.0`;`felis update` 双向报告〔rc1→v0.1.0 notify / v0.1.0 up to date〕)。
|
||||||
|
33. ~~世界归档与数据库备份只在本机~~ ✅ **已修并真机闭环**(2026-09-24:`felis offsite` + `felis-offsite.timer` 每小时把世界归档与 DB bundle 以 AES-256-GCM 分段加密推到 S3 兼容桶,`world_backups.offsite_at` 记账〔迁移 0024〕;配了 `[offsite]` 后 reaper 只在归档的异地副本确认后才删 PVC;watchdog 12 小时无成功同步即告警;安装器 `FELIS_OFFSITE_*` 生成密钥并在收尾横幅要求离机保存。真机(MinIO):首次同步 8 世界 1.2 GiB + 12 bundle、两条"记录在册但盘上已无"的历史归档如实列出;`fetch-db latest` sha256 与本机一致、错误密钥拒绝且不留半成品;`fetch-worlds` 取回被挪走的归档 sha256 一致;reaper 演练〔20 天空闲世界〕第一轮 `awaiting_offsite=1` 且 PVC 保留 → 同步后第二轮 `reaped=1`、PVC 删除、所有权释放;watchdog 把 status 改成 35 小时前 → `[warning] offsite`。演练装置已清理)。
|
||||||
|
|
||||||
## 剩余待演练队列(截至第四十三批)
|
## 剩余待演练队列(截至第四十三批)
|
||||||
|
|
||||||
@@ -925,3 +1009,5 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
|||||||
- 第四十三批追加(#68)留档:控制面 api/operator/reaper 镜像 + api/operator `FELIS_IMAGE` = `registry.felis.svc:5000/felis/felis:auditfix63`(`felis version` = `v0.0.0+fix63`;升级时 `set image` 与 `FELIS_IMAGE` env 必须成对改,否则构建 Job 的 fetch 容器会悄悄用旧镜像);`/opt/felis/src` = `b76d0ac` 快照(上一版 `src.bak43` = `2755e41`);复刻 Job `/root/job68-new.json`(由红证据 Job `build-bld-1790187851749257160` 派生:换名 `build-retry68`、剥 controller uid 标签与 selector、镜像改 auditfix63、复用 `felis-service-token` Secret);重演验证法:`kubectl -n felis scale deploy/felis-api --replicas=0` → `kubectl -n felis-build create -f /root/job68-new.json` → `kubectl -n felis-build logs <pod> -c context-fetch`(应见 `connection refused; retrying`)→ `scale --replicas=1` → Job 收敛 `Complete`;对照:api 在线直构成功且 fetch 日志无重试行。
|
- 第四十三批追加(#68)留档:控制面 api/operator/reaper 镜像 + api/operator `FELIS_IMAGE` = `registry.felis.svc:5000/felis/felis:auditfix63`(`felis version` = `v0.0.0+fix63`;升级时 `set image` 与 `FELIS_IMAGE` env 必须成对改,否则构建 Job 的 fetch 容器会悄悄用旧镜像);`/opt/felis/src` = `b76d0ac` 快照(上一版 `src.bak43` = `2755e41`);复刻 Job `/root/job68-new.json`(由红证据 Job `build-bld-1790187851749257160` 派生:换名 `build-retry68`、剥 controller uid 标签与 selector、镜像改 auditfix63、复用 `felis-service-token` Secret);重演验证法:`kubectl -n felis scale deploy/felis-api --replicas=0` → `kubectl -n felis-build create -f /root/job68-new.json` → `kubectl -n felis-build logs <pod> -c context-fetch`(应见 `connection refused; retrying`)→ `scale --replicas=1` → Job 收敛 `Complete`;对照:api 在线直构成功且 fetch 日志无重试行。
|
||||||
- 第四十四批工具与留档:200MiB 上下文构造 = 5×40MiB `head -c 41943040 /dev/urandom` + `Dockerfile`(`FROM scratch` + `COPY payload /payload`),`tar -czf /root/bigctx.tar.gz Dockerfile payload`;上传 = `curl -sk -b /tmp/owner-jar43.txt -X POST -H "Content-Type: application/gzip" --data-binary @/root/bigctx.tar.gz https://127.0.0.1:30443/api/v1/me/submissions/<id>/context`;采样 = `k3s crictl stats`(此版 **无 `--no-trunc`**,列解析取 `$2=NAME $4=MEM`);docker 缓存清理 = `systemctl start docker && docker builder prune -af && systemctl stop docker`;留档 `/root/stats44.log`、`/root/stats44-build.log`、提交 `sub-cc160b3ad19080f9`、镜像 `e2e/conc-b44-{1,2,3}`。
|
- 第四十四批工具与留档:200MiB 上下文构造 = 5×40MiB `head -c 41943040 /dev/urandom` + `Dockerfile`(`FROM scratch` + `COPY payload /payload`),`tar -czf /root/bigctx.tar.gz Dockerfile payload`;上传 = `curl -sk -b /tmp/owner-jar43.txt -X POST -H "Content-Type: application/gzip" --data-binary @/root/bigctx.tar.gz https://127.0.0.1:30443/api/v1/me/submissions/<id>/context`;采样 = `k3s crictl stats`(此版 **无 `--no-trunc`**,列解析取 `$2=NAME $4=MEM`);docker 缓存清理 = `systemctl start docker && docker builder prune -af && systemctl stop docker`;留档 `/root/stats44.log`、`/root/stats44-build.log`、提交 `sub-cc160b3ad19080f9`、镜像 `e2e/conc-b44-{1,2,3}`。
|
||||||
- 第四十五批工具与留档:从平台底座构建的配方 = 提交上下文含 `Dockerfile`(`FROM registry.felis.svc:5000/felis/paper:demo` + `COPY marker.txt /felis-probe-marker.txt`);装服验证 = `PATCH /api/v1/servers/test-one {"image":"registry.felis.svc:5000/user-uploads/<sub>:latest"}` → `POST /servers/test-one/wake` → `kubectl -n minecraft exec test-one-0 -- cat /felis-probe-marker.txt`;trivy 双 DB 镜像 = `mirror/trivy-db:2` + `mirror/trivy-java-db:1`(Java DB digest `5766dfbb…`;两键在 `/etc/felis/felis.{host,pod}.toml` 与 felis-config Secret 双副本);`bootstrap_test.sh` 需在 Linux 跑(VM 留档 `/root/btest72/`,应 140 PASS);本批提交样本 `sub-{b45d416f4af811ef(无镜像),83f677e41acd80c3,171e8f967f0de3bc,c006cbd633317ccb,cc160b3ad19080f9}`;留档 `/root/probe44/`、`/root/probe44.tar.gz`、`/root/probe44*.sid`。
|
- 第四十五批工具与留档:从平台底座构建的配方 = 提交上下文含 `Dockerfile`(`FROM registry.felis.svc:5000/felis/paper:demo` + `COPY marker.txt /felis-probe-marker.txt`);装服验证 = `PATCH /api/v1/servers/test-one {"image":"registry.felis.svc:5000/user-uploads/<sub>:latest"}` → `POST /servers/test-one/wake` → `kubectl -n minecraft exec test-one-0 -- cat /felis-probe-marker.txt`;trivy 双 DB 镜像 = `mirror/trivy-db:2` + `mirror/trivy-java-db:1`(Java DB digest `5766dfbb…`;两键在 `/etc/felis/felis.{host,pod}.toml` 与 felis-config Secret 双副本);`bootstrap_test.sh` 需在 Linux 跑(VM 留档 `/root/btest72/`,应 140 PASS);本批提交样本 `sub-{b45d416f4af811ef(无镜像),83f677e41acd80c3,171e8f967f0de3bc,c006cbd633317ccb,cc160b3ad19080f9}`;留档 `/root/probe44/`、`/root/probe44.tar.gz`、`/root/probe44*.sid`。
|
||||||
|
- 第四十六批工具与留档:控制面三处 = `registry.felis.svc:5000/felis/felis:auditfix77`(宿主 docker 源 `10.43.182.43:5000/felis/felis:auditfix77`,`v0.0.0+fix77`;api/operator + minecraft ns `felis-reaper` CronJob + api/operator `FELIS_IMAGE` 四处成对改);`/opt/felis/src` = `b9ebc87` 快照(上一版 `src.bak46`);宿主 drill 二进制 `/root/felis-fix77.bin`(Mac 侧 `CGO_ENABLED=0 GOOS=linux GOARCH=arm64 go build -ldflags="-s -w -X main.version=v0.0.0+fix77" ./cmd/felis`;scp 到 IPv6 主机时主机位必须写 `root@[fdb2:…]` 方括号);#75/#76/#77/#78 留档 `/root/probe77/`(`create-lane.log`、`lane2.log`、`lane3.log`〔含玩家会话矩阵 + 分级 ZT 对照 + curl 7.76 在 `-H Host:` 下**不发 jar cookie**的坑——host 路由 drill 用显式 `-H "Cookie: felis_session=…"`〕、`converge-1.log`/`converge-2.log`、`crs-before.yaml`、各步请求/响应 json);面板 CDP 截图(Mac):`/tmp/setup8-redirect.png`、`/tmp/withdraw-{1,2}-*.png`、`/tmp/admdelete-{1,2}-*.png`,脚本 `/tmp/cdp-setup8.js`、`/tmp/cdp-withdraw76.js`、`/tmp/cdp-admdelete76.js`;锁定会话铸法 = `sha256('drill-locked-77')` 直插 `sessions`(用户 `3f2c1b0a-…`);玩家会话铸法 = `[email protected]` 邮件 OTP(码在 felis-api 日志 `no Mailer configured` 行;`edge-dis2` 是软删死账号,登录门按设计排除);pgint 前置:`pg_hba` 需 `host felis_pgint felis 127.0.0.1/32 scram-sha-256`(本批第二次丢失并补回);发布链 smoke = tag `v0.1.0-rc1`(release run `35949621233`),prerelease 不移动 `/releases/latest`——无 stable 时 bootstrap 默认通道给出显式指引后 die、`felis update` 404 报错可读;首个稳定 tag 是产物决定(未代拍板)。
|
||||||
|
- 第四十七批工具与留档:stable tag `v0.1.0` = `b9c97ff`(annotated;release run `35950722509`;资产 amd64 61,378,722 B / arm64 57,344,162 B;`/releases/latest` = v0.1.0、rc1 保持 prerelease)。VM `/root/probe77/`:`stage1-release-download.sh`+`stage1.log`(REF=v0.1.0;host bin sha `106c4c9e…`→`a64f32c5…`)、`stage2-run.sh`+`stage2.log`(无 checkout 全量安装 rc=0/2m17s/零 fail warn;"unpacking the embedded game-stack sources" 行)、`bootstrap-funcs.sh`(可 source 副本)、`bootstrap-full.sh`(完整版,sha `89d0181d…`)、`pre77/`(事件前快照);Stage 3 = `/root/felis-rc1.bin update --all`(rc1→v0.1.0)对照现装 `felis update --all`(v0.1.0 up to date);"无 checkout 主机"演练配方 = 停 felis-velocity → `mv /opt/felis/src src.bak47` → `bash < bootstrap-full.sh` → 复原 mv;pg_hba `felis_pgint` 行**必须插在 `# END FELIS MANAGED HBA` 之后**(块内会在重装时被重写段删除——三轮同根因;本轮已按块外补回)。
|
||||||
+7
-3
@@ -21,14 +21,18 @@
|
|||||||
# minutes. The FINAL stage is deliberately NOT pinned — it must stay on the target platform
|
# minutes. The FINAL stage is deliberately NOT pinned — it must stay on the target platform
|
||||||
# or the published arm64 image would carry amd64 layers. It contains only COPY, which
|
# or the published arm64 image would carry amd64 layers. It contains only COPY, which
|
||||||
# BuildKit performs itself, so it needs no QEMU either; adding a RUN there would.
|
# BuildKit performs itself, so it needs no QEMU either; adding a RUN there would.
|
||||||
FROM --platform=$BUILDPLATFORM node:22-bookworm AS panel
|
#
|
||||||
|
# Every base image here and in deploy/{limbo,lobby,paper} is pinned by digest, so a rebuild
|
||||||
|
# of one release uses the same bytes; .github/dependabot.yml proposes the bumps (tag and
|
||||||
|
# digest together).
|
||||||
|
FROM --platform=$BUILDPLATFORM node:22-bookworm@sha256:363e1587494626837fa7f9a23bdb453d13b0ff3c67c705c2805cfc69c2d2fad7 AS panel
|
||||||
WORKDIR /panel
|
WORKDIR /panel
|
||||||
COPY panel/package*.json ./
|
COPY panel/package*.json ./
|
||||||
RUN npm ci
|
RUN npm ci
|
||||||
COPY panel/ ./
|
COPY panel/ ./
|
||||||
RUN npm run build
|
RUN npm run build
|
||||||
|
|
||||||
FROM --platform=$BUILDPLATFORM golang:1.26 AS build
|
FROM --platform=$BUILDPLATFORM golang:1.26@sha256:6c2a5538f964f1c82f97ad14988bf05de100d922d159d0e398b54c7b0ca0c6c9 AS build
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
ARG TARGETOS=linux
|
ARG TARGETOS=linux
|
||||||
ARG TARGETARCH
|
ARG TARGETARCH
|
||||||
@@ -55,7 +59,7 @@ ARG FELIS_VERSION=dev
|
|||||||
RUN CGO_ENABLED=0 GOOS="$TARGETOS" GOARCH="${TARGETARCH:-$(go env GOARCH)}" \
|
RUN CGO_ENABLED=0 GOOS="$TARGETOS" GOARCH="${TARGETARCH:-$(go env GOARCH)}" \
|
||||||
go build -trimpath -ldflags="-s -w -X main.version=${FELIS_VERSION}" -o /out/felis ./cmd/felis
|
go build -trimpath -ldflags="-s -w -X main.version=${FELIS_VERSION}" -o /out/felis ./cmd/felis
|
||||||
|
|
||||||
FROM gcr.io/distroless/static-debian12:nonroot
|
FROM gcr.io/distroless/static-debian12:nonroot@sha256:afa5c872c891853ca7fcf1f12c3edb23f7eeef36189728842dd51042ff57f7ab
|
||||||
ENV PATH=/usr/local/bin:/usr/bin:/bin
|
ENV PATH=/usr/local/bin:/usr/bin:/bin
|
||||||
COPY --chmod=0755 --from=build /out/felis /usr/local/bin/felis
|
COPY --chmod=0755 --from=build /out/felis /usr/local/bin/felis
|
||||||
# distroless "nonroot" is uid 65532; the rendered PodSecurityContext pins
|
# distroless "nonroot" is uid 65532; the rendered PodSecurityContext pins
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
|||||||
- **即开即玩**:玩家尝试连接时自动唤醒服务器,空闲后自动休眠,像游戏主机一样省资源。
|
- **即开即玩**:玩家尝试连接时自动唤醒服务器,空闲后自动休眠,像游戏主机一样省资源。
|
||||||
- **Web 控制面板**:浏览器中查看服务器状态、在线玩家与资源用量,管理备份与恢复。
|
- **Web 控制面板**:浏览器中查看服务器状态、在线玩家与资源用量,管理备份与恢复。
|
||||||
- **备份与恢复**:一键把整服数据(世界、配置、插件/模组,即整个 /data 卷)打包进集群内的归档库,支持从任意备份点回滚;默认安装就已启用(归档 PVC 与路径由安装器一并生成)。
|
- **备份与恢复**:一键把整服数据(世界、配置、插件/模组,即整个 /data 卷)打包进集群内的归档库,支持从任意备份点回滚;默认安装就已启用(归档 PVC 与路径由安装器一并生成)。
|
||||||
|
- **控制面数据库备份**:账号、服务器归属、配额与存档索引所在的数据库每天自动备份,每次升级迁移前先快照,出错可用 `felis db restore` 整库原子回滚;面板「维护与备份」页显示备份是否新鲜(见 [故障排查 §16](docs/troubleshooting.md))。
|
||||||
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。
|
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。
|
||||||
- **多核心支持**:兼容 Paper、Fabric、Forge、NeoForge,经由 Velocity 代理统一入口。
|
- **多核心支持**:兼容 Paper、Fabric、Forge、NeoForge,经由 Velocity 代理统一入口。
|
||||||
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建;构建产物进入镜像白名单,可直接选用为服务器镜像完成部署。
|
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建;构建产物进入镜像白名单,可直接选用为服务器镜像完成部署。
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ Table of Contents
|
|||||||
- **Wake on Join**: Servers start automatically when a player connects, and stop when idle — like hibernate for your server.
|
- **Wake on Join**: Servers start automatically when a player connects, and stop when idle — like hibernate for your server.
|
||||||
- **Web Dashboard**: Monitor server status, online players, and resource usage from your browser, with backup and restore management.
|
- **Web Dashboard**: Monitor server status, online players, and resource usage from your browser, with backup and restore management.
|
||||||
- **Backup & Restore**: One-click snapshots of a server's whole data volume (worlds, config, plugins/mods — the entire /data volume) into the cluster's archive store, with rollback from any backup point — enabled by default (the installer renders the archive PVC and its path).
|
- **Backup & Restore**: One-click snapshots of a server's whole data volume (worlds, config, plugins/mods — the entire /data volume) into the cluster's archive store, with rollback from any backup point — enabled by default (the installer renders the archive PVC and its path).
|
||||||
|
- **Control-plane database backups**: The database holding accounts, server ownership, quotas and the archive index is backed up daily and snapshotted before every upgrade migrates it; `felis db restore` rolls it back atomically, and the panel's Maintenance & Backups page shows whether the newest backup is fresh (see [troubleshooting §16](docs/troubleshooting.md)).
|
||||||
- **World Reaper** (opt in): Worlds idle for more than 15 days are automatically backed up and removed to free disk space. Enable it by setting `FELIS_WORLDS_HOST_PATH` at install time (on k3s: `/var/lib/rancher/k3s/storage`); without it, no world is ever deleted.
|
- **World Reaper** (opt in): Worlds idle for more than 15 days are automatically backed up and removed to free disk space. Enable it by setting `FELIS_WORLDS_HOST_PATH` at install time (on k3s: `/var/lib/rancher/k3s/storage`); without it, no world is ever deleted.
|
||||||
- **Multi-core Support**: Compatible with Paper, Fabric, Forge, and NeoForge, federated behind a Velocity proxy.
|
- **Multi-core Support**: Compatible with Paper, Fabric, Forge, and NeoForge, federated behind a Velocity proxy.
|
||||||
- **Modpack Submission**: Players submit custom modpacks; admin approval triggers an automatic build, and the result is whitelisted as a server image you can select to deploy.
|
- **Modpack Submission**: Players submit custom modpacks; admin approval triggers an automatic build, and the result is whitelisted as a server image you can select to deploy.
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ var bootstrapAssets embed.FS
|
|||||||
// otherwise be baked into every felis binary. Keep them explicit — add a source
|
// otherwise be baked into every felis binary. Keep them explicit — add a source
|
||||||
// directory here, never a parent.
|
// directory here, never a parent.
|
||||||
//
|
//
|
||||||
|
//go:embed deploy/game-stack.lock
|
||||||
//go:embed deploy/limbo/Dockerfile deploy/limbo/entrypoint.sh
|
//go:embed deploy/limbo/Dockerfile deploy/limbo/entrypoint.sh
|
||||||
//go:embed deploy/lobby/Dockerfile deploy/lobby/entrypoint.sh
|
//go:embed deploy/lobby/Dockerfile deploy/lobby/entrypoint.sh
|
||||||
//go:embed deploy/paper/Dockerfile deploy/paper/entrypoint.sh
|
//go:embed deploy/paper/Dockerfile deploy/paper/entrypoint.sh
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ package felis
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"io/fs"
|
"io/fs"
|
||||||
|
"os"
|
||||||
"regexp"
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
@@ -205,6 +206,113 @@ func requireEmbedded(t *testing.T, path string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The lock file is the install's only source of upstream builds, and bootstrap.sh reads it
|
||||||
|
// with a strict KEY=value parser that dies on anything unexpected, so a malformed lock is a
|
||||||
|
// failed install on every host. Check the shipped copy the same way here.
|
||||||
|
func TestGameStackLockIsComplete(t *testing.T) {
|
||||||
|
lock := map[string]string{}
|
||||||
|
for line := range strings.SplitSeq(readGameStackFile(t, "deploy/game-stack.lock"), "\n") {
|
||||||
|
if line == "" || strings.HasPrefix(line, "#") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
k, v, ok := strings.Cut(line, "=")
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("not a KEY=value line: %q", line)
|
||||||
|
}
|
||||||
|
lock[k] = v
|
||||||
|
}
|
||||||
|
m := regexp.MustCompile(`GAME_STACK_LOCK_KEYS="([^"]*)"`).FindStringSubmatch(BootstrapScript())
|
||||||
|
if m == nil {
|
||||||
|
t.Fatal("bootstrap.sh no longer declares GAME_STACK_LOCK_KEYS")
|
||||||
|
}
|
||||||
|
keys := strings.Fields(m[1])
|
||||||
|
sha := regexp.MustCompile(`^[0-9a-f]{64}$`)
|
||||||
|
for _, k := range keys {
|
||||||
|
v, ok := lock[k]
|
||||||
|
if !ok || v == "" {
|
||||||
|
t.Errorf("game-stack.lock does not set %s", k)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if strings.HasSuffix(k, "_SHA256") && !sha.MatchString(v) {
|
||||||
|
t.Errorf("%s=%q is not a lowercase sha256", k, v)
|
||||||
|
}
|
||||||
|
// A moving URL pins nothing: the digest check would start failing the day
|
||||||
|
// upstream publishes the next build.
|
||||||
|
if strings.HasSuffix(k, "_URL") && strings.Contains(v, "lastSuccessfulBuild") {
|
||||||
|
t.Errorf("%s names a moving build: %s", k, v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for k := range lock {
|
||||||
|
if !strings.Contains(" "+m[1]+" ", " "+k+" ") {
|
||||||
|
t.Errorf("game-stack.lock sets %s, which bootstrap.sh refuses as an unknown key", k)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Fill's URLs are content-addressed; a lock whose digest disagrees with its own URL
|
||||||
|
// was edited by hand and half-way.
|
||||||
|
for _, name := range []string{"PAPER", "VELOCITY"} {
|
||||||
|
if !strings.Contains(lock[name+"_JAR_URL"], "/objects/"+lock[name+"_JAR_SHA256"]+"/") {
|
||||||
|
t.Errorf("%s_JAR_SHA256 is not the digest in %s_JAR_URL", name, name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !strings.Contains(lock["LIMBO_JAR_URL"], "-"+lock["MC_VERSION"]+".jar") {
|
||||||
|
t.Errorf("LIMBO_JAR_URL %s is not a Minecraft %s build", lock["LIMBO_JAR_URL"], lock["MC_VERSION"])
|
||||||
|
}
|
||||||
|
if !strings.Contains(lock["PAPER_JAR_URL"], "/paper-"+lock["MC_VERSION"]+"-") {
|
||||||
|
t.Errorf("PAPER_JAR_URL %s is not a Minecraft %s build; the lobby would not speak the login gate's protocol", lock["PAPER_JAR_URL"], lock["MC_VERSION"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Each downloaded jar's digest is a build-arg bootstrap.sh passes and the Dockerfile must
|
||||||
|
// both require and spend on the file it downloaded; docker only warns about an unknown
|
||||||
|
// --build-arg, so a renamed arg would ship an unchecked jar.
|
||||||
|
func TestGameStackDigestsReachTheImageBuilds(t *testing.T) {
|
||||||
|
script := BootstrapScript()
|
||||||
|
for _, c := range []struct{ dockerfile, arg, path string }{
|
||||||
|
{"deploy/limbo/Dockerfile", "LIMBO_JAR_SHA256", "/limbo/Limbo.jar"},
|
||||||
|
{"deploy/limbo/Dockerfile", "LIMBO_SCHEM_SHA256", "/limbo/spawn.schem"},
|
||||||
|
{"deploy/lobby/Dockerfile", "LUCKPERMS_JAR_SHA256", "/paper/plugins/LuckPerms.jar"},
|
||||||
|
} {
|
||||||
|
if !strings.Contains(script, "--build-arg "+c.arg+"=\"$"+c.arg+"\"") {
|
||||||
|
t.Errorf("bootstrap.sh never passes --build-arg %s", c.arg)
|
||||||
|
}
|
||||||
|
dockerfile := readGameStackFile(t, c.dockerfile)
|
||||||
|
if !strings.Contains(dockerfile, "ARG "+c.arg) {
|
||||||
|
t.Errorf("%s declares no ARG %s", c.dockerfile, c.arg)
|
||||||
|
}
|
||||||
|
if !strings.Contains(dockerfile, `echo "$`+c.arg+` `+c.path+`" | sha256sum -c`) {
|
||||||
|
t.Errorf("%s never verifies %s against %s", c.dockerfile, c.path, c.arg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A base image named by tag alone is whatever the tag points at on build day.
|
||||||
|
func TestDockerfileBaseImagesArePinnedByDigest(t *testing.T) {
|
||||||
|
root, err := os.ReadFile("Dockerfile")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
files := map[string]string{"Dockerfile": string(root)}
|
||||||
|
for _, name := range []string{"deploy/limbo/Dockerfile", "deploy/lobby/Dockerfile", "deploy/paper/Dockerfile"} {
|
||||||
|
files[name] = readGameStackFile(t, name)
|
||||||
|
}
|
||||||
|
pinned := regexp.MustCompile(`^FROM (--platform=\S+ )?\S+:\S+@sha256:[0-9a-f]{64}( AS \S+)?$`)
|
||||||
|
for name, body := range files {
|
||||||
|
n := 0
|
||||||
|
for line := range strings.SplitSeq(body, "\n") {
|
||||||
|
if !strings.HasPrefix(line, "FROM ") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
n++
|
||||||
|
if !pinned.MatchString(line) {
|
||||||
|
t.Errorf("%s: %q is not pinned by digest", name, line)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if n == 0 {
|
||||||
|
t.Errorf("%s has no FROM line", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func readGameStackFile(t *testing.T, name string) string {
|
func readGameStackFile(t *testing.T, name string) string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
b, err := gameStackAssets.ReadFile(name)
|
b, err := gameStackAssets.ReadFile(name)
|
||||||
|
|||||||
+209
-8
@@ -5,10 +5,12 @@ import (
|
|||||||
"flag"
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
|
"log/slog"
|
||||||
"net/http"
|
"net/http"
|
||||||
"os"
|
"os"
|
||||||
"regexp"
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
|
"sync/atomic"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/api"
|
"felis.lolicon.best/internal/api"
|
||||||
@@ -17,11 +19,15 @@ import (
|
|||||||
"felis.lolicon.best/internal/build"
|
"felis.lolicon.best/internal/build"
|
||||||
"felis.lolicon.best/internal/config"
|
"felis.lolicon.best/internal/config"
|
||||||
"felis.lolicon.best/internal/fileedit"
|
"felis.lolicon.best/internal/fileedit"
|
||||||
|
"felis.lolicon.best/internal/imagepin"
|
||||||
"felis.lolicon.best/internal/mail"
|
"felis.lolicon.best/internal/mail"
|
||||||
|
"felis.lolicon.best/internal/metrics"
|
||||||
"felis.lolicon.best/internal/naming"
|
"felis.lolicon.best/internal/naming"
|
||||||
"felis.lolicon.best/internal/panel"
|
"felis.lolicon.best/internal/panel"
|
||||||
"felis.lolicon.best/internal/passkey"
|
"felis.lolicon.best/internal/passkey"
|
||||||
"felis.lolicon.best/internal/platform"
|
"felis.lolicon.best/internal/platform"
|
||||||
|
"felis.lolicon.best/internal/reaper"
|
||||||
|
"felis.lolicon.best/internal/registryprune"
|
||||||
"felis.lolicon.best/internal/restore"
|
"felis.lolicon.best/internal/restore"
|
||||||
"felis.lolicon.best/internal/store"
|
"felis.lolicon.best/internal/store"
|
||||||
"felis.lolicon.best/internal/submit"
|
"felis.lolicon.best/internal/submit"
|
||||||
@@ -109,6 +115,8 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
metrics.SetBuildInfo("api", resolvedVersion())
|
||||||
|
|
||||||
token := os.Getenv("FELIS_SERVICE_TOKEN")
|
token := os.Getenv("FELIS_SERVICE_TOKEN")
|
||||||
if token == "" {
|
if token == "" {
|
||||||
fmt.Fprintln(stderr, "felis api: warning: FELIS_SERVICE_TOKEN unset — internal face will reject all callers")
|
fmt.Fprintln(stderr, "felis api: warning: FELIS_SERVICE_TOKEN unset — internal face will reject all callers")
|
||||||
@@ -147,11 +155,13 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
// The fetch initContainer runs THIS image's fetch-context entrypoint, so the
|
// The fetch initContainer runs THIS image's fetch-context entrypoint, so the
|
||||||
// build config carries the api's own image (the platform sets FELIS_IMAGE).
|
// build config carries the api's own image (the platform sets FELIS_IMAGE).
|
||||||
buildCfg.FelisImage = os.Getenv("FELIS_IMAGE")
|
buildCfg.FelisImage = os.Getenv("FELIS_IMAGE")
|
||||||
|
buildJobs := build.NewK8sJobs(cl, buildCfg)
|
||||||
builder := &build.Builder{
|
builder := &build.Builder{
|
||||||
Store: build.NewPGStore(drv.DB()),
|
Store: build.NewPGStore(drv.DB()),
|
||||||
Jobs: build.NewK8sJobs(cl, buildCfg),
|
Jobs: buildJobs,
|
||||||
Config: buildCfg,
|
Config: buildCfg,
|
||||||
}
|
}
|
||||||
|
go probeBuildUserNamespaces(ctx, buildJobs, buildCfg, stderr)
|
||||||
|
|
||||||
// User-modpack approval lane (user-directed extension over §16; see
|
// User-modpack approval lane (user-directed extension over §16; see
|
||||||
// internal/submit). An ordinary user may only SUBMIT a
|
// internal/submit). An ordinary user may only SUBMIT a
|
||||||
@@ -203,6 +213,13 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
ContextBaseURL: internalAPIBaseURL(),
|
ContextBaseURL: internalAPIBaseURL(),
|
||||||
Blobs: blobs,
|
Blobs: blobs,
|
||||||
}
|
}
|
||||||
|
if v := cfg.Registry.UserUploadsMaxBytes; v != "" {
|
||||||
|
if n, err := parseByteSize(v); err != nil || n <= 0 {
|
||||||
|
fmt.Fprintf(stderr, "felis api: [registry] user_uploads_max_bytes %q is not a positive size such as 4Gi; keeping the default\n", v)
|
||||||
|
} else {
|
||||||
|
submissions.MaxStoredBytesTotal = n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Restore subsystem (spec §7): the weak-SA restore Job mounts the target
|
// Restore subsystem (spec §7): the weak-SA restore Job mounts the target
|
||||||
// world PVC + the backup PVC and runs `felis restore`. It needs deployment-
|
// world PVC + the backup PVC and runs `felis restore`. It needs deployment-
|
||||||
@@ -258,9 +275,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
// agree on what local auth knows.
|
// agree on what local auth knows.
|
||||||
repo := api.NewPGRepo(drv.DB())
|
repo := api.NewPGRepo(drv.DB())
|
||||||
|
|
||||||
|
// The owner's on-demand backup levers come from [archive], the same keys the
|
||||||
|
// backup Job and the reaper read. A malformed key leaves the defaults in
|
||||||
|
// place here; the reaper Job fails on it and names it.
|
||||||
|
rcfg, err := reaperConfig(cfg)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis api: %v; using the default backup limits\n", err)
|
||||||
|
rcfg = reaper.DefaultConfig()
|
||||||
|
}
|
||||||
|
|
||||||
|
cluster := api.NewK8sCluster(cl, cfg.K8s.Namespace)
|
||||||
a := &api.API{
|
a := &api.API{
|
||||||
Repo: repo,
|
Repo: repo,
|
||||||
Cluster: api.NewK8sCluster(cl, cfg.K8s.Namespace),
|
Cluster: cluster,
|
||||||
Console: api.NewK8sConsole(cl, cfg.K8s.Namespace),
|
Console: api.NewK8sConsole(cl, cfg.K8s.Namespace),
|
||||||
Logs: api.NewK8sLogStreamer(clientset, cfg.K8s.Namespace),
|
Logs: api.NewK8sLogStreamer(clientset, cfg.K8s.Namespace),
|
||||||
// Build-log stream (spec §16) is scoped to the BUILD namespace — the same
|
// Build-log stream (spec §16) is scoped to the BUILD namespace — the same
|
||||||
@@ -268,6 +295,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
|
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
|
||||||
Internal: api.BearerTokenAuth{Token: token},
|
Internal: api.BearerTokenAuth{Token: token},
|
||||||
Builder: builder,
|
Builder: builder,
|
||||||
|
Images: imagePinner(cfg.Registry.URL),
|
||||||
Restorer: restorer,
|
Restorer: restorer,
|
||||||
Backuper: backuper,
|
Backuper: backuper,
|
||||||
JobStatus: api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
|
JobStatus: api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
|
||||||
@@ -291,6 +319,10 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
AdminHostname: cfg.Auth.AdminHostname,
|
AdminHostname: cfg.Auth.AdminHostname,
|
||||||
PanelHostname: cfg.Auth.PanelHostname,
|
PanelHostname: cfg.Auth.PanelHostname,
|
||||||
WakeCooldown: 30 * time.Second,
|
WakeCooldown: 30 * time.Second,
|
||||||
|
// An owner may start one backup per server per manual_cooldown, and none
|
||||||
|
// while the store is at max_local_bytes (data-durability-9).
|
||||||
|
BackupCooldown: rcfg.ManualCooldown,
|
||||||
|
BackupStoreCap: rcfg.MaxLocalBytes,
|
||||||
// The user-modpack lane's per-user throttles: a create spaces out
|
// The user-modpack lane's per-user throttles: a create spaces out
|
||||||
// review-queue rows, an upload spaces out (up to 1 GiB) context streams.
|
// review-queue rows, an upload spaces out (up to 1 GiB) context streams.
|
||||||
// Separate keys, so the normal create→upload sequence stays immediate.
|
// Separate keys, so the normal create→upload sequence stays immediate.
|
||||||
@@ -300,8 +332,20 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
// for legitimate multi-tab / multi-server watching, while capping how many
|
// for legitimate multi-tab / multi-server watching, while capping how many
|
||||||
// upstream follow connections a single caller can tie up if their streams stall.
|
// upstream follow connections a single caller can tie up if their streams stall.
|
||||||
MaxStreamsPerPrincipal: 16,
|
MaxStreamsPerPrincipal: 16,
|
||||||
|
// Public sign-in doors, per client address: a person signing in makes a
|
||||||
|
// handful of calls, so 20 at once refilled at 20 a minute never bites a
|
||||||
|
// real user and still turns a spray into a trickle. The client address
|
||||||
|
// is the edge's header when the install names one (config.AuthConfig).
|
||||||
|
AuthDoorLimit: api.RateLimit{Burst: 20, PerMinute: 20},
|
||||||
|
ClientIPHeader: cfg.Auth.EffectiveClientIPHeader(),
|
||||||
|
MailLimit: mailLimit(cfg.SMTP.MaxPerHour),
|
||||||
}
|
}
|
||||||
fmt.Fprintln(stderr, "felis api: external face fails closed (Access JWKS key function not configured)")
|
fmt.Fprintln(stderr, "felis api: external face fails closed (Access JWKS key function not configured)")
|
||||||
|
if a.ClientIPHeader != "" {
|
||||||
|
fmt.Fprintf(stderr, "felis api: sign-in rate limit keys on the %s header\n", a.ClientIPHeader)
|
||||||
|
} else {
|
||||||
|
fmt.Fprintln(stderr, "felis api: sign-in rate limit keys on the TCP peer ([auth] client_ip_header unset)")
|
||||||
|
}
|
||||||
|
|
||||||
// Felis-nano: the multi-source hasJoined multiplexer. Mojang leads as the code-owned
|
// Felis-nano: the multi-source hasJoined multiplexer. Mojang leads as the code-owned
|
||||||
// identity anchor (正版优先); config can only append namespace-rewritten third-party
|
// identity anchor (正版优先); config can only append namespace-rewritten third-party
|
||||||
@@ -365,6 +409,11 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
|||||||
// reconciles it, but this loop converges builds nobody is polling.
|
// reconciles it, but this loop converges builds nobody is polling.
|
||||||
go reconcileBuilds(ctx, builder, stderr)
|
go reconcileBuilds(ctx, builder, stderr)
|
||||||
|
|
||||||
|
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
|
||||||
|
go pruner.Loop(ctx, registryPruneInterval)
|
||||||
|
}
|
||||||
|
go reapRejectedContexts(ctx, submissions, stderr)
|
||||||
|
|
||||||
select {
|
select {
|
||||||
case <-ctx.Done():
|
case <-ctx.Done():
|
||||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||||
@@ -418,18 +467,46 @@ func buildConfig(cfg *config.Config) build.Config {
|
|||||||
return build.Config{
|
return build.Config{
|
||||||
Namespace: cfg.Registry.BuildNamespace,
|
Namespace: cfg.Registry.BuildNamespace,
|
||||||
RegistryURL: cfg.Registry.URL,
|
RegistryURL: cfg.Registry.URL,
|
||||||
// Empty overrides fall back to the build package's defaults, so an
|
// Empty overrides fall back to the registry's copies of the tools
|
||||||
// install that has not imported kaniko/trivy keeps the compiled-in refs
|
// (build.Tools), which felis mirror-build-tools keeps current.
|
||||||
// (and fails loudly on pull rather than silently building with the wrong
|
|
||||||
// image).
|
|
||||||
KanikoImage: cfg.Registry.KanikoImage,
|
KanikoImage: cfg.Registry.KanikoImage,
|
||||||
TrivyImage: cfg.Registry.TrivyImage,
|
TrivyImage: cfg.Registry.TrivyImage,
|
||||||
CPULimit: cfg.Registry.BuildCPULimit,
|
CPULimit: cfg.Registry.BuildCPULimit,
|
||||||
MemLimit: cfg.Registry.BuildMemLimit,
|
MemLimit: cfg.Registry.BuildMemLimit,
|
||||||
// Empty keeps Trivy's own default; an install with builds points this at
|
DiskLimit: cfg.Registry.BuildDiskLimit,
|
||||||
// the internal DB mirror (see config.RegistryConfig.TrivyDBRepository).
|
// "auto" follows the startup probe (see probeBuildUserNamespaces).
|
||||||
|
UserNamespaces: cfg.Registry.BuildUserNamespaces,
|
||||||
|
UserNamespacesProbe: new(atomic.Bool),
|
||||||
|
RuntimeClass: cfg.Registry.BuildRuntimeClass,
|
||||||
|
MaxConcurrent: cfg.Registry.MaxConcurrentBuilds,
|
||||||
TrivyDBRepository: cfg.Registry.TrivyDBRepository,
|
TrivyDBRepository: cfg.Registry.TrivyDBRepository,
|
||||||
TrivyJavaDBRepository: cfg.Registry.TrivyJavaDBRepository,
|
TrivyJavaDBRepository: cfg.Registry.TrivyJavaDBRepository,
|
||||||
|
// The submit lane's derived context URLs live here; the fetch step's
|
||||||
|
// service token goes nowhere else.
|
||||||
|
ContextOrigin: internalAPIBaseURL(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// probeBuildUserNamespaces settles build_user_namespaces = "auto": one probe
|
||||||
|
// pod with hostUsers: false tells whether this node's kernel and runtime can run
|
||||||
|
// build pods in a user namespace. Builds submitted before it answers run without.
|
||||||
|
func probeBuildUserNamespaces(ctx context.Context, jobs *build.K8sJobs, cfg build.Config, stderr io.Writer) {
|
||||||
|
if mode := cfg.UserNamespaces; mode != "" && mode != build.UserNamespacesAuto {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if cfg.FelisImage == "" {
|
||||||
|
fmt.Fprintln(stderr, "felis api: FELIS_IMAGE unset — build pods run without a user namespace")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
ok, err := jobs.ProbeUserNamespaces(ctx, cfg.FelisImage)
|
||||||
|
cfg.UserNamespacesProbe.Store(ok)
|
||||||
|
switch {
|
||||||
|
case ok:
|
||||||
|
fmt.Fprintln(stderr, "felis api: build pods run in a user namespace (hostUsers: false)")
|
||||||
|
case err != nil:
|
||||||
|
fmt.Fprintf(stderr, "felis api: build pods run without a user namespace: the probe failed: %v\n", err)
|
||||||
|
default:
|
||||||
|
fmt.Fprintln(stderr, "felis api: build pods run without a user namespace: this node cannot start a pod with hostUsers: false")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -546,3 +623,127 @@ func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
|
||||||
|
// submissions rejected more than submit.RejectedContextRetention ago. Without it a
|
||||||
|
// rejected modpack keeps its bytes on the uploads store (and against its
|
||||||
|
// submitter's budget) until an admin deletes the row.
|
||||||
|
func reapRejectedContexts(ctx context.Context, m *submit.Manager, stderr io.Writer) {
|
||||||
|
t := time.NewTicker(time.Hour)
|
||||||
|
defer t.Stop()
|
||||||
|
for {
|
||||||
|
n, err := m.ReapRejected(ctx, submit.RejectedContextRetention)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis api: reap rejected uploads: %v\n", err)
|
||||||
|
}
|
||||||
|
if n > 0 {
|
||||||
|
fmt.Fprintf(stderr, "felis api: deleted the uploaded contexts of %d rejected submission(s)\n", n)
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return
|
||||||
|
case <-t.C:
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// registryPruneInterval spaces the registry pruner's runs. The registry-gc
|
||||||
|
// sidecar sweeps once a day, so pruning more often only changes which sweep frees
|
||||||
|
// a layer.
|
||||||
|
const registryPruneInterval = 6 * time.Hour
|
||||||
|
|
||||||
|
// registryPruner deletes the registry manifests nothing references
|
||||||
|
// (internal/registryprune); the registry-gc sidecar frees their layers on its next
|
||||||
|
// sweep. It acts as the gate's prune principal, whose token the api Deployment
|
||||||
|
// injects from felis-registry-auth. Without the token the registry only grows,
|
||||||
|
// which is said once here.
|
||||||
|
func registryPruner(cfg *config.Config, store imageRefStore, servers serverLister, stderr io.Writer) *registryprune.Pruner {
|
||||||
|
if cfg.Registry.URL == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
token := os.Getenv(platform.RegistryPruneTokenEnv)
|
||||||
|
if token == "" {
|
||||||
|
fmt.Fprintf(stderr, "felis api: registry pruner disabled (%s unset) — images nothing uses are never deleted from the registry\n", platform.RegistryPruneTokenEnv)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
static := append([]string{os.Getenv("FELIS_IMAGE")}, buildConfig(cfg).ToolRefs()...)
|
||||||
|
return ®istryprune.Pruner{
|
||||||
|
Registry: ®istryprune.Client{Endpoint: "http://" + cfg.Registry.URL, Token: token},
|
||||||
|
Host: cfg.Registry.URL,
|
||||||
|
Refs: func(ctx context.Context) ([]string, error) {
|
||||||
|
return inUseImageRefs(ctx, store, servers, static)
|
||||||
|
},
|
||||||
|
Log: slog.New(slog.NewTextHandler(stderr, nil)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type imageRefStore interface {
|
||||||
|
ListImages(ctx context.Context) ([]build.Image, error)
|
||||||
|
ListUnfinishedBuilds(ctx context.Context) ([]build.Build, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
type serverLister interface {
|
||||||
|
ListServers(ctx context.Context) ([]api.ServerInfo, error)
|
||||||
|
PodImages(ctx context.Context) ([]string, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// inUseImageRefs lists every image reference the platform still depends on: the
|
||||||
|
// whitelist (disabled rows too, an admin may enable them again), every server's
|
||||||
|
// spec, the images the game pods run, builds still running, and the images the
|
||||||
|
// control plane and the build Jobs run. Any source failing fails the whole list,
|
||||||
|
// so the pruner never decides on a partial view.
|
||||||
|
//
|
||||||
|
// The pods matter for the felis image: a running server keeps the one it started
|
||||||
|
// with across platform upgrades (operator.PodTemplateAnnotation), which after a
|
||||||
|
// few releases is no longer among the newest tags the pruner keeps anyway, and
|
||||||
|
// the pod needs it again whenever it is recreated.
|
||||||
|
func inUseImageRefs(ctx context.Context, store imageRefStore, servers serverLister, static []string) ([]string, error) {
|
||||||
|
refs := append([]string(nil), static...)
|
||||||
|
images, err := store.ListImages(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("image whitelist: %w", err)
|
||||||
|
}
|
||||||
|
for _, img := range images {
|
||||||
|
refs = append(refs, img.ImageRef)
|
||||||
|
}
|
||||||
|
srvs, err := servers.ListServers(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("servers: %w", err)
|
||||||
|
}
|
||||||
|
for _, s := range srvs {
|
||||||
|
refs = append(refs, s.Image)
|
||||||
|
}
|
||||||
|
podImages, err := servers.PodImages(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("game pods: %w", err)
|
||||||
|
}
|
||||||
|
refs = append(refs, podImages...)
|
||||||
|
builds, err := store.ListUnfinishedBuilds(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("running builds: %w", err)
|
||||||
|
}
|
||||||
|
for _, b := range builds {
|
||||||
|
refs = append(refs, b.ImageRef)
|
||||||
|
}
|
||||||
|
return refs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// mailLimit turns smtp.max_per_hour into the API's install-wide mail bucket:
|
||||||
|
// the hourly cap as the refill rate, with a quarter of it (at least 5) allowed
|
||||||
|
// at once so a burst of real sign-ins is not queued behind the average.
|
||||||
|
func mailLimit(perHour int) api.RateLimit {
|
||||||
|
if perHour <= 0 {
|
||||||
|
perHour = config.DefaultMailPerHour
|
||||||
|
}
|
||||||
|
return api.RateLimit{Burst: max(perHour/4, 5), PerMinute: float64(perHour) / 60}
|
||||||
|
}
|
||||||
|
|
||||||
|
// imagePinner resolves a new server's image against the platform registry
|
||||||
|
// through its in-cluster Service, the address its refs already spell. An install
|
||||||
|
// without a registry has no platform-built images to pin.
|
||||||
|
func imagePinner(registry string) api.ImagePinner {
|
||||||
|
if registry == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return imagepin.Resolver{Registry: registry}
|
||||||
|
}
|
||||||
@@ -1,9 +1,14 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
"net/http"
|
"net/http"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/api"
|
||||||
|
"felis.lolicon.best/internal/build"
|
||||||
"felis.lolicon.best/internal/config"
|
"felis.lolicon.best/internal/config"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -91,3 +96,62 @@ func TestNewAPIServerSetsHardenedTimeouts(t *testing.T) {
|
|||||||
t.Errorf("ReadTimeout = %v, want 0 (unset) so a slow SSE attach is not capped", srv.ReadTimeout)
|
t.Errorf("ReadTimeout = %v, want 0 (unset) so a slow SSE attach is not capped", srv.ReadTimeout)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type fakeRefStore struct {
|
||||||
|
images []build.Image
|
||||||
|
builds []build.Build
|
||||||
|
err error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f fakeRefStore) ListImages(context.Context) ([]build.Image, error) { return f.images, f.err }
|
||||||
|
func (f fakeRefStore) ListUnfinishedBuilds(context.Context) ([]build.Build, error) {
|
||||||
|
return f.builds, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type fakeServers struct {
|
||||||
|
list []api.ServerInfo
|
||||||
|
pods []string
|
||||||
|
podsErr error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f fakeServers) ListServers(context.Context) ([]api.ServerInfo, error) { return f.list, nil }
|
||||||
|
func (f fakeServers) PodImages(context.Context) ([]string, error) { return f.pods, f.podsErr }
|
||||||
|
|
||||||
|
// The registry pruner deletes whatever this list does not name, so every source of
|
||||||
|
// a reference has to be in it, and a failing source must fail the list.
|
||||||
|
func TestInUseImageRefsCoversEverySource(t *testing.T) {
|
||||||
|
const reg = "registry.felis.svc:5000/"
|
||||||
|
store := fakeRefStore{
|
||||||
|
images: []build.Image{{ImageRef: reg + "modpacks/pack:*"}, {ImageRef: reg + "felis/paper:demo"}},
|
||||||
|
builds: []build.Build{{ImageRef: reg + "user-uploads/sub-9:latest"}},
|
||||||
|
}
|
||||||
|
servers := fakeServers{
|
||||||
|
list: []api.ServerInfo{{Name: "s1", Image: reg + "felis/paper:demo@sha256:" + fmt.Sprintf("%064d", 1)}},
|
||||||
|
pods: []string{reg + "felis/felis:v1.0.0"},
|
||||||
|
}
|
||||||
|
got, err := inUseImageRefs(context.Background(), store, servers, []string{reg + "felis/felis:b60"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
want := []string{
|
||||||
|
reg + "felis/felis:b60",
|
||||||
|
reg + "modpacks/pack:*", reg + "felis/paper:demo",
|
||||||
|
reg + "felis/paper:demo@sha256:" + fmt.Sprintf("%064d", 1),
|
||||||
|
reg + "felis/felis:v1.0.0",
|
||||||
|
reg + "user-uploads/sub-9:latest",
|
||||||
|
}
|
||||||
|
if fmt.Sprint(got) != fmt.Sprint(want) {
|
||||||
|
t.Fatalf("refs = %v\nwant %v", got, want)
|
||||||
|
}
|
||||||
|
|
||||||
|
servers.podsErr = errors.New("apiserver down")
|
||||||
|
if _, err := inUseImageRefs(context.Background(), store, servers, nil); err == nil {
|
||||||
|
t.Fatal("a failing pod list produced a reference list")
|
||||||
|
}
|
||||||
|
servers.podsErr = nil
|
||||||
|
|
||||||
|
store.err = errors.New("db down")
|
||||||
|
if _, err := inUseImageRefs(context.Background(), store, servers, nil); err == nil {
|
||||||
|
t.Fatal("a failing whitelist read produced a reference list")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -243,6 +243,19 @@ func buildMinecraftServerFromApplyRequest(req applyRequest, namespace string) (*
|
|||||||
AutostartPolicy: policy,
|
AutostartPolicy: policy,
|
||||||
Storage: v1alpha1.StorageSpec{Size: storageQ.String()},
|
Storage: v1alpha1.StorageSpec{Size: storageQ.String()},
|
||||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: requests},
|
Resources: corev1.ResourceRequirements{Limits: limits, Requests: requests},
|
||||||
|
// The rest matches what felis-api's create writes (K8sCluster.CreateServer):
|
||||||
|
// fall back to the login gate while stopped, RCON on (readiness, the
|
||||||
|
// player count and the console all ride it; the operator mints the
|
||||||
|
// password), and the default idle stop.
|
||||||
|
FallbackServer: naming.SystemLoginServer,
|
||||||
|
Rcon: v1alpha1.RconSpec{
|
||||||
|
Enabled: true,
|
||||||
|
SecretRef: v1alpha1.SecretKeyRef{
|
||||||
|
Name: naming.RconSecretName(req.Name),
|
||||||
|
Key: naming.RconSecretKey,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
Idle: v1alpha1.DefaultIdle(),
|
||||||
},
|
},
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import (
|
|||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
"felis.lolicon.best/internal/naming"
|
||||||
corev1 "k8s.io/api/core/v1"
|
corev1 "k8s.io/api/core/v1"
|
||||||
"k8s.io/apimachinery/pkg/api/resource"
|
"k8s.io/apimachinery/pkg/api/resource"
|
||||||
)
|
)
|
||||||
@@ -167,6 +168,18 @@ func TestBuildMinecraftServerFromApplyRequest_Valid(t *testing.T) {
|
|||||||
if ms.Spec.Storage.Size != "20Gi" {
|
if ms.Spec.Storage.Size != "20Gi" {
|
||||||
t.Errorf("Storage.Size = %q, want 20Gi", ms.Spec.Storage.Size)
|
t.Errorf("Storage.Size = %q, want 20Gi", ms.Spec.Storage.Size)
|
||||||
}
|
}
|
||||||
|
// Same operational defaults as the API create path: without RCON the server
|
||||||
|
// never reports players and the console answers 503; without spec.idle it
|
||||||
|
// never stops on its own.
|
||||||
|
if !ms.Spec.Rcon.Enabled || ms.Spec.Rcon.SecretRef.Name != naming.RconSecretName("test-server") {
|
||||||
|
t.Errorf("Rcon = %+v, want enabled with the operator-minted secret", ms.Spec.Rcon)
|
||||||
|
}
|
||||||
|
if ms.Spec.Idle != v1alpha1.DefaultIdle() {
|
||||||
|
t.Errorf("Idle = %+v, want the default %+v", ms.Spec.Idle, v1alpha1.DefaultIdle())
|
||||||
|
}
|
||||||
|
if ms.Spec.FallbackServer != naming.SystemLoginServer {
|
||||||
|
t.Errorf("FallbackServer = %q, want the login gate", ms.Spec.FallbackServer)
|
||||||
|
}
|
||||||
mem, ok := ms.Spec.Resources.Limits[corev1.ResourceMemory]
|
mem, ok := ms.Spec.Resources.Limits[corev1.ResourceMemory]
|
||||||
if !ok {
|
if !ok {
|
||||||
t.Fatal("memory limit missing")
|
t.Fatal("memory limit missing")
|
||||||
|
|||||||
+37
-4
@@ -1,6 +1,7 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
"crypto/rand"
|
"crypto/rand"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
"flag"
|
"flag"
|
||||||
@@ -52,8 +53,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
|
fmt.Fprintf(stderr, "felis backup: archive store %q is not implemented in this build (only tarLocal)\n", cfg.Archive.Store)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
// Reuse the reaper's retention derivation so an on-demand backup expires on the
|
// The [archive] parse the reaper uses; an on-demand backup takes its
|
||||||
// same clock as an inactivity backup — one retention policy, not two.
|
// manual_retention and manual_keep.
|
||||||
rcfg, err := reaperConfig(cfg)
|
rcfg, err := reaperConfig(cfg)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||||
@@ -72,6 +73,13 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
|
|
||||||
ctx := ctrl.SetupSignalHandler()
|
ctx := ctrl.SetupSignalHandler()
|
||||||
|
|
||||||
|
// The archive store shares the node's disk with every world and the
|
||||||
|
// database: an owner's backup must not be what tips it into eviction.
|
||||||
|
if err := backup.CheckRoom(cfg.Archive.LocalPath, *worldsRoot, backup.MinFreeAfter); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
|
ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
|
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
|
||||||
@@ -92,9 +100,10 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
BackupRef: string(ref),
|
BackupRef: string(ref),
|
||||||
SizeBytes: size,
|
SizeBytes: size,
|
||||||
Reason: "manual",
|
Reason: "manual",
|
||||||
ExpiresAt: time.Now().Add(rcfg.Retention),
|
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
|
||||||
}
|
}
|
||||||
if err := reaper.NewPGStore(drv.DB()).InsertBackup(ctx, rec); err != nil {
|
st := reaper.NewPGStore(drv.DB())
|
||||||
|
if err := st.InsertBackup(ctx, rec); err != nil {
|
||||||
// The archive is written but unrecorded — an orphan the retention pass would
|
// The archive is written but unrecorded — an orphan the retention pass would
|
||||||
// never expire. Delete it so a failed backup leaves no leaked bytes, mirroring
|
// never expire. Delete it so a failed backup leaves no leaked bytes, mirroring
|
||||||
// the reaper's archive-then-record atomicity.
|
// the reaper's archive-then-record atomicity.
|
||||||
@@ -107,9 +116,33 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
||||||
|
pruneManualBackups(ctx, st, archiver, *server, rcfg.ManualKeep, stdout, stderr)
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// pruneManualBackups keeps server's newest keep on-demand backups and removes
|
||||||
|
// the rest, oldest first, so repeated backups of one world cannot fill the
|
||||||
|
// shared archive store. The new backup is already recorded; a removal that
|
||||||
|
// fails is reported and retried after the next backup.
|
||||||
|
func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server string, keep int, stdout, stderr io.Writer) {
|
||||||
|
excess, err := st.ExcessManualBackups(ctx, server, keep)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
for _, b := range excess {
|
||||||
|
if err := archiver.Delete(ctx, backup.ArchiveRef(b.BackupRef)); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: remove older backup %s: %v\n", b.ID, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := st.MarkBackupDeleted(ctx, b.ID, time.Now()); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis backup: removed older backup %s of %s (keeping the newest %d)\n", b.ID, server, keep)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex
|
// newBackupID mints a world_backups primary key, matching the reaper's "bk-"+hex
|
||||||
// scheme so a manual and an inactivity backup are indistinguishable downstream.
|
// scheme so a manual and an inactivity backup are indistinguishable downstream.
|
||||||
func newBackupID() string {
|
func newBackupID() string {
|
||||||
|
|||||||
+34
-2
@@ -9,12 +9,14 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
"felis.lolicon.best/internal/config"
|
"felis.lolicon.best/internal/config"
|
||||||
"felis.lolicon.best/internal/platform"
|
"felis.lolicon.best/internal/platform"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdConverge is the explicit convergence pass over already-installed system
|
// cmdConverge is the explicit convergence pass over already-installed system
|
||||||
// servers (#1). Provisioning is create-if-absent, so a field the desired spec
|
// servers (#1), plus the idle-stop default for user servers that predate it. Provisioning is create-if-absent, so a field the desired spec
|
||||||
// gained after an install (spec.rcon, spec.startup.healthHTTPPort, a derived env
|
// gained after an install (spec.rcon, spec.startup.healthHTTPPort, a derived env
|
||||||
// key) never reaches the existing CR — and nothing says so. This command fills
|
// key) never reaches the existing CR — and nothing says so. This command fills
|
||||||
// exactly those zero-value fields; see convergeSystemServers for the full contract
|
// exactly those zero-value fields; see convergeSystemServers for the full contract
|
||||||
@@ -56,7 +58,9 @@ func cmdConverge(args []string, stdout, stderr io.Writer) int {
|
|||||||
platform.InternalAPIBaseURL(controlNS), cfg.Server.RootDomain,
|
platform.InternalAPIBaseURL(controlNS), cfg.Server.RootDomain,
|
||||||
defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
|
defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
|
||||||
|
|
||||||
fmt.Fprintln(stdout, "felis converge: filling fields an installed system server predates (operator-set values are never overwritten):")
|
outcomes = append(outcomes, convergeUserServerIdle(context.Background(), cl, cfg.K8s.Namespace)...)
|
||||||
|
|
||||||
|
fmt.Fprintln(stdout, "felis converge: filling fields an installed server predates (operator-set values are never overwritten):")
|
||||||
exit := 0
|
exit := 0
|
||||||
for _, o := range outcomes {
|
for _, o := range outcomes {
|
||||||
switch {
|
switch {
|
||||||
@@ -71,3 +75,31 @@ func cmdConverge(args []string, stdout, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
return exit
|
return exit
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// convergeUserServerIdle gives every user server that predates the idle default
|
||||||
|
// (spec.idle entirely unset) the default idle stop. A server whose idle stop was
|
||||||
|
// turned off keeps a duration on its spec, so it is not "unset" and is left
|
||||||
|
// alone; system servers never idle out and are skipped. Servers that already
|
||||||
|
// carry a value produce no line, so a converged fleet prints nothing here.
|
||||||
|
func convergeUserServerIdle(ctx context.Context, cl client.Client, namespace string) []systemServerOutcome {
|
||||||
|
var list v1alpha1.MinecraftServerList
|
||||||
|
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
|
||||||
|
return []systemServerOutcome{{name: "user servers", err: fmt.Errorf("list servers: %w", err)}}
|
||||||
|
}
|
||||||
|
var out []systemServerOutcome
|
||||||
|
for i := range list.Items {
|
||||||
|
ms := &list.Items[i]
|
||||||
|
if ms.Labels[v1alpha1.LabelSystemRole] != "" || ms.Spec.Idle != (v1alpha1.IdleSpec{}) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
patch := client.MergeFrom(ms.DeepCopy())
|
||||||
|
ms.Spec.Idle = v1alpha1.DefaultIdle()
|
||||||
|
if err := cl.Patch(ctx, ms, patch); err != nil {
|
||||||
|
out = append(out, systemServerOutcome{name: ms.Name, err: fmt.Errorf("converge %s: %w", ms.Name, err)})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
out = append(out, systemServerOutcome{name: ms.Name, available: true, updated: true,
|
||||||
|
changes: []string{fmt.Sprintf("spec.idle (stop after %ds empty)", v1alpha1.DefaultEmptySecondsBeforeStop)}})
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
@@ -183,3 +183,49 @@ func TestConvergeSystemServersGuards(t *testing.T) {
|
|||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestConvergeUserServerIdle fills the idle default only where spec.idle was
|
||||||
|
// never set: a server whose idle stop was turned off (duration kept), one with
|
||||||
|
// its own duration, and a system server all stay as they are.
|
||||||
|
func TestConvergeUserServerIdle(t *testing.T) {
|
||||||
|
scheme := newSystemServerScheme(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
mk := func(name string, idle v1alpha1.IdleSpec, role string) *v1alpha1.MinecraftServer {
|
||||||
|
ms := &v1alpha1.MinecraftServer{}
|
||||||
|
ms.Name, ms.Namespace = name, "minecraft"
|
||||||
|
ms.Spec.Idle = idle
|
||||||
|
if role != "" {
|
||||||
|
ms.Labels = map[string]string{v1alpha1.LabelSystemRole: role}
|
||||||
|
}
|
||||||
|
return ms
|
||||||
|
}
|
||||||
|
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||||
|
mk("legacy", v1alpha1.IdleSpec{}, ""),
|
||||||
|
mk("off", v1alpha1.IdleSpec{EmptySecondsBeforeStop: 600}, ""),
|
||||||
|
mk("custom", v1alpha1.IdleSpec{AutoStopEnabled: true, EmptySecondsBeforeStop: 1800}, ""),
|
||||||
|
mk(naming.SystemLobbyServer, v1alpha1.IdleSpec{}, naming.SystemLobbyServer),
|
||||||
|
).Build()
|
||||||
|
|
||||||
|
outcomes := convergeUserServerIdle(ctx, cl, "minecraft")
|
||||||
|
if len(outcomes) != 1 || outcomes[0].name != "legacy" || outcomes[0].err != nil {
|
||||||
|
t.Fatalf("outcomes = %+v, want exactly one fill for legacy", outcomes)
|
||||||
|
}
|
||||||
|
want := map[string]v1alpha1.IdleSpec{
|
||||||
|
"legacy": v1alpha1.DefaultIdle(),
|
||||||
|
"off": {EmptySecondsBeforeStop: 600},
|
||||||
|
"custom": {AutoStopEnabled: true, EmptySecondsBeforeStop: 1800},
|
||||||
|
naming.SystemLobbyServer: {},
|
||||||
|
}
|
||||||
|
for name, idle := range want {
|
||||||
|
var ms v1alpha1.MinecraftServer
|
||||||
|
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
|
||||||
|
t.Fatalf("get %s: %v", name, err)
|
||||||
|
}
|
||||||
|
if ms.Spec.Idle != idle {
|
||||||
|
t.Errorf("%s idle = %+v, want %+v", name, ms.Spec.Idle, idle)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if again := convergeUserServerIdle(ctx, cl, "minecraft"); len(again) != 0 {
|
||||||
|
t.Fatalf("second pass = %+v, want nothing to do", again)
|
||||||
|
}
|
||||||
|
}
|
||||||
+334
@@ -0,0 +1,334 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
|
)
|
||||||
|
|
||||||
|
const dbUsage = `usage:
|
||||||
|
felis db backup [-config path] [-dir dir] [-label daily|manual|...] [-keep n] [-state-dir dir]
|
||||||
|
[-no-servers] [-metrics-file path]
|
||||||
|
felis db restore [-config path] [-dir dir] [-yes] [-force] [-no-safety-backup] <bundle>
|
||||||
|
felis db verify [-dir dir] <bundle>
|
||||||
|
felis db list [-dir dir]
|
||||||
|
felis db check [-dir dir] [-max-age 26h]
|
||||||
|
`
|
||||||
|
|
||||||
|
// defaultKeep is how many bundles of a label a backup leaves behind. Manual
|
||||||
|
// bundles are the operator's own and are never pruned.
|
||||||
|
var defaultKeep = map[string]int{
|
||||||
|
dbbackup.LabelDaily: 14,
|
||||||
|
dbbackup.LabelPreMigrate: 10,
|
||||||
|
dbbackup.LabelPreRestore: 5,
|
||||||
|
}
|
||||||
|
|
||||||
|
// cmdDB implements `felis db`: logical backups of the control-plane database
|
||||||
|
// together with the host state a rebuild needs (internal/dbbackup). The verb
|
||||||
|
// comes first for the same reason as `felis migrate up`.
|
||||||
|
func cmdDB(args []string, stdout, stderr io.Writer) int {
|
||||||
|
if len(args) == 0 {
|
||||||
|
fmt.Fprint(stderr, dbUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
verb, rest := args[0], args[1:]
|
||||||
|
fs := flag.NewFlagSet("db "+verb, flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
fs.Usage = func() { fmt.Fprint(stderr, dbUsage) }
|
||||||
|
dir := fs.String("dir", dbbackup.DefaultDir, "bundle directory")
|
||||||
|
switch verb {
|
||||||
|
case "backup":
|
||||||
|
return dbBackup(fs, dir, rest, stdout, stderr)
|
||||||
|
case "restore":
|
||||||
|
return dbRestore(fs, dir, rest, stdout, stderr)
|
||||||
|
case "verify":
|
||||||
|
return dbVerify(fs, dir, rest, stdout, stderr)
|
||||||
|
case "list":
|
||||||
|
return dbList(fs, dir, rest, stdout, stderr)
|
||||||
|
case "check":
|
||||||
|
return dbCheck(fs, dir, rest, stdout, stderr)
|
||||||
|
case "-h", "--help", "help":
|
||||||
|
fmt.Fprint(stdout, dbUsage)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stderr, "felis db: unknown verb %q\n%s", verb, dbUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseWithArg parses flags that may sit on either side of one positional
|
||||||
|
// argument (`restore -yes x.tar` and `restore x.tar -yes` both work) and
|
||||||
|
// returns that argument.
|
||||||
|
func parseWithArg(fs *flag.FlagSet, args []string) (string, bool) {
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
if fs.NArg() == 0 {
|
||||||
|
return "", true
|
||||||
|
}
|
||||||
|
arg := fs.Arg(0)
|
||||||
|
if err := fs.Parse(fs.Args()[1:]); err != nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
if fs.NArg() > 0 {
|
||||||
|
fmt.Fprintf(fs.Output(), "felis db: unexpected argument %q\n", fs.Arg(0))
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
return arg, true
|
||||||
|
}
|
||||||
|
|
||||||
|
func dbDatabaseURL(path string) (string, error) {
|
||||||
|
cfg, err := config.Load(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return cfg.Database.URL, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
|
label := fs.String("label", dbbackup.LabelManual, "bundle label; daily/pre-migrate/pre-restore bundles are pruned, manual ones never")
|
||||||
|
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, manual all)")
|
||||||
|
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory to bundle ("" for none)`)
|
||||||
|
noServers := fs.Bool("no-servers", false, "leave the MinecraftServer objects out of the bundle")
|
||||||
|
metrics := fs.String("metrics-file", "", "node-exporter textfile to rewrite on success (e.g. /var/lib/node_exporter/textfile_collector/felis_db_backup.prom)")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if fs.NArg() > 0 {
|
||||||
|
fmt.Fprint(stderr, dbUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
url, err := dbDatabaseURL(*cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if *keep < 0 {
|
||||||
|
*keep = defaultKeep[*label]
|
||||||
|
}
|
||||||
|
o := dbbackup.BackupOptions{
|
||||||
|
DatabaseURL: url, Dir: *dir, Label: *label, Keep: *keep,
|
||||||
|
StateDir: *stateDir, Version: resolvedVersion(), Log: stderr,
|
||||||
|
MetricsFile: *metrics, Record: true,
|
||||||
|
}
|
||||||
|
if !*noServers {
|
||||||
|
o.ExportServers = exportMinecraftServers
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
path, err := dbbackup.Backup(ctx, o)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db backup: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis db backup: wrote %s\n", path)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveBundle accepts a path, or a bare bundle name looked up in dir.
|
||||||
|
func resolveBundle(dir, arg string) string {
|
||||||
|
if strings.ContainsRune(arg, os.PathSeparator) {
|
||||||
|
return arg
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(arg); err == nil {
|
||||||
|
return arg
|
||||||
|
}
|
||||||
|
return filepath.Join(dir, arg)
|
||||||
|
}
|
||||||
|
|
||||||
|
func dbRestore(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
|
yes := fs.Bool("yes", false, "replace the database's contents (required)")
|
||||||
|
force := fs.Bool("force", false, "restore even while other clients are connected")
|
||||||
|
noSafety := fs.Bool("no-safety-backup", false, "skip the bundle of the current database taken first")
|
||||||
|
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, "host state directory for the safety bundle")
|
||||||
|
arg, ok := parseWithArg(fs, args)
|
||||||
|
if !ok {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if arg == "" {
|
||||||
|
fmt.Fprint(stderr, dbUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
bundle := resolveBundle(*dir, arg)
|
||||||
|
m, err := dbbackup.Verify(bundle)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if !*yes {
|
||||||
|
fmt.Fprintf(stderr, "felis db restore: this replaces every table in the felis database with %s (%s, taken %s, schema %d).\n",
|
||||||
|
filepath.Base(bundle), m.Label, m.CreatedAt.Format(time.RFC3339), m.SchemaVersion)
|
||||||
|
fmt.Fprintln(stderr, "Scale felis-api and felis-operator to 0 first, then re-run with -yes.")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
url, err := dbDatabaseURL(*cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
_, safety, err := dbbackup.Restore(ctx, dbbackup.RestoreOptions{
|
||||||
|
DatabaseURL: url, Bundle: bundle, Dir: *dir, Force: *force, SkipSafetyBackup: *noSafety,
|
||||||
|
Safety: dbbackup.BackupOptions{Keep: defaultKeep[dbbackup.LabelPreRestore], StateDir: *stateDir,
|
||||||
|
Version: resolvedVersion(), ExportServers: exportMinecraftServers},
|
||||||
|
Log: stderr,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db restore: %v\n", err)
|
||||||
|
if errors.Is(err, dbbackup.ErrClientsConnected) {
|
||||||
|
fmt.Fprintln(stderr, " kubectl -n felis scale deployment felis-api felis-operator --replicas=0")
|
||||||
|
}
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis db restore: restored %s (schema %d)\n", filepath.Base(bundle), m.SchemaVersion)
|
||||||
|
if safety != "" {
|
||||||
|
fmt.Fprintf(stdout, " the database as it was before is in %s\n", safety)
|
||||||
|
}
|
||||||
|
// Nothing migrates at startup, so a control plane newer than the bundle needs
|
||||||
|
// its migrations re-applied; rolling back to the release that wrote the bundle
|
||||||
|
// must skip that, or the rollback is undone.
|
||||||
|
fmt.Fprintf(stdout, " next: felis migrate up -config %s (skip it when rolling back to felis %s, which wrote this bundle)\n", *cfgPath, orUnknown(m.FelisVersion))
|
||||||
|
fmt.Fprintln(stdout, " kubectl -n felis scale deployment felis-api felis-operator --replicas=1")
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func dbVerify(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||||
|
arg, ok := parseWithArg(fs, args)
|
||||||
|
if !ok {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if arg == "" {
|
||||||
|
fmt.Fprint(stderr, dbUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
bundle := resolveBundle(*dir, arg)
|
||||||
|
m, err := dbbackup.Verify(bundle)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db verify: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "%s: ok\n taken %s (%s)\n felis %s\n schema %d\n %s\n",
|
||||||
|
filepath.Base(bundle), m.CreatedAt.Format(time.RFC3339), m.Label, orUnknown(m.FelisVersion), m.SchemaVersion, orUnknown(m.PGDumpVersion))
|
||||||
|
for _, f := range m.Files {
|
||||||
|
if f.Link != "" {
|
||||||
|
fmt.Fprintf(stdout, " %-40s -> %s\n", f.Name, f.Link)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, " %-40s %d bytes\n", f.Name, f.Size)
|
||||||
|
}
|
||||||
|
if m.ServersError != "" {
|
||||||
|
fmt.Fprintf(stdout, " (no MinecraftServer objects: %s)\n", m.ServersError)
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func orUnknown(s string) string {
|
||||||
|
if s == "" {
|
||||||
|
return "unknown"
|
||||||
|
}
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
func dbList(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
all, err := dbbackup.List(*dir)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db list: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if len(all) == 0 {
|
||||||
|
fmt.Fprintf(stdout, "no database backups in %s\n", *dir)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
now := time.Now()
|
||||||
|
for _, b := range all {
|
||||||
|
fmt.Fprintf(stdout, "%-50s %-12s %10s %s ago\n", b.Name, b.Label, humanBytes(b.Size), dbbackup.Age(now.Sub(b.Created)))
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func humanBytes(n int64) string {
|
||||||
|
const unit = 1024
|
||||||
|
if n < unit {
|
||||||
|
return fmt.Sprintf("%d B", n)
|
||||||
|
}
|
||||||
|
div, exp := int64(unit), 0
|
||||||
|
for m := n / unit; m >= unit; m /= unit {
|
||||||
|
div *= unit
|
||||||
|
exp++
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp])
|
||||||
|
}
|
||||||
|
|
||||||
|
// dbCheck is the freshness probe: exit 1 when the newest bundle is missing or
|
||||||
|
// older than -max-age, for a monitor or the break-glass console to act on.
|
||||||
|
func dbCheck(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||||
|
maxAge := fs.Duration("max-age", dbbackup.StaleAfter, "oldest acceptable newest bundle")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
b, err := dbbackup.Check(*dir, *maxAge, time.Now())
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis db check: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis db check: ok, newest backup %s (%s ago)\n", b.Name, dbbackup.Age(time.Since(b.Created)))
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// exportMinecraftServers reads every MinecraftServer through the host's k3s
|
||||||
|
// kubectl and strips what the API server owns, so the result can be fed back
|
||||||
|
// with `kubectl apply -f` on a rebuilt cluster.
|
||||||
|
func exportMinecraftServers(ctx context.Context) ([]byte, error) {
|
||||||
|
ctx, cancel := context.WithTimeout(ctx, 30*time.Second)
|
||||||
|
defer cancel()
|
||||||
|
// Output, not the CombinedOutput kubectlOutput uses: a deprecation warning
|
||||||
|
// on stderr must not end up inside the JSON.
|
||||||
|
cmd := exec.CommandContext(ctx, "k3s", "kubectl", "get", "minecraftservers.felis.lolicon.best", "-A", "-o", "json")
|
||||||
|
cmd.Env = append(os.Environ(), "KUBECONFIG="+hostBootstrapKubeconfigPath)
|
||||||
|
var errBuf strings.Builder
|
||||||
|
cmd.Stderr = &errBuf
|
||||||
|
out, err := cmd.Output()
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("k3s kubectl get minecraftservers: %w: %s", err, strings.TrimSpace(errBuf.String()))
|
||||||
|
}
|
||||||
|
return cleanServerList(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// cleanServerList drops status and the server-assigned metadata from a
|
||||||
|
// `kubectl get -o json` List.
|
||||||
|
func cleanServerList(raw []byte) ([]byte, error) {
|
||||||
|
var list struct {
|
||||||
|
Items []map[string]any `json:"items"`
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(raw, &list); err != nil {
|
||||||
|
return nil, fmt.Errorf("parse MinecraftServer list: %w", err)
|
||||||
|
}
|
||||||
|
for _, it := range list.Items {
|
||||||
|
delete(it, "status")
|
||||||
|
if md, ok := it["metadata"].(map[string]any); ok {
|
||||||
|
for _, k := range []string{"resourceVersion", "uid", "creationTimestamp", "generation", "managedFields", "selfLink"} {
|
||||||
|
delete(md, k)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if list.Items == nil {
|
||||||
|
list.Items = []map[string]any{}
|
||||||
|
}
|
||||||
|
return json.MarshalIndent(map[string]any{"apiVersion": "v1", "kind": "List", "items": list.Items}, "", " ")
|
||||||
|
}
|
||||||
@@ -0,0 +1,151 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"flag"
|
||||||
|
"io"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestDBUsage(t *testing.T) {
|
||||||
|
for _, args := range [][]string{{"db"}, {"db", "frobnicate"}, {"db", "restore"}, {"db", "verify"}, {"db", "backup", "extra"}} {
|
||||||
|
var out, errBuf bytes.Buffer
|
||||||
|
if code := run(args, &out, &errBuf); code != 2 {
|
||||||
|
t.Errorf("%v: exit %d, want 2", args, code)
|
||||||
|
}
|
||||||
|
if !strings.Contains(errBuf.String(), "felis db restore") {
|
||||||
|
t.Errorf("%v: no usage on stderr: %q", args, errBuf.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDBRestoreNeedsYes(t *testing.T) {
|
||||||
|
// A bundle that does not exist fails verification (1) before -yes matters;
|
||||||
|
// the -yes gate itself is exercised against a real bundle in internal/dbbackup
|
||||||
|
// and on the VM. Here: the refusal path never reaches the config or database.
|
||||||
|
var out, errBuf bytes.Buffer
|
||||||
|
if code := run([]string{"db", "restore", "-dir", t.TempDir(), "missing.tar"}, &out, &errBuf); code != 1 {
|
||||||
|
t.Fatalf("exit %d, stderr %q", code, errBuf.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseWithArg(t *testing.T) {
|
||||||
|
for _, args := range [][]string{{"-yes", "b.tar"}, {"b.tar", "-yes"}} {
|
||||||
|
fs := flag.NewFlagSet("t", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(io.Discard)
|
||||||
|
yes := fs.Bool("yes", false, "")
|
||||||
|
arg, ok := parseWithArg(fs, args)
|
||||||
|
if !ok || arg != "b.tar" || !*yes {
|
||||||
|
t.Errorf("%v -> %q ok=%v yes=%v", args, arg, ok, *yes)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fs := flag.NewFlagSet("t", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(io.Discard)
|
||||||
|
if _, ok := parseWithArg(fs, []string{"a.tar", "b.tar"}); ok {
|
||||||
|
t.Error("two positional arguments accepted")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolveBundle(t *testing.T) {
|
||||||
|
if got := resolveBundle("/var/lib/felis/db-backups", "felis-db-x.tar"); got != "/var/lib/felis/db-backups/felis-db-x.tar" {
|
||||||
|
t.Errorf("bare name -> %s", got)
|
||||||
|
}
|
||||||
|
if got := resolveBundle("/var/lib/felis/db-backups", "/root/copy.tar"); got != "/root/copy.tar" {
|
||||||
|
t.Errorf("path -> %s", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCleanServerList(t *testing.T) {
|
||||||
|
raw := `{"apiVersion":"v1","kind":"List","metadata":{"resourceVersion":""},"items":[{
|
||||||
|
"apiVersion":"felis.lolicon.best/v1alpha1","kind":"MinecraftServer",
|
||||||
|
"metadata":{"name":"survival","namespace":"minecraft","uid":"u","resourceVersion":"42","generation":3,
|
||||||
|
"creationTimestamp":"2026-09-01T00:00:00Z","managedFields":[{}],"labels":{"a":"b"}},
|
||||||
|
"spec":{"desiredState":"Running"},"status":{"phase":"Running"}}]}`
|
||||||
|
out, err := cleanServerList([]byte(raw))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got struct {
|
||||||
|
Kind string `json:"kind"`
|
||||||
|
Items []map[string]any `json:"items"`
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(out, &got); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got.Kind != "List" || len(got.Items) != 1 {
|
||||||
|
t.Fatalf("got %s", out)
|
||||||
|
}
|
||||||
|
it := got.Items[0]
|
||||||
|
if _, ok := it["status"]; ok {
|
||||||
|
t.Error("status kept")
|
||||||
|
}
|
||||||
|
md := it["metadata"].(map[string]any)
|
||||||
|
for _, k := range []string{"uid", "resourceVersion", "generation", "creationTimestamp", "managedFields"} {
|
||||||
|
if _, ok := md[k]; ok {
|
||||||
|
t.Errorf("metadata.%s kept", k)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if md["name"] != "survival" || md["namespace"] != "minecraft" || md["labels"] == nil {
|
||||||
|
t.Errorf("identity lost: %v", md)
|
||||||
|
}
|
||||||
|
if it["spec"].(map[string]any)["desiredState"] != "Running" {
|
||||||
|
t.Error("spec lost")
|
||||||
|
}
|
||||||
|
|
||||||
|
empty, err := cleanServerList([]byte(`{"items":null}`))
|
||||||
|
if err != nil || !strings.Contains(string(empty), `"items": []`) {
|
||||||
|
t.Errorf("empty list -> %s, %v", empty, err)
|
||||||
|
}
|
||||||
|
if _, err := cleanServerList([]byte("Warning: x\n{")); err == nil {
|
||||||
|
t.Error("garbage parsed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHasPending(t *testing.T) {
|
||||||
|
ms := []store.Migration{{Version: 1}, {Version: 2}, {Version: 3}}
|
||||||
|
if hasPending(map[int]struct{}{1: {}, 2: {}, 3: {}}, ms) {
|
||||||
|
t.Error("fully applied reported pending")
|
||||||
|
}
|
||||||
|
if !hasPending(map[int]struct{}{1: {}, 2: {}}, ms) {
|
||||||
|
t.Error("missing 3 not reported")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type appliedDriver struct {
|
||||||
|
store.Driver
|
||||||
|
done map[int]struct{}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (d appliedDriver) EnsureVersionTable(context.Context) error { return nil }
|
||||||
|
func (d appliedDriver) AppliedVersions(context.Context) (map[int]struct{}, error) {
|
||||||
|
return d.done, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPreMigrateBackupOnlyGuardsAPopulatedDatabase(t *testing.T) {
|
||||||
|
ms := []store.Migration{{Version: 1}, {Version: 2}}
|
||||||
|
// An unusable URL makes an attempted backup observable as an error without
|
||||||
|
// any PostgreSQL tooling.
|
||||||
|
const badURL = "not-a-url"
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
done map[int]struct{}
|
||||||
|
attempt bool
|
||||||
|
}{
|
||||||
|
{"fresh database", map[int]struct{}{}, false},
|
||||||
|
{"up to date", map[int]struct{}{1: {}, 2: {}}, false},
|
||||||
|
{"pending on a populated database", map[int]struct{}{1: {}}, true},
|
||||||
|
} {
|
||||||
|
path, err := preMigrateBackup(context.Background(), appliedDriver{done: tc.done}, ms, badURL, t.TempDir(), io.Discard)
|
||||||
|
if attempted := err != nil; attempted != tc.attempt {
|
||||||
|
t.Errorf("%s: attempted = %v (err %v), want %v", tc.name, attempted, err, tc.attempt)
|
||||||
|
}
|
||||||
|
if path != "" {
|
||||||
|
t.Errorf("%s: path = %q", tc.name, path)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"net"
|
||||||
|
"os"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Vars so tests can shrink them. A dial that neither connects nor is refused
|
||||||
|
// within egressDialTimeout counts as blocked: a policy that drops packets looks
|
||||||
|
// exactly like that.
|
||||||
|
var (
|
||||||
|
egressDialTimeout = 500 * time.Millisecond
|
||||||
|
egressPollInterval = 200 * time.Millisecond
|
||||||
|
)
|
||||||
|
|
||||||
|
// cmdEgressGate is the first initContainer of every build pod. The pod's
|
||||||
|
// NetworkPolicy is programmed asynchronously after the pod starts (live on k3s:
|
||||||
|
// a build-labelled pod reached the internet and the Kubernetes API for its first
|
||||||
|
// ~0.7 s), so the gate dials a destination the policy denies until it stops
|
||||||
|
// answering, and only then lets the pod's next container, eventually the
|
||||||
|
// untrusted Dockerfile, start.
|
||||||
|
//
|
||||||
|
// The default probe is the Kubernetes API Service, which the kubelet names in
|
||||||
|
// every pod's environment and the build policy never admits. A probe that still
|
||||||
|
// answers after --wait means the policy is not enforced at all (a CNI without
|
||||||
|
// NetworkPolicy support, or k3s run with --disable-network-policy), and the
|
||||||
|
// build fails closed.
|
||||||
|
func cmdEgressGate(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("egress-gate", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
probe := fs.String("probe", "", "host:port the build NetworkPolicy denies (default: the Kubernetes API Service from KUBERNETES_SERVICE_HOST/PORT)")
|
||||||
|
wait := fs.Duration("wait", 2*time.Minute, "how long the probe may keep answering before the build is refused")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *probe == "" {
|
||||||
|
host, port := os.Getenv("KUBERNETES_SERVICE_HOST"), os.Getenv("KUBERNETES_SERVICE_PORT")
|
||||||
|
if host == "" || port == "" {
|
||||||
|
fmt.Fprintln(stderr, "felis egress-gate: no --probe and no KUBERNETES_SERVICE_HOST/PORT to default to")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
*probe = net.JoinHostPort(host, port)
|
||||||
|
}
|
||||||
|
|
||||||
|
start := time.Now()
|
||||||
|
for {
|
||||||
|
conn, err := net.DialTimeout("tcp", *probe, egressDialTimeout)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stdout, "felis egress-gate: %s is unreachable after %s (%v); the egress lock is in effect\n",
|
||||||
|
*probe, time.Since(start).Round(time.Millisecond), err)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
_ = conn.Close()
|
||||||
|
if time.Since(start) >= *wait {
|
||||||
|
fmt.Fprintf(stderr, "felis egress-gate: %s still answers after %s: the build namespace's NetworkPolicy is not enforced "+
|
||||||
|
"(a CNI without NetworkPolicy support, or k3s started with --disable-network-policy); refusing to run the build\n",
|
||||||
|
*probe, *wait)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
time.Sleep(egressPollInterval)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"net"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
func shrinkEgressGate(t *testing.T) {
|
||||||
|
t.Helper()
|
||||||
|
dial, poll := egressDialTimeout, egressPollInterval
|
||||||
|
egressDialTimeout, egressPollInterval = 200*time.Millisecond, 10*time.Millisecond
|
||||||
|
t.Cleanup(func() { egressDialTimeout, egressPollInterval = dial, poll })
|
||||||
|
}
|
||||||
|
|
||||||
|
// The gate holds while the probe answers and lets the pod go on once the policy
|
||||||
|
// lands, which the test plays by closing the listener.
|
||||||
|
func TestEgressGateWaitsForTheLock(t *testing.T) {
|
||||||
|
shrinkEgressGate(t)
|
||||||
|
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
accepted := make(chan struct{}, 100)
|
||||||
|
go func() {
|
||||||
|
for {
|
||||||
|
c, err := ln.Accept()
|
||||||
|
if err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
_ = c.Close()
|
||||||
|
accepted <- struct{}{}
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
go func() {
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
<-accepted
|
||||||
|
}
|
||||||
|
_ = ln.Close()
|
||||||
|
}()
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "10s"}, &out, &errb); code != 0 {
|
||||||
|
t.Fatalf("exit %d: %s", code, errb.String())
|
||||||
|
}
|
||||||
|
if !strings.Contains(out.String(), "egress lock is in effect") {
|
||||||
|
t.Errorf("stdout = %q", out.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A probe that keeps answering means no policy is enforced: the build must not run.
|
||||||
|
func TestEgressGateRefusesAnOpenNetwork(t *testing.T) {
|
||||||
|
shrinkEgressGate(t)
|
||||||
|
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
defer ln.Close()
|
||||||
|
go func() {
|
||||||
|
for {
|
||||||
|
c, err := ln.Accept()
|
||||||
|
if err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
_ = c.Close()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
if code := cmdEgressGate([]string{"--probe", ln.Addr().String(), "--wait", "100ms"}, &out, &errb); code != 1 {
|
||||||
|
t.Fatalf("exit %d, want 1", code)
|
||||||
|
}
|
||||||
|
if !strings.Contains(errb.String(), "not enforced") {
|
||||||
|
t.Errorf("stderr = %q", errb.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEgressGateDefaultsToTheKubernetesService(t *testing.T) {
|
||||||
|
shrinkEgressGate(t)
|
||||||
|
t.Setenv("KUBERNETES_SERVICE_HOST", "")
|
||||||
|
t.Setenv("KUBERNETES_SERVICE_PORT", "")
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
if code := cmdEgressGate(nil, &out, &errb); code != 2 {
|
||||||
|
t.Fatalf("exit %d without a probe, want 2", code)
|
||||||
|
}
|
||||||
|
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
host, port, _ := net.SplitHostPort(ln.Addr().String())
|
||||||
|
_ = ln.Close() // closed: the lock reads as in effect at once
|
||||||
|
t.Setenv("KUBERNETES_SERVICE_HOST", host)
|
||||||
|
t.Setenv("KUBERNETES_SERVICE_PORT", port)
|
||||||
|
out.Reset()
|
||||||
|
if code := cmdEgressGate(nil, &out, &errb); code != 0 || !strings.Contains(out.String(), ln.Addr().String()) {
|
||||||
|
t.Fatalf("exit %d, stdout %q", code, out.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -4,6 +4,8 @@ import (
|
|||||||
"archive/tar"
|
"archive/tar"
|
||||||
"compress/gzip"
|
"compress/gzip"
|
||||||
"context"
|
"context"
|
||||||
|
"crypto/sha256"
|
||||||
|
"encoding/hex"
|
||||||
"errors"
|
"errors"
|
||||||
"flag"
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
@@ -15,6 +17,8 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"syscall"
|
"syscall"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/build"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdFetchContext is the in-Pod entrypoint the build Job's context-fetch
|
// cmdFetchContext is the in-Pod entrypoint the build Job's context-fetch
|
||||||
@@ -41,9 +45,14 @@ func cmdFetchContext(args []string, _, stderr io.Writer) int {
|
|||||||
fs.SetOutput(stderr)
|
fs.SetOutput(stderr)
|
||||||
url := fs.String("url", "", "internal-face URL of the submission's build-context tarball")
|
url := fs.String("url", "", "internal-face URL of the submission's build-context tarball")
|
||||||
out := fs.String("out", "/context", "directory to extract the build context into")
|
out := fs.String("out", "/context", "directory to extract the build context into")
|
||||||
|
want := fs.String("sha256", "", "refuse the context unless the tarball's sha256 is this lowercase hex digest")
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
|
if *want != "" && !build.IsSHA256Hex(*want) {
|
||||||
|
fmt.Fprintf(stderr, "felis fetch-context: --sha256 %q is not a lowercase hex sha256\n", *want)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
if *url == "" {
|
if *url == "" {
|
||||||
fmt.Fprintln(stderr, "felis fetch-context: --url is required")
|
fmt.Fprintln(stderr, "felis fetch-context: --url is required")
|
||||||
return 2
|
return 2
|
||||||
@@ -66,7 +75,12 @@ func cmdFetchContext(args []string, _, stderr io.Writer) int {
|
|||||||
// No overall client timeout: a legitimate modpack context can be large and the
|
// No overall client timeout: a legitimate modpack context can be large and the
|
||||||
// Job's activeDeadlineSeconds is the real bound. The header timeout catches a
|
// Job's activeDeadlineSeconds is the real bound. The header timeout catches a
|
||||||
// wedged endpoint without capping a healthy download.
|
// wedged endpoint without capping a healthy download.
|
||||||
client := &http.Client{Transport: &http.Transport{ResponseHeaderTimeout: time.Minute}}
|
// Redirects are refused: the request carries the service token, and the
|
||||||
|
// internal face never redirects, so a 3xx is someone steering the token.
|
||||||
|
client := &http.Client{
|
||||||
|
Transport: &http.Transport{ResponseHeaderTimeout: time.Minute},
|
||||||
|
CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse },
|
||||||
|
}
|
||||||
resp, err := fetchContextWithRetry(ctx, client, *url, token, stderr)
|
resp, err := fetchContextWithRetry(ctx, client, *url, token, stderr)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
||||||
@@ -74,10 +88,28 @@ func cmdFetchContext(args []string, _, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
defer resp.Body.Close()
|
defer resp.Body.Close()
|
||||||
|
|
||||||
if err := extractTarGz(resp.Body, *out); err != nil {
|
h := sha256.New()
|
||||||
|
body := io.TeeReader(resp.Body, h)
|
||||||
|
if err := extractTarGz(body, *out); err != nil {
|
||||||
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
if *want == "" {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
// The tar end marker comes before the gzip trailer and whatever follows it,
|
||||||
|
// so read to EOF: the digest must cover every byte the blob holds. The blob
|
||||||
|
// itself is size-capped at upload, which bounds this read.
|
||||||
|
if _, err := io.Copy(io.Discard, io.LimitReader(body, maxContextBytes)); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if got := hex.EncodeToString(h.Sum(nil)); got != *want {
|
||||||
|
// The init container failing is what keeps Kaniko from ever starting on
|
||||||
|
// the extracted tree.
|
||||||
|
fmt.Fprintf(stderr, "felis fetch-context: the context's sha256 is %s, the approved digest is %s: it changed after approval; refusing to build\n", got, *want)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -136,13 +168,27 @@ func fetchContextWithRetry(ctx context.Context, client *http.Client, url, token
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// maxContextBytes / maxContextEntries bound what one context may expand to. The
|
||||||
|
// compressed upload is capped at 1 GiB, but gzip turns that into hundreds of GiB
|
||||||
|
// or millions of empty files, and the emptyDir's 4 GiB sizeLimit is only
|
||||||
|
// enforced by the kubelet's periodic sweep, after the disk has filled. The byte
|
||||||
|
// cap matches that sizeLimit; the entry cap is far above any real modpack (a
|
||||||
|
// large one is a few thousand files) and far below an inode exhaustion.
|
||||||
|
//
|
||||||
|
// Vars, not consts, so tests can shrink them.
|
||||||
|
var (
|
||||||
|
maxContextBytes int64 = 4 << 30
|
||||||
|
maxContextEntries = 200_000
|
||||||
|
)
|
||||||
|
|
||||||
// extractTarGz streams a gzip'd tarball into root, creating directories as
|
// extractTarGz streams a gzip'd tarball into root, creating directories as
|
||||||
// needed. Every entry is vetted BEFORE anything is written: a path that is
|
// needed. Every entry is vetted BEFORE anything is written: a path that is
|
||||||
// absolute or escapes root (via ".."), a link (symlink or hardlink), or any
|
// absolute or escapes root (via ".."), a link (symlink or hardlink), or any
|
||||||
// special file kind aborts the whole extraction. Refusing rather than skipping is
|
// special file kind aborts the whole extraction. Refusing rather than skipping is
|
||||||
// deliberate — a context that needs one of those constructs is not a context this
|
// deliberate — a context that needs one of those constructs is not a context this
|
||||||
// transport carries, and silently dropping entries would build from a corpus the
|
// transport carries, and silently dropping entries would build from a corpus the
|
||||||
// submitter did not upload.
|
// submitter did not upload. The whole extraction is also bounded by
|
||||||
|
// maxContextBytes and maxContextEntries.
|
||||||
func extractTarGz(r io.Reader, root string) error {
|
func extractTarGz(r io.Reader, root string) error {
|
||||||
if err := os.MkdirAll(root, 0o755); err != nil {
|
if err := os.MkdirAll(root, 0o755); err != nil {
|
||||||
return fmt.Errorf("create context dir: %w", err)
|
return fmt.Errorf("create context dir: %w", err)
|
||||||
@@ -153,6 +199,8 @@ func extractTarGz(r io.Reader, root string) error {
|
|||||||
}
|
}
|
||||||
defer zr.Close()
|
defer zr.Close()
|
||||||
tr := tar.NewReader(zr)
|
tr := tar.NewReader(zr)
|
||||||
|
var written int64
|
||||||
|
entries := 0
|
||||||
for {
|
for {
|
||||||
hdr, err := tr.Next()
|
hdr, err := tr.Next()
|
||||||
if errors.Is(err, io.EOF) {
|
if errors.Is(err, io.EOF) {
|
||||||
@@ -161,6 +209,9 @@ func extractTarGz(r io.Reader, root string) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read context tarball: %w", err)
|
return fmt.Errorf("read context tarball: %w", err)
|
||||||
}
|
}
|
||||||
|
if entries++; entries > maxContextEntries {
|
||||||
|
return fmt.Errorf("the build context has more than %d entries", maxContextEntries)
|
||||||
|
}
|
||||||
name := filepath.Clean(hdr.Name)
|
name := filepath.Clean(hdr.Name)
|
||||||
if name == "." {
|
if name == "." {
|
||||||
continue
|
continue
|
||||||
@@ -176,7 +227,7 @@ func extractTarGz(r io.Reader, root string) error {
|
|||||||
if err := os.MkdirAll(target, 0o755); err != nil {
|
if err := os.MkdirAll(target, 0o755); err != nil {
|
||||||
return fmt.Errorf("create %q: %w", name, err)
|
return fmt.Errorf("create %q: %w", name, err)
|
||||||
}
|
}
|
||||||
case tar.TypeReg, tar.TypeRegA:
|
case tar.TypeReg:
|
||||||
if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil {
|
if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil {
|
||||||
return fmt.Errorf("create parent of %q: %w", name, err)
|
return fmt.Errorf("create parent of %q: %w", name, err)
|
||||||
}
|
}
|
||||||
@@ -188,10 +239,16 @@ func extractTarGz(r io.Reader, root string) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("create %q: %w", name, err)
|
return fmt.Errorf("create %q: %w", name, err)
|
||||||
}
|
}
|
||||||
if _, err := io.Copy(f, tr); err != nil {
|
n, err := io.Copy(f, io.LimitReader(tr, maxContextBytes-written+1))
|
||||||
|
written += n
|
||||||
|
if err != nil {
|
||||||
_ = f.Close()
|
_ = f.Close()
|
||||||
return fmt.Errorf("write %q: %w", name, err)
|
return fmt.Errorf("write %q: %w", name, err)
|
||||||
}
|
}
|
||||||
|
if written > maxContextBytes {
|
||||||
|
_ = f.Close()
|
||||||
|
return fmt.Errorf("the build context expands past %d bytes", maxContextBytes)
|
||||||
|
}
|
||||||
if err := f.Close(); err != nil {
|
if err := f.Close(); err != nil {
|
||||||
return fmt.Errorf("close %q: %w", name, err)
|
return fmt.Errorf("close %q: %w", name, err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,8 @@ import (
|
|||||||
"archive/tar"
|
"archive/tar"
|
||||||
"bytes"
|
"bytes"
|
||||||
"compress/gzip"
|
"compress/gzip"
|
||||||
|
"crypto/sha256"
|
||||||
|
"encoding/hex"
|
||||||
"io"
|
"io"
|
||||||
"net"
|
"net"
|
||||||
"net/http"
|
"net/http"
|
||||||
@@ -121,6 +123,30 @@ func TestExtractTarGzRefusesEscapes(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A context that expands past the byte or entry cap is refused, however small
|
||||||
|
// it was compressed: gzip bombs and inode floods stop at the cap.
|
||||||
|
func TestExtractTarGzCapsExpansion(t *testing.T) {
|
||||||
|
bytesCap, entriesCap := maxContextBytes, maxContextEntries
|
||||||
|
t.Cleanup(func() { maxContextBytes, maxContextEntries = bytesCap, entriesCap })
|
||||||
|
maxContextBytes, maxContextEntries = 1000, 5
|
||||||
|
|
||||||
|
fits := tgzBody(t, tarEntry{name: "a", body: strings.Repeat("x", 600)}, tarEntry{name: "b", body: strings.Repeat("y", 400)})
|
||||||
|
if err := extractTarGz(bytes.NewReader(fits), t.TempDir()); err != nil {
|
||||||
|
t.Fatalf("a context exactly at the byte cap: %v", err)
|
||||||
|
}
|
||||||
|
big := tgzBody(t, tarEntry{name: "a", body: strings.Repeat("x", 600)}, tarEntry{name: "b", body: strings.Repeat("y", 401)})
|
||||||
|
if err := extractTarGz(bytes.NewReader(big), t.TempDir()); err == nil || !strings.Contains(err.Error(), "expands past") {
|
||||||
|
t.Fatalf("one byte over the cap: err = %v", err)
|
||||||
|
}
|
||||||
|
var many []tarEntry
|
||||||
|
for i := 0; i < 6; i++ {
|
||||||
|
many = append(many, tarEntry{name: "d" + string(rune('0'+i)) + "/", typ: tar.TypeDir})
|
||||||
|
}
|
||||||
|
if err := extractTarGz(bytes.NewReader(tgzBody(t, many...)), t.TempDir()); err == nil || !strings.Contains(err.Error(), "entries") {
|
||||||
|
t.Fatalf("six entries over a cap of five: err = %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// The command end to end: it dials the URL with the bearer token from the
|
// The command end to end: it dials the URL with the bearer token from the
|
||||||
// environment, and refuses to run without it (the internal face would 401
|
// environment, and refuses to run without it (the internal face would 401
|
||||||
// anyway; failing at parse time is the honest earlier error).
|
// anyway; failing at parse time is the honest earlier error).
|
||||||
@@ -163,6 +189,69 @@ func TestCmdFetchContextFetchAndExtract(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// With --sha256 the fetch refuses any bytes but the approved ones, including
|
||||||
|
// a tarball that extracts cleanly: that is exactly the context an uploader
|
||||||
|
// swapped in after the review.
|
||||||
|
func TestCmdFetchContextChecksDigest(t *testing.T) {
|
||||||
|
body := tgzBody(t, tarEntry{name: "Dockerfile", body: "FROM scratch\n"})
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||||
|
_, _ = w.Write(body)
|
||||||
|
}))
|
||||||
|
defer srv.Close()
|
||||||
|
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||||
|
sum := sha256.Sum256(body)
|
||||||
|
good := hex.EncodeToString(sum[:])
|
||||||
|
args := func(digest string) []string {
|
||||||
|
return []string{"--url=" + srv.URL + "/sub-1/context", "--out=" + t.TempDir(), "--sha256=" + digest}
|
||||||
|
}
|
||||||
|
|
||||||
|
if code := cmdFetchContext(args(good), io.Discard, io.Discard); code != 0 {
|
||||||
|
t.Fatalf("matching digest exit = %d, want 0", code)
|
||||||
|
}
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
other := strings.Repeat("0", 64)
|
||||||
|
if code := cmdFetchContext(args(other), io.Discard, &stderr); code != 1 || !strings.Contains(stderr.String(), "changed after approval") {
|
||||||
|
t.Fatalf("mismatched digest exit = %d, stderr %q; want 1 naming the change", code, stderr.String())
|
||||||
|
}
|
||||||
|
stderr.Reset()
|
||||||
|
if code := cmdFetchContext(args("ABC"), io.Discard, &stderr); code != 2 {
|
||||||
|
t.Fatalf("malformed digest exit = %d, want 2 (stderr %q)", code, stderr.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
// Bytes after the tar end marker still count: appending to an approved blob
|
||||||
|
// must change what the fetch accepts.
|
||||||
|
padded := append(append([]byte{}, body...), "trailing"...)
|
||||||
|
srvPadded := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||||
|
_, _ = w.Write(padded)
|
||||||
|
}))
|
||||||
|
defer srvPadded.Close()
|
||||||
|
if code := cmdFetchContext([]string{"--url=" + srvPadded.URL + "/c", "--out=" + t.TempDir(), "--sha256=" + good}, io.Discard, io.Discard); code != 1 {
|
||||||
|
t.Fatalf("padded blob exit = %d, want 1", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The request carries the service token, so a redirect is a failure: the token
|
||||||
|
// never follows it to another host (build-supply-chain-13).
|
||||||
|
func TestCmdFetchContextRefusesRedirects(t *testing.T) {
|
||||||
|
var leaked bool
|
||||||
|
elsewhere := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
leaked = true
|
||||||
|
}))
|
||||||
|
defer elsewhere.Close()
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
http.Redirect(w, r, elsewhere.URL+"/steal", http.StatusFound)
|
||||||
|
}))
|
||||||
|
defer srv.Close()
|
||||||
|
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
if code := cmdFetchContext([]string{"--url=" + srv.URL + "/c", "--out=" + t.TempDir()}, io.Discard, &stderr); code != 1 {
|
||||||
|
t.Fatalf("redirect exit = %d, want 1 (stderr %q)", code, stderr.String())
|
||||||
|
}
|
||||||
|
if leaked {
|
||||||
|
t.Fatal("the fetch followed the redirect")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// A body that is not a gzip tarball must fail the extraction rather than produce
|
// A body that is not a gzip tarball must fail the extraction rather than produce
|
||||||
// an empty (or partial) context Kaniko would then try to build.
|
// an empty (or partial) context Kaniko would then try to build.
|
||||||
func TestExtractTarGzRejectsNonGzip(t *testing.T) {
|
func TestExtractTarGzRejectsNonGzip(t *testing.T) {
|
||||||
|
|||||||
+11
-16
@@ -20,18 +20,14 @@ const forwardingSecretEnv = "FELIS_FORWARDING_SECRET"
|
|||||||
// config/ and server.properties live under it.
|
// config/ and server.properties live under it.
|
||||||
const defaultForwardingDataDir = "/data"
|
const defaultForwardingDataDir = "/data"
|
||||||
|
|
||||||
// fwd*Mode make the written config readable AND rewritable by the main server
|
// fwd*Mode are the modes the written config lands with. The initContainer runs as
|
||||||
// container, whose UID we do not control (an arbitrary user image). The
|
// the same uid as the server container (naming.GameUID, pinned by the operator in
|
||||||
// initContainer runs as root (see buildStatefulSet) so it can write into a data
|
// the pod securityContext) after the prepare-data initContainer has handed the
|
||||||
// volume of unknown ownership; 0666/0777 then let a non-root Paper rewrite the
|
// whole volume to that uid, so owner read/write is all the server needs to rewrite
|
||||||
// same files on boot.
|
// these files on boot and nothing else on the node gets write access to them.
|
||||||
//
|
|
||||||
// This relies on the initContainer running as root to write into a volume of
|
|
||||||
// unknown ownership; that is how the operator schedules it. If that ever changes,
|
|
||||||
// give the server pod an fsGroup so the shared volume is group-writable instead.
|
|
||||||
const (
|
const (
|
||||||
fwdFileMode os.FileMode = 0o666
|
fwdFileMode os.FileMode = 0o644
|
||||||
fwdDirMode os.FileMode = 0o777
|
fwdDirMode os.FileMode = 0o755
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdInitForwarding is the felis-image initContainer entrypoint that makes an
|
// cmdInitForwarding is the felis-image initContainer entrypoint that makes an
|
||||||
@@ -96,9 +92,8 @@ func writePaperGlobal(dataDir, secret string) error {
|
|||||||
if err := os.MkdirAll(dir, fwdDirMode); err != nil {
|
if err := os.MkdirAll(dir, fwdDirMode); err != nil {
|
||||||
return fmt.Errorf("create %s: %w", dir, err)
|
return fmt.Errorf("create %s: %w", dir, err)
|
||||||
}
|
}
|
||||||
// MkdirAll honours the process umask (root's is typically 022 → 0755); chmod
|
// MkdirAll honours the process umask; chmod does not, so a directory an older
|
||||||
// does not, and a non-root main container must be able to place/replace the
|
// release left at 0777 is brought back to fwdDirMode here.
|
||||||
// file in this directory on boot.
|
|
||||||
if err := os.Chmod(dir, fwdDirMode); err != nil {
|
if err := os.Chmod(dir, fwdDirMode); err != nil {
|
||||||
return fmt.Errorf("chmod %s: %w", dir, err)
|
return fmt.Errorf("chmod %s: %w", dir, err)
|
||||||
}
|
}
|
||||||
@@ -188,8 +183,8 @@ func upsertProperty(content []byte, key, value string) []byte {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// writeFileMode writes data then forces the mode, since WriteFile honours the
|
// writeFileMode writes data then forces the mode, since WriteFile honours the
|
||||||
// umask (root's is typically 022 → 0644) but a non-root main container must be
|
// umask and leaves an existing file's mode alone: a file an older release wrote
|
||||||
// able to rewrite these files on boot.
|
// world-writable (0666) is tightened back to fwdFileMode on the next boot.
|
||||||
func writeFileMode(path string, data []byte) error {
|
func writeFileMode(path string, data []byte) error {
|
||||||
if err := os.WriteFile(path, data, fwdFileMode); err != nil {
|
if err := os.WriteFile(path, data, fwdFileMode); err != nil {
|
||||||
return fmt.Errorf("write %s: %w", path, err)
|
return fmt.Errorf("write %s: %w", path, err)
|
||||||
|
|||||||
@@ -164,8 +164,9 @@ func TestUpsertPropertyAppends(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// The written files must be group/world writable so a non-root main container can
|
// The written files land at fwdFileMode: owner-writable for the game uid the init
|
||||||
// rewrite them. chmod semantics are POSIX-only, so this asserts on non-Windows.
|
// shares with the server container, and no longer world-writable. chmod semantics
|
||||||
|
// are POSIX-only, so this asserts on non-Windows.
|
||||||
func TestWriteForwardingFileModes(t *testing.T) {
|
func TestWriteForwardingFileModes(t *testing.T) {
|
||||||
if runtime.GOOS == "windows" {
|
if runtime.GOOS == "windows" {
|
||||||
t.Skip("POSIX file modes not represented on Windows")
|
t.Skip("POSIX file modes not represented on Windows")
|
||||||
|
|||||||
@@ -0,0 +1,117 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"io/fs"
|
||||||
|
"os"
|
||||||
|
"syscall"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/naming"
|
||||||
|
)
|
||||||
|
|
||||||
|
// cmdInitVolume is the felis-image `prepare-data` initContainer entrypoint: it
|
||||||
|
// hands every entry of a server's world volume to the game uid/gid before the
|
||||||
|
// server container starts. The operator runs the server itself as naming.GameUID,
|
||||||
|
// so a world written by an earlier release (whose server ran as root), a restore
|
||||||
|
// Job (which extracts as root), or a storage provisioner that creates the volume
|
||||||
|
// root-owned would otherwise leave files the server cannot write — a world that
|
||||||
|
// boots and then fails every save.
|
||||||
|
//
|
||||||
|
// fsGroup covers only part of this: kubelet applies it to volume types that
|
||||||
|
// support ownership management, and a k3s local-path PV is a hostPath underneath,
|
||||||
|
// which it skips. A walk from inside the pod works for every volume type.
|
||||||
|
//
|
||||||
|
// Only mismatched entries are touched, so a volume already owned by the game uid
|
||||||
|
// costs one lstat per entry and no writes. The walk runs inside an os.Root at the
|
||||||
|
// data dir and uses lchown, so a symlink a plugin planted is re-owned as a link
|
||||||
|
// and never followed out of the volume.
|
||||||
|
//
|
||||||
|
// A single entry that cannot be chowned is reported and skipped: failing the pod
|
||||||
|
// over one odd file would keep the whole server down, while the server itself
|
||||||
|
// reports the one file it cannot write. Only an unreadable data dir fails.
|
||||||
|
func cmdInitVolume(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("init-volume", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
dataDir := fs.String("data", defaultForwardingDataDir, "world volume mount to hand to the game uid")
|
||||||
|
uid := fs.Int64("uid", naming.GameUID, "owner uid for every entry")
|
||||||
|
gid := fs.Int64("gid", naming.GameGID, "owner gid for every entry")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
root, err := os.OpenRoot(*dataDir)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis init-volume: open %s: %v\n", *dataDir, err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer root.Close()
|
||||||
|
res, err := chownTree(root, int(*uid), int(*gid), root.Lchown)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis init-volume: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
for _, f := range res.failures {
|
||||||
|
fmt.Fprintf(stderr, "felis init-volume: %s\n", f)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis init-volume: %d entries checked, %d handed to %d:%d, %d failed\n",
|
||||||
|
res.checked, res.changed, *uid, *gid, len(res.failures))
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// chownResult tallies one walk; failures is capped so a volume of thousands of
|
||||||
|
// unownable files cannot flood the pod log.
|
||||||
|
type chownResult struct {
|
||||||
|
checked int
|
||||||
|
changed int
|
||||||
|
failures []string
|
||||||
|
}
|
||||||
|
|
||||||
|
const maxReportedChownFailures = 20
|
||||||
|
|
||||||
|
// chownTree walks root and calls chown on every entry (the root dir included)
|
||||||
|
// whose owner is not uid:gid. It returns an error only when the root itself
|
||||||
|
// cannot be read; per-entry failures are collected in the result.
|
||||||
|
func chownTree(root *os.Root, uid, gid int, chown func(name string, uid, gid int) error) (chownResult, error) {
|
||||||
|
var res chownResult
|
||||||
|
fail := func(name string, err error) {
|
||||||
|
if len(res.failures) < maxReportedChownFailures {
|
||||||
|
res.failures = append(res.failures, fmt.Sprintf("%s: %v", name, err))
|
||||||
|
} else if len(res.failures) == maxReportedChownFailures {
|
||||||
|
res.failures = append(res.failures, "further failures not listed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
err := fs.WalkDir(root.FS(), ".", func(name string, d fs.DirEntry, walkErr error) error {
|
||||||
|
if walkErr != nil {
|
||||||
|
if name == "." {
|
||||||
|
return walkErr
|
||||||
|
}
|
||||||
|
fail(name, walkErr)
|
||||||
|
// A directory that cannot be listed is skipped as a whole; a file
|
||||||
|
// error has nothing below it to skip.
|
||||||
|
if d != nil && d.IsDir() {
|
||||||
|
return fs.SkipDir
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
res.checked++
|
||||||
|
info, err := d.Info()
|
||||||
|
if err != nil {
|
||||||
|
fail(name, err)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if st, ok := info.Sys().(*syscall.Stat_t); ok && int(st.Uid) == uid && int(st.Gid) == gid {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if err := chown(name, uid, gid); err != nil {
|
||||||
|
if !errors.Is(err, fs.ErrNotExist) { // gone mid-walk: nothing left to own
|
||||||
|
fail(name, err)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
res.changed++
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
return res, err
|
||||||
|
}
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"slices"
|
||||||
|
"strconv"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// openTree builds a small world under a temp dir: nested dirs, a file, and a
|
||||||
|
// symlink pointing out of the volume that the walk must not follow.
|
||||||
|
func openTree(t *testing.T) *os.Root {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
for _, d := range []string{"world/region", "plugins"} {
|
||||||
|
if err := os.MkdirAll(filepath.Join(dir, d), 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, f := range []string{"level.dat", "world/region/r.0.0.mca"} {
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, f), []byte("x"), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
outside := t.TempDir()
|
||||||
|
if err := os.Symlink(outside, filepath.Join(dir, "plugins", "escape")); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
root, err := os.OpenRoot(dir)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { root.Close() })
|
||||||
|
return root
|
||||||
|
}
|
||||||
|
|
||||||
|
// Every entry owned by someone else is handed over, the root dir included, and a
|
||||||
|
// symlink is re-owned as a link rather than walked into.
|
||||||
|
func TestChownTreeHandsOverMismatchedEntries(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("POSIX ownership not represented on Windows")
|
||||||
|
}
|
||||||
|
root := openTree(t)
|
||||||
|
var got []string
|
||||||
|
res, err := chownTree(root, os.Getuid()+1, os.Getgid(), func(name string, uid, gid int) error {
|
||||||
|
got = append(got, name)
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("chownTree: %v", err)
|
||||||
|
}
|
||||||
|
want := []string{".", "level.dat", "plugins", "plugins/escape", "world", "world/region", "world/region/r.0.0.mca"}
|
||||||
|
slices.Sort(got)
|
||||||
|
if !slices.Equal(got, want) {
|
||||||
|
t.Errorf("chowned %v, want %v", got, want)
|
||||||
|
}
|
||||||
|
if res.changed != len(want) || res.checked != len(want) || len(res.failures) != 0 {
|
||||||
|
t.Errorf("result = %+v, want %d checked and changed, no failures", res, len(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A volume already owned by the game uid costs no chown at all: this is the steady
|
||||||
|
// state every restart after the first one hits.
|
||||||
|
func TestChownTreeSkipsMatchingOwner(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("POSIX ownership not represented on Windows")
|
||||||
|
}
|
||||||
|
root := openTree(t)
|
||||||
|
calls := 0
|
||||||
|
res, err := chownTree(root, os.Getuid(), os.Getgid(), func(string, int, int) error {
|
||||||
|
calls++
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("chownTree: %v", err)
|
||||||
|
}
|
||||||
|
if calls != 0 || res.changed != 0 {
|
||||||
|
t.Errorf("chown called %d times on an already-owned tree (result %+v)", calls, res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// One entry that refuses the chown is reported and the walk carries on: a single
|
||||||
|
// odd file must not keep the whole server from starting.
|
||||||
|
func TestChownTreeContinuesPastFailures(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("POSIX ownership not represented on Windows")
|
||||||
|
}
|
||||||
|
root := openTree(t)
|
||||||
|
res, err := chownTree(root, os.Getuid()+1, os.Getgid(), func(name string, uid, gid int) error {
|
||||||
|
if name == "level.dat" {
|
||||||
|
return os.ErrPermission
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("chownTree: %v", err)
|
||||||
|
}
|
||||||
|
if len(res.failures) != 1 || res.changed != 6 {
|
||||||
|
t.Errorf("result = %+v, want 1 failure and 6 changed", res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Against a real directory the owner already matches, so the command succeeds
|
||||||
|
// without needing CAP_CHOWN — the path every test runner (non-root) can take.
|
||||||
|
func TestCmdInitVolumeOwnedTree(t *testing.T) {
|
||||||
|
if runtime.GOOS == "windows" {
|
||||||
|
t.Skip("POSIX ownership not represented on Windows")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "server.properties"), []byte("x"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
code := cmdInitVolume([]string{"--data", dir, "--uid", strconv.Itoa(os.Getuid()), "--gid", strconv.Itoa(os.Getgid())}, &out, &errb)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("exit %d, stderr %q", code, errb.String())
|
||||||
|
}
|
||||||
|
if code := cmdInitVolume([]string{"--data", filepath.Join(dir, "missing")}, &out, &errb); code != 1 {
|
||||||
|
t.Errorf("missing data dir exit = %d, want 1", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
+27
-6
@@ -45,15 +45,22 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
|||||||
registryPort := fs.Int("registry-port", 5000, "port the in-cluster registry listens on")
|
registryPort := fs.Int("registry-port", 5000, "port the in-cluster registry listens on")
|
||||||
panelNodePort := fs.Int("panel-node-port", int(platform.DefaultPanelNodePort), "NodePort that exposes the built-in HTTPS panel/API origin")
|
panelNodePort := fs.Int("panel-node-port", int(platform.DefaultPanelNodePort), "NodePort that exposes the built-in HTTPS panel/API origin")
|
||||||
felisImage := fs.String("felis-image", "", "container image the felis-api/operator Deployments run, also passed through as FELIS_IMAGE (REQUIRED)")
|
felisImage := fs.String("felis-image", "", "container image the felis-api/operator Deployments run, also passed through as FELIS_IMAGE (REQUIRED)")
|
||||||
registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry:2)")
|
registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry 2.8.3, pinned by digest)")
|
||||||
backupPVC := fs.String("backup-pvc", "felis-backups", "name of the world-archive PVC this bundle renders in the Minecraft namespace and advertises to the backup/restore executors via FELIS_BACKUP_PVC (default: felis-backups; pass an empty value to render none, leaving backup/restore answering 503)")
|
backupPVC := fs.String("backup-pvc", "felis-backups", "name of the world-archive PVC this bundle renders in the Minecraft namespace and advertises to the backup/restore executors via FELIS_BACKUP_PVC (default: felis-backups; pass an empty value to render none, leaving backup/restore answering 503)")
|
||||||
worldsHostPath := fs.String("worlds-host-path", "", "node directory the reaper reads worlds from: each world PVC resolves as <path>/<pvc>, or as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (k3s storage root: /var/lib/rancher/k3s/storage); enables the reaper CronJob (requires --archive-local-path and a non-empty --backup-pvc)")
|
worldsHostPath := fs.String("worlds-host-path", "", "node directory the reaper reads worlds from: each world PVC resolves as <path>/<pvc>, or as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (k3s storage root: /var/lib/rancher/k3s/storage); enables the reaper CronJob (requires --archive-local-path and a non-empty --backup-pvc)")
|
||||||
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path")
|
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path")
|
||||||
|
registryStorage := fs.String("registry-storage", "", "capacity the registry PVC requests (default 10Gi; k3s local-path does not enforce it)")
|
||||||
|
uploadsStorage := fs.String("uploads-storage", "", "capacity the uploads PVC requests (default 5Gi; k3s local-path does not enforce it)")
|
||||||
|
backupStorage := fs.String("backup-storage", "", "capacity the world-archive PVC requests (default 10Gi; k3s local-path does not enforce it)")
|
||||||
reaperNode := fs.String("reaper-node", "", "node that holds --worlds-host-path: pins the reaper CronJob's pod there via nodeSelector kubernetes.io/hostname (multi-node clusters need this, or the reaper may schedule where the hostPath is empty)")
|
reaperNode := fs.String("reaper-node", "", "node that holds --worlds-host-path: pins the reaper CronJob's pod there via nodeSelector kubernetes.io/hostname (multi-node clusters need this, or the reaper may schedule where the hostPath is empty)")
|
||||||
var velocityCIDRs multiFlag
|
var velocityCIDRs multiFlag
|
||||||
fs.Var(&velocityCIDRs, "velocity-cidr", "CIDR of a Velocity proxy host allowed to reach game port 25565 (repeatable, REQUIRED)")
|
fs.Var(&velocityCIDRs, "velocity-cidr", "CIDR of a Velocity proxy host allowed to reach game port 25565 (repeatable, REQUIRED)")
|
||||||
var packageCIDRs multiFlag
|
var packageCIDRs multiFlag
|
||||||
fs.Var(&packageCIDRs, "package-cidr", "CIDR of a package mirror build Pods may reach (repeatable; default none = no internet egress)")
|
fs.Var(&packageCIDRs, "package-cidr", "CIDR of a package mirror build Pods may reach (repeatable; default none = no internet egress)")
|
||||||
|
var serverDenyCIDRs multiFlag
|
||||||
|
fs.Var(&serverDenyCIDRs, "server-egress-deny-cidr", "extra CIDR game server pods may never reach, e.g. the node's public address (repeatable)")
|
||||||
|
var serverAllowCIDRs multiFlag
|
||||||
|
fs.Var(&serverAllowCIDRs, "server-egress-allow-cidr", "private CIDR game server pods may reach despite the private-range block, e.g. a LAN database (repeatable)")
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
@@ -74,7 +81,9 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
|||||||
"(the felis-api/operator Deployments run it and it is passed through as FELIS_IMAGE, e.g. --felis-image registry.felis.svc:5000/felis:v1)")
|
"(the felis-api/operator Deployments run it and it is passed through as FELIS_IMAGE, e.g. --felis-image registry.felis.svc:5000/felis:v1)")
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
for _, cidr := range append(append([]string{}, velocityCIDRs...), packageCIDRs...) {
|
allCIDRs := append(append([]string{}, velocityCIDRs...), packageCIDRs...)
|
||||||
|
allCIDRs = append(append(allCIDRs, serverDenyCIDRs...), serverAllowCIDRs...)
|
||||||
|
for _, cidr := range allCIDRs {
|
||||||
if _, _, err := net.ParseCIDR(cidr); err != nil {
|
if _, _, err := net.ParseCIDR(cidr); err != nil {
|
||||||
fmt.Fprintf(stderr, "felis manifests: invalid CIDR %q: %v\n", cidr, err)
|
fmt.Fprintf(stderr, "felis manifests: invalid CIDR %q: %v\n", cidr, err)
|
||||||
return 2
|
return 2
|
||||||
@@ -118,8 +127,8 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
|||||||
"multi-node cluster you MUST pass --reaper-node <name> (or add a nodeSelector) for the node holding the " +
|
"multi-node cluster you MUST pass --reaper-node <name> (or add a nodeSelector) for the node holding the " +
|
||||||
"worlds, or the reaper may schedule where the hostPath is empty"
|
"worlds, or the reaper may schedule where the hostPath is empty"
|
||||||
if *reaperNode != "" {
|
if *reaperNode != "" {
|
||||||
pin = fmt.Sprintf("the CronJob is pinned to node %q via kubernetes.io/hostname — keep this pointed at the "+
|
pin = fmt.Sprintf("the CronJob and its worlds-root PV are pinned to node %q via kubernetes.io/hostname — "+
|
||||||
"node that actually holds the world volumes", *reaperNode)
|
"keep this pointed at the node that actually holds the world volumes", *reaperNode)
|
||||||
}
|
}
|
||||||
fmt.Fprintf(stderr, "felis manifests: note: rendering the retention reaper CronJob (worlds hostPath %q). "+
|
fmt.Fprintf(stderr, "felis manifests: note: rendering the retention reaper CronJob (worlds hostPath %q). "+
|
||||||
"These points are NOT verified here:\n"+
|
"These points are NOT verified here:\n"+
|
||||||
@@ -133,7 +142,7 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
|||||||
"(pass --worlds-host-path and --archive-local-path — the archive PVC defaults to felis-backups — to enable it)")
|
"(pass --worlds-host-path and --archive-local-path — the archive PVC defaults to felis-backups — to enable it)")
|
||||||
}
|
}
|
||||||
|
|
||||||
out, err := platform.RenderYAML(platform.Params{
|
params := platform.Params{
|
||||||
ControlNamespace: *controlNS,
|
ControlNamespace: *controlNS,
|
||||||
MinecraftNamespace: *minecraftNS,
|
MinecraftNamespace: *minecraftNS,
|
||||||
BuildNamespace: *buildNS,
|
BuildNamespace: *buildNS,
|
||||||
@@ -148,7 +157,19 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
|||||||
ArchiveLocalPath: *archiveLocalPath,
|
ArchiveLocalPath: *archiveLocalPath,
|
||||||
VelocityCIDRs: []string(velocityCIDRs),
|
VelocityCIDRs: []string(velocityCIDRs),
|
||||||
PackageSourceCIDRs: []string(packageCIDRs),
|
PackageSourceCIDRs: []string(packageCIDRs),
|
||||||
})
|
|
||||||
|
ServerEgressDenyCIDRs: []string(serverDenyCIDRs),
|
||||||
|
ServerEgressAllowCIDRs: []string(serverAllowCIDRs),
|
||||||
|
|
||||||
|
RegistryStorage: *registryStorage,
|
||||||
|
UploadsStorage: *uploadsStorage,
|
||||||
|
BackupStorage: *backupStorage,
|
||||||
|
}
|
||||||
|
if err := params.Validate(); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis manifests: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
out, err := platform.RenderYAML(params)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis manifests: render: %v\n", err)
|
fmt.Fprintf(stderr, "felis manifests: render: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
@@ -219,3 +219,27 @@ func TestManifestsReaperNodePin(t *testing.T) {
|
|||||||
t.Errorf("--reaper-node without --worlds-host-path: exit = %d, want 2", code)
|
t.Errorf("--reaper-node without --worlds-host-path: exit = %d, want 2", code)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestManifestsStorageSizes proves the PVC size flags reach the rendered claims
|
||||||
|
// and a size the API server would reject fails before anything is applied.
|
||||||
|
func TestManifestsStorageSizes(t *testing.T) {
|
||||||
|
var out, errBuf bytes.Buffer
|
||||||
|
code := run([]string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "10.0.0.5/32",
|
||||||
|
"--registry-storage", "40Gi", "--uploads-storage", "8Gi"}, &out, &errBuf)
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("exit code = %d, stderr = %s", code, errBuf.String())
|
||||||
|
}
|
||||||
|
for _, want := range []string{"storage: 40Gi", "storage: 8Gi"} {
|
||||||
|
if !strings.Contains(out.String(), want) {
|
||||||
|
t.Errorf("bundle lacks %q", want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
out.Reset()
|
||||||
|
errBuf.Reset()
|
||||||
|
code = run([]string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "10.0.0.5/32",
|
||||||
|
"--registry-storage", "lots"}, &out, &errBuf)
|
||||||
|
if code == 0 || !strings.Contains(errBuf.String(), "registry storage") {
|
||||||
|
t.Errorf("--registry-storage lots: exit %d, stderr %q; want a refusal naming the flag", code, errBuf.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -7,15 +7,24 @@ import (
|
|||||||
"io"
|
"io"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/config"
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
"felis.lolicon.best/internal/store"
|
"felis.lolicon.best/internal/store"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdMigrate implements `felis migrate up`: load config, open the database, and
|
// cmdMigrate implements `felis migrate up`: load config, open the database, and
|
||||||
// apply every pending embedded migration under the advisory lock (spec §6).
|
// apply every pending embedded migration under the advisory lock (spec §6).
|
||||||
|
//
|
||||||
|
// Migrations only roll forward, and some drop data (0017_drop_password), so a
|
||||||
|
// database that already holds a schema and has migrations pending is bundled
|
||||||
|
// first (internal/dbbackup, label pre-migrate). A failed snapshot stops the
|
||||||
|
// upgrade; -no-backup is the explicit way past it, e.g. for an external
|
||||||
|
// database whose server is newer than the host's pg_dump.
|
||||||
func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
||||||
fs := flag.NewFlagSet("migrate", flag.ContinueOnError)
|
fs := flag.NewFlagSet("migrate", flag.ContinueOnError)
|
||||||
fs.SetOutput(stderr)
|
fs.SetOutput(stderr)
|
||||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
|
backupDir := fs.String("backup-dir", dbbackup.DefaultDir, "where the pre-migration snapshot goes")
|
||||||
|
noBackup := fs.Bool("no-backup", false, "apply pending migrations without snapshotting the database first")
|
||||||
// The "up" verb precedes any flags (felis migrate up -config path). Go's
|
// The "up" verb precedes any flags (felis migrate up -config path). Go's
|
||||||
// flag.Parse stops at the first non-flag token and would never see a flag
|
// flag.Parse stops at the first non-flag token and would never see a flag
|
||||||
// placed after "up", silently falling back to the default -config. Pull the
|
// placed after "up", silently falling back to the default -config. Pull the
|
||||||
@@ -48,6 +57,18 @@ func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if !*noBackup {
|
||||||
|
path, err := preMigrateBackup(ctx, drv, migrations, cfg.Database.URL, *backupDir, stderr)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis migrate: pre-migration backup failed, nothing applied: %v\n", err)
|
||||||
|
fmt.Fprintln(stderr, " fix the backup, or re-run with -no-backup to migrate without one")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if path != "" {
|
||||||
|
fmt.Fprintf(stdout, "felis migrate: database snapshot %s\n", path)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
applied, err := store.Up(ctx, drv, migrations)
|
applied, err := store.Up(ctx, drv, migrations)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(stderr, "felis migrate: %v\n", err)
|
fmt.Fprintf(stderr, "felis migrate: %v\n", err)
|
||||||
@@ -60,3 +81,33 @@ func cmdMigrate(args []string, stdout, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// preMigrateBackup bundles the database when it already carries a schema and
|
||||||
|
// some of migrations are not applied yet, and returns the bundle's path ("" when
|
||||||
|
// there was nothing to protect: a fresh database, or nothing pending).
|
||||||
|
func preMigrateBackup(ctx context.Context, drv store.Driver, migrations []store.Migration, dbURL, dir string, log io.Writer) (string, error) {
|
||||||
|
if err := drv.EnsureVersionTable(ctx); err != nil {
|
||||||
|
return "", fmt.Errorf("ensure version table: %w", err)
|
||||||
|
}
|
||||||
|
done, err := drv.AppliedVersions(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("read applied versions: %w", err)
|
||||||
|
}
|
||||||
|
if len(done) == 0 || !hasPending(done, migrations) {
|
||||||
|
return "", nil
|
||||||
|
}
|
||||||
|
return dbbackup.Backup(ctx, dbbackup.BackupOptions{
|
||||||
|
DatabaseURL: dbURL, Dir: dir, Label: dbbackup.LabelPreMigrate,
|
||||||
|
Keep: defaultKeep[dbbackup.LabelPreMigrate], StateDir: dbbackup.DefaultStateDir,
|
||||||
|
Version: resolvedVersion(), Log: log, Record: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func hasPending(done map[int]struct{}, migrations []store.Migration) bool {
|
||||||
|
for _, m := range migrations {
|
||||||
|
if _, ok := done[m.Version]; !ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -0,0 +1,125 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"os/signal"
|
||||||
|
"strings"
|
||||||
|
"syscall"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/build"
|
||||||
|
"felis.lolicon.best/internal/imagepush"
|
||||||
|
"felis.lolicon.best/internal/registrygate"
|
||||||
|
)
|
||||||
|
|
||||||
|
// defaultBuildToolsStatus is where mirror-build-tools records its last run; the
|
||||||
|
// watchdog reads it to tell a vulnerability DB that stopped refreshing.
|
||||||
|
const defaultBuildToolsStatus = "/var/lib/felis/build-tools/status.json"
|
||||||
|
|
||||||
|
// cmdMirrorBuildTools copies the build lane's tools (build.Tools: the kaniko and
|
||||||
|
// trivy images, Trivy's vulnerability and Java DBs) from upstream into the
|
||||||
|
// platform registry, where build Jobs pull them. deploy/bootstrap.sh runs it at
|
||||||
|
// install and from felis-build-tools.timer twice a day, which is what keeps the
|
||||||
|
// DBs fresh; a root shell can run it the same way to refresh now.
|
||||||
|
//
|
||||||
|
// It writes as the platform principal through the node's loopback hostPort, the
|
||||||
|
// same way the installer pushes, reading the token from the environment or from
|
||||||
|
// /etc/felis/secrets.env.
|
||||||
|
func cmdMirrorBuildTools(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("mirror-build-tools", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
endpoint := fs.String("endpoint", "127.0.0.1:5000", "host[:port] of the registry to write to (plain HTTP)")
|
||||||
|
only := fs.String("only", "", "comma-separated tool names to copy (default: all of "+toolNames()+")")
|
||||||
|
status := fs.String("status", defaultBuildToolsStatus, `file to record the run in ("" records nothing)`)
|
||||||
|
secrets := fs.String("secrets-env", "/etc/felis/secrets.env", "installer secrets file holding REGISTRY_PLATFORM_TOKEN, read when FELIS_REGISTRY_PASSWORD is unset")
|
||||||
|
platformFlag := fs.String("platform", "", "os/arch of the images to copy (default: this machine's)")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
if errors.Is(err, flag.ErrHelp) {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
tools, err := selectTools(*only)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis mirror-build-tools: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if err := loadEnvFile(*secrets); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis mirror-build-tools: read %s: %v\n", *secrets, err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
user, pass := os.Getenv("FELIS_REGISTRY_USERNAME"), os.Getenv("FELIS_REGISTRY_PASSWORD")
|
||||||
|
if pass == "" {
|
||||||
|
user, pass = registrygate.PrincipalPlatform, os.Getenv("REGISTRY_PLATFORM_TOKEN")
|
||||||
|
}
|
||||||
|
if pass == "" {
|
||||||
|
fmt.Fprintln(stderr, "felis mirror-build-tools: no registry credential: set FELIS_REGISTRY_PASSWORD or run as root on the node (REGISTRY_PLATFORM_TOKEN in /etc/felis/secrets.env)")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
|
||||||
|
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||||
|
defer stop()
|
||||||
|
p := &imagepush.Pusher{Scheme: "http", Username: user, Password: pass, Log: stdout}
|
||||||
|
src := &imagepush.Source{Platform: *platformFlag}
|
||||||
|
started := time.Now()
|
||||||
|
var failed []string
|
||||||
|
for _, t := range tools {
|
||||||
|
dst := strings.TrimSuffix(*endpoint, "/") + "/" + t.Mirror
|
||||||
|
if _, err := p.Mirror(ctx, src, t.Source, dst); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis mirror-build-tools: %s: %v\n", t.Name, err)
|
||||||
|
failed = append(failed, t.Name+": "+err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if *status != "" {
|
||||||
|
st, err := imagepush.ReadMirrorStatus(*status)
|
||||||
|
if err != nil || st == nil {
|
||||||
|
st = &imagepush.MirrorStatus{}
|
||||||
|
}
|
||||||
|
st.LastAttempt = started
|
||||||
|
st.LastError = strings.Join(failed, "; ")
|
||||||
|
if len(failed) == 0 {
|
||||||
|
st.LastSuccess = started
|
||||||
|
}
|
||||||
|
if err := imagepush.WriteMirrorStatus(*status, *st); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis mirror-build-tools: record %s: %v\n", *status, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(failed) > 0 {
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func toolNames() string {
|
||||||
|
var names []string
|
||||||
|
for _, t := range build.Tools {
|
||||||
|
names = append(names, t.Name)
|
||||||
|
}
|
||||||
|
return strings.Join(names, ",")
|
||||||
|
}
|
||||||
|
|
||||||
|
func selectTools(only string) ([]build.Tool, error) {
|
||||||
|
if only == "" {
|
||||||
|
return build.Tools, nil
|
||||||
|
}
|
||||||
|
var out []build.Tool
|
||||||
|
for _, name := range strings.Split(only, ",") {
|
||||||
|
name = strings.TrimSpace(name)
|
||||||
|
found := false
|
||||||
|
for _, t := range build.Tools {
|
||||||
|
if t.Name == name {
|
||||||
|
out = append(out, t)
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
return nil, fmt.Errorf("unknown tool %q (known: %s)", name, toolNames())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,556 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bufio"
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
|
"felis.lolicon.best/internal/offsite"
|
||||||
|
"felis.lolicon.best/internal/platform"
|
||||||
|
"felis.lolicon.best/internal/store"
|
||||||
|
appsv1 "k8s.io/api/apps/v1"
|
||||||
|
corev1 "k8s.io/api/core/v1"
|
||||||
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||||
|
"k8s.io/apimachinery/pkg/types"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||||
|
)
|
||||||
|
|
||||||
|
const offsiteUsage = `usage:
|
||||||
|
felis offsite sync [-config path] [-archive-dir dir] [-db-dir dir] [-status-file path]
|
||||||
|
felis offsite status [-config path] [-status-file path]
|
||||||
|
felis offsite list [-config path]
|
||||||
|
felis offsite fetch-db [-config path | -endpoint url -bucket name [-region r] [-prefix p]]
|
||||||
|
[-dir dir] latest|<bundle>
|
||||||
|
felis offsite fetch-worlds [-config path] [-archive-dir dir]
|
||||||
|
felis offsite keygen
|
||||||
|
|
||||||
|
Every verb but keygen reads the bucket credentials and the encryption key from
|
||||||
|
the variables [offsite] names (default FELIS_OFFSITE_ACCESS_KEY,
|
||||||
|
FELIS_OFFSITE_SECRET_KEY, FELIS_OFFSITE_KEY), taking any that are unset from
|
||||||
|
-env-file (default /etc/felis/offsite.env).
|
||||||
|
`
|
||||||
|
|
||||||
|
// defaultOffsiteEnvFile is where bootstrap keeps the [offsite] secrets; the
|
||||||
|
// felis-offsite.service unit loads it as its EnvironmentFile.
|
||||||
|
const defaultOffsiteEnvFile = "/etc/felis/offsite.env"
|
||||||
|
|
||||||
|
// cmdOffsite implements `felis offsite`: the off-site copy of the world
|
||||||
|
// archives and the database bundles (internal/offsite). felis-offsite.timer
|
||||||
|
// runs `sync` hourly on the host; the fetch verbs are the way back after the
|
||||||
|
// node is lost (docs/troubleshooting.md §16).
|
||||||
|
func cmdOffsite(args []string, stdout, stderr io.Writer) int {
|
||||||
|
if len(args) == 0 {
|
||||||
|
fmt.Fprint(stderr, offsiteUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
verb, rest := args[0], args[1:]
|
||||||
|
fs := flag.NewFlagSet("offsite "+verb, flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
fs.Usage = func() { fmt.Fprint(stderr, offsiteUsage) }
|
||||||
|
switch verb {
|
||||||
|
case "sync":
|
||||||
|
return offsiteSync(fs, rest, stdout, stderr)
|
||||||
|
case "status":
|
||||||
|
return offsiteStatus(fs, rest, stdout, stderr)
|
||||||
|
case "list":
|
||||||
|
return offsiteList(fs, rest, stdout, stderr)
|
||||||
|
case "fetch-db":
|
||||||
|
return offsiteFetchDB(fs, rest, stdout, stderr)
|
||||||
|
case "fetch-worlds":
|
||||||
|
return offsiteFetchWorlds(fs, rest, stdout, stderr)
|
||||||
|
case "keygen":
|
||||||
|
k, err := offsite.NewKey()
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite keygen: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintln(stdout, k)
|
||||||
|
return 0
|
||||||
|
case "-h", "--help", "help":
|
||||||
|
fmt.Fprint(stdout, offsiteUsage)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stderr, "felis offsite: unknown verb %q\n%s", verb, offsiteUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
|
||||||
|
// offsiteEnv is the resolved [offsite] binding: the bucket and the key.
|
||||||
|
type offsiteEnv struct {
|
||||||
|
cfg config.OffsiteConfig
|
||||||
|
bucket *offsite.S3
|
||||||
|
key []byte
|
||||||
|
}
|
||||||
|
|
||||||
|
// loadEnvFile sets each KEY=VALUE of path that is not already in the
|
||||||
|
// environment, so a root shell runs a command the same way its unit does.
|
||||||
|
// A missing file is not an error.
|
||||||
|
func loadEnvFile(path string) error {
|
||||||
|
if path == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
f, err := os.Open(path)
|
||||||
|
if errors.Is(err, os.ErrNotExist) {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
defer f.Close()
|
||||||
|
sc := bufio.NewScanner(f)
|
||||||
|
for sc.Scan() {
|
||||||
|
line := strings.TrimSpace(sc.Text())
|
||||||
|
if line == "" || strings.HasPrefix(line, "#") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
k, v, ok := strings.Cut(line, "=")
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
k = strings.TrimSpace(strings.TrimPrefix(k, "export "))
|
||||||
|
v = strings.TrimSpace(v)
|
||||||
|
if len(v) >= 2 && (v[0] == '"' || v[0] == '\'') && v[len(v)-1] == v[0] {
|
||||||
|
v = v[1 : len(v)-1]
|
||||||
|
}
|
||||||
|
if os.Getenv(k) == "" {
|
||||||
|
os.Setenv(k, v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return sc.Err()
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveOffsite builds the bucket client and parses the key for c.
|
||||||
|
func resolveOffsite(c config.OffsiteConfig) (*offsiteEnv, error) {
|
||||||
|
if !c.Enabled() {
|
||||||
|
return nil, errors.New("no [offsite] bucket is configured (docs/troubleshooting.md §16, \"Keep a copy somewhere else\")")
|
||||||
|
}
|
||||||
|
need := func(ref, what string) (string, error) {
|
||||||
|
v := os.Getenv(ref)
|
||||||
|
if v == "" {
|
||||||
|
return "", fmt.Errorf("%s: environment variable %s is empty (set it, or put it in %s)", what, ref, defaultOffsiteEnvFile)
|
||||||
|
}
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
|
ak, err := need(c.AccessKeyRef, "access key")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
sk, err := need(c.SecretKeyRef, "secret key")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
rawKey, err := need(c.KeyRef, "encryption key")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
key, err := offsite.ParseKey(rawKey)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
b, err := offsite.NewS3(offsite.S3Config{
|
||||||
|
Endpoint: c.Endpoint, Region: c.Region, Bucket: c.Bucket, Prefix: c.Prefix,
|
||||||
|
AccessKey: ak, SecretKey: sk,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return &offsiteEnv{cfg: c, bucket: b, key: key}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// loadOffsite loads felis.toml and the env file and resolves [offsite].
|
||||||
|
func loadOffsite(cfgPath, envFile string) (*config.Config, *offsiteEnv, error) {
|
||||||
|
if err := loadEnvFile(envFile); err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("read %s: %w", envFile, err)
|
||||||
|
}
|
||||||
|
cfg, err := config.Load(cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, err
|
||||||
|
}
|
||||||
|
env, err := resolveOffsite(cfg.Offsite)
|
||||||
|
if err != nil {
|
||||||
|
return cfg, nil, err
|
||||||
|
}
|
||||||
|
return cfg, env, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
|
||||||
|
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||||
|
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
|
||||||
|
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
|
||||||
|
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
|
||||||
|
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "where the result of this run is recorded for the watchdog and `status`")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
cfg, env, err := loadOffsite(*cfgPath, *envFile)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
st := offsite.Status{
|
||||||
|
LastAttempt: time.Now().UTC(), Endpoint: env.cfg.Endpoint, Bucket: env.cfg.Bucket,
|
||||||
|
Prefix: env.cfg.Prefix, KeyID: offsite.KeyID(env.key),
|
||||||
|
}
|
||||||
|
if prev, _ := offsite.ReadStatus(*statusFile); prev != nil {
|
||||||
|
st.LastSuccess = prev.LastSuccess
|
||||||
|
}
|
||||||
|
res, err := runOffsiteSync(cfg, env, *archiveDir, *backupPVC, *dbDir, stderr)
|
||||||
|
st.Result = res
|
||||||
|
if err != nil {
|
||||||
|
st.LastError = err.Error()
|
||||||
|
} else {
|
||||||
|
st.LastSuccess = st.LastAttempt
|
||||||
|
}
|
||||||
|
if werr := offsite.WriteStatus(*statusFile, st); werr != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite sync: record status: %v\n", werr)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; bucket holds %d worlds (%s) and %d bundles\n",
|
||||||
|
res.WorldsUploaded, res.WorldsPending, len(res.WorldsMissing), res.WorldsExpired,
|
||||||
|
res.DBUploaded, res.DBPruned, res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB)
|
||||||
|
for _, m := range res.WorldsMissing {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite sync: recorded archive not on the volume, nothing to copy: %s\n", m)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, archiveDir, backupPVC, dbDir string, log io.Writer) (offsite.Result, error) {
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 50*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
checkCtx, checkCancel := context.WithTimeout(ctx, 30*time.Second)
|
||||||
|
err := env.bucket.Check(checkCtx)
|
||||||
|
checkCancel()
|
||||||
|
if err != nil {
|
||||||
|
return offsite.Result{}, err
|
||||||
|
}
|
||||||
|
if archiveDir == "" && backupPVC != "" {
|
||||||
|
dir, err := resolveArchiveDir(ctx, cfg.K8s.Namespace, backupPVC, false, log)
|
||||||
|
if err != nil {
|
||||||
|
return offsite.Result{}, err
|
||||||
|
}
|
||||||
|
archiveDir = dir
|
||||||
|
}
|
||||||
|
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||||
|
if err != nil {
|
||||||
|
return offsite.Result{}, fmt.Errorf("open database: %w", err)
|
||||||
|
}
|
||||||
|
defer drv.Close()
|
||||||
|
s := &offsite.Syncer{
|
||||||
|
Bucket: env.bucket, Catalog: offsite.PGCatalog{DB: drv.DB()}, Key: env.key,
|
||||||
|
ArchiveDir: archiveDir, DBDir: dbDir, DBKeep: env.cfg.DBKeep, Log: log,
|
||||||
|
}
|
||||||
|
return s.Run(ctx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveArchiveDir finds the host directory behind the world archive PVC: a
|
||||||
|
// local-path volume is a directory on this node. A PVC still waiting for its
|
||||||
|
// first consumer holds nothing yet: without bind that is "" (no archives),
|
||||||
|
// with bind it is bound first, for fetch-worlds to write into.
|
||||||
|
func resolveArchiveDir(ctx context.Context, ns, pvcName string, bind bool, log io.Writer) (string, error) {
|
||||||
|
if ns == "" {
|
||||||
|
ns = platform.DefaultMinecraftNamespace
|
||||||
|
}
|
||||||
|
cl, err := buildSystemServerClient()
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("reach the cluster to find the archive volume (or pass -archive-dir): %w", err)
|
||||||
|
}
|
||||||
|
var pvc corev1.PersistentVolumeClaim
|
||||||
|
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err != nil {
|
||||||
|
return "", fmt.Errorf("archive volume %s/%s: %w", ns, pvcName, err)
|
||||||
|
}
|
||||||
|
if pvc.Spec.VolumeName == "" {
|
||||||
|
if !bind {
|
||||||
|
fmt.Fprintf(log, "felis offsite: archive volume %s/%s is not bound yet; no world has been archived\n", ns, pvcName)
|
||||||
|
return "", nil
|
||||||
|
}
|
||||||
|
if err := bindVolume(ctx, cl, ns, pvcName, log); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var pv corev1.PersistentVolume
|
||||||
|
if err := cl.Get(ctx, types.NamespacedName{Name: pvc.Spec.VolumeName}, &pv); err != nil {
|
||||||
|
return "", fmt.Errorf("archive volume %s: %w", pvc.Spec.VolumeName, err)
|
||||||
|
}
|
||||||
|
var dir string
|
||||||
|
switch {
|
||||||
|
case pv.Spec.Local != nil:
|
||||||
|
dir = pv.Spec.Local.Path
|
||||||
|
case pv.Spec.HostPath != nil:
|
||||||
|
dir = pv.Spec.HostPath.Path
|
||||||
|
default:
|
||||||
|
return "", fmt.Errorf("archive volume %s is not a directory on a node (local or hostPath); pass -archive-dir with where it is mounted on this host", pv.Name)
|
||||||
|
}
|
||||||
|
if fi, err := os.Stat(dir); err != nil || !fi.IsDir() {
|
||||||
|
return "", fmt.Errorf("archive volume %s is %s on its node, which is not a directory here; run this on the node that holds it, or pass -archive-dir", pv.Name, dir)
|
||||||
|
}
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// bindVolume runs a pod that mounts the PVC and exits, which is what makes a
|
||||||
|
// WaitForFirstConsumer volume (k3s local-path) get provisioned. The pod uses
|
||||||
|
// the control plane's own image, which every install already has.
|
||||||
|
func bindVolume(ctx context.Context, cl client.Client, ns, pvcName string, log io.Writer) error {
|
||||||
|
var api appsv1.Deployment
|
||||||
|
if err := cl.Get(ctx, types.NamespacedName{Namespace: platform.DefaultControlNamespace, Name: "felis-api"}, &api); err != nil {
|
||||||
|
return fmt.Errorf("find the felis image to bind the archive volume with: %w", err)
|
||||||
|
}
|
||||||
|
if len(api.Spec.Template.Spec.Containers) == 0 {
|
||||||
|
return errors.New("felis-api has no container to take the image from")
|
||||||
|
}
|
||||||
|
image := api.Spec.Template.Spec.Containers[0].Image
|
||||||
|
pod := platform.VolumeBinderPod(ns, pvcName, image)
|
||||||
|
if err := cl.Create(ctx, pod); err != nil {
|
||||||
|
return fmt.Errorf("start a pod to bind the archive volume: %w", err)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(log, "felis offsite: binding the archive volume %s/%s (pod %s)\n", ns, pvcName, pod.Name)
|
||||||
|
defer func() {
|
||||||
|
_ = cl.Delete(context.Background(), pod, client.PropagationPolicy(metav1.DeletePropagationBackground))
|
||||||
|
}()
|
||||||
|
deadline := time.Now().Add(3 * time.Minute)
|
||||||
|
for time.Now().Before(deadline) {
|
||||||
|
var pvc corev1.PersistentVolumeClaim
|
||||||
|
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err == nil && pvc.Spec.VolumeName != "" && pvc.Status.Phase == corev1.ClaimBound {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return ctx.Err()
|
||||||
|
case <-time.After(2 * time.Second):
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return fmt.Errorf("the archive volume %s/%s did not bind within 3 minutes; see kubectl -n %s describe pod %s", ns, pvcName, ns, pod.Name)
|
||||||
|
}
|
||||||
|
|
||||||
|
func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
|
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "the record `sync` writes")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
cfg, err := config.Load(*cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite status: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if !cfg.Offsite.Enabled() {
|
||||||
|
fmt.Fprintln(stdout, "off-site copy: not configured. World archives and database bundles exist on this machine only.")
|
||||||
|
fmt.Fprintln(stdout, "See docs/troubleshooting.md §16, \"Keep a copy somewhere else\".")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
o := cfg.Offsite
|
||||||
|
fmt.Fprintf(stdout, "bucket: %s at %s", o.Bucket, o.Endpoint)
|
||||||
|
if o.Prefix != "" {
|
||||||
|
fmt.Fprintf(stdout, ", prefix %s", o.Prefix)
|
||||||
|
}
|
||||||
|
fmt.Fprintln(stdout)
|
||||||
|
st, err := offsite.ReadStatus(*statusFile)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite status: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if st == nil {
|
||||||
|
fmt.Fprintln(stdout, "last sync: never (sudo systemctl start felis-offsite.service)")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
now := time.Now()
|
||||||
|
fmt.Fprintf(stdout, "key id: %s\n", st.KeyID)
|
||||||
|
fmt.Fprintf(stdout, "last attempt: %s (%s ago)\n", st.LastAttempt.Local().Format(time.DateTime), dbbackup.Age(now.Sub(st.LastAttempt)))
|
||||||
|
if st.LastSuccess.IsZero() {
|
||||||
|
fmt.Fprintln(stdout, "last success: never")
|
||||||
|
} else {
|
||||||
|
fmt.Fprintf(stdout, "last success: %s (%s ago)\n", st.LastSuccess.Local().Format(time.DateTime), dbbackup.Age(now.Sub(st.LastSuccess)))
|
||||||
|
}
|
||||||
|
if st.LastError != "" {
|
||||||
|
fmt.Fprintf(stdout, "last error: %s\n", st.LastError)
|
||||||
|
}
|
||||||
|
r := st.Result
|
||||||
|
fmt.Fprintf(stdout, "bucket holds: %d world archives (%s), %d database bundles, newest %s\n",
|
||||||
|
r.RemoteWorlds, offsite.HumanBytes(r.RemoteBytes), r.RemoteDB, orNone(r.NewestDB))
|
||||||
|
fmt.Fprintf(stdout, "waiting: %d world archives not yet copied\n", r.WorldsPending)
|
||||||
|
for _, m := range r.WorldsMissing {
|
||||||
|
fmt.Fprintf(stdout, "missing: %s is recorded but not on the volume\n", m)
|
||||||
|
}
|
||||||
|
if st.LastSuccess.IsZero() || now.Sub(st.LastSuccess) > offsite.StaleAfter {
|
||||||
|
fmt.Fprintf(stdout, "\nThe last successful sync is older than %s: journalctl -u felis-offsite -n 50\n", dbbackup.Age(offsite.StaleAfter))
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func orNone(s string) string {
|
||||||
|
if s == "" {
|
||||||
|
return "none"
|
||||||
|
}
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
func offsiteList(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
|
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
_, env, err := loadOffsite(*cfgPath, *envFile)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return printOffsiteList(env, stdout, stderr)
|
||||||
|
}
|
||||||
|
|
||||||
|
func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
bundles, err := offsite.ListDB(ctx, env.bucket)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
worlds, err := env.bucket.List(ctx, "worlds/")
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "database bundles (%d, newest first):\n", len(bundles))
|
||||||
|
for _, b := range bundles {
|
||||||
|
fmt.Fprintf(stdout, " %s %s\n", b.Key, offsite.HumanBytes(b.Size))
|
||||||
|
}
|
||||||
|
var total int64
|
||||||
|
for _, w := range worlds {
|
||||||
|
total += w.Size
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "world archives: %d (%s)\n", len(worlds), offsite.HumanBytes(total))
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func offsiteFetchDB(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml; on a host with no install yet, give -endpoint and -bucket instead")
|
||||||
|
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||||
|
endpoint := fs.String("endpoint", "", "bucket endpoint, when there is no felis.toml")
|
||||||
|
bucket := fs.String("bucket", "", "bucket name, when there is no felis.toml")
|
||||||
|
region := fs.String("region", "", "bucket region, when there is no felis.toml")
|
||||||
|
prefix := fs.String("prefix", "", "key prefix, when there is no felis.toml")
|
||||||
|
dir := fs.String("dir", dbbackup.DefaultDir, "directory to write the bundle to")
|
||||||
|
arg, ok := parseWithArg(fs, args)
|
||||||
|
if !ok {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if arg == "" {
|
||||||
|
fmt.Fprint(stderr, offsiteUsage)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if err := loadEnvFile(*envFile); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: read %s: %v\n", *envFile, err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
var oc config.OffsiteConfig
|
||||||
|
if *bucket != "" {
|
||||||
|
oc = config.OffsiteConfig{
|
||||||
|
Endpoint: *endpoint, Bucket: *bucket, Region: *region, Prefix: *prefix,
|
||||||
|
AccessKeyRef: config.DefaultOffsiteAccessKeyEnv, SecretKeyRef: config.DefaultOffsiteSecretKeyEnv,
|
||||||
|
KeyRef: config.DefaultOffsiteKeyEnv,
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
cfg, err := config.Load(*cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: %v (on a host with no install yet, pass -endpoint and -bucket)\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
oc = cfg.Offsite
|
||||||
|
}
|
||||||
|
env, err := resolveOffsite(oc)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
name := arg
|
||||||
|
if name == "latest" {
|
||||||
|
bundles, err := offsite.ListDB(ctx, env.bucket)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if len(bundles) == 0 {
|
||||||
|
fmt.Fprintln(stderr, "felis offsite fetch-db: the bucket holds no database bundle")
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
name = bundles[0].Key
|
||||||
|
}
|
||||||
|
if _, _, ok := dbbackup.ParseBundleName(name); !ok {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: %q is not a bundle name (felis-db-<stamp>-<label>.tar); see `felis offsite list`\n", name)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if err := os.MkdirAll(*dir, 0o700); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
dst := filepath.Join(*dir, name)
|
||||||
|
if err := offsite.FetchObject(ctx, env.bucket, env.key, offsite.DBKey(name), dst, 0o600); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if _, err := dbbackup.Verify(dst); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-db: fetched %s but it does not verify: %v\n", dst, err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis offsite fetch-db: wrote %s (verified)\n", dst)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
func offsiteFetchWorlds(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy)")
|
||||||
|
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||||
|
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC, binding it if needed)")
|
||||||
|
backupPVC := fs.String("backup-pvc", "felis-backups", "the world archive PVC, in the [k8s] namespace")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
cfg, env, err := loadOffsite(*cfgPath, *envFile)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 6*time.Hour)
|
||||||
|
defer cancel()
|
||||||
|
dir := *archiveDir
|
||||||
|
if dir == "" {
|
||||||
|
if dir, err = resolveArchiveDir(ctx, cfg.K8s.Namespace, *backupPVC, true, stderr); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-worlds: open database: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer drv.Close()
|
||||||
|
res, err := offsite.FetchWorlds(ctx, env.bucket, offsite.PGCatalog{DB: drv.DB()}, env.key, dir, stderr)
|
||||||
|
fmt.Fprintf(stdout, "felis offsite fetch-worlds: %d recorded archives, %d fetched into %s, %d with no copy in the bucket\n",
|
||||||
|
res.Present, len(res.Fetched), dir, len(res.Missing))
|
||||||
|
for _, m := range res.Missing {
|
||||||
|
fmt.Fprintf(stdout, " no off-site copy: %s\n", m)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/offsite"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestLoadOffsiteEnvFile(t *testing.T) {
|
||||||
|
path := filepath.Join(t.TempDir(), "offsite.env")
|
||||||
|
body := `# written by bootstrap
|
||||||
|
FELIS_OFFSITE_ACCESS_KEY=AKIA123
|
||||||
|
export FELIS_OFFSITE_SECRET_KEY="se=cret"
|
||||||
|
FELIS_OFFSITE_KEY='k'
|
||||||
|
|
||||||
|
not a line
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(path, []byte(body), 0o600); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
t.Setenv("FELIS_OFFSITE_ACCESS_KEY", "from-the-shell")
|
||||||
|
t.Setenv("FELIS_OFFSITE_SECRET_KEY", "")
|
||||||
|
t.Setenv("FELIS_OFFSITE_KEY", "")
|
||||||
|
if err := loadEnvFile(path); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
for k, want := range map[string]string{
|
||||||
|
"FELIS_OFFSITE_ACCESS_KEY": "from-the-shell", // the environment wins
|
||||||
|
"FELIS_OFFSITE_SECRET_KEY": "se=cret",
|
||||||
|
"FELIS_OFFSITE_KEY": "k",
|
||||||
|
} {
|
||||||
|
if got := os.Getenv(k); got != want {
|
||||||
|
t.Errorf("%s = %q, want %q", k, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := loadEnvFile(filepath.Join(t.TempDir(), "absent")); err != nil {
|
||||||
|
t.Errorf("a missing env file is not an error: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolveOffsiteNamesTheMissingVariable(t *testing.T) {
|
||||||
|
c := config.OffsiteConfig{
|
||||||
|
Endpoint: "https://s3.example", Bucket: "b",
|
||||||
|
AccessKeyRef: "T_AK", SecretKeyRef: "T_SK", KeyRef: "T_KEY",
|
||||||
|
}
|
||||||
|
t.Setenv("T_AK", "ak")
|
||||||
|
t.Setenv("T_SK", "sk")
|
||||||
|
t.Setenv("T_KEY", "")
|
||||||
|
if _, err := resolveOffsite(c); err == nil || !strings.Contains(err.Error(), "T_KEY") {
|
||||||
|
t.Fatalf("err = %v, want it to name T_KEY", err)
|
||||||
|
}
|
||||||
|
t.Setenv("T_KEY", "not base64 at all")
|
||||||
|
if _, err := resolveOffsite(c); err == nil {
|
||||||
|
t.Fatal("a malformed key was accepted")
|
||||||
|
}
|
||||||
|
key, _ := offsite.NewKey()
|
||||||
|
t.Setenv("T_KEY", key)
|
||||||
|
env, err := resolveOffsite(c)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if len(env.key) != offsite.KeySize {
|
||||||
|
t.Fatalf("key is %d bytes", len(env.key))
|
||||||
|
}
|
||||||
|
if _, err := resolveOffsite(config.OffsiteConfig{}); err == nil {
|
||||||
|
t.Fatal("an unconfigured [offsite] resolved")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOffsiteKeygen(t *testing.T) {
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
if code := cmdOffsite([]string{"keygen"}, &out, &errb); code != 0 {
|
||||||
|
t.Fatalf("exit %d: %s", code, errb.String())
|
||||||
|
}
|
||||||
|
if _, err := offsite.ParseKey(strings.TrimSpace(out.String())); err != nil {
|
||||||
|
t.Fatalf("keygen printed %q: %v", out.String(), err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOffsiteFetchDBRejectsOddNames(t *testing.T) {
|
||||||
|
key, _ := offsite.NewKey()
|
||||||
|
t.Setenv("FELIS_OFFSITE_ACCESS_KEY", "ak")
|
||||||
|
t.Setenv("FELIS_OFFSITE_SECRET_KEY", "sk")
|
||||||
|
t.Setenv("FELIS_OFFSITE_KEY", key)
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
code := cmdOffsite([]string{"fetch-db", "-env-file", "", "-endpoint", "http://127.0.0.1:1", "-bucket", "b",
|
||||||
|
"-dir", t.TempDir(), "../../etc/shadow"}, &out, &errb)
|
||||||
|
if code != 2 || !strings.Contains(errb.String(), "not a bundle name") {
|
||||||
|
t.Fatalf("exit %d: %s", code, errb.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
+34
-7
@@ -1,11 +1,15 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
"flag"
|
"flag"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
"log/slog"
|
"log/slog"
|
||||||
|
"net/http"
|
||||||
"os"
|
"os"
|
||||||
|
"time"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
felismetrics "felis.lolicon.best/internal/metrics"
|
felismetrics "felis.lolicon.best/internal/metrics"
|
||||||
@@ -75,16 +79,19 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace)
|
fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace)
|
||||||
|
|
||||||
// Register the two probe endpoints. controller-runtime only mounts /healthz and
|
// /healthz fails while a reconcile pass has been stuck past its limit, so the
|
||||||
// /readyz once at least one check is registered, so a bare listener would 404.
|
// liveness probe restarts an operator whose workers are wedged (a Pod whose
|
||||||
// The checks are the canonical always-pass ping: the probes' contract is "the
|
// process answers but no server starts or stops). /readyz waits for the
|
||||||
// manager process is up and serving", and a dependency hiccup (e.g. an API blip)
|
// informer caches: until they sync the operator acts on nothing, and one that
|
||||||
// must not restart the operator.
|
// never syncs (lost RBAC, an unreachable API) never reports Available.
|
||||||
if err := mgr.AddHealthzCheck("ping", healthz.Ping); err != nil {
|
// A dependency hiccup fails neither: the caches ride through API blips, and
|
||||||
|
// each pass is bounded well inside the stuck limit.
|
||||||
|
watch := &operator.ReconcileWatch{}
|
||||||
|
if err := mgr.AddHealthzCheck("reconcile", watch.Check); err != nil {
|
||||||
fmt.Fprintf(stderr, "felis operator: register healthz check: %v\n", err)
|
fmt.Fprintf(stderr, "felis operator: register healthz check: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
if err := mgr.AddReadyzCheck("ping", healthz.Ping); err != nil {
|
if err := mgr.AddReadyzCheck("informers", cacheSynced(mgr.GetCache())); err != nil {
|
||||||
fmt.Fprintf(stderr, "felis operator: register readyz check: %v\n", err)
|
fmt.Fprintf(stderr, "felis operator: register readyz check: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
@@ -98,6 +105,7 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
|||||||
fmt.Fprintf(stderr, "felis operator: register metrics: %v\n", err)
|
fmt.Fprintf(stderr, "felis operator: register metrics: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
felismetrics.SetBuildInfo("operator", resolvedVersion())
|
||||||
|
|
||||||
r := &operator.Reconciler{
|
r := &operator.Reconciler{
|
||||||
Client: mgr.GetClient(),
|
Client: mgr.GetClient(),
|
||||||
@@ -107,6 +115,10 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
|||||||
// injects into user servers. The Deployment passes it as FELIS_IMAGE (see
|
// injects into user servers. The Deployment passes it as FELIS_IMAGE (see
|
||||||
// platform.OperatorDeployment); absent, that injection is simply skipped.
|
// platform.OperatorDeployment); absent, that injection is simply skipped.
|
||||||
FelisImage: os.Getenv("FELIS_IMAGE"),
|
FelisImage: os.Getenv("FELIS_IMAGE"),
|
||||||
|
// Uncached: the maintenance-lock check lists Jobs only when a server is
|
||||||
|
// about to start, which does not justify a namespace-wide Job informer.
|
||||||
|
Jobs: mgr.GetAPIReader(),
|
||||||
|
Watch: watch,
|
||||||
}
|
}
|
||||||
if err := r.SetupWithManager(mgr); err != nil {
|
if err := r.SetupWithManager(mgr); err != nil {
|
||||||
fmt.Fprintf(stderr, "felis operator: setup controller: %v\n", err)
|
fmt.Fprintf(stderr, "felis operator: setup controller: %v\n", err)
|
||||||
@@ -128,3 +140,18 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
|||||||
}
|
}
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// cacheSynced is a readyz check that passes once every informer the manager
|
||||||
|
// started has synced. It waits at most a second, well inside the probe timeout.
|
||||||
|
func cacheSynced(c interface {
|
||||||
|
WaitForCacheSync(ctx context.Context) bool
|
||||||
|
}) healthz.Checker {
|
||||||
|
return func(req *http.Request) error {
|
||||||
|
ctx, cancel := context.WithTimeout(req.Context(), time.Second)
|
||||||
|
defer cancel()
|
||||||
|
if !c.WaitForCacheSync(ctx) {
|
||||||
|
return errors.New("informer caches not synced")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"net/http/httptest"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
type fakeCache bool
|
||||||
|
|
||||||
|
func (f fakeCache) WaitForCacheSync(ctx context.Context) bool {
|
||||||
|
if !f {
|
||||||
|
<-ctx.Done()
|
||||||
|
}
|
||||||
|
return bool(f)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCacheSynced: the operator reports ready only once its informers synced,
|
||||||
|
// and a check against caches that never sync returns within its own deadline.
|
||||||
|
func TestCacheSynced(t *testing.T) {
|
||||||
|
req := httptest.NewRequest("GET", "/readyz", nil)
|
||||||
|
if err := cacheSynced(fakeCache(true))(req); err != nil {
|
||||||
|
t.Errorf("synced: %v", err)
|
||||||
|
}
|
||||||
|
if err := cacheSynced(fakeCache(false))(req); err == nil {
|
||||||
|
t.Error("unsynced caches reported ready")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,126 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
"felis.lolicon.best/internal/imagepin"
|
||||||
|
"felis.lolicon.best/internal/platform"
|
||||||
|
"k8s.io/apimachinery/pkg/api/meta"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||||
|
)
|
||||||
|
|
||||||
|
// defaultRegistryURL is the [registry] url every install uses; deploy/bootstrap.sh
|
||||||
|
// spells the same value as REGISTRY_URL.
|
||||||
|
const defaultRegistryURL = "registry.felis.svc:5000"
|
||||||
|
|
||||||
|
// cmdPinImages pins every user server whose spec.image still names a mutable tag
|
||||||
|
// in the platform registry to the digest that tag names now (internal/imagepin).
|
||||||
|
// felis-api pins on create, so this covers the servers created before it did.
|
||||||
|
//
|
||||||
|
// deploy/bootstrap.sh runs it before it rebuilds the game images and pushes them
|
||||||
|
// over the same tags: run after the push, it would pin those servers to the new
|
||||||
|
// build, which is exactly the silent Minecraft upgrade pinning exists to stop.
|
||||||
|
// It reaches the registry through the node's loopback hostPort, the same way the
|
||||||
|
// installer pushes.
|
||||||
|
func cmdPinImages(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("pin-images", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
namespace := fs.String("namespace", platform.DefaultMinecraftNamespace, "namespace the MinecraftServers live in")
|
||||||
|
registry := fs.String("registry", defaultRegistryURL, "registry host[:port] the image refs spell")
|
||||||
|
endpoint := fs.String("endpoint", "", "host[:port] to reach the registry at (default: 127.0.0.1 on the registry's port, its hostPort on this node)")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
if errors.Is(err, flag.ErrHelp) {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *endpoint == "" {
|
||||||
|
*endpoint = loopbackEndpoint(*registry)
|
||||||
|
}
|
||||||
|
cl, err := buildSystemServerClient()
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis pin-images: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
outcomes, err := pinUserServerImages(ctx, cl, *namespace, imagepin.Resolver{Registry: *registry, Endpoint: *endpoint})
|
||||||
|
if meta.IsNoMatchError(err) {
|
||||||
|
fmt.Fprintln(stdout, "felis pin-images: no MinecraftServer CRD yet, so no server to pin")
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis pin-images: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
if len(outcomes) == 0 {
|
||||||
|
fmt.Fprintln(stdout, "felis pin-images: every user server already runs a pinned image")
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
fmt.Fprintln(stdout, "felis pin-images: pinning user servers to the build their tag names now:")
|
||||||
|
exit := 0
|
||||||
|
for _, o := range outcomes {
|
||||||
|
if o.err != nil {
|
||||||
|
fmt.Fprintf(stdout, " - %s: ERROR %v\n", o.name, o.err)
|
||||||
|
exit = 1
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, " - %s: %s\n", o.name, strings.Join(o.changes, ", "))
|
||||||
|
}
|
||||||
|
return exit
|
||||||
|
}
|
||||||
|
|
||||||
|
// loopbackEndpoint is the registry's port on 127.0.0.1: the registry Deployment
|
||||||
|
// binds it as a hostPort, and containerd's mirror and the installer's pushes use
|
||||||
|
// the same address.
|
||||||
|
func loopbackEndpoint(registry string) string {
|
||||||
|
if i := strings.LastIndex(registry, ":"); i >= 0 {
|
||||||
|
return "127.0.0.1" + registry[i:]
|
||||||
|
}
|
||||||
|
return "127.0.0.1"
|
||||||
|
}
|
||||||
|
|
||||||
|
// pinUserServerImages patches spec.image of every user server whose image the
|
||||||
|
// resolver covers and is not yet pinned. System servers are left on their tags:
|
||||||
|
// the installer rebuilds and restarts them on purpose (restart_existing_system_servers).
|
||||||
|
// A server that is already pinned, or runs an image from elsewhere, produces no
|
||||||
|
// outcome, so a pinned fleet reports nothing. A running server restarts once as
|
||||||
|
// the operator rolls its StatefulSet onto the pinned ref, which is the build it
|
||||||
|
// already runs.
|
||||||
|
func pinUserServerImages(ctx context.Context, cl client.Client, namespace string, r imagepin.Resolver) ([]systemServerOutcome, error) {
|
||||||
|
var list v1alpha1.MinecraftServerList
|
||||||
|
if err := cl.List(ctx, &list, client.InNamespace(namespace)); err != nil {
|
||||||
|
return nil, fmt.Errorf("list servers: %w", err)
|
||||||
|
}
|
||||||
|
var out []systemServerOutcome
|
||||||
|
for i := range list.Items {
|
||||||
|
ms := &list.Items[i]
|
||||||
|
if ms.Labels[v1alpha1.LabelSystemRole] != "" || imagepin.Pinned(ms.Spec.Image) || !r.Covers(ms.Spec.Image) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
pinned, err := r.Pin(ctx, ms.Spec.Image)
|
||||||
|
if errors.Is(err, imagepin.ErrNotFound) {
|
||||||
|
err = fmt.Errorf("%s is not in the registry, so there is no build to pin it to; left unpinned: %w", ms.Spec.Image, err)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
out = append(out, systemServerOutcome{name: ms.Name, err: err})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
patch := client.MergeFrom(ms.DeepCopy())
|
||||||
|
ms.Spec.Image = pinned
|
||||||
|
if err := cl.Patch(ctx, ms, patch); err != nil {
|
||||||
|
out = append(out, systemServerOutcome{name: ms.Name, err: fmt.Errorf("patch %s: %w", ms.Name, err)})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
out = append(out, systemServerOutcome{name: ms.Name, available: true, updated: true,
|
||||||
|
changes: []string{"spec.image pinned to " + pinned}})
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,112 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
"felis.lolicon.best/internal/imagepin"
|
||||||
|
"felis.lolicon.best/internal/naming"
|
||||||
|
"k8s.io/apimachinery/pkg/api/meta"
|
||||||
|
"k8s.io/apimachinery/pkg/runtime/schema"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
|
||||||
|
)
|
||||||
|
|
||||||
|
const pinTestDigest = "sha256:2222222222222222222222222222222222222222222222222222222222222222"
|
||||||
|
|
||||||
|
// TestPinUserServerImages pins exactly the user servers still on a platform tag,
|
||||||
|
// reports a tag the registry lost as an error without touching that server, and
|
||||||
|
// has nothing left to do on a second pass.
|
||||||
|
func TestPinUserServerImages(t *testing.T) {
|
||||||
|
reg := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
if r.URL.Path != "/v2/felis/paper/manifests/demo" {
|
||||||
|
http.NotFound(w, r)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.Header().Set("Docker-Content-Digest", pinTestDigest)
|
||||||
|
}))
|
||||||
|
defer reg.Close()
|
||||||
|
res := imagepin.Resolver{Registry: defaultRegistryURL, Endpoint: strings.TrimPrefix(reg.URL, "http://")}
|
||||||
|
|
||||||
|
paper := defaultRegistryURL + "/felis/paper:demo"
|
||||||
|
mk := func(name, image, role string) *v1alpha1.MinecraftServer {
|
||||||
|
ms := &v1alpha1.MinecraftServer{}
|
||||||
|
ms.Name, ms.Namespace = name, "minecraft"
|
||||||
|
ms.Spec.Image = image
|
||||||
|
if role != "" {
|
||||||
|
ms.Labels = map[string]string{v1alpha1.LabelSystemRole: role}
|
||||||
|
}
|
||||||
|
return ms
|
||||||
|
}
|
||||||
|
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(
|
||||||
|
mk("legacy", paper, ""),
|
||||||
|
mk("pinned", paper+"@sha256:"+strings.Repeat("3", 64), ""),
|
||||||
|
mk("external", "docker.io/itzg/minecraft-server:java21", ""),
|
||||||
|
mk("gone", defaultRegistryURL+"/felis/paper:old", ""),
|
||||||
|
mk(naming.SystemLobbyServer, defaultRegistryURL+"/felis/felis-lobby:demo", naming.SystemLobbyServer),
|
||||||
|
).Build()
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
outcomes, err := pinUserServerImages(ctx, cl, "minecraft", res)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pinUserServerImages: %v", err)
|
||||||
|
}
|
||||||
|
byName := map[string]systemServerOutcome{}
|
||||||
|
for _, o := range outcomes {
|
||||||
|
byName[o.name] = o
|
||||||
|
}
|
||||||
|
if len(outcomes) != 2 || byName["legacy"].err != nil || byName["gone"].err == nil {
|
||||||
|
t.Fatalf("outcomes = %+v, want legacy pinned and gone reported", outcomes)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := map[string]string{
|
||||||
|
"legacy": paper + "@" + pinTestDigest,
|
||||||
|
"pinned": paper + "@sha256:" + strings.Repeat("3", 64),
|
||||||
|
"external": "docker.io/itzg/minecraft-server:java21",
|
||||||
|
"gone": defaultRegistryURL + "/felis/paper:old",
|
||||||
|
naming.SystemLobbyServer: defaultRegistryURL + "/felis/felis-lobby:demo",
|
||||||
|
}
|
||||||
|
for name, image := range want {
|
||||||
|
var ms v1alpha1.MinecraftServer
|
||||||
|
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
|
||||||
|
t.Fatalf("get %s: %v", name, err)
|
||||||
|
}
|
||||||
|
if ms.Spec.Image != image {
|
||||||
|
t.Errorf("%s image = %q, want %q", name, ms.Spec.Image, image)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
again, err := pinUserServerImages(ctx, cl, "minecraft", res)
|
||||||
|
if err != nil || len(again) != 1 || again[0].name != "gone" {
|
||||||
|
t.Fatalf("second pass = %+v, %v; want only the unresolvable server again", again, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A fresh install has no CRD yet; the command must read that as nothing to pin.
|
||||||
|
func TestPinUserServerImagesNoCRD(t *testing.T) {
|
||||||
|
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithInterceptorFuncs(interceptor.Funcs{
|
||||||
|
List: func(context.Context, client.WithWatch, client.ObjectList, ...client.ListOption) error {
|
||||||
|
return &meta.NoKindMatchError{GroupKind: schema.GroupKind{Group: "felis.lolicon.best", Kind: "MinecraftServer"}}
|
||||||
|
},
|
||||||
|
}).Build()
|
||||||
|
_, err := pinUserServerImages(context.Background(), cl, "minecraft", imagepin.Resolver{Registry: defaultRegistryURL})
|
||||||
|
if !meta.IsNoMatchError(err) {
|
||||||
|
t.Fatalf("err = %v, want a NoMatch error the command can recognise", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoopbackEndpoint(t *testing.T) {
|
||||||
|
for in, want := range map[string]string{
|
||||||
|
"registry.felis.svc:5000": "127.0.0.1:5000",
|
||||||
|
"registry.example": "127.0.0.1",
|
||||||
|
} {
|
||||||
|
if got := loopbackEndpoint(in); got != want {
|
||||||
|
t.Errorf("loopbackEndpoint(%q) = %q, want %q", in, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+40
-4
@@ -131,10 +131,24 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
|||||||
fmt.Fprintf(stderr, "felis reaper: %v\n", err)
|
fmt.Fprintf(stderr, "felis reaper: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d warned=%d skipped=%d evicted=%d expired=%d\n",
|
return reportReaperRun(sum, stdout, stderr)
|
||||||
sum.Evaluated, sum.WorldsReaped, sum.Warned, sum.Skipped, sum.EvictedEarly, sum.BackupsExpired)
|
}
|
||||||
|
|
||||||
|
// reportReaperRun prints the run's tally and turns a run that left work undone
|
||||||
|
// into exit 1, so the Job fails and the watchdog's job-failed check (and the
|
||||||
|
// FelisWorldJobFailed rule) reach the operator: a world that cannot be archived
|
||||||
|
// is kept, and without this nobody would learn that it is never reaped.
|
||||||
|
func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int {
|
||||||
|
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d\n",
|
||||||
|
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||||
|
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed)
|
||||||
|
if !sum.Failed() {
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
fmt.Fprintf(stderr, "felis reaper: %d servers failed (%d kept because the backup store is full) and %d expired backups were not removed; the errors are above, and each is retried next run\n",
|
||||||
|
sum.Skipped, sum.StoreFull, sum.ExpireFailed)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
// mailWarner delivers a pre-reap notice to the owner's verified email — the
|
// mailWarner delivers a pre-reap notice to the owner's verified email — the
|
||||||
// only channel this build can reach. Unowned owners and owners who never proved
|
// only channel this build can reach. Unowned owners and owners who never proved
|
||||||
@@ -169,8 +183,9 @@ func (w *mailWarner) Warn(ctx context.Context, ownerID, server, remaining string
|
|||||||
}
|
}
|
||||||
|
|
||||||
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
|
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
|
||||||
// idle deadline is fixed by §18; only the warning offsets, retention, and the
|
// idle deadline is fixed by §18; only the warning offsets, retention, the
|
||||||
// store soft-cap are configurable (§24).
|
// store soft-cap and the on-demand backup bounds are configurable (§24). The
|
||||||
|
// backup Job and felis-api read the manual_* bounds through it too.
|
||||||
func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
||||||
rc := reaper.DefaultConfig()
|
rc := reaper.DefaultConfig()
|
||||||
if v := cfg.Archive.Retention; v != "" {
|
if v := cfg.Archive.Retention; v != "" {
|
||||||
@@ -198,6 +213,27 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
|||||||
}
|
}
|
||||||
rc.MaxLocalBytes = b
|
rc.MaxLocalBytes = b
|
||||||
}
|
}
|
||||||
|
if v := cfg.Archive.ManualRetention; v != "" {
|
||||||
|
d, err := parseSpanDuration(v)
|
||||||
|
if err != nil || d <= 0 {
|
||||||
|
return rc, fmt.Errorf("[archive] manual_retention %q: want a positive span such as 30d", v)
|
||||||
|
}
|
||||||
|
rc.ManualRetention = d
|
||||||
|
}
|
||||||
|
switch n := cfg.Archive.ManualKeep; {
|
||||||
|
case n < 0:
|
||||||
|
return rc, fmt.Errorf("[archive] manual_keep %d: want 1 or more", n)
|
||||||
|
case n > 0:
|
||||||
|
rc.ManualKeep = n
|
||||||
|
}
|
||||||
|
if v := cfg.Archive.ManualCooldown; v != "" {
|
||||||
|
d, err := parseSpanDuration(v)
|
||||||
|
if err != nil || d < 0 {
|
||||||
|
return rc, fmt.Errorf("[archive] manual_cooldown %q: want a span such as 10m (0s for none)", v)
|
||||||
|
}
|
||||||
|
rc.ManualCooldown = d
|
||||||
|
}
|
||||||
|
rc.RequireOffsite = cfg.Offsite.Enabled()
|
||||||
return rc, nil
|
return rc, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,22 +1,85 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
"context"
|
"context"
|
||||||
"errors"
|
"errors"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
corev1 "k8s.io/api/core/v1"
|
corev1 "k8s.io/api/core/v1"
|
||||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/reaper"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// TestReportReaperRunFailsTheJob: a run that could not process a server, or
|
||||||
|
// could not remove an expired backup, exits 1 so the Job shows as failed.
|
||||||
|
func TestReportReaperRunFailsTheJob(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
sum reaper.Summary
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0},
|
||||||
|
{"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1},
|
||||||
|
{"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1},
|
||||||
|
{"expiry failed", reaper.Summary{Evaluated: 3, ExpireFailed: 2}, 1},
|
||||||
|
} {
|
||||||
|
var out, errb bytes.Buffer
|
||||||
|
if got := reportReaperRun(tc.sum, &out, &errb); got != tc.want {
|
||||||
|
t.Errorf("%s: exit %d, want %d", tc.name, got, tc.want)
|
||||||
|
}
|
||||||
|
if !strings.Contains(out.String(), "skipped=") || !strings.Contains(out.String(), "expire_failed=") {
|
||||||
|
t.Errorf("%s: summary line = %q", tc.name, out.String())
|
||||||
|
}
|
||||||
|
if (tc.want == 1) != (errb.Len() > 0) {
|
||||||
|
t.Errorf("%s: stderr = %q", tc.name, errb.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestResolveWorldDir pins the two world layouts the reaper must find, and the
|
// TestResolveWorldDir pins the two world layouts the reaper must find, and the
|
||||||
// fail-closed miss. The stock local-path arm is derived from the live PVC's
|
// fail-closed miss. The stock local-path arm is derived from the live PVC's
|
||||||
// volumeName — a name-based guess (glob) could tar a stale deleted PV's bytes and
|
// volumeName — a name-based guess (glob) could tar a stale deleted PV's bytes and
|
||||||
// then delete the current world, which is why it is read from the API instead.
|
// then delete the current world, which is why it is read from the API instead.
|
||||||
|
// TestReaperConfigManualKeys: the on-demand backup keys default to 30 days,
|
||||||
|
// five per server and a ten-minute cooldown, accept overrides, and refuse
|
||||||
|
// values that would keep nothing or throttle backwards.
|
||||||
|
func TestReaperConfigManualKeys(t *testing.T) {
|
||||||
|
rc, err := reaperConfig(&config.Config{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if rc.ManualRetention != 30*reaper.Day || rc.ManualKeep != 5 || rc.ManualCooldown != 10*time.Minute {
|
||||||
|
t.Fatalf("defaults = %v / %d / %v", rc.ManualRetention, rc.ManualKeep, rc.ManualCooldown)
|
||||||
|
}
|
||||||
|
rc, err = reaperConfig(&config.Config{Archive: config.ArchiveConfig{
|
||||||
|
ManualRetention: "7d", ManualKeep: 2, ManualCooldown: "0s"}})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if rc.ManualRetention != 7*reaper.Day || rc.ManualKeep != 2 || rc.ManualCooldown != 0 {
|
||||||
|
t.Fatalf("overrides = %v / %d / %v", rc.ManualRetention, rc.ManualKeep, rc.ManualCooldown)
|
||||||
|
}
|
||||||
|
for _, bad := range []config.ArchiveConfig{
|
||||||
|
{ManualRetention: "0d"},
|
||||||
|
{ManualRetention: "soon"},
|
||||||
|
{ManualKeep: -1},
|
||||||
|
{ManualCooldown: "-5m"},
|
||||||
|
{ManualCooldown: "often"},
|
||||||
|
} {
|
||||||
|
if _, err := reaperConfig(&config.Config{Archive: bad}); err == nil {
|
||||||
|
t.Errorf("%+v was accepted", bad)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestResolveWorldDir(t *testing.T) {
|
func TestResolveWorldDir(t *testing.T) {
|
||||||
ctx := context.Background()
|
ctx := context.Background()
|
||||||
root := t.TempDir()
|
root := t.TempDir()
|
||||||
|
|||||||
@@ -0,0 +1,164 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"log/slog"
|
||||||
|
"net"
|
||||||
|
"net/http"
|
||||||
|
"net/url"
|
||||||
|
"os"
|
||||||
|
"os/signal"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"syscall"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/imagepush"
|
||||||
|
"felis.lolicon.best/internal/registrygate"
|
||||||
|
)
|
||||||
|
|
||||||
|
// cmdRegistryGate is the sidecar entrypoint in the registry pod: it owns the
|
||||||
|
// registry port (and the loopback hostPort containerd pulls through), lets reads
|
||||||
|
// through anonymously, and forwards writes to the loopback-only registry:2 only
|
||||||
|
// for an authenticated principal allowed to write that repository. See
|
||||||
|
// internal/registrygate for the policy.
|
||||||
|
//
|
||||||
|
// Tokens are files under --auth-dir, one per principal (platform, build, prune),
|
||||||
|
// mounted from the registry-auth Secret. A missing file disables that principal:
|
||||||
|
// writes fail closed while every pull keeps working, which is the right way round
|
||||||
|
// for a registry the running workloads depend on.
|
||||||
|
//
|
||||||
|
// --maint-listen is the GC sidecar's read-only handshake (registrygate.MaintHandler).
|
||||||
|
// It has no authentication, so it must name a loopback address; --maint-dir keeps
|
||||||
|
// an open window across a gate restart.
|
||||||
|
func cmdRegistryGate(args []string, _, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("registry-gate", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
listen := fs.String("listen", ":5000", "address the gate serves the registry API on")
|
||||||
|
upstream := fs.String("upstream", "http://127.0.0.1:5001", "the loopback registry the gate forwards to")
|
||||||
|
authDir := fs.String("auth-dir", "/etc/felis-registry-auth", "directory holding one token file per principal")
|
||||||
|
maintListen := fs.String("maint-listen", "", "loopback address for the GC sidecar's read-only handshake (empty disables it)")
|
||||||
|
maintDir := fs.String("maint-dir", "", "directory that keeps an open read-only window across a gate restart")
|
||||||
|
quiet := fs.Duration("maint-quiet", registrygate.DefaultQuiet, "how long writes must be idle before a read-only window is granted")
|
||||||
|
dataDir := fs.String("data-dir", "", "the registry's storage root, mounted read-only, for the manifest index (empty disables it)")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *maintListen != "" && !loopbackAddr(*maintListen) {
|
||||||
|
fmt.Fprintf(stderr, "felis registry-gate: --maint-listen %q must be a loopback address: the handshake has no authentication\n", *maintListen)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
target, err := url.Parse(*upstream)
|
||||||
|
if err != nil || target.Scheme == "" || target.Host == "" {
|
||||||
|
fmt.Fprintf(stderr, "felis registry-gate: bad --upstream %q\n", *upstream)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
log := slog.New(slog.NewTextHandler(stderr, nil))
|
||||||
|
tokens := map[string]string{}
|
||||||
|
for _, p := range registrygate.Principals {
|
||||||
|
b, err := os.ReadFile(filepath.Join(*authDir, p))
|
||||||
|
tok := strings.TrimSpace(string(b))
|
||||||
|
if err != nil || tok == "" {
|
||||||
|
log.Warn("registry principal disabled: no token", "principal", p, "dir", *authDir)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
tokens[p] = tok
|
||||||
|
}
|
||||||
|
|
||||||
|
gate := registrygate.New(target, tokens, log)
|
||||||
|
gate.SetQuiet(*quiet)
|
||||||
|
gate.DataDir = *dataDir
|
||||||
|
if *maintDir != "" {
|
||||||
|
if err := gate.SetMaintenanceState(registrygate.MaintStatePath(*maintDir)); err != nil {
|
||||||
|
// A corrupt file must not keep the registry from serving pulls.
|
||||||
|
log.Warn("ignoring the saved read-only window", "err", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
srv := &http.Server{
|
||||||
|
Addr: *listen,
|
||||||
|
Handler: gate,
|
||||||
|
ReadHeaderTimeout: 10 * time.Second,
|
||||||
|
}
|
||||||
|
var maint *http.Server
|
||||||
|
if *maintListen != "" {
|
||||||
|
maint = &http.Server{Addr: *maintListen, Handler: gate.MaintHandler(), ReadHeaderTimeout: 10 * time.Second}
|
||||||
|
go func() {
|
||||||
|
if err := maint.ListenAndServe(); err != nil && !errors.Is(err, http.ErrServerClosed) {
|
||||||
|
log.Error("maintenance listener stopped; garbage collection cannot get a read-only window", "err", err)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
}
|
||||||
|
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||||
|
defer stop()
|
||||||
|
go func() {
|
||||||
|
<-ctx.Done()
|
||||||
|
shutdown, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||||
|
defer cancel()
|
||||||
|
_ = srv.Shutdown(shutdown)
|
||||||
|
if maint != nil {
|
||||||
|
_ = maint.Shutdown(shutdown)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
log.Info("registry gate listening", "addr", *listen, "upstream", target.String(), "principals", len(tokens))
|
||||||
|
if err := srv.ListenAndServe(); err != nil && !errors.Is(err, http.ErrServerClosed) {
|
||||||
|
fmt.Fprintf(stderr, "felis registry-gate: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
// loopbackAddr reports whether a host:port listen address binds loopback only.
|
||||||
|
func loopbackAddr(addr string) bool {
|
||||||
|
host, _, err := net.SplitHostPort(addr)
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if host == "localhost" {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
ip := net.ParseIP(host)
|
||||||
|
return ip != nil && ip.IsLoopback()
|
||||||
|
}
|
||||||
|
|
||||||
|
// cmdPushImage is the build Job's publish step. It runs after Kaniko built the
|
||||||
|
// image into a tarball (--no-push) and Trivy passed that tarball, and it is the
|
||||||
|
// only container of the build pod that holds the registry credential — the one
|
||||||
|
// executing the untrusted Dockerfile never sees it.
|
||||||
|
func cmdPushImage(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("push-image", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
tarPath := fs.String("tar", "", "image tarball Kaniko wrote with --tar-path")
|
||||||
|
ref := fs.String("ref", "", "host/repository:tag to publish it as")
|
||||||
|
scheme := fs.String("scheme", "http", "registry scheme: http for the in-cluster registry, https otherwise")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *tarPath == "" || *ref == "" {
|
||||||
|
fmt.Fprintln(stderr, "felis push-image: --tar and --ref are required")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
if *scheme != "http" && *scheme != "https" {
|
||||||
|
fmt.Fprintf(stderr, "felis push-image: bad --scheme %q\n", *scheme)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
user := os.Getenv("FELIS_REGISTRY_USERNAME")
|
||||||
|
pass := os.Getenv("FELIS_REGISTRY_PASSWORD")
|
||||||
|
if user == "" || pass == "" {
|
||||||
|
fmt.Fprintln(stderr, "felis push-image: FELIS_REGISTRY_USERNAME/FELIS_REGISTRY_PASSWORD are empty — the registry refuses anonymous writes")
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||||
|
defer stop()
|
||||||
|
p := &imagepush.Pusher{Scheme: *scheme, Username: user, Password: pass, Log: stderr}
|
||||||
|
digest, err := p.Push(ctx, *tarPath, *ref)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis push-image: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintln(stdout, digest)
|
||||||
|
return 0
|
||||||
|
}
|
||||||
+17
-1
@@ -11,7 +11,9 @@ Usage:
|
|||||||
felis <command> [flags]
|
felis <command> [flags]
|
||||||
|
|
||||||
Commands:
|
Commands:
|
||||||
migrate up Apply embedded database migrations under an advisory lock
|
migrate up Apply embedded database migrations under an advisory lock (snapshots the database first)
|
||||||
|
db Back up, verify, list and restore the control-plane database (backup|restore|verify|list|check)
|
||||||
|
offsite Copy world archives and database bundles to an off-site bucket, and fetch them back (sync|status|list|fetch-db|fetch-worlds|keygen)
|
||||||
operator Run the MinecraftServer controller-manager
|
operator Run the MinecraftServer controller-manager
|
||||||
api Run the felis-api HTTP server
|
api Run the felis-api HTTP server
|
||||||
nano Run the Felis-nano hasJoined multiplexer (multi-Yggdrasil, no control plane)
|
nano Run the Felis-nano hasJoined multiplexer (multi-Yggdrasil, no control plane)
|
||||||
@@ -19,11 +21,16 @@ Commands:
|
|||||||
restore Extract a world archive into a world volume (internal Job entrypoint)
|
restore Extract a world archive into a world volume (internal Job entrypoint)
|
||||||
backup Archive a world into the backup store and record it (internal Job entrypoint)
|
backup Archive a world into the backup store and record it (internal Job entrypoint)
|
||||||
files List/read/write one file in a stopped server's world (internal Job entrypoint)
|
files List/read/write one file in a stopped server's world (internal Job entrypoint)
|
||||||
|
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
|
||||||
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
|
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
|
||||||
|
push-image Push a scanned image tarball to the registry (internal Job entrypoint)
|
||||||
|
mirror-build-tools Copy kaniko, trivy and Trivy's DBs into the registry (run by felis-build-tools.timer)
|
||||||
|
registry-gate Authorize registry writes in front of registry:2 (internal sidecar entrypoint)
|
||||||
manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML
|
manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML
|
||||||
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
||||||
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
||||||
converge Fill in fields a newer desired spec added to already-installed system servers
|
converge Fill in fields a newer desired spec added to already-installed system servers
|
||||||
|
watchdog Check the platform once and mail the owners what has gone wrong (run by felis-watchdog.timer)
|
||||||
version Print the build stamp of this binary
|
version Print the build stamp of this binary
|
||||||
update Report which platform components have updates available
|
update Report which platform components have updates available
|
||||||
breakGlass Open the local break-glass emergency console (TUI; requires root/sudo)
|
breakGlass Open the local break-glass emergency console (TUI; requires root/sudo)
|
||||||
@@ -42,6 +49,8 @@ Run "felis <command> -h" for command-specific flags.
|
|||||||
// subcommand, and listing them would make the table disagree with the command list.
|
// subcommand, and listing them would make the table disagree with the command list.
|
||||||
var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||||
"migrate": cmdMigrate,
|
"migrate": cmdMigrate,
|
||||||
|
"db": cmdDB,
|
||||||
|
"offsite": cmdOffsite,
|
||||||
"operator": cmdOperator,
|
"operator": cmdOperator,
|
||||||
"api": cmdAPI,
|
"api": cmdAPI,
|
||||||
"nano": cmdNano,
|
"nano": cmdNano,
|
||||||
@@ -49,7 +58,11 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
|||||||
"restore": cmdRestore,
|
"restore": cmdRestore,
|
||||||
"backup": cmdBackup,
|
"backup": cmdBackup,
|
||||||
"files": cmdFiles,
|
"files": cmdFiles,
|
||||||
|
"egress-gate": cmdEgressGate,
|
||||||
"fetch-context": cmdFetchContext,
|
"fetch-context": cmdFetchContext,
|
||||||
|
"push-image": cmdPushImage,
|
||||||
|
"mirror-build-tools": cmdMirrorBuildTools,
|
||||||
|
"registry-gate": cmdRegistryGate,
|
||||||
"manifests": cmdManifests,
|
"manifests": cmdManifests,
|
||||||
"apply": cmdApply,
|
"apply": cmdApply,
|
||||||
"setup": cmdSetup,
|
"setup": cmdSetup,
|
||||||
@@ -57,8 +70,11 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
|||||||
"breakGlass": cmdBreakGlass,
|
"breakGlass": cmdBreakGlass,
|
||||||
"bootstrap-assets": cmdBootstrapAssets,
|
"bootstrap-assets": cmdBootstrapAssets,
|
||||||
"init-forwarding": cmdInitForwarding,
|
"init-forwarding": cmdInitForwarding,
|
||||||
|
"init-volume": cmdInitVolume,
|
||||||
|
"pin-images": cmdPinImages,
|
||||||
"version": cmdVersion,
|
"version": cmdVersion,
|
||||||
"update": cmdUpdate,
|
"update": cmdUpdate,
|
||||||
|
"watchdog": cmdWatchdog,
|
||||||
}
|
}
|
||||||
|
|
||||||
// run dispatches a subcommand. It is separate from main so the router is
|
// run dispatches a subcommand. It is separate from main so the router is
|
||||||
|
|||||||
@@ -38,9 +38,13 @@ func TestRunUnknownCommand(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// undocumentedCommands are routable on purpose but kept out of the usage text: they
|
// undocumentedCommands are routable on purpose but kept out of the usage text: they
|
||||||
// are called by deploy/bootstrap.sh, not by a human at a prompt. Listing them here is
|
// are called by deploy/bootstrap.sh or the operator's initContainers, not by a human
|
||||||
// what makes their absence from usage a deliberate decision rather than an oversight.
|
// at a prompt. Listing them here is what makes their absence from usage a deliberate
|
||||||
var undocumentedCommands = map[string]bool{"bootstrap-assets": true, "init-forwarding": true}
|
// decision rather than an oversight.
|
||||||
|
var undocumentedCommands = map[string]bool{
|
||||||
|
"bootstrap-assets": true, "init-forwarding": true, "init-volume": true,
|
||||||
|
"pin-images": true,
|
||||||
|
}
|
||||||
|
|
||||||
// The usage text and the dispatch table must describe the same set of commands.
|
// The usage text and the dispatch table must describe the same set of commands.
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -437,6 +437,10 @@ func (m *edgeModel) errorView() string {
|
|||||||
if m.lastErr != nil {
|
if m.lastErr != nil {
|
||||||
b.WriteString(tuiHint.Render(m.lastErr.Error()) + "\n")
|
b.WriteString(tuiHint.Render(m.lastErr.Error()) + "\n")
|
||||||
}
|
}
|
||||||
|
// Nothing done above is rolled back, and nothing needs to be: every step finds what an
|
||||||
|
// earlier attempt created (the tunnel, its DNS route, the Access app and policy) and
|
||||||
|
// carries on from it.
|
||||||
|
b.WriteString("\n" + tuiHint.Render("Retrying is safe: it reuses the tunnel, DNS record and Access app created so far instead of making duplicates.") + "\n")
|
||||||
b.WriteString("\n" + tuiAction("enter", "retry", "esc", "edit"))
|
b.WriteString("\n" + tuiAction("enter", "retry", "esc", "edit"))
|
||||||
return b.String()
|
return b.String()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -32,7 +32,9 @@ func applyCloudflareEdge(ctx context.Context, result *cfsetup.Result, panelHost,
|
|||||||
if adminHost == "" {
|
if adminHost == "" {
|
||||||
return fmt.Errorf("admin hostname is required")
|
return fmt.Errorf("admin hostname is required")
|
||||||
}
|
}
|
||||||
if err := writeConnectionConfig(panelHost, adminHost, result.AccessAud); err != nil {
|
// cloudflared is the only way in once the NodePort is fenced, so the
|
||||||
|
// visitor address it writes can key the sign-in rate limit.
|
||||||
|
if err := writeConnectionConfig(panelHost, adminHost, result.AccessAud, "CF-Connecting-IP"); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if err := applyFelisConfigSecret(ctx); err != nil {
|
if err := applyFelisConfigSecret(ctx); err != nil {
|
||||||
@@ -78,7 +80,8 @@ func applyReverseProxy(ctx context.Context, panelHost, adminHost string) error {
|
|||||||
if adminHost == "" {
|
if adminHost == "" {
|
||||||
return fmt.Errorf("admin hostname is required")
|
return fmt.Errorf("admin hostname is required")
|
||||||
}
|
}
|
||||||
if err := writeConnectionConfig(panelHost, adminHost, ""); err != nil {
|
// Caddy, nginx and Traefik all append the peer they saw to X-Forwarded-For.
|
||||||
|
if err := writeConnectionConfig(panelHost, adminHost, "", "X-Forwarded-For"); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if err := applyFelisConfigSecret(ctx); err != nil {
|
if err := applyFelisConfigSecret(ctx); err != nil {
|
||||||
@@ -93,16 +96,17 @@ func applyReverseProxy(ctx context.Context, panelHost, adminHost string) error {
|
|||||||
// writeConnectionConfig stamps the chosen hostnames (and optional Access audience)
|
// writeConnectionConfig stamps the chosen hostnames (and optional Access audience)
|
||||||
// into both the host and pod config files. An empty aud clears any prior
|
// into both the host and pod config files. An empty aud clears any prior
|
||||||
// Cloudflare audience, which is correct when switching to a non-Access front.
|
// Cloudflare audience, which is correct when switching to a non-Access front.
|
||||||
func writeConnectionConfig(panelHost, adminHost, aud string) error {
|
// clientIPHeader is the header that front writes the visitor address into.
|
||||||
|
func writeConnectionConfig(panelHost, adminHost, aud, clientIPHeader string) error {
|
||||||
for _, path := range []string{hostSetupConfigPath, podSetupConfigPath} {
|
for _, path := range []string{hostSetupConfigPath, podSetupConfigPath} {
|
||||||
if err := updateAuthConfig(path, panelHost, adminHost, aud); err != nil {
|
if err := updateAuthConfig(path, panelHost, adminHost, aud, clientIPHeader); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func updateAuthConfig(path, panelHost, adminHost, aud string) error {
|
func updateAuthConfig(path, panelHost, adminHost, aud, clientIPHeader string) error {
|
||||||
cfg, err := config.Load(path)
|
cfg, err := config.Load(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -112,6 +116,7 @@ func updateAuthConfig(path, panelHost, adminHost, aud string) error {
|
|||||||
}
|
}
|
||||||
cfg.Auth.AdminHostname = adminHost
|
cfg.Auth.AdminHostname = adminHost
|
||||||
cfg.Auth.AccessJWTAud = aud
|
cfg.Auth.AccessJWTAud = aud
|
||||||
|
cfg.Auth.ClientIPHeader = clientIPHeader
|
||||||
return writeConfig(path, cfg)
|
return writeConfig(path, cfg)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -3,8 +3,10 @@ package main
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
"felis.lolicon.best/internal/store"
|
"felis.lolicon.best/internal/store"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -12,7 +14,7 @@ import (
|
|||||||
// applied count. Used by the preflight stage to self-heal a freshly bootstrapped
|
// applied count. Used by the preflight stage to self-heal a freshly bootstrapped
|
||||||
// (or upgraded) database.
|
// (or upgraded) database.
|
||||||
func applyMigrations(dbURL string) (int, error) {
|
func applyMigrations(dbURL string) (int, error) {
|
||||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||||
defer cancel()
|
defer cancel()
|
||||||
drv, err := store.Open(ctx, dbURL)
|
drv, err := store.Open(ctx, dbURL)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -23,6 +25,11 @@ func applyMigrations(dbURL string) (int, error) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
|
// Same guard as `felis migrate up`: never roll a populated database forward
|
||||||
|
// without a snapshot to roll back to.
|
||||||
|
if _, err := preMigrateBackup(ctx, drv, migrations, dbURL, dbbackup.DefaultDir, io.Discard); err != nil {
|
||||||
|
return 0, fmt.Errorf("pre-migration backup: %w", err)
|
||||||
|
}
|
||||||
if _, err := store.Up(ctx, drv, migrations); err != nil {
|
if _, err := store.Up(ctx, drv, migrations); err != nil {
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
|
|||||||
+35
-4
@@ -50,10 +50,13 @@ type updateTarget struct {
|
|||||||
// and even on the bootstrap path it re-images felis-api from the binary setup is already
|
// and even on the bootstrap path it re-images felis-api from the binary setup is already
|
||||||
// running (FELIS_BOOTSTRAP_BINARY), which looks like an update and changes nothing.
|
// running (FELIS_BOOTSTRAP_BINARY), which looks like an update and changes nothing.
|
||||||
//
|
//
|
||||||
// The URL is the same one-liner both READMEs hand out. While the repo is private it
|
// The URL is the one-liner both READMEs hand out, read at a tag rather than main: the
|
||||||
|
// script's release channel installs the newest release's binary, and main can carry
|
||||||
|
// installer changes that binary was never tested with. installerRef picks the tag and
|
||||||
|
// renderApplyGuidance substitutes it for {ref}. While the repo is private the URL
|
||||||
// answers 404 (raw.githubusercontent.com hides private repos), which is why the trailer
|
// answers 404 (raw.githubusercontent.com hides private repos), which is why the trailer
|
||||||
// below points at the README's token'd form for that case.
|
// below points at the README's token'd form for that case.
|
||||||
const installerRerun = "curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash"
|
const installerRerun = "curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/{ref}/deploy/bootstrap.sh | sudo bash"
|
||||||
|
|
||||||
// updateTargets is the selector table. panel and plugins both resolve to felis-api
|
// updateTargets is the selector table. panel and plugins both resolve to felis-api
|
||||||
// because they are not separately versioned: the panel is compiled into the felis
|
// because they are not separately versioned: the panel is compiled into the felis
|
||||||
@@ -69,7 +72,7 @@ var updateTargets = []updateTarget{
|
|||||||
{
|
{
|
||||||
selector: "velocity",
|
selector: "velocity",
|
||||||
component: "velocity",
|
component: "velocity",
|
||||||
note: "re-runs install_velocity: newest BUILD of the pinned minor (FELIS_VELOCITY_VERSION), atomic jar install, then restarts felis-velocity",
|
note: "re-runs install_velocity: the build the release pins in deploy/game-stack.lock (FELIS_VELOCITY_VERSION=<minor> takes that minor's newest build instead), sha256-checked, atomic jar install, then restarts felis-velocity only if the jar or its config changed",
|
||||||
command: installerRerun,
|
command: installerRerun,
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
@@ -258,7 +261,7 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
|||||||
// component, and reinstalling the current release is a valid repair action.
|
// component, and reinstalling the current release is a valid repair action.
|
||||||
fmt.Fprintf(&b, " note: cannot tell whether %s is current — its latest version could not be discovered (see above); this reinstalls it either way\n", t.component)
|
fmt.Fprintf(&b, " note: cannot tell whether %s is current — its latest version could not be discovered (see above); this reinstalls it either way\n", t.component)
|
||||||
}
|
}
|
||||||
fmt.Fprintf(&b, " run: %s\n", t.command)
|
fmt.Fprintf(&b, " run: %s\n", strings.ReplaceAll(t.command, "{ref}", installerRef(byComponent)))
|
||||||
offeredCommand = true
|
offeredCommand = true
|
||||||
}
|
}
|
||||||
// Only explain the command when one was actually offered; a --mc-only run has
|
// Only explain the command when one was actually offered; a --mc-only run has
|
||||||
@@ -277,3 +280,31 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
|||||||
}
|
}
|
||||||
return b.String()
|
return b.String()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// installerRef is the git ref the installer re-run reads bootstrap.sh from: the newest
|
||||||
|
// stable felis release when the feed answered, which is the release that script then
|
||||||
|
// installs; else the release this host runs; main only when neither is a release tag.
|
||||||
|
func installerRef(byComponent map[string]updates.Action) string {
|
||||||
|
a, ok := byComponent["felis-api"]
|
||||||
|
if !ok {
|
||||||
|
return "main"
|
||||||
|
}
|
||||||
|
if a.LatestKnown && isReleaseTag(a.Latest) {
|
||||||
|
return a.Latest.String()
|
||||||
|
}
|
||||||
|
if isReleaseTag(a.Current) {
|
||||||
|
return a.Current.String()
|
||||||
|
}
|
||||||
|
return "main"
|
||||||
|
}
|
||||||
|
|
||||||
|
// isReleaseTag reports whether v was read from a stable vX.Y.Z tag, the only refs
|
||||||
|
// release.yml publishes a binary for.
|
||||||
|
func isReleaseTag(v updates.Version) bool {
|
||||||
|
s := v.String()
|
||||||
|
if !strings.HasPrefix(s, "v") || v.IsPrerelease() {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
_, err := updates.Parse(s)
|
||||||
|
return err == nil
|
||||||
|
}
|
||||||
@@ -167,3 +167,35 @@ func TestApplyGuidancePointsEveryComponentAtTheInstaller(t *testing.T) {
|
|||||||
t.Fatalf("--mc offers no command; the trailer is a non-sequitur:\n%s", mc)
|
t.Fatalf("--mc offers no command; the trailer is a non-sequitur:\n%s", mc)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The re-run reads bootstrap.sh at the tag whose binary it installs. main can carry
|
||||||
|
// installer changes no release was tested with.
|
||||||
|
func TestApplyGuidanceReadsTheInstallerAtTheReleaseTag(t *testing.T) {
|
||||||
|
v := func(s string) updates.Version {
|
||||||
|
t.Helper()
|
||||||
|
out, err := updates.Parse(s)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
api []updates.Action
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"latest known", []updates.Action{{Component: "felis-api", Kind: updates.ActionNotify, Current: v("v1.3.0"), Latest: v("v1.4.0"), LatestKnown: true}}, "/FelisMC/Felis/v1.4.0/deploy/bootstrap.sh"},
|
||||||
|
{"latest unknown", []updates.Action{{Component: "felis-api", Kind: updates.ActionNone, Current: v("v1.3.0")}}, "/FelisMC/Felis/v1.3.0/deploy/bootstrap.sh"},
|
||||||
|
{"prerelease latest", []updates.Action{{Component: "felis-api", Kind: updates.ActionNone, Current: v("v1.3.0"), Latest: v("v1.4.0-rc.1"), LatestKnown: true}}, "/FelisMC/Felis/v1.3.0/deploy/bootstrap.sh"},
|
||||||
|
{"nothing known", nil, "/FelisMC/Felis/main/deploy/bootstrap.sh"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
out := renderApplyGuidance(planResult(c.api), map[string]bool{"velocity": true, "panel": true}, true)
|
||||||
|
if !strings.Contains(out, c.want) {
|
||||||
|
t.Errorf("%s: want %q in:\n%s", c.name, c.want, out)
|
||||||
|
}
|
||||||
|
if strings.Contains(out, "{ref}") {
|
||||||
|
t.Errorf("%s: placeholder left in:\n%s", c.name, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,269 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"io"
|
||||||
|
"net"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/config"
|
||||||
|
"felis.lolicon.best/internal/mail"
|
||||||
|
"felis.lolicon.best/internal/offsite"
|
||||||
|
"felis.lolicon.best/internal/platform"
|
||||||
|
"felis.lolicon.best/internal/store"
|
||||||
|
"felis.lolicon.best/internal/watchdog"
|
||||||
|
corev1 "k8s.io/api/core/v1"
|
||||||
|
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||||
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||||
|
)
|
||||||
|
|
||||||
|
// proxyFor is how long the game proxy may refuse connections before it is
|
||||||
|
// mailed: a restart takes seconds.
|
||||||
|
const proxyFor = 3 * time.Minute
|
||||||
|
|
||||||
|
// cmdWatchdog runs one pass of the platform watchdog (internal/watchdog): it
|
||||||
|
// checks the cluster, PostgreSQL, the game proxy, the database backups and the
|
||||||
|
// host, prints every finding, and mails the platform owners what came due.
|
||||||
|
// deploy/bootstrap.sh runs it every two minutes from felis-watchdog.timer.
|
||||||
|
func cmdWatchdog(args []string, stdout, stderr io.Writer) int {
|
||||||
|
fs := flag.NewFlagSet("watchdog", flag.ContinueOnError)
|
||||||
|
fs.SetOutput(stderr)
|
||||||
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
|
||||||
|
statePath := fs.String("state", "/var/lib/felis/watchdog/state.json", "state kept between runs (root only: it caches the relay password)")
|
||||||
|
quietPath := fs.String("quiet-file", "/run/felis/watchdog-quiet-until", "Unix time before which nothing is mailed; the installer writes it while it restarts things on purpose")
|
||||||
|
backupDir := fs.String("backup-dir", "/var/lib/felis/db-backups", `control-plane database backups to check for freshness ("" skips the check)`)
|
||||||
|
diskPaths := fs.String("disk-paths", "/,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis", "comma-separated paths whose filesystems must keep free space")
|
||||||
|
proxyAddr := fs.String("proxy-addr", "", `game proxy address to dial, e.g. 127.0.0.1:25565 ("" skips the check)`)
|
||||||
|
controlNS := fs.String("control-namespace", platform.DefaultControlNamespace, "namespace of the control plane")
|
||||||
|
offsiteStatus := fs.String("offsite-status", offsite.DefaultStatusFile, "the record `felis offsite sync` leaves, checked when [offsite] is configured")
|
||||||
|
toolsStatus := fs.String("build-tools-status", defaultBuildToolsStatus, "the record `felis mirror-build-tools` leaves, checked when builds scan against the registry's DB copy")
|
||||||
|
dryRun := fs.Bool("dry-run", false, "print every finding and the mail that is due; send nothing and keep the state as it was")
|
||||||
|
if err := fs.Parse(args); err != nil {
|
||||||
|
if errors.Is(err, flag.ErrHelp) {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
cfg, err := config.Load(*cfgPath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis watchdog: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
state, err := watchdog.LoadState(*statePath)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis watchdog: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
|
||||||
|
defer cancel()
|
||||||
|
now := time.Now()
|
||||||
|
|
||||||
|
var report watchdog.Report
|
||||||
|
add := func(f *watchdog.Finding) {
|
||||||
|
if f != nil {
|
||||||
|
report.Findings = append(report.Findings, *f)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The cluster: one unreachable API server stands in for every check behind it.
|
||||||
|
minecraftNS := cfg.K8s.Namespace
|
||||||
|
if minecraftNS == "" {
|
||||||
|
minecraftNS = platform.DefaultMinecraftNamespace
|
||||||
|
}
|
||||||
|
cl, err := buildSystemServerClient()
|
||||||
|
var found []watchdog.Finding
|
||||||
|
if err == nil {
|
||||||
|
found, err = watchdog.Cluster{Client: cl, ControlNamespace: *controlNS, MinecraftNamespace: minecraftNS}.Check(ctx, now)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
f := watchdog.KubeAPIDown(err)
|
||||||
|
add(&f)
|
||||||
|
report.Unknown = append(report.Unknown, watchdog.ClusterPrefixes...)
|
||||||
|
} else {
|
||||||
|
report.Findings = append(report.Findings, found...)
|
||||||
|
if cfg.SMTP.Host != "" {
|
||||||
|
refreshSMTPPassword(ctx, cl, *controlNS, state, stderr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if recipients, err := ownerEmails(ctx, cfg.Database.URL); err != nil {
|
||||||
|
f := watchdog.PostgresDown(err)
|
||||||
|
add(&f)
|
||||||
|
} else {
|
||||||
|
state.Recipients = recipients
|
||||||
|
}
|
||||||
|
|
||||||
|
if *proxyAddr != "" {
|
||||||
|
add(proxyFinding(ctx, *proxyAddr))
|
||||||
|
}
|
||||||
|
if *backupDir != "" {
|
||||||
|
add(watchdog.BackupFinding(*backupDir, now))
|
||||||
|
}
|
||||||
|
if cfg.Offsite.Enabled() {
|
||||||
|
add(watchdog.OffsiteFinding(*offsiteStatus, now))
|
||||||
|
}
|
||||||
|
if usesMirroredScanDB(cfg) {
|
||||||
|
add(watchdog.ScanDBFinding(*toolsStatus, now))
|
||||||
|
}
|
||||||
|
report.Findings = append(report.Findings, watchdog.DiskFindings(splitList(*diskPaths))...)
|
||||||
|
add(watchdog.MemoryFinding("/proc/meminfo"))
|
||||||
|
|
||||||
|
if len(report.Findings) == 0 {
|
||||||
|
fmt.Fprintln(stdout, "felis watchdog: every check passed")
|
||||||
|
}
|
||||||
|
for _, f := range report.Findings {
|
||||||
|
fmt.Fprintf(stdout, "felis watchdog: [%s] %s: %s\n", f.Severity, f.Key, f.SummaryEN)
|
||||||
|
}
|
||||||
|
|
||||||
|
plan := state.Observe(report, now)
|
||||||
|
host, _ := os.Hostname()
|
||||||
|
subject, body := plan.Message(host, now)
|
||||||
|
if *dryRun {
|
||||||
|
if plan.Empty() {
|
||||||
|
fmt.Fprintln(stdout, "felis watchdog: nothing is due to be mailed")
|
||||||
|
} else {
|
||||||
|
fmt.Fprintf(stdout, "felis watchdog: due to be mailed to %s:\nSubject: %s\n\n%s", strings.Join(state.Recipients, ", "), subject, strings.ReplaceAll(body, "\r\n", "\n"))
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
save := func() int {
|
||||||
|
if err := watchdog.SaveState(*statePath, state); err != nil {
|
||||||
|
fmt.Fprintf(stderr, "felis watchdog: save state: %v\n", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
if plan.Empty() {
|
||||||
|
return save()
|
||||||
|
}
|
||||||
|
if until := watchdog.QuietUntil(*quietPath); now.Before(until) {
|
||||||
|
fmt.Fprintf(stdout, "felis watchdog: quiet until %s (installer running); holding this mail: %s\n", until.UTC().Format(time.RFC3339), subject)
|
||||||
|
return save()
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case cfg.SMTP.Host == "":
|
||||||
|
fmt.Fprintf(stdout, "felis watchdog: no [smtp] relay configured, so this is logged only: %s\n", subject)
|
||||||
|
case len(state.Recipients) == 0:
|
||||||
|
fmt.Fprintf(stdout, "felis watchdog: no owner account has a verified email, so this is logged only: %s\n", subject)
|
||||||
|
default:
|
||||||
|
if err := sendAlert(ctx, cfg, state, subject, body); err != nil {
|
||||||
|
// Not committed: the same alerts come due again next run.
|
||||||
|
fmt.Fprintf(stderr, "felis watchdog: mail %q: %v\n", subject, err)
|
||||||
|
save()
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
fmt.Fprintf(stdout, "felis watchdog: mailed %s: %s\n", strings.Join(state.Recipients, ", "), subject)
|
||||||
|
}
|
||||||
|
state.Commit(plan, now)
|
||||||
|
return save()
|
||||||
|
}
|
||||||
|
|
||||||
|
// usesMirroredScanDB reports whether build scans read the vulnerability DB copy
|
||||||
|
// felis mirror-build-tools keeps in the platform registry: the default, or an
|
||||||
|
// explicit trivy_db_repository under the registry's mirror/.
|
||||||
|
func usesMirroredScanDB(cfg *config.Config) bool {
|
||||||
|
if cfg.Registry.URL == "" {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
repo := cfg.Registry.TrivyDBRepository
|
||||||
|
return repo == "" || strings.HasPrefix(repo, cfg.Registry.URL+"/mirror/")
|
||||||
|
}
|
||||||
|
|
||||||
|
// refreshSMTPPassword caches the relay password from the felis-smtp Secret, or
|
||||||
|
// forgets it when the Secret is gone (a relay without AUTH). An env var named by
|
||||||
|
// [smtp] password_ref, when set, wins at send time instead.
|
||||||
|
func refreshSMTPPassword(ctx context.Context, cl client.Client, ns string, state *watchdog.State, stderr io.Writer) {
|
||||||
|
var sec corev1.Secret
|
||||||
|
err := cl.Get(ctx, client.ObjectKey{Namespace: ns, Name: platform.SMTPSecretName}, &sec)
|
||||||
|
switch {
|
||||||
|
case apierrors.IsNotFound(err):
|
||||||
|
state.SMTPPassword = ""
|
||||||
|
case err != nil:
|
||||||
|
fmt.Fprintf(stderr, "felis watchdog: read %s/%s (keeping the cached relay password): %v\n", ns, platform.SMTPSecretName, err)
|
||||||
|
default:
|
||||||
|
state.SMTPPassword = string(sec.Data[platform.SMTPSecretPasswordKey])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ownerEmails pings PostgreSQL and returns the verified addresses of the
|
||||||
|
// enabled owner accounts, the people who can act on an alert.
|
||||||
|
func ownerEmails(ctx context.Context, url string) ([]string, error) {
|
||||||
|
ctx, cancel := context.WithTimeout(ctx, 15*time.Second)
|
||||||
|
defer cancel()
|
||||||
|
drv, err := store.Open(ctx, url)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer drv.Close()
|
||||||
|
rows, err := drv.DB().QueryContext(ctx,
|
||||||
|
`SELECT email FROM users
|
||||||
|
WHERE role = 'owner' AND email_verified AND COALESCE(email, '') <> ''
|
||||||
|
AND NOT disabled AND deleted_at IS NULL
|
||||||
|
ORDER BY email`)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
var out []string
|
||||||
|
for rows.Next() {
|
||||||
|
var email sql.NullString
|
||||||
|
if err := rows.Scan(&email); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
out = append(out, email.String)
|
||||||
|
}
|
||||||
|
return out, rows.Err()
|
||||||
|
}
|
||||||
|
|
||||||
|
// proxyFinding dials the game proxy; players reach every server through it.
|
||||||
|
func proxyFinding(ctx context.Context, addr string) *watchdog.Finding {
|
||||||
|
d := net.Dialer{Timeout: 5 * time.Second}
|
||||||
|
conn, err := d.DialContext(ctx, "tcp", addr)
|
||||||
|
if err == nil {
|
||||||
|
conn.Close()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return &watchdog.Finding{
|
||||||
|
Key: "proxy", Severity: watchdog.Critical, For: proxyFor,
|
||||||
|
Summary: fmt.Sprintf("游戏代理 %s 无法连接:玩家进不了任何服务器", addr),
|
||||||
|
SummaryEN: fmt.Sprintf("the game proxy at %s refuses connections: players cannot reach any server", addr),
|
||||||
|
Hint: fmt.Sprintf("systemctl status felis-velocity; journalctl -u felis-velocity -n 200 (%v)", err),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sendAlert mails subject/body to every recipient; it fails only when no
|
||||||
|
// recipient got it.
|
||||||
|
func sendAlert(ctx context.Context, cfg *config.Config, state *watchdog.State, subject, body string) error {
|
||||||
|
password := state.SMTPPassword
|
||||||
|
if ref := cfg.SMTP.PasswordRef; ref != "" && os.Getenv(ref) != "" {
|
||||||
|
password = os.Getenv(ref)
|
||||||
|
}
|
||||||
|
relay := &mail.SMTP{Host: cfg.SMTP.Host, Port: cfg.SMTP.Port, From: cfg.SMTP.From, Username: cfg.SMTP.Username, Password: password}
|
||||||
|
var errs []error
|
||||||
|
for _, to := range state.Recipients {
|
||||||
|
if err := relay.SendNotice(ctx, to, subject, body); err != nil {
|
||||||
|
errs = append(errs, fmt.Errorf("%s: %w", to, err))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(errs) == len(state.Recipients) {
|
||||||
|
return errors.Join(errs...)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func splitList(s string) []string {
|
||||||
|
var out []string
|
||||||
|
for _, p := range strings.Split(s, ",") {
|
||||||
|
if p = strings.TrimSpace(p); p != "" {
|
||||||
|
out = append(out, p)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"net"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestProxyFinding: a listening proxy is healthy; a closed port is the critical
|
||||||
|
// "players cannot reach any server" finding.
|
||||||
|
func TestProxyFinding(t *testing.T) {
|
||||||
|
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
addr := ln.Addr().String()
|
||||||
|
go func() {
|
||||||
|
for {
|
||||||
|
c, err := ln.Accept()
|
||||||
|
if err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
c.Close()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
if f := proxyFinding(context.Background(), addr); f != nil {
|
||||||
|
t.Fatalf("listening proxy reported: %+v", f)
|
||||||
|
}
|
||||||
|
ln.Close()
|
||||||
|
f := proxyFinding(context.Background(), addr)
|
||||||
|
if f == nil || f.Key != "proxy" || !strings.Contains(f.SummaryEN, addr) {
|
||||||
|
t.Fatalf("closed proxy = %+v, want the proxy finding", f)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSplitList(t *testing.T) {
|
||||||
|
got := splitList(" /, /var/lib/felis ,,")
|
||||||
|
if strings.Join(got, "|") != "/|/var/lib/felis" {
|
||||||
|
t.Fatalf("splitList = %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,10 +1,21 @@
|
|||||||
# Felis alert rules — plain Prometheus format (also the promtool-tested source
|
# Felis alert rules — plain Prometheus format (also the promtool-tested source
|
||||||
# for felis-prometheusrule.yaml). See docs/troubleshooting.md §14 for scraping
|
# for felis-prometheusrule.yaml). See docs/troubleshooting.md §14 for scraping
|
||||||
# and loading instructions.
|
# and loading instructions. These are for a deployment that brings its own
|
||||||
|
# Prometheus; every install already runs `felis watchdog` on the host, which
|
||||||
|
# checks the same conditions without one and mails the owners (§14).
|
||||||
#
|
#
|
||||||
# felis_* series come from two processes:
|
# felis_* series come from these processes:
|
||||||
# - felis-operator pod :8080/metrics → felis_servers_total, felis_start_duration_seconds
|
# - felis-operator-metrics Service :8080 → felis_servers_total, felis_server_phase,
|
||||||
# - felis-api internal :8081/metrics → felis_image_build_failures_total
|
# felis_start_duration_seconds,
|
||||||
|
# felis_build_info{component="operator"},
|
||||||
|
# controller_runtime_*, workqueue_*
|
||||||
|
# - felis-api-internal Service :8081 → felis_image_build_failures_total,
|
||||||
|
# felis_mail_total, felis_rate_limited_total,
|
||||||
|
# felis_auth_otp_lockouts_total,
|
||||||
|
# felis_auth_failures_total,
|
||||||
|
# felis_audit_write_failures_total,
|
||||||
|
# felis_build_info{component="api"}
|
||||||
|
# - node-exporter textfile collector → felis_db_backup_* (felis-db-backup.timer)
|
||||||
# node_* / kube_* series come from node-exporter / kube-state-metrics.
|
# node_* / kube_* series come from node-exporter / kube-state-metrics.
|
||||||
groups:
|
groups:
|
||||||
- name: felis.rules
|
- name: felis.rules
|
||||||
@@ -32,6 +43,116 @@ groups:
|
|||||||
observed when readiness is first reached). A start that never completes
|
observed when readiness is first reached). A start that never completes
|
||||||
records nothing — cross-check desiredState=Running servers with no ready
|
records nothing — cross-check desiredState=Running servers with no ready
|
||||||
phase (troubleshooting §1).
|
phase (troubleshooting §1).
|
||||||
|
- name: felis.platform.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisOperatorDown
|
||||||
|
expr: absent(felis_build_info{component="operator"})
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "felis-operator is down or not scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_build_info{component="operator"} series for 10 minutes. Without
|
||||||
|
the operator no server starts, stops or recovers. Check
|
||||||
|
`kubectl -n felis get deploy felis-operator` and its log; if the pod is
|
||||||
|
healthy, the felis-operator-metrics Service is not being scraped
|
||||||
|
(troubleshooting §14).
|
||||||
|
- alert: FelisAPIDown
|
||||||
|
expr: absent(felis_build_info{component="api"})
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "felis-api is down or not scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_build_info{component="api"} series for 10 minutes. The panel,
|
||||||
|
sign-in and the proxy's player lookups all go through felis-api. Check
|
||||||
|
`kubectl -n felis get deploy felis-api` and its log; if the pod is
|
||||||
|
healthy, the felis-api-internal Service is not being scraped
|
||||||
|
(troubleshooting §14).
|
||||||
|
- alert: FelisLoginGateDown
|
||||||
|
expr: felis_server_phase{role="login",desired="Running",phase!="Running"} == 1
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "the login gate {{ $labels.server }} is {{ $labels.phase }}"
|
||||||
|
description: >-
|
||||||
|
Every player connection passes through the login server first, so no one
|
||||||
|
can join. The MinecraftServer's conditions carry the reason:
|
||||||
|
`kubectl -n minecraft describe minecraftserver {{ $labels.server }}`
|
||||||
|
(troubleshooting §1, §2).
|
||||||
|
- alert: FelisSystemServerDown
|
||||||
|
expr: felis_server_phase{role!="",role!="login",desired="Running",phase!="Running"} == 1
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "system server {{ $labels.server }} ({{ $labels.role }}) is {{ $labels.phase }}"
|
||||||
|
description: >-
|
||||||
|
Players who sign in are sent to the lobby; while it is down they stay at
|
||||||
|
the gate. `kubectl -n minecraft describe minecraftserver {{ $labels.server }}`
|
||||||
|
shows the reason (troubleshooting §1, §2).
|
||||||
|
- alert: FelisServerFailed
|
||||||
|
expr: felis_server_phase{role="",phase="Failed"} == 1
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "server {{ $labels.server }} is Failed"
|
||||||
|
description: >-
|
||||||
|
The operator gave up on this server (a crash loop, an image that will not
|
||||||
|
pull, a world volume that will not mount). Its conditions carry the
|
||||||
|
reason: `kubectl -n minecraft describe minecraftserver {{ $labels.server }}`
|
||||||
|
(troubleshooting §2).
|
||||||
|
- alert: FelisReconcileErrors
|
||||||
|
expr: sum(increase(controller_runtime_reconcile_errors_total{controller="minecraftserver"}[15m])) > 10
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the operator failed over 10 reconciles in 15 minutes"
|
||||||
|
description: >-
|
||||||
|
Server changes are being retried instead of applied. The felis-operator
|
||||||
|
log names each failing server and its error.
|
||||||
|
- alert: FelisReconcileStuck
|
||||||
|
expr: max(workqueue_longest_running_processor_seconds{name="minecraftserver"}) > 300
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "an operator reconcile has been running for over 5 minutes"
|
||||||
|
description: >-
|
||||||
|
Each reconcile is bounded at 3 minutes, so this one is ignoring its
|
||||||
|
deadline and holding a worker. The liveness probe restarts the operator
|
||||||
|
once a pass passes 10 minutes; the log from before the restart shows
|
||||||
|
where it hung.
|
||||||
|
- name: felis.jobs.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisWorldJobFailed
|
||||||
|
expr: kube_job_failed{namespace="minecraft",condition="true"} == 1
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Job {{ $labels.job_name }} failed"
|
||||||
|
description: >-
|
||||||
|
A world backup, restore or reaper run failed; after a failed backup that
|
||||||
|
world's newest archive is older than planned.
|
||||||
|
`kubectl -n minecraft logs job/{{ $labels.job_name }}` has the error
|
||||||
|
(troubleshooting §10).
|
||||||
|
- alert: FelisReaperStale
|
||||||
|
expr: time() - kube_cronjob_status_last_successful_time{namespace="minecraft",cronjob="felis-reaper"} > 26 * 3600
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the world reaper has not succeeded in over 26h"
|
||||||
|
description: >-
|
||||||
|
felis-reaper runs daily; idle worlds are neither backed up nor reclaimed
|
||||||
|
while it fails. `kubectl -n minecraft get jobs --sort-by=.metadata.creationTimestamp`
|
||||||
|
lists its runs, and the newest one's log
|
||||||
|
shows why (troubleshooting §10).
|
||||||
- name: felis.node.rules
|
- name: felis.node.rules
|
||||||
rules:
|
rules:
|
||||||
- alert: FelisNodeDiskSpaceLow
|
- alert: FelisNodeDiskSpaceLow
|
||||||
@@ -67,3 +188,97 @@ groups:
|
|||||||
description: >-
|
description: >-
|
||||||
PostgreSQL, the control plane, the registry and game servers share one
|
PostgreSQL, the control plane, the registry and game servers share one
|
||||||
node; sustained memory pressure risks OOM kills.
|
node; sustained memory pressure risks OOM kills.
|
||||||
|
- name: felis.backup.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisDBBackupStale
|
||||||
|
expr: time() - max(felis_db_backup_last_success_timestamp_seconds) > 26 * 3600
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "no control-plane database backup in over 26h"
|
||||||
|
description: >-
|
||||||
|
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
||||||
|
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
||||||
|
with `sudo felis db backup` (troubleshooting §16).
|
||||||
|
- alert: FelisDBBackupMetricMissing
|
||||||
|
expr: absent(felis_db_backup_last_success_timestamp_seconds)
|
||||||
|
for: 2h
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "database backup freshness is not being scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_db_backup_last_success_timestamp_seconds series, so
|
||||||
|
FelisDBBackupStale cannot fire. Point node-exporter's
|
||||||
|
--collector.textfile.directory at the directory of
|
||||||
|
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
||||||
|
(troubleshooting §16).
|
||||||
|
- name: felis.auth.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisMailBudgetExhausted
|
||||||
|
expr: sum(increase(felis_mail_total{result="throttled"}[15m])) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the install-wide mail budget refused mail"
|
||||||
|
description: >-
|
||||||
|
felis_mail_total{result="throttled"} increased: [smtp] max_per_hour is
|
||||||
|
spent, and every sign-in code is refused with 429 mail_rate_limited until
|
||||||
|
it refills. Check felis_rate_limited_total for a flood before raising the
|
||||||
|
budget (troubleshooting §17).
|
||||||
|
- alert: FelisMailDeliveryFailing
|
||||||
|
expr: sum(increase(felis_mail_total{result="failed"}[15m])) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the SMTP relay refused mail in the last 15m"
|
||||||
|
description: >-
|
||||||
|
felis_mail_total{result="failed"} increased: sign-in codes are not being
|
||||||
|
delivered (502 mail_undeliverable). The relay's reason is in the
|
||||||
|
felis-api log (troubleshooting §17).
|
||||||
|
- alert: FelisSignInFlood
|
||||||
|
expr: sum(rate(felis_rate_limited_total{scope="auth_door"}[5m])) * 60 > 10
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "sign-in doors refusing over 10 requests a minute"
|
||||||
|
description: >-
|
||||||
|
The per-address sign-in limit has been refusing callers for 10 minutes.
|
||||||
|
A script is hammering the auth doors; if real users report rate_limited
|
||||||
|
at once instead, [auth] client_ip_header is missing and everyone shares
|
||||||
|
the proxy's address (troubleshooting §17).
|
||||||
|
- alert: FelisOTPAccountLocked
|
||||||
|
expr: sum by (purpose) (increase(felis_auth_otp_lockouts_total[1h])) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "an account's email-code sign-in locked after 10 wrong codes"
|
||||||
|
description: >-
|
||||||
|
Someone entered 10 wrong codes for one account within 24h ({{ $labels.purpose }}).
|
||||||
|
The audit log names the account (action auth.otp.locked); the owner was
|
||||||
|
mailed. Unless they fumbled codes, someone is guessing at it
|
||||||
|
(troubleshooting §17).
|
||||||
|
- alert: FelisSignInFailures
|
||||||
|
expr: sum(increase(felis_auth_failures_total[15m])) > 30
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "over 30 refused sign-ins in 15 minutes"
|
||||||
|
description: >-
|
||||||
|
Wrong codes, unknown addresses or bad passkey assertions well above people
|
||||||
|
mistyping: someone is guessing or enumerating. `sum by (door, reason)
|
||||||
|
(increase(felis_auth_failures_total[15m]))` shows where; the audit rows
|
||||||
|
(action auth.<door>.failed) carry each caller's client_ip (troubleshooting §17).
|
||||||
|
- alert: FelisAuditWriteFailing
|
||||||
|
expr: increase(felis_audit_write_failures_total[15m]) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "felis-api failed to write audit rows"
|
||||||
|
description: >-
|
||||||
|
The actions went through but their audit rows were lost. The felis-api log
|
||||||
|
names each lost row (`audit: lost ...`); the usual cause is PostgreSQL
|
||||||
|
being unreachable or out of disk.
|
||||||
@@ -105,3 +105,405 @@ tests:
|
|||||||
description: >-
|
description: >-
|
||||||
PostgreSQL, the control plane, the registry and game servers share one
|
PostgreSQL, the control plane, the registry and game servers share one
|
||||||
node; sustained memory pressure risks OOM kills.
|
node; sustained memory pressure risks OOM kills.
|
||||||
|
- name: database backup freshness
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
# The newest bundle was taken at t=0 and none since.
|
||||||
|
- series: 'felis_db_backup_last_success_timestamp_seconds{instance="node1",job="node-exporter",label="daily"}'
|
||||||
|
values: '0x1630'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 25h
|
||||||
|
alertname: FelisDBBackupStale
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 27h
|
||||||
|
alertname: FelisDBBackupStale
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: critical
|
||||||
|
exp_annotations:
|
||||||
|
summary: "no control-plane database backup in over 26h"
|
||||||
|
description: >-
|
||||||
|
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
||||||
|
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
||||||
|
with `sudo felis db backup` (troubleshooting §16).
|
||||||
|
- eval_time: 27h
|
||||||
|
alertname: FelisDBBackupMetricMissing
|
||||||
|
exp_alerts: []
|
||||||
|
- name: database backup freshness not scraped
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'up{job="node-exporter"}'
|
||||||
|
values: '1x200'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 1h
|
||||||
|
alertname: FelisDBBackupMetricMissing
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 3h
|
||||||
|
alertname: FelisDBBackupMetricMissing
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
exp_annotations:
|
||||||
|
summary: "database backup freshness is not being scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_db_backup_last_success_timestamp_seconds series, so
|
||||||
|
FelisDBBackupStale cannot fire. Point node-exporter's
|
||||||
|
--collector.textfile.directory at the directory of
|
||||||
|
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
||||||
|
(troubleshooting §16).
|
||||||
|
- name: sign-in mail budget and relay
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
# Created at zero on start; the budget refuses one mail at t=3m.
|
||||||
|
- series: 'felis_mail_total{kind="otp",result="throttled",job="felis-api"}'
|
||||||
|
values: '0 0 0 1x30'
|
||||||
|
- series: 'felis_mail_total{kind="otp",result="failed",job="felis-api"}'
|
||||||
|
values: '0x33'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 2m
|
||||||
|
alertname: FelisMailBudgetExhausted
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisMailBudgetExhausted
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
exp_annotations:
|
||||||
|
summary: "the install-wide mail budget refused mail"
|
||||||
|
description: >-
|
||||||
|
felis_mail_total{result="throttled"} increased: [smtp] max_per_hour is
|
||||||
|
spent, and every sign-in code is refused with 429 mail_rate_limited until
|
||||||
|
it refills. Check felis_rate_limited_total for a flood before raising the
|
||||||
|
budget (troubleshooting §17).
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisMailDeliveryFailing
|
||||||
|
exp_alerts: []
|
||||||
|
- name: sign-in flood
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
# 30 refusals a minute from t=0; a lone refused script at 2/min stays quiet.
|
||||||
|
- series: 'felis_rate_limited_total{scope="auth_door",job="felis-api"}'
|
||||||
|
values: '0+30x40'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 10m
|
||||||
|
alertname: FelisSignInFlood
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 20m
|
||||||
|
alertname: FelisSignInFlood
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
exp_annotations:
|
||||||
|
summary: "sign-in doors refusing over 10 requests a minute"
|
||||||
|
description: >-
|
||||||
|
The per-address sign-in limit has been refusing callers for 10 minutes.
|
||||||
|
A script is hammering the auth doors; if real users report rate_limited
|
||||||
|
at once instead, [auth] client_ip_header is missing and everyone shares
|
||||||
|
the proxy's address (troubleshooting §17).
|
||||||
|
- name: sign-in trickle stays quiet
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_rate_limited_total{scope="auth_door",job="felis-api"}'
|
||||||
|
values: '0+2x40'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 30m
|
||||||
|
alertname: FelisSignInFlood
|
||||||
|
exp_alerts: []
|
||||||
|
- name: account email-code lock
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_auth_otp_lockouts_total{purpose="login_email",job="felis-api"}'
|
||||||
|
values: '0 0 1x90'
|
||||||
|
- series: 'felis_auth_otp_lockouts_total{purpose="op_login",job="felis-api"}'
|
||||||
|
values: '0x92'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 1m
|
||||||
|
alertname: FelisOTPAccountLocked
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 10m
|
||||||
|
alertname: FelisOTPAccountLocked
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
purpose: login_email
|
||||||
|
exp_annotations:
|
||||||
|
summary: "an account's email-code sign-in locked after 10 wrong codes"
|
||||||
|
description: >-
|
||||||
|
Someone entered 10 wrong codes for one account within 24h (login_email).
|
||||||
|
The audit log names the account (action auth.otp.locked); the owner was
|
||||||
|
mailed. Unless they fumbled codes, someone is guessing at it
|
||||||
|
(troubleshooting §17).
|
||||||
|
- eval_time: 90m
|
||||||
|
alertname: FelisOTPAccountLocked
|
||||||
|
exp_alerts: []
|
||||||
|
- name: sign-in failure rate
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
# Two doors failing at 3/min between them from t=0.
|
||||||
|
- series: 'felis_auth_failures_total{door="login_email",reason="bad_code",job="felis-api"}'
|
||||||
|
values: '0+2x40'
|
||||||
|
- series: 'felis_auth_failures_total{door="op_login",reason="no_account",job="felis-api"}'
|
||||||
|
values: '0+1x40'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 8m
|
||||||
|
alertname: FelisSignInFailures
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 25m
|
||||||
|
alertname: FelisSignInFailures
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
exp_annotations:
|
||||||
|
summary: "over 30 refused sign-ins in 15 minutes"
|
||||||
|
description: >-
|
||||||
|
Wrong codes, unknown addresses or bad passkey assertions well above people
|
||||||
|
mistyping: someone is guessing or enumerating. `sum by (door, reason)
|
||||||
|
(increase(felis_auth_failures_total[15m]))` shows where; the audit rows
|
||||||
|
(action auth.<door>.failed) carry each caller's client_ip (troubleshooting §17).
|
||||||
|
- name: people mistyping stays quiet
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_auth_failures_total{door="login_email",reason="bad_code",job="felis-api"}'
|
||||||
|
values: '0 0 1 1 2 2 3 3 4 4 5x30'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 30m
|
||||||
|
alertname: FelisSignInFailures
|
||||||
|
exp_alerts: []
|
||||||
|
- name: audit rows lost
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_audit_write_failures_total{job="felis-api",instance="api-0"}'
|
||||||
|
values: '0 0 0 2x20'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 2m
|
||||||
|
alertname: FelisAuditWriteFailing
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisAuditWriteFailing
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
job: felis-api
|
||||||
|
instance: api-0
|
||||||
|
exp_annotations:
|
||||||
|
summary: "felis-api failed to write audit rows"
|
||||||
|
description: >-
|
||||||
|
The actions went through but their audit rows were lost. The felis-api log
|
||||||
|
names each lost row (`audit: lost ...`); the usual cause is PostgreSQL
|
||||||
|
being unreachable or out of disk.
|
||||||
|
- name: operator and api presence
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_build_info{component="operator",version="v1",job="felis-operator",instance="op-0"}'
|
||||||
|
values: '1x30'
|
||||||
|
# felis-api stops being scraped after 5m; the series goes stale 5m later.
|
||||||
|
- series: 'felis_build_info{component="api",version="v1",job="felis-api",instance="api-0"}'
|
||||||
|
values: '1x5'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 25m
|
||||||
|
alertname: FelisOperatorDown
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 15m
|
||||||
|
alertname: FelisAPIDown
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 25m
|
||||||
|
alertname: FelisAPIDown
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: critical
|
||||||
|
component: api
|
||||||
|
exp_annotations:
|
||||||
|
summary: "felis-api is down or not scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_build_info{component="api"} series for 10 minutes. The panel,
|
||||||
|
sign-in and the proxy's player lookups all go through felis-api. Check
|
||||||
|
`kubectl -n felis get deploy felis-api` and its log; if the pod is
|
||||||
|
healthy, the felis-api-internal Service is not being scraped
|
||||||
|
(troubleshooting §14).
|
||||||
|
- name: operator never scraped
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_build_info{component="api",version="v1",job="felis-api",instance="api-0"}'
|
||||||
|
values: '1x30'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisOperatorDown
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 15m
|
||||||
|
alertname: FelisOperatorDown
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: critical
|
||||||
|
component: operator
|
||||||
|
exp_annotations:
|
||||||
|
summary: "felis-operator is down or not scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_build_info{component="operator"} series for 10 minutes. Without
|
||||||
|
the operator no server starts, stops or recovers. Check
|
||||||
|
`kubectl -n felis get deploy felis-operator` and its log; if the pod is
|
||||||
|
healthy, the felis-operator-metrics Service is not being scraped
|
||||||
|
(troubleshooting §14).
|
||||||
|
- name: system and user servers down
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'felis_server_phase{server="login",role="login",phase="Starting",desired="Running"}'
|
||||||
|
values: '1x20'
|
||||||
|
- series: 'felis_server_phase{server="lobby",role="lobby",phase="Failed",desired="Running"}'
|
||||||
|
values: '1x20'
|
||||||
|
# Stopped on purpose: not an outage.
|
||||||
|
- series: 'felis_server_phase{server="lobby2",role="lobby",phase="Stopped",desired="Stopped"}'
|
||||||
|
values: '1x20'
|
||||||
|
# A user server carries no role label (the operator publishes role="").
|
||||||
|
- series: 'felis_server_phase{server="survival",phase="Failed",desired="Running"}'
|
||||||
|
values: '1x20'
|
||||||
|
- series: 'felis_server_phase{server="creative",phase="Running",desired="Running"}'
|
||||||
|
values: '1x20'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 9m
|
||||||
|
alertname: FelisLoginGateDown
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 11m
|
||||||
|
alertname: FelisLoginGateDown
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: critical
|
||||||
|
server: login
|
||||||
|
role: login
|
||||||
|
phase: Starting
|
||||||
|
desired: Running
|
||||||
|
exp_annotations:
|
||||||
|
summary: "the login gate login is Starting"
|
||||||
|
description: >-
|
||||||
|
Every player connection passes through the login server first, so no one
|
||||||
|
can join. The MinecraftServer's conditions carry the reason:
|
||||||
|
`kubectl -n minecraft describe minecraftserver login`
|
||||||
|
(troubleshooting §1, §2).
|
||||||
|
- eval_time: 11m
|
||||||
|
alertname: FelisSystemServerDown
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
server: lobby
|
||||||
|
role: lobby
|
||||||
|
phase: Failed
|
||||||
|
desired: Running
|
||||||
|
exp_annotations:
|
||||||
|
summary: "system server lobby (lobby) is Failed"
|
||||||
|
description: >-
|
||||||
|
Players who sign in are sent to the lobby; while it is down they stay at
|
||||||
|
the gate. `kubectl -n minecraft describe minecraftserver lobby`
|
||||||
|
shows the reason (troubleshooting §1, §2).
|
||||||
|
- eval_time: 3m
|
||||||
|
alertname: FelisServerFailed
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 6m
|
||||||
|
alertname: FelisServerFailed
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
server: survival
|
||||||
|
phase: Failed
|
||||||
|
desired: Running
|
||||||
|
exp_annotations:
|
||||||
|
summary: "server survival is Failed"
|
||||||
|
description: >-
|
||||||
|
The operator gave up on this server (a crash loop, an image that will not
|
||||||
|
pull, a world volume that will not mount). Its conditions carry the
|
||||||
|
reason: `kubectl -n minecraft describe minecraftserver survival`
|
||||||
|
(troubleshooting §2).
|
||||||
|
- name: operator reconcile errors and a stuck pass
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
# Two failed reconciles a minute from 6m on.
|
||||||
|
- series: 'controller_runtime_reconcile_errors_total{controller="minecraftserver",job="felis-operator"}'
|
||||||
|
values: '0x5 0+2x20'
|
||||||
|
# One pass that started at 5m and never returns.
|
||||||
|
- series: 'workqueue_longest_running_processor_seconds{name="minecraftserver",controller="minecraftserver",job="felis-operator"}'
|
||||||
|
values: '0x5 60+60x20'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisReconcileErrors
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 25m
|
||||||
|
alertname: FelisReconcileErrors
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
exp_annotations:
|
||||||
|
summary: "the operator failed over 10 reconciles in 15 minutes"
|
||||||
|
description: >-
|
||||||
|
Server changes are being retried instead of applied. The felis-operator
|
||||||
|
log names each failing server and its error.
|
||||||
|
- eval_time: 12m
|
||||||
|
alertname: FelisReconcileStuck
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 20m
|
||||||
|
alertname: FelisReconcileStuck
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: critical
|
||||||
|
exp_annotations:
|
||||||
|
summary: "an operator reconcile has been running for over 5 minutes"
|
||||||
|
description: >-
|
||||||
|
Each reconcile is bounded at 3 minutes, so this one is ignoring its
|
||||||
|
deadline and holding a worker. The liveness probe restarts the operator
|
||||||
|
once a pass passes 10 minutes; the log from before the restart shows
|
||||||
|
where it hung.
|
||||||
|
- name: world job failures and a late reaper
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'kube_job_failed{namespace="minecraft",job_name="backup-survival-abc",condition="true"}'
|
||||||
|
values: '0x2 1x10'
|
||||||
|
- series: 'kube_job_failed{namespace="minecraft",job_name="backup-survival-abc",condition="false"}'
|
||||||
|
values: '1x2 0x10'
|
||||||
|
# A build Job in another namespace is FelisImageBuildFailures' business.
|
||||||
|
- series: 'kube_job_failed{namespace="felis-build",job_name="build-x",condition="true"}'
|
||||||
|
values: '1x12'
|
||||||
|
# Evaluation starts at the epoch, so "over a day ago" is a negative timestamp.
|
||||||
|
- series: 'kube_cronjob_status_last_successful_time{namespace="minecraft",cronjob="felis-reaper"}'
|
||||||
|
values: '-100000x30'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 1m
|
||||||
|
alertname: FelisWorldJobFailed
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisWorldJobFailed
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
namespace: minecraft
|
||||||
|
job_name: backup-survival-abc
|
||||||
|
condition: "true"
|
||||||
|
exp_annotations:
|
||||||
|
summary: "Job backup-survival-abc failed"
|
||||||
|
description: >-
|
||||||
|
A world backup, restore or reaper run failed; after a failed backup that
|
||||||
|
world's newest archive is older than planned.
|
||||||
|
`kubectl -n minecraft logs job/backup-survival-abc` has the error
|
||||||
|
(troubleshooting §10).
|
||||||
|
- eval_time: 5m
|
||||||
|
alertname: FelisReaperStale
|
||||||
|
exp_alerts: []
|
||||||
|
- eval_time: 15m
|
||||||
|
alertname: FelisReaperStale
|
||||||
|
exp_alerts:
|
||||||
|
- exp_labels:
|
||||||
|
severity: warning
|
||||||
|
namespace: minecraft
|
||||||
|
cronjob: felis-reaper
|
||||||
|
exp_annotations:
|
||||||
|
summary: "the world reaper has not succeeded in over 26h"
|
||||||
|
description: >-
|
||||||
|
felis-reaper runs daily; idle worlds are neither backed up nor reclaimed
|
||||||
|
while it fails. `kubectl -n minecraft get jobs --sort-by=.metadata.creationTimestamp`
|
||||||
|
lists its runs, and the newest one's log
|
||||||
|
shows why (troubleshooting §10).
|
||||||
|
- name: a reaper that ran yesterday stays quiet
|
||||||
|
interval: 1m
|
||||||
|
input_series:
|
||||||
|
- series: 'kube_cronjob_status_last_successful_time{namespace="minecraft",cronjob="felis-reaper"}'
|
||||||
|
values: '-50000x30'
|
||||||
|
alert_rule_test:
|
||||||
|
- eval_time: 25m
|
||||||
|
alertname: FelisReaperStale
|
||||||
|
exp_alerts: []
|
||||||
@@ -37,6 +37,116 @@ spec:
|
|||||||
observed when readiness is first reached). A start that never completes
|
observed when readiness is first reached). A start that never completes
|
||||||
records nothing — cross-check desiredState=Running servers with no ready
|
records nothing — cross-check desiredState=Running servers with no ready
|
||||||
phase (troubleshooting §1).
|
phase (troubleshooting §1).
|
||||||
|
- name: felis.platform.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisOperatorDown
|
||||||
|
expr: absent(felis_build_info{component="operator"})
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "felis-operator is down or not scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_build_info{component="operator"} series for 10 minutes. Without
|
||||||
|
the operator no server starts, stops or recovers. Check
|
||||||
|
`kubectl -n felis get deploy felis-operator` and its log; if the pod is
|
||||||
|
healthy, the felis-operator-metrics Service is not being scraped
|
||||||
|
(troubleshooting §14).
|
||||||
|
- alert: FelisAPIDown
|
||||||
|
expr: absent(felis_build_info{component="api"})
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "felis-api is down or not scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_build_info{component="api"} series for 10 minutes. The panel,
|
||||||
|
sign-in and the proxy's player lookups all go through felis-api. Check
|
||||||
|
`kubectl -n felis get deploy felis-api` and its log; if the pod is
|
||||||
|
healthy, the felis-api-internal Service is not being scraped
|
||||||
|
(troubleshooting §14).
|
||||||
|
- alert: FelisLoginGateDown
|
||||||
|
expr: felis_server_phase{role="login",desired="Running",phase!="Running"} == 1
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "the login gate {{ $labels.server }} is {{ $labels.phase }}"
|
||||||
|
description: >-
|
||||||
|
Every player connection passes through the login server first, so no one
|
||||||
|
can join. The MinecraftServer's conditions carry the reason:
|
||||||
|
`kubectl -n minecraft describe minecraftserver {{ $labels.server }}`
|
||||||
|
(troubleshooting §1, §2).
|
||||||
|
- alert: FelisSystemServerDown
|
||||||
|
expr: felis_server_phase{role!="",role!="login",desired="Running",phase!="Running"} == 1
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "system server {{ $labels.server }} ({{ $labels.role }}) is {{ $labels.phase }}"
|
||||||
|
description: >-
|
||||||
|
Players who sign in are sent to the lobby; while it is down they stay at
|
||||||
|
the gate. `kubectl -n minecraft describe minecraftserver {{ $labels.server }}`
|
||||||
|
shows the reason (troubleshooting §1, §2).
|
||||||
|
- alert: FelisServerFailed
|
||||||
|
expr: felis_server_phase{role="",phase="Failed"} == 1
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "server {{ $labels.server }} is Failed"
|
||||||
|
description: >-
|
||||||
|
The operator gave up on this server (a crash loop, an image that will not
|
||||||
|
pull, a world volume that will not mount). Its conditions carry the
|
||||||
|
reason: `kubectl -n minecraft describe minecraftserver {{ $labels.server }}`
|
||||||
|
(troubleshooting §2).
|
||||||
|
- alert: FelisReconcileErrors
|
||||||
|
expr: sum(increase(controller_runtime_reconcile_errors_total{controller="minecraftserver"}[15m])) > 10
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the operator failed over 10 reconciles in 15 minutes"
|
||||||
|
description: >-
|
||||||
|
Server changes are being retried instead of applied. The felis-operator
|
||||||
|
log names each failing server and its error.
|
||||||
|
- alert: FelisReconcileStuck
|
||||||
|
expr: max(workqueue_longest_running_processor_seconds{name="minecraftserver"}) > 300
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "an operator reconcile has been running for over 5 minutes"
|
||||||
|
description: >-
|
||||||
|
Each reconcile is bounded at 3 minutes, so this one is ignoring its
|
||||||
|
deadline and holding a worker. The liveness probe restarts the operator
|
||||||
|
once a pass passes 10 minutes; the log from before the restart shows
|
||||||
|
where it hung.
|
||||||
|
- name: felis.jobs.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisWorldJobFailed
|
||||||
|
expr: kube_job_failed{namespace="minecraft",condition="true"} == 1
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Job {{ $labels.job_name }} failed"
|
||||||
|
description: >-
|
||||||
|
A world backup, restore or reaper run failed; after a failed backup that
|
||||||
|
world's newest archive is older than planned.
|
||||||
|
`kubectl -n minecraft logs job/{{ $labels.job_name }}` has the error
|
||||||
|
(troubleshooting §10).
|
||||||
|
- alert: FelisReaperStale
|
||||||
|
expr: time() - kube_cronjob_status_last_successful_time{namespace="minecraft",cronjob="felis-reaper"} > 26 * 3600
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the world reaper has not succeeded in over 26h"
|
||||||
|
description: >-
|
||||||
|
felis-reaper runs daily; idle worlds are neither backed up nor reclaimed
|
||||||
|
while it fails. `kubectl -n minecraft get jobs --sort-by=.metadata.creationTimestamp`
|
||||||
|
lists its runs, and the newest one's log
|
||||||
|
shows why (troubleshooting §10).
|
||||||
- name: felis.node.rules
|
- name: felis.node.rules
|
||||||
rules:
|
rules:
|
||||||
- alert: FelisNodeDiskSpaceLow
|
- alert: FelisNodeDiskSpaceLow
|
||||||
@@ -72,3 +182,97 @@ spec:
|
|||||||
description: >-
|
description: >-
|
||||||
PostgreSQL, the control plane, the registry and game servers share one
|
PostgreSQL, the control plane, the registry and game servers share one
|
||||||
node; sustained memory pressure risks OOM kills.
|
node; sustained memory pressure risks OOM kills.
|
||||||
|
- name: felis.backup.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisDBBackupStale
|
||||||
|
expr: time() - max(felis_db_backup_last_success_timestamp_seconds) > 26 * 3600
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: "no control-plane database backup in over 26h"
|
||||||
|
description: >-
|
||||||
|
felis-db-backup.timer runs daily; the newest bundle is more than a day
|
||||||
|
old. Read `journalctl -u felis-db-backup` on the host, then take one now
|
||||||
|
with `sudo felis db backup` (troubleshooting §16).
|
||||||
|
- alert: FelisDBBackupMetricMissing
|
||||||
|
expr: absent(felis_db_backup_last_success_timestamp_seconds)
|
||||||
|
for: 2h
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "database backup freshness is not being scraped"
|
||||||
|
description: >-
|
||||||
|
No felis_db_backup_last_success_timestamp_seconds series, so
|
||||||
|
FelisDBBackupStale cannot fire. Point node-exporter's
|
||||||
|
--collector.textfile.directory at the directory of
|
||||||
|
FELIS_DB_BACKUP_METRICS (default /var/lib/node_exporter/textfile_collector)
|
||||||
|
(troubleshooting §16).
|
||||||
|
- name: felis.auth.rules
|
||||||
|
rules:
|
||||||
|
- alert: FelisMailBudgetExhausted
|
||||||
|
expr: sum(increase(felis_mail_total{result="throttled"}[15m])) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the install-wide mail budget refused mail"
|
||||||
|
description: >-
|
||||||
|
felis_mail_total{result="throttled"} increased: [smtp] max_per_hour is
|
||||||
|
spent, and every sign-in code is refused with 429 mail_rate_limited until
|
||||||
|
it refills. Check felis_rate_limited_total for a flood before raising the
|
||||||
|
budget (troubleshooting §17).
|
||||||
|
- alert: FelisMailDeliveryFailing
|
||||||
|
expr: sum(increase(felis_mail_total{result="failed"}[15m])) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "the SMTP relay refused mail in the last 15m"
|
||||||
|
description: >-
|
||||||
|
felis_mail_total{result="failed"} increased: sign-in codes are not being
|
||||||
|
delivered (502 mail_undeliverable). The relay's reason is in the
|
||||||
|
felis-api log (troubleshooting §17).
|
||||||
|
- alert: FelisSignInFlood
|
||||||
|
expr: sum(rate(felis_rate_limited_total{scope="auth_door"}[5m])) * 60 > 10
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "sign-in doors refusing over 10 requests a minute"
|
||||||
|
description: >-
|
||||||
|
The per-address sign-in limit has been refusing callers for 10 minutes.
|
||||||
|
A script is hammering the auth doors; if real users report rate_limited
|
||||||
|
at once instead, [auth] client_ip_header is missing and everyone shares
|
||||||
|
the proxy's address (troubleshooting §17).
|
||||||
|
- alert: FelisOTPAccountLocked
|
||||||
|
expr: sum by (purpose) (increase(felis_auth_otp_lockouts_total[1h])) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "an account's email-code sign-in locked after 10 wrong codes"
|
||||||
|
description: >-
|
||||||
|
Someone entered 10 wrong codes for one account within 24h ({{ $labels.purpose }}).
|
||||||
|
The audit log names the account (action auth.otp.locked); the owner was
|
||||||
|
mailed. Unless they fumbled codes, someone is guessing at it
|
||||||
|
(troubleshooting §17).
|
||||||
|
- alert: FelisSignInFailures
|
||||||
|
expr: sum(increase(felis_auth_failures_total[15m])) > 30
|
||||||
|
for: 5m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "over 30 refused sign-ins in 15 minutes"
|
||||||
|
description: >-
|
||||||
|
Wrong codes, unknown addresses or bad passkey assertions well above people
|
||||||
|
mistyping: someone is guessing or enumerating. `sum by (door, reason)
|
||||||
|
(increase(felis_auth_failures_total[15m]))` shows where; the audit rows
|
||||||
|
(action auth.<door>.failed) carry each caller's client_ip (troubleshooting §17).
|
||||||
|
- alert: FelisAuditWriteFailing
|
||||||
|
expr: increase(felis_audit_write_failures_total[15m]) > 0
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "felis-api failed to write audit rows"
|
||||||
|
description: >-
|
||||||
|
The actions went through but their audit rows were lost. The felis-api log
|
||||||
|
names each lost row (`audit: lost ...`); the usual cause is PostgreSQL
|
||||||
|
being unreachable or out of disk.
|
||||||
+1229
-125
File diff suppressed because it is too large.
Load diff
+1014
-39
File diff suppressed because it is too large.
Load diff
@@ -123,10 +123,6 @@ spec:
|
|||||||
lifecycle:
|
lifecycle:
|
||||||
description: Lifecycle tunes graceful shutdown (spec §7).
|
description: Lifecycle tunes graceful shutdown (spec §7).
|
||||||
properties:
|
properties:
|
||||||
preStopSaveAndStop:
|
|
||||||
description: PreStopSaveAndStop enables the operator-injected
|
|
||||||
RCON save+stop preStop.
|
|
||||||
type: boolean
|
|
||||||
terminationGracePeriodSeconds:
|
terminationGracePeriodSeconds:
|
||||||
description: TerminationGracePeriodSeconds is the pod grace period
|
description: TerminationGracePeriodSeconds is the pod grace period
|
||||||
(default 300).
|
(default 300).
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
# The upstream builds this release installs. deploy/bootstrap.sh reads this file (the
|
||||||
|
# default FELIS_GAME_STACK=pinned), downloads exactly these artifacts and refuses any whose
|
||||||
|
# sha256 differs, so every host installing one release gets the same login gate, lobby,
|
||||||
|
# plain-Paper image and proxy, and a rerun rebuilds nothing that did not change.
|
||||||
|
#
|
||||||
|
# MC_VERSION is the protocol the whole stack speaks: Limbo speaks exactly one, and Paper
|
||||||
|
# follows it so a client that passes the login gate can also reach the lobby.
|
||||||
|
#
|
||||||
|
# Refresh with deploy/update-game-stack-lock.sh, which resolves upstream's newest builds and
|
||||||
|
# hashes them. Plain KEY=value lines only; bootstrap reads it without evaluating it.
|
||||||
|
MC_VERSION=26.3
|
||||||
|
LIMBO_VERSION=2026.0.3-ALPHA
|
||||||
|
LIMBO_JAR_URL=https://ci.loohpjames.com/job/Limbo/76/artifact/target/Limbo-2026.0.3-ALPHA-26.3.jar
|
||||||
|
LIMBO_JAR_SHA256=a2de91fcaa2255aed8a111b8786f370213423c7798eee778c00d016d83aa65d2
|
||||||
|
LIMBO_SCHEM_URL=https://ci.loohpjames.com/job/Limbo/76/artifact/spawn.schem
|
||||||
|
LIMBO_SCHEM_SHA256=70c85dae2db157971ef513e318820c5a5e10a96b813b70e21f8af2b632cbcbc7
|
||||||
|
PAPER_JAR_URL=https://fill-data.papermc.io/v1/objects/49399919246cbf443efc8507447dc948eb7477c41be560b0e87e2a455aff824a/paper-26.3-40.jar
|
||||||
|
PAPER_JAR_SHA256=49399919246cbf443efc8507447dc948eb7477c41be560b0e87e2a455aff824a
|
||||||
|
LUCKPERMS_JAR_URL=https://download.luckperms.net/1672/bukkit/loader/LuckPerms-Bukkit-5.5.85.jar
|
||||||
|
LUCKPERMS_JAR_SHA256=dc637ce18f48d3b75a7ffd1784b85be16a627359090dfe5adf15dab6d145dc7d
|
||||||
|
VELOCITY_VERSION=3.5.1
|
||||||
|
VELOCITY_JAR_URL=https://fill-data.papermc.io/v1/objects/b4e3164df5377346854dc6cb9e6a78022b1946ff69e89676313f5f6f1c6f0fb3/velocity-3.5.1-615.jar
|
||||||
|
VELOCITY_JAR_SHA256=b4e3164df5377346854dc6cb9e6a78022b1946ff69e89676313f5f6f1c6f0fb3
|
||||||
+27
-7
@@ -10,10 +10,11 @@
|
|||||||
# There is no bundled server.properties — Limbo writes a default on first run.
|
# There is no bundled server.properties — Limbo writes a default on first run.
|
||||||
# So the runtime is assembled from those two URLs (not a zip) via --build-arg:
|
# So the runtime is assembled from those two URLs (not a zip) via --build-arg:
|
||||||
#
|
#
|
||||||
|
# . <(grep '^LIMBO_' deploy/game-stack.lock)
|
||||||
# docker build -f deploy/limbo/Dockerfile \
|
# docker build -f deploy/limbo/Dockerfile \
|
||||||
# --build-arg LIMBO_JAR_URL=https://ci.loohpjames.com/job/Limbo/<n>/artifact/target/Limbo-<ver>.jar \
|
# --build-arg LIMBO_JAR_URL="$LIMBO_JAR_URL" --build-arg LIMBO_JAR_SHA256="$LIMBO_JAR_SHA256" \
|
||||||
# --build-arg LIMBO_SCHEM_URL=https://ci.loohpjames.com/job/Limbo/<n>/artifact/spawn.schem \
|
# --build-arg LIMBO_SCHEM_URL="$LIMBO_SCHEM_URL" --build-arg LIMBO_SCHEM_SHA256="$LIMBO_SCHEM_SHA256" \
|
||||||
# --build-arg LIMBO_VERSION=<maven-api-version> \
|
# --build-arg LIMBO_VERSION="$LIMBO_VERSION" \
|
||||||
# -t felis-limbo:demo .
|
# -t felis-limbo:demo .
|
||||||
#
|
#
|
||||||
# Note LIMBO_VERSION (the maven API version the plugin compiles against, e.g.
|
# Note LIMBO_VERSION (the maven API version the plugin compiles against, e.g.
|
||||||
@@ -33,7 +34,7 @@
|
|||||||
# JDK 17 fails to read them with "wrong version 65.0, should be 61.0". The image
|
# JDK 17 fails to read them with "wrong version 65.0, should be 61.0". The image
|
||||||
# also provides the `gradle` binary (this tree vendors no Gradle wrapper).
|
# also provides the `gradle` binary (this tree vendors no Gradle wrapper).
|
||||||
# build.gradle still targets release 17 bytecode so the plugin loads on Java 17+.
|
# build.gradle still targets release 17 bytecode so the plugin loads on Java 17+.
|
||||||
FROM gradle:8.14-jdk21 AS plugin
|
FROM gradle:8.14-jdk21@sha256:5c4c0c4284de4a19951e82ac78f86dbcda2e136644bbfe159beba7ea3420cc80 AS plugin
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
# Copy what the limbo module needs: its own tree plus the shared link core it
|
# Copy what the limbo module needs: its own tree plus the shared link core it
|
||||||
# srcDir-includes (../shared/src/main/java → /src/plugins/shared/src/main/java), so
|
# srcDir-includes (../shared/src/main/java → /src/plugins/shared/src/main/java), so
|
||||||
@@ -50,21 +51,33 @@ RUN cd plugins/limbo \
|
|||||||
# 21-jre: the Limbo jar is Java 21 bytecode (class-file major 65), so a Java 17
|
# 21-jre: the Limbo jar is Java 21 bytecode (class-file major 65), so a Java 17
|
||||||
# JRE cannot run it (UnsupportedClassVersionError). A 21 JRE also runs the
|
# JRE cannot run it (UnsupportedClassVersionError). A 21 JRE also runs the
|
||||||
# plugin's release-17 bytecode fine.
|
# plugin's release-17 bytecode fine.
|
||||||
FROM eclipse-temurin:21-jre
|
FROM eclipse-temurin:21-jre@sha256:49e21e16e3c86eb7816a44a67549910ed090fbeb40c29c525d58bf5e02e91b0f
|
||||||
ARG LIMBO_JAR_URL
|
ARG LIMBO_JAR_URL
|
||||||
|
ARG LIMBO_JAR_SHA256
|
||||||
ARG LIMBO_SCHEM_URL
|
ARG LIMBO_SCHEM_URL
|
||||||
|
ARG LIMBO_SCHEM_SHA256
|
||||||
WORKDIR /limbo
|
WORKDIR /limbo
|
||||||
# Pull the two loose LOOHP/Limbo CI artifacts: the server jar (required, saved as
|
# Pull the two loose LOOHP/Limbo CI artifacts: the server jar (required, saved as
|
||||||
# Limbo.jar) and the default spawn schematic (optional). Fail loudly if the jar
|
# Limbo.jar) and the default spawn schematic (optional). Each is checked against the
|
||||||
# URL was not supplied.
|
# digest deploy/game-stack.lock names (bootstrap.sh passes it): the login gate is the
|
||||||
|
# first thing every player's connection reaches, and Limbo's CI publishes no digest of
|
||||||
|
# its own.
|
||||||
RUN set -eu; \
|
RUN set -eu; \
|
||||||
if [ -z "${LIMBO_JAR_URL:-}" ]; then \
|
if [ -z "${LIMBO_JAR_URL:-}" ]; then \
|
||||||
echo "ERROR: --build-arg LIMBO_JAR_URL=<Limbo server jar> is required" >&2; exit 1; \
|
echo "ERROR: --build-arg LIMBO_JAR_URL=<Limbo server jar> is required" >&2; exit 1; \
|
||||||
fi; \
|
fi; \
|
||||||
|
if [ -z "${LIMBO_JAR_SHA256:-}" ]; then \
|
||||||
|
echo "ERROR: --build-arg LIMBO_JAR_SHA256=<Limbo jar sha256> is required" >&2; exit 1; \
|
||||||
|
fi; \
|
||||||
|
if [ -n "${LIMBO_SCHEM_URL:-}" ] && [ -z "${LIMBO_SCHEM_SHA256:-}" ]; then \
|
||||||
|
echo "ERROR: --build-arg LIMBO_SCHEM_SHA256=<spawn.schem sha256> is required with LIMBO_SCHEM_URL" >&2; exit 1; \
|
||||||
|
fi; \
|
||||||
apt-get update && apt-get install -y --no-install-recommends curl ca-certificates; \
|
apt-get update && apt-get install -y --no-install-recommends curl ca-certificates; \
|
||||||
curl -fSL "$LIMBO_JAR_URL" -o /limbo/Limbo.jar; \
|
curl -fSL "$LIMBO_JAR_URL" -o /limbo/Limbo.jar; \
|
||||||
|
echo "$LIMBO_JAR_SHA256 /limbo/Limbo.jar" | sha256sum -c; \
|
||||||
if [ -n "${LIMBO_SCHEM_URL:-}" ]; then \
|
if [ -n "${LIMBO_SCHEM_URL:-}" ]; then \
|
||||||
curl -fSL "$LIMBO_SCHEM_URL" -o /limbo/spawn.schem; \
|
curl -fSL "$LIMBO_SCHEM_URL" -o /limbo/spawn.schem; \
|
||||||
|
echo "$LIMBO_SCHEM_SHA256 /limbo/spawn.schem" | sha256sum -c; \
|
||||||
fi; \
|
fi; \
|
||||||
apt-get purge -y curl && apt-get autoremove -y && rm -rf /var/lib/apt/lists/*; \
|
apt-get purge -y curl && apt-get autoremove -y && rm -rf /var/lib/apt/lists/*; \
|
||||||
mkdir -p /limbo/plugins
|
mkdir -p /limbo/plugins
|
||||||
@@ -78,6 +91,13 @@ COPY deploy/limbo/entrypoint.sh /usr/local/bin/felis-entrypoint.sh
|
|||||||
# The operator mounts the world PVC at /data. Runtime state lives there; /limbo
|
# The operator mounts the world PVC at /data. Runtime state lives there; /limbo
|
||||||
# remains the immutable image seed copied into the volume by the entrypoint.
|
# remains the immutable image seed copied into the volume by the entrypoint.
|
||||||
WORKDIR /data
|
WORKDIR /data
|
||||||
|
# Run as the game uid (naming.GameUID in the Go tree). The operator pins the same uid in
|
||||||
|
# the pod securityContext whatever USER an image declares; declaring it here as well
|
||||||
|
# keeps a plain `docker run` of this image off root, and chowning the empty /data seed
|
||||||
|
# lets that run write its world. The jar seed above stays root-owned and read-only to
|
||||||
|
# the server.
|
||||||
|
RUN chown 1000:1000 /data
|
||||||
|
USER 1000:1000
|
||||||
|
|
||||||
ENV FELIS_HEALTH_PORT=8080
|
ENV FELIS_HEALTH_PORT=8080
|
||||||
# FELIS_GAME_PORT is the port the entrypoint pins Limbo to; it MUST equal the operator's
|
# FELIS_GAME_PORT is the port the entrypoint pins Limbo to; it MUST equal the operator's
|
||||||
|
|||||||
@@ -143,11 +143,14 @@ set them by hand:
|
|||||||
pod's internal port 8081. That Service is deliberately separate from the external
|
pod's internal port 8081. That Service is deliberately separate from the external
|
||||||
NodePort `felis-api` (443) so the no-Zero-Trust internal face is never published on
|
NodePort `felis-api` (443) so the no-Zero-Trust internal face is never published on
|
||||||
a node's external IP.
|
a node's external IP.
|
||||||
- **NetworkPolicy:** none is required today — neither the minecraft-namespace egress
|
- **NetworkPolicy:** the minecraft namespace is egress-locked
|
||||||
nor the control-namespace ingress is policy-locked, so the login pod's call to the
|
(`felis-server-egress`: DNS plus the public internet, every private range
|
||||||
API internal port is reachable. If a future deployment adds a minecraft egress lock
|
excluded), so the internal API is unreachable from a game server by default.
|
||||||
or a control-namespace ingress fence, it must also open the login-pod →
|
`felis-login-to-internal-api` opens exactly the login pod → felis-api (8081) path,
|
||||||
felis-api-internal (8081) path.
|
selecting on the reserved `login` name AND the setup-owned
|
||||||
|
`felis.lolicon.best/system-role=login` label the operator copies onto the pod — the
|
||||||
|
same pair that decides who receives `FELIS_SERVICE_TOKEN`, so a user server cannot
|
||||||
|
match it by picking a name.
|
||||||
|
|
||||||
The Velocity gate/lobby wiring is printed by `felis setup` and enforces the
|
The Velocity gate/lobby wiring is printed by `felis setup` and enforces the
|
||||||
invariant: fresh connections hit `login` first, and only an authenticated release
|
invariant: fresh connections hit `login` first, and only an authenticated release
|
||||||
|
|||||||
+24
-11
@@ -5,13 +5,13 @@
|
|||||||
# the POST-auth /menu hub: it is reached only when the login gate transfers an
|
# the POST-auth /menu hub: it is reached only when the login gate transfers an
|
||||||
# authenticated player onward, and it must never be a fallback target.
|
# authenticated player onward, and it must never be a fallback target.
|
||||||
#
|
#
|
||||||
# Build (deploy/bootstrap.sh does this for you; the PAPER_JAR_URL comes from PaperMC's
|
# Build (deploy/bootstrap.sh does this for you, with the URLs and digests
|
||||||
# Fill v3 API — api.papermc.io v2 has returned HTTP 410 since 2026-07-01):
|
# deploy/game-stack.lock names):
|
||||||
|
# . <(grep -E '^(PAPER|LUCKPERMS)_' deploy/game-stack.lock)
|
||||||
# docker build -f deploy/lobby/Dockerfile \
|
# docker build -f deploy/lobby/Dockerfile \
|
||||||
# --build-arg PAPER_JAR_URL=https://fill-data.papermc.io/v1/objects/<sha>/paper-26.2-<build>.jar \
|
# --build-arg PAPER_JAR_URL="$PAPER_JAR_URL" --build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \
|
||||||
# --build-arg PAPER_JAR_SHA256=<that same sha — the objects/ path segment> \
|
# --build-arg LUCKPERMS_JAR_URL="$LUCKPERMS_JAR_URL" \
|
||||||
# --build-arg LUCKPERMS_JAR_URL="$(curl -fsSL https://metadata.luckperms.net/data/all \
|
# --build-arg LUCKPERMS_JAR_SHA256="$LUCKPERMS_JAR_SHA256" \
|
||||||
# | grep -o 'https://download.luckperms.net/[^"]*/bukkit/loader/[^"]*\.jar')" \
|
|
||||||
# -t felis-lobby:demo .
|
# -t felis-lobby:demo .
|
||||||
# docker save felis-lobby:demo | sudo k3s ctr images import -
|
# docker save felis-lobby:demo | sudo k3s ctr images import -
|
||||||
# # felis.toml → [velocity] lobby_image = "felis-lobby:demo"
|
# # felis.toml → [velocity] lobby_image = "felis-lobby:demo"
|
||||||
@@ -28,7 +28,7 @@
|
|||||||
# ---- build the felis-paper plugin jar (Paper API is Java 21) ----
|
# ---- build the felis-paper plugin jar (Paper API is Java 21) ----
|
||||||
# gradle:8.14-jdk21 — an official Gradle image on JDK 21 (this tree vendors no Gradle
|
# gradle:8.14-jdk21 — an official Gradle image on JDK 21 (this tree vendors no Gradle
|
||||||
# wrapper, and a bare JDK image ships no `gradle`). JDK 21 matches the Paper API.
|
# wrapper, and a bare JDK image ships no `gradle`). JDK 21 matches the Paper API.
|
||||||
FROM gradle:8.14-jdk21 AS plugin
|
FROM gradle:8.14-jdk21@sha256:5c4c0c4284de4a19951e82ac78f86dbcda2e136644bbfe159beba7ea3420cc80 AS plugin
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
COPY plugins/paper/ ./plugins/paper/
|
COPY plugins/paper/ ./plugins/paper/
|
||||||
COPY plugins/shared/ ./plugins/shared/
|
COPY plugins/shared/ ./plugins/shared/
|
||||||
@@ -41,7 +41,7 @@ RUN cd plugins/paper \
|
|||||||
# 25-jre, not 21: Paper 26.2 declares `java.version.minimum = 25` (PaperMC Fill v3,
|
# 25-jre, not 21: Paper 26.2 declares `java.version.minimum = 25` (PaperMC Fill v3,
|
||||||
# GET /v3/projects/paper/versions/26.2) and refuses to boot on anything older. A 25 JRE
|
# GET /v3/projects/paper/versions/26.2) and refuses to boot on anything older. A 25 JRE
|
||||||
# also runs the plugin's Java-21 bytecode, so only the runtime moves.
|
# also runs the plugin's Java-21 bytecode, so only the runtime moves.
|
||||||
FROM eclipse-temurin:25-jre
|
FROM eclipse-temurin:25-jre@sha256:bb036ed6cfdc57e3da7c22634d15f1b840d2caf76183861c80e81ca4b5104abb
|
||||||
ARG PAPER_JAR_URL
|
ARG PAPER_JAR_URL
|
||||||
# Required alongside the URL: Fill's URLs are content-addressed, but nothing enforces
|
# Required alongside the URL: Fill's URLs are content-addressed, but nothing enforces
|
||||||
# that shape at build time. Checking the digest after the download turns a truncated or
|
# that shape at build time. Checking the digest after the download turns a truncated or
|
||||||
@@ -51,10 +51,12 @@ ARG PAPER_JAR_SHA256
|
|||||||
# (internal/api/handlers_access.go) issues `lp user ...` over RCON, so a lobby built
|
# (internal/api/handlers_access.go) issues `lp user ...` over RCON, so a lobby built
|
||||||
# without it answers every grant with "Unknown command" — a failure the operator only
|
# without it answers every grant with "Unknown command" — a failure the operator only
|
||||||
# discovers in production, because the server itself starts and runs perfectly well.
|
# discovers in production, because the server itself starts and runs perfectly well.
|
||||||
# Failing the build is the cheap place to notice. Resolved by URL rather than pinned
|
# Failing the build is the cheap place to notice. Passed in rather than pinned here for
|
||||||
# here for the same reason PAPER_JAR_URL is: bootstrap.sh asks upstream for the current
|
# the same reason PAPER_JAR_URL is: deploy/game-stack.lock names the build, so this file
|
||||||
# build, so this file does not go stale on every LuckPerms release.
|
# does not change on every LuckPerms release. The digest is required like Paper's; the
|
||||||
|
# jar runs inside the lobby with the server's full permissions.
|
||||||
ARG LUCKPERMS_JAR_URL
|
ARG LUCKPERMS_JAR_URL
|
||||||
|
ARG LUCKPERMS_JAR_SHA256
|
||||||
WORKDIR /paper
|
WORKDIR /paper
|
||||||
RUN set -eu; \
|
RUN set -eu; \
|
||||||
if [ -z "${PAPER_JAR_URL:-}" ]; then \
|
if [ -z "${PAPER_JAR_URL:-}" ]; then \
|
||||||
@@ -66,11 +68,15 @@ RUN set -eu; \
|
|||||||
if [ -z "${LUCKPERMS_JAR_URL:-}" ]; then \
|
if [ -z "${LUCKPERMS_JAR_URL:-}" ]; then \
|
||||||
echo "ERROR: --build-arg LUCKPERMS_JAR_URL=<luckperms bukkit jar> is required" >&2; exit 1; \
|
echo "ERROR: --build-arg LUCKPERMS_JAR_URL=<luckperms bukkit jar> is required" >&2; exit 1; \
|
||||||
fi; \
|
fi; \
|
||||||
|
if [ -z "${LUCKPERMS_JAR_SHA256:-}" ]; then \
|
||||||
|
echo "ERROR: --build-arg LUCKPERMS_JAR_SHA256=<luckperms jar sha256> is required" >&2; exit 1; \
|
||||||
|
fi; \
|
||||||
apt-get update && apt-get install -y --no-install-recommends curl ca-certificates; \
|
apt-get update && apt-get install -y --no-install-recommends curl ca-certificates; \
|
||||||
mkdir -p /paper/plugins; \
|
mkdir -p /paper/plugins; \
|
||||||
curl -fSL "$PAPER_JAR_URL" -o /paper/paper.jar; \
|
curl -fSL "$PAPER_JAR_URL" -o /paper/paper.jar; \
|
||||||
echo "$PAPER_JAR_SHA256 /paper/paper.jar" | sha256sum -c; \
|
echo "$PAPER_JAR_SHA256 /paper/paper.jar" | sha256sum -c; \
|
||||||
curl -fSL "$LUCKPERMS_JAR_URL" -o /paper/plugins/LuckPerms.jar; \
|
curl -fSL "$LUCKPERMS_JAR_URL" -o /paper/plugins/LuckPerms.jar; \
|
||||||
|
echo "$LUCKPERMS_JAR_SHA256 /paper/plugins/LuckPerms.jar" | sha256sum -c; \
|
||||||
apt-get purge -y curl && apt-get autoremove -y && rm -rf /var/lib/apt/lists/*; \
|
apt-get purge -y curl && apt-get autoremove -y && rm -rf /var/lib/apt/lists/*; \
|
||||||
echo "eula=true" > /paper/eula.txt
|
echo "eula=true" > /paper/eula.txt
|
||||||
COPY --from=plugin /felis-paper.jar /paper/plugins/felis-paper.jar
|
COPY --from=plugin /felis-paper.jar /paper/plugins/felis-paper.jar
|
||||||
@@ -82,6 +88,13 @@ COPY deploy/lobby/entrypoint.sh /usr/local/bin/felis-entrypoint.sh
|
|||||||
# The operator mounts the world PVC at /data. Runtime state lives there; /paper
|
# The operator mounts the world PVC at /data. Runtime state lives there; /paper
|
||||||
# remains the immutable image seed copied into the volume by the entrypoint.
|
# remains the immutable image seed copied into the volume by the entrypoint.
|
||||||
WORKDIR /data
|
WORKDIR /data
|
||||||
|
# Run as the game uid (naming.GameUID in the Go tree). The operator pins the same uid in
|
||||||
|
# the pod securityContext whatever USER an image declares; declaring it here as well
|
||||||
|
# keeps a plain `docker run` of this image off root, and chowning the empty /data seed
|
||||||
|
# lets that run write its world. The jar seed above stays root-owned and read-only to
|
||||||
|
# the server.
|
||||||
|
RUN chown 1000:1000 /data
|
||||||
|
USER 1000:1000
|
||||||
|
|
||||||
# FELIS_GAME_PORT is the port the entrypoint pins Paper to; it MUST equal the operator's
|
# FELIS_GAME_PORT is the port the entrypoint pins Paper to; it MUST equal the operator's
|
||||||
# GamePort (internal/operator/builders.go). Default 25565 — override only in lockstep
|
# GamePort (internal/operator/builders.go). Default 25565 — override only in lockstep
|
||||||
|
|||||||
+13
-6
@@ -5,7 +5,7 @@
|
|||||||
# internal/store/migrations/0019_recommended_paper.sql). It is NOT a system server: it
|
# internal/store/migrations/0019_recommended_paper.sql). It is NOT a system server: it
|
||||||
# carries no felis-paper /menu plugin, no LuckPerms, and no forwarding-secret gate.
|
# carries no felis-paper /menu plugin, no LuckPerms, and no forwarding-secret gate.
|
||||||
#
|
#
|
||||||
# It writes NO Velocity forwarding config itself. The operator injects a root
|
# It writes NO Velocity forwarding config itself. The operator injects a
|
||||||
# `felis init-forwarding` initContainer into every USER server (internal/operator/
|
# `felis init-forwarding` initContainer into every USER server (internal/operator/
|
||||||
# builders.go: buildStatefulSet) that writes config/paper-global.yml + server.properties
|
# builders.go: buildStatefulSet) that writes config/paper-global.yml + server.properties
|
||||||
# online-mode=false onto the /data PVC before this container starts. That external step is
|
# online-mode=false onto the /data PVC before this container starts. That external step is
|
||||||
@@ -14,11 +14,11 @@
|
|||||||
# image a user brings is made joinable the same way. If the initContainer is absent (no
|
# image a user brings is made joinable the same way. If the initContainer is absent (no
|
||||||
# FELIS_IMAGE configured) Paper boots as a standalone online server: degraded, not broken.
|
# FELIS_IMAGE configured) Paper boots as a standalone online server: degraded, not broken.
|
||||||
#
|
#
|
||||||
# Build (deploy/bootstrap.sh does this for you; PAPER_JAR_URL comes from PaperMC's Fill v3
|
# Build (deploy/bootstrap.sh does this for you, with the SAME Paper build the lobby uses —
|
||||||
# API — the SAME url the lobby build resolves, so this reuses it and adds no new dependency):
|
# deploy/game-stack.lock names it — so this adds no new dependency):
|
||||||
|
# . <(grep '^PAPER_' deploy/game-stack.lock)
|
||||||
# docker build -f deploy/paper/Dockerfile \
|
# docker build -f deploy/paper/Dockerfile \
|
||||||
# --build-arg PAPER_JAR_URL=https://fill-data.papermc.io/v1/objects/<sha>/paper-<ver>-<build>.jar \
|
# --build-arg PAPER_JAR_URL="$PAPER_JAR_URL" --build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \
|
||||||
# --build-arg PAPER_JAR_SHA256=<that same sha — the objects/ path segment> \
|
|
||||||
# -t felis-paper:demo .
|
# -t felis-paper:demo .
|
||||||
# docker save felis-paper:demo | sudo k3s ctr images import -
|
# docker save felis-paper:demo | sudo k3s ctr images import -
|
||||||
# # felis.toml → recommended via 0019_recommended_paper.sql (no [velocity] key points here)
|
# # felis.toml → recommended via 0019_recommended_paper.sql (no [velocity] key points here)
|
||||||
@@ -29,7 +29,7 @@
|
|||||||
|
|
||||||
# 25-jre, not 21: Paper 26.2 declares java.version.minimum=25 (PaperMC Fill v3) and refuses
|
# 25-jre, not 21: Paper 26.2 declares java.version.minimum=25 (PaperMC Fill v3) and refuses
|
||||||
# to boot on anything older.
|
# to boot on anything older.
|
||||||
FROM eclipse-temurin:25-jre
|
FROM eclipse-temurin:25-jre@sha256:bb036ed6cfdc57e3da7c22634d15f1b840d2caf76183861c80e81ca4b5104abb
|
||||||
ARG PAPER_JAR_URL
|
ARG PAPER_JAR_URL
|
||||||
# Required alongside the URL: Fill's URLs are content-addressed, but nothing enforces
|
# Required alongside the URL: Fill's URLs are content-addressed, but nothing enforces
|
||||||
# that shape at build time. Checking the digest after the download turns a truncated or
|
# that shape at build time. Checking the digest after the download turns a truncated or
|
||||||
@@ -54,6 +54,13 @@ COPY deploy/paper/entrypoint.sh /usr/local/bin/felis-entrypoint.sh
|
|||||||
# on the PVC. /paper stays the immutable image seed: the jar is never copied onto the
|
# on the PVC. /paper stays the immutable image seed: the jar is never copied onto the
|
||||||
# volume, so the panel file editor (which sees only /data) cannot tamper with it.
|
# volume, so the panel file editor (which sees only /data) cannot tamper with it.
|
||||||
WORKDIR /data
|
WORKDIR /data
|
||||||
|
# Run as the game uid (naming.GameUID in the Go tree). The operator pins the same uid in
|
||||||
|
# the pod securityContext whatever USER an image declares; declaring it here as well
|
||||||
|
# keeps a plain `docker run` of this image off root, and chowning the empty /data seed
|
||||||
|
# lets that run write its world. The jar seed above stays root-owned and read-only to
|
||||||
|
# the server.
|
||||||
|
RUN chown 1000:1000 /data
|
||||||
|
USER 1000:1000
|
||||||
|
|
||||||
# FELIS_GAME_PORT is the port the entrypoint pins Paper to; it MUST equal the operator's
|
# FELIS_GAME_PORT is the port the entrypoint pins Paper to; it MUST equal the operator's
|
||||||
# GamePort (internal/operator/builders.go). Default 25565 — override only in lockstep with
|
# GamePort (internal/operator/builders.go). Default 25565 — override only in lockstep with
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
#
|
#
|
||||||
# A plain Paper backend for a user's OWN world — NOT a system server. Unlike deploy/limbo
|
# A plain Paper backend for a user's OWN world — NOT a system server. Unlike deploy/limbo
|
||||||
# and deploy/lobby it writes no Velocity forwarding config and has no secret gate: the
|
# and deploy/lobby it writes no Velocity forwarding config and has no secret gate: the
|
||||||
# operator injects a root `felis init-forwarding` initContainer that writes
|
# operator injects a `felis init-forwarding` initContainer that writes
|
||||||
# config/paper-global.yml + server.properties online-mode=false onto /data BEFORE this
|
# config/paper-global.yml + server.properties online-mode=false onto /data BEFORE this
|
||||||
# container starts, so forwarding is configured externally and this stays a drop-in Paper
|
# container starts, so forwarding is configured externally and this stays a drop-in Paper
|
||||||
# image. With no initContainer (no FELIS_IMAGE) Paper just boots standalone-online —
|
# image. With no initContainer (no FELIS_IMAGE) Paper just boots standalone-online —
|
||||||
|
|||||||
Executable
+73
@@ -0,0 +1,73 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Refreshes deploy/game-stack.lock to upstream's newest builds, hashed.
|
||||||
|
#
|
||||||
|
# bash deploy/update-game-stack-lock.sh # rewrite the lock in place
|
||||||
|
# bash deploy/update-game-stack-lock.sh --check # exit 1 if upstream moved on
|
||||||
|
#
|
||||||
|
# It runs the same resolver bootstrap.sh uses for FELIS_GAME_STACK=latest (the functions are
|
||||||
|
# lifted out of bootstrap.sh, so the two cannot drift), then pins Velocity's newest build of
|
||||||
|
# VELOCITY_LATEST_MINOR. Limbo and LuckPerms publish no digest, so their jars are downloaded
|
||||||
|
# and hashed here; Paper and Velocity come from Fill's content-addressed URLs.
|
||||||
|
#
|
||||||
|
# Review the diff before committing: MC_VERSION moves the login gate's protocol, and the
|
||||||
|
# lobby, plain-Paper image and every client follow it.
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
BS="${here}/bootstrap.sh"
|
||||||
|
LOCK="${here}/game-stack.lock"
|
||||||
|
check=0
|
||||||
|
case "${1:-}" in
|
||||||
|
--check) check=1 ;;
|
||||||
|
"") ;;
|
||||||
|
*) printf 'usage: %s [--check]\n' "$0" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
log() { printf '[lock] %s\n' "$*" >&2; }
|
||||||
|
ok() { printf '[ ok ] %s\n' "$*" >&2; }
|
||||||
|
warn() { :; }
|
||||||
|
die() { printf '[fail] %s\n' "$*" >&2; exit 1; }
|
||||||
|
|
||||||
|
lift() { # function-name
|
||||||
|
local body
|
||||||
|
body="$(awk -v f="$1" '$0 ~ "^" f "\\(\\) \\{" {on=1} on {print} on && /^}/ {exit}' "$BS")"
|
||||||
|
[ -n "$body" ] || die "bootstrap.sh no longer defines $1"
|
||||||
|
eval "$body"
|
||||||
|
}
|
||||||
|
for fn in meta_get papermc_latest_jar luckperms_latest_jar url_sha256 resolve_latest_game_jars; do
|
||||||
|
lift "$fn"
|
||||||
|
done
|
||||||
|
eval "$(grep '^VELOCITY_LATEST_MINOR=' "$BS")"
|
||||||
|
[ -n "${VELOCITY_LATEST_MINOR:-}" ] || die "bootstrap.sh no longer sets VELOCITY_LATEST_MINOR"
|
||||||
|
|
||||||
|
resolve_latest_game_jars
|
||||||
|
log "resolving the newest Velocity ${VELOCITY_LATEST_MINOR} build"
|
||||||
|
velocity="$(papermc_latest_jar velocity "$VELOCITY_LATEST_MINOR")" \
|
||||||
|
|| die "no Velocity build for ${VELOCITY_LATEST_MINOR}"
|
||||||
|
# shellcheck disable=SC2034 # read back through ${!key} below
|
||||||
|
VELOCITY_VERSION="$VELOCITY_LATEST_MINOR"
|
||||||
|
# shellcheck disable=SC2034
|
||||||
|
VELOCITY_JAR_URL="${velocity% *}"
|
||||||
|
# shellcheck disable=SC2034
|
||||||
|
VELOCITY_JAR_SHA256="${velocity##* }"
|
||||||
|
|
||||||
|
tmp="$(mktemp)"
|
||||||
|
trap 'rm -f "$tmp"' EXIT
|
||||||
|
# The comment header is kept as it is; only the KEY=value lines are regenerated.
|
||||||
|
sed -n '/^#/p;/^#/!q' "$LOCK" > "$tmp"
|
||||||
|
for key in MC_VERSION LIMBO_VERSION LIMBO_JAR_URL LIMBO_JAR_SHA256 LIMBO_SCHEM_URL LIMBO_SCHEM_SHA256 \
|
||||||
|
PAPER_JAR_URL PAPER_JAR_SHA256 LUCKPERMS_JAR_URL LUCKPERMS_JAR_SHA256 \
|
||||||
|
VELOCITY_VERSION VELOCITY_JAR_URL VELOCITY_JAR_SHA256; do
|
||||||
|
printf '%s=%s\n' "$key" "${!key}" >> "$tmp"
|
||||||
|
done
|
||||||
|
|
||||||
|
if cmp -s "$tmp" "$LOCK"; then
|
||||||
|
ok "game-stack.lock already pins upstream's newest builds"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
diff -u "$LOCK" "$tmp" >&2 || true
|
||||||
|
if [ "$check" = 1 ]; then
|
||||||
|
die "upstream has newer builds than game-stack.lock"
|
||||||
|
fi
|
||||||
|
cp "$tmp" "$LOCK"
|
||||||
|
ok "game-stack.lock updated; run go test . and the bootstrap tests, then commit"
|
||||||
+20
-18
@@ -52,17 +52,10 @@ A grep across `*.md` and `*.go` returns both sets; only the Go ones are seams.
|
|||||||
Secret-replica mechanism the login gate uses (bootstrap + `felis setup`), and the
|
Secret-replica mechanism the login gate uses (bootstrap + `felis setup`), and the
|
||||||
build egress lock allows exactly the control namespace on the internal port.
|
build egress lock allows exactly the control namespace on the internal port.
|
||||||
Uniform for local and s3:// stores — neither hands the sandboxed build Pod a
|
Uniform for local and s3:// stores — neither hands the sandboxed build Pod a
|
||||||
filesystem view or object-store credentials. Kaniko/Trivy images are
|
filesystem view or object-store credentials. Kaniko, Trivy and Trivy's two DBs
|
||||||
external-only by default; `[registry] kaniko_image / trivy_image /
|
come from the registry's `mirror/` copies, which the installer and
|
||||||
build_cpu_limit / build_mem_limit` override them for mirrored or air-gapped
|
felis-build-tools.timer keep current (`felis mirror-build-tools`,
|
||||||
installs. Trivy's vulnerability DB is the same story, and now has its own knob:
|
docs/troubleshooting.md §8e); the `[registry]` keys override them.
|
||||||
`[registry] trivy_db_repository` points `--db-repository` at an internal mirror
|
|
||||||
(recipe in docs/troubleshooting.md §8e); `trivy_java_db_repository` does the
|
|
||||||
same for the Java DB, which Trivy fetches so soon as the scanned image contains
|
|
||||||
a jar — i.e. for every real modpack build. Left unset on an egress-locked box
|
|
||||||
the scan step fails closed — Kaniko pushes, Trivy exits on the DB download —
|
|
||||||
which is the correct fail direction but leaves the build unfinished, so the
|
|
||||||
mirrors are part of a production build install.
|
|
||||||
|
|
||||||
## Built; only its I/O is unverifiable from this repo
|
## Built; only its I/O is unverifiable from this repo
|
||||||
|
|
||||||
@@ -95,10 +88,19 @@ worth revisiting.
|
|||||||
moved inside `ClaimServer` (advisory lock + re-check + UPDATE in one transaction),
|
moved inside `ClaimServer` (advisory lock + re-check + UPDATE in one transaction),
|
||||||
red-then-green in the pgint suite, which is exactly the real-Postgres harness this
|
red-then-green in the pgint suite, which is exactly the real-Postgres harness this
|
||||||
line was waiting for.
|
line was waiting for.
|
||||||
- `internal/api/api.go:671` — `cooldownLimiter` is process-local, so across N api
|
- `internal/api/api.go:773` — `cooldownLimiter` is process-local, so across N api
|
||||||
replicas a caller could draw up to N OTP codes per window. The intra-replica burst
|
replicas a caller could draw up to N OTP codes per window. The intra-replica burst
|
||||||
is closed; cross-replica bounding needs a shared store, out of scope for a
|
is closed; cross-replica bounding needs a shared store, out of scope for a
|
||||||
single-replica install.
|
single-replica install. Revisit before the api Deployment runs more than one
|
||||||
|
replica.
|
||||||
|
- `internal/submit/submit.go:524` — the per-user upload storage budget reads the
|
||||||
|
stored bytes, then writes. On one replica the API's per-user upload reservation
|
||||||
|
serializes it; across replicas a burst can overshoot by one blob per interleaved
|
||||||
|
upload, each still under the single-blob cap. The pending-submission cap no
|
||||||
|
longer has this shape: `CreateSubmission` counts and inserts under a
|
||||||
|
per-submitter advisory lock (pgint `TestSubmitPendingCapHoldsUnderConcurrency`).
|
||||||
|
Revisit with the cooldown above, before scaling api replicas: a reservation row
|
||||||
|
per upload in the same kind of transaction closes it.
|
||||||
- `internal/submit/submit.go:436` and `internal/submit/submit_test.go:351` — a
|
- `internal/submit/submit.go:436` and `internal/submit/submit_test.go:351` — a
|
||||||
post-CAS `Approve`
|
post-CAS `Approve`
|
||||||
failure leaves a row indistinguishable from the benign case, so `Approve` returns a
|
failure leaves a row indistinguishable from the benign case, so `Approve` returns a
|
||||||
@@ -132,8 +134,8 @@ worth revisiting.
|
|||||||
|
|
||||||
## Recorded outside the code
|
## Recorded outside the code
|
||||||
|
|
||||||
- `deploy/limbo/README.md:139` — no NetworkPolicy locks the minecraft-namespace
|
- The minecraft-namespace egress is locked (`felis-server-egress`, DNS plus the
|
||||||
egress or the control-namespace ingress today, which is why the login pod reaches
|
public internet with every private range and the node's own global addresses
|
||||||
`felis-api-internal:8081`. This is a conditional obligation rather than a seam: if
|
excluded) and `felis-login-to-internal-api` opens the one platform path a game pod
|
||||||
a future deployment adds either lock, it must also open that path. Spec v4.1 §21
|
needs — login → felis-api:8081. Any new in-cluster service a game server must call
|
||||||
asks for those policies; `cmd/felis/manifests.go` renders the game-port one.
|
needs its own allow policy next to that one (`internal/platform/netpol.go`).
|
||||||
+209
-31
@@ -147,6 +147,19 @@ components:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
RateLimited:
|
||||||
|
description: >
|
||||||
|
This client address called the public sign-in doors faster than the per-address
|
||||||
|
limit allows (code rate_limited); Retry-After gives the seconds until the next
|
||||||
|
call is admitted. The address is the visitor header the install's edge writes
|
||||||
|
([auth] client_ip_header: CF-Connecting-IP behind the Cloudflare tunnel), else
|
||||||
|
the TCP peer; IPv6 clients share one limit per /64.
|
||||||
|
headers:
|
||||||
|
Retry-After:
|
||||||
|
schema: { type: integer }
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
AccessResult:
|
AccessResult:
|
||||||
description: The structured access mutation succeeded; the raw RCON reply is in output.
|
description: The structured access mutation succeeded; the raw RCON reply is in output.
|
||||||
content:
|
content:
|
||||||
@@ -201,6 +214,49 @@ components:
|
|||||||
nullable: true
|
nullable: true
|
||||||
description: Window end (RFC3339, exclusive), or null when unset.
|
description: Window end (RFC3339, exclusive), or null when unset.
|
||||||
|
|
||||||
|
DBBackupStatus:
|
||||||
|
type: object
|
||||||
|
description: >
|
||||||
|
The newest control-plane database backup the host recorded
|
||||||
|
(internal/api/handlers_dbbackup.go dbBackupView; the record itself is
|
||||||
|
internal/dbbackup Status, written by `felis db backup`).
|
||||||
|
required: [last, stale, max_age_seconds]
|
||||||
|
properties:
|
||||||
|
last:
|
||||||
|
type: object
|
||||||
|
nullable: true
|
||||||
|
description: Null until the first backup has been recorded.
|
||||||
|
required: [at, name, label, size_bytes, dir]
|
||||||
|
properties:
|
||||||
|
at:
|
||||||
|
type: string
|
||||||
|
format: date-time
|
||||||
|
description: When the bundle was written.
|
||||||
|
name:
|
||||||
|
type: string
|
||||||
|
description: Bundle file name, felis-db-<UTC stamp>-<label>.tar.
|
||||||
|
label:
|
||||||
|
type: string
|
||||||
|
enum: [daily, pre-migrate, pre-restore, manual]
|
||||||
|
size_bytes:
|
||||||
|
type: integer
|
||||||
|
format: int64
|
||||||
|
felis_version:
|
||||||
|
type: string
|
||||||
|
schema_version:
|
||||||
|
type: integer
|
||||||
|
description: Newest applied migration at backup time.
|
||||||
|
dir:
|
||||||
|
type: string
|
||||||
|
description: Backup directory on the host.
|
||||||
|
stale:
|
||||||
|
type: boolean
|
||||||
|
description: True when there is no record or it is older than max_age_seconds.
|
||||||
|
max_age_seconds:
|
||||||
|
type: integer
|
||||||
|
format: int64
|
||||||
|
description: The freshness limit (26h), shared with `felis db check` and FelisDBBackupStale.
|
||||||
|
|
||||||
PasskeyCredential:
|
PasskeyCredential:
|
||||||
type: object
|
type: object
|
||||||
description: >
|
description: >
|
||||||
@@ -242,6 +298,18 @@ components:
|
|||||||
endpointAddress: { type: string }
|
endpointAddress: { type: string }
|
||||||
playersOnline: { type: integer, format: int32 }
|
playersOnline: { type: integer, format: int32 }
|
||||||
playersMax: { type: integer, format: int32 }
|
playersMax: { type: integer, format: int32 }
|
||||||
|
displayName: { type: string }
|
||||||
|
image: { type: string }
|
||||||
|
javaMemory: { type: string }
|
||||||
|
storageSize: { type: string }
|
||||||
|
cpu: { type: string }
|
||||||
|
idleStopSeconds:
|
||||||
|
type: integer
|
||||||
|
format: int32
|
||||||
|
description: Seconds the server may sit empty before idle auto-stop scales it down; 0 when it never idles out (off, RCON disabled, or a system server).
|
||||||
|
playerCountUnknown:
|
||||||
|
type: boolean
|
||||||
|
description: Present and true while the operator cannot read the player count over RCON; idle auto-stop waits until it can.
|
||||||
|
|
||||||
MyServerView:
|
MyServerView:
|
||||||
type: object
|
type: object
|
||||||
@@ -573,7 +641,12 @@ paths:
|
|||||||
tags: [admin-servers]
|
tags: [admin-servers]
|
||||||
operationId: createServer
|
operationId: createServer
|
||||||
summary: Create a server (admin).
|
summary: Create a server (admin).
|
||||||
description: Requires the admin Access path; the image must be whitelisted.
|
description: >-
|
||||||
|
Requires the admin Access path; the image must be whitelisted. An image in the
|
||||||
|
platform registry is stored pinned to the digest its tag names at creation
|
||||||
|
(name:tag@sha256:…), so a later push over the tag never moves the server;
|
||||||
|
400 image_not_in_registry when the registry lacks the tag, 503
|
||||||
|
registry_unavailable when it cannot be asked.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: admin
|
x-felis-tier: admin
|
||||||
security: [{ accessJWT: [] }]
|
security: [{ accessJWT: [] }]
|
||||||
@@ -741,6 +814,11 @@ paths:
|
|||||||
$ref: '#/components/responses/Forbidden'
|
$ref: '#/components/responses/Forbidden'
|
||||||
'404':
|
'404':
|
||||||
$ref: '#/components/responses/NotFound'
|
$ref: '#/components/responses/NotFound'
|
||||||
|
'409':
|
||||||
|
description: A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started.
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: Wake cooldown is still active for this server.
|
description: Wake cooldown is still active for this server.
|
||||||
content:
|
content:
|
||||||
@@ -1193,7 +1271,7 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'409':
|
'409':
|
||||||
description: Server is not stopped (its world PVC is still mounted).
|
description: Server is not stopped (not_stopped), or a restore, backup or file write already holds its world volume (maintenance_in_progress).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -1228,6 +1306,11 @@ paths:
|
|||||||
$ref: '#/components/responses/Forbidden'
|
$ref: '#/components/responses/Forbidden'
|
||||||
'404':
|
'404':
|
||||||
$ref: '#/components/responses/NotFound'
|
$ref: '#/components/responses/NotFound'
|
||||||
|
'409':
|
||||||
|
description: A restore, backup or file write holds the server's world volume (maintenance_in_progress); nothing was started.
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: Wake cooldown is still active.
|
description: Wake cooldown is still active.
|
||||||
content:
|
content:
|
||||||
@@ -1822,8 +1905,8 @@ paths:
|
|||||||
array. It never reveals staffness: methods are computed identically for every
|
array. It never reveals staffness: methods are computed identically for every
|
||||||
resolved account (no role branch), so a staff and a player address in the same
|
resolved account (no role branch), so a staff and a player address in the same
|
||||||
credential state return byte-identical bodies. passkey is offered only when a
|
credential state return byte-identical bodies. passkey is offered only when a
|
||||||
verifier is wired. Sends no mail and mutates nothing; not rate-limited at the app
|
verifier is wired. Sends no mail and mutates nothing; bounded by the per-address
|
||||||
layer (volumetric abuse is bounded at the edge). Gated on local_auth_enabled.
|
sign-in rate limit (429 rate_limited). Gated on local_auth_enabled.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: public
|
x-felis-tier: public
|
||||||
security: []
|
security: []
|
||||||
@@ -1865,6 +1948,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
|
|
||||||
/api/v1/auth/passkey/login/begin:
|
/api/v1/auth/passkey/login/begin:
|
||||||
post:
|
post:
|
||||||
@@ -1920,7 +2005,9 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: A passkey login for this recipient was started too recently (otp_resend_cooldown).
|
description: >-
|
||||||
|
A passkey login for this recipient was started too recently (otp_resend_cooldown);
|
||||||
|
or this client address called the sign-in doors too often (rate_limited, with Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -1992,6 +2079,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
'503':
|
'503':
|
||||||
description: No passkey verifier is wired on this deployment (passkey_unavailable).
|
description: No passkey verifier is wired on this deployment (passkey_unavailable).
|
||||||
content:
|
content:
|
||||||
@@ -2013,9 +2102,9 @@ paths:
|
|||||||
userHandle inside the signed assertion at finish. The challenge cannot be
|
userHandle inside the signed assertion at finish. The challenge cannot be
|
||||||
user-keyed, so it is stashed under login_id in a non-user-keyed store and echoed
|
user-keyed, so it is stashed under login_id in a non-user-keyed store and echoed
|
||||||
back at finish. Mounted Public and gated on local_auth_enabled. There is no
|
back at finish. Mounted Public and gated on local_auth_enabled. There is no
|
||||||
recipient or principal to key a per-caller cooldown on (that volumetric limiting
|
recipient or principal to key a per-caller cooldown on, so one client is bounded
|
||||||
is delegated to the edge), so the server-side brake is a hard global cap on live
|
by the per-address sign-in rate limit (429 rate_limited) and the table by a hard
|
||||||
challenges (429 too_many_challenges). Inert for a credential until its owner
|
global cap on live challenges (429 too_many_challenges). Inert for a credential until its owner
|
||||||
enrolls a resident passkey; email-OTP and username-first passkey remain the
|
enrolls a resident passkey; email-OTP and username-first passkey remain the
|
||||||
fallbacks, so no authenticator is ever locked out.
|
fallbacks, so no authenticator is ever locked out.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
@@ -2061,8 +2150,9 @@ paths:
|
|||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: >-
|
description: >-
|
||||||
Too many discoverable logins are in flight server-wide; the global cap is hit
|
Too many discoverable logins are in flight server-wide (too_many_challenges;
|
||||||
(too_many_challenges). No per-recipient signal is leaked — the cap is global.
|
the cap is global, so no per-recipient signal leaks); or this client address
|
||||||
|
called the sign-in doors too often (rate_limited, with Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -2139,6 +2229,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
'503':
|
'503':
|
||||||
description: No passkey verifier is wired on this deployment (passkey_unavailable).
|
description: No passkey verifier is wired on this deployment (passkey_unavailable).
|
||||||
content:
|
content:
|
||||||
@@ -2156,7 +2248,9 @@ paths:
|
|||||||
purpose. An address with no account returns the SAME 202 with no code minted,
|
purpose. An address with no account returns the SAME 202 with no code minted,
|
||||||
and the per-recipient cooldown is kept on that path too, so probing reveals
|
and the per-recipient cooldown is kept on that path too, so probing reveals
|
||||||
nothing (existence is learnt only at the sanctioned /auth/options oracle).
|
nothing (existence is learnt only at the sanctioned /auth/options oracle).
|
||||||
Gated on local_auth_enabled.
|
An account that spent its daily wrong-code budget (10 per 24h, across every
|
||||||
|
code) also gets the same 202 and no mail until the window ends. Gated on
|
||||||
|
local_auth_enabled.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: public
|
x-felis-tier: public
|
||||||
security: []
|
security: []
|
||||||
@@ -2198,7 +2292,10 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: A code for this recipient was requested too recently (otp_resend_cooldown).
|
description: >-
|
||||||
|
A code for this recipient was requested too recently (otp_resend_cooldown);
|
||||||
|
or this client address called the sign-in doors too often (rate_limited, with Retry-After);
|
||||||
|
or the install-wide mail budget is spent (mail_rate_limited, with Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -2215,7 +2312,10 @@ paths:
|
|||||||
under the login purpose, and on success mints a host-only felis_session. An
|
under the login purpose, and on success mints a host-only felis_session. An
|
||||||
unknown address, a wrong or expired code, and an attempt-exhausted code all
|
unknown address, a wrong or expired code, and an attempt-exhausted code all
|
||||||
return the IDENTICAL 400 invalid_code, so the door is not an existence or
|
return the IDENTICAL 400 invalid_code, so the door is not an existence or
|
||||||
lockout oracle. Staff are refused (403) — but only AFTER a valid code is
|
lockout oracle. The 10th wrong code in 24h locks the door for that account
|
||||||
|
until the window ends (the right code then also reads as invalid_code); the
|
||||||
|
owner is told by mail once, and the lock is audited as auth.otp.locked.
|
||||||
|
Staff are refused (403) — but only AFTER a valid code is
|
||||||
redeemed, so only the account owner can ever reach that refusal.
|
redeemed, so only the account owner can ever reach that refusal.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: public
|
x-felis-tier: public
|
||||||
@@ -2260,6 +2360,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
|
|
||||||
/api/v1/auth/op-login/start:
|
/api/v1/auth/op-login/start:
|
||||||
post:
|
post:
|
||||||
@@ -2271,7 +2373,8 @@ paths:
|
|||||||
staff address, opens an op_login request, and mails a one-time code under the
|
staff address, opens an op_login request, and mails a one-time code under the
|
||||||
op_login purpose, returning the request handle the browser polls. A non-staff
|
op_login purpose, returning the request handle the browser polls. A non-staff
|
||||||
or unknown address gets the SAME 202 with a random, non-persisted handle and no
|
or unknown address gets the SAME 202 with a random, non-persisted handle and no
|
||||||
mail, so this never becomes a staff-enumeration oracle. Gated on
|
mail, so this never becomes a staff-enumeration oracle. A staff account that
|
||||||
|
spent its daily wrong-code budget gets the same neutral 202. Gated on
|
||||||
local_auth_enabled.
|
local_auth_enabled.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: public
|
x-felis-tier: public
|
||||||
@@ -2314,7 +2417,10 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: A code for this recipient was requested too recently (otp_resend_cooldown).
|
description: >-
|
||||||
|
A code for this recipient was requested too recently (otp_resend_cooldown);
|
||||||
|
or this client address called the sign-in doors too often (rate_limited, with Retry-After);
|
||||||
|
or the install-wide mail budget is spent (mail_rate_limited, with Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -2362,7 +2468,7 @@ paths:
|
|||||||
Public, pre-session final leg: mints a host-only staff session only when BOTH
|
Public, pre-session final leg: mints a host-only staff session only when BOTH
|
||||||
factors have landed — the request is approved-and-live AND the mailed code
|
factors have landed — the request is approved-and-live AND the mailed code
|
||||||
verifies. Every failure (unknown handle, not-yet-approved, wrong or locked code,
|
verifies. Every failure (unknown handle, not-yet-approved, wrong or locked code,
|
||||||
lost race) collapses into one uniform 400 op_login_invalid, so a code-less
|
an account past its daily wrong-code budget, lost race) collapses into one uniform 400 op_login_invalid, so a code-less
|
||||||
caller learns nothing. Admin is re-asserted before the session is issued.
|
caller learns nothing. Admin is re-asserted before the session is issued.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: public
|
x-felis-tier: public
|
||||||
@@ -2408,6 +2514,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
|
|
||||||
/api/v1/auth/setup/redeem:
|
/api/v1/auth/setup/redeem:
|
||||||
post:
|
post:
|
||||||
@@ -2465,6 +2573,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
|
|
||||||
/api/v1/auth/setup/status:
|
/api/v1/auth/setup/status:
|
||||||
get:
|
get:
|
||||||
@@ -2581,6 +2691,8 @@ paths:
|
|||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
|
'429':
|
||||||
|
$ref: '#/components/responses/RateLimited'
|
||||||
|
|
||||||
/api/v1/me:
|
/api/v1/me:
|
||||||
get:
|
get:
|
||||||
@@ -2714,6 +2826,31 @@ paths:
|
|||||||
'403':
|
'403':
|
||||||
$ref: '#/components/responses/Forbidden'
|
$ref: '#/components/responses/Forbidden'
|
||||||
|
|
||||||
|
/api/v1/platform/db-backup:
|
||||||
|
get:
|
||||||
|
tags: [admin-updates]
|
||||||
|
operationId: getDBBackup
|
||||||
|
summary: Freshness of the newest control-plane database backup (admin).
|
||||||
|
description: >-
|
||||||
|
What the host's felis-db-backup.timer (or a manual `felis db backup`)
|
||||||
|
last recorded in platform_settings. last is null before the first
|
||||||
|
backup; stale is true then, and whenever the newest backup is older than
|
||||||
|
max_age_seconds. Read-only: backups run on the host, never through the API.
|
||||||
|
x-felis-face: [external]
|
||||||
|
x-felis-tier: admin
|
||||||
|
security: [{ accessJWT: [] }]
|
||||||
|
responses:
|
||||||
|
'200':
|
||||||
|
description: The newest recorded backup and whether it is stale.
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema:
|
||||||
|
$ref: '#/components/schemas/DBBackupStatus'
|
||||||
|
'401':
|
||||||
|
$ref: '#/components/responses/Unauthorized'
|
||||||
|
'403':
|
||||||
|
$ref: '#/components/responses/Forbidden'
|
||||||
|
|
||||||
/api/v1/fleet:
|
/api/v1/fleet:
|
||||||
get:
|
get:
|
||||||
tags: [admin-servers]
|
tags: [admin-servers]
|
||||||
@@ -2828,7 +2965,7 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'409':
|
'409':
|
||||||
description: Submission has already been reviewed.
|
description: Server is not stopped (not_stopped), or a restore, backup or file write already holds its world volume (maintenance_in_progress).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -2873,7 +3010,7 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'409':
|
'409':
|
||||||
description: Server is not stopped (its world PVC is still mounted).
|
description: Server is not stopped (not_stopped), or a restore, backup or file write already holds its world volume (maintenance_in_progress).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -3122,7 +3259,7 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'409':
|
'409':
|
||||||
description: Server is not stopped (its world PVC is still mounted).
|
description: Server is not stopped (not_stopped), or a restore, backup or file write already holds its world volume (maintenance_in_progress).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -3690,6 +3827,15 @@ paths:
|
|||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'401':
|
'401':
|
||||||
$ref: '#/components/responses/Unauthorized'
|
$ref: '#/components/responses/Unauthorized'
|
||||||
|
'429':
|
||||||
|
description: >-
|
||||||
|
Resend requested before the cooldown elapsed (otp_resend_cooldown); or the
|
||||||
|
account spent its daily wrong-code budget (otp_account_locked, with
|
||||||
|
Retry-After); or the install-wide mail budget is spent
|
||||||
|
(mail_rate_limited, with Retry-After).
|
||||||
|
content:
|
||||||
|
application/json:
|
||||||
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'502':
|
'502':
|
||||||
$ref: '#/components/responses/MailUndeliverable'
|
$ref: '#/components/responses/MailUndeliverable'
|
||||||
|
|
||||||
@@ -3701,8 +3847,10 @@ paths:
|
|||||||
description: >
|
description: >
|
||||||
Consumes a previously delivered code for the authenticated principal. On
|
Consumes a previously delivered code for the authenticated principal. On
|
||||||
success the user's email is written and email_verified is set true. Too many
|
success the user's email is written and email_verified is set true. Too many
|
||||||
incorrect attempts lock the code (429); an unknown, expired, consumed, or
|
incorrect attempts lock the code (429 otp_locked); 10 wrong codes in 24h,
|
||||||
mismatched code is a 400.
|
counted across every code, lock the account's email-code door until the
|
||||||
|
window ends (429 otp_account_locked with Retry-After). An unknown, expired,
|
||||||
|
consumed, or mismatched code is a 400.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: app
|
x-felis-tier: app
|
||||||
security: [{ accessJWT: [] }]
|
security: [{ accessJWT: [] }]
|
||||||
@@ -3734,7 +3882,9 @@ paths:
|
|||||||
'401':
|
'401':
|
||||||
$ref: '#/components/responses/Unauthorized'
|
$ref: '#/components/responses/Unauthorized'
|
||||||
'429':
|
'429':
|
||||||
description: Too many incorrect attempts; the code is locked.
|
description: >-
|
||||||
|
Too many incorrect attempts on this code (otp_locked), or the account's
|
||||||
|
daily wrong-code budget is spent (otp_account_locked, with Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -3992,7 +4142,11 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: Resend requested before the cooldown elapsed.
|
description: >-
|
||||||
|
Resend requested before the cooldown elapsed (otp_resend_cooldown), or the
|
||||||
|
account's daily wrong-code budget is spent (otp_account_locked, with
|
||||||
|
Retry-After); or the install-wide mail budget is spent
|
||||||
|
(mail_rate_limited, with Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -4007,8 +4161,9 @@ paths:
|
|||||||
description: >
|
description: >
|
||||||
Consumes the fresh migrate-purpose email code for the caller's initiated
|
Consumes the fresh migrate-purpose email code for the caller's initiated
|
||||||
migration and advances it to confirmed with confirm_factor email_otp. Too many
|
migration and advances it to confirmed with confirm_factor email_otp. Too many
|
||||||
wrong attempts lock the code (429 otp_locked); an unknown, expired, consumed, or
|
wrong attempts lock the code (429 otp_locked), and 10 wrong codes in 24h lock
|
||||||
mismatched code is a 400 invalid_code.
|
the account's email-code door (429 otp_account_locked with Retry-After); an
|
||||||
|
unknown, expired, consumed, or mismatched code is a 400 invalid_code.
|
||||||
x-felis-face: [external]
|
x-felis-face: [external]
|
||||||
x-felis-tier: app
|
x-felis-tier: app
|
||||||
security: [{ accessJWT: [] }]
|
security: [{ accessJWT: [] }]
|
||||||
@@ -4049,7 +4204,10 @@ paths:
|
|||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
'429':
|
'429':
|
||||||
description: The code is locked after too many wrong attempts (otp_locked).
|
description: >-
|
||||||
|
The code is locked after too many wrong attempts (otp_locked), or the
|
||||||
|
account's daily wrong-code budget is spent (otp_account_locked, with
|
||||||
|
Retry-After).
|
||||||
content:
|
content:
|
||||||
application/json:
|
application/json:
|
||||||
schema: { $ref: '#/components/schemas/Error' }
|
schema: { $ref: '#/components/schemas/Error' }
|
||||||
@@ -4409,9 +4567,21 @@ paths:
|
|||||||
type: object
|
type: object
|
||||||
description: Only the supplied fields are patched; an empty patch is rejected.
|
description: Only the supplied fields are patched; an empty patch is rejected.
|
||||||
properties:
|
properties:
|
||||||
display_name: { type: string }
|
displayName: { type: string }
|
||||||
autostart_policy: { type: string }
|
autostartPolicy: { type: string }
|
||||||
image: { type: string }
|
image:
|
||||||
|
type: string
|
||||||
|
description: >-
|
||||||
|
Re-admitted against the whitelist (a pinned name:tag@sha256:… ref is
|
||||||
|
admitted by its name:tag) and pinned like create does. A pin equal to
|
||||||
|
the current image is no change; any other needs confirmImageChange.
|
||||||
|
confirmImageChange:
|
||||||
|
type: boolean
|
||||||
|
description: >-
|
||||||
|
Acknowledges that the new image opens the world with its Minecraft
|
||||||
|
version, whose chunk upgrades the old one cannot read. Without it an
|
||||||
|
image that would move the server is refused with 409
|
||||||
|
image_change_unconfirmed. The audit row records image_from/image_to.
|
||||||
memory: { type: string }
|
memory: { type: string }
|
||||||
storage:
|
storage:
|
||||||
type: string
|
type: string
|
||||||
@@ -4420,9 +4590,13 @@ paths:
|
|||||||
type: object
|
type: object
|
||||||
properties:
|
properties:
|
||||||
cpu: { type: string }
|
cpu: { type: string }
|
||||||
cpu_request: { type: string }
|
cpuRequest: { type: string }
|
||||||
memory: { type: string }
|
memory: { type: string }
|
||||||
memory_request: { type: string }
|
memoryRequest: { type: string }
|
||||||
|
idleStopSeconds:
|
||||||
|
type: integer
|
||||||
|
format: int32
|
||||||
|
description: Idle auto-stop. 0 turns it off; otherwise the server stops after this many seconds with nobody online (60–86400, else 400 bad_idle_stop).
|
||||||
responses:
|
responses:
|
||||||
'200':
|
'200':
|
||||||
description: Patched.
|
description: Patched.
|
||||||
@@ -4444,6 +4618,10 @@ paths:
|
|||||||
$ref: '#/components/responses/Forbidden'
|
$ref: '#/components/responses/Forbidden'
|
||||||
'404':
|
'404':
|
||||||
$ref: '#/components/responses/NotFound'
|
$ref: '#/components/responses/NotFound'
|
||||||
|
'409':
|
||||||
|
$ref: '#/components/responses/Conflict'
|
||||||
|
'503':
|
||||||
|
$ref: '#/components/responses/ServiceUnavailable'
|
||||||
|
|
||||||
/api/v1/images/build:
|
/api/v1/images/build:
|
||||||
post:
|
post:
|
||||||
|
|||||||
+1021
-119
File diff suppressed because it is too large.
Load diff
+83
-15
@@ -33,6 +33,11 @@ type API struct {
|
|||||||
// still exercised even before the subsystem is wired in.
|
// still exercised even before the subsystem is wired in.
|
||||||
Builder ImageBuilder
|
Builder ImageBuilder
|
||||||
|
|
||||||
|
// Images pins a whitelisted image ref to the digest it names when a server is
|
||||||
|
// created or its image is changed (internal/imagepin), so a later push over
|
||||||
|
// the same tag never reaches an existing world. Nil stores refs as given.
|
||||||
|
Images ImagePinner
|
||||||
|
|
||||||
// Console is the synchronous RCON write channel (spec §8 写=RCON). It is
|
// Console is the synchronous RCON write channel (spec §8 写=RCON). It is
|
||||||
// wired in production (cmd/felis); a nil Console makes the command route report
|
// wired in production (cmd/felis); a nil Console makes the command route report
|
||||||
// 503 rather than panic, so the ownership boundary is still exercised in tests.
|
// 503 rather than panic, so the ownership boundary is still exercised in tests.
|
||||||
@@ -123,6 +128,14 @@ type API struct {
|
|||||||
// on the wake lever). Zero disables throttling.
|
// on the wake lever). Zero disables throttling.
|
||||||
WakeCooldown time.Duration
|
WakeCooldown time.Duration
|
||||||
|
|
||||||
|
// BackupCooldown spaces out an owner's on-demand backups of one server, and
|
||||||
|
// BackupStoreCap refuses them once the present backups reach [archive]
|
||||||
|
// max_local_bytes (data-durability-9): each archive lands on the node disk
|
||||||
|
// the worlds and the database share. Admins and the break-glass console are
|
||||||
|
// exempt. Zero disables each lever.
|
||||||
|
BackupCooldown time.Duration
|
||||||
|
BackupStoreCap int64
|
||||||
|
|
||||||
// SubmitCreateCooldown / SubmitUploadCooldown throttle the user-modpack
|
// SubmitCreateCooldown / SubmitUploadCooldown throttle the user-modpack
|
||||||
// submission lane per user: create bounds how quickly review-queue rows can
|
// submission lane per user: create bounds how quickly review-queue rows can
|
||||||
// appear, upload bounds how often a user may stream a (up to 1 GiB) build
|
// appear, upload bounds how often a user may stream a (up to 1 GiB) build
|
||||||
@@ -158,6 +171,16 @@ type API struct {
|
|||||||
// Consumed by handleHasJoined (handlers_hasjoined.go).
|
// Consumed by handleHasJoined (handlers_hasjoined.go).
|
||||||
AuthSources []AuthSource
|
AuthSources []AuthSource
|
||||||
|
|
||||||
|
// AuthDoorLimit bounds how often one client address may call the public
|
||||||
|
// pre-session auth doors (ratelimit.go). MailLimit bounds all mail the API
|
||||||
|
// sends, install-wide. Zero values disable them; cmd/felis wires both.
|
||||||
|
AuthDoorLimit RateLimit
|
||||||
|
MailLimit RateLimit
|
||||||
|
// ClientIPHeader names the header the install's edge writes the client
|
||||||
|
// address into (CF-Connecting-IP behind the Cloudflare tunnel,
|
||||||
|
// X-Forwarded-For behind an operator proxy). Empty means the TCP peer.
|
||||||
|
ClientIPHeader string
|
||||||
|
|
||||||
// Now is the clock, injectable for tests. Defaults to time.Now.
|
// Now is the clock, injectable for tests. Defaults to time.Now.
|
||||||
Now func() time.Time
|
Now func() time.Time
|
||||||
|
|
||||||
@@ -172,6 +195,11 @@ type API struct {
|
|||||||
|
|
||||||
streamCapOnce sync.Once
|
streamCapOnce sync.Once
|
||||||
streamCap *streamLimiter
|
streamCap *streamLimiter
|
||||||
|
|
||||||
|
authDoorOnce sync.Once
|
||||||
|
authDoorBuckets *bucketSet
|
||||||
|
mailOnce sync.Once
|
||||||
|
mailBuckets *bucketSet
|
||||||
}
|
}
|
||||||
|
|
||||||
// panelURL returns the public player-console origin ("https://console.<root>"),
|
// panelURL returns the public player-console origin ("https://console.<root>"),
|
||||||
@@ -286,6 +314,11 @@ type apiRoute struct {
|
|||||||
// whose EmailVerified is false is restricted to these routes only.
|
// whose EmailVerified is false is restricted to these routes only.
|
||||||
SetupAllowed bool
|
SetupAllowed bool
|
||||||
|
|
||||||
|
// AuthDoor marks a public pre-session auth door: it is rate limited per
|
||||||
|
// client address (throttleAuthDoor). The op-login status poll is left off,
|
||||||
|
// since the browser calls it every few seconds while it waits.
|
||||||
|
AuthDoor bool
|
||||||
|
|
||||||
h http.HandlerFunc
|
h http.HandlerFunc
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -381,21 +414,21 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
|||||||
// counter-slice to the anti-enumeration doors — the ONE sanctioned place existence
|
// counter-slice to the anti-enumeration doors — the ONE sanctioned place existence
|
||||||
// is disclosed — but it never reveals staffness (methods computed with no role
|
// is disclosed — but it never reveals staffness (methods computed with no role
|
||||||
// branch, so a staff and a player address in the same state are indistinguishable).
|
// branch, so a staff and a player address in the same state are indistinguishable).
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/options", Public: true, h: a.handleAuthOptions},
|
{Method: "POST", Pattern: "/api/v1/auth/options", Public: true, AuthDoor: true, h: a.handleAuthOptions},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/setup/redeem", Public: true, h: a.handleSetupRedeem},
|
{Method: "POST", Pattern: "/api/v1/auth/setup/redeem", Public: true, AuthDoor: true, h: a.handleSetupRedeem},
|
||||||
{Method: "GET", Pattern: "/api/v1/auth/setup/status", SetupAllowed: true, h: a.handleSetupStatus},
|
{Method: "GET", Pattern: "/api/v1/auth/setup/status", SetupAllowed: true, h: a.handleSetupStatus},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/begin", Public: true, h: a.handlePasskeyLoginBegin},
|
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/begin", Public: true, AuthDoor: true, h: a.handlePasskeyLoginBegin},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/finish", Public: true, h: a.handlePasskeyLoginFinish},
|
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/finish", Public: true, AuthDoor: true, h: a.handlePasskeyLoginFinish},
|
||||||
// Discoverable ("usernameless") passkey login (task #40): the from-zero sibling of the
|
// Discoverable ("usernameless") passkey login (task #40): the from-zero sibling of the
|
||||||
// email-first pair above — no identifier typed, the account is resolved from the
|
// email-first pair above — no identifier typed, the account is resolved from the
|
||||||
// userHandle inside the signed assertion (handlers_passkey_discoverable.go).
|
// userHandle inside the signed assertion (handlers_passkey_discoverable.go).
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/discoverable/begin", Public: true, h: a.handlePasskeyLoginDiscoverableBegin},
|
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/discoverable/begin", Public: true, AuthDoor: true, h: a.handlePasskeyLoginDiscoverableBegin},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/discoverable/finish", Public: true, h: a.handlePasskeyLoginDiscoverableFinish},
|
{Method: "POST", Pattern: "/api/v1/auth/passkey/login/discoverable/finish", Public: true, AuthDoor: true, h: a.handlePasskeyLoginDiscoverableFinish},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/email/start", Public: true, h: a.handleLoginEmailStart},
|
{Method: "POST", Pattern: "/api/v1/auth/email/start", Public: true, AuthDoor: true, h: a.handleLoginEmailStart},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/email/verify", Public: true, h: a.handleLoginEmailVerify},
|
{Method: "POST", Pattern: "/api/v1/auth/email/verify", Public: true, AuthDoor: true, h: a.handleLoginEmailVerify},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/op-login/start", Public: true, h: a.handleOpLoginStart},
|
{Method: "POST", Pattern: "/api/v1/auth/op-login/start", Public: true, AuthDoor: true, h: a.handleOpLoginStart},
|
||||||
{Method: "GET", Pattern: "/api/v1/auth/op-login/status/{id}", Public: true, h: a.handleOpLoginStatus},
|
{Method: "GET", Pattern: "/api/v1/auth/op-login/status/{id}", Public: true, h: a.handleOpLoginStatus},
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/op-login/finish", Public: true, h: a.handleOpLoginFinish},
|
{Method: "POST", Pattern: "/api/v1/auth/op-login/finish", Public: true, AuthDoor: true, h: a.handleOpLoginFinish},
|
||||||
// Player-console bootstrap (console-tier access model): the account-less
|
// Player-console bootstrap (console-tier access model): the account-less
|
||||||
// player's door into console.<root_domain>. Public — like login there is no prior
|
// player's door into console.<root_domain>. Public — like login there is no prior
|
||||||
// principal — and session-minting, but the artifact it consumes is a one-time
|
// principal — and session-minting, but the artifact it consumes is a one-time
|
||||||
@@ -403,7 +436,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
|||||||
// possession already proves a Minecraft identity. A code whose UUID belongs to
|
// possession already proves a Minecraft identity. A code whose UUID belongs to
|
||||||
// staff is refused (403) so this never yields an admin session; op.console stays
|
// staff is refused (403) so this never yields an admin session; op.console stays
|
||||||
// behind Zero Trust (handlers_onboard.go).
|
// behind Zero Trust (handlers_onboard.go).
|
||||||
{Method: "POST", Pattern: "/api/v1/auth/bind", Public: true, h: a.handleBindRedeem},
|
{Method: "POST", Pattern: "/api/v1/auth/bind", Public: true, AuthDoor: true, h: a.handleBindRedeem},
|
||||||
|
|
||||||
// App-auth tier: operations on your own servers (spec §14).
|
// App-auth tier: operations on your own servers (spec §14).
|
||||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/wake", h: a.handleWake},
|
{Method: "POST", Pattern: "/api/v1/servers/{name}/wake", h: a.handleWake},
|
||||||
@@ -565,6 +598,9 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
|||||||
// only — the runner/executors that consume the window are still INTEGRATION-ONLY.
|
// only — the runner/executors that consume the window are still INTEGRATION-ONLY.
|
||||||
{Method: "GET", Pattern: "/api/v1/updates/window", Admin: true, h: a.handleGetUpdateWindow},
|
{Method: "GET", Pattern: "/api/v1/updates/window", Admin: true, h: a.handleGetUpdateWindow},
|
||||||
{Method: "PUT", Pattern: "/api/v1/updates/window", Admin: true, h: a.handleSetUpdateWindow},
|
{Method: "PUT", Pattern: "/api/v1/updates/window", Admin: true, h: a.handleSetUpdateWindow},
|
||||||
|
// Control-plane database backup freshness, as the host's felis-db-backup.timer
|
||||||
|
// last recorded it. Admin-tier: it names the host backup directory.
|
||||||
|
{Method: "GET", Pattern: "/api/v1/platform/db-backup", Admin: true, h: a.handleGetDBBackup},
|
||||||
|
|
||||||
// User admin (spec §7, owner-only). Every route gates on the admin Zero-Trust
|
// User admin (spec §7, owner-only). Every route gates on the admin Zero-Trust
|
||||||
// path AND the owner role: listing, mutating, disabling, or deleting users is
|
// path AND the owner role: listing, mutating, disabling, or deleting users is
|
||||||
@@ -611,7 +647,11 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler
|
|||||||
for _, rt := range routes {
|
for _, rt := range routes {
|
||||||
pattern := rt.Method + " " + rt.Pattern
|
pattern := rt.Method + " " + rt.Pattern
|
||||||
if rt.Public {
|
if rt.Public {
|
||||||
mux.HandleFunc(pattern, rt.h)
|
h := rt.h
|
||||||
|
if rt.AuthDoor {
|
||||||
|
h = a.throttleAuthDoor(h)
|
||||||
|
}
|
||||||
|
mux.HandleFunc(pattern, h)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
h := rt.h
|
h := rt.h
|
||||||
@@ -724,10 +764,35 @@ func principalFromContext(ctx context.Context) *Principal {
|
|||||||
// reserve/release pair closes the intra-replica concurrent burst (the bug fixed in
|
// reserve/release pair closes the intra-replica concurrent burst (the bug fixed in
|
||||||
// #35); cross-replica bounding would need a shared store (out of scope for the
|
// #35); cross-replica bounding would need a shared store (out of scope for the
|
||||||
// single-replica demo).
|
// single-replica demo).
|
||||||
|
//
|
||||||
|
// Entries older than the longest window the limiter has been asked about can
|
||||||
|
// no longer block anything, so checks sweep them out (at most once per
|
||||||
|
// bucketSweepEvery). Without that, every distinct address typed into a public
|
||||||
|
// door, whose neutral branch keeps its reservation, stayed in the map for the
|
||||||
|
// life of the process.
|
||||||
type cooldownLimiter struct {
|
type cooldownLimiter struct {
|
||||||
mu sync.Mutex
|
mu sync.Mutex
|
||||||
now func() time.Time
|
now func() time.Time
|
||||||
last map[string]time.Time
|
last map[string]time.Time
|
||||||
|
maxWindow time.Duration
|
||||||
|
swept time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
// noteWindow widens the retention to window and sweeps stale entries when due.
|
||||||
|
// The caller holds mu.
|
||||||
|
func (c *cooldownLimiter) noteWindow(window time.Duration, now time.Time) {
|
||||||
|
if window > c.maxWindow {
|
||||||
|
c.maxWindow = window
|
||||||
|
}
|
||||||
|
if c.maxWindow <= 0 || now.Sub(c.swept) < bucketSweepEvery {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
c.swept = now
|
||||||
|
for k, t := range c.last {
|
||||||
|
if now.Sub(t) >= c.maxWindow {
|
||||||
|
delete(c.last, k)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// allowed reports whether name may wake now WITHOUT recording the attempt. A
|
// allowed reports whether name may wake now WITHOUT recording the attempt. A
|
||||||
@@ -742,7 +807,9 @@ func (c *cooldownLimiter) allowed(name string, window time.Duration) bool {
|
|||||||
}
|
}
|
||||||
c.mu.Lock()
|
c.mu.Lock()
|
||||||
defer c.mu.Unlock()
|
defer c.mu.Unlock()
|
||||||
if last, ok := c.last[name]; ok && c.now().Sub(last) < window {
|
now := c.now()
|
||||||
|
c.noteWindow(window, now)
|
||||||
|
if last, ok := c.last[name]; ok && now.Sub(last) < window {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
return true
|
return true
|
||||||
@@ -775,10 +842,11 @@ func (c *cooldownLimiter) reserve(name string, window time.Duration) (time.Time,
|
|||||||
}
|
}
|
||||||
c.mu.Lock()
|
c.mu.Lock()
|
||||||
defer c.mu.Unlock()
|
defer c.mu.Unlock()
|
||||||
if last, ok := c.last[name]; ok && c.now().Sub(last) < window {
|
t := c.now()
|
||||||
|
c.noteWindow(window, t)
|
||||||
|
if last, ok := c.last[name]; ok && t.Sub(last) < window {
|
||||||
return time.Time{}, false
|
return time.Time{}, false
|
||||||
}
|
}
|
||||||
t := c.now()
|
|
||||||
c.last[name] = t
|
c.last[name] = t
|
||||||
return t, true
|
return t, true
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -47,6 +47,10 @@ type fakeRepo struct {
|
|||||||
serverResources map[string]ResourceSpec
|
serverResources map[string]ResourceSpec
|
||||||
resourceUpdates map[string]ResourceSpec
|
resourceUpdates map[string]ResourceSpec
|
||||||
audits []AuditEntry
|
audits []AuditEntry
|
||||||
|
failAudit error // Audit fails with it (a store outage)
|
||||||
|
// backupRequested mirrors the newest backup.create audit row per server,
|
||||||
|
// stamped by Audit with the wall clock (LastBackupRequest).
|
||||||
|
backupRequested map[string]time.Time
|
||||||
joins []string
|
joins []string
|
||||||
// create-server seeding (spec §15)
|
// create-server seeding (spec §15)
|
||||||
seeded map[string]bool // name -> servers row exists
|
seeded map[string]bool // name -> servers row exists
|
||||||
@@ -74,6 +78,8 @@ type fakeRepo struct {
|
|||||||
// player email OTPs (spec §B2). Keyed by row id; the verify path scans for the
|
// player email OTPs (spec §B2). Keyed by row id; the verify path scans for the
|
||||||
// newest live (user, purpose) just as the PG query does.
|
// newest live (user, purpose) just as the PG query does.
|
||||||
otps map[string]*fakeEmailOTP
|
otps map[string]*fakeEmailOTP
|
||||||
|
// otpBudget mirrors otp_failure_windows, keyed user|purpose.
|
||||||
|
otpBudget map[string]*fakeOTPBudget
|
||||||
// op-login requests (spec §B op-login). opLogins mirrors op_login_requests keyed
|
// op-login requests (spec §B op-login). opLogins mirrors op_login_requests keyed
|
||||||
// by id; the in-game approve/finish paths mutate status/consumed in place, and
|
// by id; the in-game approve/finish paths mutate status/consumed in place, and
|
||||||
// tests plant rows directly to drive the status/finish/pending-list paths.
|
// tests plant rows directly to drive the status/finish/pending-list paths.
|
||||||
@@ -151,6 +157,37 @@ type fakeDataHold struct {
|
|||||||
expiresAt time.Time
|
expiresAt time.Time
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fakeOTPBudget mirrors an otp_failure_windows row.
|
||||||
|
type fakeOTPBudget struct {
|
||||||
|
windowStart time.Time
|
||||||
|
failures int
|
||||||
|
}
|
||||||
|
|
||||||
|
// OTPLockedUntil mirrors PGRepo.OTPLockedUntil through the shared otpLockEnd rule.
|
||||||
|
func (f *fakeRepo) OTPLockedUntil(_ context.Context, userID, purpose string, now time.Time) (time.Time, error) {
|
||||||
|
b := f.otpBudget[userID+"|"+purpose]
|
||||||
|
if b == nil {
|
||||||
|
return time.Time{}, nil
|
||||||
|
}
|
||||||
|
return otpLockEnd(b.windowStart, b.failures, now), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// chargeOTP mirrors chargeOTPMismatch: one wrong guess on the code and the budget.
|
||||||
|
func (f *fakeRepo) chargeOTP(live *fakeEmailOTP, now time.Time) error {
|
||||||
|
live.attempts++
|
||||||
|
key := live.userID + "|" + live.purpose
|
||||||
|
b := f.otpBudget[key]
|
||||||
|
if b == nil || !b.windowStart.Add(otpFailureWindow).After(now) {
|
||||||
|
b = &fakeOTPBudget{windowStart: now}
|
||||||
|
f.otpBudget[key] = b
|
||||||
|
}
|
||||||
|
b.failures++
|
||||||
|
if b.failures == otpFailureBudget {
|
||||||
|
return &OTPAccountLockedError{Until: b.windowStart.Add(otpFailureWindow), JustLocked: true}
|
||||||
|
}
|
||||||
|
return ErrOTPInvalid
|
||||||
|
}
|
||||||
|
|
||||||
// fakeEmailOTP mirrors an email_otps row: only the code hash is held (never the
|
// fakeEmailOTP mirrors an email_otps row: only the code hash is held (never the
|
||||||
// digits), attempts caps brute force, consumed marks single-use, and createdAt
|
// digits), attempts caps brute force, consumed marks single-use, and createdAt
|
||||||
// orders the newest-live lookup.
|
// orders the newest-live lookup.
|
||||||
@@ -227,6 +264,7 @@ func newFakeRepo() *fakeRepo {
|
|||||||
sessions: map[string]*fakeSession{},
|
sessions: map[string]*fakeSession{},
|
||||||
settings: map[string][]byte{},
|
settings: map[string][]byte{},
|
||||||
otps: map[string]*fakeEmailOTP{},
|
otps: map[string]*fakeEmailOTP{},
|
||||||
|
otpBudget: map[string]*fakeOTPBudget{},
|
||||||
opLogins: map[string]*fakeOpLogin{},
|
opLogins: map[string]*fakeOpLogin{},
|
||||||
setupTokens: map[string]fakeSetupToken{},
|
setupTokens: map[string]fakeSetupToken{},
|
||||||
blacklist: map[string]bool{},
|
blacklist: map[string]bool{},
|
||||||
@@ -373,12 +411,14 @@ func (f *fakeRepo) VerifyEmailOTP(_ context.Context, userID, purpose, codeHash s
|
|||||||
if !live.expiresAt.After(now) {
|
if !live.expiresAt.After(now) {
|
||||||
return "", ErrOTPInvalid
|
return "", ErrOTPInvalid
|
||||||
}
|
}
|
||||||
|
if until, _ := f.OTPLockedUntil(context.Background(), userID, purpose, now); !until.IsZero() {
|
||||||
|
return "", &OTPAccountLockedError{Until: until}
|
||||||
|
}
|
||||||
if live.attempts >= otpMaxAttempts {
|
if live.attempts >= otpMaxAttempts {
|
||||||
return "", ErrOTPLocked
|
return "", ErrOTPLocked
|
||||||
}
|
}
|
||||||
if live.codeHash != codeHash {
|
if live.codeHash != codeHash {
|
||||||
live.attempts++ // a typo costs an attempt but does not consume the code
|
return "", f.chargeOTP(live, now) // a typo costs an attempt but does not consume the code
|
||||||
return "", ErrOTPInvalid
|
|
||||||
}
|
}
|
||||||
// A DIFFERENT verified holder of the same address → ErrEmailTaken, code left
|
// A DIFFERENT verified holder of the same address → ErrEmailTaken, code left
|
||||||
// live — mirrors PGRepo's guard + the users_verified_email_unique index.
|
// live — mirrors PGRepo's guard + the users_verified_email_unique index.
|
||||||
@@ -709,10 +749,36 @@ func (f *fakeRepo) SeedServer(_ context.Context, name, subdomain string, _, _, _
|
|||||||
}
|
}
|
||||||
func (f *fakeRepo) Ping(_ context.Context) error { return f.pingErr }
|
func (f *fakeRepo) Ping(_ context.Context) error { return f.pingErr }
|
||||||
func (f *fakeRepo) Audit(_ context.Context, e AuditEntry) error {
|
func (f *fakeRepo) Audit(_ context.Context, e AuditEntry) error {
|
||||||
|
if f.failAudit != nil {
|
||||||
|
return f.failAudit
|
||||||
|
}
|
||||||
f.audits = append(f.audits, e)
|
f.audits = append(f.audits, e)
|
||||||
|
if e.Action == "backup.create" {
|
||||||
|
if f.backupRequested == nil {
|
||||||
|
f.backupRequested = map[string]time.Time{}
|
||||||
|
}
|
||||||
|
f.backupRequested[e.ServerName] = time.Now()
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (f *fakeRepo) LastBackupRequest(_ context.Context, serverName string, since time.Time) (time.Time, error) {
|
||||||
|
if at, ok := f.backupRequested[serverName]; ok && !at.Before(since) {
|
||||||
|
return at, nil
|
||||||
|
}
|
||||||
|
return time.Time{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeRepo) BackupStoreBytes(context.Context) (int64, error) {
|
||||||
|
var n int64
|
||||||
|
for _, b := range f.backups {
|
||||||
|
if b.view.Status == "present" {
|
||||||
|
n += b.view.SizeBytes
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, nil
|
||||||
|
}
|
||||||
|
|
||||||
// AllBackups / BackupsForUser / LatestBackup mirror the PG queries' contract so
|
// AllBackups / BackupsForUser / LatestBackup mirror the PG queries' contract so
|
||||||
// the hermetic tests can't pass against a too-lenient fake: only status='present'
|
// the hermetic tests can't pass against a too-lenient fake: only status='present'
|
||||||
// rows are visible, the user scope is the former_owner column, and LatestBackup
|
// rows are visible, the user scope is the former_owner column, and LatestBackup
|
||||||
@@ -821,7 +887,8 @@ func (f *fakeRepo) SessionUser(_ context.Context, tokenHash string, now time.Tim
|
|||||||
return nil, ErrNotFound
|
return nil, ErrNotFound
|
||||||
}
|
}
|
||||||
return &SessionedUser{
|
return &SessionedUser{
|
||||||
ID: u.ID, Email: u.Email, Role: u.Role,
|
ID: u.ID, Username: u.Username, Email: u.Email, Role: u.Role,
|
||||||
|
EmailVerified: u.EmailVerified,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1319,12 +1386,14 @@ func (f *fakeRepo) ConsumeLoginEmailOTP(_ context.Context, userID, purpose, code
|
|||||||
if live == nil || !live.expiresAt.After(now) {
|
if live == nil || !live.expiresAt.After(now) {
|
||||||
return ErrOTPInvalid
|
return ErrOTPInvalid
|
||||||
}
|
}
|
||||||
|
if until, _ := f.OTPLockedUntil(context.Background(), userID, purpose, now); !until.IsZero() {
|
||||||
|
return &OTPAccountLockedError{Until: until}
|
||||||
|
}
|
||||||
if live.attempts >= otpMaxAttempts {
|
if live.attempts >= otpMaxAttempts {
|
||||||
return ErrOTPLocked
|
return ErrOTPLocked
|
||||||
}
|
}
|
||||||
if live.codeHash != codeHash {
|
if live.codeHash != codeHash {
|
||||||
live.attempts++ // a typo costs an attempt but does not consume the code
|
return f.chargeOTP(live, now) // a typo costs an attempt but does not consume the code
|
||||||
return ErrOTPInvalid
|
|
||||||
}
|
}
|
||||||
live.consumed = true
|
live.consumed = true
|
||||||
return nil
|
return nil
|
||||||
@@ -1454,12 +1523,19 @@ type fakeCluster struct {
|
|||||||
noWorld map[string]bool // server names modeled WITHOUT a world volume (never started / reaped)
|
noWorld map[string]bool // server names modeled WITHOUT a world volume (never started / reaped)
|
||||||
createErr error
|
createErr error
|
||||||
pingErr error
|
pingErr error
|
||||||
|
// maintErr / wakeErr: what AcquireMaintenance / SetDesiredState(Running)
|
||||||
|
// return for a server (the world-volume lock, internal/maintenance).
|
||||||
|
maintErr map[string]error
|
||||||
|
wakeErr map[string]error
|
||||||
|
acquired []string // "name:kind" per admitted AcquireMaintenance
|
||||||
|
released []string // names per ReleaseMaintenance
|
||||||
}
|
}
|
||||||
|
|
||||||
func newFakeCluster() *fakeCluster {
|
func newFakeCluster() *fakeCluster {
|
||||||
return &fakeCluster{byName: map[string]*ServerInfo{}, bySub: map[string]*ServerInfo{},
|
return &fakeCluster{byName: map[string]*ServerInfo{}, bySub: map[string]*ServerInfo{},
|
||||||
desired: map[string]v1alpha1.DesiredState{}, created: map[string]CreateServerInput{},
|
desired: map[string]v1alpha1.DesiredState{}, created: map[string]CreateServerInput{},
|
||||||
patched: map[string]ServerSpecPatch{}, noWorld: map[string]bool{}}
|
patched: map[string]ServerSpecPatch{}, noWorld: map[string]bool{},
|
||||||
|
maintErr: map[string]error{}, wakeErr: map[string]error{}}
|
||||||
}
|
}
|
||||||
func (c *fakeCluster) GetServer(_ context.Context, n string) (*ServerInfo, error) {
|
func (c *fakeCluster) GetServer(_ context.Context, n string) (*ServerInfo, error) {
|
||||||
if s, ok := c.byName[n]; ok {
|
if s, ok := c.byName[n]; ok {
|
||||||
@@ -1483,9 +1559,23 @@ func (c *fakeCluster) WorldVolumeExists(_ context.Context, n string) (bool, erro
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.DesiredState) error {
|
func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.DesiredState) error {
|
||||||
|
if err := c.wakeErr[n]; err != nil && s == v1alpha1.DesiredRunning {
|
||||||
|
return err
|
||||||
|
}
|
||||||
c.desired[n] = s
|
c.desired[n] = s
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
func (c *fakeCluster) AcquireMaintenance(_ context.Context, n, kind string) error {
|
||||||
|
if err := c.maintErr[n]; err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
c.acquired = append(c.acquired, n+":"+kind)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
func (c *fakeCluster) ReleaseMaintenance(_ context.Context, n string) error {
|
||||||
|
c.released = append(c.released, n)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
func (c *fakeCluster) CreateServer(_ context.Context, in CreateServerInput) error {
|
func (c *fakeCluster) CreateServer(_ context.Context, in CreateServerInput) error {
|
||||||
if c.createErr != nil {
|
if c.createErr != nil {
|
||||||
return c.createErr
|
return c.createErr
|
||||||
@@ -1512,6 +1602,9 @@ func (c *fakeCluster) PatchServerSpec(_ context.Context, n string, p ServerSpecP
|
|||||||
if p.AutostartPolicy != nil {
|
if p.AutostartPolicy != nil {
|
||||||
info.AutostartPolicy = string(*p.AutostartPolicy)
|
info.AutostartPolicy = string(*p.AutostartPolicy)
|
||||||
}
|
}
|
||||||
|
if p.IdleStopSeconds != nil {
|
||||||
|
info.IdleStopSeconds = *p.IdleStopSeconds
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,185 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"cmp"
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"log"
|
||||||
|
"net/http"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
"unicode/utf8"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/metrics"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Audit rows (audit_logs, spec §6).
|
||||||
|
//
|
||||||
|
// Every row an HTTP request writes carries the acting account's id
|
||||||
|
// (actor_user_id), the caller's address (the one the sign-in limit keys on) and
|
||||||
|
// user agent; actor is display text. A failed write never fails the operation
|
||||||
|
// it records, which already happened, but it is logged and counted
|
||||||
|
// (felis_audit_write_failures_total, FelisAuditWriteFailing): a silent drop is
|
||||||
|
// how a database blip erases the trail.
|
||||||
|
|
||||||
|
const (
|
||||||
|
// auditWriteTimeout bounds one audit insert. The write outlives the caller's
|
||||||
|
// request context, so a client that hangs up right after the action cannot
|
||||||
|
// cancel its own audit row.
|
||||||
|
auditWriteTimeout = 5 * time.Second
|
||||||
|
// auditUserAgentMax bounds the stored user agent; the header is the caller's
|
||||||
|
// to write.
|
||||||
|
auditUserAgentMax = 256
|
||||||
|
// anonymousActor names a caller no account was resolved for.
|
||||||
|
anonymousActor = "anonymous"
|
||||||
|
)
|
||||||
|
|
||||||
|
// auditActor is the display name for a principal: an email only when something
|
||||||
|
// vouches for it (an Access JWT, or a session whose address was verified), else
|
||||||
|
// the username. A player can set their address to anyone's before verifying it,
|
||||||
|
// so an unverified email would let them sign rows as that person.
|
||||||
|
func auditActor(p *Principal) string {
|
||||||
|
switch {
|
||||||
|
case p == nil:
|
||||||
|
return anonymousActor
|
||||||
|
case p.Email != "" && (p.EmailVerified || !p.ViaSession):
|
||||||
|
return p.Email
|
||||||
|
case p.Username != "":
|
||||||
|
return p.Username
|
||||||
|
case p.UserID != "":
|
||||||
|
return p.UserID
|
||||||
|
}
|
||||||
|
return anonymousActor
|
||||||
|
}
|
||||||
|
|
||||||
|
// audit records an action by the signed-in caller. target is the object acted
|
||||||
|
// on (a server name, a user or credential id) and lands in server_name.
|
||||||
|
func (a *API) audit(r *http.Request, action, target string) {
|
||||||
|
p := principalFromContext(r.Context())
|
||||||
|
e := AuditEntry{Actor: auditActor(p), Action: action, ServerName: target}
|
||||||
|
if p != nil {
|
||||||
|
e.ActorUserID = p.UserID
|
||||||
|
}
|
||||||
|
a.auditEntry(r, e)
|
||||||
|
}
|
||||||
|
|
||||||
|
// auditImageChange records a confirmed image change as server.patch with the
|
||||||
|
// image it replaced and the one it set, so the audit log alone can say which
|
||||||
|
// build a world ran before it was moved.
|
||||||
|
func (a *API) auditImageChange(r *http.Request, server, from, to string) {
|
||||||
|
p := principalFromContext(r.Context())
|
||||||
|
e := AuditEntry{Actor: auditActor(p), Action: "server.patch", ServerName: server}
|
||||||
|
if p != nil {
|
||||||
|
e.ActorUserID = p.UserID
|
||||||
|
}
|
||||||
|
e.Payload = auditPayload(map[string]any{"image_from": from, "image_to": to})
|
||||||
|
a.auditEntry(r, e)
|
||||||
|
}
|
||||||
|
|
||||||
|
// auditAccount records an action a pre-session door took for the account it
|
||||||
|
// resolved (u nil: none was). The username is the actor: the door has not yet
|
||||||
|
// proven anything about the address.
|
||||||
|
func (a *API) auditAccount(r *http.Request, u *StaffUser, action, target string) {
|
||||||
|
e := AuditEntry{Actor: anonymousActor, Action: action, ServerName: target}
|
||||||
|
if u != nil {
|
||||||
|
e.Actor, e.ActorUserID = u.Username, u.ID
|
||||||
|
}
|
||||||
|
a.auditEntry(r, e)
|
||||||
|
}
|
||||||
|
|
||||||
|
// auditEntry fills the request detail into e and writes it. Source defaults to
|
||||||
|
// external; internal callers set it and the component actor themselves.
|
||||||
|
func (a *API) auditEntry(r *http.Request, e AuditEntry) {
|
||||||
|
if e.Source == "" {
|
||||||
|
e.Source = "external"
|
||||||
|
}
|
||||||
|
e.RequestID = requestIDFromContext(r.Context())
|
||||||
|
if e.Source == "external" {
|
||||||
|
if ip := a.clientIP(r); ip.IsValid() {
|
||||||
|
e.ClientIP = ip.String()
|
||||||
|
}
|
||||||
|
e.UserAgent = truncateUTF8(r.UserAgent(), auditUserAgentMax)
|
||||||
|
}
|
||||||
|
a.writeAudit(r.Context(), e)
|
||||||
|
}
|
||||||
|
|
||||||
|
// writeAudit inserts e, logging and counting a failure.
|
||||||
|
func (a *API) writeAudit(ctx context.Context, e AuditEntry) {
|
||||||
|
ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), auditWriteTimeout)
|
||||||
|
defer cancel()
|
||||||
|
if err := a.Repo.Audit(ctx, e); err != nil {
|
||||||
|
metrics.AuditWriteFailuresTotal.Inc()
|
||||||
|
log.Printf("audit: lost %s by %s (user %q, request_id=%s): %v",
|
||||||
|
e.Action, e.Actor, e.ActorUserID, e.RequestID, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// authFailure records one refused sign-in attempt: felis_auth_failures_total
|
||||||
|
// by door and reason, and an auth.<door>.failed row naming the account when
|
||||||
|
// the door resolved one (u nil: the signed-in caller if any, else anonymous).
|
||||||
|
// The doors keep their answers uniform so a prober learns nothing; the reason
|
||||||
|
// is for the operator.
|
||||||
|
func (a *API) authFailure(r *http.Request, door, reason string, u *StaffUser) {
|
||||||
|
metrics.AuthFailuresTotal.WithLabelValues(door, reason).Inc()
|
||||||
|
e := AuditEntry{Action: "auth." + door + ".failed", Payload: auditPayload(map[string]any{"reason": reason})}
|
||||||
|
switch p := principalFromContext(r.Context()); {
|
||||||
|
case u != nil:
|
||||||
|
e.Actor, e.ActorUserID = cmp.Or(u.Username, u.ID), u.ID
|
||||||
|
case p != nil:
|
||||||
|
e.Actor, e.ActorUserID = auditActor(p), p.UserID
|
||||||
|
default:
|
||||||
|
e.Actor = anonymousActor
|
||||||
|
}
|
||||||
|
a.auditEntry(r, e)
|
||||||
|
}
|
||||||
|
|
||||||
|
// passkeyCloneRejected records an assertion refused for a regressed signature
|
||||||
|
// counter. It keeps its own action so a cloned authenticator stands out from
|
||||||
|
// ordinary failures, and counts as a failure of its door. u nil: a signed-in
|
||||||
|
// step-up, attributed to the caller.
|
||||||
|
func (a *API) passkeyCloneRejected(r *http.Request, door string, u *StaffUser, credentialID string) {
|
||||||
|
metrics.AuthFailuresTotal.WithLabelValues(door, "clone_rejected").Inc()
|
||||||
|
if u == nil {
|
||||||
|
a.audit(r, "auth.passkey_clone_rejected", credentialID)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
a.auditAccount(r, u, "auth.passkey_clone_rejected", credentialID)
|
||||||
|
}
|
||||||
|
|
||||||
|
// isOTPRefusal reports whether err is a refused code (wrong, spent, or the
|
||||||
|
// account's budget locked), as opposed to a fault.
|
||||||
|
func isOTPRefusal(err error) bool {
|
||||||
|
return errors.Is(err, ErrOTPInvalid) || errors.Is(err, ErrOTPLocked) || errors.Is(err, ErrOTPAccountLocked)
|
||||||
|
}
|
||||||
|
|
||||||
|
// otpFailureReason names a refused code for authFailure.
|
||||||
|
func otpFailureReason(err error) string {
|
||||||
|
switch {
|
||||||
|
case errors.Is(err, ErrOTPAccountLocked):
|
||||||
|
return "account_locked"
|
||||||
|
case errors.Is(err, ErrOTPLocked):
|
||||||
|
return "code_locked"
|
||||||
|
}
|
||||||
|
return "bad_code"
|
||||||
|
}
|
||||||
|
|
||||||
|
// auditPayload marshals a small detail map for AuditEntry.Payload.
|
||||||
|
func auditPayload(v map[string]any) []byte {
|
||||||
|
b, _ := json.Marshal(v)
|
||||||
|
return b
|
||||||
|
}
|
||||||
|
|
||||||
|
// truncateUTF8 makes s valid UTF-8 (a header may carry any byte, a text
|
||||||
|
// column refuses invalid sequences) and cuts it to at most n bytes on a rune
|
||||||
|
// boundary.
|
||||||
|
func truncateUTF8(s string, n int) string {
|
||||||
|
s = strings.ToValidUTF8(s, "\uFFFD")
|
||||||
|
if len(s) <= n {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
for n > 0 && !utf8.RuneStart(s[n]) {
|
||||||
|
n--
|
||||||
|
}
|
||||||
|
return s[:n]
|
||||||
|
}
|
||||||
@@ -0,0 +1,177 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"net/http"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/prometheus/client_golang/prometheus/testutil"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/metrics"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Audit attribution: a row names the acting account by id and never by an
|
||||||
|
// address the caller merely asserted; refused sign-ins, throttling and logouts
|
||||||
|
// leave rows; a failed write is counted instead of vanishing.
|
||||||
|
|
||||||
|
func TestAuditActorIgnoresUnverifiedEmail(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
p *Principal
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"verified session email", &Principal{UserID: "u1", Username: "alice", Email: "[email protected]", EmailVerified: true, ViaSession: true}, "[email protected]"},
|
||||||
|
{"unverified session email", &Principal{UserID: "u2", Username: "mallory", Email: "[email protected]", ViaSession: true}, "mallory"},
|
||||||
|
{"access jwt email", &Principal{UserID: "sub", Email: "[email protected]"}, "[email protected]"},
|
||||||
|
{"no email", &Principal{UserID: "u3", Username: "bob", ViaSession: true}, "bob"},
|
||||||
|
{"id only", &Principal{UserID: "u4", ViaSession: true}, "u4"},
|
||||||
|
{"nobody", nil, anonymousActor},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if got := auditActor(c.p); got != c.want {
|
||||||
|
t.Errorf("%s: auditActor = %q, want %q", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A player who sets their address to the owner's still signs every row as
|
||||||
|
// themselves, by username and by id.
|
||||||
|
func TestAuditCannotBeSignedWithAnotherPersonsEmail(t *testing.T) {
|
||||||
|
repo := newFakeRepo()
|
||||||
|
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||||
|
repo.staff["owner"] = &StaffUser{ID: "u1", Username: "owner", Email: "[email protected]", Role: "owner", EmailVerified: true}
|
||||||
|
repo.staff["mallory"] = &StaffUser{ID: "u2", Username: "mallory", Email: "[email protected]", Role: "user", EmailVerified: true}
|
||||||
|
repo.sessions[hashCookie("tok")] = &fakeSession{userID: "u2", expiresAt: time.Unix(1_700_000_000, 0).Add(time.Hour)}
|
||||||
|
api := newTestAPI(repo, newFakeCluster())
|
||||||
|
api.External = SessionAuth{Repo: repo, RootDomain: testRoot, Now: api.now}
|
||||||
|
api.ClientIPHeader = "CF-Connecting-IP"
|
||||||
|
eh := api.ExternalHandler()
|
||||||
|
hdr := map[string]string{
|
||||||
|
"Content-Type": "application/json", "Cookie": sessionCookieName + "=tok",
|
||||||
|
"CF-Connecting-IP": "203.0.113.5", "User-Agent": "probe/1.0",
|
||||||
|
}
|
||||||
|
for _, email := range []string{"[email protected]", "[email protected]"} {
|
||||||
|
if w := do(eh, "POST", "/api/v1/account/email", `{"email":"`+email+`"}`, hdr); w.Code != http.StatusOK {
|
||||||
|
t.Fatalf("set email = %d (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The first write ran while the address was still verified; the second
|
||||||
|
// ran with the owner's address set and unverified.
|
||||||
|
last := repo.audits[len(repo.audits)-1]
|
||||||
|
if last.Actor != "mallory" || last.ActorUserID != "u2" {
|
||||||
|
t.Fatalf("audit after spoofing = actor %q user %q, want mallory/u2", last.Actor, last.ActorUserID)
|
||||||
|
}
|
||||||
|
if last.ClientIP != "203.0.113.5" || last.UserAgent != "probe/1.0" || last.RequestID == "" {
|
||||||
|
t.Fatalf("audit request detail = %+v", last)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSignInFailuresAreAuditedAndCounted(t *testing.T) {
|
||||||
|
api, repo, mailer := seedLoginEmailAPI(t)
|
||||||
|
eh := api.ExternalHandler()
|
||||||
|
noAccount := metrics.AuthFailuresTotal.WithLabelValues("login_email", "no_account")
|
||||||
|
badCode := metrics.AuthFailuresTotal.WithLabelValues("login_email", "bad_code")
|
||||||
|
n0, b0 := testutil.ToFloat64(noAccount), testutil.ToFloat64(badCode)
|
||||||
|
|
||||||
|
do(eh, "POST", "/api/v1/auth/email/verify", `{"email":"[email protected]","code":"123456"}`, jsonHeader)
|
||||||
|
if w := do(eh, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("start = %d", w.Code)
|
||||||
|
}
|
||||||
|
wrong := "000000"
|
||||||
|
if mailer.code == wrong {
|
||||||
|
wrong = "111111"
|
||||||
|
}
|
||||||
|
do(eh, "POST", "/api/v1/auth/email/verify", `{"email":"[email protected]","code":"`+wrong+`"}`, jsonHeader)
|
||||||
|
|
||||||
|
if got := testutil.ToFloat64(noAccount) - n0; got != 1 {
|
||||||
|
t.Errorf("no_account failures counted %v, want 1", got)
|
||||||
|
}
|
||||||
|
if got := testutil.ToFloat64(badCode) - b0; got != 1 {
|
||||||
|
t.Errorf("bad_code failures counted %v, want 1", got)
|
||||||
|
}
|
||||||
|
var failed []AuditEntry
|
||||||
|
for _, e := range repo.audits {
|
||||||
|
if e.Action == "auth.login_email.failed" {
|
||||||
|
failed = append(failed, e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(failed) != 2 {
|
||||||
|
t.Fatalf("failure audits = %+v, want 2", failed)
|
||||||
|
}
|
||||||
|
if failed[0].Actor != anonymousActor || failed[0].ActorUserID != "" || !strings.Contains(string(failed[0].Payload), "no_account") {
|
||||||
|
t.Errorf("unknown-address failure = %+v", failed[0])
|
||||||
|
}
|
||||||
|
if failed[1].Actor != "player" || failed[1].ActorUserID != "u1" || !strings.Contains(string(failed[1].Payload), "bad_code") {
|
||||||
|
t.Errorf("wrong-code failure = %+v", failed[1])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestThrottleAuditsOncePerEpisode(t *testing.T) {
|
||||||
|
api, repo, _ := seedLoginEmailAPI(t)
|
||||||
|
api.AuthDoorLimit = RateLimit{Burst: 1, PerMinute: 1}
|
||||||
|
clock := time.Unix(1_700_000_000, 0)
|
||||||
|
api.Now = func() time.Time { return clock }
|
||||||
|
eh := api.ExternalHandler()
|
||||||
|
options := func() int {
|
||||||
|
return do(eh, "POST", "/api/v1/auth/options", `{"email":"[email protected]"}`, jsonHeader).Code
|
||||||
|
}
|
||||||
|
throttled := func() int {
|
||||||
|
n := 0
|
||||||
|
for _, e := range repo.audits {
|
||||||
|
if e.Action == "auth.rate_limited" {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
options()
|
||||||
|
for i := 0; i < 5; i++ {
|
||||||
|
if c := options(); c != http.StatusTooManyRequests {
|
||||||
|
t.Fatalf("call %d = %d, want 429", i+2, c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if n := throttled(); n != 1 {
|
||||||
|
t.Fatalf("5 refusals left %d audit rows, want 1", n)
|
||||||
|
}
|
||||||
|
clock = clock.Add(time.Minute)
|
||||||
|
options()
|
||||||
|
options()
|
||||||
|
if n := throttled(); n != 2 {
|
||||||
|
t.Fatalf("a second episode left %d rows in total, want 2", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAuditWriteFailureIsCountedNotFatal(t *testing.T) {
|
||||||
|
api, repo, _ := seedLoginEmailAPI(t)
|
||||||
|
repo.failAudit = errors.New("db down")
|
||||||
|
before := testutil.ToFloat64(metrics.AuditWriteFailuresTotal)
|
||||||
|
if w := do(api.ExternalHandler(), "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("start with the audit store down = %d, want 202", w.Code)
|
||||||
|
}
|
||||||
|
if got := testutil.ToFloat64(metrics.AuditWriteFailuresTotal) - before; got != 1 {
|
||||||
|
t.Fatalf("audit write failures counted %v, want 1", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLogoutAuditsTheLiveSession(t *testing.T) {
|
||||||
|
api, repo, _ := seedLoginEmailAPI(t)
|
||||||
|
repo.sessions[hashCookie("tok")] = &fakeSession{userID: "u1", expiresAt: time.Unix(1_700_000_000, 0).Add(time.Hour)}
|
||||||
|
eh := api.ExternalHandler()
|
||||||
|
cookie := map[string]string{"Content-Type": "application/json", "Cookie": sessionCookieName + "=tok"}
|
||||||
|
do(eh, "POST", "/api/v1/auth/logout", "", cookie)
|
||||||
|
do(eh, "POST", "/api/v1/auth/logout", "", cookie) // already revoked: no second row
|
||||||
|
if len(repo.audits) != 1 || repo.audits[0].Action != "auth.logout" || repo.audits[0].ActorUserID != "u1" || repo.audits[0].Actor != "player" {
|
||||||
|
t.Fatalf("logout audits = %+v, want one auth.logout by player/u1", repo.audits)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestTruncateUTF8KeepsRunesWhole(t *testing.T) {
|
||||||
|
if got := truncateUTF8("ab\xffc", 10); got != "ab�c" {
|
||||||
|
t.Errorf("invalid byte = %q", got)
|
||||||
|
}
|
||||||
|
if got := truncateUTF8("猫猫", 4); got != "猫" {
|
||||||
|
t.Errorf("cut mid-rune = %q, want 猫", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -15,7 +15,10 @@ import (
|
|||||||
type Principal struct {
|
type Principal struct {
|
||||||
// UserID is the stable web identity (SSO subject → users.id).
|
// UserID is the stable web identity (SSO subject → users.id).
|
||||||
UserID string
|
UserID string
|
||||||
// Email is the audited actor identity (spec §14: audit actor = Access email).
|
// Username is the account's login name; empty for an Access-JWT caller.
|
||||||
|
Username string
|
||||||
|
// Email is the account's address. Only an Access JWT or EmailVerified vouches
|
||||||
|
// for it: a player can set any address before verifying it (auditActor).
|
||||||
Email string
|
Email string
|
||||||
// Role is "owner", "admin", or "user" (mirrors users.role).
|
// Role is "owner", "admin", or "user" (mirrors users.role).
|
||||||
Role string
|
Role string
|
||||||
|
|||||||
+20
-1
@@ -28,6 +28,12 @@ type ServerInfo struct {
|
|||||||
JavaMemory string `json:"javaMemory,omitempty"`
|
JavaMemory string `json:"javaMemory,omitempty"`
|
||||||
StorageSize string `json:"storageSize,omitempty"`
|
StorageSize string `json:"storageSize,omitempty"`
|
||||||
CPU string `json:"cpu,omitempty"`
|
CPU string `json:"cpu,omitempty"`
|
||||||
|
// IdleStopSeconds is how long the server may sit empty before idle
|
||||||
|
// auto-stop scales it down; 0 means it never idles out.
|
||||||
|
IdleStopSeconds int32 `json:"idleStopSeconds"`
|
||||||
|
// PlayerCountUnknown is true while the operator cannot read the player
|
||||||
|
// count over RCON; idle auto-stop waits until it can.
|
||||||
|
PlayerCountUnknown bool `json:"playerCountUnknown,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// CreateServerInput is the validated, structured create-server form (spec §15).
|
// CreateServerInput is the validated, structured create-server form (spec §15).
|
||||||
@@ -67,6 +73,9 @@ type ServerSpecPatch struct {
|
|||||||
// (felis-api resolves both from the same form) or both stay nil.
|
// (felis-api resolves both from the same form) or both stay nil.
|
||||||
JavaMemory *string
|
JavaMemory *string
|
||||||
Resources *corev1.ResourceRequirements
|
Resources *corev1.ResourceRequirements
|
||||||
|
// IdleStopSeconds sets idle auto-stop: 0 turns it off, anything else is the
|
||||||
|
// empty duration before the stop (already range-checked).
|
||||||
|
IdleStopSeconds *int32
|
||||||
}
|
}
|
||||||
|
|
||||||
// Cluster is the lifecycle-layer access the API depends on: reads of the
|
// Cluster is the lifecycle-layer access the API depends on: reads of the
|
||||||
@@ -94,8 +103,18 @@ type Cluster interface {
|
|||||||
// velocity registration pull (spec §7 GET /servers).
|
// velocity registration pull (spec §7 GET /servers).
|
||||||
ListServers(ctx context.Context) ([]ServerInfo, error)
|
ListServers(ctx context.Context) ([]ServerInfo, error)
|
||||||
// SetDesiredState flips spec.desiredState — the only write the API performs
|
// SetDesiredState flips spec.desiredState — the only write the API performs
|
||||||
// against the CRD (spec §9.1). It is idempotent.
|
// against the CRD (spec §9.1). It is idempotent. Flipping to Running returns a
|
||||||
|
// *MaintenanceBusyError (errors.Is ErrMaintenanceInProgress) while a restore,
|
||||||
|
// backup or file write holds the world volume.
|
||||||
SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
|
SetDesiredState(ctx context.Context, name string, state v1alpha1.DesiredState) error
|
||||||
|
// AcquireMaintenance admits one world-volume operation (internal/maintenance
|
||||||
|
// kind): ErrNotStopped unless the server is fully stopped, a
|
||||||
|
// *MaintenanceBusyError while another operation holds the volume. The check
|
||||||
|
// and the lock are one atomic write against a concurrent wake.
|
||||||
|
AcquireMaintenance(ctx context.Context, name, kind string) error
|
||||||
|
// ReleaseMaintenance drops the admission lock once the operation's Job exists
|
||||||
|
// (or could not be created). It is idempotent.
|
||||||
|
ReleaseMaintenance(ctx context.Context, name string) error
|
||||||
// CreateServer creates a MinecraftServer CRD from the validated form (spec
|
// CreateServer creates a MinecraftServer CRD from the validated form (spec
|
||||||
// §15). It returns ErrConflict if a server of that name already exists.
|
// §15). It returns ErrConflict if a server of that name already exists.
|
||||||
CreateServer(ctx context.Context, in CreateServerInput) error
|
CreateServer(ctx context.Context, in CreateServerInput) error
|
||||||
|
|||||||
@@ -6,6 +6,8 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"strconv"
|
||||||
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Sentinel errors the repository and cluster layers return so handlers can map
|
// Sentinel errors the repository and cluster layers return so handlers can map
|
||||||
@@ -44,6 +46,11 @@ var (
|
|||||||
// can answer 429 (back off / request a new code) rather than inviting another
|
// can answer 429 (back off / request a new code) rather than inviting another
|
||||||
// guess against a code that will never accept one.
|
// guess against a code that will never accept one.
|
||||||
ErrOTPLocked = errors.New("email code locked: too many attempts")
|
ErrOTPLocked = errors.New("email code locked: too many attempts")
|
||||||
|
// ErrOTPAccountLocked means the (user, purpose) has spent its wrong-code budget
|
||||||
|
// for the current window (otpFailureBudget): every code for that door is refused,
|
||||||
|
// the right one included, until the window ends. The repo returns it as an
|
||||||
|
// *OTPAccountLockedError carrying the end of the lock.
|
||||||
|
ErrOTPAccountLocked = errors.New("email codes locked for this account: too many wrong codes")
|
||||||
// ErrPasskeyChallengeInvalid means a passkey enrollment ceremony cannot be
|
// ErrPasskeyChallengeInvalid means a passkey enrollment ceremony cannot be
|
||||||
// finished: there is no live (unconsumed, unexpired) challenge for the caller and
|
// finished: there is no live (unconsumed, unexpired) challenge for the caller and
|
||||||
// purpose (Phase 6 WebAuthn bind). Like ErrOTPInvalid it is a client error — the
|
// purpose (Phase 6 WebAuthn bind). Like ErrOTPInvalid it is a client error — the
|
||||||
@@ -82,8 +89,40 @@ var (
|
|||||||
// sentinels so the handler answers 429 (a transient "too busy, retry" — the cap self-clears
|
// sentinels so the handler answers 429 (a transient "too busy, retry" — the cap self-clears
|
||||||
// as challenges expire), never a 400 that invites an immediate retry.
|
// as challenges expire), never a 400 that invites an immediate retry.
|
||||||
ErrTooManyDiscoverableChallenges = errors.New("too many discoverable login challenges in flight")
|
ErrTooManyDiscoverableChallenges = errors.New("too many discoverable login challenges in flight")
|
||||||
|
// ErrNotStopped means a world-volume operation was refused because the server is
|
||||||
|
// not fully stopped: desiredState is not Stopped, or its pod is still shutting
|
||||||
|
// down (phase Stopping) and holds the volume while it saves.
|
||||||
|
ErrNotStopped = errors.New("server is not stopped")
|
||||||
|
// ErrMaintenanceInProgress means another operation holds the server's world
|
||||||
|
// volume (internal/maintenance). Cluster methods return it wrapped in a
|
||||||
|
// *MaintenanceBusyError that names the holder.
|
||||||
|
ErrMaintenanceInProgress = errors.New("world maintenance in progress")
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// MaintenanceBusyError names what holds a server's world volume. errors.Is
|
||||||
|
// matches it against ErrMaintenanceInProgress.
|
||||||
|
type MaintenanceBusyError struct{ Kind string }
|
||||||
|
|
||||||
|
func (e *MaintenanceBusyError) Error() string {
|
||||||
|
return "world maintenance in progress: " + e.Kind
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *MaintenanceBusyError) Is(target error) bool { return target == ErrMaintenanceInProgress }
|
||||||
|
|
||||||
|
// OTPAccountLockedError is ErrOTPAccountLocked with its detail. JustLocked is set
|
||||||
|
// only on the wrong guess that spent the budget, so the handler notifies and
|
||||||
|
// audits the lock exactly once.
|
||||||
|
type OTPAccountLockedError struct {
|
||||||
|
Until time.Time
|
||||||
|
JustLocked bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *OTPAccountLockedError) Error() string {
|
||||||
|
return ErrOTPAccountLocked.Error() + " until " + e.Until.UTC().Format(time.RFC3339)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *OTPAccountLockedError) Is(target error) bool { return target == ErrOTPAccountLocked }
|
||||||
|
|
||||||
// apiError is a handler-level error carrying an HTTP status and a stable,
|
// apiError is a handler-level error carrying an HTTP status and a stable,
|
||||||
// machine-readable code. The error envelope matches the platform convention:
|
// machine-readable code. The error envelope matches the platform convention:
|
||||||
//
|
//
|
||||||
@@ -92,10 +131,19 @@ type apiError struct {
|
|||||||
status int
|
status int
|
||||||
code string
|
code string
|
||||||
msg string
|
msg string
|
||||||
|
// wait, when positive, is sent as Retry-After (whole seconds, rounded up).
|
||||||
|
wait time.Duration
|
||||||
}
|
}
|
||||||
|
|
||||||
func (e *apiError) Error() string { return e.msg }
|
func (e *apiError) Error() string { return e.msg }
|
||||||
|
|
||||||
|
// retryAfter returns a copy of e that tells the client when to retry.
|
||||||
|
func (e *apiError) retryAfter(d time.Duration) *apiError {
|
||||||
|
c := *e
|
||||||
|
c.wait = d
|
||||||
|
return &c
|
||||||
|
}
|
||||||
|
|
||||||
// newError builds an apiError with a formatted message.
|
// newError builds an apiError with a formatted message.
|
||||||
func newError(status int, code, format string, a ...any) *apiError {
|
func newError(status int, code, format string, a ...any) *apiError {
|
||||||
return &apiError{status: status, code: code, msg: fmt.Sprintf(format, a...)}
|
return &apiError{status: status, code: code, msg: fmt.Sprintf(format, a...)}
|
||||||
@@ -138,6 +186,9 @@ func writeError(w http.ResponseWriter, r *http.Request, err error) {
|
|||||||
r.Method, r.URL.Path, requestIDFromContext(r.Context()), err)
|
r.Method, r.URL.Path, requestIDFromContext(r.Context()), err)
|
||||||
ae = newError(http.StatusInternalServerError, "internal", "internal error")
|
ae = newError(http.StatusInternalServerError, "internal", "internal error")
|
||||||
}
|
}
|
||||||
|
if ae.wait > 0 {
|
||||||
|
w.Header().Set("Retry-After", strconv.FormatInt(int64((ae.wait+time.Second-1)/time.Second), 10))
|
||||||
|
}
|
||||||
body := map[string]any{
|
body := map[string]any{
|
||||||
"error": map[string]string{
|
"error": map[string]string{
|
||||||
"code": ae.code,
|
"code": ae.code,
|
||||||
|
|||||||
@@ -159,7 +159,7 @@ func (a *API) handleAccessWhitelist(w http.ResponseWriter, r *http.Request) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, principalFromContext(r.Context()).Email, "access.whitelist."+body.Action, name)
|
a.audit(r, "access.whitelist."+body.Action, name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"name": name, "action": body.Action, "player": body.Player, "output": out,
|
"name": name, "action": body.Action, "player": body.Player, "output": out,
|
||||||
})
|
})
|
||||||
@@ -249,7 +249,7 @@ func (a *API) handleAccessBan(w http.ResponseWriter, r *http.Request) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, principalFromContext(r.Context()).Email, "access.ban."+body.Action, name)
|
a.audit(r, "access.ban."+body.Action, name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"name": name, "action": body.Action, "player": body.Player, "output": out,
|
"name": name, "action": body.Action, "player": body.Player, "output": out,
|
||||||
})
|
})
|
||||||
@@ -282,7 +282,7 @@ func (a *API) handleAccessKick(w http.ResponseWriter, r *http.Request) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, principalFromContext(r.Context()).Email, "access.kick", name)
|
a.audit(r, "access.kick", name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"name": name, "player": body.Player, "output": out,
|
"name": name, "player": body.Player, "output": out,
|
||||||
})
|
})
|
||||||
@@ -354,7 +354,7 @@ func (a *API) handleAccessPermission(w http.ResponseWriter, r *http.Request) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, principalFromContext(r.Context()).Email, "access.permission."+body.Action, name)
|
a.audit(r, "access.permission."+body.Action, name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"name": name, "action": body.Action, "player": body.Player,
|
"name": name, "action": body.Action, "player": body.Player,
|
||||||
"node": body.Node, "output": out,
|
"node": body.Node, "output": out,
|
||||||
@@ -397,7 +397,7 @@ func (a *API) handleAccessGroup(w http.ResponseWriter, r *http.Request) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, principalFromContext(r.Context()).Email, "access.group."+body.Action, name)
|
a.audit(r, "access.group."+body.Action, name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"name": name, "action": body.Action, "player": body.Player, "group": body.Group, "output": out,
|
"name": name, "action": body.Action, "player": body.Player, "group": body.Group, "output": out,
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -229,7 +229,7 @@ func (a *API) handleLinkVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "account.link", "")
|
a.audit(r, "account.link", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"linked": true, "mc_uuid": mcUUID, "auth_source": authSource,
|
"linked": true, "mc_uuid": mcUUID, "auth_source": authSource,
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -114,11 +114,10 @@ func (a *API) handleMigrateStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
// Internal-face event: attribute to the in-game initiator, Source 'internal'.
|
// Internal-face event: attribute to the in-game initiator, Source 'internal'.
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: "mc:" + mcUUID,
|
Actor: "mc:" + mcUUID,
|
||||||
Source: "internal",
|
Source: "internal",
|
||||||
Action: "account.migrate.start",
|
Action: "account.migrate.start",
|
||||||
RequestID: requestIDFromContext(r.Context()),
|
|
||||||
})
|
})
|
||||||
writeJSON(w, http.StatusCreated, map[string]any{"started": true, "state": "initiated"})
|
writeJSON(w, http.StatusCreated, map[string]any{"started": true, "state": "initiated"})
|
||||||
}
|
}
|
||||||
@@ -206,6 +205,13 @@ func (a *API) handleMigrateConfirmOTPStart(w http.ResponseWriter, r *http.Reques
|
|||||||
}
|
}
|
||||||
// Per-recipient cooldown, namespaced apart from the other OTP doors so they never
|
// Per-recipient cooldown, namespaced apart from the other OTP doors so they never
|
||||||
// perturb each other's throttle.
|
// perturb each other's throttle.
|
||||||
|
if until, err := a.Repo.OTPLockedUntil(r.Context(), p.UserID, otpPurposeMigrate, a.now()); err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
} else if !until.IsZero() {
|
||||||
|
writeOTPAccountLocked(w, r, until, a.now())
|
||||||
|
return
|
||||||
|
}
|
||||||
emailKey := "migrate:confirm:" + strings.ToLower(p.Email)
|
emailKey := "migrate:confirm:" + strings.ToLower(p.Email)
|
||||||
lim := a.otpLimiter()
|
lim := a.otpLimiter()
|
||||||
emailAt, ok := lim.reserve(emailKey, otpResendCooldown)
|
emailAt, ok := lim.reserve(emailKey, otpResendCooldown)
|
||||||
@@ -240,7 +246,7 @@ func (a *API) handleMigrateConfirmOTPStart(w http.ResponseWriter, r *http.Reques
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
committed = true
|
committed = true
|
||||||
a.audit(r, auditActor(p), "account.migrate.confirm_otp_sent", "")
|
a.audit(r, "account.migrate.confirm_otp_sent", "")
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -267,7 +273,16 @@ func (a *API) handleMigrateConfirmOTPVerify(w http.ResponseWriter, r *http.Reque
|
|||||||
if _, ok := a.requireInitiatedMigration(w, r, p.UserID); !ok {
|
if _, ok := a.requireInitiatedMigration(w, r, p.UserID); !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
switch err := a.Repo.ConsumeLoginEmailOTP(r.Context(), p.UserID, otpPurposeMigrate, otpCodeHash(code), a.now()); {
|
var lock *OTPAccountLockedError
|
||||||
|
err := a.Repo.ConsumeLoginEmailOTP(r.Context(), p.UserID, otpPurposeMigrate, otpCodeHash(code), a.now())
|
||||||
|
if isOTPRefusal(err) {
|
||||||
|
a.authFailure(r, "migrate_confirm", otpFailureReason(err), nil)
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case errors.As(err, &lock):
|
||||||
|
a.noteOTPLock(r, err, p.UserID, otpPurposeMigrate)
|
||||||
|
writeOTPAccountLocked(w, r, lock.Until, a.now())
|
||||||
|
return
|
||||||
case errors.Is(err, ErrOTPLocked):
|
case errors.Is(err, ErrOTPLocked):
|
||||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_locked",
|
writeError(w, r, newError(http.StatusTooManyRequests, "otp_locked",
|
||||||
"too many incorrect attempts; request a new code"))
|
"too many incorrect attempts; request a new code"))
|
||||||
@@ -288,7 +303,7 @@ func (a *API) handleMigrateConfirmOTPVerify(w http.ResponseWriter, r *http.Reque
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.migrate.confirmed", "")
|
a.audit(r, "account.migrate.confirmed", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"confirmed": true})
|
writeJSON(w, http.StatusOK, map[string]any{"confirmed": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -375,6 +390,7 @@ func (a *API) handleMigrateConfirmPasskeyFinish(w http.ResponseWriter, r *http.R
|
|||||||
sessionData, err := a.Repo.ConsumePasskeyChallengeByUser(r.Context(), p.UserID, passkeyPurposeMigrate, a.now())
|
sessionData, err := a.Repo.ConsumePasskeyChallengeByUser(r.Context(), p.UserID, passkeyPurposeMigrate, a.now())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
||||||
|
a.authFailure(r, "migrate_passkey", "challenge_invalid", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey confirmation could not be completed; begin again"))
|
"passkey confirmation could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -389,6 +405,7 @@ func (a *API) handleMigrateConfirmPasskeyFinish(w http.ResponseWriter, r *http.R
|
|||||||
}
|
}
|
||||||
va, err := a.Passkey.FinishLogin(migratePasskeyUser(p, creds), sessionData, bytes.NewReader(req.Assertion))
|
va, err := a.Passkey.FinishLogin(migratePasskeyUser(p, creds), sessionData, bytes.NewReader(req.Assertion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
a.authFailure(r, "migrate_passkey", "bad_assertion", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey confirmation could not be completed; begin again"))
|
"passkey confirmation could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -400,7 +417,7 @@ func (a *API) handleMigrateConfirmPasskeyFinish(w http.ResponseWriter, r *http.R
|
|||||||
// next login.
|
// next login.
|
||||||
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
||||||
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
||||||
a.audit(r, auditActor(p), "auth.passkey_clone_rejected", va.CredentialID)
|
a.passkeyCloneRejected(r, "migrate_passkey", nil, va.CredentialID)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey confirmation could not be completed; begin again"))
|
"passkey confirmation could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -417,7 +434,7 @@ func (a *API) handleMigrateConfirmPasskeyFinish(w http.ResponseWriter, r *http.R
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.migrate.confirmed", "")
|
a.audit(r, "account.migrate.confirmed", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"confirmed": true})
|
writeJSON(w, http.StatusOK, map[string]any{"confirmed": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -494,7 +511,7 @@ func (a *API) handleMigrateIssueCode(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.migrate.code_issued", targetID)
|
a.audit(r, "account.migrate.code_issued", targetID)
|
||||||
writeJSON(w, http.StatusCreated, map[string]any{"code": code, "expires_at": expiresAt.UTC()})
|
writeJSON(w, http.StatusCreated, map[string]any{"code": code, "expires_at": expiresAt.UTC()})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -532,7 +549,7 @@ func (a *API) handleMigrateRedeem(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.migrate.redeemed", sourceUserID)
|
a.audit(r, "account.migrate.redeemed", sourceUserID)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"migrated": true,
|
"migrated": true,
|
||||||
"servers_moved": len(moved),
|
"servers_moved": len(moved),
|
||||||
|
|||||||
@@ -1,6 +1,10 @@
|
|||||||
package api
|
package api
|
||||||
|
|
||||||
import "net/http"
|
import (
|
||||||
|
"net/http"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/metrics"
|
||||||
|
)
|
||||||
|
|
||||||
// Passwordless auth handlers (spec §B). Staff (Owner/Operator) authenticate via
|
// Passwordless auth handlers (spec §B). Staff (Owner/Operator) authenticate via
|
||||||
// email-OTP / passkey + in-game approve on op.console; players via bind code or
|
// email-OTP / passkey + in-game approve on op.console; players via bind code or
|
||||||
@@ -10,10 +14,16 @@ import "net/http"
|
|||||||
|
|
||||||
// handleLogout revokes the presented session and clears the cookie (spec §B). It
|
// handleLogout revokes the presented session and clears the cookie (spec §B). It
|
||||||
// is mounted Public and idempotent: it reads the cookie directly, so it works even
|
// is mounted Public and idempotent: it reads the cookie directly, so it works even
|
||||||
// when the session has already expired and never errors on a missing one.
|
// when the session has already expired and never errors on a missing one. Ending
|
||||||
|
// a live session is audited under its account; a dead cookie leaves no row.
|
||||||
func (a *API) handleLogout(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handleLogout(w http.ResponseWriter, r *http.Request) {
|
||||||
if c, err := r.Cookie(sessionCookieName); err == nil && c.Value != "" {
|
if c, err := r.Cookie(sessionCookieName); err == nil && c.Value != "" {
|
||||||
_ = a.Repo.RevokeSession(r.Context(), hashCookie(c.Value))
|
hash := hashCookie(c.Value)
|
||||||
|
u, uerr := a.Repo.SessionUser(r.Context(), hash, a.now())
|
||||||
|
if err := a.Repo.RevokeSession(r.Context(), hash); err == nil && uerr == nil {
|
||||||
|
metrics.SessionsRevokedTotal.WithLabelValues("logout").Inc()
|
||||||
|
a.auditEntry(r, AuditEntry{Actor: u.Username, ActorUserID: u.ID, Action: "auth.logout"})
|
||||||
|
}
|
||||||
}
|
}
|
||||||
clearSessionCookie(w)
|
clearSessionCookie(w)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
||||||
|
|||||||
@@ -24,12 +24,9 @@ import (
|
|||||||
// this separator).
|
// this separator).
|
||||||
// - No principal. The throttle cannot key off a user id (there is none yet); it
|
// - No principal. The throttle cannot key off a user id (there is none yet); it
|
||||||
// keys off the typed recipient address, the same anti-bomb dimension the onboard
|
// keys off the typed recipient address, the same anti-bomb dimension the onboard
|
||||||
// start uses. Per-source (client-IP) aggregate limiting is deliberately NOT done
|
// start uses. Volume from one client is bounded separately by the per-address
|
||||||
// here: cooldownLimiter is a one-per-window primitive, so keying it on client IP
|
// token bucket every public auth door sits behind (throttleAuthDoor), and total
|
||||||
// would false-positive on shared egress (CGNAT / office NAT), and behind
|
// mail by the install-wide mail budget (ratelimit.go).
|
||||||
// Cloudflare RemoteAddr is the proxy anyway. The only real harm — bombing one
|
|
||||||
// mailbox — is already bounded per recipient; volumetric per-source limiting
|
|
||||||
// belongs at the edge.
|
|
||||||
// - Refuse staff. Like handleBindRedeem this public door provably never mints a
|
// - Refuse staff. Like handleBindRedeem this public door provably never mints a
|
||||||
// session for an admin identity: op.console stays behind Zero Trust (and its own
|
// session for an admin identity: op.console stays behind Zero Trust (and its own
|
||||||
// in-game approval gate). The refusal happens only AFTER a valid code is
|
// in-game approval gate). The refusal happens only AFTER a valid code is
|
||||||
@@ -84,6 +81,13 @@ func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The install-wide mail budget is checked before the address is resolved,
|
||||||
|
// so while it is spent every address gets the same 429.
|
||||||
|
if err := a.checkMailBudget(); err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
// Atomically reserve the per-recipient cooldown BEFORE any work, so a burst of
|
// Atomically reserve the per-recipient cooldown BEFORE any work, so a burst of
|
||||||
// truly concurrent starts yields exactly one winner and each admitted send is one
|
// truly concurrent starts yields exactly one winner and each admitted send is one
|
||||||
// real, non-idempotent email. The key is namespaced apart from the onboard door's
|
// real, non-idempotent email. The key is namespaced apart from the onboard door's
|
||||||
@@ -126,6 +130,19 @@ func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A locked door (wrong-code budget spent) gets the same neutral 202 and no
|
||||||
|
// mail: the owner was told by the lock notice, and a distinct answer here
|
||||||
|
// would tell a prober the address has an account.
|
||||||
|
switch until, err := a.Repo.OTPLockedUntil(r.Context(), u.ID, otpPurposeLogin, a.now()); {
|
||||||
|
case err != nil:
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
case !until.IsZero():
|
||||||
|
committed = true
|
||||||
|
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
code, err := newEmailOTP()
|
code, err := newEmailOTP()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
@@ -150,7 +167,7 @@ func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
committed = true
|
committed = true
|
||||||
a.audit(r, u.Username, "auth.login_email.otp_sent", "")
|
a.auditAccount(r, u, "auth.login_email.otp_sent", "")
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -202,6 +219,7 @@ func (a *API) handleLoginEmailVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
// Uniform with a wrong code: a caller probing whether an address has an account
|
// Uniform with a wrong code: a caller probing whether an address has an account
|
||||||
// gets the same invalid_code either way. (The /auth/options oracle is the
|
// gets the same invalid_code either way. (The /auth/options oracle is the
|
||||||
// sanctioned place to learn existence; this door does not double as one.)
|
// sanctioned place to learn existence; this door does not double as one.)
|
||||||
|
a.authFailure(r, "login_email", "no_account", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "email code is invalid or expired"))
|
writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "email code is invalid or expired"))
|
||||||
return
|
return
|
||||||
case err != nil:
|
case err != nil:
|
||||||
@@ -220,7 +238,10 @@ func (a *API) handleLoginEmailVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
// verified; touching the row here would let a stale OTP-snapshot address overwrite
|
// verified; touching the row here would let a stale OTP-snapshot address overwrite
|
||||||
// the live one and could 500 a correct code on a spurious collision.
|
// the live one and could 500 a correct code on a spurious collision.
|
||||||
switch err := a.Repo.ConsumeLoginEmailOTP(r.Context(), u.ID, otpPurposeLogin, otpCodeHash(code), a.now()); {
|
switch err := a.Repo.ConsumeLoginEmailOTP(r.Context(), u.ID, otpPurposeLogin, otpCodeHash(code), a.now()); {
|
||||||
case errors.Is(err, ErrOTPInvalid), errors.Is(err, ErrOTPLocked):
|
case errors.Is(err, ErrOTPInvalid), errors.Is(err, ErrOTPLocked), errors.Is(err, ErrOTPAccountLocked):
|
||||||
|
// The account lock answers the same way; its owner hears about it by mail.
|
||||||
|
a.noteOTPLock(r, err, u.ID, otpPurposeLogin)
|
||||||
|
a.authFailure(r, "login_email", otpFailureReason(err), u)
|
||||||
// Both a wrong/expired code and an attempt-exhausted one return the SAME 400
|
// Both a wrong/expired code and an attempt-exhausted one return the SAME 400
|
||||||
// invalid_code, byte-identical to the unknown-account branch above. Surfacing
|
// invalid_code, byte-identical to the unknown-account branch above. Surfacing
|
||||||
// otp_locked as a distinct 429 (as the authenticated onboarding door does) would
|
// otp_locked as a distinct 429 (as the authenticated onboarding door does) would
|
||||||
@@ -244,6 +265,7 @@ func (a *API) handleLoginEmailVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
// Staff means anything above role=user: an admin OR the role=owner identity. The
|
// Staff means anything above role=user: an admin OR the role=owner identity. The
|
||||||
// player door must yield only player sessions.
|
// player door must yield only player sessions.
|
||||||
if u.Role != "user" {
|
if u.Role != "user" {
|
||||||
|
a.authFailure(r, "login_email", "staff_account", u)
|
||||||
writeError(w, r, newError(http.StatusForbidden, "staff_account",
|
writeError(w, r, newError(http.StatusForbidden, "staff_account",
|
||||||
"that account is staff; sign in at the operator console"))
|
"that account is staff; sign in at the operator console"))
|
||||||
return
|
return
|
||||||
@@ -260,7 +282,7 @@ func (a *API) handleLoginEmailVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
setSessionCookie(w, token, expires)
|
setSessionCookie(w, token, expires)
|
||||||
a.audit(r, u.Username, "auth.login_email", "")
|
a.auditAccount(r, u, "auth.login_email", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"user_id": u.ID,
|
"user_id": u.ID,
|
||||||
"role": u.Role,
|
"role": u.Role,
|
||||||
|
|||||||
@@ -16,11 +16,10 @@ import (
|
|||||||
// address has an account precisely because THIS endpoint is the one sanctioned place
|
// address has an account precisely because THIS endpoint is the one sanctioned place
|
||||||
// existence is revealed. An empty methods array means "no (verified) account". That
|
// existence is revealed. An empty methods array means "no (verified) account". That
|
||||||
// makes it a mass-enumeration surface by design — an accepted product decision, the
|
// makes it a mass-enumeration surface by design — an accepted product decision, the
|
||||||
// same one the email door's header records. It is bounded only at the edge: the
|
// same one the email door's header records. The handler sends no mail and mutates
|
||||||
// handler sends no mail and mutates nothing, so a per-recipient cooldown would merely
|
// nothing, so a per-recipient cooldown would merely block a legitimate retry; what
|
||||||
// block a legitimate retry, and per-source (client-IP) limiting is the edge's job
|
// bounds enumeration is the per-client-address token bucket shared by every public
|
||||||
// (behind Cloudflare RemoteAddr is the proxy, and CGNAT would false-positive) — see
|
// auth door (throttleAuthDoor in ratelimit.go).
|
||||||
// the handlers_auth_email.go header for the same reasoning.
|
|
||||||
//
|
//
|
||||||
// It never reveals STAFFNESS. Methods are computed by the SAME rule for every resolved
|
// It never reveals STAFFNESS. Methods are computed by the SAME rule for every resolved
|
||||||
// account — no role branch, no operator hint — so a staff email and a player email in
|
// account — no role branch, no operator hint — so a staff email and a player email in
|
||||||
|
|||||||
@@ -5,7 +5,9 @@ import (
|
|||||||
"encoding/json"
|
"encoding/json"
|
||||||
"errors"
|
"errors"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"strconv"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
)
|
)
|
||||||
@@ -191,6 +193,78 @@ func TestBackupNow(t *testing.T) {
|
|||||||
t.Fatalf("code = %d, want 400", w.Code)
|
t.Fatalf("code = %d, want 400", w.Code)
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
|
|
||||||
|
// data-durability-9: an owner's backups are rationed per server; an admin's
|
||||||
|
// are not.
|
||||||
|
t.Run("owner inside the cooldown -> 429 backup_cooldown with Retry-After", func(t *testing.T) {
|
||||||
|
api, _, _, backuper := mk()
|
||||||
|
api.BackupCooldown = 10 * time.Minute
|
||||||
|
api.Now = time.Now // the fake stamps backup.create audits with the wall clock
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("first backup: code = %d (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "backup_cooldown" {
|
||||||
|
t.Fatalf("second backup: code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if ra, _ := strconv.Atoi(w.Header().Get("Retry-After")); ra < 590 || ra > 600 {
|
||||||
|
t.Fatalf("Retry-After = %q, want about 600", w.Header().Get("Retry-After"))
|
||||||
|
}
|
||||||
|
if backuper.calls != 1 {
|
||||||
|
t.Fatalf("backuper called %d times, want 1", backuper.calls)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("cooldown elapsed -> 202", func(t *testing.T) {
|
||||||
|
api, repo, _, _ := mk()
|
||||||
|
api.BackupCooldown = 10 * time.Minute
|
||||||
|
api.Now = time.Now
|
||||||
|
repo.backupRequested = map[string]time.Time{"survival": time.Now().Add(-11 * time.Minute)}
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("admin bypasses the cooldown and the store cap", func(t *testing.T) {
|
||||||
|
api, repo, _, backuper := mk()
|
||||||
|
api.BackupCooldown = 10 * time.Minute
|
||||||
|
api.BackupStoreCap = 100
|
||||||
|
api.Now = time.Now
|
||||||
|
repo.backupRequested = map[string]time.Time{"survival": time.Now()}
|
||||||
|
repo.backups = []fakeBackup{{view: BackupView{ID: "b1", ServerName: "other", Status: "present", SizeBytes: 500}}}
|
||||||
|
api.External = staticExternal{p: &Principal{UserID: "admin1", Email: "[email protected]",
|
||||||
|
Role: "admin", ViaAdminAccess: true}}
|
||||||
|
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if backuper.calls != 1 {
|
||||||
|
t.Fatal("the admin's backup did not start")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("owner with the store at its cap -> 507 backup_store_full", func(t *testing.T) {
|
||||||
|
api, repo, _, backuper := mk()
|
||||||
|
api.BackupStoreCap = 1000
|
||||||
|
repo.backups = []fakeBackup{
|
||||||
|
{view: BackupView{ID: "b1", ServerName: "other", Status: "present", SizeBytes: 600}},
|
||||||
|
{view: BackupView{ID: "b2", ServerName: "survival", Status: "present", SizeBytes: 400}},
|
||||||
|
{view: BackupView{ID: "b3", ServerName: "survival", Status: "deleted", SizeBytes: 9000}},
|
||||||
|
}
|
||||||
|
api.External = staticExternal{p: owner}
|
||||||
|
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||||
|
if w.Code != http.StatusInsufficientStorage || decodeErr(t, w) != "backup_store_full" {
|
||||||
|
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if backuper.calls != 0 {
|
||||||
|
t.Fatal("a full store still started a backup")
|
||||||
|
}
|
||||||
|
repo.backups[0].view.Status = "deleted"
|
||||||
|
if w := do(api.ExternalHandler(), "POST", path, "", nil); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("below the cap: code = %d (%s)", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestInternalBackup exercises POST /api/v1/internal/servers/{name}/backup, the
|
// TestInternalBackup exercises POST /api/v1/internal/servers/{name}/backup, the
|
||||||
|
|||||||
@@ -1,11 +1,14 @@
|
|||||||
package api
|
package api
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
"errors"
|
"errors"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strings"
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
"felis.lolicon.best/internal/maintenance"
|
||||||
"felis.lolicon.best/internal/naming"
|
"felis.lolicon.best/internal/naming"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -69,7 +72,10 @@ func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
|
|||||||
// re-claims a released server could otherwise resurrect user A's world (the
|
// re-claims a released server could otherwise resurrect user A's world (the
|
||||||
// backup still carries former_owner=A), a data leak. Admin skips this check.
|
// backup still carries former_owner=A), a data leak. Admin skips this check.
|
||||||
// ⑦ stopped gate: the world PVC must be free, so restore is refused unless the
|
// ⑦ stopped gate: the world PVC must be free, so restore is refused unless the
|
||||||
// server is fully stopped.
|
// server is fully stopped. The world-volume lock (internal/maintenance) then
|
||||||
|
// makes that atomic against a wake and refuses a second restore, backup or
|
||||||
|
// file write on the same world with 409 maintenance_in_progress until the
|
||||||
|
// restore Job finishes.
|
||||||
// ⑧ hand off to the Restorer. Restore is asynchronous (a restore Job, like an
|
// ⑧ hand off to the Restorer. Restore is asynchronous (a restore Job, like an
|
||||||
// image build Job), so success means "enqueued" and the handler answers 202.
|
// image build Job), so success means "enqueued" and the handler answers 202.
|
||||||
//
|
//
|
||||||
@@ -152,7 +158,9 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
|||||||
// Refuse unless the server is fully stopped — Ready means it is up, and any
|
// Refuse unless the server is fully stopped — Ready means it is up, and any
|
||||||
// desiredState other than Stopped means it is up or coming up and still owns the
|
// desiredState other than Stopped means it is up or coming up and still owns the
|
||||||
// RWO volume (spec §141 readiness is an RCON probe; DesiredStopped is the
|
// RWO volume (spec §141 readiness is an RCON probe; DesiredStopped is the
|
||||||
// intent). This yields a specific 409 instead of a restore Job that cannot mount.
|
// intent). This is the early, readable refusal from a snapshot; acquireWorld
|
||||||
|
// below is the atomic one (RWO is per node, so on a single node a restore Job
|
||||||
|
// WOULD mount beside a running server).
|
||||||
info, err := a.Cluster.GetServer(r.Context(), name)
|
info, err := a.Cluster.GetServer(r.Context(), name)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
a.writeLookupError(w, r, err)
|
a.writeLookupError(w, r, err)
|
||||||
@@ -186,13 +194,23 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// World-volume lock (internal/maintenance). The stopped gate above reads a
|
||||||
|
// snapshot; this is the atomic check, and it keeps a wake — the owner's, or a
|
||||||
|
// player's join through velocity — from booting the server on a half-extracted
|
||||||
|
// world until the restore Job has finished.
|
||||||
|
release, ok := a.acquireWorld(w, r, name, maintenance.KindRestore, "stop the server before restoring a backup")
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer release()
|
||||||
|
|
||||||
if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil {
|
if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil {
|
||||||
// ErrNotFound (server vanished from the execution backend) → 404; else 500.
|
// ErrNotFound (server vanished from the execution backend) → 404; else 500.
|
||||||
a.writeLookupError(w, r, err)
|
a.writeLookupError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "backup.restore", name)
|
a.audit(r, "backup.restore", name)
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||||
"name": name,
|
"name": name,
|
||||||
"status": "restoring",
|
"status": "restoring",
|
||||||
@@ -212,9 +230,10 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
|||||||
// ② ServerByName — an unknown server is 404
|
// ② ServerByName — an unknown server is 404
|
||||||
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
|
// ③ owner-or-admin, else 403 (an unowned server passes only for admin, so a
|
||||||
// released world can still be snapshotted by an operator before disposal)
|
// released world can still be snapshotted by an operator before disposal)
|
||||||
// ④ stopped gate: the world PVC is RWO and held by a running server, so a backup
|
// ④ stopped gate: refuse unless the server is fully stopped, so the archive is
|
||||||
// Job cannot double-mount it — refuse unless the server is fully stopped. This
|
// quiescent and non-torn. The world-volume lock (internal/maintenance) keeps it
|
||||||
// also guarantees a quiescent, non-torn archive.
|
// that way until the backup Job finishes: a wake meanwhile, or a second
|
||||||
|
// restore/backup/file write, gets 409 maintenance_in_progress.
|
||||||
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
|
// ⑤ hand off to the Backuper. Backup is asynchronous (a backup Job), so success
|
||||||
// means "enqueued" and the handler answers 202.
|
// means "enqueued" and the handler answers 202.
|
||||||
//
|
//
|
||||||
@@ -238,8 +257,47 @@ func (a *API) handleBackupNow(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, errForbidden)
|
writeError(w, r, errForbidden)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
if !p.IsAdmin() {
|
||||||
|
if err := a.backupAllowance(r.Context(), name); err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
a.enqueueBackup(w, r, name, rec, p.Email, "external")
|
a.enqueueBackup(w, r, name, rec, auditActor(p), "external")
|
||||||
|
}
|
||||||
|
|
||||||
|
// backupAllowance rations an owner's on-demand backups (data-durability-9):
|
||||||
|
// one per BackupCooldown per server, and none while the present backups fill
|
||||||
|
// BackupStoreCap. The owner's older backups are pruned by the Job itself
|
||||||
|
// ([archive] manual_keep), so these two gates bound the rate and the total.
|
||||||
|
func (a *API) backupAllowance(ctx context.Context, name string) error {
|
||||||
|
if a.BackupCooldown > 0 {
|
||||||
|
now := a.now()
|
||||||
|
last, err := a.Repo.LastBackupRequest(ctx, name, now.Add(-a.BackupCooldown))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if !last.IsZero() {
|
||||||
|
wait := last.Add(a.BackupCooldown).Sub(now)
|
||||||
|
if wait > 0 {
|
||||||
|
return newError(http.StatusTooManyRequests, "backup_cooldown",
|
||||||
|
"a backup of this server was started %s ago; the next one can start in %s",
|
||||||
|
now.Sub(last).Round(time.Second), wait.Round(time.Second)).retryAfter(wait)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if a.BackupStoreCap > 0 {
|
||||||
|
used, err := a.Repo.BackupStoreBytes(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if used >= a.BackupStoreCap {
|
||||||
|
return newError(http.StatusInsufficientStorage, "backup_store_full",
|
||||||
|
"the backup store is full; ask an administrator to free space")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleInternalBackup is the internal-face backup trigger. The break-glass console
|
// handleInternalBackup is the internal-face backup trigger. The break-glass console
|
||||||
@@ -294,9 +352,10 @@ func (a *API) handleInternalBackup(w http.ResponseWriter, r *http.Request) {
|
|||||||
// (Principal vs trusted service token) and the audit actor/source — keeping the
|
// (Principal vs trusted service token) and the audit actor/source — keeping the
|
||||||
// security-critical stopped-gate single-sourced so the two faces cannot diverge.
|
// security-critical stopped-gate single-sourced so the two faces cannot diverge.
|
||||||
func (a *API) enqueueBackup(w http.ResponseWriter, r *http.Request, name string, rec *ServerRecord, actor, source string) {
|
func (a *API) enqueueBackup(w http.ResponseWriter, r *http.Request, name string, rec *ServerRecord, actor, source string) {
|
||||||
// Stopped gate: the world PVC is RWO and held by a running server, so a backup
|
// Stopped gate (mirrors the restore gate): Ready means it is up; any
|
||||||
// Job cannot double-mount it (mirrors the restore gate). Ready means it is up;
|
// desiredState other than Stopped means it is up or coming up. RWO is per node,
|
||||||
// any desiredState other than Stopped means it owns the RWO volume.
|
// so on a single node the Job WOULD mount beside a live server and archive a
|
||||||
|
// torn world; acquireWorld below makes this check atomic.
|
||||||
info, err := a.Cluster.GetServer(r.Context(), name)
|
info, err := a.Cluster.GetServer(r.Context(), name)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
a.writeLookupError(w, r, err)
|
a.writeLookupError(w, r, err)
|
||||||
@@ -329,15 +388,24 @@ func (a *API) enqueueBackup(w http.ResponseWriter, r *http.Request, name string,
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// World-volume lock (internal/maintenance): a server woken mid-backup would
|
||||||
|
// leave a torn archive that a later restore makes permanent.
|
||||||
|
release, ok := a.acquireWorld(w, r, name, maintenance.KindBackup, "stop the server before backing up its world")
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer release()
|
||||||
|
|
||||||
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
|
if err := a.Backuper.Backup(r.Context(), name, rec.OwnerID); err != nil {
|
||||||
a.writeLookupError(w, r, err)
|
a.writeLookupError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
e := AuditEntry{Actor: actor, Source: source, Action: "backup.create", ServerName: name}
|
||||||
Actor: actor, Source: source, Action: "backup.create",
|
if p := principalFromContext(r.Context()); p != nil {
|
||||||
ServerName: name, RequestID: requestIDFromContext(r.Context()),
|
e.ActorUserID = p.UserID
|
||||||
})
|
}
|
||||||
|
a.auditEntry(r, e)
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||||
"name": name,
|
"name": name,
|
||||||
"status": "backing_up",
|
"status": "backing_up",
|
||||||
|
|||||||
@@ -117,6 +117,6 @@ func (a *API) handleCommand(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "console.command", name)
|
a.audit(r, "console.command", name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"name": name, "output": output})
|
writeJSON(w, http.StatusOK, map[string]any{"name": name, "output": output})
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"net/http"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Control-plane database backup freshness (admin-tier, read-only). The backups
|
||||||
|
// themselves run on the host — felis-db-backup.timer calls `felis db backup`,
|
||||||
|
// which records its newest success in platform_settings[dbbackup.StatusKey] — so
|
||||||
|
// the API can report them without reaching the host's backup directory. The
|
||||||
|
// panel shows this next to the update window: both answer "is it safe to change
|
||||||
|
// something on this install right now".
|
||||||
|
|
||||||
|
// dbBackupView is the wire shape. Last is null until the first backup has been
|
||||||
|
// recorded; Stale is true for a missing record too, so the panel has a single
|
||||||
|
// flag for "nobody could restore today's state".
|
||||||
|
type dbBackupView struct {
|
||||||
|
Last *dbbackup.Status `json:"last"`
|
||||||
|
Stale bool `json:"stale"`
|
||||||
|
MaxAgeSeconds int64 `json:"max_age_seconds"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleGetDBBackup reports the newest recorded control-plane database backup.
|
||||||
|
// Only a missing key reads as "never"; any other store error is a 500 so a DB
|
||||||
|
// blip is never shown as a healthy or an absent backup.
|
||||||
|
func (a *API) handleGetDBBackup(w http.ResponseWriter, r *http.Request) {
|
||||||
|
view := dbBackupView{Stale: true, MaxAgeSeconds: int64(dbbackup.StaleAfter.Seconds())}
|
||||||
|
raw, err := a.Repo.GetSetting(r.Context(), dbbackup.StatusKey)
|
||||||
|
switch {
|
||||||
|
case errors.Is(err, ErrNotFound):
|
||||||
|
writeJSON(w, http.StatusOK, view)
|
||||||
|
return
|
||||||
|
case err != nil:
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var st dbbackup.Status
|
||||||
|
if err := json.Unmarshal(raw, &st); err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
view.Last = &st
|
||||||
|
view.Stale = st.At.IsZero() || a.now().Sub(st.At) > dbbackup.StaleAfter
|
||||||
|
writeJSON(w, http.StatusOK, view)
|
||||||
|
}
|
||||||
@@ -0,0 +1,87 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The panel's backup card reads one flag, stale, so these pin when it is set:
|
||||||
|
// never backed up, too old, and not for a fresh backup. A store outage must
|
||||||
|
// not read as either answer.
|
||||||
|
|
||||||
|
func getDBBackup(t *testing.T, api *API) (int, dbBackupView) {
|
||||||
|
t.Helper()
|
||||||
|
w := do(api.ExternalHandler(), "GET", "/api/v1/platform/db-backup", "", nil)
|
||||||
|
var v dbBackupView
|
||||||
|
if w.Code == http.StatusOK {
|
||||||
|
if err := json.Unmarshal(w.Body.Bytes(), &v); err != nil {
|
||||||
|
t.Fatalf("body not JSON: %v (%s)", err, w.Body.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return w.Code, v
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDBBackupNeverRecordedIsStale(t *testing.T) {
|
||||||
|
api, _ := seedUpdatesAPI(t)
|
||||||
|
code, v := getDBBackup(t, api)
|
||||||
|
if code != http.StatusOK || v.Last != nil || !v.Stale {
|
||||||
|
t.Fatalf("never backed up = %d %+v, want 200 last=null stale", code, v)
|
||||||
|
}
|
||||||
|
if v.MaxAgeSeconds != int64(dbbackup.StaleAfter/time.Second) {
|
||||||
|
t.Fatalf("max_age_seconds = %d", v.MaxAgeSeconds)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDBBackupFreshness(t *testing.T) {
|
||||||
|
now := time.Date(2026, 9, 24, 12, 0, 0, 0, time.UTC)
|
||||||
|
for _, tc := range []struct {
|
||||||
|
name string
|
||||||
|
age time.Duration
|
||||||
|
stale bool
|
||||||
|
}{
|
||||||
|
{"taken this morning", 8 * time.Hour, false},
|
||||||
|
{"yesterday's, timer slightly late", 25 * time.Hour, false},
|
||||||
|
{"missed a day", 27 * time.Hour, true},
|
||||||
|
} {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
api, repo := seedUpdatesAPI(t)
|
||||||
|
api.Now = func() time.Time { return now }
|
||||||
|
st := dbbackup.Status{At: now.Add(-tc.age), Name: "felis-db-x-daily.tar", Label: "daily", SizeBytes: 4096, SchemaVersion: 31, Dir: "/var/lib/felis/db-backups"}
|
||||||
|
raw, _ := json.Marshal(st)
|
||||||
|
repo.settings[dbbackup.StatusKey] = raw
|
||||||
|
|
||||||
|
code, v := getDBBackup(t, api)
|
||||||
|
if code != http.StatusOK || v.Last == nil {
|
||||||
|
t.Fatalf("code %d, view %+v", code, v)
|
||||||
|
}
|
||||||
|
if v.Stale != tc.stale {
|
||||||
|
t.Fatalf("stale = %v, want %v", v.Stale, tc.stale)
|
||||||
|
}
|
||||||
|
if v.Last.Name != st.Name || v.Last.SizeBytes != 4096 || v.Last.SchemaVersion != 31 || !v.Last.At.Equal(st.At) {
|
||||||
|
t.Fatalf("record not passed through: %+v", v.Last)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDBBackupStoreOutageIsAnError(t *testing.T) {
|
||||||
|
api, repo := seedUpdatesAPI(t)
|
||||||
|
repo.failGetSetting = errors.New("connection reset")
|
||||||
|
if code, _ := getDBBackup(t, api); code == http.StatusOK {
|
||||||
|
t.Fatal("a failed settings read answered 200")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDBBackupAdminOnly(t *testing.T) {
|
||||||
|
repo := newFakeRepo()
|
||||||
|
api := newTestAPI(repo, newFakeCluster())
|
||||||
|
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
|
||||||
|
if code, _ := getDBBackup(t, api); code != http.StatusForbidden {
|
||||||
|
t.Fatalf("player read = %d, want 403", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,6 +11,8 @@ import (
|
|||||||
"net/http"
|
"net/http"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/metrics"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Player email verification (spec §B2 onboarding). Forced web onboarding proves a
|
// Player email verification (spec §B2 onboarding). Forced web onboarding proves a
|
||||||
@@ -47,6 +49,14 @@ const (
|
|||||||
// keys (principal and recipient) so neither one account fanning out across many
|
// keys (principal and recipient) so neither one account fanning out across many
|
||||||
// addresses, nor many accounts converging on one address, can flood a mailbox.
|
// addresses, nor many accounts converging on one address, can flood a mailbox.
|
||||||
otpResendCooldown = 60 * time.Second
|
otpResendCooldown = 60 * time.Second
|
||||||
|
// otpFailureBudget caps wrong codes per (user, purpose) inside otpFailureWindow,
|
||||||
|
// across every code minted in it. The per-code cap alone resets on each resend,
|
||||||
|
// and the public login door can mint a code a minute, so without this an
|
||||||
|
// attacker gets ~7,200 guesses a day at one account. Ten a day puts a blind hit
|
||||||
|
// at 1e-5 per day; a person who mistypes that often can wait out the window or
|
||||||
|
// use a passkey.
|
||||||
|
otpFailureBudget = 10
|
||||||
|
otpFailureWindow = 24 * time.Hour
|
||||||
)
|
)
|
||||||
|
|
||||||
// OTPMailer delivers a one-time code to an email address. It is a seam, not a
|
// OTPMailer delivers a one-time code to an email address. It is a seam, not a
|
||||||
@@ -129,6 +139,15 @@ func (a *API) handleEmailOTPStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
// can't sidestep the per-mailbox cap. If any later step fails the deferred rollback
|
// can't sidestep the per-mailbox cap. If any later step fails the deferred rollback
|
||||||
// frees both windows, so a failed mint or delivery never consumes the cooldown —
|
// frees both windows, so a failed mint or delivery never consumes the cooldown —
|
||||||
// the same property the old record-after-send gave, now race-free.
|
// the same property the old record-after-send gave, now race-free.
|
||||||
|
// A spent wrong-code budget locks the door; mailing another code into it
|
||||||
|
// would only spend a send.
|
||||||
|
if until, err := a.Repo.OTPLockedUntil(r.Context(), p.UserID, otpPurposeOnboard, a.now()); err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
} else if !until.IsZero() {
|
||||||
|
writeOTPAccountLocked(w, r, until, a.now())
|
||||||
|
return
|
||||||
|
}
|
||||||
userKey, emailKey := "user:"+p.UserID, "email:"+strings.ToLower(email)
|
userKey, emailKey := "user:"+p.UserID, "email:"+strings.ToLower(email)
|
||||||
lim := a.otpLimiter()
|
lim := a.otpLimiter()
|
||||||
userAt, ok := lim.reserve(userKey, otpResendCooldown)
|
userAt, ok := lim.reserve(userKey, otpResendCooldown)
|
||||||
@@ -173,7 +192,7 @@ func (a *API) handleEmailOTPStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
// The send succeeded: keep both reservations (the deferred rollback becomes a
|
// The send succeeded: keep both reservations (the deferred rollback becomes a
|
||||||
// no-op) so the cooldown windows stand.
|
// no-op) so the cooldown windows stand.
|
||||||
committed = true
|
committed = true
|
||||||
a.audit(r, auditActor(p), "account.email.otp_sent", "")
|
a.audit(r, "account.email.otp_sent", "")
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||||
"sent": true,
|
"sent": true,
|
||||||
"expires_at": expiresAt.UTC(),
|
"expires_at": expiresAt.UTC(),
|
||||||
@@ -205,7 +224,15 @@ func (a *API) handleEmailOTPVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
email, err := a.Repo.VerifyEmailOTP(r.Context(), p.UserID, otpPurposeOnboard, otpCodeHash(code), a.now())
|
email, err := a.Repo.VerifyEmailOTP(r.Context(), p.UserID, otpPurposeOnboard, otpCodeHash(code), a.now())
|
||||||
|
if isOTPRefusal(err) {
|
||||||
|
a.authFailure(r, "onboard_email", otpFailureReason(err), nil)
|
||||||
|
}
|
||||||
|
var lock *OTPAccountLockedError
|
||||||
switch {
|
switch {
|
||||||
|
case errors.As(err, &lock):
|
||||||
|
a.noteOTPLock(r, err, p.UserID, otpPurposeOnboard)
|
||||||
|
writeOTPAccountLocked(w, r, lock.Until, a.now())
|
||||||
|
return
|
||||||
case errors.Is(err, ErrOTPLocked):
|
case errors.Is(err, ErrOTPLocked):
|
||||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_locked",
|
writeError(w, r, newError(http.StatusTooManyRequests, "otp_locked",
|
||||||
"too many incorrect attempts; request a new code"))
|
"too many incorrect attempts; request a new code"))
|
||||||
@@ -221,19 +248,28 @@ func (a *API) handleEmailOTPVerify(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.email.verified", "")
|
a.audit(r, "account.email.verified", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"verified": true, "email": email})
|
writeJSON(w, http.StatusOK, map[string]any{"verified": true, "email": email})
|
||||||
}
|
}
|
||||||
|
|
||||||
// deliverOTP hands the code to the configured Mailer, or — when none is wired (the
|
// deliverOTP hands the code to the configured Mailer, or — when none is wired (the
|
||||||
// demo) — logs it server-side as a KNOWN-LIMITATION. The code is logged ONLY in the
|
// demo) — logs it server-side as a KNOWN-LIMITATION. The code is logged ONLY in the
|
||||||
// no-mailer fallback and ONLY to the server log; it is never put in an HTTP response.
|
// no-mailer fallback and ONLY to the server log; it is never put in an HTTP response.
|
||||||
|
//
|
||||||
|
// Every real send spends one token of the install-wide mail budget (mailGate);
|
||||||
|
// a spent budget is a 429 mail_rate_limited and nothing reaches the relay.
|
||||||
func (a *API) deliverOTP(ctx context.Context, email, code string) error {
|
func (a *API) deliverOTP(ctx context.Context, email, code string) error {
|
||||||
if a.Mailer == nil {
|
if a.Mailer == nil {
|
||||||
log.Printf("email-otp: no Mailer configured; code for %s is %s (KNOWN-LIMITATION: demo has no SMTP)", email, code)
|
log.Printf("email-otp: no Mailer configured; code for %s is %s (KNOWN-LIMITATION: demo has no SMTP)", email, code)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
if ok, wait := a.mailGate().take(mailGateKey); !ok {
|
||||||
|
metrics.MailTotal.WithLabelValues("otp", "throttled").Inc()
|
||||||
|
log.Printf("api: OTP mail refused by the install-wide mail budget (request_id=%s)", requestIDFromContext(ctx))
|
||||||
|
return errMailRateLimited(wait)
|
||||||
|
}
|
||||||
if err := a.Mailer.SendOTP(ctx, email, code); err != nil {
|
if err := a.Mailer.SendOTP(ctx, email, code); err != nil {
|
||||||
|
metrics.MailTotal.WithLabelValues("otp", "failed").Inc()
|
||||||
// Mapped here rather than at each of the four call sites, so every door that
|
// Mapped here rather than at each of the four call sites, so every door that
|
||||||
// mails a code answers the same way. A relay refusal is neither the caller's
|
// mails a code answers the same way. A relay refusal is neither the caller's
|
||||||
// fault nor a bug in Felis, and a bare 500 says neither — it reads as "the
|
// fault nor a bug in Felis, and a bare 500 says neither — it reads as "the
|
||||||
@@ -247,6 +283,7 @@ func (a *API) deliverOTP(ctx context.Context, email, code string) error {
|
|||||||
return newError(http.StatusBadGateway, "mail_undeliverable",
|
return newError(http.StatusBadGateway, "mail_undeliverable",
|
||||||
"the mail relay refused this message; ask the server operator to check the SMTP settings")
|
"the mail relay refused this message; ask the server operator to check the SMTP settings")
|
||||||
}
|
}
|
||||||
|
metrics.MailTotal.WithLabelValues("otp", "sent").Inc()
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -282,16 +319,6 @@ func (a *API) handleSetEmail(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.email.set", "")
|
a.audit(r, "account.email.set", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"email": email})
|
writeJSON(w, http.StatusOK, map[string]any{"email": email})
|
||||||
}
|
}
|
||||||
|
|
||||||
// auditActor picks the most identifying actor string for a principal: the audited
|
|
||||||
// Access email when present, else the stable user id. A player mid-onboarding may
|
|
||||||
// not have a verified email yet, so the id keeps the audit row attributable.
|
|
||||||
func auditActor(p *Principal) string {
|
|
||||||
if p.Email != "" {
|
|
||||||
return p.Email
|
|
||||||
}
|
|
||||||
return p.UserID
|
|
||||||
}
|
|
||||||
@@ -7,6 +7,7 @@ import (
|
|||||||
|
|
||||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
"felis.lolicon.best/internal/fileedit"
|
"felis.lolicon.best/internal/fileedit"
|
||||||
|
"felis.lolicon.best/internal/maintenance"
|
||||||
"felis.lolicon.best/internal/naming"
|
"felis.lolicon.best/internal/naming"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -149,13 +150,21 @@ func (a *API) handleWriteFile(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A write holds the world volume for its Job's lifetime (internal/maintenance);
|
||||||
|
// reads and listings do not, since a read-only mount cannot hurt a server
|
||||||
|
// starting beside it.
|
||||||
|
release, ok := a.acquireWorld(w, r, name, maintenance.KindFileWrite, "stop the server before editing its files")
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer release()
|
||||||
|
|
||||||
if err := a.Files.Write(r.Context(), name, path, *body.Content); err != nil {
|
if err := a.Files.Write(r.Context(), name, path, *body.Content); err != nil {
|
||||||
writeFileEditError(w, r, err)
|
writeFileEditError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
p := principalFromContext(r.Context())
|
a.audit(r, "file.write", name+":"+path)
|
||||||
a.audit(r, p.Email, "file.write", name+":"+path)
|
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"path": path, "status": "written"})
|
writeJSON(w, http.StatusOK, map[string]any{"path": path, "status": "written"})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -63,9 +63,9 @@ func mkFiles() (*API, *fakeRepo, *fakeCluster, *fakeFileEditor) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestFileEditorStoppedGate is the gate this whole subsystem hinges on. The world
|
// TestFileEditorStoppedGate is the gate this whole subsystem hinges on. The world
|
||||||
// PVC is ReadWriteOnce, so a running server holds it and a file Job physically
|
// PVC is ReadWriteOnce, but RWO is per node: on a single node a file Job mounts it
|
||||||
// cannot mount it — an ungated request would not fail cleanly, it would hang
|
// right beside a running server, and a write lands under a live world that the
|
||||||
// waiting for a Pod that can never be scheduled. Every one of the three routes
|
// server's next save overwrites or tears. Every one of the three routes
|
||||||
// must therefore refuse a non-stopped server with 409 not_stopped BEFORE reaching
|
// must therefore refuse a non-stopped server with 409 not_stopped BEFORE reaching
|
||||||
// the executor, which is why each asserts calls == 0 as well as the status.
|
// the executor, which is why each asserts calls == 0 as well as the status.
|
||||||
func TestFileEditorStoppedGate(t *testing.T) {
|
func TestFileEditorStoppedGate(t *testing.T) {
|
||||||
|
|||||||
@@ -51,9 +51,8 @@ func (a *API) handleReady(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: "backend", Source: "internal", Action: "ready", ServerName: name,
|
Actor: "backend", Source: "internal", Action: "ready", ServerName: name,
|
||||||
RequestID: requestIDFromContext(r.Context()),
|
|
||||||
})
|
})
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
}
|
}
|
||||||
@@ -153,17 +152,18 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A 409 maintenance_in_progress tells velocity nothing is coming up until the
|
||||||
|
// restore/backup/file write finishes, so it does not enqueue the player.
|
||||||
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
|
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
|
||||||
writeError(w, r, err)
|
a.writeLookupError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// Consume the shared per-server cooldown only after the wake flips, so a join
|
// Consume the shared per-server cooldown only after the wake flips, so a join
|
||||||
// the cap held with 503 (or a SetDesiredState error) leaves the cooldown
|
// the cap held with 503 (or a SetDesiredState error) leaves the cooldown
|
||||||
// untouched and the next join attempt is not also throttled.
|
// untouched and the next join attempt is not also throttled.
|
||||||
a.limiter().record(name)
|
a.limiter().record(name)
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: "velocity", Source: "internal", Action: "wake", ServerName: name,
|
Actor: "velocity", Source: "internal", Action: "wake", ServerName: name,
|
||||||
RequestID: requestIDFromContext(r.Context()),
|
|
||||||
})
|
})
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||||
"name": name, "desiredState": "Running",
|
"name": name, "desiredState": "Running",
|
||||||
@@ -254,9 +254,8 @@ func (a *API) handleInternalClaim(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: "velocity", Source: "internal", Action: "claim", ServerName: name,
|
Actor: "velocity", Source: "internal", Action: "claim", ServerName: name,
|
||||||
RequestID: requestIDFromContext(r.Context()),
|
|
||||||
})
|
})
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true})
|
writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true})
|
||||||
}
|
}
|
||||||
@@ -365,6 +364,8 @@ func (a *API) writeLookupError(w http.ResponseWriter, r *http.Request, err error
|
|||||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "not found"))
|
writeError(w, r, newError(http.StatusNotFound, "not_found", "not found"))
|
||||||
case errors.Is(err, ErrConflict):
|
case errors.Is(err, ErrConflict):
|
||||||
writeError(w, r, newError(http.StatusConflict, "conflict", "conflict"))
|
writeError(w, r, newError(http.StatusConflict, "conflict", "conflict"))
|
||||||
|
case errors.Is(err, ErrMaintenanceInProgress), errors.Is(err, ErrNotStopped):
|
||||||
|
writeError(w, r, maintenanceError(err, "stop the server completely first"))
|
||||||
default:
|
default:
|
||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -98,6 +98,6 @@ func (a *API) handleServerConsole(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "console.attach", name)
|
a.audit(r, "console.attach", name)
|
||||||
relayLogStream(w, r, src)
|
relayLogStream(w, r, src)
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,176 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"slices"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||||
|
"felis.lolicon.best/internal/maintenance"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The world-volume lock (internal/maintenance) from the handlers' side: a wake
|
||||||
|
// refused because a restore/backup/file write holds the world, and those three
|
||||||
|
// operations refused while another one does. The lock itself (atomicity, stale
|
||||||
|
// locks, Job-backed holds) is K8sCluster's and is tested in k8scluster_test.
|
||||||
|
|
||||||
|
func TestWakeRefusedDuringMaintenance(t *testing.T) {
|
||||||
|
busy := &MaintenanceBusyError{Kind: maintenance.KindRestore}
|
||||||
|
|
||||||
|
t.Run("external wake -> 409 maintenance_in_progress, cooldown kept", func(t *testing.T) {
|
||||||
|
repo := newFakeRepo()
|
||||||
|
cl := newFakeCluster()
|
||||||
|
cl.byName["survival"] = &ServerInfo{Name: "survival", AutostartPolicy: "public"}
|
||||||
|
cl.wakeErr["survival"] = busy
|
||||||
|
api := newTestAPI(repo, cl)
|
||||||
|
api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
|
||||||
|
|
||||||
|
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil)
|
||||||
|
if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
|
||||||
|
t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if _, set := cl.desired["survival"]; set {
|
||||||
|
t.Fatal("a refused wake must not flip desiredState")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The refusal did not burn the cooldown: once the restore is done the very
|
||||||
|
// next wake goes through.
|
||||||
|
delete(cl.wakeErr, "survival")
|
||||||
|
if w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/wake", "", nil); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("wake after maintenance: code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("internal wake (velocity) -> 409 maintenance_in_progress, cooldown kept", func(t *testing.T) {
|
||||||
|
api, cl := newInternalWakeAPI("public")
|
||||||
|
cl.wakeErr["survival"] = busy
|
||||||
|
body := `{"mc_uuid":"` + wakeUUID + `"}`
|
||||||
|
|
||||||
|
w := internalWake(api, body)
|
||||||
|
if w.Code != http.StatusConflict || decodeErr(t, w) != "maintenance_in_progress" {
|
||||||
|
t.Fatalf("code = %d body %s, want 409 maintenance_in_progress", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
delete(cl.wakeErr, "survival")
|
||||||
|
if w := internalWake(api, body); w.Code != http.StatusAccepted {
|
||||||
|
t.Fatalf("wake after maintenance: code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// maintenanceOp is one world-volume operation as the external face serves it.
|
||||||
|
type maintenanceOp struct {
|
||||||
|
name string
|
||||||
|
kind string
|
||||||
|
method string
|
||||||
|
path string
|
||||||
|
body string
|
||||||
|
calls func() int
|
||||||
|
}
|
||||||
|
|
||||||
|
func maintenanceOps() (*API, *fakeCluster, []maintenanceOp) {
|
||||||
|
repo := newFakeRepo()
|
||||||
|
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||||
|
repo.backups = []fakeBackup{{view: BackupView{ID: "bk1", ServerName: "survival",
|
||||||
|
FormerOwner: "owner1", Status: "present"}, ref: "world-archive-ref"}}
|
||||||
|
cl := newFakeCluster()
|
||||||
|
cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: "Stopped",
|
||||||
|
DesiredState: string(v1alpha1.DesiredStopped)}
|
||||||
|
restorer, backuper, files := &fakeRestorer{}, &fakeBackuper{}, &fakeFileEditor{}
|
||||||
|
api := newTestAPI(repo, cl)
|
||||||
|
api.Restorer, api.Backuper, api.Files = restorer, backuper, files
|
||||||
|
api.External = staticExternal{p: &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}}
|
||||||
|
return api, cl, []maintenanceOp{
|
||||||
|
{"restore", maintenance.KindRestore, "POST", "/api/v1/servers/survival/restore-backup", "",
|
||||||
|
func() int { return restorer.calls }},
|
||||||
|
{"backup", maintenance.KindBackup, "POST", "/api/v1/servers/survival/backup", "",
|
||||||
|
func() int { return backuper.calls }},
|
||||||
|
{"file write", maintenance.KindFileWrite, "PUT", "/api/v1/servers/survival/file?path=server.properties",
|
||||||
|
`{"content":"aGk="}`, func() int { return files.calls }},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (op maintenanceOp) do(api *API) *httptest.ResponseRecorder {
|
||||||
|
var hdr map[string]string
|
||||||
|
if op.body != "" {
|
||||||
|
hdr = jsonHeader
|
||||||
|
}
|
||||||
|
return do(api.ExternalHandler(), op.method, op.path, op.body, hdr)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintenanceOpsTakeAndReleaseTheLock(t *testing.T) {
|
||||||
|
_, _, ops := maintenanceOps()
|
||||||
|
for i := range ops {
|
||||||
|
t.Run(ops[i].name, func(t *testing.T) {
|
||||||
|
api, cl, ops := maintenanceOps()
|
||||||
|
op := ops[i]
|
||||||
|
if w := op.do(api); w.Code/100 != 2 {
|
||||||
|
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||||
|
}
|
||||||
|
if op.calls() != 1 {
|
||||||
|
t.Fatalf("executor calls = %d, want 1", op.calls())
|
||||||
|
}
|
||||||
|
if want := []string{"survival:" + op.kind}; !slices.Equal(cl.acquired, want) {
|
||||||
|
t.Fatalf("acquired = %v, want %v", cl.acquired, want)
|
||||||
|
}
|
||||||
|
if want := []string{"survival"}; !slices.Equal(cl.released, want) {
|
||||||
|
t.Fatalf("released = %v, want %v (the Job is the lock from here on)", cl.released, want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintenanceOpsRefusedWhileHeld(t *testing.T) {
|
||||||
|
refusals := []struct {
|
||||||
|
name string
|
||||||
|
err error
|
||||||
|
code string
|
||||||
|
}{
|
||||||
|
{"another holder", &MaintenanceBusyError{Kind: maintenance.KindBackup}, "maintenance_in_progress"},
|
||||||
|
// The snapshot said Stopped but the atomic re-check found it waking: the
|
||||||
|
// wake won the race.
|
||||||
|
{"server not stopped", fmt.Errorf("wrapped: %w", ErrNotStopped), "not_stopped"},
|
||||||
|
}
|
||||||
|
_, _, ops := maintenanceOps()
|
||||||
|
for i := range ops {
|
||||||
|
for _, rf := range refusals {
|
||||||
|
t.Run(ops[i].name+" / "+rf.name, func(t *testing.T) {
|
||||||
|
api, cl, ops := maintenanceOps()
|
||||||
|
op := ops[i]
|
||||||
|
cl.maintErr["survival"] = rf.err
|
||||||
|
w := op.do(api)
|
||||||
|
if w.Code != http.StatusConflict || decodeErr(t, w) != rf.code {
|
||||||
|
t.Fatalf("code = %d body %s, want 409 %s", w.Code, w.Body.String(), rf.code)
|
||||||
|
}
|
||||||
|
if op.calls() != 0 {
|
||||||
|
t.Fatal("a refused operation must not reach its executor")
|
||||||
|
}
|
||||||
|
if len(cl.released) != 0 {
|
||||||
|
t.Fatalf("released %v a lock that was never taken", cl.released)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestMaintenanceErrorMapping(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
err error
|
||||||
|
code string
|
||||||
|
}{
|
||||||
|
{&MaintenanceBusyError{Kind: maintenance.KindFileWrite}, "maintenance_in_progress"},
|
||||||
|
{fmt.Errorf("x: %w", ErrMaintenanceInProgress), "maintenance_in_progress"},
|
||||||
|
{ErrNotStopped, "not_stopped"},
|
||||||
|
} {
|
||||||
|
var ae *apiError
|
||||||
|
if !errors.As(maintenanceError(tc.err, "stop it"), &ae) || ae.code != tc.code || ae.status != http.StatusConflict {
|
||||||
|
t.Errorf("%v -> %+v, want 409 %s", tc.err, ae, tc.code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
other := errors.New("boom")
|
||||||
|
if got := maintenanceError(other, "stop it"); got != other {
|
||||||
|
t.Errorf("unrelated error rewritten to %v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -103,13 +103,16 @@ func (a *API) handleBindRedeem(w http.ResponseWriter, r *http.Request) {
|
|||||||
userID, mcUUID, authSource, err := a.Repo.RedeemPlayerBindCode(r.Context(), uid, code, a.now())
|
userID, mcUUID, authSource, err := a.Repo.RedeemPlayerBindCode(r.Context(), uid, code, a.now())
|
||||||
switch {
|
switch {
|
||||||
case errors.Is(err, ErrLinkCodeInvalid):
|
case errors.Is(err, ErrLinkCodeInvalid):
|
||||||
|
a.authFailure(r, "bind_redeem", "bad_code", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "bind code is invalid or expired"))
|
writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "bind code is invalid or expired"))
|
||||||
return
|
return
|
||||||
case errors.Is(err, ErrPlayerBindForbidden):
|
case errors.Is(err, ErrPlayerBindForbidden):
|
||||||
|
a.authFailure(r, "bind_redeem", "staff_account", nil)
|
||||||
writeError(w, r, newError(http.StatusForbidden, "staff_account",
|
writeError(w, r, newError(http.StatusForbidden, "staff_account",
|
||||||
"that Minecraft account belongs to staff; sign in at the operator console"))
|
"that Minecraft account belongs to staff; sign in at the operator console"))
|
||||||
return
|
return
|
||||||
case errors.Is(err, ErrPlayerAccountRetired):
|
case errors.Is(err, ErrPlayerAccountRetired):
|
||||||
|
a.authFailure(r, "bind_redeem", "account_retired", nil)
|
||||||
// The linked Felis account is disabled or soft-deleted: the door refuses to
|
// The linked Felis account is disabled or soft-deleted: the door refuses to
|
||||||
// reuse it, because minting a session here would resurrect the account the
|
// reuse it, because minting a session here would resurrect the account the
|
||||||
// owner just retired (audit #33). The code survives, so re-enabling the
|
// owner just retired (audit #33). The code survives, so re-enabling the
|
||||||
@@ -133,7 +136,8 @@ func (a *API) handleBindRedeem(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
setSessionCookie(w, token, expires)
|
setSessionCookie(w, token, expires)
|
||||||
a.audit(r, userID, "account.bind_redeem", "")
|
// The in-game code proved the Minecraft account; it names the actor.
|
||||||
|
a.auditEntry(r, AuditEntry{Actor: "mc:" + mcUUID, ActorUserID: userID, Action: "account.bind_redeem"})
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"user_id": userID, "linked": true, "mc_uuid": mcUUID, "auth_source": authSource,
|
"user_id": userID, "linked": true, "mc_uuid": mcUUID, "auth_source": authSource,
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -86,6 +86,13 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The install-wide mail budget is checked before the address is resolved,
|
||||||
|
// so while it is spent every address gets the same 429.
|
||||||
|
if err := a.checkMailBudget(); err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
// Per-recipient cooldown reserved BEFORE any work, identical to the console email
|
// Per-recipient cooldown reserved BEFORE any work, identical to the console email
|
||||||
// door: one winner per window, and the neutral (non-staff) branch keeps the
|
// door: one winner per window, and the neutral (non-staff) branch keeps the
|
||||||
// reservation too so probing an address is throttled exactly like a real send. The
|
// reservation too so probing an address is throttled exactly like a real send. The
|
||||||
@@ -129,6 +136,7 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
u, err := a.Repo.UserByEmail(r.Context(), email)
|
u, err := a.Repo.UserByEmail(r.Context(), email)
|
||||||
switch {
|
switch {
|
||||||
case errors.Is(err, ErrNotFound):
|
case errors.Is(err, ErrNotFound):
|
||||||
|
a.authFailure(r, "op_login", "no_account", nil)
|
||||||
neutral()
|
neutral()
|
||||||
return
|
return
|
||||||
case err != nil:
|
case err != nil:
|
||||||
@@ -141,6 +149,16 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
// on console.<root_domain>) gets the neutral response, never a request or a code.
|
// on console.<root_domain>) gets the neutral response, never a request or a code.
|
||||||
// Staff means admin OR owner — the Owner is the primary op.console user.
|
// Staff means admin OR owner — the Owner is the primary op.console user.
|
||||||
if !staffRole(u.Role) {
|
if !staffRole(u.Role) {
|
||||||
|
a.authFailure(r, "op_login", "not_staff", u)
|
||||||
|
neutral()
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// A locked door (wrong-code budget spent) is neutral too: no request, no mail.
|
||||||
|
switch until, err := a.Repo.OTPLockedUntil(r.Context(), u.ID, otpPurposeOpLogin, a.now()); {
|
||||||
|
case err != nil:
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
case !until.IsZero():
|
||||||
neutral()
|
neutral()
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -175,7 +193,7 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
committed = true
|
committed = true
|
||||||
a.audit(r, u.Username, "auth.op_login.otp_sent", "")
|
a.auditAccount(r, u, "auth.op_login.otp_sent", "")
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||||
"request_id": id, "expires_at": expiresAt.UTC(),
|
"request_id": id, "expires_at": expiresAt.UTC(),
|
||||||
})
|
})
|
||||||
@@ -255,6 +273,7 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
loginReq, err := a.Repo.OpLoginRequestByID(r.Context(), requestID)
|
loginReq, err := a.Repo.OpLoginRequestByID(r.Context(), requestID)
|
||||||
switch {
|
switch {
|
||||||
case errors.Is(err, ErrNotFound):
|
case errors.Is(err, ErrNotFound):
|
||||||
|
a.authFailure(r, "op_login", "unknown_request", nil)
|
||||||
writeError(w, r, invalid)
|
writeError(w, r, invalid)
|
||||||
return
|
return
|
||||||
case err != nil:
|
case err != nil:
|
||||||
@@ -265,6 +284,7 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
// code before an admin approved) must not consume the code. Not-approved collapses
|
// code before an admin approved) must not consume the code. Not-approved collapses
|
||||||
// into the same uniform failure as a bad code, so the ordering leaks nothing.
|
// into the same uniform failure as a bad code, so the ordering leaks nothing.
|
||||||
if loginReq.Status != "approved" || loginReq.Consumed || !loginReq.ExpiresAt.After(now) {
|
if loginReq.Status != "approved" || loginReq.Consumed || !loginReq.ExpiresAt.After(now) {
|
||||||
|
a.authFailure(r, "op_login", "not_approved", a.opLoginAccount(r, loginReq.UserID))
|
||||||
writeError(w, r, invalid)
|
writeError(w, r, invalid)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -272,7 +292,9 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
// attempt without minting anything and returns the uniform failure — the code, not
|
// attempt without minting anything and returns the uniform failure — the code, not
|
||||||
// the request, is the problem, and the request stays approved for a retry.
|
// the request, is the problem, and the request stays approved for a retry.
|
||||||
switch err := a.Repo.ConsumeLoginEmailOTP(r.Context(), loginReq.UserID, otpPurposeOpLogin, otpCodeHash(code), now); {
|
switch err := a.Repo.ConsumeLoginEmailOTP(r.Context(), loginReq.UserID, otpPurposeOpLogin, otpCodeHash(code), now); {
|
||||||
case errors.Is(err, ErrOTPInvalid), errors.Is(err, ErrOTPLocked):
|
case errors.Is(err, ErrOTPInvalid), errors.Is(err, ErrOTPLocked), errors.Is(err, ErrOTPAccountLocked):
|
||||||
|
a.noteOTPLock(r, err, loginReq.UserID, otpPurposeOpLogin)
|
||||||
|
a.authFailure(r, "op_login", otpFailureReason(err), a.opLoginAccount(r, loginReq.UserID))
|
||||||
writeError(w, r, invalid)
|
writeError(w, r, invalid)
|
||||||
return
|
return
|
||||||
case err != nil:
|
case err != nil:
|
||||||
@@ -300,6 +322,7 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !staffRole(u.Role) {
|
if !staffRole(u.Role) {
|
||||||
|
a.authFailure(r, "op_login", "not_staff", u)
|
||||||
writeError(w, r, newError(http.StatusForbidden, "staff_account", "that account is not an operator"))
|
writeError(w, r, newError(http.StatusForbidden, "staff_account", "that account is not an operator"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -314,10 +337,19 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
setSessionCookie(w, token, expires)
|
setSessionCookie(w, token, expires)
|
||||||
a.audit(r, u.Username, "auth.op_login", "")
|
a.auditAccount(r, u, "auth.op_login", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"user_id": u.ID, "role": u.Role})
|
writeJSON(w, http.StatusOK, map[string]any{"user_id": u.ID, "role": u.Role})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// opLoginAccount loads the account a login request belongs to for a failure's
|
||||||
|
// audit row, falling back to the bare id.
|
||||||
|
func (a *API) opLoginAccount(r *http.Request, userID string) *StaffUser {
|
||||||
|
if u, err := a.Repo.UserByID(r.Context(), userID); err == nil {
|
||||||
|
return u
|
||||||
|
}
|
||||||
|
return &StaffUser{ID: userID}
|
||||||
|
}
|
||||||
|
|
||||||
// handleOpLoginPending lists live pending staff login requests, oldest first (internal
|
// handleOpLoginPending lists live pending staff login requests, oldest first (internal
|
||||||
// face). Today no plugin consumes it: the approver learns the request id out-of-band
|
// face). Today no plugin consumes it: the approver learns the request id out-of-band
|
||||||
// (the op.console start screen shows it to the person logging in) and runs
|
// (the op.console start screen shows it to the person logging in) and runs
|
||||||
@@ -404,9 +436,9 @@ func (a *API) handleOpLoginApprove(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
payload, _ := json.Marshal(map[string]string{"request_id": id, "approver_user_id": approverID})
|
payload, _ := json.Marshal(map[string]string{"request_id": id, "approver_user_id": approverID})
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: approver.Username, Source: "internal", Action: "auth.op_login.approved",
|
Actor: approver.Username, ActorUserID: approverID, Source: "internal",
|
||||||
RequestID: requestIDFromContext(r.Context()), Payload: payload,
|
Action: "auth.op_login.approved", Payload: payload,
|
||||||
})
|
})
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"approved": true})
|
writeJSON(w, http.StatusOK, map[string]any{"approved": true})
|
||||||
}
|
}
|
||||||
@@ -3,6 +3,7 @@ package api
|
|||||||
import (
|
import (
|
||||||
"net/http"
|
"net/http"
|
||||||
"net/http/httptest"
|
"net/http/httptest"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
@@ -149,13 +150,14 @@ func TestOpLoginVertical(t *testing.T) {
|
|||||||
t.Fatalf("session row for the cookie = %+v (ok=%v), want userID a1", s, ok)
|
t.Fatalf("session row for the cookie = %+v (ok=%v), want userID a1", s, ok)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Three audits by "op": otp_sent (start), approved (in-game vouch), op_login (finish).
|
// Four audits by "op": otp_sent (start), the early finish refused before
|
||||||
if n := len(repo.audits); n != 3 {
|
// approval, approved (in-game vouch), op_login (finish).
|
||||||
t.Fatalf("want 3 audits, got %d: %+v", n, repo.audits)
|
if n := len(repo.audits); n != 4 {
|
||||||
|
t.Fatalf("want 4 audits, got %d: %+v", n, repo.audits)
|
||||||
}
|
}
|
||||||
wantActions := []string{"auth.op_login.otp_sent", "auth.op_login.approved", "auth.op_login"}
|
wantActions := []string{"auth.op_login.otp_sent", "auth.op_login.failed", "auth.op_login.approved", "auth.op_login"}
|
||||||
for i, want := range wantActions {
|
for i, want := range wantActions {
|
||||||
if repo.audits[i].Action != want || repo.audits[i].Actor != "op" {
|
if repo.audits[i].Action != want || repo.audits[i].Actor != "op" || repo.audits[i].ActorUserID != "a1" {
|
||||||
t.Errorf("audit[%d] = %+v, want action %q by op", i, repo.audits[i], want)
|
t.Errorf("audit[%d] = %+v, want action %q by op", i, repo.audits[i], want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -213,7 +215,7 @@ func TestOpLoginOwnerAdmitted(t *testing.T) {
|
|||||||
// a request_id + expires_at, mint/mail nothing, and still burn the per-recipient
|
// a request_id + expires_at, mint/mail nothing, and still burn the per-recipient
|
||||||
// cooldown — so neither the response nor the throttle tells a caller who is staff.
|
// cooldown — so neither the response nor the throttle tells a caller who is staff.
|
||||||
func TestOpLoginStartNeutral(t *testing.T) {
|
func TestOpLoginStartNeutral(t *testing.T) {
|
||||||
check := func(t *testing.T, seed func(*fakeRepo), email string) {
|
check := func(t *testing.T, seed func(*fakeRepo), email, wantReason, wantUser string) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
repo := newFakeRepo()
|
repo := newFakeRepo()
|
||||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||||
@@ -236,9 +238,15 @@ func TestOpLoginStartNeutral(t *testing.T) {
|
|||||||
if s, _ := b["expires_at"].(string); s == "" {
|
if s, _ := b["expires_at"].(string); s == "" {
|
||||||
t.Error("neutral start must still return expires_at")
|
t.Error("neutral start must still return expires_at")
|
||||||
}
|
}
|
||||||
if len(repo.opLogins) != 0 || len(repo.otps) != 0 || mailer.calls != 0 || len(repo.audits) != 0 {
|
if len(repo.opLogins) != 0 || len(repo.otps) != 0 || mailer.calls != 0 {
|
||||||
t.Errorf("neutral start must mint/mail/audit nothing: reqs=%d otps=%d mails=%d audits=%d",
|
t.Errorf("neutral start must mint/mail nothing: reqs=%d otps=%d mails=%d",
|
||||||
len(repo.opLogins), len(repo.otps), mailer.calls, len(repo.audits))
|
len(repo.opLogins), len(repo.otps), mailer.calls)
|
||||||
|
}
|
||||||
|
// The response is neutral; the operator's record is not.
|
||||||
|
if len(repo.audits) != 1 || repo.audits[0].Action != "auth.op_login.failed" ||
|
||||||
|
!strings.Contains(string(repo.audits[0].Payload), `"reason":"`+wantReason+`"`) ||
|
||||||
|
repo.audits[0].ActorUserID != wantUser {
|
||||||
|
t.Errorf("neutral start audits = %+v, want one auth.op_login.failed %s by %q", repo.audits, wantReason, wantUser)
|
||||||
}
|
}
|
||||||
// The reservation is KEPT: re-probing the same address is throttled like a resend.
|
// The reservation is KEPT: re-probing the same address is throttled like a resend.
|
||||||
if w := startOp(eh, email); w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "otp_resend_cooldown" {
|
if w := startOp(eh, email); w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "otp_resend_cooldown" {
|
||||||
@@ -247,12 +255,12 @@ func TestOpLoginStartNeutral(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
t.Run("unknown address", func(t *testing.T) {
|
t.Run("unknown address", func(t *testing.T) {
|
||||||
check(t, nil, "[email protected]")
|
check(t, nil, "[email protected]", "no_account", "")
|
||||||
})
|
})
|
||||||
t.Run("non-staff (role=user) address is ignored by the staff door", func(t *testing.T) {
|
t.Run("non-staff (role=user) address is ignored by the staff door", func(t *testing.T) {
|
||||||
check(t, func(repo *fakeRepo) {
|
check(t, func(repo *fakeRepo) {
|
||||||
repo.staff["p"] = &StaffUser{ID: "u9", Username: "p", Email: "[email protected]", Role: "user", EmailVerified: true}
|
repo.staff["p"] = &StaffUser{ID: "u9", Username: "p", Email: "[email protected]", Role: "user", EmailVerified: true}
|
||||||
}, "[email protected]")
|
}, "[email protected]", "not_staff", "u9")
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -355,7 +355,7 @@ func (a *API) handlePasskeyRegisterFinish(w http.ResponseWriter, r *http.Request
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.passkey.registered", cred.ID)
|
a.audit(r, "account.passkey.registered", cred.ID)
|
||||||
writeJSON(w, http.StatusCreated, passkeyView(cred))
|
writeJSON(w, http.StatusCreated, passkeyView(cred))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -415,7 +415,7 @@ func (a *API) handlePasskeyDelete(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, auditActor(p), "account.passkey.removed", id)
|
a.audit(r, "account.passkey.removed", id)
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -593,6 +593,7 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
u, err := a.Repo.UserByEmail(r.Context(), email)
|
u, err := a.Repo.UserByEmail(r.Context(), email)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, ErrNotFound) {
|
if errors.Is(err, ErrNotFound) {
|
||||||
|
a.authFailure(r, "passkey", "no_account", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -604,6 +605,7 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
sessionData, err := a.Repo.ConsumePasskeyChallengeByUser(r.Context(), u.ID, passkeyPurposeLogin, a.now())
|
sessionData, err := a.Repo.ConsumePasskeyChallengeByUser(r.Context(), u.ID, passkeyPurposeLogin, a.now())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
||||||
|
a.authFailure(r, "passkey", "challenge_invalid", u)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -625,6 +627,7 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
}
|
}
|
||||||
va, err := a.Passkey.FinishLogin(user, sessionData, bytes.NewReader(req.Assertion))
|
va, err := a.Passkey.FinishLogin(user, sessionData, bytes.NewReader(req.Assertion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
a.authFailure(r, "passkey", "bad_assertion", u)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -634,7 +637,7 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
// distinctly; a successful assertion advances the stored counter and stamps last_used_at.
|
// distinctly; a successful assertion advances the stored counter and stamps last_used_at.
|
||||||
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
||||||
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
||||||
a.audit(r, u.Username, "auth.passkey_clone_rejected", va.CredentialID)
|
a.passkeyCloneRejected(r, "passkey", u, va.CredentialID)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -654,7 +657,7 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
setSessionCookie(w, token, expires)
|
setSessionCookie(w, token, expires)
|
||||||
a.audit(r, u.Username, "auth.passkey_login", "")
|
a.auditAccount(r, u, "auth.passkey_login", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"user_id": u.ID,
|
"user_id": u.ID,
|
||||||
"role": u.Role,
|
"role": u.Role,
|
||||||
|
|||||||
@@ -21,10 +21,10 @@ import (
|
|||||||
//
|
//
|
||||||
// Anti-abuse divergence from the email-first door: that door reserves a per-recipient cooldown
|
// Anti-abuse divergence from the email-first door: that door reserves a per-recipient cooldown
|
||||||
// (a.otpLimiter) keyed on the typed email. A usernameless begin has no recipient OR principal to
|
// (a.otpLimiter) keyed on the typed email. A usernameless begin has no recipient OR principal to
|
||||||
// key a fair per-caller limit on, so — matching the stance in handlers_auth_email.go (behind
|
// key a fair per-caller limit on, so one client is bounded by the per-address token bucket every
|
||||||
// Cloudflare RemoteAddr is the proxy; CGNAT false-positives) — volumetric per-source limiting is
|
// public auth door sits behind (throttleAuthDoor), and the table by a hard global cap on live
|
||||||
// left to the edge, and the server-side bound is a hard global cap on live challenges enforced
|
// challenges enforced atomically in CreateDiscoverableChallenge (ErrTooManyDiscoverableChallenges
|
||||||
// atomically in CreateDiscoverableChallenge (ErrTooManyDiscoverableChallenges → 429).
|
// → 429).
|
||||||
|
|
||||||
// handlePasskeyLoginDiscoverableBegin starts a usernameless assertion ceremony (Public,
|
// handlePasskeyLoginDiscoverableBegin starts a usernameless assertion ceremony (Public,
|
||||||
// pre-session). It has no request body — the whole point is that the caller supplies no
|
// pre-session). It has no request body — the whole point is that the caller supplies no
|
||||||
@@ -128,6 +128,7 @@ func (a *API) handlePasskeyLoginDiscoverableFinish(w http.ResponseWriter, r *htt
|
|||||||
sessionData, err := a.Repo.ConsumeDiscoverableChallenge(r.Context(), req.LoginID, a.now())
|
sessionData, err := a.Repo.ConsumeDiscoverableChallenge(r.Context(), req.LoginID, a.now())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
||||||
|
a.authFailure(r, "passkey_discoverable", "challenge_invalid", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -166,6 +167,9 @@ func (a *API) handlePasskeyLoginDiscoverableFinish(w http.ResponseWriter, r *htt
|
|||||||
}
|
}
|
||||||
va, err := a.Passkey.FinishDiscoverableLogin(resolve, sessionData, bytes.NewReader(req.Assertion))
|
va, err := a.Passkey.FinishDiscoverableLogin(resolve, sessionData, bytes.NewReader(req.Assertion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
// resolved is set when the credential named a live account and only the
|
||||||
|
// signature (or the credential's binding) failed.
|
||||||
|
a.authFailure(r, "passkey_discoverable", "bad_assertion", resolved)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -184,7 +188,7 @@ func (a *API) handlePasskeyLoginDiscoverableFinish(w http.ResponseWriter, r *htt
|
|||||||
// resolved account; a successful assertion advances the stored counter and stamps last_used_at.
|
// resolved account; a successful assertion advances the stored counter and stamps last_used_at.
|
||||||
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
||||||
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
||||||
a.audit(r, resolved.Username, "auth.passkey_clone_rejected", va.CredentialID)
|
a.passkeyCloneRejected(r, "passkey_discoverable", resolved, va.CredentialID)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||||
"passkey login could not be completed; begin again"))
|
"passkey login could not be completed; begin again"))
|
||||||
return
|
return
|
||||||
@@ -204,7 +208,7 @@ func (a *API) handlePasskeyLoginDiscoverableFinish(w http.ResponseWriter, r *htt
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
setSessionCookie(w, token, expires)
|
setSessionCookie(w, token, expires)
|
||||||
a.audit(r, resolved.Username, "auth.passkey_login_discoverable", "")
|
a.auditAccount(r, resolved, "auth.passkey_login_discoverable", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"user_id": resolved.ID,
|
"user_id": resolved.ID,
|
||||||
"role": resolved.Role,
|
"role": resolved.Role,
|
||||||
|
|||||||
@@ -75,7 +75,7 @@ func TestPatchServerAutostartPolicy(t *testing.T) {
|
|||||||
func TestPatchServerImageReAdmitted(t *testing.T) {
|
func TestPatchServerImageReAdmitted(t *testing.T) {
|
||||||
api, _, cl, _ := newPatchAPI()
|
api, _, cl, _ := newPatchAPI()
|
||||||
|
|
||||||
w := patchSurvival(api, `{"image":"`+admittedImage+`"}`)
|
w := patchSurvival(api, `{"image":"`+admittedImage+`","confirmImageChange":true}`)
|
||||||
if w.Code != http.StatusOK {
|
if w.Code != http.StatusOK {
|
||||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||||
}
|
}
|
||||||
@@ -273,3 +273,39 @@ func TestPatchServerImageWithoutBuilderIs503(t *testing.T) {
|
|||||||
t.Error("no image may be patched without a Builder")
|
t.Error("no image may be patched without a Builder")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestPatchServerIdleStop covers the idle auto-stop knob: 0 turns it off, a
|
||||||
|
// value inside the range is carried to the cluster, and one outside is refused
|
||||||
|
// before anything is written.
|
||||||
|
func TestPatchServerIdleStop(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
body string
|
||||||
|
wantCode int
|
||||||
|
want int32
|
||||||
|
}{
|
||||||
|
{`{"idleStopSeconds":0}`, http.StatusOK, 0},
|
||||||
|
{`{"idleStopSeconds":900}`, http.StatusOK, 900},
|
||||||
|
{`{"idleStopSeconds":59}`, http.StatusBadRequest, 0},
|
||||||
|
{`{"idleStopSeconds":86401}`, http.StatusBadRequest, 0},
|
||||||
|
{`{"idleStopSeconds":-5}`, http.StatusBadRequest, 0},
|
||||||
|
} {
|
||||||
|
api, _, cl, _ := newPatchAPI()
|
||||||
|
w := patchSurvival(api, tc.body)
|
||||||
|
if w.Code != tc.wantCode {
|
||||||
|
t.Fatalf("%s: code = %d, want %d (%s)", tc.body, w.Code, tc.wantCode, w.Body.String())
|
||||||
|
}
|
||||||
|
p, patched := cl.patched["survival"]
|
||||||
|
if tc.wantCode != http.StatusOK {
|
||||||
|
if patched {
|
||||||
|
t.Fatalf("%s: a refused value reached the cluster: %+v", tc.body, p)
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !patched || p.IdleStopSeconds == nil || *p.IdleStopSeconds != tc.want {
|
||||||
|
t.Fatalf("%s: patched idle = %v, want %d", tc.body, p.IdleStopSeconds, tc.want)
|
||||||
|
}
|
||||||
|
if got := cl.byName["survival"].IdleStopSeconds; got != tc.want {
|
||||||
|
t.Fatalf("%s: view idleStopSeconds = %d, want %d", tc.body, got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -98,9 +98,8 @@ func (a *API) handleReclaimUsername(w http.ResponseWriter, r *http.Request) {
|
|||||||
// accountability record shows the reclaim was declined, and why.
|
// accountability record shows the reclaim was declined, and why.
|
||||||
payload, _ := json.Marshal(map[string]string{
|
payload, _ := json.Marshal(map[string]string{
|
||||||
"username": req.Username, "squatter_uuid": req.SquatterUUID, "reason": "protected_admin"})
|
"username": req.Username, "squatter_uuid": req.SquatterUUID, "reason": "protected_admin"})
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: "velocity", Source: "internal", Action: "player.reclaim.refused",
|
Actor: "velocity", Source: "internal", Action: "player.reclaim.refused", Payload: payload,
|
||||||
RequestID: requestIDFromContext(r.Context()), Payload: payload,
|
|
||||||
})
|
})
|
||||||
writeError(w, r, newError(http.StatusConflict, "protected_admin",
|
writeError(w, r, newError(http.StatusConflict, "protected_admin",
|
||||||
"that username belongs to a linked administrator on the login server and cannot be reclaimed"))
|
"that username belongs to a linked administrator on the login server and cannot be reclaimed"))
|
||||||
@@ -126,9 +125,8 @@ func (a *API) handleReclaimUsername(w http.ResponseWriter, r *http.Request) {
|
|||||||
// payload (the flat columns model a server op, not this), keyed by Source
|
// payload (the flat columns model a server op, not this), keyed by Source
|
||||||
// internal since velocity, not a human, drives it.
|
// internal since velocity, not a human, drives it.
|
||||||
payload, _ := json.Marshal(map[string]string{"username": req.Username, "squatter_uuid": req.SquatterUUID})
|
payload, _ := json.Marshal(map[string]string{"username": req.Username, "squatter_uuid": req.SquatterUUID})
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
a.auditEntry(r, AuditEntry{
|
||||||
Actor: "velocity", Source: "internal", Action: "player.reclaim",
|
Actor: "velocity", Source: "internal", Action: "player.reclaim", Payload: payload,
|
||||||
RequestID: requestIDFromContext(r.Context()), Payload: payload,
|
|
||||||
})
|
})
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"blacklisted": true,
|
"blacklisted": true,
|
||||||
|
|||||||
@@ -67,6 +67,7 @@ func (a *API) handleSetupRedeem(w http.ResponseWriter, r *http.Request) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
// Unknown, already-consumed, or expired — uniform 400 so the token cannot
|
// Unknown, already-consumed, or expired — uniform 400 so the token cannot
|
||||||
// be used as an oracle.
|
// be used as an oracle.
|
||||||
|
a.authFailure(r, "setup_redeem", "bad_token", nil)
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "setup_token_invalid",
|
writeError(w, r, newError(http.StatusBadRequest, "setup_token_invalid",
|
||||||
"this setup link is invalid or has already been used"))
|
"this setup link is invalid or has already been used"))
|
||||||
return
|
return
|
||||||
@@ -96,7 +97,7 @@ func (a *API) handleSetupRedeem(w http.ResponseWriter, r *http.Request) {
|
|||||||
creds, _ := a.Repo.PasskeyCredentialsForUser(r.Context(), u.ID)
|
creds, _ := a.Repo.PasskeyCredentialsForUser(r.Context(), u.ID)
|
||||||
hasPasskey := len(creds) > 0
|
hasPasskey := len(creds) > 0
|
||||||
|
|
||||||
a.audit(r, u.Username, "auth.setup_redeem", "")
|
a.auditAccount(r, u, "auth.setup_redeem", "")
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"user_id": u.ID,
|
"user_id": u.ID,
|
||||||
"username": u.Username,
|
"username": u.Username,
|
||||||
|
|||||||
@@ -113,6 +113,6 @@ func (a *API) handleSetUpdateWindow(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, principalFromContext(r.Context()).Email, "updates.window_set", "")
|
a.audit(r, "updates.window_set", "")
|
||||||
writeJSON(w, http.StatusOK, body)
|
writeJSON(w, http.StatusOK, body)
|
||||||
}
|
}
|
||||||
@@ -56,8 +56,10 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Refused with 409 maintenance_in_progress while a restore, backup or file
|
||||||
|
// write holds the world volume: starting on a half-written world corrupts it.
|
||||||
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
|
if err := a.Cluster.SetDesiredState(r.Context(), name, v1alpha1.DesiredRunning); err != nil {
|
||||||
writeError(w, r, err)
|
a.writeLookupError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// The wake actually flipped, so consume the per-server cooldown only now: a 503
|
// The wake actually flipped, so consume the per-server cooldown only now: a 503
|
||||||
@@ -65,7 +67,7 @@ func (a *API) handleWake(w http.ResponseWriter, r *http.Request) {
|
|||||||
// held at capacity should retry the instant a slot frees, not wait out a
|
// held at capacity should retry the instant a slot frees, not wait out a
|
||||||
// cooldown their refused wake never earned).
|
// cooldown their refused wake never earned).
|
||||||
a.limiter().record(name)
|
a.limiter().record(name)
|
||||||
a.audit(r, p.Email, "wake", name)
|
a.audit(r, "wake", name)
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
|
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Running"})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -93,7 +95,7 @@ func (a *API) handleStop(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "stop", name)
|
a.audit(r, "stop", name)
|
||||||
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Stopped"})
|
writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Stopped"})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -158,7 +160,7 @@ func (a *API) handleClaim(w http.ResponseWriter, r *http.Request) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// ④ audit. Allowlist population happens on first successful join (spec §9.4).
|
// ④ audit. Allowlist population happens on first successful join (spec §9.4).
|
||||||
a.audit(r, p.Email, "claim", name)
|
a.audit(r, "claim", name)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true})
|
writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -319,7 +321,6 @@ func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, errBuildUnavailable)
|
writeError(w, r, errBuildUnavailable)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
|
|
||||||
var body createServerRequest
|
var body createServerRequest
|
||||||
if err := decodeJSON(w, r, &body); err != nil {
|
if err := decodeJSON(w, r, &body); err != nil {
|
||||||
@@ -374,6 +375,13 @@ func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
"image %q is not on the whitelist", body.Image))
|
"image %q is not on the whitelist", body.Image))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
// The spec keeps the digest the tag names now, not the tag: the world is
|
||||||
|
// created on this build and stays on it until an admin changes the image.
|
||||||
|
image, err := a.pinImage(r.Context(), body.Image)
|
||||||
|
if err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
// Quota is intentionally NOT enforced here. §15 creates an UNOWNED server
|
// Quota is intentionally NOT enforced here. §15 creates an UNOWNED server
|
||||||
// (owner_id NULL); the per-user quota is charged at claim time (spec §9.3 /
|
// (owner_id NULL); the per-user quota is charged at claim time (spec §9.3 /
|
||||||
@@ -431,7 +439,7 @@ func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
Name: body.Name,
|
Name: body.Name,
|
||||||
Subdomain: body.Subdomain,
|
Subdomain: body.Subdomain,
|
||||||
DisplayName: body.DisplayName,
|
DisplayName: body.DisplayName,
|
||||||
Image: body.Image,
|
Image: image,
|
||||||
JavaMemory: javaMemory,
|
JavaMemory: javaMemory,
|
||||||
StorageSize: storage,
|
StorageSize: storage,
|
||||||
AutostartPolicy: policy,
|
AutostartPolicy: policy,
|
||||||
@@ -447,7 +455,7 @@ func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "server.create", body.Name)
|
a.audit(r, "server.create", body.Name)
|
||||||
writeJSON(w, http.StatusCreated, map[string]any{
|
writeJSON(w, http.StatusCreated, map[string]any{
|
||||||
"name": body.Name,
|
"name": body.Name,
|
||||||
"subdomain": body.Subdomain,
|
"subdomain": body.Subdomain,
|
||||||
@@ -615,8 +623,16 @@ type patchServerRequest struct {
|
|||||||
DisplayName *string `json:"displayName,omitempty"`
|
DisplayName *string `json:"displayName,omitempty"`
|
||||||
AutostartPolicy *string `json:"autostartPolicy,omitempty"`
|
AutostartPolicy *string `json:"autostartPolicy,omitempty"`
|
||||||
Image *string `json:"image,omitempty"`
|
Image *string `json:"image,omitempty"`
|
||||||
|
// ConfirmImageChange acknowledges that a new image opens the world with
|
||||||
|
// whatever Minecraft version it carries. Chunks a newer version has upgraded
|
||||||
|
// cannot be read by the older one again, so without it an image change that
|
||||||
|
// would actually move the server is refused (image_change_unconfirmed).
|
||||||
|
ConfirmImageChange bool `json:"confirmImageChange,omitempty"`
|
||||||
Memory *string `json:"memory,omitempty"`
|
Memory *string `json:"memory,omitempty"`
|
||||||
Resources *resourceRequest `json:"resources,omitempty"`
|
Resources *resourceRequest `json:"resources,omitempty"`
|
||||||
|
// IdleStopSeconds sets idle auto-stop: 0 turns it off, otherwise the server
|
||||||
|
// stops after that many seconds with nobody online (60 to 86400).
|
||||||
|
IdleStopSeconds *int32 `json:"idleStopSeconds,omitempty"`
|
||||||
// Storage is recognized only so the endpoint can reject it with a precise
|
// Storage is recognized only so the endpoint can reject it with a precise
|
||||||
// reason rather than an opaque "unknown field": a StatefulSet's PVC capacity
|
// reason rather than an opaque "unknown field": a StatefulSet's PVC capacity
|
||||||
// is immutable except for storage-class-gated expansion, which this build does
|
// is immutable except for storage-class-gated expansion, which this build does
|
||||||
@@ -625,6 +641,12 @@ type patchServerRequest struct {
|
|||||||
Storage *string `json:"storage,omitempty"`
|
Storage *string `json:"storage,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The idle auto-stop range an admin may pick through PATCH /servers/{name}.
|
||||||
|
const (
|
||||||
|
minIdleStopSeconds = 60
|
||||||
|
maxIdleStopSeconds = 86400
|
||||||
|
)
|
||||||
|
|
||||||
// handlePatchServer (spec §7 PATCH /servers/{name}) is the admin-tier spec
|
// handlePatchServer (spec §7 PATCH /servers/{name}) is the admin-tier spec
|
||||||
// mutation: it validates the structured form, re-admits any new image against the
|
// mutation: it validates the structured form, re-admits any new image against the
|
||||||
// whitelist, re-derives the §22 memory ceiling, and applies a merge patch to the
|
// whitelist, re-derives the §22 memory ceiling, and applies a merge patch to the
|
||||||
@@ -632,7 +654,6 @@ type patchServerRequest struct {
|
|||||||
// (Postgres) is untouched, so the two never desync (spec §22). The admin gate is
|
// (Postgres) is untouched, so the two never desync (spec §22). The admin gate is
|
||||||
// the adminOnly wrapper in routing — every caller here is already an admin.
|
// the adminOnly wrapper in routing — every caller here is already an admin.
|
||||||
func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
name := r.PathValue("name")
|
name := r.PathValue("name")
|
||||||
if err := naming.ValidateServerName(name); err != nil {
|
if err := naming.ValidateServerName(name); err != nil {
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||||
@@ -647,7 +668,7 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
|
|
||||||
// An empty patch is a client mistake, not a no-op success.
|
// An empty patch is a client mistake, not a no-op success.
|
||||||
if body.DisplayName == nil && body.AutostartPolicy == nil && body.Image == nil &&
|
if body.DisplayName == nil && body.AutostartPolicy == nil && body.Image == nil &&
|
||||||
body.Memory == nil && body.Resources == nil && body.Storage == nil {
|
body.Memory == nil && body.Resources == nil && body.Storage == nil && body.IdleStopSeconds == nil {
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
||||||
"patch must set at least one field"))
|
"patch must set at least one field"))
|
||||||
return
|
return
|
||||||
@@ -662,7 +683,11 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
// the SAME helpers the create form uses. `changed` records what actually moves
|
// the SAME helpers the create form uses. `changed` records what actually moves
|
||||||
// so the response and audit name the real mutation.
|
// so the response and audit name the real mutation.
|
||||||
var patch ServerSpecPatch
|
var patch ServerSpecPatch
|
||||||
var changed []string
|
// Non-nil so a patch that moves nothing (re-picking the image a server is
|
||||||
|
// already pinned to) still answers "patched": [], as the API documents.
|
||||||
|
changed := []string{}
|
||||||
|
// imageFrom is the image a confirmed image change replaced, for the audit row.
|
||||||
|
var imageFrom string
|
||||||
|
|
||||||
if body.DisplayName != nil {
|
if body.DisplayName != nil {
|
||||||
patch.DisplayName = body.DisplayName
|
patch.DisplayName = body.DisplayName
|
||||||
@@ -686,6 +711,19 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
changed = append(changed, "autostartPolicy")
|
changed = append(changed, "autostartPolicy")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if body.IdleStopSeconds != nil {
|
||||||
|
// A minute is the floor: below it a player who drops for a reconnect
|
||||||
|
// finds the server stopping under them. A day is the ceiling; longer is
|
||||||
|
// what "off" is for.
|
||||||
|
if s := *body.IdleStopSeconds; s != 0 && (s < minIdleStopSeconds || s > maxIdleStopSeconds) {
|
||||||
|
writeError(w, r, newError(http.StatusBadRequest, "bad_idle_stop",
|
||||||
|
"idleStopSeconds must be 0 (off) or between %d and %d", minIdleStopSeconds, maxIdleStopSeconds))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
patch.IdleStopSeconds = body.IdleStopSeconds
|
||||||
|
changed = append(changed, "idleStopSeconds")
|
||||||
|
}
|
||||||
|
|
||||||
if body.Image != nil {
|
if body.Image != nil {
|
||||||
// A new image must be re-admitted against the whitelist, exactly as create
|
// A new image must be re-admitted against the whitelist, exactly as create
|
||||||
// does — admission is the only source of a legal image. With no Builder
|
// does — admission is the only source of a legal image. With no Builder
|
||||||
@@ -708,8 +746,31 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
"image %q is not on the whitelist", *body.Image))
|
"image %q is not on the whitelist", *body.Image))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
patch.Image = body.Image
|
image, err := a.pinImage(r.Context(), *body.Image)
|
||||||
|
if err != nil {
|
||||||
|
writeError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
info, err := a.Cluster.GetServer(r.Context(), name)
|
||||||
|
if err != nil {
|
||||||
|
a.writeLookupError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// Re-picking the tag a server was created from resolves to that tag's
|
||||||
|
// newest build, which is as much a version move as picking another image.
|
||||||
|
// Only a pin that lands on exactly the current image is no change at all.
|
||||||
|
if image != info.Image {
|
||||||
|
if !body.ConfirmImageChange {
|
||||||
|
writeError(w, r, newError(http.StatusConflict, "image_change_unconfirmed",
|
||||||
|
"changing the image from %q to %q opens this world with the new image's Minecraft version, "+
|
||||||
|
"and chunks it upgrades cannot be opened by the old one again; back the world up first, "+
|
||||||
|
"then resend with confirmImageChange", info.Image, image))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
patch.Image = &image
|
||||||
changed = append(changed, "image")
|
changed = append(changed, "image")
|
||||||
|
imageFrom = info.Image
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Memory and the resource overrides move together: resolveResources derives the
|
// Memory and the resource overrides move together: resolveResources derives the
|
||||||
@@ -792,7 +853,11 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "server.patch", name)
|
if patch.Image != nil {
|
||||||
|
a.auditImageChange(r, name, imageFrom, *patch.Image)
|
||||||
|
} else {
|
||||||
|
a.audit(r, "server.patch", name)
|
||||||
|
}
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"name": name,
|
"name": name,
|
||||||
"patched": changed,
|
"patched": changed,
|
||||||
@@ -836,18 +901,6 @@ func (a *API) isOwnerOrAdmin(p *Principal, rec *ServerRecord) bool {
|
|||||||
return rec != nil && rec.OwnerID != "" && rec.OwnerID == p.UserID
|
return rec != nil && rec.OwnerID != "" && rec.OwnerID == p.UserID
|
||||||
}
|
}
|
||||||
|
|
||||||
// audit writes a best-effort audit row; a logging failure must not fail the
|
|
||||||
// underlying operation, which already succeeded.
|
|
||||||
func (a *API) audit(r *http.Request, actor, action, server string) {
|
|
||||||
_ = a.Repo.Audit(r.Context(), AuditEntry{
|
|
||||||
Actor: actor,
|
|
||||||
Source: "external",
|
|
||||||
Action: action,
|
|
||||||
ServerName: server,
|
|
||||||
RequestID: requestIDFromContext(r.Context()),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
// quantityToMilli converts a K8s resource.Quantity to millicores (e.g. "2"→2000,
|
// quantityToMilli converts a K8s resource.Quantity to millicores (e.g. "2"→2000,
|
||||||
// "500m"→500). A zero/unset quantity returns 0.
|
// "500m"→500). A zero/unset quantity returns 0.
|
||||||
func quantityToMilli(q resource.Quantity) int {
|
func quantityToMilli(q resource.Quantity) int {
|
||||||
|
|||||||
@@ -5,6 +5,8 @@ import (
|
|||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/metrics"
|
||||||
)
|
)
|
||||||
|
|
||||||
// ---- user CRUD ----
|
// ---- user CRUD ----
|
||||||
@@ -99,7 +101,7 @@ func (a *API) handleCreateUser(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.create", u.ID)
|
a.audit(r, "user.create", u.ID)
|
||||||
writeJSON(w, http.StatusCreated, u)
|
writeJSON(w, http.StatusCreated, u)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -179,7 +181,7 @@ func (a *API) handlePatchUser(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.patch", id)
|
a.audit(r, "user.patch", id)
|
||||||
writeJSON(w, http.StatusOK, u)
|
writeJSON(w, http.StatusOK, u)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -215,7 +217,7 @@ func (a *API) handleDeleteUser(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.delete", id)
|
a.audit(r, "user.delete", id)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"deleted": true})
|
writeJSON(w, http.StatusOK, map[string]any{"deleted": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -266,7 +268,7 @@ func (a *API) handleDisableUser(w http.ResponseWriter, r *http.Request) {
|
|||||||
if body.Disabled {
|
if body.Disabled {
|
||||||
action = "user.disable"
|
action = "user.disable"
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, action, id)
|
a.audit(r, action, id)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"id": id, "disabled": body.Disabled})
|
writeJSON(w, http.StatusOK, map[string]any{"id": id, "disabled": body.Disabled})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -323,7 +325,7 @@ func (a *API) handleSetQuotas(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.set_quotas", id)
|
a.audit(r, "user.set_quotas", id)
|
||||||
writeJSON(w, http.StatusOK, v)
|
writeJSON(w, http.StatusOK, v)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -350,7 +352,6 @@ func (a *API) handleListUserSessions(w http.ResponseWriter, r *http.Request) {
|
|||||||
// handleRevokeUserSessions revokes every live session of a user
|
// handleRevokeUserSessions revokes every live session of a user
|
||||||
// (DELETE /users/{id}/sessions).
|
// (DELETE /users/{id}/sessions).
|
||||||
func (a *API) handleRevokeUserSessions(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handleRevokeUserSessions(w http.ResponseWriter, r *http.Request) {
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
id := r.PathValue("id")
|
id := r.PathValue("id")
|
||||||
if id == "" {
|
if id == "" {
|
||||||
writeError(w, r, errBadRequest)
|
writeError(w, r, errBadRequest)
|
||||||
@@ -362,14 +363,14 @@ func (a *API) handleRevokeUserSessions(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.revoke_sessions", id)
|
metrics.SessionsRevokedTotal.WithLabelValues("admin").Inc()
|
||||||
|
a.audit(r, "user.revoke_sessions", id)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleRevokeUserSession revokes a single session of a user
|
// handleRevokeUserSession revokes a single session of a user
|
||||||
// (DELETE /users/{id}/sessions/{hash}).
|
// (DELETE /users/{id}/sessions/{hash}).
|
||||||
func (a *API) handleRevokeUserSession(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handleRevokeUserSession(w http.ResponseWriter, r *http.Request) {
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
id := r.PathValue("id")
|
id := r.PathValue("id")
|
||||||
tokenHash := r.PathValue("hash")
|
tokenHash := r.PathValue("hash")
|
||||||
if id == "" || tokenHash == "" {
|
if id == "" || tokenHash == "" {
|
||||||
@@ -382,7 +383,8 @@ func (a *API) handleRevokeUserSession(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.revoke_session", id)
|
metrics.SessionsRevokedTotal.WithLabelValues("admin").Inc()
|
||||||
|
a.audit(r, "user.revoke_session", id)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -397,7 +399,6 @@ func (a *API) handleRevokeUserSession(w http.ResponseWriter, r *http.Request) {
|
|||||||
// DeleteAllPasskeyCredentialsForUser treats removing zero rows as success, so
|
// DeleteAllPasskeyCredentialsForUser treats removing zero rows as success, so
|
||||||
// unbinding an account that holds no passkeys is a 200 no-op, not a 404.
|
// unbinding an account that holds no passkeys is a 200 no-op, not a 404.
|
||||||
func (a *API) handleUnbindUserPasskeys(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handleUnbindUserPasskeys(w http.ResponseWriter, r *http.Request) {
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
id := r.PathValue("id")
|
id := r.PathValue("id")
|
||||||
if id == "" {
|
if id == "" {
|
||||||
writeError(w, r, errBadRequest)
|
writeError(w, r, errBadRequest)
|
||||||
@@ -409,7 +410,7 @@ func (a *API) handleUnbindUserPasskeys(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.unbind_passkeys", id)
|
a.audit(r, "user.unbind_passkeys", id)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
writeJSON(w, http.StatusOK, map[string]any{"ok": true})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -418,7 +419,6 @@ func (a *API) handleUnbindUserPasskeys(w http.ResponseWriter, r *http.Request) {
|
|||||||
// handleUnlinkAccount removes a single (user_id, mc_uuid) binding
|
// handleUnlinkAccount removes a single (user_id, mc_uuid) binding
|
||||||
// (DELETE /users/{id}/links/{mc_uuid}).
|
// (DELETE /users/{id}/links/{mc_uuid}).
|
||||||
func (a *API) handleUnlinkAccount(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handleUnlinkAccount(w http.ResponseWriter, r *http.Request) {
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
userID := r.PathValue("id")
|
userID := r.PathValue("id")
|
||||||
mcUUID := r.PathValue("mc_uuid")
|
mcUUID := r.PathValue("mc_uuid")
|
||||||
if userID == "" || mcUUID == "" {
|
if userID == "" || mcUUID == "" {
|
||||||
@@ -436,14 +436,13 @@ func (a *API) handleUnlinkAccount(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.unlink_account", userID)
|
a.audit(r, "user.unlink_account", userID)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{"ok": true, "mc_uuid": mcUUID})
|
writeJSON(w, http.StatusOK, map[string]any{"ok": true, "mc_uuid": mcUUID})
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleLinkAccount force-binds a UUID to a user
|
// handleLinkAccount force-binds a UUID to a user
|
||||||
// (POST /users/{id}/links).
|
// (POST /users/{id}/links).
|
||||||
func (a *API) handleLinkAccount(w http.ResponseWriter, r *http.Request) {
|
func (a *API) handleLinkAccount(w http.ResponseWriter, r *http.Request) {
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
userID := r.PathValue("id")
|
userID := r.PathValue("id")
|
||||||
if userID == "" {
|
if userID == "" {
|
||||||
writeError(w, r, errBadRequest)
|
writeError(w, r, errBadRequest)
|
||||||
@@ -489,7 +488,7 @@ func (a *API) handleLinkAccount(w http.ResponseWriter, r *http.Request) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
a.audit(r, p.Email, "user.link_account", userID)
|
a.audit(r, "user.link_account", userID)
|
||||||
writeJSON(w, http.StatusOK, map[string]any{
|
writeJSON(w, http.StatusOK, map[string]any{
|
||||||
"ok": true,
|
"ok": true,
|
||||||
"mc_uuid": body.MCUUID,
|
"mc_uuid": body.MCUUID,
|
||||||
|
|||||||
+35
-7
@@ -6,6 +6,7 @@ import (
|
|||||||
"net/http"
|
"net/http"
|
||||||
|
|
||||||
"felis.lolicon.best/internal/build"
|
"felis.lolicon.best/internal/build"
|
||||||
|
"felis.lolicon.best/internal/imagepin"
|
||||||
"k8s.io/apimachinery/pkg/util/validation"
|
"k8s.io/apimachinery/pkg/util/validation"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -73,7 +74,7 @@ func (a *API) handleBuildImage(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeBuildError(w, r, err)
|
writeBuildError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "image.build", bld.ImageRef)
|
a.audit(r, "image.build", bld.ImageRef)
|
||||||
writeJSON(w, http.StatusAccepted, bld)
|
writeJSON(w, http.StatusAccepted, bld)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -151,7 +152,7 @@ func (a *API) handleBuildLogs(w http.ResponseWriter, r *http.Request) {
|
|||||||
"could not open build logs"))
|
"could not open build logs"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "image.build.logs", id)
|
a.audit(r, "image.build.logs", id)
|
||||||
relayLogStream(w, r, src)
|
relayLogStream(w, r, src)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -162,14 +163,13 @@ func (a *API) handleCancelBuild(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, errBuildUnavailable)
|
writeError(w, r, errBuildUnavailable)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
id := r.PathValue("id")
|
id := r.PathValue("id")
|
||||||
bld, err := a.Builder.Cancel(r.Context(), id)
|
bld, err := a.Builder.Cancel(r.Context(), id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
writeBuildError(w, r, err)
|
writeBuildError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "image.build.cancel", bld.ImageRef)
|
a.audit(r, "image.build.cancel", bld.ImageRef)
|
||||||
writeJSON(w, http.StatusOK, bld)
|
writeJSON(w, http.StatusOK, bld)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -206,7 +206,7 @@ func (a *API) handleAddImage(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeBuildError(w, r, err)
|
writeBuildError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "image.admit", img.ImageRef)
|
a.audit(r, "image.admit", img.ImageRef)
|
||||||
writeJSON(w, http.StatusCreated, img)
|
writeJSON(w, http.StatusCreated, img)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -218,7 +218,6 @@ func (a *API) handleRemoveImage(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeError(w, r, errBuildUnavailable)
|
writeError(w, r, errBuildUnavailable)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
p := principalFromContext(r.Context())
|
|
||||||
ref := r.URL.Query().Get("ref")
|
ref := r.URL.Query().Get("ref")
|
||||||
if ref == "" {
|
if ref == "" {
|
||||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
||||||
@@ -229,7 +228,7 @@ func (a *API) handleRemoveImage(w http.ResponseWriter, r *http.Request) {
|
|||||||
writeBuildError(w, r, err)
|
writeBuildError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
a.audit(r, p.Email, "image.remove", ref)
|
a.audit(r, "image.remove", ref)
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -254,3 +253,32 @@ func writeBuildError(w http.ResponseWriter, r *http.Request, err error) {
|
|||||||
writeError(w, r, err)
|
writeError(w, r, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ImagePinner resolves an image ref to the immutable form a server's spec keeps
|
||||||
|
// (imagepin.Resolver). A ref it does not manage comes back unchanged.
|
||||||
|
type ImagePinner interface {
|
||||||
|
Pin(ctx context.Context, ref string) (string, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// pinImage pins an admitted ref for a server spec. A tag the registry does not
|
||||||
|
// hold is the caller's to fix (build or push it first); any other failure is the
|
||||||
|
// registry being unreachable, and the server is not created or changed without a
|
||||||
|
// pin, since an unpinned ref is exactly what lets a later push move its world.
|
||||||
|
func (a *API) pinImage(ctx context.Context, ref string) (string, error) {
|
||||||
|
if a.Images == nil {
|
||||||
|
return ref, nil
|
||||||
|
}
|
||||||
|
pinned, err := a.Images.Pin(ctx, ref)
|
||||||
|
switch {
|
||||||
|
case errors.Is(err, imagepin.ErrNotFound) && imagepin.Pinned(ref):
|
||||||
|
return "", newError(http.StatusBadRequest, "image_not_in_registry",
|
||||||
|
"the registry no longer holds build %q (nothing referenced it, so it was pruned); pick a current tag", ref)
|
||||||
|
case errors.Is(err, imagepin.ErrNotFound):
|
||||||
|
return "", newError(http.StatusBadRequest, "image_not_in_registry",
|
||||||
|
"image %q is whitelisted but the registry does not hold it; build or push it first", ref)
|
||||||
|
case err != nil:
|
||||||
|
return "", newError(http.StatusServiceUnavailable, "registry_unavailable",
|
||||||
|
"could not resolve image %q to a digest: %v", ref, err)
|
||||||
|
}
|
||||||
|
return pinned, nil
|
||||||
|
}
|
||||||
Loaded 100 of 280 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user