Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
abbe40fa8b | ||
|
|
001f060027 | ||
|
|
d9453a6488 | ||
|
|
2d1592bf01 | ||
|
|
f1414b5cc8 | ||
|
|
0fb43232b5 | ||
|
|
7bf81a0921 | ||
|
|
7f160e2feb | ||
|
|
24373823cc | ||
|
|
d76683f5cb | ||
|
|
b384f6281f | ||
|
|
925cfcf8f8 | ||
|
|
b7d4275ca9 | ||
|
|
a977e229ca | ||
|
|
6ec1b2726c | ||
|
|
234498b85a | ||
|
|
b7fdef7522 | ||
|
|
516c58b543 | ||
|
|
87dee3e5a5 | ||
|
|
e7326e6315 | ||
|
|
1054e62fa9 | ||
|
|
0a66385d9f | ||
|
|
7d82402c18 | ||
|
|
d93c1b6913 | ||
|
|
4757353324 | ||
|
|
084ba1ed9e | ||
|
|
38288e1c60 | ||
|
|
a883c1fe07 | ||
|
|
d3769b5c31 | ||
|
|
9e7f23ca13 | ||
|
|
98295e630e | ||
|
|
88d3dd7121 | ||
|
|
fde677c07e | ||
|
|
dee4d87fd1 | ||
|
|
9f60178ba8 | ||
|
|
73f4c858bf | ||
|
|
f156e385f0 | ||
|
|
bb9c2168df | ||
|
|
06d5e652c6 | ||
|
|
9d03386c83 | ||
|
|
a5307d44d4 | ||
|
|
367ef2678a | ||
|
|
9a5225f77e | ||
|
|
6c3421fe8c | ||
|
|
1a8cccf245 | ||
|
|
34ea733775 | ||
|
|
175c9721d4 | ||
|
|
2151e0cf92 | ||
|
|
3d79293218 | ||
|
|
2f99874d5e | ||
|
|
ea425cffa4 | ||
|
|
90c39afb0c | ||
|
|
3a2166ca87 | ||
|
|
bcf245410a | ||
|
|
6595a2c581 | ||
|
|
e0fa0268e0 | ||
|
|
2779d8f5cf | ||
|
|
8fb3d298ae | ||
|
|
e3ac9cd545 | ||
|
|
0e93e961ed | ||
|
|
5d1da31a48 | ||
|
|
e1c325d594 | ||
|
|
074bd1783c | ||
|
|
0770a4d676 | ||
|
|
fe15b56074 | ||
|
|
7766e8efa4 | ||
|
|
89a0134707 | ||
|
|
489eff4494 | ||
|
|
d44243c0ac | ||
|
|
a660b01efa | ||
|
|
f8f112b8ca | ||
|
|
cc85aac906 | ||
|
|
3ed6603201 | ||
|
|
b1678f78c8 | ||
|
|
6f7b8d7b30 | ||
|
|
989be55b4a |
No files matched your search
@@ -135,6 +135,7 @@ jobs:
|
||||
./shellcheck-v0.11.0/shellcheck -S warning $(git ls-files '*.sh')
|
||||
|
||||
- run: sh deploy/bootstrap_test.sh
|
||||
- run: sh deploy/uninstall_test.sh
|
||||
|
||||
panel:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -164,26 +165,53 @@ jobs:
|
||||
- run: npm test
|
||||
working-directory: panel
|
||||
|
||||
# `tsc -b`: tsconfig.json is a solution file (files: [] plus references), so a plain
|
||||
# `tsc --noEmit` checked nothing and passed with type errors in the tree.
|
||||
- run: npm run typecheck
|
||||
working-directory: panel
|
||||
|
||||
# Rules of hooks and effect dependency lists, with --max-warnings 0.
|
||||
- run: npm run lint
|
||||
working-directory: panel
|
||||
|
||||
# openapi.gen.ts is generated from docs/openapi.yaml and checked in, so the
|
||||
# compile-time parity in src/lib/types.parity.ts needs no generator in the build;
|
||||
# a schema edit that was not regenerated fails here.
|
||||
- name: openapi.gen.ts matches docs/openapi.yaml
|
||||
working-directory: panel
|
||||
run: |
|
||||
npm run gen:api
|
||||
git diff --exit-code -- src/lib/openapi.gen.ts
|
||||
|
||||
# Browser smoke over the mock-mode dev server, in the runner's installed Chrome
|
||||
# (playwright.config.ts sets channel: chrome, so nothing is downloaded).
|
||||
- run: npm run test:e2e
|
||||
working-directory: panel
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
if: failure()
|
||||
with:
|
||||
name: panel-playwright-report
|
||||
path: |
|
||||
panel/playwright-report
|
||||
panel/test-results
|
||||
retention-days: 7
|
||||
|
||||
plugins:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
# The other jobs never touch the Java layer: the plugin jars were only ever
|
||||
# compiled by bootstrap on a live host, and the three test mains under
|
||||
# plugins/*/test were run by hand. JDK 21 plus the Gradle major the plugin
|
||||
# Dockerfiles pin (8.14) is that same toolchain, in CI.
|
||||
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
|
||||
# compiled by bootstrap on a live host, and the test mains under plugins/*/test
|
||||
# were run by hand. JDK 25 is what the plugin build image runs (paper-api 26.x
|
||||
# needs it); each module's wrapper brings the Gradle the image pins.
|
||||
- uses: actions/setup-java@de7274f081f381c8f8158605e0321c36c376e2e6 # v6.0.1
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '21'
|
||||
java-version: '25'
|
||||
|
||||
- uses: gradle/actions/setup-gradle@ed408507eac070d1f99cc633dbcf757c94c7933a # v4.4.3
|
||||
with:
|
||||
gradle-version: '8.14'
|
||||
|
||||
- run: bash plugins/test.sh
|
||||
|
||||
@@ -197,7 +225,7 @@ jobs:
|
||||
# this job nothing ever built them: no install path touches them, and their
|
||||
# gradlew scripts were committed without the exec bit, so the README's
|
||||
# one-liners failed on a fresh clone.
|
||||
- uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4.9.1
|
||||
- uses: actions/setup-java@de7274f081f381c8f8158605e0321c36c376e2e6 # v6.0.1
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '17'
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
# deploy/bootstrap.sh end to end on a fresh Ubuntu 24.04 x86_64 runner: the paths every
|
||||
# host goes through, run for real instead of by hand on a VM.
|
||||
#
|
||||
# install a full install of this commit, then the same commit again (a rerun must
|
||||
# converge without restarting what did not change)
|
||||
# upgrade the newest published release, then this commit on top of it; skipped until
|
||||
# a release exists
|
||||
#
|
||||
# Each job builds the control plane, the game images and the proxy from scratch, about
|
||||
# half an hour of runner time, so this runs on pushes that touch what gets installed, by
|
||||
# hand, and weekly (a moving upstream: apt mirrors, k3s's install script, Adoptium).
|
||||
# deploy/e2e_check.sh holds the assertions.
|
||||
name: e2e
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'deploy/**'
|
||||
- 'cmd/**'
|
||||
- 'internal/**'
|
||||
- 'panel/**'
|
||||
- 'plugins/**'
|
||||
- 'Dockerfile'
|
||||
- 'go.mod'
|
||||
- 'go.sum'
|
||||
- '.github/workflows/e2e.yml'
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: '23 4 * * 1'
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: e2e-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
install:
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 90
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
|
||||
# The images, k3s and the JRE need more room than a stock runner leaves free.
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
# FELIS_SKIP_FETCH builds whatever sits in /opt/felis/src, the way the VM runs
|
||||
# have always been done; the .git directory is what stamps the build. Root owns it
|
||||
# so git, run as root by the installer, does not refuse it as dubious.
|
||||
- name: Stage this commit as the installer's source
|
||||
run: |
|
||||
sudo mkdir -p /opt/felis
|
||||
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
|
||||
sudo chown -R root:root /opt/felis/src
|
||||
|
||||
- name: Install
|
||||
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee install.log
|
||||
|
||||
- name: Check the install
|
||||
run: sudo bash deploy/e2e_check.sh install
|
||||
|
||||
- name: Rerun the same commit
|
||||
run: sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee rerun.log
|
||||
|
||||
- name: Check the rerun
|
||||
run: |
|
||||
sudo bash deploy/e2e_check.sh rerun
|
||||
grep -q 'felis-velocity unchanged; left running' rerun.log
|
||||
|
||||
- name: Diagnostics
|
||||
if: failure()
|
||||
run: |
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
k() { sudo -E /usr/local/bin/k3s kubectl "$@"; }
|
||||
k get pods -A -o wide || true
|
||||
k get events -A --sort-by=.lastTimestamp | tail -n 60 || true
|
||||
for p in $(k get pods -A --no-headers 2>/dev/null | awk '$4 != "Running" && $4 != "Completed" {print $1 "/" $2}'); do
|
||||
k -n "${p%%/*}" describe pod "${p#*/}" | tail -n 40 || true
|
||||
k -n "${p%%/*}" logs "${p#*/}" --all-containers --tail=60 || true
|
||||
done
|
||||
sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
if: always()
|
||||
with:
|
||||
name: e2e-install-logs
|
||||
path: '*.log'
|
||||
if-no-files-found: ignore
|
||||
|
||||
upgrade:
|
||||
runs-on: ubuntu-24.04
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Free disk space
|
||||
run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL
|
||||
|
||||
- name: Find the newest release
|
||||
id: release
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
tag="$(gh release view --repo "$GITHUB_REPOSITORY" --json tagName --jq .tagName 2>/dev/null || true)"
|
||||
if [ -z "$tag" ]; then
|
||||
echo "::notice::no published release yet; the upgrade path has nothing to start from"
|
||||
fi
|
||||
echo "tag=${tag}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The release's own installer, fetching the release's own binary: what a host that
|
||||
# installed that release is running today.
|
||||
- name: Install the newest release
|
||||
if: steps.release.outputs.tag != ''
|
||||
env:
|
||||
TAG: ${{ steps.release.outputs.tag }}
|
||||
TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
git show "${TAG}:deploy/bootstrap.sh" > release-bootstrap.sh
|
||||
sudo FELIS_GITHUB_TOKEN="$TOKEN" FELIS_REF="$TAG" FELIS_INSTALL_MODE=full bash release-bootstrap.sh 2>&1 | tee release.log
|
||||
|
||||
- name: Check the release install
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: sudo bash deploy/e2e_check.sh release
|
||||
|
||||
- name: Upgrade to this commit
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: |
|
||||
sudo rm -rf /opt/felis/src
|
||||
sudo cp -a "$GITHUB_WORKSPACE" /opt/felis/src
|
||||
sudo chown -R root:root /opt/felis/src
|
||||
sudo FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full bash /opt/felis/src/deploy/bootstrap.sh 2>&1 | tee upgrade.log
|
||||
|
||||
- name: Check the upgrade
|
||||
if: steps.release.outputs.tag != ''
|
||||
run: sudo bash deploy/e2e_check.sh upgrade
|
||||
|
||||
- name: Diagnostics
|
||||
if: failure()
|
||||
run: |
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
sudo -E /usr/local/bin/k3s kubectl get pods -A -o wide || true
|
||||
sudo journalctl -u k3s -u felis-velocity -u postgresql --no-pager -n 120 || true
|
||||
|
||||
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
||||
if: always()
|
||||
with:
|
||||
name: e2e-upgrade-logs
|
||||
path: '*.log'
|
||||
if-no-files-found: ignore
|
||||
@@ -19,7 +19,7 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
||||
- **Web 控制面板**:浏览器中查看服务器状态、在线玩家与资源用量,管理备份与恢复。
|
||||
- **备份与恢复**:一键把整服数据(世界、配置、插件/模组,即整个 /data 卷)打包进集群内的归档库,支持从任意备份点回滚;默认安装就已启用(归档 PVC 与路径由安装器一并生成)。
|
||||
- **控制面数据库备份**:账号、服务器归属、配额与存档索引所在的数据库每天自动备份,每次升级迁移前先快照,出错可用 `felis db restore` 整库原子回滚;面板「维护与备份」页显示备份是否新鲜(见 [故障排查 §16](docs/troubleshooting.md))。
|
||||
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。
|
||||
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。过期备份无论是否开启都会每天清理。
|
||||
- **多核心支持**:兼容 Paper、Fabric、Forge、NeoForge,经由 Velocity 代理统一入口。
|
||||
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建;构建产物进入镜像白名单,可直接选用为服务器镜像完成部署。
|
||||
- **Passkey 登录**:支持指纹、面容、硬件密钥等无密码认证方式。
|
||||
@@ -27,7 +27,7 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
||||
|
||||
## 使用方式
|
||||
|
||||
在准备好的 Linux 主机上执行:
|
||||
在准备好的 Linux 主机上执行(已验证的发行版与架构见 [运维手册 §1](docs/operations.md#1-supported-hosts):CentOS Stream 9 aarch64 实机验证,Ubuntu 24.04 x86_64 每次推送由 CI 跑全新安装、重跑与升级):
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
|
||||
+4
-1
@@ -27,7 +27,10 @@ Table of Contents
|
||||
|
||||
## Getting Started
|
||||
|
||||
On a prepared Linux host, run:
|
||||
On a prepared Linux host, run (the verified distributions and architectures are listed in
|
||||
[operations §1](docs/operations.md#1-supported-hosts): CentOS Stream 9 on aarch64 is verified
|
||||
on a real host, and Ubuntu 24.04 on x86_64 gets a fresh install, rerun and upgrade in CI on
|
||||
every push):
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
|
||||
+6
-4
@@ -22,15 +22,17 @@ var bootstrapAssets embed.FS
|
||||
// developer's working tree carries gradle output (plugins/*/build, plugins/*/bin,
|
||||
// and for the modded loaders a decompiled Minecraft under build/) which would
|
||||
// otherwise be baked into every felis binary. Keep them explicit — add a source
|
||||
// directory here, never a parent.
|
||||
// directory here, never a parent. Each module's gradle/verification-metadata.xml rides
|
||||
// along, since Gradle builds unverified without it; the wrapper stays out, because the
|
||||
// image builds run the pinned build image's own gradle.
|
||||
//
|
||||
//go:embed deploy/game-stack.lock
|
||||
//go:embed deploy/limbo/Dockerfile deploy/limbo/entrypoint.sh
|
||||
//go:embed deploy/lobby/Dockerfile deploy/lobby/entrypoint.sh
|
||||
//go:embed deploy/paper/Dockerfile deploy/paper/entrypoint.sh
|
||||
//go:embed plugins/limbo/build.gradle plugins/limbo/settings.gradle plugins/limbo/src
|
||||
//go:embed plugins/paper/build.gradle plugins/paper/settings.gradle plugins/paper/src
|
||||
//go:embed plugins/velocity/build.gradle plugins/velocity/settings.gradle plugins/velocity/src
|
||||
//go:embed plugins/limbo/build.gradle plugins/limbo/settings.gradle plugins/limbo/src plugins/limbo/gradle/verification-metadata.xml
|
||||
//go:embed plugins/paper/build.gradle plugins/paper/settings.gradle plugins/paper/src plugins/paper/gradle/verification-metadata.xml
|
||||
//go:embed plugins/velocity/build.gradle plugins/velocity/settings.gradle plugins/velocity/src plugins/velocity/gradle/verification-metadata.xml
|
||||
//go:embed plugins/shared/src
|
||||
var gameStackAssets embed.FS
|
||||
|
||||
|
||||
+185
-11
@@ -1,6 +1,7 @@
|
||||
package felis
|
||||
|
||||
import (
|
||||
"encoding/xml"
|
||||
"io/fs"
|
||||
"os"
|
||||
"regexp"
|
||||
@@ -210,17 +211,7 @@ func requireEmbedded(t *testing.T, path string) {
|
||||
// with a strict KEY=value parser that dies on anything unexpected, so a malformed lock is a
|
||||
// failed install on every host. Check the shipped copy the same way here.
|
||||
func TestGameStackLockIsComplete(t *testing.T) {
|
||||
lock := map[string]string{}
|
||||
for line := range strings.SplitSeq(readGameStackFile(t, "deploy/game-stack.lock"), "\n") {
|
||||
if line == "" || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
k, v, ok := strings.Cut(line, "=")
|
||||
if !ok {
|
||||
t.Fatalf("not a KEY=value line: %q", line)
|
||||
}
|
||||
lock[k] = v
|
||||
}
|
||||
lock := gameStackLock(t)
|
||||
m := regexp.MustCompile(`GAME_STACK_LOCK_KEYS="([^"]*)"`).FindStringSubmatch(BootstrapScript())
|
||||
if m == nil {
|
||||
t.Fatal("bootstrap.sh no longer declares GAME_STACK_LOCK_KEYS")
|
||||
@@ -313,6 +304,189 @@ func TestDockerfileBaseImagesArePinnedByDigest(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The plugin jars are built in three places the installer controls — the lobby and limbo
|
||||
// image builds and bootstrap's Velocity build — and through each module's wrapper by a
|
||||
// developer or CI. A tag alone is whatever it points at on build day, and two Gradle
|
||||
// versions are two chances for a build to pass in one place and break in the other, so
|
||||
// all of them run one image, pinned by digest, whose Gradle is the wrappers' Gradle.
|
||||
func TestPluginBuildsRunOnePinnedGradle(t *testing.T) {
|
||||
sources := map[string]string{
|
||||
"deploy/lobby/Dockerfile": readGameStackFile(t, "deploy/lobby/Dockerfile"),
|
||||
"deploy/limbo/Dockerfile": readGameStackFile(t, "deploy/limbo/Dockerfile"),
|
||||
"deploy/bootstrap.sh": BootstrapScript(),
|
||||
}
|
||||
anyRef := regexp.MustCompile(`gradle:[\w.-]+(@sha256:\w+)?`)
|
||||
pinned := regexp.MustCompile(`^gradle:(\d+\.\d+(?:\.\d+)?)-jdk\d+@sha256:[0-9a-f]{64}$`)
|
||||
images := map[string]bool{}
|
||||
gradle := ""
|
||||
for name, body := range sources {
|
||||
refs := anyRef.FindAllString(body, -1)
|
||||
if len(refs) == 0 {
|
||||
t.Errorf("%s names no gradle image", name)
|
||||
}
|
||||
for _, ref := range refs {
|
||||
m := pinned.FindStringSubmatch(ref)
|
||||
if m == nil {
|
||||
t.Errorf("%s: %s is not a gradle image pinned by digest", name, ref)
|
||||
continue
|
||||
}
|
||||
images[ref] = true
|
||||
gradle = m[1]
|
||||
}
|
||||
}
|
||||
if len(images) != 1 {
|
||||
t.Fatalf("the plugin builds use %d different gradle images, want one: %v", len(images), images)
|
||||
}
|
||||
// bootstrap names the image once and has to spend it where it builds the jar.
|
||||
if !strings.Contains(BootstrapScript(), `"$PLUGIN_BUILD_IMAGE" gradle --no-daemon clean build`) {
|
||||
t.Error("build_velocity_plugin does not build in $PLUGIN_BUILD_IMAGE")
|
||||
}
|
||||
|
||||
for _, module := range []string{"velocity", "paper", "limbo"} {
|
||||
props := wrapperProperties(t, module)
|
||||
if want := "gradle-" + gradle + "-bin.zip"; !strings.HasSuffix(props["distributionUrl"], "/"+want) {
|
||||
t.Errorf("plugins/%s wrapper runs %s; the image builds run Gradle %s", module, props["distributionUrl"], gradle)
|
||||
}
|
||||
}
|
||||
// The mods are no part of the install, but a wrapper without a checksum runs whatever
|
||||
// the download handed it.
|
||||
for _, module := range []string{"velocity", "paper", "limbo", "fabric", "forge", "neoforge"} {
|
||||
if sum := wrapperProperties(t, module)["distributionSha256Sum"]; !regexp.MustCompile(`^[0-9a-f]{64}$`).MatchString(sum) {
|
||||
t.Errorf("plugins/%s wrapper pins no distribution sha256 (got %q)", module, sum)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Each plugin compiles against the API of the exact build the install runs, and Gradle
|
||||
// checks those bytes against the module's verification file. Nothing but this test ties
|
||||
// the three to deploy/game-stack.lock: a lock refresh that leaves them behind builds the
|
||||
// lobby against yesterday's API, or fails every image build on a checksum the file does
|
||||
// not have.
|
||||
func TestPluginApisAreTheLockedBuilds(t *testing.T) {
|
||||
lock := gameStackLock(t)
|
||||
|
||||
// Paper: paper-<mc>-<build>.jar runs; paper-api <mc>.build.<build>-<channel> compiles.
|
||||
jar := regexp.MustCompile(`/paper-([^/]+)-(\d+)\.jar$`).FindStringSubmatch(lock["PAPER_JAR_URL"])
|
||||
if jar == nil {
|
||||
t.Fatalf("PAPER_JAR_URL %s does not name paper-<mc>-<build>.jar", lock["PAPER_JAR_URL"])
|
||||
}
|
||||
dep := regexp.MustCompile(`compileOnly 'io\.papermc\.paper:paper-api:([^']+)'`).
|
||||
FindStringSubmatch(readGameStackFile(t, "plugins/paper/build.gradle"))
|
||||
if dep == nil {
|
||||
t.Fatal("plugins/paper/build.gradle declares no paper-api dependency")
|
||||
}
|
||||
if !regexp.MustCompile(`^` + regexp.QuoteMeta(jar[1]+".build."+jar[2]) + `(-[a-z]+)?$`).MatchString(dep[1]) {
|
||||
t.Errorf("paper-api %s is not the API of the locked server paper-%s-%s.jar", dep[1], jar[1], jar[2])
|
||||
}
|
||||
requireVerified(t, "paper", "io.papermc.paper", "paper-api", dep[1])
|
||||
|
||||
// Limbo: the lock's release, passed to the image build, which refuses to guess one.
|
||||
limbo := lock["LIMBO_VERSION"]
|
||||
if !strings.Contains(BootstrapScript(), `--build-arg LIMBO_VERSION="$LIMBO_VERSION"`) {
|
||||
t.Error("bootstrap.sh does not pass the locked LIMBO_VERSION to the limbo image build")
|
||||
}
|
||||
dockerfile := readGameStackFile(t, "deploy/limbo/Dockerfile")
|
||||
if !regexp.MustCompile(`(?m)^ARG LIMBO_VERSION$`).MatchString(dockerfile) ||
|
||||
!strings.Contains(dockerfile, `if [ -z "${LIMBO_VERSION:-}" ]`) {
|
||||
t.Error("deploy/limbo/Dockerfile does not require LIMBO_VERSION; a build without it would " +
|
||||
"compile against a version nobody chose")
|
||||
}
|
||||
// LOOHP publishes the jar the login gate runs as the Limbo API artifact itself, so the
|
||||
// checksum Gradle holds for it is the lock's: compiled-against and running are one file.
|
||||
if got := requireVerified(t, "limbo", "com.loohp", "Limbo", limbo)["Limbo-"+limbo+".jar"]; got != lock["LIMBO_JAR_SHA256"] {
|
||||
t.Errorf("verification-metadata.xml holds %q for Limbo-%s.jar; the login gate runs %s", got, limbo, lock["LIMBO_JAR_SHA256"])
|
||||
}
|
||||
|
||||
// Velocity: the API default is the proxy the install runs.
|
||||
api := regexp.MustCompile(`findProperty\('velocityApi'\) \?: '([^']+)'`).
|
||||
FindStringSubmatch(readGameStackFile(t, "plugins/velocity/build.gradle"))
|
||||
if api == nil {
|
||||
t.Fatal("plugins/velocity/build.gradle has no velocityApi default")
|
||||
}
|
||||
if api[1] != lock["VELOCITY_VERSION"] {
|
||||
t.Errorf("velocity-api defaults to %s; the install runs Velocity %s", api[1], lock["VELOCITY_VERSION"])
|
||||
}
|
||||
requireVerified(t, "velocity", "com.velocitypowered", "velocity-api", api[1])
|
||||
}
|
||||
|
||||
// requireVerified asserts the module's shipped verification file checks metadata and
|
||||
// pins group:name:version, and returns that component's artifact sha256s by file name.
|
||||
func requireVerified(t *testing.T, module, group, name, version string) map[string]string {
|
||||
t.Helper()
|
||||
path := "plugins/" + module + "/gradle/verification-metadata.xml"
|
||||
var doc struct {
|
||||
VerifyMetadata bool `xml:"configuration>verify-metadata"`
|
||||
Components []struct {
|
||||
Group string `xml:"group,attr"`
|
||||
Name string `xml:"name,attr"`
|
||||
Version string `xml:"version,attr"`
|
||||
Artifacts []struct {
|
||||
Name string `xml:"name,attr"`
|
||||
SHA256 []struct {
|
||||
Value string `xml:"value,attr"`
|
||||
} `xml:"sha256"`
|
||||
} `xml:"artifact"`
|
||||
} `xml:"components>component"`
|
||||
}
|
||||
if err := xml.Unmarshal([]byte(readGameStackFile(t, path)), &doc); err != nil {
|
||||
t.Fatalf("%s: %v", path, err)
|
||||
}
|
||||
if !doc.VerifyMetadata {
|
||||
t.Errorf("%s does not verify metadata; a swapped pom could redirect the graph", path)
|
||||
}
|
||||
for _, c := range doc.Components {
|
||||
if c.Group != group || c.Name != name || c.Version != version {
|
||||
continue
|
||||
}
|
||||
sums := map[string]string{}
|
||||
for _, a := range c.Artifacts {
|
||||
if len(a.SHA256) > 0 {
|
||||
sums[a.Name] = a.SHA256[0].Value
|
||||
}
|
||||
}
|
||||
if sums[name+"-"+version+".jar"] == "" {
|
||||
t.Errorf("%s pins %s:%s:%s but no sha256 for its jar", path, group, name, version)
|
||||
}
|
||||
return sums
|
||||
}
|
||||
t.Errorf("%s has no checksum for %s:%s:%s; the build would refuse it", path, group, name, version)
|
||||
return nil
|
||||
}
|
||||
|
||||
// wrapperProperties reads a module's gradle-wrapper.properties off disk: the wrappers are
|
||||
// for developers and CI, and nothing embeds them.
|
||||
func wrapperProperties(t *testing.T, module string) map[string]string {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile("plugins/" + module + "/gradle/wrapper/gradle-wrapper.properties")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
props := map[string]string{}
|
||||
for line := range strings.SplitSeq(string(b), "\n") {
|
||||
if k, v, ok := strings.Cut(strings.TrimSpace(line), "="); ok && !strings.HasPrefix(k, "#") {
|
||||
props[k] = strings.ReplaceAll(v, `\:`, ":")
|
||||
}
|
||||
}
|
||||
return props
|
||||
}
|
||||
|
||||
// gameStackLock parses the shipped deploy/game-stack.lock the way bootstrap.sh does.
|
||||
func gameStackLock(t *testing.T) map[string]string {
|
||||
t.Helper()
|
||||
lock := map[string]string{}
|
||||
for line := range strings.SplitSeq(readGameStackFile(t, "deploy/game-stack.lock"), "\n") {
|
||||
if line == "" || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
k, v, ok := strings.Cut(line, "=")
|
||||
if !ok {
|
||||
t.Fatalf("not a KEY=value line: %q", line)
|
||||
}
|
||||
lock[k] = v
|
||||
}
|
||||
return lock
|
||||
}
|
||||
|
||||
func readGameStackFile(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
b, err := gameStackAssets.ReadFile(name)
|
||||
|
||||
+231
-51
@@ -8,8 +8,10 @@ import (
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
@@ -29,13 +31,15 @@ import (
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
"felis.lolicon.best/internal/registryprune"
|
||||
"felis.lolicon.best/internal/restore"
|
||||
"felis.lolicon.best/internal/store"
|
||||
"felis.lolicon.best/internal/retention"
|
||||
"felis.lolicon.best/internal/submit"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
"k8s.io/client-go/rest"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/cache"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
@@ -60,10 +64,8 @@ func authSourcesFromConfig(configured []config.AuthSourceConfig) []api.AuthSourc
|
||||
}
|
||||
|
||||
// cmdAPI runs felis-api: two listeners, two middleware chains (spec §7). The
|
||||
// internal face (service token) is fully wired. The external face is wired but
|
||||
// fails closed until an Access JWKS key function is configured — the verifier's
|
||||
// audience logic is unit-tested (internal/api), the JWKS source is a deployment
|
||||
// integration point.
|
||||
// internal face authenticates per-caller service tokens; the external face
|
||||
// authenticates the local session cookie.
|
||||
func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("api", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
@@ -86,9 +88,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
// Load already refused a malformed [audit] retention.
|
||||
auditRetention, _ := cfg.Audit.RetentionPeriod()
|
||||
if auditRetention == 0 {
|
||||
fmt.Fprintln(stdout, "felis api: audit rows are kept forever ([audit] retention = \"forever\")")
|
||||
} else {
|
||||
fmt.Fprintf(stdout, "felis api: audit rows older than %d days are deleted ([audit] retention; export them first with felis db audit-export)\n", int(auditRetention/(24*time.Hour)))
|
||||
}
|
||||
|
||||
ctx := ctrl.SetupSignalHandler()
|
||||
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
// Before anything serves: an api on a schema it was not built for answers with
|
||||
// errors, or writes rows the other version cannot read.
|
||||
drv, err := openStore(ctx, cfg.Database.URL, false)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: open database: %v\n", err)
|
||||
return 1
|
||||
@@ -117,15 +129,22 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
|
||||
metrics.SetBuildInfo("api", resolvedVersion())
|
||||
|
||||
token := os.Getenv("FELIS_SERVICE_TOKEN")
|
||||
if token == "" {
|
||||
fmt.Fprintln(stderr, "felis api: warning: FELIS_SERVICE_TOKEN unset — internal face will reject all callers")
|
||||
internalAuth, err := internalCallerTokens(os.Getenv)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: internal face tokens: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
for _, ct := range naming.CallerTokens {
|
||||
if internalAuth[api.Caller(ct.Caller)] == "" {
|
||||
fmt.Fprintf(stderr, "felis api: warning: %s unset — the internal face turns the %s caller away\n", ct.APIEnv, ct.Caller)
|
||||
}
|
||||
}
|
||||
|
||||
// Email one-time codes go through the [smtp] relay when one is configured; the
|
||||
// password is read from the env var password_ref names (default SMTPPasswordEnv,
|
||||
// injected from the felis-smtp Secret). No [smtp] host ⇒ mailer stays nil and
|
||||
// deliverOTP logs each code server-side (the pre-SMTP bootstrap posture).
|
||||
// every door that mails a code answers 503 mail_unavailable: a code that is
|
||||
// not mailed is never written anywhere else either.
|
||||
var mailer api.OTPMailer
|
||||
if cfg.SMTP.Host != "" {
|
||||
passRef := cfg.SMTP.PasswordRef
|
||||
@@ -136,15 +155,12 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
if cfg.SMTP.Username != "" && password == "" {
|
||||
fmt.Fprintf(stderr, "felis api: warning: [smtp] username is set but credentials env %s is empty — OTP sends will fail AUTH\n", passRef)
|
||||
}
|
||||
mailer = &mail.SMTP{
|
||||
Host: cfg.SMTP.Host,
|
||||
Port: cfg.SMTP.Port,
|
||||
From: cfg.SMTP.From,
|
||||
Username: cfg.SMTP.Username,
|
||||
Password: password,
|
||||
mailer = smtpRelay(cfg.SMTP, password)
|
||||
if !cfg.SMTP.TLSRequired() {
|
||||
fmt.Fprintf(stderr, "felis api: warning: [smtp] %s may be sent codes without TLS (require_tls off or a relay on this host)\n", cfg.SMTP.Host)
|
||||
}
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis api: [smtp] not configured — email one-time codes are logged, not mailed")
|
||||
fmt.Fprintln(stderr, "felis api: [smtp] not configured — email sign-in and verification are off (503 mail_unavailable); sign in with a passkey, or run felis setup to add a relay")
|
||||
}
|
||||
|
||||
// Build subsystem (spec §16): the weak-SA build Job runs in the configured
|
||||
@@ -157,9 +173,10 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
buildCfg.FelisImage = os.Getenv("FELIS_IMAGE")
|
||||
buildJobs := build.NewK8sJobs(cl, buildCfg)
|
||||
builder := &build.Builder{
|
||||
Store: build.NewPGStore(drv.DB()),
|
||||
Jobs: buildJobs,
|
||||
Config: buildCfg,
|
||||
Store: build.NewPGStore(drv.DB()),
|
||||
Jobs: buildJobs,
|
||||
Config: buildCfg,
|
||||
Outcomes: build.NewK8sOutcomes(clientset, buildCfg),
|
||||
}
|
||||
go probeBuildUserNamespaces(ctx, buildJobs, buildCfg, stderr)
|
||||
|
||||
@@ -213,6 +230,9 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
ContextBaseURL: internalAPIBaseURL(),
|
||||
Blobs: blobs,
|
||||
}
|
||||
if blobs != nil {
|
||||
submissions.Parts = &submit.PartStore{Dir: uploadPartsDir(contextBase)}
|
||||
}
|
||||
if v := cfg.Registry.UserUploadsMaxBytes; v != "" {
|
||||
if n, err := parseByteSize(v); err != nil || n <= 0 {
|
||||
fmt.Fprintf(stderr, "felis api: [registry] user_uploads_max_bytes %q is not a positive size such as 4Gi; keeping the default\n", v)
|
||||
@@ -220,6 +240,11 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
submissions.MaxStoredBytesTotal = n
|
||||
}
|
||||
}
|
||||
if n, err := contextMaxBytes(cfg); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: [registry] context_max_bytes %q is not a positive size such as 512Mi; keeping the default\n", cfg.Registry.ContextMaxBytes)
|
||||
} else {
|
||||
submissions.MaxContextBytes = n
|
||||
}
|
||||
|
||||
// Restore subsystem (spec §7): the weak-SA restore Job mounts the target
|
||||
// world PVC + the backup PVC and runs `felis restore`. It needs deployment-
|
||||
@@ -274,6 +299,9 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// the same store the auth handlers write to, so a login and the next request
|
||||
// agree on what local auth knows.
|
||||
repo := api.NewPGRepo(drv.DB())
|
||||
if err := api.RegisterStorePool(drv.DB()); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: store pool metrics unavailable: %v\n", err)
|
||||
}
|
||||
|
||||
// The owner's on-demand backup levers come from [archive], the same keys the
|
||||
// backup Job and the reaper read. A malformed key leaves the defaults in
|
||||
@@ -284,7 +312,13 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
rcfg = reaper.DefaultConfig()
|
||||
}
|
||||
|
||||
cluster := api.NewK8sCluster(cl, cfg.K8s.Namespace)
|
||||
serverCache, serversSynced, err := startServerCache(ctx, restCfg, scheme, cfg.K8s.Namespace, stderr)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: MinecraftServer cache: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
cluster := api.NewK8sCluster(cl, cfg.K8s.Namespace).WithServerCache(serverCache, serversSynced)
|
||||
jobStatus := api.NewK8sJobStatus(cl, cfg.K8s.Namespace)
|
||||
a := &api.API{
|
||||
Repo: repo,
|
||||
Cluster: cluster,
|
||||
@@ -292,26 +326,24 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
Logs: api.NewK8sLogStreamer(clientset, cfg.K8s.Namespace),
|
||||
// Build-log stream (spec §16) is scoped to the BUILD namespace — the same
|
||||
// value the Builder renders Jobs into — so it follows where build Pods run.
|
||||
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
|
||||
Internal: api.BearerTokenAuth{Token: token},
|
||||
Builder: builder,
|
||||
Images: imagePinner(cfg.Registry.URL),
|
||||
Restorer: restorer,
|
||||
Backuper: backuper,
|
||||
JobStatus: api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
|
||||
Files: files,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
// The external face is fronted by SessionAuth: it prefers a local session
|
||||
// cookie (minted by the passwordless doors) and otherwise delegates to the
|
||||
// Cloudflare-Access JWT verifier, so both auth models coexist on one face. The
|
||||
// delegate's Keyfunc is intentionally nil — the JWT path fails closed until a
|
||||
// JWKS-backed key function is wired (deployment integration point) — while the
|
||||
// local session path is live the moment `felis breakGlass` flips
|
||||
// local_auth_enabled on.
|
||||
BuildLogs: api.NewK8sBuildLogStreamer(clientset, cfg.Registry.BuildNamespace),
|
||||
Internal: internalAuth,
|
||||
Builder: builder,
|
||||
Images: imagePinner(cfg.Registry.URL),
|
||||
Restorer: restorer,
|
||||
Backuper: backuper,
|
||||
JobStatus: jobStatus,
|
||||
// A restore starts with a safety snapshot; settleRestoreChains starts the
|
||||
// restore behind each one.
|
||||
RestoreChains: jobStatus,
|
||||
Files: files,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
// The external face authenticates the local session cookie the sign-in doors
|
||||
// mint, live once `felis breakGlass` flips local_auth_enabled on. Cloudflare
|
||||
// Access, when the install sits behind it, is enforced at the edge only.
|
||||
External: api.SessionAuth{
|
||||
Repo: repo,
|
||||
Delegate: api.AccessVerifier{Audience: cfg.Auth.AccessJWTAud},
|
||||
RootDomain: cfg.Server.RootDomain,
|
||||
AdminHostname: cfg.Auth.AdminHostname,
|
||||
},
|
||||
@@ -340,7 +372,6 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
ClientIPHeader: cfg.Auth.EffectiveClientIPHeader(),
|
||||
MailLimit: mailLimit(cfg.SMTP.MaxPerHour),
|
||||
}
|
||||
fmt.Fprintln(stderr, "felis api: external face fails closed (Access JWKS key function not configured)")
|
||||
if a.ClientIPHeader != "" {
|
||||
fmt.Fprintf(stderr, "felis api: sign-in rate limit keys on the %s header\n", a.ClientIPHeader)
|
||||
} else {
|
||||
@@ -386,7 +417,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
externalHandler := panel.Handler(a.ExternalHandler(), cfg.Server.RootDomain,
|
||||
defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname),
|
||||
defaultAdminHostname(cfg.Server.RootDomain, cfg.Auth.AdminHostname),
|
||||
resolvedVersion())
|
||||
cfg.Velocity.GamePort, resolvedVersion())
|
||||
internalSrv := newAPIServer(*internalAddr, a.InternalHandler())
|
||||
externalSrv := newAPIServer(cfg.Server.Listen, externalHandler)
|
||||
|
||||
@@ -408,21 +439,25 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// and advance any whose Job has reached a terminal phase. GET on a build also
|
||||
// reconciles it, but this loop converges builds nobody is polling.
|
||||
go reconcileBuilds(ctx, builder, stderr)
|
||||
go settleRestoreChains(ctx, a, stderr)
|
||||
|
||||
if pruner := registryPruner(cfg, builder.Store, cluster, stderr); pruner != nil {
|
||||
go pruner.Loop(ctx, registryPruneInterval)
|
||||
}
|
||||
go reapRejectedContexts(ctx, submissions, stderr)
|
||||
go retention.Loop(ctx, drv.DB(), retention.Policy{Audit: auditRetention}, retentionInterval, slog.Default())
|
||||
|
||||
servers := []*http.Server{internalSrv, externalSrv}
|
||||
if httpsSrv != nil {
|
||||
servers = append(servers, httpsSrv)
|
||||
}
|
||||
for _, srv := range servers {
|
||||
srv.RegisterOnShutdown(a.CloseStreams)
|
||||
}
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
_ = internalSrv.Shutdown(shutdownCtx)
|
||||
_ = externalSrv.Shutdown(shutdownCtx)
|
||||
if httpsSrv != nil {
|
||||
_ = httpsSrv.Shutdown(shutdownCtx)
|
||||
}
|
||||
shutdownServers(servers, apiShutdownGrace, stderr)
|
||||
return 0
|
||||
case err := <-errc:
|
||||
if err != nil && err != http.ErrServerClosed {
|
||||
@@ -442,8 +477,30 @@ const (
|
||||
// apiIdleTimeout caps how long a kept-alive connection may sit idle between
|
||||
// requests before the server closes it, bounding idle-connection exhaustion.
|
||||
apiIdleTimeout = 120 * time.Second
|
||||
// apiShutdownGrace is how long the listeners drain after SIGTERM. The pod gets
|
||||
// the Kubernetes default of 30s before SIGKILL; this leaves the rest for the
|
||||
// process to exit.
|
||||
apiShutdownGrace = 20 * time.Second
|
||||
)
|
||||
|
||||
// shutdownServers drains every listener at once under one deadline: in turn, a
|
||||
// slow first listener would spend the time the others needed. Log streams end
|
||||
// through RegisterOnShutdown (API.CloseStreams); what is still running when the
|
||||
// deadline passes is cut off with the process.
|
||||
func shutdownServers(servers []*http.Server, grace time.Duration, stderr io.Writer) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), grace)
|
||||
defer cancel()
|
||||
var wg sync.WaitGroup
|
||||
for _, srv := range servers {
|
||||
wg.Go(func() {
|
||||
if err := srv.Shutdown(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: shutdown %s: %v\n", srv.Addr, err)
|
||||
}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// newAPIServer builds an http.Server with hardened header/idle timeouts (gosec
|
||||
// G112) shared by all three felis-api listeners (internal, external, https).
|
||||
// WriteTimeout and ReadTimeout are deliberately LEFT UNSET: the external and https
|
||||
@@ -481,6 +538,9 @@ func buildConfig(cfg *config.Config) build.Config {
|
||||
MaxConcurrent: cfg.Registry.MaxConcurrentBuilds,
|
||||
TrivyDBRepository: cfg.Registry.TrivyDBRepository,
|
||||
TrivyJavaDBRepository: cfg.Registry.TrivyJavaDBRepository,
|
||||
ScanFailOn: cfg.Registry.ScanFailOn,
|
||||
ScanFailUnfixed: cfg.Registry.ScanFailUnfixed,
|
||||
ScanAccept: cfg.Registry.ScanAccept,
|
||||
// The submit lane's derived context URLs live here; the fetch step's
|
||||
// service token goes nowhere else.
|
||||
ContextOrigin: internalAPIBaseURL(),
|
||||
@@ -624,10 +684,30 @@ func reconcileBuilds(ctx context.Context, b *build.Builder, stderr io.Writer) {
|
||||
}
|
||||
}
|
||||
|
||||
// settleRestoreChains starts the restore behind each safety snapshot that has
|
||||
// finished (and gives up the one behind a snapshot that failed). The world stays
|
||||
// locked in between, so the interval is how long a finished snapshot keeps the
|
||||
// server down before its restore begins.
|
||||
func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) {
|
||||
t := time.NewTicker(5 * time.Second)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
if err := a.SettleRestoreChains(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: restore chains: %v\n", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// reapRejectedContexts deletes, once an hour, the uploaded contexts of
|
||||
// submissions rejected more than submit.RejectedContextRetention ago. Without it a
|
||||
// rejected modpack keeps its bytes on the uploads store (and against its
|
||||
// submitter's budget) until an admin deletes the row.
|
||||
// submissions rejected more than submit.RejectedContextRetention ago, and the
|
||||
// chunked uploads left untouched for submit.StalePartRetention. Without it a
|
||||
// rejected modpack or an abandoned upload keeps its bytes on the uploads store
|
||||
// (and against its submitter's budget) until an admin deletes the row.
|
||||
func reapRejectedContexts(ctx context.Context, m *submit.Manager, stderr io.Writer) {
|
||||
t := time.NewTicker(time.Hour)
|
||||
defer t.Stop()
|
||||
@@ -639,6 +719,13 @@ func reapRejectedContexts(ctx context.Context, m *submit.Manager, stderr io.Writ
|
||||
if n > 0 {
|
||||
fmt.Fprintf(stderr, "felis api: deleted the uploaded contexts of %d rejected submission(s)\n", n)
|
||||
}
|
||||
n, err = m.ReapStaleParts(submit.StalePartRetention)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: reap abandoned uploads: %v\n", err)
|
||||
}
|
||||
if n > 0 {
|
||||
fmt.Fprintf(stderr, "felis api: deleted %d abandoned chunked upload(s)\n", n)
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
@@ -647,6 +734,11 @@ func reapRejectedContexts(ctx context.Context, m *submit.Manager, stderr io.Writ
|
||||
}
|
||||
}
|
||||
|
||||
// retentionInterval spaces the runs that delete spent sign-in rows and audit rows
|
||||
// past [audit] retention. The rows are spent for weeks before they go, so a few
|
||||
// runs a day keep the tables flat.
|
||||
const retentionInterval = 6 * time.Hour
|
||||
|
||||
// registryPruneInterval spaces the registry pruner's runs. The registry-gc
|
||||
// sidecar sweeps once a day, so pruning more often only changes which sweep frees
|
||||
// a layer.
|
||||
@@ -747,3 +839,91 @@ func imagePinner(registry string) api.ImagePinner {
|
||||
}
|
||||
return imagepin.Resolver{Registry: registry}
|
||||
}
|
||||
|
||||
// internalCallerTokens reads each internal caller's token from the env var the
|
||||
// Deployment feeds it from (naming.CallerTokens). Two callers sharing a value
|
||||
// would make the caller ambiguous, so that refuses to start.
|
||||
func internalCallerTokens(getenv func(string) string) (api.CallerTokens, error) {
|
||||
tokens := map[api.Caller]string{}
|
||||
for _, ct := range naming.CallerTokens {
|
||||
tokens[api.Caller(ct.Caller)] = strings.TrimSpace(getenv(ct.APIEnv))
|
||||
}
|
||||
return api.NewCallerTokens(tokens)
|
||||
}
|
||||
|
||||
// smtpRelay is the relay [smtp] names, with the resolved password and the TLS
|
||||
// posture config.SMTPConfig.TLSRequired picks. felis api, the reaper and the
|
||||
// watchdog all send through it, so none can drift to a weaker posture.
|
||||
func smtpRelay(c config.SMTPConfig, password string) *mail.SMTP {
|
||||
return &mail.SMTP{
|
||||
Host: c.Host,
|
||||
Port: c.Port,
|
||||
From: c.From,
|
||||
Username: c.Username,
|
||||
Password: password,
|
||||
RequireTLS: c.TLSRequired(),
|
||||
}
|
||||
}
|
||||
|
||||
// startServerCache starts the informer that serves the api's fleet-wide
|
||||
// MinecraftServer reads (api.K8sCluster.WithServerCache): one watch on the
|
||||
// namespace instead of a full List per velocity pull, fleet page and wake. It
|
||||
// caches MinecraftServers only — ReaderFailOnMissingInformer turns any other read
|
||||
// through it into an error rather than a new informer the api's Role cannot back —
|
||||
// indexes spec.subdomain for GetBySubdomain, and drops managedFields to keep the
|
||||
// copy small. It returns without waiting: the reads block until the first list
|
||||
// lands and /readyz reports not-ready until then.
|
||||
func startServerCache(ctx context.Context, cfg *rest.Config, scheme *runtime.Scheme, namespace string, stderr io.Writer) (cache.Cache, func() bool, error) {
|
||||
c, err := cache.New(cfg, cache.Options{
|
||||
Scheme: scheme,
|
||||
DefaultNamespaces: map[string]cache.Config{namespace: {}},
|
||||
DefaultTransform: cache.TransformStripManagedFields(),
|
||||
ReaderFailOnMissingInformer: true,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if err := c.IndexField(ctx, &v1alpha1.MinecraftServer{}, api.SubdomainIndex, api.SubdomainOf); err != nil {
|
||||
return nil, nil, fmt.Errorf("index %s: %w", api.SubdomainIndex, err)
|
||||
}
|
||||
inf, err := c.GetInformer(ctx, &v1alpha1.MinecraftServer{})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
go func() {
|
||||
if err := c.Start(ctx); err != nil {
|
||||
fmt.Fprintf(stderr, "felis api: MinecraftServer cache stopped: %v\n", err)
|
||||
}
|
||||
}()
|
||||
return c, inf.HasSynced, nil
|
||||
}
|
||||
|
||||
// uploadPartsDir is where chunked uploads are staged: beside a local store's
|
||||
// contexts, so the room check and the budget see one disk and a staged upload
|
||||
// survives an API restart; for an s3:// store, on the uploads volume the
|
||||
// platform mounts either way, or the pod's /tmp when run by hand without it.
|
||||
func uploadPartsDir(contextBase string) string {
|
||||
if isLocalUploadsPath(contextBase) {
|
||||
return filepath.Join(strings.TrimPrefix(contextBase, "file://"), ".parts")
|
||||
}
|
||||
if fi, err := os.Stat(platform.UploadsLocalPath); err == nil && fi.IsDir() {
|
||||
return filepath.Join(platform.UploadsLocalPath, ".parts")
|
||||
}
|
||||
return filepath.Join(os.TempDir(), "felis-upload-parts")
|
||||
}
|
||||
|
||||
// contextMaxBytes resolves [registry] context_max_bytes. 0 keeps the submit
|
||||
// package's own default (1 GiB). The Cloudflare edge refuses a single request
|
||||
// body over 100 MB, which the panel's chunked upload stays under, so the edge
|
||||
// does not lower the cap.
|
||||
func contextMaxBytes(cfg *config.Config) (int64, error) {
|
||||
v := cfg.Registry.ContextMaxBytes
|
||||
if v == "" {
|
||||
return 0, nil
|
||||
}
|
||||
n, err := parseByteSize(v)
|
||||
if err != nil || n <= 0 {
|
||||
return 0, fmt.Errorf("not a positive size: %q", v)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
@@ -4,8 +4,12 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/build"
|
||||
@@ -97,6 +101,50 @@ func TestNewAPIServerSetsHardenedTimeouts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Every listener drains at once, and the shutdown hook (API.CloseStreams in
|
||||
// cmdAPI) runs on each: two listeners each holding a request that ends when
|
||||
// the hook fires take one hook's worth of time, well inside the deadline.
|
||||
func TestShutdownServersDrainsListenersTogether(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
var hooks int32
|
||||
started := make(chan struct{}, 2)
|
||||
var servers []*http.Server
|
||||
for range 2 {
|
||||
srv := newAPIServer("", http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
started <- struct{}{}
|
||||
<-release
|
||||
}))
|
||||
srv.RegisterOnShutdown(func() {
|
||||
if atomic.AddInt32(&hooks, 1) == 1 {
|
||||
time.AfterFunc(100*time.Millisecond, func() { close(release) })
|
||||
}
|
||||
})
|
||||
l, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
srv.Addr = l.Addr().String()
|
||||
go func() { _ = srv.Serve(l) }()
|
||||
go func() {
|
||||
if resp, err := http.Get("http://" + srv.Addr); err == nil {
|
||||
resp.Body.Close()
|
||||
}
|
||||
}()
|
||||
servers = append(servers, srv)
|
||||
}
|
||||
<-started
|
||||
<-started
|
||||
|
||||
begun := time.Now()
|
||||
shutdownServers(servers, 5*time.Second, io.Discard)
|
||||
if took := time.Since(begun); took > 2*time.Second {
|
||||
t.Fatalf("shutdown took %v", took)
|
||||
}
|
||||
if got := atomic.LoadInt32(&hooks); got != 2 {
|
||||
t.Fatalf("shutdown hook ran %d times, want once per listener", got)
|
||||
}
|
||||
}
|
||||
|
||||
type fakeRefStore struct {
|
||||
images []build.Image
|
||||
builds []build.Build
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
)
|
||||
|
||||
// felis-api maps each env var onto the caller the Deployment feeds it for, so the
|
||||
// build Job's token (FELIS_BUILD_TOKEN) authenticates as build and only as build.
|
||||
func TestInternalCallerTokensReadEachCallersEnv(t *testing.T) {
|
||||
env := map[string]string{
|
||||
"FELIS_SERVICE_TOKEN": "v-tok",
|
||||
"FELIS_LIMBO_TOKEN": "l-tok",
|
||||
"FELIS_BUILD_TOKEN": " b-tok\n",
|
||||
"FELIS_OPS_TOKEN": "o-tok",
|
||||
}
|
||||
auth, err := internalCallerTokens(func(k string) string { return env[k] })
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for tok, want := range map[string]api.Caller{"v-tok": api.CallerVelocity, "l-tok": api.CallerLimbo, "b-tok": api.CallerBuild, "o-tok": api.CallerOps} {
|
||||
r := httptest.NewRequest("GET", "/", nil)
|
||||
r.Header.Set("Authorization", "Bearer "+tok)
|
||||
if got, err := auth.Authenticate(r); err != nil || got != want {
|
||||
t.Errorf("%s: got (%q, %v), want %q", tok, got, err, want)
|
||||
}
|
||||
}
|
||||
|
||||
// An install where the build namespace still holds a copy of the proxy's token
|
||||
// would let that copy act as the proxy; the api refuses to start on it.
|
||||
env["FELIS_BUILD_TOKEN"] = "v-tok"
|
||||
if _, err := internalCallerTokens(func(k string) string { return env[k] }); err == nil || !strings.Contains(err.Error(), "same value") {
|
||||
t.Fatalf("shared token: err = %v, want a same-value refusal", err)
|
||||
}
|
||||
}
|
||||
+39
-10
@@ -7,9 +7,11 @@ import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/backup"
|
||||
"felis.lolicon.best/internal/backupjob"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
@@ -36,6 +38,8 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
server := fs.String("server", "", "server name whose world is being backed up")
|
||||
formerOwner := fs.String("former-owner", "", "owner recorded on the backup row (empty for an unowned server)")
|
||||
worldsRoot := fs.String("worlds-root", "/world", "mount path of the world PVC being archived")
|
||||
reason := fs.String("reason", reasonManual, "world_backups reason: manual, or pre_restore for the safety snapshot in front of a restore")
|
||||
protect := fs.String("protect", "", "backup id the prune must keep (the one a chained restore extracts)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -43,6 +47,11 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintln(stderr, "felis backup: --server is required")
|
||||
return 2
|
||||
}
|
||||
keep, ok := map[string]int{reasonManual: -1, backupjob.ReasonPreRestore: preRestoreKeep}[*reason]
|
||||
if !ok {
|
||||
fmt.Fprintf(stderr, "felis backup: unknown --reason %q (manual or %s)\n", *reason, backupjob.ReasonPreRestore)
|
||||
return 2
|
||||
}
|
||||
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
if err != nil {
|
||||
@@ -60,6 +69,9 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis backup: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if keep < 0 {
|
||||
keep = rcfg.ManualKeep
|
||||
}
|
||||
|
||||
// The world PVC is mounted directly at worldsRoot; the resolver returns it for
|
||||
// any target, exactly as in cmdRestore. This is the same TarLocal the reaper
|
||||
@@ -80,11 +92,16 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
ref, size, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
|
||||
a, err := archiver.Archive(ctx, *server, naming.WorldPVCName(*server))
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup: archive: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ref, size := a.Ref, a.Size
|
||||
if len(a.Skipped) > 0 {
|
||||
fmt.Fprintf(stderr, "felis backup: %d entries are not plain files or directories and are not in the archive: %s\n",
|
||||
len(a.Skipped), strings.Join(a.Skipped[:min(len(a.Skipped), 10)], ", "))
|
||||
}
|
||||
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
@@ -99,8 +116,11 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
FormerOwner: *formerOwner,
|
||||
BackupRef: string(ref),
|
||||
SizeBytes: size,
|
||||
Reason: "manual",
|
||||
Reason: *reason,
|
||||
ExpiresAt: time.Now().Add(rcfg.ManualRetention),
|
||||
|
||||
SHA256: a.SHA256,
|
||||
SkippedEntries: len(a.Skipped),
|
||||
}
|
||||
st := reaper.NewPGStore(drv.DB())
|
||||
if err := st.InsertBackup(ctx, rec); err != nil {
|
||||
@@ -116,16 +136,25 @@ func cmdBackup(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
fmt.Fprintf(stdout, "felis backup: server=%s archived %d bytes to %s (backup %s)\n", *server, size, ref, rec.ID)
|
||||
pruneManualBackups(ctx, st, archiver, *server, rcfg.ManualKeep, stdout, stderr)
|
||||
pruneBackups(ctx, st, archiver, *server, *reason, keep, *protect, stdout, stderr)
|
||||
return 0
|
||||
}
|
||||
|
||||
// pruneManualBackups keeps server's newest keep on-demand backups and removes
|
||||
// the rest, oldest first, so repeated backups of one world cannot fill the
|
||||
// shared archive store. The new backup is already recorded; a removal that
|
||||
// fails is reported and retried after the next backup.
|
||||
func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server string, keep int, stdout, stderr io.Writer) {
|
||||
excess, err := st.ExcessManualBackups(ctx, server, keep)
|
||||
const (
|
||||
reasonManual = "manual"
|
||||
// preRestoreKeep is how many safety snapshots a server keeps: enough to walk
|
||||
// back a couple of restores in a row, without every restore adding a world's
|
||||
// worth of bytes for the full manual retention.
|
||||
preRestoreKeep = 3
|
||||
)
|
||||
|
||||
// pruneBackups keeps server's newest keep backups of this reason and removes the
|
||||
// rest, oldest first, so repeated backups of one world cannot fill the shared
|
||||
// archive store. protect is never removed: it is the backup a chained restore is
|
||||
// about to extract. The new backup is already recorded; a removal that fails is
|
||||
// reported and retried after the next backup.
|
||||
func pruneBackups(ctx context.Context, st *reaper.PGStore, archiver backup.WorldArchiver, server, reason string, keep int, protect string, stdout, stderr io.Writer) {
|
||||
excess, err := st.ExcessBackups(ctx, server, reason, keep, protect)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis backup: list older backups of %s: %v\n", server, err)
|
||||
return
|
||||
@@ -139,7 +168,7 @@ func pruneManualBackups(ctx context.Context, st *reaper.PGStore, archiver backup
|
||||
fmt.Fprintf(stderr, "felis backup: record the removal of %s: %v\n", b.ID, err)
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis backup: removed older backup %s of %s (keeping the newest %d)\n", b.ID, server, keep)
|
||||
fmt.Fprintf(stdout, "felis backup: removed older %s backup %s of %s (keeping the newest %d)\n", reason, b.ID, server, keep)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The reason decides which backups the new one's prune may remove, so an
|
||||
// unknown one is refused before anything is archived.
|
||||
func TestBackupSubcommandRejectsUnknownReason(t *testing.T) {
|
||||
var stderr bytes.Buffer
|
||||
if code := cmdBackup([]string{"--server", "survival", "--reason", "inactive_15d"}, &bytes.Buffer{}, &stderr); code != 2 {
|
||||
t.Fatalf("exit = %d, want 2 (%s)", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "unknown --reason") {
|
||||
t.Fatalf("stderr = %q", stderr.String())
|
||||
}
|
||||
}
|
||||
@@ -20,7 +20,7 @@ import (
|
||||
// backupnow is the break-glass "back up a world now" op (§B4 "Sync"). Unlike halt —
|
||||
// which writes the CRD directly — a backup needs felis-api's deployment coordinates
|
||||
// (FELIS_IMAGE / FELIS_BACKUP_PVC) to render the one-shot backup Job, so the console
|
||||
// cannot do it in-process. It POSTs the felis-api INTERNAL face (service-token auth)
|
||||
// cannot do it in-process. It POSTs the felis-api INTERNAL face (ops-token auth)
|
||||
// while the API is alive, and the API renders the Job and audits the action. This file
|
||||
// is the pure core (no bubbletea); tui_backupnow.go is the terminal glue.
|
||||
|
||||
@@ -33,7 +33,7 @@ type backupNowOutcome struct {
|
||||
|
||||
// resolveInternalAPI reads the two things the on-node console needs to reach the
|
||||
// felis-api internal face: the felis-api-internal Service ClusterIP (the host's
|
||||
// resolver is not CoreDNS, so the cluster-DNS name is useless here) and the service
|
||||
// resolver is not CoreDNS, so the cluster-DNS name is useless here) and the ops
|
||||
// token. Both live in the control namespace.
|
||||
func resolveInternalAPI(ctx context.Context, cl client.Client, controlNamespace string) (baseURL, token string, err error) {
|
||||
var svc corev1.Service
|
||||
@@ -45,13 +45,14 @@ func resolveInternalAPI(ctx context.Context, cl client.Client, controlNamespace
|
||||
return "", "", fmt.Errorf("%s Service has no ClusterIP yet", platform.APIInternalServiceName)
|
||||
}
|
||||
|
||||
// The console's own token, which the api serves on the backup route alone.
|
||||
var sec corev1.Secret
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: controlNamespace, Name: naming.ServiceTokenSecretName}, &sec); err != nil {
|
||||
return "", "", fmt.Errorf("get %s Secret: %w", naming.ServiceTokenSecretName, err)
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: controlNamespace, Name: naming.OpsTokenSecretName}, &sec); err != nil {
|
||||
return "", "", fmt.Errorf("get %s Secret (re-run the installer to create it): %w", naming.OpsTokenSecretName, err)
|
||||
}
|
||||
token = string(sec.Data[naming.ServiceTokenSecretKey])
|
||||
if token == "" {
|
||||
return "", "", fmt.Errorf("secret %s has no %s key", naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey)
|
||||
return "", "", fmt.Errorf("secret %s has no %s key", naming.OpsTokenSecretName, naming.ServiceTokenSecretKey)
|
||||
}
|
||||
|
||||
return fmt.Sprintf("http://%s:%d", ip, platform.APIInternalPort), token, nil
|
||||
|
||||
@@ -11,7 +11,6 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
@@ -28,9 +27,15 @@ func internalAPIObjs(clusterIP, token string) []client.Object {
|
||||
ObjectMeta: metav1.ObjectMeta{Name: platform.APIInternalServiceName, Namespace: bgControlNS},
|
||||
Spec: corev1.ServiceSpec{ClusterIP: clusterIP},
|
||||
},
|
||||
// The console presents the ops token; the proxy's felis-service-token sits
|
||||
// beside it and must not be the one picked up.
|
||||
&corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ServiceTokenSecretName, Namespace: bgControlNS},
|
||||
Data: map[string][]byte{naming.ServiceTokenSecretKey: []byte(token)},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-ops-token", Namespace: bgControlNS},
|
||||
Data: map[string][]byte{"token": []byte(token)},
|
||||
},
|
||||
&corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-service-token", Namespace: bgControlNS},
|
||||
Data: map[string][]byte{"token": []byte("proxy-" + token)},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,358 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/imagepin"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/interceptor"
|
||||
)
|
||||
|
||||
// racingClient lands one concurrent write (race) on the stored object just before
|
||||
// setup's first write of it goes out, the way the operator's status update or
|
||||
// felis-api's idle patch can, and counts setup's writes.
|
||||
func racingClient(t *testing.T, race func(ctx context.Context, c client.WithWatch), objs ...client.Object) (client.Client, *int) {
|
||||
t.Helper()
|
||||
writes := 0
|
||||
before := func(ctx context.Context, c client.WithWatch) {
|
||||
writes++
|
||||
if writes == 1 {
|
||||
race(ctx, c)
|
||||
}
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(objs...).
|
||||
WithStatusSubresource(&v1alpha1.MinecraftServer{}).
|
||||
WithInterceptorFuncs(interceptor.Funcs{
|
||||
Update: func(ctx context.Context, c client.WithWatch, obj client.Object, opts ...client.UpdateOption) error {
|
||||
before(ctx, c)
|
||||
return c.Update(ctx, obj, opts...)
|
||||
},
|
||||
Patch: func(ctx context.Context, c client.WithWatch, obj client.Object, patch client.Patch, opts ...client.PatchOption) error {
|
||||
before(ctx, c)
|
||||
return c.Patch(ctx, obj, patch, opts...)
|
||||
},
|
||||
}).Build()
|
||||
return cl, &writes
|
||||
}
|
||||
|
||||
func staleLoginGate(t *testing.T) *v1alpha1.MinecraftServer {
|
||||
t.Helper()
|
||||
ms, err := loginSystemServer("felis-limbo:demo", "minecraft",
|
||||
"http://old.internal:8081", "203.0.113.10.nip.io", "console.203.0.113.10.nip.io")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return ms
|
||||
}
|
||||
|
||||
func getLogin(t *testing.T, cl client.Client) *v1alpha1.MinecraftServer {
|
||||
t.Helper()
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return &ms
|
||||
}
|
||||
|
||||
func envMap(ms *v1alpha1.MinecraftServer) map[string]string {
|
||||
m := map[string]string{}
|
||||
for _, e := range ms.Spec.Env {
|
||||
m[e.Name] = e.Value
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// A hand edit that lands while setup refreshes the console hostnames costs setup a
|
||||
// re-read and a second write; both the edit and the refresh survive.
|
||||
func TestRefreshDerivedEnvRetriesAConcurrentEdit(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
cl, writes := racingClient(t, func(ctx context.Context, c client.WithWatch) {
|
||||
ms := getLogin(t, c)
|
||||
ms.Spec.Env = append(ms.Spec.Env, v1alpha1.EnvVar{Name: "HAND_TUNED", Value: "1"})
|
||||
if err := c.Update(ctx, ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}, staleLoginGate(t))
|
||||
|
||||
desired, err := loginSystemServer("felis-limbo:demo", "minecraft",
|
||||
"http://felis-api-internal.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
refreshed, err := refreshDerivedEnv(ctx, cl, getLogin(t, cl), desired)
|
||||
if err != nil || !refreshed {
|
||||
t.Fatalf("refreshed=%v err=%v, want a refresh after the retry", refreshed, err)
|
||||
}
|
||||
env := envMap(getLogin(t, cl))
|
||||
if env[envPanelHostname] != "console.mc.example.net" || env[envRootDomain] != "mc.example.net" {
|
||||
t.Errorf("env = %v, want the new hostnames", env)
|
||||
}
|
||||
if env["HAND_TUNED"] != "1" {
|
||||
t.Errorf("env = %v: the concurrent hand edit was dropped", env)
|
||||
}
|
||||
if *writes != 2 {
|
||||
t.Errorf("writes = %d, want 2 (one conflict, one retry)", *writes)
|
||||
}
|
||||
}
|
||||
|
||||
// converge reports each fill once even when a status write forced a retry.
|
||||
func TestConvergeRetriesAConcurrentStatusWrite(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
lobby, err := lobbySystemServer("reg/lobby:1", "minecraft")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
lobby.Spec.Rcon = v1alpha1.RconSpec{}
|
||||
cl, writes := racingClient(t, func(ctx context.Context, c client.WithWatch) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLobbyServer}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ms.Status.Phase = v1alpha1.PhaseRunning
|
||||
if err := c.Status().Update(ctx, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}, lobby)
|
||||
|
||||
var got systemServerOutcome
|
||||
for _, o := range convergeSystemServers(ctx, cl, "minecraft", "", "reg/lobby:1",
|
||||
"http://felis-api-internal.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net") {
|
||||
if o.name == naming.SystemLobbyServer {
|
||||
got = o
|
||||
}
|
||||
}
|
||||
if got.err != nil || !got.updated || !slices.Equal(got.changes, []string{"spec.rcon"}) {
|
||||
t.Fatalf("lobby outcome = %+v, want spec.rcon filled once", got)
|
||||
}
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLobbyServer}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !ms.Spec.Rcon.Enabled || ms.Status.Phase != v1alpha1.PhaseRunning {
|
||||
t.Errorf("rcon.enabled=%v phase=%q, want the fill and the status write both kept", ms.Spec.Rcon.Enabled, ms.Status.Phase)
|
||||
}
|
||||
if *writes != 2 {
|
||||
t.Errorf("writes = %d, want 2", *writes)
|
||||
}
|
||||
}
|
||||
|
||||
// A replica refresh racing another writer keeps that writer's key.
|
||||
func TestSecretReplicaRefreshRetriesAConcurrentWrite(t *testing.T) {
|
||||
secret := func(ns, body string) *corev1.Secret {
|
||||
return &corev1.Secret{ObjectMeta: metav1.ObjectMeta{Name: "felis-config", Namespace: ns},
|
||||
Data: map[string][]byte{"felis.toml": []byte(body)}}
|
||||
}
|
||||
cl, writes := racingClient(t, func(ctx context.Context, c client.WithWatch) {
|
||||
var s corev1.Secret
|
||||
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: "felis-config"}, &s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s.Data["extra"] = []byte("x")
|
||||
if err := c.Update(ctx, &s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}, secret("felis", "current"), secret("minecraft", "stale"))
|
||||
|
||||
out := ensureSecretReplica(context.Background(), cl, "felis", "minecraft",
|
||||
"felis-config", "felis.toml", "config", "minecraft ns", true)
|
||||
if out.err != nil || !out.updated {
|
||||
t.Fatalf("outcome = %+v, want refreshed", out)
|
||||
}
|
||||
var s corev1.Secret
|
||||
if err := cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "felis-config"}, &s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if string(s.Data["felis.toml"]) != "current" || string(s.Data["extra"]) != "x" {
|
||||
t.Errorf("replica data = %q, want felis.toml=current and extra=x", s.Data)
|
||||
}
|
||||
if *writes != 2 {
|
||||
t.Errorf("writes = %d, want 2", *writes)
|
||||
}
|
||||
}
|
||||
|
||||
const (
|
||||
sysOldDigest = "sha256:1111111111111111111111111111111111111111111111111111111111111111"
|
||||
sysNewDigest = "sha256:4444444444444444444444444444444444444444444444444444444444444444"
|
||||
)
|
||||
|
||||
func systemPinRegistry(t *testing.T) imagepin.Resolver {
|
||||
t.Helper()
|
||||
reg := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
switch r.URL.Path {
|
||||
case "/v2/felis/limbo/manifests/demo", "/v2/felis/lobby/manifests/demo":
|
||||
w.Header().Set("Docker-Content-Digest", sysNewDigest)
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
t.Cleanup(reg.Close)
|
||||
return imagepin.Resolver{Registry: defaultRegistryURL, Endpoint: strings.TrimPrefix(reg.URL, "http://")}
|
||||
}
|
||||
|
||||
func systemServer(name, image string) *v1alpha1.MinecraftServer {
|
||||
ms := &v1alpha1.MinecraftServer{}
|
||||
ms.Name, ms.Namespace = name, "minecraft"
|
||||
ms.Labels = map[string]string{v1alpha1.LabelSystemRole: name}
|
||||
ms.Spec.Image = image
|
||||
return ms
|
||||
}
|
||||
|
||||
// The installer's --system pin moves a system server onto the build its tag names
|
||||
// now, from the bare tag or from an earlier digest, and is a no-op the second time.
|
||||
func TestPinSystemServerImage(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
res := systemPinRegistry(t)
|
||||
limbo := defaultRegistryURL + "/felis/limbo:demo"
|
||||
lobby := defaultRegistryURL + "/felis/lobby:demo"
|
||||
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(
|
||||
systemServer(naming.SystemLoginServer, limbo),
|
||||
systemServer(naming.SystemLobbyServer, lobby+"@"+sysOldDigest),
|
||||
).Build()
|
||||
image := func(name string) string {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: name}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return ms.Spec.Image
|
||||
}
|
||||
|
||||
for name, want := range map[string]string{
|
||||
naming.SystemLoginServer: limbo + "@" + sysNewDigest,
|
||||
naming.SystemLobbyServer: lobby + "@" + sysNewDigest,
|
||||
} {
|
||||
o, err := pinSystemServerImage(ctx, cl, "minecraft", name, res)
|
||||
if err != nil || o.err != nil || !o.updated {
|
||||
t.Fatalf("%s: outcome=%+v err=%v, want pinned", name, o, err)
|
||||
}
|
||||
if got := image(name); got != want {
|
||||
t.Errorf("%s image = %q, want %q", name, got, want)
|
||||
}
|
||||
again, err := pinSystemServerImage(ctx, cl, "minecraft", name, res)
|
||||
if err != nil || again.updated || again.skipped != "already runs "+want {
|
||||
t.Errorf("%s second pass = %+v, %v; want already runs %s", name, again, err, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPinSystemServerImageLeavesOthersAlone(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
res := systemPinRegistry(t)
|
||||
pin := func(t *testing.T, ms *v1alpha1.MinecraftServer) (systemServerOutcome, string) {
|
||||
t.Helper()
|
||||
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).WithObjects(ms).Build()
|
||||
o, err := pinSystemServerImage(ctx, cl, "minecraft", naming.SystemLoginServer, res)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKeyFromObject(ms), &got); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return o, got.Spec.Image
|
||||
}
|
||||
|
||||
t.Run("external image", func(t *testing.T) {
|
||||
o, img := pin(t, systemServer(naming.SystemLoginServer, "docker.io/example/limbo:1.2"))
|
||||
if o.err != nil || o.updated || img != "docker.io/example/limbo:1.2" ||
|
||||
o.skipped != "runs docker.io/example/limbo:1.2, which names no platform registry tag to follow; left alone" {
|
||||
t.Fatalf("outcome=%+v image=%q, want left alone", o, img)
|
||||
}
|
||||
})
|
||||
t.Run("digest without a tag", func(t *testing.T) {
|
||||
ref := defaultRegistryURL + "/felis/limbo@" + sysOldDigest
|
||||
o, img := pin(t, systemServer(naming.SystemLoginServer, ref))
|
||||
if o.err != nil || o.updated || img != ref {
|
||||
t.Fatalf("outcome=%+v image=%q, want left alone", o, img)
|
||||
}
|
||||
})
|
||||
t.Run("not a system server", func(t *testing.T) {
|
||||
ms := systemServer(naming.SystemLoginServer, defaultRegistryURL+"/felis/limbo:demo")
|
||||
ms.Labels = nil
|
||||
o, img := pin(t, ms)
|
||||
if o.err == nil || img != defaultRegistryURL+"/felis/limbo:demo" {
|
||||
t.Fatalf("outcome=%+v image=%q, want refused", o, img)
|
||||
}
|
||||
})
|
||||
t.Run("tag the registry lost", func(t *testing.T) {
|
||||
o, img := pin(t, systemServer(naming.SystemLoginServer, defaultRegistryURL+"/felis/limbo:gone"))
|
||||
if o.err == nil || img != defaultRegistryURL+"/felis/limbo:gone" {
|
||||
t.Fatalf("outcome=%+v image=%q, want an error the installer falls back on", o, img)
|
||||
}
|
||||
})
|
||||
t.Run("absent", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(newSystemServerScheme(t)).Build()
|
||||
o, err := pinSystemServerImage(ctx, cl, "minecraft", naming.SystemLoginServer, res)
|
||||
if err != nil || o.err != nil || o.skipped != "not present yet; sudo felis setup creates it" {
|
||||
t.Fatalf("outcome=%+v err=%v, want a skip", o, err)
|
||||
}
|
||||
})
|
||||
t.Run("retargeted while pinning", func(t *testing.T) {
|
||||
cl, _ := racingClient(t, func(ctx context.Context, c client.WithWatch) {
|
||||
ms := getLogin(t, c)
|
||||
ms.Spec.Image = "docker.io/example/limbo:1.2"
|
||||
if err := c.Update(ctx, ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}, systemServer(naming.SystemLoginServer, defaultRegistryURL+"/felis/limbo:demo"))
|
||||
o, err := pinSystemServerImage(ctx, cl, "minecraft", naming.SystemLoginServer, res)
|
||||
if err != nil || o.err != nil || o.updated {
|
||||
t.Fatalf("outcome=%+v err=%v, want nothing written", o, err)
|
||||
}
|
||||
if img := getLogin(t, cl).Spec.Image; img != "docker.io/example/limbo:1.2" {
|
||||
t.Errorf("image = %q, want the admin's retarget kept", img)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// --system names one of the two system servers; anything else is a usage error
|
||||
// before any cluster is touched.
|
||||
func TestPinImagesSystemFlagTakesOnlySystemServers(t *testing.T) {
|
||||
var stdout, stderr strings.Builder
|
||||
if code := cmdPinImages([]string{"--system", "survival"}, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("exit = %d, want 2", code)
|
||||
}
|
||||
if got := stderr.String(); got != "felis pin-images: --system takes login or lobby, not \"survival\"\n" {
|
||||
t.Errorf("stderr = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// An idle setting the panel saves while converge fills the default is kept.
|
||||
func TestConvergeIdleKeepsAConcurrentPanelEdit(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
srv := &v1alpha1.MinecraftServer{}
|
||||
srv.Name, srv.Namespace = "survival", "minecraft"
|
||||
cl, writes := racingClient(t, func(ctx context.Context, c client.WithWatch) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := c.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: "survival"}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ms.Spec.Idle = v1alpha1.IdleSpec{AutoStopEnabled: false, EmptySecondsBeforeStop: 1800}
|
||||
if err := c.Update(ctx, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}, srv)
|
||||
|
||||
if out := convergeUserServerIdle(ctx, cl, "minecraft"); len(out) != 0 {
|
||||
t.Fatalf("outcomes = %+v, want none: the server has a setting by the time converge writes", out)
|
||||
}
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: "survival"}, &ms); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if want := (v1alpha1.IdleSpec{AutoStopEnabled: false, EmptySecondsBeforeStop: 1800}); ms.Spec.Idle != want {
|
||||
t.Errorf("idle = %+v, want the panel's %+v", ms.Spec.Idle, want)
|
||||
}
|
||||
if *writes != 1 {
|
||||
t.Errorf("writes = %d, want 1 (the conflicted attempt; the retry sends nothing)", *writes)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
)
|
||||
|
||||
func TestContextMaxBytes(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
reg config.RegistryConfig
|
||||
auth config.AuthConfig
|
||||
want int64
|
||||
wantErr bool
|
||||
}{
|
||||
{name: "direct install keeps the package default", want: 0},
|
||||
// The panel uploads in parts under the edge's 100 MB body limit, so the
|
||||
// Cloudflare edge keeps the full default.
|
||||
{name: "the Cloudflare edge keeps the package default", auth: config.AuthConfig{AccessJWTAud: "aud-1"}, want: 0},
|
||||
{name: "CF-Connecting-IP keeps the package default", auth: config.AuthConfig{ClientIPHeader: "cf-connecting-ip"}, want: 0},
|
||||
{name: "explicit value behind the edge", reg: config.RegistryConfig{ContextMaxBytes: "50Mi"}, auth: config.AuthConfig{AccessJWTAud: "aud-1"}, want: 52428800},
|
||||
{name: "explicit value on a direct install", reg: config.RegistryConfig{ContextMaxBytes: "2Gi"}, want: 2147483648},
|
||||
{name: "garbage is refused", reg: config.RegistryConfig{ContextMaxBytes: "lots"}, wantErr: true},
|
||||
{name: "zero is refused", reg: config.RegistryConfig{ContextMaxBytes: "0"}, wantErr: true},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got, err := contextMaxBytes(&config.Config{Registry: tc.reg, Auth: tc.auth})
|
||||
if (err != nil) != tc.wantErr {
|
||||
t.Fatalf("err = %v, wantErr %v", err, tc.wantErr)
|
||||
}
|
||||
if got != tc.want {
|
||||
t.Fatalf("contextMaxBytes = %d, want %d", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestUploadPartsDir(t *testing.T) {
|
||||
if got := uploadPartsDir("/var/lib/felis/uploads"); got != "/var/lib/felis/uploads/.parts" {
|
||||
t.Errorf("local store: parts dir = %q, want beside the contexts", got)
|
||||
}
|
||||
if got := uploadPartsDir("file:///srv/uploads"); got != "/srv/uploads/.parts" {
|
||||
t.Errorf("file:// store: parts dir = %q, want /srv/uploads/.parts", got)
|
||||
}
|
||||
// This machine has no /var/lib/felis/uploads mount, so an s3:// store falls
|
||||
// back to the temp dir.
|
||||
if _, err := os.Stat("/var/lib/felis/uploads"); err == nil {
|
||||
t.Skip("/var/lib/felis/uploads exists here")
|
||||
}
|
||||
if got := uploadPartsDir("s3://bucket/uploads"); got != filepath.Join(os.TempDir(), "felis-upload-parts") {
|
||||
t.Errorf("s3 store without the uploads mount: parts dir = %q", got)
|
||||
}
|
||||
}
|
||||
+13
-3
@@ -92,12 +92,22 @@ func convergeUserServerIdle(ctx context.Context, cl client.Client, namespace str
|
||||
if ms.Labels[v1alpha1.LabelSystemRole] != "" || ms.Spec.Idle != (v1alpha1.IdleSpec{}) {
|
||||
continue
|
||||
}
|
||||
patch := client.MergeFrom(ms.DeepCopy())
|
||||
ms.Spec.Idle = v1alpha1.DefaultIdle()
|
||||
if err := cl.Patch(ctx, ms, patch); err != nil {
|
||||
// Re-checked on the copy each attempt reads: an idle setting the panel saved
|
||||
// meanwhile is the user's, and the default must not land over it.
|
||||
changed, err := patchOnConflictRetry(ctx, cl, ms, func() bool {
|
||||
if ms.Spec.Idle != (v1alpha1.IdleSpec{}) {
|
||||
return false
|
||||
}
|
||||
ms.Spec.Idle = v1alpha1.DefaultIdle()
|
||||
return true
|
||||
})
|
||||
if err != nil {
|
||||
out = append(out, systemServerOutcome{name: ms.Name, err: fmt.Errorf("converge %s: %w", ms.Name, err)})
|
||||
continue
|
||||
}
|
||||
if !changed {
|
||||
continue
|
||||
}
|
||||
out = append(out, systemServerOutcome{name: ms.Name, available: true, updated: true,
|
||||
changes: []string{fmt.Sprintf("spec.idle (stop after %ds empty)", v1alpha1.DefaultEmptySecondsBeforeStop)}})
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ import (
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/retention"
|
||||
)
|
||||
|
||||
const dbUsage = `usage:
|
||||
@@ -24,6 +25,7 @@ const dbUsage = `usage:
|
||||
felis db verify [-dir dir] <bundle>
|
||||
felis db list [-dir dir]
|
||||
felis db check [-dir dir] [-max-age 26h]
|
||||
felis db audit-export [-config path] [-since date] [-until date] [-out file]
|
||||
`
|
||||
|
||||
// defaultKeep is how many bundles of a label a backup leaves behind. Manual
|
||||
@@ -58,6 +60,8 @@ func cmdDB(args []string, stdout, stderr io.Writer) int {
|
||||
return dbList(fs, dir, rest, stdout, stderr)
|
||||
case "check":
|
||||
return dbCheck(fs, dir, rest, stdout, stderr)
|
||||
case "audit-export":
|
||||
return dbAuditExport(fs, rest, stdout, stderr)
|
||||
case "-h", "--help", "help":
|
||||
fmt.Fprint(stdout, dbUsage)
|
||||
return 0
|
||||
@@ -262,6 +266,88 @@ func dbList(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writ
|
||||
return 0
|
||||
}
|
||||
|
||||
// dbAuditExport writes audit rows to a file (or stdout) as JSON lines, so an
|
||||
// install can keep them past [audit] retention, after which felis-api deletes them.
|
||||
func dbAuditExport(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||
sinceFlag := fs.String("since", "", "first day (or RFC 3339 instant) to export, inclusive; empty starts at the oldest row")
|
||||
untilFlag := fs.String("until", "", "day (or RFC 3339 instant) to stop before, exclusive; empty runs to the newest row")
|
||||
out := fs.String("out", "", "file to write (created 0600, never overwritten); empty writes to stdout")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
if fs.NArg() > 0 {
|
||||
fmt.Fprint(stderr, dbUsage)
|
||||
return 2
|
||||
}
|
||||
since, err := parseExportBound(*sinceFlag)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db audit-export: -since: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
until, err := parseExportBound(*untilFlag)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db audit-export: -until: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
if !since.IsZero() && !until.IsZero() && !until.After(since) {
|
||||
fmt.Fprintf(stderr, "felis db audit-export: -until %s is not after -since %s\n", *untilFlag, *sinceFlag)
|
||||
return 2
|
||||
}
|
||||
url, err := dbDatabaseURL(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db audit-export: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
w := stdout
|
||||
var f *os.File
|
||||
if *out != "" {
|
||||
if f, err = os.OpenFile(*out, os.O_WRONLY|os.O_CREATE|os.O_EXCL, 0o600); err != nil {
|
||||
fmt.Fprintf(stderr, "felis db audit-export: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
w = f
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
|
||||
defer cancel()
|
||||
n, err := exportAudit(ctx, url, since, until, w)
|
||||
if f != nil {
|
||||
if cerr := f.Close(); err == nil {
|
||||
err = cerr
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis db audit-export: %v (%d rows written)\n", err, n)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis db audit-export: %d audit rows written\n", n)
|
||||
return 0
|
||||
}
|
||||
|
||||
func exportAudit(ctx context.Context, url string, since, until time.Time, w io.Writer) (int, error) {
|
||||
drv, err := openStore(ctx, url, false)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("open database: %w", err)
|
||||
}
|
||||
defer drv.Close()
|
||||
return retention.ExportAudit(ctx, drv.DB(), since, until, w)
|
||||
}
|
||||
|
||||
// parseExportBound reads a -since/-until value: a day (midnight UTC) or an
|
||||
// RFC 3339 instant; empty is an open bound.
|
||||
func parseExportBound(v string) (time.Time, error) {
|
||||
if v == "" {
|
||||
return time.Time{}, nil
|
||||
}
|
||||
if t, err := time.Parse(time.DateOnly, v); err == nil {
|
||||
return t, nil
|
||||
}
|
||||
if t, err := time.Parse(time.RFC3339, v); err == nil {
|
||||
return t.UTC(), nil
|
||||
}
|
||||
return time.Time{}, fmt.Errorf("%q is neither a day (2026-01-31) nor an RFC 3339 instant (2026-01-31T12:00:00Z)", v)
|
||||
}
|
||||
|
||||
func humanBytes(n int64) string {
|
||||
const unit = 1024
|
||||
if n < unit {
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"io"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/store"
|
||||
)
|
||||
@@ -149,3 +150,36 @@ func TestPreMigrateBackupOnlyGuardsAPopulatedDatabase(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// audit-export takes a day or an RFC 3339 instant for each bound, and refuses a
|
||||
// malformed or inverted window before it opens the config or the database.
|
||||
func TestAuditExportBounds(t *testing.T) {
|
||||
for _, tc := range []struct{ in, want string }{
|
||||
{"", "0001-01-01T00:00:00Z"},
|
||||
{"2026-01-31", "2026-01-31T00:00:00Z"},
|
||||
{"2026-01-31T12:30:00+08:00", "2026-01-31T04:30:00Z"},
|
||||
} {
|
||||
got, err := parseExportBound(tc.in)
|
||||
if err != nil || got.Format(time.RFC3339) != tc.want {
|
||||
t.Errorf("parseExportBound(%q) = %v, %v; want %s", tc.in, got, err, tc.want)
|
||||
}
|
||||
}
|
||||
if _, err := parseExportBound("31/01/2026"); err == nil || err.Error() != `"31/01/2026" is neither a day (2026-01-31) nor an RFC 3339 instant (2026-01-31T12:00:00Z)` {
|
||||
t.Errorf("parseExportBound(31/01/2026) err = %v", err)
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
args []string
|
||||
wantErr string
|
||||
}{
|
||||
{[]string{"db", "audit-export", "-since", "yesterday"}, `felis db audit-export: -since: "yesterday" is neither`},
|
||||
{[]string{"db", "audit-export", "-until", "2026-13-01"}, `felis db audit-export: -until: "2026-13-01" is neither`},
|
||||
{[]string{"db", "audit-export", "-since", "2026-02-01", "-until", "2026-02-01"}, "felis db audit-export: -until 2026-02-01 is not after -since 2026-02-01"},
|
||||
{[]string{"db", "audit-export", "extra"}, "felis db audit-export [-config path]"},
|
||||
} {
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run(append(tc.args, "-config", "/nonexistent/felis.toml"), &out, &errBuf)
|
||||
if code != 2 || !strings.Contains(errBuf.String(), tc.wantErr) {
|
||||
t.Errorf("%v: exit %d, stderr %q; want 2 and %q", tc.args, code, errBuf.String(), tc.wantErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
+3
-2
@@ -19,7 +19,7 @@ import (
|
||||
// Like cmdRestore it deliberately holds NO database credentials and never calls
|
||||
// config.Load: felis-api made the authorization decision (the caller owns this
|
||||
// server, and the server is stopped so the RWO world volume is free); this process
|
||||
// is the unprivileged hands that touch bytes. Its entire input is the three flags
|
||||
// is the unprivileged hands that touch bytes. Its entire input is the flags
|
||||
// below plus, for a write, one environment variable. Every isolation guarantee
|
||||
// lives in the Pod spec (internal/fileedit/jobspec.go), and the path-containment
|
||||
// guarantee lives in fileedit.Execute, which resolves the path through os.Root and
|
||||
@@ -37,6 +37,7 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
|
||||
op := fs.String("op", "", "operation: list, read, or write")
|
||||
path := fs.String("path", "", "path to operate on, relative to the world root (empty = the root itself)")
|
||||
worldsRoot := fs.String("worlds-root", "/data", "mount path of the world PVC; every path resolves under it")
|
||||
expect := fs.String("expect-sha256", "", "write only: refuse unless the file's current SHA-256 (hex) is this")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -67,7 +68,7 @@ func cmdFiles(args []string, stdout, stderr io.Writer) int {
|
||||
content = decoded
|
||||
}
|
||||
|
||||
res, err := fileedit.Execute(*worldsRoot, *op, *path, content)
|
||||
res, err := fileedit.Execute(*worldsRoot, *op, *path, content, *expect)
|
||||
if err != nil {
|
||||
// The operation could not be attempted — infrastructure, not caller fault.
|
||||
fmt.Fprintf(stderr, "felis files: %v\n", err)
|
||||
|
||||
+17
-3
@@ -48,7 +48,7 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry 2.8.3, pinned by digest)")
|
||||
backupPVC := fs.String("backup-pvc", "felis-backups", "name of the world-archive PVC this bundle renders in the Minecraft namespace and advertises to the backup/restore executors via FELIS_BACKUP_PVC (default: felis-backups; pass an empty value to render none, leaving backup/restore answering 503)")
|
||||
worldsHostPath := fs.String("worlds-host-path", "", "node directory the reaper reads worlds from: each world PVC resolves as <path>/<pvc>, or as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (k3s storage root: /var/lib/rancher/k3s/storage); enables the reaper CronJob (requires --archive-local-path and a non-empty --backup-pvc)")
|
||||
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path")
|
||||
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path. With the backup PVC alone it renders the retention-only CronJob, which deletes backups past their expiry and never touches a world")
|
||||
registryStorage := fs.String("registry-storage", "", "capacity the registry PVC requests (default 10Gi; k3s local-path does not enforce it)")
|
||||
uploadsStorage := fs.String("uploads-storage", "", "capacity the uploads PVC requests (default 5Gi; k3s local-path does not enforce it)")
|
||||
backupStorage := fs.String("backup-storage", "", "capacity the world-archive PVC requests (default 10Gi; k3s local-path does not enforce it)")
|
||||
@@ -138,8 +138,22 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
"<path>/<pvc>, or each candidate's archive fails and the world is preserved;\n"+
|
||||
" - %s.\n", *worldsHostPath, *worldsHostPath, *worldsHostPath, pin)
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis manifests: note: retention reaper CronJob not rendered "+
|
||||
"(pass --worlds-host-path and --archive-local-path — the archive PVC defaults to felis-backups — to enable it)")
|
||||
switch {
|
||||
case *archiveLocalPath != "" && *backupPVC == "":
|
||||
fmt.Fprintln(stderr, "felis manifests: --archive-local-path names where the backup PVC is mounted, "+
|
||||
"but --backup-pvc is empty (no archive store renders); drop one or the other")
|
||||
return 2
|
||||
case *archiveLocalPath != "":
|
||||
fmt.Fprintln(stderr, "felis manifests: note: rendering the reaper CronJob retention-only: backups past "+
|
||||
"their expiry are deleted daily, idle worlds are never archived or deleted "+
|
||||
"(pass --worlds-host-path to reap them too)")
|
||||
case *backupPVC != "":
|
||||
fmt.Fprintln(stderr, "felis manifests: note: reaper CronJob not rendered: backups are never expired, so "+
|
||||
"the archive store only grows until the disk fills (pass --archive-local-path, equal to felis.toml "+
|
||||
"[archive] local_path, for the retention-only CronJob, and --worlds-host-path as well to reap idle worlds)")
|
||||
default:
|
||||
fmt.Fprintln(stderr, "felis manifests: note: reaper CronJob not rendered (backups are disabled)")
|
||||
}
|
||||
}
|
||||
|
||||
params := platform.Params{
|
||||
|
||||
@@ -90,12 +90,13 @@ func TestManifestsRendersBundle(t *testing.T) {
|
||||
t.Error("rendered bundle must not contain ClusterRole/ClusterRoleBinding")
|
||||
}
|
||||
// Without the retention flags, the reaper CronJob is not rendered and the
|
||||
// generator says so on stderr.
|
||||
// generator says so on stderr, naming what that leaves: backups that never
|
||||
// expire.
|
||||
if strings.Contains(text, "kind: CronJob") {
|
||||
t.Error("no reaper CronJob must render without --worlds-host-path")
|
||||
t.Error("no reaper CronJob must render without --archive-local-path")
|
||||
}
|
||||
if !strings.Contains(errBuf.String(), "not rendered") {
|
||||
t.Errorf("expected a 'reaper not rendered' notice on stderr, got %q", errBuf.String())
|
||||
if !strings.Contains(errBuf.String(), "not rendered") || !strings.Contains(errBuf.String(), "never expired") {
|
||||
t.Errorf("expected a 'reaper not rendered, backups never expired' notice on stderr, got %q", errBuf.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -144,6 +145,40 @@ func TestManifestsBackupPVCOptOut(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsRendersRetentionOnly: the archive store without a worlds root
|
||||
// still gets the daily CronJob, retention-only, so backups past their expiry
|
||||
// leave the store on an install that never reaps a world; the operator is told
|
||||
// which of the two it got. An archive path with the store switched off is a
|
||||
// mistake and fails loud.
|
||||
func TestManifestsRendersRetentionOnly(t *testing.T) {
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run([]string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "10.0.0.5/32",
|
||||
"--archive-local-path", "/var/lib/felis/archives"}, &out, &errBuf)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String())
|
||||
}
|
||||
text := out.String()
|
||||
for _, want := range []string{"kind: CronJob", "name: felis-reaper", "--retention-only", "claimName: felis-backups"} {
|
||||
if !strings.Contains(text, want) {
|
||||
t.Errorf("retention-only bundle missing %q", want)
|
||||
}
|
||||
}
|
||||
if strings.Contains(text, "kind: PersistentVolume\n") || strings.Contains(text, "--worlds-root") {
|
||||
t.Error("a retention-only bundle must not reach for a worlds root")
|
||||
}
|
||||
if !strings.Contains(errBuf.String(), "retention-only") {
|
||||
t.Errorf("stderr must say the CronJob is retention-only, got %q", errBuf.String())
|
||||
}
|
||||
|
||||
out.Reset()
|
||||
errBuf.Reset()
|
||||
code = run([]string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "10.0.0.5/32",
|
||||
"--archive-local-path", "/var/lib/felis/archives", "--backup-pvc="}, &out, &errBuf)
|
||||
if code != 2 || out.Len() != 0 || !strings.Contains(errBuf.String(), "--backup-pvc is empty") {
|
||||
t.Errorf("archive path without an archive store: exit=%d out=%d bytes stderr=%q, want a fail-loud 2", code, out.Len(), errBuf.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsRendersReaper proves the happy path with the full retention trio:
|
||||
// a batch/v1 CronJob is emitted, named felis-reaper, mounting the backup PVC at the
|
||||
// supplied archive path.
|
||||
|
||||
@@ -111,3 +111,26 @@ func hasPending(done map[int]struct{}, migrations []store.Migration) bool {
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// openStore opens the business database for a command that reads and writes its
|
||||
// tables, and refuses one whose schema this build was not written against: a newer
|
||||
// Felis migrated it (a rolled-back binary), or, unless allowPending, migrations this
|
||||
// build embeds have not run yet (a binary swapped in ahead of `felis migrate up`).
|
||||
func openStore(ctx context.Context, url string, allowPending bool) (*store.PostgresDriver, error) {
|
||||
drv, err := store.Open(ctx, url)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
s, err := store.ReadSchema(ctx, drv)
|
||||
if err == nil {
|
||||
err = s.Err()
|
||||
if allowPending {
|
||||
err = s.Newer()
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
drv.Close()
|
||||
return nil, err
|
||||
}
|
||||
return drv, nil
|
||||
}
|
||||
@@ -14,7 +14,6 @@ import (
|
||||
|
||||
"felis.lolicon.best/internal/build"
|
||||
"felis.lolicon.best/internal/imagepush"
|
||||
"felis.lolicon.best/internal/registrygate"
|
||||
)
|
||||
|
||||
// defaultBuildToolsStatus is where mirror-build-tools records its last run; the
|
||||
@@ -49,16 +48,9 @@ func cmdMirrorBuildTools(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis mirror-build-tools: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
if err := loadEnvFile(*secrets); err != nil {
|
||||
fmt.Fprintf(stderr, "felis mirror-build-tools: read %s: %v\n", *secrets, err)
|
||||
return 1
|
||||
}
|
||||
user, pass := os.Getenv("FELIS_REGISTRY_USERNAME"), os.Getenv("FELIS_REGISTRY_PASSWORD")
|
||||
if pass == "" {
|
||||
user, pass = registrygate.PrincipalPlatform, os.Getenv("REGISTRY_PLATFORM_TOKEN")
|
||||
}
|
||||
if pass == "" {
|
||||
fmt.Fprintln(stderr, "felis mirror-build-tools: no registry credential: set FELIS_REGISTRY_PASSWORD or run as root on the node (REGISTRY_PLATFORM_TOKEN in /etc/felis/secrets.env)")
|
||||
user, pass, err := registryWriteCredential(*secrets)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis mirror-build-tools: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
|
||||
|
||||
+130
-33
@@ -16,7 +16,6 @@ import (
|
||||
"felis.lolicon.best/internal/dbbackup"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
appsv1 "k8s.io/api/apps/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
@@ -25,12 +24,15 @@ import (
|
||||
)
|
||||
|
||||
const offsiteUsage = `usage:
|
||||
felis offsite sync [-config path] [-archive-dir dir] [-db-dir dir] [-status-file path]
|
||||
felis offsite sync [-config path] [-archive-dir dir] [-db-dir dir] [-registry host:port|off]
|
||||
[-uploads-dir dir] [-status-file path]
|
||||
felis offsite status [-config path] [-status-file path]
|
||||
felis offsite list [-config path]
|
||||
felis offsite fetch-db [-config path | -endpoint url -bucket name [-region r] [-prefix p]]
|
||||
[-dir dir] latest|<bundle>
|
||||
felis offsite fetch-worlds [-config path] [-archive-dir dir]
|
||||
felis offsite fetch-images [-config path] [-registry host:port] [-at version]
|
||||
felis offsite fetch-uploads [-config path] [-uploads-dir dir] [-at version]
|
||||
felis offsite keygen
|
||||
|
||||
Every verb but keygen reads the bucket credentials and the encryption key from
|
||||
@@ -44,9 +46,10 @@ FELIS_OFFSITE_SECRET_KEY, FELIS_OFFSITE_KEY), taking any that are unset from
|
||||
const defaultOffsiteEnvFile = "/etc/felis/offsite.env"
|
||||
|
||||
// cmdOffsite implements `felis offsite`: the off-site copy of the world
|
||||
// archives and the database bundles (internal/offsite). felis-offsite.timer
|
||||
// runs `sync` hourly on the host; the fetch verbs are the way back after the
|
||||
// node is lost (docs/troubleshooting.md §16).
|
||||
// archives, the database bundles, the registry's user images and the
|
||||
// submission uploads (internal/offsite). felis-offsite.timer runs `sync`
|
||||
// hourly on the host; the fetch verbs are the way back after the node is lost
|
||||
// (docs/troubleshooting.md §16).
|
||||
func cmdOffsite(args []string, stdout, stderr io.Writer) int {
|
||||
if len(args) == 0 {
|
||||
fmt.Fprint(stderr, offsiteUsage)
|
||||
@@ -67,6 +70,10 @@ func cmdOffsite(args []string, stdout, stderr io.Writer) int {
|
||||
return offsiteFetchDB(fs, rest, stdout, stderr)
|
||||
case "fetch-worlds":
|
||||
return offsiteFetchWorlds(fs, rest, stdout, stderr)
|
||||
case "fetch-images":
|
||||
return offsiteFetchImages(fs, rest, stdout, stderr)
|
||||
case "fetch-uploads":
|
||||
return offsiteFetchUploads(fs, rest, stdout, stderr)
|
||||
case "keygen":
|
||||
k, err := offsite.NewKey()
|
||||
if err != nil {
|
||||
@@ -187,6 +194,9 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
|
||||
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
|
||||
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
|
||||
registry := fs.String("registry", "", `host[:port] of the registry whose user images are copied (default: the in-cluster registry's loopback hostPort; "off" copies none)`)
|
||||
uploadsDir := fs.String("uploads-dir", "", "host directory of the submission uploads volume (default: resolved from the uploads PVC through the cluster)")
|
||||
uploadsPVC := fs.String("uploads-pvc", platform.UploadsPVCName, `the submission uploads PVC, in the control-plane namespace ("" copies no uploads)`)
|
||||
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "where the result of this run is recorded for the watchdog and `status`")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
@@ -203,7 +213,11 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
if prev, _ := offsite.ReadStatus(*statusFile); prev != nil {
|
||||
st.LastSuccess = prev.LastSuccess
|
||||
}
|
||||
res, err := runOffsiteSync(cfg, env, *archiveDir, *backupPVC, *dbDir, stderr)
|
||||
res, err := runOffsiteSync(cfg, env, offsiteSources{
|
||||
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir,
|
||||
registry: offsiteRegistryEndpoint(*registry, cfg.Registry),
|
||||
uploadsDir: *uploadsDir, uploadsPVC: *uploadsPVC,
|
||||
}, stderr)
|
||||
st.Result = res
|
||||
if err != nil {
|
||||
st.LastError = err.Error()
|
||||
@@ -213,12 +227,18 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
if werr := offsite.WriteStatus(*statusFile, st); werr != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite sync: record status: %v\n", werr)
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; bucket holds %d worlds (%s) and %d bundles\n",
|
||||
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; images copied=%d blobs=%d pruned=%d; uploads copied=%d pruned=%d; bucket holds %d worlds (%s), %d bundles, %d images in %d repositories (%s), %d uploads (%s)\n",
|
||||
res.WorldsUploaded, res.WorldsPending, len(res.WorldsMissing), res.WorldsExpired,
|
||||
res.DBUploaded, res.DBPruned, res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB)
|
||||
res.DBUploaded, res.DBPruned, res.ImagesUploaded, res.ImageBlobsUploaded, res.ImageObjectsPruned,
|
||||
res.UploadsUploaded, res.UploadObjectsPruned,
|
||||
res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB, res.Images, res.ImageRepos, offsite.HumanBytes(res.RemoteImageBytes),
|
||||
res.Uploads, offsite.HumanBytes(res.RemoteUploadBytes))
|
||||
for _, m := range res.WorldsMissing {
|
||||
fmt.Fprintf(stderr, "felis offsite sync: recorded archive not on the volume, nothing to copy: %s\n", m)
|
||||
}
|
||||
for _, m := range res.ImagesIncomplete {
|
||||
fmt.Fprintf(stderr, "felis offsite sync: registry image not whole: %s\n", m)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
|
||||
return 1
|
||||
@@ -226,7 +246,18 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
||||
return 0
|
||||
}
|
||||
|
||||
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, archiveDir, backupPVC, dbDir string, log io.Writer) (offsite.Result, error) {
|
||||
// offsiteSources is where one sync pass reads from: the world archive volume
|
||||
// (archiveDir, or the backupPVC's directory), the bundle directory, the
|
||||
// registry's loopback endpoint and the uploads volume (uploadsDir, or the
|
||||
// uploadsPVC's directory). An empty source is skipped.
|
||||
type offsiteSources struct {
|
||||
archiveDir, backupPVC string
|
||||
dbDir string
|
||||
registry string
|
||||
uploadsDir, uploadsPVC string
|
||||
}
|
||||
|
||||
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, log io.Writer) (offsite.Result, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 50*time.Minute)
|
||||
defer cancel()
|
||||
checkCtx, checkCancel := context.WithTimeout(ctx, 30*time.Second)
|
||||
@@ -235,47 +266,70 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, archiveDir, backupPVC,
|
||||
if err != nil {
|
||||
return offsite.Result{}, err
|
||||
}
|
||||
if archiveDir == "" && backupPVC != "" {
|
||||
dir, err := resolveArchiveDir(ctx, cfg.K8s.Namespace, backupPVC, false, log)
|
||||
archiveDir, uploadsDir := src.archiveDir, src.uploadsDir
|
||||
if archiveDir == "" && src.backupPVC != "" {
|
||||
dir, err := resolveVolumeDir(ctx, cfg.K8s.Namespace, src.backupPVC, archiveVolume, false, log)
|
||||
if err != nil {
|
||||
return offsite.Result{}, err
|
||||
}
|
||||
archiveDir = dir
|
||||
}
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
// An s3:// uploads store is off the host already; only a local one, on
|
||||
// the uploads PVC, needs the copy.
|
||||
if uploadsDir == "" && src.uploadsPVC != "" && isLocalUploadsPath(cfg.Registry.UserUploadsContext) {
|
||||
dir, err := resolveVolumeDir(ctx, platform.DefaultControlNamespace, src.uploadsPVC, uploadsVolume, false, log)
|
||||
if err != nil {
|
||||
return offsite.Result{}, err
|
||||
}
|
||||
uploadsDir = dir
|
||||
}
|
||||
drv, err := openStore(ctx, cfg.Database.URL, false)
|
||||
if err != nil {
|
||||
return offsite.Result{}, fmt.Errorf("open database: %w", err)
|
||||
}
|
||||
defer drv.Close()
|
||||
s := &offsite.Syncer{
|
||||
Bucket: env.bucket, Catalog: offsite.PGCatalog{DB: drv.DB()}, Key: env.key,
|
||||
ArchiveDir: archiveDir, DBDir: dbDir, DBKeep: env.cfg.DBKeep, Log: log,
|
||||
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Log: log,
|
||||
}
|
||||
if src.registry != "" {
|
||||
s.Images = newRegistryImages(src.registry)
|
||||
s.ImagePins = imagePins(drv.DB(), cfg.Registry.URL)
|
||||
}
|
||||
return s.Run(ctx)
|
||||
}
|
||||
|
||||
// resolveArchiveDir finds the host directory behind the world archive PVC: a
|
||||
// local-path volume is a directory on this node. A PVC still waiting for its
|
||||
// first consumer holds nothing yet: without bind that is "" (no archives),
|
||||
// with bind it is bound first, for fetch-worlds to write into.
|
||||
func resolveArchiveDir(ctx context.Context, ns, pvcName string, bind bool, log io.Writer) (string, error) {
|
||||
// volumeKind names a PVC the off-site copy reads or restores, for messages,
|
||||
// with the flag that bypasses finding it through the cluster.
|
||||
type volumeKind struct{ what, dirFlag, empty string }
|
||||
|
||||
var (
|
||||
archiveVolume = volumeKind{"archive volume", "-archive-dir", "no world has been archived"}
|
||||
uploadsVolume = volumeKind{"uploads volume", "-uploads-dir", "no modpack has been uploaded"}
|
||||
)
|
||||
|
||||
// resolveVolumeDir finds the host directory behind a PVC: a local-path volume
|
||||
// is a directory on this node. A PVC still waiting for its first consumer
|
||||
// holds nothing yet: without bind that is "" (nothing to copy), with bind it
|
||||
// is bound first, for a fetch to write into.
|
||||
func resolveVolumeDir(ctx context.Context, ns, pvcName string, kind volumeKind, bind bool, log io.Writer) (string, error) {
|
||||
if ns == "" {
|
||||
ns = platform.DefaultMinecraftNamespace
|
||||
}
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("reach the cluster to find the archive volume (or pass -archive-dir): %w", err)
|
||||
return "", fmt.Errorf("reach the cluster to find the %s (or pass %s): %w", kind.what, kind.dirFlag, err)
|
||||
}
|
||||
var pvc corev1.PersistentVolumeClaim
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err != nil {
|
||||
return "", fmt.Errorf("archive volume %s/%s: %w", ns, pvcName, err)
|
||||
return "", fmt.Errorf("%s %s/%s: %w", kind.what, ns, pvcName, err)
|
||||
}
|
||||
if pvc.Spec.VolumeName == "" {
|
||||
if !bind {
|
||||
fmt.Fprintf(log, "felis offsite: archive volume %s/%s is not bound yet; no world has been archived\n", ns, pvcName)
|
||||
fmt.Fprintf(log, "felis offsite: %s %s/%s is not bound yet; %s\n", kind.what, ns, pvcName, kind.empty)
|
||||
return "", nil
|
||||
}
|
||||
if err := bindVolume(ctx, cl, ns, pvcName, log); err != nil {
|
||||
if err := bindVolume(ctx, cl, ns, pvcName, kind, log); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err != nil {
|
||||
@@ -284,7 +338,7 @@ func resolveArchiveDir(ctx context.Context, ns, pvcName string, bind bool, log i
|
||||
}
|
||||
var pv corev1.PersistentVolume
|
||||
if err := cl.Get(ctx, types.NamespacedName{Name: pvc.Spec.VolumeName}, &pv); err != nil {
|
||||
return "", fmt.Errorf("archive volume %s: %w", pvc.Spec.VolumeName, err)
|
||||
return "", fmt.Errorf("%s %s: %w", kind.what, pvc.Spec.VolumeName, err)
|
||||
}
|
||||
var dir string
|
||||
switch {
|
||||
@@ -293,10 +347,10 @@ func resolveArchiveDir(ctx context.Context, ns, pvcName string, bind bool, log i
|
||||
case pv.Spec.HostPath != nil:
|
||||
dir = pv.Spec.HostPath.Path
|
||||
default:
|
||||
return "", fmt.Errorf("archive volume %s is not a directory on a node (local or hostPath); pass -archive-dir with where it is mounted on this host", pv.Name)
|
||||
return "", fmt.Errorf("%s %s is not a directory on a node (local or hostPath); pass %s with where it is mounted on this host", kind.what, pv.Name, kind.dirFlag)
|
||||
}
|
||||
if fi, err := os.Stat(dir); err != nil || !fi.IsDir() {
|
||||
return "", fmt.Errorf("archive volume %s is %s on its node, which is not a directory here; run this on the node that holds it, or pass -archive-dir", pv.Name, dir)
|
||||
return "", fmt.Errorf("%s %s is %s on its node, which is not a directory here; run this on the node that holds it, or pass %s", kind.what, pv.Name, dir, kind.dirFlag)
|
||||
}
|
||||
return dir, nil
|
||||
}
|
||||
@@ -304,10 +358,10 @@ func resolveArchiveDir(ctx context.Context, ns, pvcName string, bind bool, log i
|
||||
// bindVolume runs a pod that mounts the PVC and exits, which is what makes a
|
||||
// WaitForFirstConsumer volume (k3s local-path) get provisioned. The pod uses
|
||||
// the control plane's own image, which every install already has.
|
||||
func bindVolume(ctx context.Context, cl client.Client, ns, pvcName string, log io.Writer) error {
|
||||
func bindVolume(ctx context.Context, cl client.Client, ns, pvcName string, kind volumeKind, log io.Writer) error {
|
||||
var api appsv1.Deployment
|
||||
if err := cl.Get(ctx, types.NamespacedName{Namespace: platform.DefaultControlNamespace, Name: "felis-api"}, &api); err != nil {
|
||||
return fmt.Errorf("find the felis image to bind the archive volume with: %w", err)
|
||||
return fmt.Errorf("find the felis image to bind the %s with: %w", kind.what, err)
|
||||
}
|
||||
if len(api.Spec.Template.Spec.Containers) == 0 {
|
||||
return errors.New("felis-api has no container to take the image from")
|
||||
@@ -315,9 +369,9 @@ func bindVolume(ctx context.Context, cl client.Client, ns, pvcName string, log i
|
||||
image := api.Spec.Template.Spec.Containers[0].Image
|
||||
pod := platform.VolumeBinderPod(ns, pvcName, image)
|
||||
if err := cl.Create(ctx, pod); err != nil {
|
||||
return fmt.Errorf("start a pod to bind the archive volume: %w", err)
|
||||
return fmt.Errorf("start a pod to bind the %s: %w", kind.what, err)
|
||||
}
|
||||
fmt.Fprintf(log, "felis offsite: binding the archive volume %s/%s (pod %s)\n", ns, pvcName, pod.Name)
|
||||
fmt.Fprintf(log, "felis offsite: binding the %s %s/%s (pod %s)\n", kind.what, ns, pvcName, pod.Name)
|
||||
defer func() {
|
||||
_ = cl.Delete(context.Background(), pod, client.PropagationPolicy(metav1.DeletePropagationBackground))
|
||||
}()
|
||||
@@ -333,7 +387,7 @@ func bindVolume(ctx context.Context, cl client.Client, ns, pvcName string, log i
|
||||
case <-time.After(2 * time.Second):
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("the archive volume %s/%s did not bind within 3 minutes; see kubectl -n %s describe pod %s", ns, pvcName, ns, pod.Name)
|
||||
return fmt.Errorf("the %s %s/%s did not bind within 3 minutes; see kubectl -n %s describe pod %s", kind.what, ns, pvcName, ns, pod.Name)
|
||||
}
|
||||
|
||||
func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
@@ -348,7 +402,7 @@ func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) in
|
||||
return 1
|
||||
}
|
||||
if !cfg.Offsite.Enabled() {
|
||||
fmt.Fprintln(stdout, "off-site copy: not configured. World archives and database bundles exist on this machine only.")
|
||||
fmt.Fprintln(stdout, "off-site copy: not configured. World archives, database bundles, user images and uploaded modpacks exist on this machine only.")
|
||||
fmt.Fprintln(stdout, "See docs/troubleshooting.md §16, \"Keep a copy somewhere else\".")
|
||||
return 1
|
||||
}
|
||||
@@ -381,10 +435,25 @@ func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) in
|
||||
r := st.Result
|
||||
fmt.Fprintf(stdout, "bucket holds: %d world archives (%s), %d database bundles, newest %s\n",
|
||||
r.RemoteWorlds, offsite.HumanBytes(r.RemoteBytes), r.RemoteDB, orNone(r.NewestDB))
|
||||
if r.ImageIndex != "" {
|
||||
fmt.Fprintf(stdout, "images: %d in %d repositories (%s), registry index %s\n",
|
||||
r.Images, r.ImageRepos, offsite.HumanBytes(r.RemoteImageBytes), r.ImageIndex)
|
||||
} else {
|
||||
fmt.Fprintln(stdout, "images: not copied (no in-cluster registry, or no sync has reached it yet)")
|
||||
}
|
||||
if r.UploadIndex != "" {
|
||||
fmt.Fprintf(stdout, "uploads: %d submission contexts (%s), uploads index %s\n",
|
||||
r.Uploads, offsite.HumanBytes(r.RemoteUploadBytes), r.UploadIndex)
|
||||
} else {
|
||||
fmt.Fprintln(stdout, "uploads: not copied (an s3:// uploads store, or no sync has reached the volume yet)")
|
||||
}
|
||||
fmt.Fprintf(stdout, "waiting: %d world archives not yet copied\n", r.WorldsPending)
|
||||
for _, m := range r.WorldsMissing {
|
||||
fmt.Fprintf(stdout, "missing: %s is recorded but not on the volume\n", m)
|
||||
}
|
||||
for _, m := range r.ImagesIncomplete {
|
||||
fmt.Fprintf(stdout, "not whole: %s\n", m)
|
||||
}
|
||||
if st.LastSuccess.IsZero() || now.Sub(st.LastSuccess) > offsite.StaleAfter {
|
||||
fmt.Fprintf(stdout, "\nThe last successful sync is older than %s: journalctl -u felis-offsite -n 50\n", dbbackup.Age(offsite.StaleAfter))
|
||||
return 1
|
||||
@@ -435,6 +504,34 @@ func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
|
||||
total += w.Size
|
||||
}
|
||||
fmt.Fprintf(stdout, "world archives: %d (%s)\n", len(worlds), offsite.HumanBytes(total))
|
||||
versions, err := offsite.ImageIndexes(ctx, env.bucket)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "registry index versions (%d, newest first; restore one with fetch-images -at):\n", len(versions))
|
||||
for i := len(versions) - 1; i >= 0; i-- {
|
||||
x, err := offsite.LoadImageIndex(ctx, env.bucket, env.key, versions[i])
|
||||
if err != nil {
|
||||
fmt.Fprintf(stdout, " %s unreadable: %v\n", versions[i], err)
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(stdout, " %s %d images in %d repositories\n", versions[i], x.Images(), len(x.Repositories))
|
||||
}
|
||||
uploads, err := offsite.UploadIndexes(ctx, env.bucket)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "uploads index versions (%d, newest first; restore one with fetch-uploads -at):\n", len(uploads))
|
||||
for i := len(uploads) - 1; i >= 0; i-- {
|
||||
x, err := offsite.LoadUploadIndex(ctx, env.bucket, env.key, uploads[i])
|
||||
if err != nil {
|
||||
fmt.Fprintf(stdout, " %s unreadable: %v\n", uploads[i], err)
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(stdout, " %s %d submission contexts (%s)\n", uploads[i], len(x.Contexts), offsite.HumanBytes(x.Bytes()))
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
@@ -531,12 +628,12 @@ func offsiteFetchWorlds(fs *flag.FlagSet, args []string, stdout, stderr io.Write
|
||||
defer cancel()
|
||||
dir := *archiveDir
|
||||
if dir == "" {
|
||||
if dir, err = resolveArchiveDir(ctx, cfg.K8s.Namespace, *backupPVC, true, stderr); err != nil {
|
||||
if dir, err = resolveVolumeDir(ctx, cfg.K8s.Namespace, *backupPVC, archiveVolume, true, stderr); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
drv, err := openStore(ctx, cfg.Database.URL, false)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-worlds: open database: %v\n", err)
|
||||
return 1
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"slices"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/imagepush"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/registrygate"
|
||||
"felis.lolicon.best/internal/registryprune"
|
||||
)
|
||||
|
||||
// registryImages is the platform registry as the off-site copy sees it: read
|
||||
// anonymously through the gate (the catalog, its manifest index, manifests and
|
||||
// blobs) and, for a restore, written as the platform principal. Both go through
|
||||
// the node's loopback hostPort, the way the installer pushes.
|
||||
type registryImages struct {
|
||||
host string
|
||||
index *registryprune.Client
|
||||
source *imagepush.Source
|
||||
pusher *imagepush.Pusher
|
||||
}
|
||||
|
||||
func newRegistryImages(endpoint string) *registryImages {
|
||||
return ®istryImages{
|
||||
host: endpoint,
|
||||
index: ®istryprune.Client{Endpoint: "http://" + endpoint},
|
||||
source: &imagepush.Source{Scheme: "http"},
|
||||
}
|
||||
}
|
||||
|
||||
func (r *registryImages) Repositories(ctx context.Context) ([]string, error) {
|
||||
return r.index.Repositories(ctx)
|
||||
}
|
||||
|
||||
func (r *registryImages) Revisions(ctx context.Context, repo string) ([]string, map[string]string, error) {
|
||||
idx, err := r.index.Index(ctx, repo)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
digests := make([]string, 0, len(idx.Revisions))
|
||||
for _, rev := range idx.Revisions {
|
||||
digests = append(digests, rev.Digest)
|
||||
}
|
||||
return digests, idx.Tags, nil
|
||||
}
|
||||
|
||||
func (r *registryImages) Manifest(ctx context.Context, repo, digest string) ([]byte, string, error) {
|
||||
body, mt, err := r.source.Manifest(ctx, r.host, repo, digest)
|
||||
return body, mt, registryGone(err)
|
||||
}
|
||||
|
||||
func (r *registryImages) Blob(ctx context.Context, repo, digest string) (io.ReadCloser, error) {
|
||||
rc, err := r.source.Blob(ctx, r.host, repo, digest)
|
||||
return rc, registryGone(err)
|
||||
}
|
||||
|
||||
func (r *registryImages) PutBlob(ctx context.Context, repo, digest string, size int64, open func() (io.ReadCloser, error)) error {
|
||||
return r.pusher.UploadBlob(ctx, r.host, repo, digest, size, open)
|
||||
}
|
||||
|
||||
func (r *registryImages) PutManifest(ctx context.Context, repo, reference, mediaType string, body []byte) error {
|
||||
_, err := r.pusher.PutManifest(ctx, r.host, repo, reference, mediaType, body)
|
||||
return err
|
||||
}
|
||||
|
||||
// imagePins lists, per repository of the registry refs spell as host, the
|
||||
// digests the MinecraftServers' specs and the image whitelist pin: what a
|
||||
// restored database and its servers will ask the registry for.
|
||||
func imagePins(db *sql.DB, host string) func(ctx context.Context) (map[string][]string, error) {
|
||||
return func(ctx context.Context) (map[string][]string, error) {
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("reach the cluster: %w", err)
|
||||
}
|
||||
var servers v1alpha1.MinecraftServerList
|
||||
if err := cl.List(ctx, &servers); err != nil {
|
||||
return nil, fmt.Errorf("list MinecraftServers: %w", err)
|
||||
}
|
||||
refs := make([]string, 0, len(servers.Items))
|
||||
for _, s := range servers.Items {
|
||||
refs = append(refs, s.Spec.Image)
|
||||
}
|
||||
rows, err := db.QueryContext(ctx, `SELECT image_ref FROM image_whitelist`)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read the image whitelist: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
for rows.Next() {
|
||||
var ref string
|
||||
if err := rows.Scan(&ref); err != nil {
|
||||
return nil, fmt.Errorf("read the image whitelist: %w", err)
|
||||
}
|
||||
refs = append(refs, ref)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, fmt.Errorf("read the image whitelist: %w", err)
|
||||
}
|
||||
pins := map[string][]string{}
|
||||
for _, ref := range refs {
|
||||
repo, _, digest, ok := registryprune.ParseRef(ref, host)
|
||||
if ok && digest != "" && !slices.Contains(pins[repo], digest) {
|
||||
pins[repo] = append(pins[repo], digest)
|
||||
}
|
||||
}
|
||||
return pins, nil
|
||||
}
|
||||
}
|
||||
|
||||
// registryGone marks a 404 as a manifest or blob the registry no longer holds.
|
||||
func registryGone(err error) error {
|
||||
var se *imagepush.StatusError
|
||||
if errors.As(err, &se) && se.Code == http.StatusNotFound {
|
||||
return fmt.Errorf("%w: %v", offsite.ErrImageGone, err)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// offsiteRegistryEndpoint is where the host reaches the registry whose images
|
||||
// the off-site copy covers: flag when given ("off" for none), otherwise the
|
||||
// loopback hostPort of the in-cluster registry [registry] url names. A
|
||||
// registry outside the cluster is not this install's to copy.
|
||||
func offsiteRegistryEndpoint(flag string, reg config.RegistryConfig) string {
|
||||
switch flag {
|
||||
case "off":
|
||||
return ""
|
||||
case "":
|
||||
default:
|
||||
return flag
|
||||
}
|
||||
host, _, _ := strings.Cut(reg.URL, "/")
|
||||
name, _, _ := strings.Cut(host, ":")
|
||||
if strings.HasSuffix(name, ".svc") || strings.HasSuffix(name, ".svc.cluster.local") {
|
||||
return loopbackEndpoint(host)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// registryWriteCredential is the credential a host-side push uses:
|
||||
// FELIS_REGISTRY_USERNAME/PASSWORD when set, otherwise the platform principal
|
||||
// with REGISTRY_PLATFORM_TOKEN from the installer's secrets file.
|
||||
func registryWriteCredential(secrets string) (string, string, error) {
|
||||
if err := loadEnvFile(secrets); err != nil {
|
||||
return "", "", fmt.Errorf("read %s: %w", secrets, err)
|
||||
}
|
||||
user, pass := os.Getenv("FELIS_REGISTRY_USERNAME"), os.Getenv("FELIS_REGISTRY_PASSWORD")
|
||||
if pass == "" {
|
||||
user, pass = registrygate.PrincipalPlatform, os.Getenv("REGISTRY_PLATFORM_TOKEN")
|
||||
}
|
||||
if pass == "" {
|
||||
return "", "", errors.New("no registry credential: set FELIS_REGISTRY_PASSWORD or run as root on the node (REGISTRY_PLATFORM_TOKEN in /etc/felis/secrets.env)")
|
||||
}
|
||||
return user, pass, nil
|
||||
}
|
||||
|
||||
func offsiteFetchImages(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy)")
|
||||
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||
registry := fs.String("registry", "", "host[:port] of the registry to push into (default: the loopback hostPort of the in-cluster registry)")
|
||||
secrets := fs.String("secrets-env", "/etc/felis/secrets.env", "installer secrets file holding REGISTRY_PLATFORM_TOKEN, read when FELIS_REGISTRY_PASSWORD is unset")
|
||||
at := fs.String("at", "", "registry index version to restore (default: the newest; `felis offsite list` shows them)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
cfg, env, err := loadOffsite(*cfgPath, *envFile)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-images: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
endpoint := offsiteRegistryEndpoint(*registry, cfg.Registry)
|
||||
if endpoint == "" {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-images: [registry] url %q is not the in-cluster registry; pass -registry host:port\n", cfg.Registry.URL)
|
||||
return 2
|
||||
}
|
||||
user, pass, err := registryWriteCredential(*secrets)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-images: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 6*time.Hour)
|
||||
defer cancel()
|
||||
stamp, idx, err := offsite.ChooseImageIndex(ctx, env.bucket, env.key, *at)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-images: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis offsite fetch-images: restoring registry index %s (%d repositories, %d images) into %s\n",
|
||||
stamp, len(idx.Repositories), idx.Images(), endpoint)
|
||||
target := newRegistryImages(endpoint)
|
||||
target.pusher = &imagepush.Pusher{Scheme: "http", Username: user, Password: pass}
|
||||
res, err := offsite.FetchImages(ctx, env.bucket, env.key, idx, target, stderr)
|
||||
fmt.Fprintf(stdout, "felis offsite fetch-images: %d of %d repositories restored, %d images, %d tags, %d blobs pushed (%s)\n",
|
||||
res.Repositories, len(idx.Repositories), res.Manifests, res.Tags, res.BlobsPushed, offsite.HumanBytes(res.BytesPushed))
|
||||
for _, f := range res.Failures {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-images: %s\n", f)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-images: %v (a second run pushes only what is still missing)\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -2,12 +2,14 @@ package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/imagepush"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
)
|
||||
|
||||
@@ -94,3 +96,27 @@ func TestOffsiteFetchDBRejectsOddNames(t *testing.T) {
|
||||
t.Fatalf("exit %d: %s", code, errb.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsiteRegistryEndpoint(t *testing.T) {
|
||||
for _, tc := range []struct{ flag, url, want string }{
|
||||
{"", "registry.felis.svc:5000", "127.0.0.1:5000"},
|
||||
{"", "registry.felis.svc.cluster.local:5001", "127.0.0.1:5001"},
|
||||
{"", "ghcr.io/acme", ""},
|
||||
{"", "", ""},
|
||||
{"off", "registry.felis.svc:5000", ""},
|
||||
{"10.0.0.5:5000", "ghcr.io/acme", "10.0.0.5:5000"},
|
||||
} {
|
||||
if got := offsiteRegistryEndpoint(tc.flag, config.RegistryConfig{URL: tc.url}); got != tc.want {
|
||||
t.Errorf("offsiteRegistryEndpoint(%q, %q) = %q, want %q", tc.flag, tc.url, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistryGoneMarksNotFound(t *testing.T) {
|
||||
if err := registryGone(&imagepush.StatusError{Op: "get blob", Code: 404}); !errors.Is(err, offsite.ErrImageGone) {
|
||||
t.Fatalf("404 = %v, want ErrImageGone", err)
|
||||
}
|
||||
if err := registryGone(&imagepush.StatusError{Op: "get blob", Code: 503}); errors.Is(err, offsite.ErrImageGone) {
|
||||
t.Fatalf("503 = %v, want it kept an ordinary failure", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
)
|
||||
|
||||
func offsiteFetchUploads(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy)")
|
||||
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
|
||||
uploadsDir := fs.String("uploads-dir", "", "host directory of the submission uploads volume (default: resolved from the uploads PVC, binding it if needed)")
|
||||
uploadsPVC := fs.String("uploads-pvc", platform.UploadsPVCName, "the submission uploads PVC, in the control-plane namespace")
|
||||
at := fs.String("at", "", "uploads index version to restore (default: the newest; `felis offsite list` shows them)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
cfg, env, err := loadOffsite(*cfgPath, *envFile)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-uploads: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 6*time.Hour)
|
||||
defer cancel()
|
||||
stamp, idx, err := offsite.ChooseUploadIndex(ctx, env.bucket, env.key, *at)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-uploads: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
dir := *uploadsDir
|
||||
if dir == "" {
|
||||
if !isLocalUploadsPath(cfg.Registry.UserUploadsContext) {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-uploads: [registry] user_uploads_context %q is not the uploads volume; pass -uploads-dir to restore into a directory anyway\n", cfg.Registry.UserUploadsContext)
|
||||
return 2
|
||||
}
|
||||
if dir, err = resolveVolumeDir(ctx, platform.DefaultControlNamespace, *uploadsPVC, uploadsVolume, true, stderr); err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-uploads: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(stdout, "felis offsite fetch-uploads: restoring uploads index %s (%d submission contexts, %s) into %s\n",
|
||||
stamp, len(idx.Contexts), offsite.HumanBytes(idx.Bytes()), dir)
|
||||
res, err := offsite.FetchUploads(ctx, env.bucket, env.key, idx, dir, platform.ControlPlaneUID, platform.ControlPlaneUID, stderr)
|
||||
fmt.Fprintf(stdout, "felis offsite fetch-uploads: %d written (%s), %d already in place, %d failed\n",
|
||||
res.Written, offsite.HumanBytes(res.Bytes), res.Present, len(res.Failures))
|
||||
for _, f := range res.Failures {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-uploads: %s\n", f)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis offsite fetch-uploads: %v (a second run writes only what is still missing)\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -117,8 +117,12 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
FelisImage: os.Getenv("FELIS_IMAGE"),
|
||||
// Uncached: the maintenance-lock check lists Jobs only when a server is
|
||||
// about to start, which does not justify a namespace-wide Job informer.
|
||||
Jobs: mgr.GetAPIReader(),
|
||||
Watch: watch,
|
||||
Jobs: mgr.GetAPIReader(),
|
||||
// Uncached too: RCON Secrets are read by name, so the Role grants
|
||||
// secrets:get without the list/watch an informer would need.
|
||||
Secrets: mgr.GetAPIReader(),
|
||||
Recorder: mgr.GetEventRecorderFor("felis-operator"),
|
||||
Watch: watch,
|
||||
}
|
||||
if err := r.SetupWithManager(mgr); err != nil {
|
||||
fmt.Fprintf(stderr, "felis operator: setup controller: %v\n", err)
|
||||
|
||||
+91
-2
@@ -11,7 +11,9 @@ import (
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/imagepin"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/api/meta"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
@@ -35,12 +37,17 @@ func cmdPinImages(args []string, stdout, stderr io.Writer) int {
|
||||
namespace := fs.String("namespace", platform.DefaultMinecraftNamespace, "namespace the MinecraftServers live in")
|
||||
registry := fs.String("registry", defaultRegistryURL, "registry host[:port] the image refs spell")
|
||||
endpoint := fs.String("endpoint", "", "host[:port] to reach the registry at (default: 127.0.0.1 on the registry's port, its hostPort on this node)")
|
||||
system := fs.String("system", "", "pin this system server ("+naming.SystemLoginServer+" or "+naming.SystemLobbyServer+") to the build its tag names now, instead of the user servers")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if *system != "" && *system != naming.SystemLoginServer && *system != naming.SystemLobbyServer {
|
||||
fmt.Fprintf(stderr, "felis pin-images: --system takes %s or %s, not %q\n", naming.SystemLoginServer, naming.SystemLobbyServer, *system)
|
||||
return 2
|
||||
}
|
||||
if *endpoint == "" {
|
||||
*endpoint = loopbackEndpoint(*registry)
|
||||
}
|
||||
@@ -51,6 +58,9 @@ func cmdPinImages(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
if *system != "" {
|
||||
return reportSystemPin(ctx, cl, *namespace, *system, imagepin.Resolver{Registry: *registry, Endpoint: *endpoint}, stdout, stderr)
|
||||
}
|
||||
outcomes, err := pinUserServerImages(ctx, cl, *namespace, imagepin.Resolver{Registry: *registry, Endpoint: *endpoint})
|
||||
if meta.IsNoMatchError(err) {
|
||||
fmt.Fprintln(stdout, "felis pin-images: no MinecraftServer CRD yet, so no server to pin")
|
||||
@@ -77,6 +87,84 @@ func cmdPinImages(args []string, stdout, stderr io.Writer) int {
|
||||
return exit
|
||||
}
|
||||
|
||||
// reportSystemPin runs pinSystemServerImage for `felis pin-images --system` and
|
||||
// prints what it did. Only a failed pin exits non-zero: the installer falls back
|
||||
// to restarting the pod on its tag then.
|
||||
func reportSystemPin(ctx context.Context, cl client.Client, namespace, name string, r imagepin.Resolver, stdout, stderr io.Writer) int {
|
||||
o, err := pinSystemServerImage(ctx, cl, namespace, name, r)
|
||||
switch {
|
||||
case meta.IsNoMatchError(err):
|
||||
fmt.Fprintln(stdout, "felis pin-images: no MinecraftServer CRD yet, so no server to pin")
|
||||
return 0
|
||||
case err == nil && o.err != nil:
|
||||
err = o.err
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis pin-images: %s: %v\n", name, err)
|
||||
return 1
|
||||
}
|
||||
switch {
|
||||
case o.updated:
|
||||
fmt.Fprintf(stdout, "felis pin-images: %s: %s; the operator rolls it onto that build\n", name, strings.Join(o.changes, ", "))
|
||||
default:
|
||||
fmt.Fprintf(stdout, "felis pin-images: %s: %s\n", name, o.skipped)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// pinSystemServerImage fixes a system server to the build its image tag names now,
|
||||
// replacing the digest of an earlier build. The installer runs it after pushing a
|
||||
// rebuilt login or lobby image, and the operator rolls the StatefulSet onto the new
|
||||
// ref, so the build a system server runs is written in its spec and moves only when
|
||||
// a build did. An image outside the platform registry, or one naming no tag to
|
||||
// follow, is the admin's choice and is left alone; so is a server whose image an
|
||||
// admin retargets while this runs.
|
||||
func pinSystemServerImage(ctx context.Context, cl client.Client, namespace, name string, r imagepin.Resolver) (systemServerOutcome, error) {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: namespace, Name: name}, &ms); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return systemServerOutcome{name: name, skipped: "not present yet; sudo felis setup creates it"}, nil
|
||||
}
|
||||
return systemServerOutcome{}, err
|
||||
}
|
||||
if ms.Labels[v1alpha1.LabelSystemRole] != name {
|
||||
return systemServerOutcome{name: name, err: fmt.Errorf(
|
||||
"MinecraftServer %s/%s is not marked as the Felis %q system server; left alone", namespace, name, name)}, nil
|
||||
}
|
||||
tagged := withoutDigest(ms.Spec.Image)
|
||||
if !r.Covers(tagged) || !strings.Contains(tagged[strings.LastIndex(tagged, "/")+1:], ":") {
|
||||
return systemServerOutcome{name: name, available: true, skipped: "runs " + ms.Spec.Image +
|
||||
", which names no platform registry tag to follow; left alone"}, nil
|
||||
}
|
||||
pinned, err := r.Pin(ctx, tagged)
|
||||
if err != nil {
|
||||
return systemServerOutcome{name: name, err: fmt.Errorf("resolve %s: %w", tagged, err)}, nil
|
||||
}
|
||||
changed, err := patchOnConflictRetry(ctx, cl, &ms, func() bool {
|
||||
if withoutDigest(ms.Spec.Image) != tagged || ms.Spec.Image == pinned {
|
||||
return false
|
||||
}
|
||||
ms.Spec.Image = pinned
|
||||
return true
|
||||
})
|
||||
if err != nil {
|
||||
return systemServerOutcome{name: name, err: fmt.Errorf("patch %s: %w", name, err)}, nil
|
||||
}
|
||||
if !changed {
|
||||
return systemServerOutcome{name: name, available: true, skipped: "already runs " + ms.Spec.Image}, nil
|
||||
}
|
||||
return systemServerOutcome{name: name, available: true, updated: true,
|
||||
changes: []string{"spec.image pinned to " + pinned}}, nil
|
||||
}
|
||||
|
||||
// withoutDigest drops the @sha256:… of a pinned ref, leaving the tag it came from.
|
||||
func withoutDigest(ref string) string {
|
||||
if i := strings.Index(ref, "@"); i >= 0 {
|
||||
return ref[:i]
|
||||
}
|
||||
return ref
|
||||
}
|
||||
|
||||
// loopbackEndpoint is the registry's port on 127.0.0.1: the registry Deployment
|
||||
// binds it as a hostPort, and containerd's mirror and the installer's pushes use
|
||||
// the same address.
|
||||
@@ -88,8 +176,9 @@ func loopbackEndpoint(registry string) string {
|
||||
}
|
||||
|
||||
// pinUserServerImages patches spec.image of every user server whose image the
|
||||
// resolver covers and is not yet pinned. System servers are left on their tags:
|
||||
// the installer rebuilds and restarts them on purpose (restart_existing_system_servers).
|
||||
// resolver covers and is not yet pinned. System servers are the installer's to move:
|
||||
// it re-pins them with --system when it rolls them onto a new build
|
||||
// (restart_existing_system_servers).
|
||||
// A server that is already pinned, or runs an image from elsewhere, produces no
|
||||
// outcome, so a pinned fleet reports nothing. A running server restarts once as
|
||||
// the operator rolls its StatefulSet onto the pinned ref, which is the build it
|
||||
|
||||
+52
-49
@@ -16,10 +16,8 @@ import (
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/backup"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
"felis.lolicon.best/internal/store"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
@@ -33,11 +31,17 @@ import (
|
||||
// cadence, and RunOnce is idempotent and restart-safe, so a missed or retried
|
||||
// run simply converges. Only the tarLocal archive backend is wired in this
|
||||
// build; the snapshot backends (§19) are a later integration.
|
||||
//
|
||||
// --retention-only runs the archive-store half alone (reaper.RunRetention):
|
||||
// the CronJob an install without a worlds root gets, so backups past their
|
||||
// expiry still leave the store there. It builds no Kubernetes client and reads
|
||||
// no world.
|
||||
func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("reaper", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||
worldsRoot := fs.String("worlds-root", "/worlds", "mount root under which world PVCs are visible (tarLocal: <root>/<pvc>, else the stock local-path <root>/<pv-name>_<ns>_<pvc-name>)")
|
||||
retentionOnly := fs.Bool("retention-only", false, "only expire, read back and sweep the archive store: no server is evaluated and no world is read or deleted, so neither the worlds root nor the Kubernetes API is needed")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -56,13 +60,16 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
|
||||
ctx := ctrl.SetupSignalHandler()
|
||||
|
||||
scheme := runtime.NewScheme()
|
||||
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
|
||||
utilruntime.Must(v1alpha1.AddToScheme(scheme))
|
||||
cl, err := client.New(ctrl.GetConfigOrDie(), client.Options{Scheme: scheme})
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: build k8s client: %v\n", err)
|
||||
return 1
|
||||
var cl client.Client
|
||||
if !*retentionOnly {
|
||||
scheme := runtime.NewScheme()
|
||||
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
|
||||
utilruntime.Must(v1alpha1.AddToScheme(scheme))
|
||||
cl, err = client.New(ctrl.GetConfigOrDie(), client.Options{Scheme: scheme})
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: build k8s client: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
}
|
||||
|
||||
archiver, err := buildArchiver(ctx, cfg, *worldsRoot, cl)
|
||||
@@ -71,7 +78,7 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
drv, err := openStore(ctx, cfg.Database.URL, false)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: open database: %v\n", err)
|
||||
return 1
|
||||
@@ -81,9 +88,13 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
r := &reaper.Reaper{
|
||||
Cfg: rcfg,
|
||||
Store: reaper.NewPGStore(drv.DB()),
|
||||
Cluster: reaper.NewK8sCluster(cl, cfg.K8s.Namespace),
|
||||
Archiver: archiver,
|
||||
}
|
||||
if *retentionOnly {
|
||||
fmt.Fprintln(stderr, "felis reaper: retention only — no worlds root is configured, so idle worlds are neither archived nor released")
|
||||
return reportReaperRun(r.RunRetention(ctx), stdout, stderr)
|
||||
}
|
||||
r.Cluster = reaper.NewK8sCluster(cl, cfg.K8s.Namespace)
|
||||
|
||||
// Pre-reap warnings go out by email when [smtp] is configured (the same
|
||||
// relay and password_ref convention felis-api uses); without it the channel
|
||||
@@ -114,13 +125,7 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
return email, nil
|
||||
},
|
||||
notifier: &mail.SMTP{
|
||||
Host: cfg.SMTP.Host,
|
||||
Port: cfg.SMTP.Port,
|
||||
From: cfg.SMTP.From,
|
||||
Username: cfg.SMTP.Username,
|
||||
Password: password,
|
||||
},
|
||||
notifier: smtpRelay(cfg.SMTP, password),
|
||||
}
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis reaper: [smtp] not configured — pre-reap warnings are logged and NOT marked sent")
|
||||
@@ -139,14 +144,24 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
// FelisWorldJobFailed rule) reach the operator: a world that cannot be archived
|
||||
// is kept, and without this nobody would learn that it is never reaped.
|
||||
func reportReaperRun(sum reaper.Summary, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d\n",
|
||||
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed)
|
||||
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d awaiting_stop=%d warned=%d skipped=%d store_full=%d evicted=%d expired=%d expire_failed=%d verified=%d corrupt=%d verify_failed=%d swept=%d orphan_archives=%d\n",
|
||||
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.AwaitingStop, sum.Warned, sum.Skipped, sum.StoreFull,
|
||||
sum.EvictedEarly, sum.BackupsExpired, sum.ExpireFailed,
|
||||
sum.Verified, sum.Corrupt, sum.VerifyFailed, sum.Swept, sum.OrphanArchives)
|
||||
if !sum.Failed() {
|
||||
return 0
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis reaper: %d servers failed (%d kept because the backup store is full) and %d expired backups were not removed; the errors are above, and each is retried next run\n",
|
||||
sum.Skipped, sum.StoreFull, sum.ExpireFailed)
|
||||
if sum.Skipped > 0 || sum.ExpireFailed > 0 {
|
||||
fmt.Fprintf(stderr, "felis reaper: %d servers failed (%d kept because the backup store is full) and %d expired backups were not removed; the errors are above, and each is retried next run\n",
|
||||
sum.Skipped, sum.StoreFull, sum.ExpireFailed)
|
||||
}
|
||||
if sum.Corrupt > 0 {
|
||||
fmt.Fprintf(stderr, "felis reaper: %d archives did not read back and are marked corrupt; they are no longer offered for restore (the errors are above)\n", sum.Corrupt)
|
||||
}
|
||||
if sum.VerifyFailed > 0 || sum.SweepFailed {
|
||||
fmt.Fprintf(stderr, "felis reaper: %d archives could not be read back and the store sweep completed=%t; both are retried next run\n",
|
||||
sum.VerifyFailed, !sum.SweepFailed)
|
||||
}
|
||||
return 1
|
||||
}
|
||||
|
||||
@@ -239,14 +254,20 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
||||
|
||||
// buildArchiver constructs the WorldArchiver. Only tarLocal is implemented in
|
||||
// this build; the resolver maps each world PVC to its directory under worldsRoot
|
||||
// (resolveWorldDir).
|
||||
// (resolveWorldDir). A nil cl is the retention-only run, which never archives a
|
||||
// world, so its archiver resolves none.
|
||||
func buildArchiver(ctx context.Context, cfg *config.Config, worldsRoot string, cl client.Client) (backup.WorldArchiver, error) {
|
||||
switch cfg.Archive.Store {
|
||||
case "tarLocal":
|
||||
return &backup.TarLocal{
|
||||
BackupRoot: cfg.Archive.LocalPath,
|
||||
Resolve: resolveWorldDir(ctx, cl, cfg.K8s.Namespace, worldsRoot),
|
||||
}, nil
|
||||
t := &backup.TarLocal{BackupRoot: cfg.Archive.LocalPath}
|
||||
if cl != nil {
|
||||
t.Resolve = resolveWorldDir(ctx, cl, cfg.K8s.Namespace, worldsRoot)
|
||||
} else {
|
||||
t.Resolve = func(pvc string) (string, error) {
|
||||
return "", fmt.Errorf("resolve world PVC %s: this run has no worlds root (retention only)", pvc)
|
||||
}
|
||||
}
|
||||
return t, nil
|
||||
default:
|
||||
return nil, fmt.Errorf("[archive] store %q is not implemented in this build (only tarLocal)", cfg.Archive.Store)
|
||||
}
|
||||
@@ -289,27 +310,9 @@ func resolveWorldDir(ctx context.Context, cl client.Client, namespace, worldsRoo
|
||||
}
|
||||
}
|
||||
|
||||
// parseSpanDuration parses the human spans used in felis.toml's [archive] table:
|
||||
// "3mo" (months≈30d), "15d" (days), or any time.ParseDuration unit ("12h").
|
||||
func parseSpanDuration(s string) (time.Duration, error) {
|
||||
s = strings.TrimSpace(s)
|
||||
switch {
|
||||
case strings.HasSuffix(s, "mo"):
|
||||
n, err := strconv.Atoi(strings.TrimSuffix(s, "mo"))
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return time.Duration(n) * 30 * 24 * time.Hour, nil
|
||||
case strings.HasSuffix(s, "d"):
|
||||
n, err := strconv.Atoi(strings.TrimSuffix(s, "d"))
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return time.Duration(n) * 24 * time.Hour, nil
|
||||
default:
|
||||
return time.ParseDuration(s)
|
||||
}
|
||||
}
|
||||
// parseSpanDuration parses the human spans used in felis.toml's [archive] table
|
||||
// (config.ParseSpan).
|
||||
func parseSpanDuration(s string) (time.Duration, error) { return config.ParseSpan(s) }
|
||||
|
||||
// parseByteSize parses a Kubernetes-style quantity ("200Gi", "10G") into bytes.
|
||||
// An empty string means unlimited (0).
|
||||
|
||||
@@ -27,9 +27,14 @@ func TestReportReaperRunFailsTheJob(t *testing.T) {
|
||||
want int
|
||||
}{
|
||||
{"clean", reaper.Summary{Evaluated: 3, WorldsReaped: 1, AwaitingOffsite: 1}, 0},
|
||||
{"waiting for a stop", reaper.Summary{Evaluated: 3, AwaitingStop: 1}, 0},
|
||||
{"server failed", reaper.Summary{Evaluated: 3, Skipped: 1}, 1},
|
||||
{"store full", reaper.Summary{Evaluated: 3, Skipped: 1, StoreFull: 1}, 1},
|
||||
{"expiry failed", reaper.Summary{Evaluated: 3, ExpireFailed: 2}, 1},
|
||||
{"corrupt archive", reaper.Summary{Evaluated: 3, Verified: 4, Corrupt: 1}, 1},
|
||||
{"read-back failed", reaper.Summary{Evaluated: 3, VerifyFailed: 1}, 1},
|
||||
{"sweep failed", reaper.Summary{Evaluated: 3, SweepFailed: true}, 1},
|
||||
{"orphans kept", reaper.Summary{Evaluated: 3, Swept: 2, OrphanArchives: 1}, 0},
|
||||
} {
|
||||
var out, errb bytes.Buffer
|
||||
if got := reportReaperRun(tc.sum, &out, &errb); got != tc.want {
|
||||
|
||||
@@ -32,11 +32,11 @@ func archiveTempWorld(t *testing.T, files map[string]string) (ref string, backup
|
||||
BackupRoot: backupRoot,
|
||||
Resolve: func(string) (string, error) { return srcDir, nil },
|
||||
}
|
||||
got, _, err := ar.Archive(context.Background(), "survival", "world-survival-0")
|
||||
got, err := ar.Archive(context.Background(), "survival", "world-survival-0")
|
||||
if err != nil {
|
||||
t.Fatalf("Archive: %v", err)
|
||||
}
|
||||
return string(got), backupRoot
|
||||
return string(got.Ref), backupRoot
|
||||
}
|
||||
|
||||
// The restore subcommand must extract the archived world into the target world
|
||||
|
||||
@@ -0,0 +1,309 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"syscall"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// rotate-token replaces one internal caller's token (naming.CallerTokens): a new
|
||||
// value goes into the installer's record, the control-namespace Secret and the
|
||||
// replica the caller's pods mount, felis-api rolls so it accepts only the new
|
||||
// value, and then the caller restarts so it presents it. Between the api's
|
||||
// rollout and the caller's restart the caller is turned away with 401; for the
|
||||
// login gate and the proxy that is the few seconds of a pod or unit restart.
|
||||
|
||||
const (
|
||||
defaultSecretsEnvPath = "/etc/felis/secrets.env"
|
||||
defaultLinkPropsPath = "/opt/felis/velocity/plugins/felis-link/felis-link.properties"
|
||||
velocityUnit = "felis-velocity"
|
||||
)
|
||||
|
||||
// installerTokenKeys names each caller's token in the installer's secrets.env
|
||||
// (deploy/bootstrap.sh load_or_make_secrets). A re-run of the installer applies
|
||||
// these values to the Secrets, so a rotation that skipped the file would be
|
||||
// undone by the next upgrade.
|
||||
var installerTokenKeys = map[string]string{
|
||||
"velocity": "SERVICE_TOKEN",
|
||||
"limbo": "LIMBO_TOKEN",
|
||||
"build": "BUILD_TOKEN",
|
||||
"ops": "OPS_TOKEN",
|
||||
}
|
||||
|
||||
type tokenRotator struct {
|
||||
cl client.Client
|
||||
controlNS string
|
||||
minecraftNS string
|
||||
buildNS string
|
||||
// secretsEnv and linkProps are the installer's record and the proxy's
|
||||
// felis-link.properties; a missing file is reported and skipped.
|
||||
secretsEnv string
|
||||
linkProps string
|
||||
newToken func() (string, error)
|
||||
// rollAPI restarts felis-api and waits for the rollout.
|
||||
rollAPI func(ctx context.Context) error
|
||||
// restartUnit restarts a systemd unit on this host.
|
||||
restartUnit func(ctx context.Context, unit string) error
|
||||
out io.Writer
|
||||
}
|
||||
|
||||
func cmdRotateToken(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("rotate-token", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
|
||||
secretsEnv := fs.String("secrets-env", defaultSecretsEnvPath, "the installer's secrets file, updated so a re-run keeps the new value")
|
||||
linkProps := fs.String("link-properties", defaultLinkPropsPath, "the host proxy's felis-link.properties (velocity only)")
|
||||
fs.Usage = func() {
|
||||
fmt.Fprintf(stderr, "Usage: felis rotate-token [flags] <%s>\n\n", strings.Join(callerNames(), "|"))
|
||||
fmt.Fprintln(stderr, "Replaces one internal caller's token: the Secrets, felis-api, then the caller itself.")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if fs.NArg() != 1 {
|
||||
fs.Usage()
|
||||
return 2
|
||||
}
|
||||
if _, ok := callerToken(fs.Arg(0)); !ok {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: unknown caller %q (one of %s)\n", fs.Arg(0), strings.Join(callerNames(), ", "))
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis rotate-token: refused — rotating writes the cluster Secrets and the installer's secrets file, so it must run as root (try: sudo felis rotate-token "+fs.Arg(0)+")")
|
||||
return 1
|
||||
}
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
buildNS := cfg.Registry.BuildNamespace
|
||||
if buildNS == "" {
|
||||
buildNS = platform.DefaultBuildNamespace
|
||||
}
|
||||
r := tokenRotator{
|
||||
cl: cl,
|
||||
controlNS: platform.DefaultControlNamespace,
|
||||
minecraftNS: cfg.K8s.Namespace,
|
||||
buildNS: buildNS,
|
||||
secretsEnv: *secretsEnv,
|
||||
linkProps: *linkProps,
|
||||
newToken: randomToken,
|
||||
rollAPI: func(ctx context.Context) error {
|
||||
if err := kubectl(ctx, "-n", platform.DefaultControlNamespace, "rollout", "restart", "deployment/felis-api"); err != nil {
|
||||
return err
|
||||
}
|
||||
return kubectl(ctx, "-n", platform.DefaultControlNamespace, "rollout", "status", "deployment/felis-api", "--timeout=180s")
|
||||
},
|
||||
restartUnit: func(ctx context.Context, unit string) error { return systemctl(ctx, "restart", unit) },
|
||||
out: stdout,
|
||||
}
|
||||
if err := r.rotate(context.Background(), fs.Arg(0)); err != nil {
|
||||
fmt.Fprintf(stderr, "felis rotate-token: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func callerNames() []string {
|
||||
names := make([]string, 0, len(naming.CallerTokens))
|
||||
for _, ct := range naming.CallerTokens {
|
||||
names = append(names, ct.Caller)
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
func callerToken(name string) (naming.CallerToken, bool) {
|
||||
for _, ct := range naming.CallerTokens {
|
||||
if ct.Caller == name {
|
||||
return ct, true
|
||||
}
|
||||
}
|
||||
return naming.CallerToken{}, false
|
||||
}
|
||||
|
||||
// randomToken is 32 random bytes in hex, the shape the installer generates.
|
||||
func randomToken() (string, error) {
|
||||
b := make([]byte, 32)
|
||||
if _, err := rand.Read(b); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return hex.EncodeToString(b), nil
|
||||
}
|
||||
|
||||
func (r tokenRotator) rotate(ctx context.Context, caller string) error {
|
||||
ct, ok := callerToken(caller)
|
||||
if !ok {
|
||||
return fmt.Errorf("unknown caller %q", caller)
|
||||
}
|
||||
tok, err := r.newToken()
|
||||
if err != nil {
|
||||
return fmt.Errorf("generate a token: %w", err)
|
||||
}
|
||||
|
||||
// The installer's record first: from here on, whatever fails, a re-run of the
|
||||
// installer puts the new value everywhere.
|
||||
switch err := setKeyValueLine(r.secretsEnv, installerTokenKeys[ct.Caller], "=", tok); {
|
||||
case errors.Is(err, fs.ErrNotExist):
|
||||
fmt.Fprintf(r.out, " - %s: not found, skipped (this host was not installed by deploy/bootstrap.sh)\n", r.secretsEnv)
|
||||
case err != nil:
|
||||
return fmt.Errorf("record the new token in %s: %w", r.secretsEnv, err)
|
||||
default:
|
||||
fmt.Fprintf(r.out, " - %s: %s updated\n", r.secretsEnv, installerTokenKeys[ct.Caller])
|
||||
}
|
||||
|
||||
namespaces := []string{r.controlNS}
|
||||
replica := map[string]string{"minecraft": r.minecraftNS, "build": r.buildNS}[ct.Replica]
|
||||
if replica != "" && replica != r.controlNS {
|
||||
namespaces = append(namespaces, replica)
|
||||
}
|
||||
for _, ns := range namespaces {
|
||||
if err := writeTokenSecret(ctx, r.cl, ns, ct.Secret, tok); err != nil {
|
||||
return fmt.Errorf("write Secret %s/%s: %w", ns, ct.Secret, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - Secret %s/%s: updated\n", ns, ct.Secret)
|
||||
}
|
||||
|
||||
hostProxy := false
|
||||
if ct.Caller == "velocity" {
|
||||
switch err := setKeyValueLine(r.linkProps, "service-token", "=", tok); {
|
||||
case errors.Is(err, fs.ErrNotExist):
|
||||
fmt.Fprintf(r.out, " - %s: not found; set service-token in your proxy's felis-link.properties to the value in Secret %s/%s and restart it\n",
|
||||
r.linkProps, r.controlNS, ct.Secret)
|
||||
case err != nil:
|
||||
return fmt.Errorf("write the proxy's token into %s: %w", r.linkProps, err)
|
||||
default:
|
||||
hostProxy = true
|
||||
fmt.Fprintf(r.out, " - %s: service-token updated\n", r.linkProps)
|
||||
}
|
||||
}
|
||||
|
||||
if err := r.rollAPI(ctx); err != nil {
|
||||
return fmt.Errorf("roll felis-api: %w", err)
|
||||
}
|
||||
fmt.Fprintln(r.out, " - felis-api: rolled out, accepting only the new token")
|
||||
|
||||
switch ct.Caller {
|
||||
case "velocity":
|
||||
if hostProxy {
|
||||
if err := r.restartUnit(ctx, velocityUnit); err != nil {
|
||||
return fmt.Errorf("restart %s: %w", velocityUnit, err)
|
||||
}
|
||||
fmt.Fprintf(r.out, " - %s: restarted (players on the proxy were disconnected and can rejoin)\n", velocityUnit)
|
||||
}
|
||||
case "limbo":
|
||||
if err := r.cl.DeleteAllOf(ctx, &corev1.Pod{}, client.InNamespace(r.minecraftNS),
|
||||
client.MatchingLabels{v1alpha1.LabelServer: naming.SystemLoginServer}); err != nil {
|
||||
return fmt.Errorf("restart the login gate: %w", err)
|
||||
}
|
||||
fmt.Fprintln(r.out, " - login gate: pod restarted to read the new token")
|
||||
case "build":
|
||||
fmt.Fprintln(r.out, " - builds: the next build Job reads the new token; one fetching its context right now fails and can be submitted again")
|
||||
case "ops":
|
||||
fmt.Fprintln(r.out, " - felis backup-now reads the new token on its next run")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// writeTokenSecret sets the token in a Secret, creating it when absent.
|
||||
func writeTokenSecret(ctx context.Context, cl client.Client, namespace, name, token string) error {
|
||||
var sec corev1.Secret
|
||||
err := cl.Get(ctx, client.ObjectKey{Namespace: namespace, Name: name}, &sec)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return cl.Create(ctx, &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: namespace, Name: name},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
Data: map[string][]byte{naming.ServiceTokenSecretKey: []byte(token)},
|
||||
})
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if sec.Data == nil {
|
||||
sec.Data = map[string][]byte{}
|
||||
}
|
||||
sec.Data[naming.ServiceTokenSecretKey] = []byte(token)
|
||||
return cl.Update(ctx, &sec)
|
||||
}
|
||||
|
||||
// setKeyValueLine rewrites the `key<sep>value` line of a flat key/value file
|
||||
// (secrets.env, a .properties file), appending one when the key is absent. The
|
||||
// file is replaced atomically and keeps its mode and owner: felis-link.properties
|
||||
// is root:felis-velocity 0640, and the proxy must still be able to read it.
|
||||
func setKeyValueLine(path, key, sep, value string) error {
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
lines := strings.Split(strings.TrimRight(string(raw), "\n"), "\n")
|
||||
found := false
|
||||
for i, ln := range lines {
|
||||
k, _, ok := strings.Cut(ln, sep)
|
||||
if ok && strings.TrimSpace(k) == key {
|
||||
lines[i] = key + sep + value
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
lines = append(lines, key+sep+value)
|
||||
}
|
||||
tmp, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.Remove(tmp.Name())
|
||||
if err := tmp.Chmod(info.Mode().Perm()); err != nil {
|
||||
tmp.Close()
|
||||
return err
|
||||
}
|
||||
if st, ok := info.Sys().(*syscall.Stat_t); ok {
|
||||
if err := tmp.Chown(int(st.Uid), int(st.Gid)); err != nil {
|
||||
tmp.Close()
|
||||
return err
|
||||
}
|
||||
}
|
||||
if _, err := tmp.WriteString(strings.Join(lines, "\n") + "\n"); err != nil {
|
||||
tmp.Close()
|
||||
return err
|
||||
}
|
||||
if err := tmp.Sync(); err != nil {
|
||||
tmp.Close()
|
||||
return err
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.Rename(tmp.Name(), path)
|
||||
}
|
||||
@@ -0,0 +1,283 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
type rotationRig struct {
|
||||
r tokenRotator
|
||||
cl client.Client
|
||||
out *bytes.Buffer
|
||||
events []string
|
||||
dir string
|
||||
}
|
||||
|
||||
func tokenSecret(ns, name, val string) *corev1.Secret {
|
||||
return &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name},
|
||||
Data: map[string][]byte{"token": []byte(val)},
|
||||
}
|
||||
}
|
||||
|
||||
func serverPod(ns, name, server string) *corev1.Pod {
|
||||
return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name,
|
||||
Labels: map[string]string{"felis.lolicon.best/server": server}}}
|
||||
}
|
||||
|
||||
func newRotationRig(t *testing.T, objs ...client.Object) *rotationRig {
|
||||
t.Helper()
|
||||
rig := &rotationRig{out: &bytes.Buffer{}, dir: t.TempDir()}
|
||||
rig.cl = fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(objs...).Build()
|
||||
rig.r = tokenRotator{
|
||||
cl: rig.cl,
|
||||
controlNS: "felis",
|
||||
minecraftNS: "minecraft",
|
||||
buildNS: "felis-build",
|
||||
secretsEnv: filepath.Join(rig.dir, "secrets.env"),
|
||||
linkProps: filepath.Join(rig.dir, "felis-link.properties"),
|
||||
newToken: func() (string, error) { return "NEWTOKEN", nil },
|
||||
rollAPI: func(context.Context) error {
|
||||
rig.events = append(rig.events, "roll-api")
|
||||
return nil
|
||||
},
|
||||
restartUnit: func(_ context.Context, unit string) error {
|
||||
rig.events = append(rig.events, "restart "+unit)
|
||||
return nil
|
||||
},
|
||||
out: rig.out,
|
||||
}
|
||||
return rig
|
||||
}
|
||||
|
||||
func (rig *rotationRig) secret(t *testing.T, ns, name string) string {
|
||||
t.Helper()
|
||||
var s corev1.Secret
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
|
||||
return "<missing>"
|
||||
}
|
||||
return string(s.Data["token"])
|
||||
}
|
||||
|
||||
func writeTestFile(t *testing.T, path, body string, mode os.FileMode) {
|
||||
t.Helper()
|
||||
if err := os.WriteFile(path, []byte(body), mode); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.Chmod(path, mode); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateLimboToken(t *testing.T) {
|
||||
rig := newRotationRig(t,
|
||||
tokenSecret("felis", "felis-limbo-token", "old"),
|
||||
tokenSecret("minecraft", "felis-limbo-token", "old"),
|
||||
tokenSecret("felis", "felis-service-token", "proxy"),
|
||||
serverPod("minecraft", "login-0", "login"),
|
||||
serverPod("minecraft", "survival-0", "survival"),
|
||||
)
|
||||
writeTestFile(t, rig.r.secretsEnv, "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=old\nOPS_TOKEN=ops\n", 0o600)
|
||||
|
||||
// felis-api must roll only after both copies hold the new value, and the login
|
||||
// pod must still be there then: restarting it earlier would have it present
|
||||
// the new token to an api that does not know it yet.
|
||||
rig.r.rollAPI = func(context.Context) error {
|
||||
rig.events = append(rig.events, "roll-api")
|
||||
if got := rig.secret(t, "felis", "felis-limbo-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("api rolled while the control Secret held %q", got)
|
||||
}
|
||||
if got := rig.secret(t, "minecraft", "felis-limbo-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("api rolled while the minecraft replica held %q", got)
|
||||
}
|
||||
var pod corev1.Pod
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "login-0"}, &pod); err != nil {
|
||||
t.Errorf("the login pod was restarted before the api rolled")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := rig.r.rotate(context.Background(), "limbo"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
raw, _ := os.ReadFile(rig.r.secretsEnv)
|
||||
if string(raw) != "DB_PASSWORD=db\nSERVICE_TOKEN=proxy\nLIMBO_TOKEN=NEWTOKEN\nOPS_TOKEN=ops\n" {
|
||||
t.Errorf("secrets.env = %q", raw)
|
||||
}
|
||||
if info, _ := os.Stat(rig.r.secretsEnv); info.Mode().Perm() != 0o600 {
|
||||
t.Errorf("secrets.env mode = %v, want 0600", info.Mode().Perm())
|
||||
}
|
||||
if got := rig.secret(t, "felis", "felis-service-token"); got != "proxy" {
|
||||
t.Errorf("the proxy's token changed to %q", got)
|
||||
}
|
||||
var pods corev1.PodList
|
||||
if err := rig.cl.List(context.Background(), &pods, client.InNamespace("minecraft")); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(pods.Items) != 1 || pods.Items[0].Name != "survival-0" {
|
||||
t.Errorf("pods left = %v, want only survival-0 (the login pod restarted, user servers untouched)", pods.Items)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api" {
|
||||
t.Errorf("events = %v, want only the api roll (no unit restart for limbo)", rig.events)
|
||||
}
|
||||
if strings.Contains(rig.out.String(), "NEWTOKEN") {
|
||||
t.Errorf("the new token was printed: %s", rig.out.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateVelocityTokenOnTheHostProxy(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=old\n", 0o600)
|
||||
writeTestFile(t, rig.r.linkProps, "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=old\nroot-domain=example.com\n", 0o640)
|
||||
|
||||
if err := rig.r.rotate(context.Background(), "velocity"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, _ := os.ReadFile(rig.r.linkProps)
|
||||
if string(raw) != "# Generated\napi-base-url=http://10.0.0.1:8081\nservice-token=NEWTOKEN\nroot-domain=example.com\n" {
|
||||
t.Errorf("felis-link.properties = %q", raw)
|
||||
}
|
||||
if info, _ := os.Stat(rig.r.linkProps); info.Mode().Perm() != 0o640 {
|
||||
t.Errorf("properties mode = %v, want 0640 (the proxy's group must still read it)", info.Mode().Perm())
|
||||
}
|
||||
if got := rig.secret(t, "felis", "felis-service-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
// The proxy's token has no replica: it must not appear in a workload namespace.
|
||||
if got := rig.secret(t, "minecraft", "felis-service-token"); got != "<missing>" {
|
||||
t.Errorf("rotation copied the proxy token into minecraft (%q)", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api,restart felis-velocity" {
|
||||
t.Errorf("events = %v, want the api roll then the proxy restart", rig.events)
|
||||
}
|
||||
}
|
||||
|
||||
// An external proxy has no felis-link.properties here: the Secret still rotates,
|
||||
// nothing is restarted on this host, and the output says where the value is
|
||||
// without printing it.
|
||||
func TestRotateVelocityTokenForAnExternalProxy(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-service-token", "old"))
|
||||
if err := rig.r.rotate(context.Background(), "velocity"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := rig.secret(t, "felis", "felis-service-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
if strings.Join(rig.events, ",") != "roll-api" {
|
||||
t.Errorf("events = %v, want no proxy restart", rig.events)
|
||||
}
|
||||
out := rig.out.String()
|
||||
if !strings.Contains(out, "felis/felis-service-token") || strings.Contains(out, "NEWTOKEN") {
|
||||
t.Errorf("output should point at the Secret without the value: %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateBuildTokenReachesTheBuildNamespace(t *testing.T) {
|
||||
rig := newRotationRig(t, tokenSecret("felis", "felis-build-token", "old"))
|
||||
// An install from before per-caller tokens has no BUILD_TOKEN line yet.
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=proxy", 0o600)
|
||||
if err := rig.r.rotate(context.Background(), "build"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := rig.secret(t, "felis-build", "felis-build-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("build replica = %q, want NEWTOKEN (created when absent)", got)
|
||||
}
|
||||
if got := rig.secret(t, "felis", "felis-build-token"); got != "NEWTOKEN" {
|
||||
t.Errorf("control Secret = %q, want NEWTOKEN", got)
|
||||
}
|
||||
raw, _ := os.ReadFile(rig.r.secretsEnv)
|
||||
if string(raw) != "SERVICE_TOKEN=proxy\nBUILD_TOKEN=NEWTOKEN\n" {
|
||||
t.Errorf("secrets.env = %q", raw)
|
||||
}
|
||||
}
|
||||
|
||||
// A failed api rollout stops the rotation before the caller restarts: the login
|
||||
// gate keeps running on the old value the old api pods still accept.
|
||||
func TestRotateStopsWhenTheAPIDoesNotRoll(t *testing.T) {
|
||||
rig := newRotationRig(t,
|
||||
tokenSecret("felis", "felis-limbo-token", "old"),
|
||||
serverPod("minecraft", "login-0", "login"),
|
||||
)
|
||||
rig.r.rollAPI = func(context.Context) error { return errors.New("rollout timed out") }
|
||||
err := rig.r.rotate(context.Background(), "limbo")
|
||||
if err == nil || !strings.Contains(err.Error(), "rollout timed out") {
|
||||
t.Fatalf("err = %v, want the rollout failure", err)
|
||||
}
|
||||
var pod corev1.Pod
|
||||
if err := rig.cl.Get(context.Background(), client.ObjectKey{Namespace: "minecraft", Name: "login-0"}, &pod); err != nil {
|
||||
t.Error("the login pod was restarted although the api never rolled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRotateRefusesAnUnknownCaller(t *testing.T) {
|
||||
rig := newRotationRig(t)
|
||||
writeTestFile(t, rig.r.secretsEnv, "SERVICE_TOKEN=proxy\n", 0o600)
|
||||
if err := rig.r.rotate(context.Background(), "admin"); err == nil {
|
||||
t.Fatal("rotated a token for an unknown caller")
|
||||
}
|
||||
if raw, _ := os.ReadFile(rig.r.secretsEnv); string(raw) != "SERVICE_TOKEN=proxy\n" {
|
||||
t.Errorf("secrets.env = %q, want it untouched", raw)
|
||||
}
|
||||
if len(rig.events) != 0 {
|
||||
t.Errorf("events = %v, want nothing touched", rig.events)
|
||||
}
|
||||
}
|
||||
|
||||
// The installer and rotate-token must agree on where each caller's token lives:
|
||||
// rotate-token writes installerTokenKeys into secrets.env, and the installer
|
||||
// applies those same keys to the Secrets on its next run. A key the installer
|
||||
// does not read would be silently reverted by the next upgrade.
|
||||
func TestInstallerProvisionsEveryCallerToken(t *testing.T) {
|
||||
raw, err := os.ReadFile(filepath.Join("..", "..", "deploy", "bootstrap.sh"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
script := string(raw)
|
||||
for caller, key := range installerTokenKeys {
|
||||
ct, ok := callerToken(caller)
|
||||
if !ok {
|
||||
t.Fatalf("installerTokenKeys names unknown caller %q", caller)
|
||||
}
|
||||
if !strings.Contains(script, key+`="${`+key+`:-$(openssl rand -hex 32)}"`) {
|
||||
t.Errorf("bootstrap.sh does not generate %s", key)
|
||||
}
|
||||
if !strings.Contains(script, key+"=${"+key+"}\n") {
|
||||
t.Errorf("bootstrap.sh does not persist %s to secrets.env", key)
|
||||
}
|
||||
apply := regexp.MustCompile(`apply_literal_secret "\$CONTROL_NS" ` + regexp.QuoteMeta(ct.Secret) + ` token "\$` + key + `"`)
|
||||
if !apply.MatchString(script) {
|
||||
t.Errorf("bootstrap.sh does not apply %s from %s in the control namespace", ct.Secret, key)
|
||||
}
|
||||
}
|
||||
// The replicas the installer applies straight into the workload namespaces, so an
|
||||
// upgrade has them before the new operator and build Jobs reference them.
|
||||
for _, line := range []string{
|
||||
`apply_literal_secret "$MINECRAFT_NS" felis-limbo-token token "$LIMBO_TOKEN"`,
|
||||
`apply_literal_secret "$BUILD_NS" felis-build-token token "$BUILD_TOKEN"`,
|
||||
`kube -n "$MINECRAFT_NS" delete secret felis-service-token --ignore-not-found`,
|
||||
`kube -n "$BUILD_NS" delete secret felis-service-token --ignore-not-found`,
|
||||
} {
|
||||
if !strings.Contains(script, line) {
|
||||
t.Errorf("bootstrap.sh lacks %s", line)
|
||||
}
|
||||
}
|
||||
for _, ns := range []string{"MINECRAFT_NS", "BUILD_NS"} {
|
||||
if strings.Contains(script, `apply_literal_secret "$`+ns+`" felis-service-token`) {
|
||||
t.Errorf("bootstrap.sh still copies the proxy's token into %s", ns)
|
||||
}
|
||||
}
|
||||
if len(installerTokenKeys) != len(callerNames()) {
|
||||
t.Errorf("installerTokenKeys covers %d callers, naming.CallerTokens lists %d", len(installerTokenKeys), len(callerNames()))
|
||||
}
|
||||
}
|
||||
+5
-1
@@ -13,7 +13,7 @@ Usage:
|
||||
Commands:
|
||||
migrate up Apply embedded database migrations under an advisory lock (snapshots the database first)
|
||||
db Back up, verify, list and restore the control-plane database (backup|restore|verify|list|check)
|
||||
offsite Copy world archives and database bundles to an off-site bucket, and fetch them back (sync|status|list|fetch-db|fetch-worlds|keygen)
|
||||
offsite Copy world archives, database bundles, user images and uploads to an off-site bucket, and fetch them back (sync|status|list|fetch-db|fetch-worlds|fetch-images|fetch-uploads|keygen)
|
||||
operator Run the MinecraftServer controller-manager
|
||||
api Run the felis-api HTTP server
|
||||
nano Run the Felis-nano hasJoined multiplexer (multi-Yggdrasil, no control plane)
|
||||
@@ -23,6 +23,7 @@ Commands:
|
||||
files List/read/write one file in a stopped server's world (internal Job entrypoint)
|
||||
egress-gate Hold a build pod until its egress NetworkPolicy is enforced (internal Job entrypoint)
|
||||
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
|
||||
scan-gate Apply the scan policy to a build's Trivy report and hand felis-api the report and SBOM (internal Job entrypoint)
|
||||
push-image Push a scanned image tarball to the registry (internal Job entrypoint)
|
||||
mirror-build-tools Copy kaniko, trivy and Trivy's DBs into the registry (run by felis-build-tools.timer)
|
||||
registry-gate Authorize registry writes in front of registry:2 (internal sidecar entrypoint)
|
||||
@@ -30,6 +31,7 @@ Commands:
|
||||
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
||||
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
||||
converge Fill in fields a newer desired spec added to already-installed system servers
|
||||
rotate-token Replace one internal caller's token and restart what holds it (velocity|limbo|build|ops; requires root/sudo)
|
||||
watchdog Check the platform once and mail the owners what has gone wrong (run by felis-watchdog.timer)
|
||||
version Print the build stamp of this binary
|
||||
update Report which platform components have updates available
|
||||
@@ -60,6 +62,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||
"files": cmdFiles,
|
||||
"egress-gate": cmdEgressGate,
|
||||
"fetch-context": cmdFetchContext,
|
||||
"scan-gate": cmdScanGate,
|
||||
"push-image": cmdPushImage,
|
||||
"mirror-build-tools": cmdMirrorBuildTools,
|
||||
"registry-gate": cmdRegistryGate,
|
||||
@@ -67,6 +70,7 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||
"apply": cmdApply,
|
||||
"setup": cmdSetup,
|
||||
"converge": cmdConverge,
|
||||
"rotate-token": cmdRotateToken,
|
||||
"breakGlass": cmdBreakGlass,
|
||||
"bootstrap-assets": cmdBootstrapAssets,
|
||||
"init-forwarding": cmdInitForwarding,
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"felis.lolicon.best/internal/build"
|
||||
)
|
||||
|
||||
// maxScanDocument bounds each document scan-gate reads: a modpack report lists a
|
||||
// few thousand packages, far below this. Tests shrink it.
|
||||
var maxScanDocument int64 = 64 << 20
|
||||
|
||||
// cmdScanGate is the build pod's verdict step, after trivy wrote its full JSON
|
||||
// report and trivy convert wrote the CycloneDX SBOM. It applies the scan policy
|
||||
// to the report, prints the verdict and every blocking finding, then appends the
|
||||
// verdict, the report and the SBOM to its log as the envelope felis-api keeps on
|
||||
// the build (build.WriteScanEnvelope). It exits 1 when a finding blocks, which
|
||||
// fails the pod before the push step runs, and 2 when the report cannot be read,
|
||||
// so a scan that produced nothing usable never admits an image.
|
||||
func cmdScanGate(args []string, stdout, stderr io.Writer) int {
|
||||
fset := flag.NewFlagSet("scan-gate", flag.ContinueOnError)
|
||||
fset.SetOutput(stderr)
|
||||
reportPath := fset.String("report", "", "trivy JSON report (required)")
|
||||
sbomPath := fset.String("sbom", "", "CycloneDX SBOM to keep with the report")
|
||||
failOn := fset.String("fail-on", strings.Join(build.DefaultScanFailOn, ","), "comma-separated severities that block the image")
|
||||
failUnfixed := fset.Bool("fail-unfixed", false, "block on vulnerabilities that have no fixed release too")
|
||||
accept := fset.String("accept", "", "comma-separated vulnerability ids and secret rule ids that never block")
|
||||
termLog := fset.String("termination-log", "/dev/termination-log", "where the one-line verdict goes for the pod status")
|
||||
if err := fset.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
// A stray argument is a policy the gate would otherwise drop without a word
|
||||
// (a comma split out of --fail-on, say).
|
||||
if fset.NArg() > 0 {
|
||||
fmt.Fprintf(stderr, "felis scan-gate: unexpected argument %q\n", fset.Arg(0))
|
||||
return 2
|
||||
}
|
||||
sevs, err := build.ParseSeverities(*failOn)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis scan-gate: --fail-on: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
accepted, err := build.ParseScanAccept(*accept)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis scan-gate: --accept: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
if *reportPath == "" {
|
||||
fmt.Fprintln(stderr, "felis scan-gate: --report is required")
|
||||
return 2
|
||||
}
|
||||
fail := func(msg string) int {
|
||||
fmt.Fprintln(stderr, "felis scan-gate: "+msg)
|
||||
writeTerminationLog(*termLog, msg)
|
||||
return 2
|
||||
}
|
||||
report, err := readScanDocument(*reportPath)
|
||||
if err != nil {
|
||||
return fail("the scan report is unreadable: " + err.Error())
|
||||
}
|
||||
policy := build.ScanPolicy{FailOn: sevs, FailUnfixed: *failUnfixed, Accept: accepted}
|
||||
summary, err := build.Summarize(report, policy)
|
||||
if err != nil {
|
||||
return fail("the scan report is unreadable: " + err.Error())
|
||||
}
|
||||
env := build.ScanEnvelope{Summary: summary, Report: report}
|
||||
if *sbomPath != "" {
|
||||
switch sbom, err := readScanDocument(*sbomPath); {
|
||||
case err == nil:
|
||||
env.SBOM = sbom
|
||||
case errors.Is(err, fs.ErrNotExist):
|
||||
fmt.Fprintf(stdout, "felis scan-gate: no SBOM at %s; keeping the report alone\n", *sbomPath)
|
||||
default:
|
||||
return fail("the SBOM is unreadable: " + err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
printScanVerdict(stdout, summary)
|
||||
written, err := build.WriteScanEnvelope(stdout, env)
|
||||
if err != nil {
|
||||
return fail("could not write the scan envelope: " + err.Error())
|
||||
}
|
||||
for _, doc := range written.Summary.Omitted {
|
||||
fmt.Fprintf(stderr, "felis scan-gate: the %s is too large to keep with the build and was left out\n", doc)
|
||||
}
|
||||
if summary.Blocked {
|
||||
writeTerminationLog(*termLog, summary.Reason())
|
||||
return 1
|
||||
}
|
||||
writeTerminationLog(*termLog, "the scan passed")
|
||||
return 0
|
||||
}
|
||||
|
||||
// printScanVerdict writes the human half of scan-gate's log.
|
||||
func printScanVerdict(w io.Writer, s build.ScanSummary) {
|
||||
var counts []string
|
||||
for _, sev := range build.Severities {
|
||||
counts = append(counts, fmt.Sprintf("%s %d", sev, s.Counts[sev]))
|
||||
}
|
||||
fmt.Fprintf(w, "felis scan-gate: %d packages; findings: %s\n", s.Packages, strings.Join(counts, ", "))
|
||||
unfixed := "vulnerabilities with no fixed release do not block"
|
||||
if s.Policy.FailUnfixed {
|
||||
unfixed = "vulnerabilities with no fixed release block too"
|
||||
}
|
||||
fmt.Fprintf(w, "felis scan-gate: blocking on %s (%s)\n", strings.Join(s.Policy.FailOn, ", "), unfixed)
|
||||
if len(s.Policy.Accept) > 0 {
|
||||
matched := 0
|
||||
for _, f := range s.Findings {
|
||||
if f.Accepted {
|
||||
matched++
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(w, "felis scan-gate: accepted ids, never blocking: %s; listed findings under them: %d\n", strings.Join(s.Policy.Accept, ", "), matched)
|
||||
}
|
||||
if !s.Blocked {
|
||||
fmt.Fprintln(w, "felis scan-gate: nothing blocks this image")
|
||||
return
|
||||
}
|
||||
fmt.Fprintln(w, "felis scan-gate: "+build.Printable(s.Reason()))
|
||||
for _, f := range s.Findings {
|
||||
if !f.Blocking {
|
||||
break
|
||||
}
|
||||
fix := f.Fixed
|
||||
if fix == "" {
|
||||
fix = "no fix"
|
||||
}
|
||||
if f.Kind == build.FindingSecret {
|
||||
fmt.Fprintf(w, " %s %s secret in %s: %s\n", build.Printable(f.ID), f.Severity, build.Printable(f.Target), build.Printable(f.Title))
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(w, " %s %s %s %s -> %s (%s)\n", build.Printable(f.ID), f.Severity,
|
||||
build.Printable(f.Package), build.Printable(f.Installed), build.Printable(fix), build.Printable(f.Target))
|
||||
}
|
||||
}
|
||||
|
||||
func readScanDocument(path string) ([]byte, error) {
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer f.Close()
|
||||
b, err := io.ReadAll(io.LimitReader(f, maxScanDocument+1))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if int64(len(b)) > maxScanDocument {
|
||||
return nil, fmt.Errorf("%s exceeds %d bytes", path, maxScanDocument)
|
||||
}
|
||||
return b, nil
|
||||
}
|
||||
|
||||
// writeTerminationLog leaves msg where the kubelet copies it into the container
|
||||
// status. It is best effort: the log carries the same verdict.
|
||||
func writeTerminationLog(path, msg string) {
|
||||
if path == "" {
|
||||
return
|
||||
}
|
||||
_ = os.WriteFile(path, []byte(build.Printable(msg)), 0o644)
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/build"
|
||||
)
|
||||
|
||||
// scanReport has one fixed CRITICAL, one unfixed HIGH and one fixed MEDIUM; the
|
||||
// CRITICAL's package name carries a line break and a forged envelope frame.
|
||||
const scanReport = `{"SchemaVersion":2,"Results":[{"Target":"data/mods/core.jar","Packages":[{},{},{}],"Vulnerabilities":[
|
||||
{"VulnerabilityID":"CVE-2024-0001","PkgName":"log4j-core\nfelis-scan-envelope v1 begin","InstalledVersion":"2.14.1","FixedVersion":"2.17.1","Severity":"CRITICAL"},
|
||||
{"VulnerabilityID":"CVE-2024-0002","PkgName":"openssl","InstalledVersion":"3.0.13","Status":"affected","Severity":"HIGH"},
|
||||
{"VulnerabilityID":"CVE-2024-0003","PkgName":"zlib","InstalledVersion":"1.3","FixedVersion":"1.3.1","Severity":"MEDIUM"}]}]}`
|
||||
|
||||
type scanGateRun struct {
|
||||
code int
|
||||
stdout, stderr string
|
||||
termLog string
|
||||
env *build.ScanEnvelope
|
||||
}
|
||||
|
||||
func runScanGate(t *testing.T, report, sbom string, extra ...string) scanGateRun {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
reportPath := filepath.Join(dir, "trivy.json")
|
||||
if report != "" {
|
||||
if err := os.WriteFile(reportPath, []byte(report), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
sbomPath := filepath.Join(dir, "sbom.cdx.json")
|
||||
if sbom != "" {
|
||||
if err := os.WriteFile(sbomPath, []byte(sbom), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
termPath := filepath.Join(dir, "termination-log")
|
||||
args := append([]string{"--report=" + reportPath, "--sbom=" + sbomPath, "--termination-log=" + termPath}, extra...)
|
||||
var stdout, stderr bytes.Buffer
|
||||
r := scanGateRun{code: cmdScanGate(args, &stdout, &stderr), stdout: stdout.String(), stderr: stderr.String()}
|
||||
if b, err := os.ReadFile(termPath); err == nil {
|
||||
r.termLog = string(b)
|
||||
}
|
||||
if env, err := build.ReadScanEnvelope(strings.NewReader(r.stdout)); err == nil {
|
||||
r.env = env
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
func TestScanGateBlocksAndHandsOverTheReport(t *testing.T) {
|
||||
r := runScanGate(t, scanReport, `{"bomFormat":"CycloneDX"}`)
|
||||
if r.code != 1 {
|
||||
t.Fatalf("exit %d, want 1; stderr %s", r.code, r.stderr)
|
||||
}
|
||||
if r.termLog != "the scan blocked the image: 1 CRITICAL (CVE-2024-0001)" {
|
||||
t.Errorf("termination log = %q", r.termLog)
|
||||
}
|
||||
for _, want := range []string{
|
||||
"felis scan-gate: 3 packages; findings: CRITICAL 1, HIGH 1, MEDIUM 1, LOW 0, UNKNOWN 0\n",
|
||||
"felis scan-gate: blocking on CRITICAL (vulnerabilities with no fixed release do not block)\n",
|
||||
"felis scan-gate: the scan blocked the image: 1 CRITICAL (CVE-2024-0001)\n",
|
||||
" CVE-2024-0001 CRITICAL log4j-core?felis-scan-envelope v1 begin 2.14.1 -> 2.17.1 (data/mods/core.jar)\n",
|
||||
} {
|
||||
if !strings.Contains(r.stdout, want) {
|
||||
t.Errorf("stdout lacks %q:\n%s", want, r.stdout)
|
||||
}
|
||||
}
|
||||
if strings.Contains(r.stdout, " CVE-2024-0002") {
|
||||
t.Errorf("an unfixed HIGH was listed as blocking:\n%s", r.stdout)
|
||||
}
|
||||
if strings.Count(r.stdout, "\nfelis-scan-envelope v1 begin\n") != 1 {
|
||||
t.Errorf("stdout must hold exactly one frame start on a line of its own:\n%s", r.stdout)
|
||||
}
|
||||
if r.env == nil {
|
||||
t.Fatal("no envelope in stdout")
|
||||
}
|
||||
if !r.env.Summary.Blocked || string(r.env.SBOM) != `{"bomFormat":"CycloneDX"}` || !strings.Contains(string(r.env.Report), "CVE-2024-0003") {
|
||||
t.Errorf("envelope = blocked %t, sbom %s, report %d bytes", r.env.Summary.Blocked, r.env.SBOM, len(r.env.Report))
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanGatePolicyFlags(t *testing.T) {
|
||||
r := runScanGate(t, scanReport, "", "--fail-on=high", "--fail-unfixed")
|
||||
if r.code != 1 || r.termLog != "the scan blocked the image: 1 HIGH (CVE-2024-0002)" {
|
||||
t.Errorf("HIGH + unfixed: exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
for _, want := range []string{
|
||||
"felis scan-gate: blocking on HIGH (vulnerabilities with no fixed release block too)\n",
|
||||
" CVE-2024-0002 HIGH openssl 3.0.13 -> no fix (data/mods/core.jar)\n",
|
||||
} {
|
||||
if !strings.Contains(r.stdout, want) {
|
||||
t.Errorf("stdout lacks %q:\n%s", want, r.stdout)
|
||||
}
|
||||
}
|
||||
r = runScanGate(t, scanReport, `{"bomFormat":"CycloneDX"}`, "--fail-on=LOW")
|
||||
if r.code != 0 || r.termLog != "the scan passed" || !strings.Contains(r.stdout, "felis scan-gate: nothing blocks this image\n") {
|
||||
t.Errorf("LOW only: exit %d, termination log %q, stdout:\n%s", r.code, r.termLog, r.stdout)
|
||||
}
|
||||
if r.env == nil || r.env.Summary.Blocked || r.env.SBOM == nil {
|
||||
t.Errorf("a passing scan must still hand over its report and SBOM: %+v", r.env)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanGateAcceptedIDsNeverBlock(t *testing.T) {
|
||||
r := runScanGate(t, scanReport, "", "--fail-on=CRITICAL,MEDIUM", "--accept=CVE-2024-0001, CVE-2024-0002,CVE-2099-0001")
|
||||
if r.code != 1 || r.termLog != "the scan blocked the image: 1 MEDIUM (CVE-2024-0003)" {
|
||||
t.Errorf("exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
if !strings.Contains(r.stdout, "felis scan-gate: accepted ids, never blocking: CVE-2024-0001, CVE-2024-0002, CVE-2099-0001; listed findings under them: 2\n") {
|
||||
t.Errorf("stdout:\n%s", r.stdout)
|
||||
}
|
||||
if r.env == nil || strings.Join(r.env.Summary.Policy.Accept, ",") != "CVE-2024-0001,CVE-2024-0002,CVE-2099-0001" {
|
||||
t.Fatalf("envelope = %+v", r.env)
|
||||
}
|
||||
for _, f := range r.env.Summary.Findings {
|
||||
if f.ID == "CVE-2024-0001" && (f.Blocking || !f.Accepted) {
|
||||
t.Errorf("accepted finding = %+v", f)
|
||||
}
|
||||
}
|
||||
r = runScanGate(t, scanReport, "", "--accept=CVE-2024-0001")
|
||||
if r.code != 0 || r.termLog != "the scan passed" {
|
||||
t.Errorf("only an accepted CRITICAL: exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanGateWithoutAnSBOMKeepsTheReport(t *testing.T) {
|
||||
r := runScanGate(t, scanReport, "", "--fail-on=LOW")
|
||||
if r.code != 0 || !strings.Contains(r.stdout, "felis scan-gate: no SBOM at ") {
|
||||
t.Errorf("exit %d, stdout:\n%s", r.code, r.stdout)
|
||||
}
|
||||
if r.env == nil || r.env.SBOM != nil || r.env.Report == nil {
|
||||
t.Errorf("envelope = %+v", r.env)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanGateListsABlockingSecret(t *testing.T) {
|
||||
r := runScanGate(t, `{"SchemaVersion":2,"Results":[{"Target":"config/keys.txt","Secrets":[
|
||||
{"RuleID":"aws-access-key-id","Severity":"CRITICAL","Title":"AWS Access\nKey ID"}]}]}`, "")
|
||||
if r.code != 1 || r.termLog != "the scan blocked the image: 1 CRITICAL (aws-access-key-id)" {
|
||||
t.Errorf("exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
if !strings.Contains(r.stdout, " aws-access-key-id CRITICAL secret in config/keys.txt: AWS Access?Key ID\n") {
|
||||
t.Errorf("stdout:\n%s", r.stdout)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScanGateFailsClosed(t *testing.T) {
|
||||
r := runScanGate(t, "", "")
|
||||
if r.code != 2 || !strings.HasPrefix(r.termLog, "the scan report is unreadable: open ") || r.env != nil {
|
||||
t.Errorf("missing report: exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
r = runScanGate(t, `{"SchemaVersion":1}`, "")
|
||||
if r.code != 2 || r.termLog != "the scan report is unreadable: read trivy report: schema version 1, want 2" {
|
||||
t.Errorf("old schema: exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
// An SBOM step that exited 0 but left something unreadable fails the gate; the
|
||||
// line break in the path stays out of the one-line termination message.
|
||||
sbomDir := filepath.Join(t.TempDir(), "sb\nom")
|
||||
if err := os.Mkdir(sbomDir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
r = runScanGate(t, scanReport, "", "--sbom="+sbomDir)
|
||||
if r.code != 2 || !strings.HasPrefix(r.termLog, "the SBOM is unreadable: read ") ||
|
||||
!strings.HasSuffix(r.termLog, "/sb?om: is a directory") || r.env != nil {
|
||||
t.Errorf("unreadable SBOM: exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
defer func(n int64) { maxScanDocument = n }(maxScanDocument)
|
||||
maxScanDocument = 64
|
||||
r = runScanGate(t, scanReport, "")
|
||||
if r.code != 2 || !strings.HasSuffix(r.termLog, "/trivy.json exceeds 64 bytes") || r.env != nil {
|
||||
t.Errorf("oversized report: exit %d, termination log %q", r.code, r.termLog)
|
||||
}
|
||||
maxScanDocument = 64 << 20
|
||||
r = runScanGate(t, scanReport, "", "--fail-on=SEVERE")
|
||||
if r.code != 2 || !strings.Contains(r.stderr, `felis scan-gate: --fail-on: unknown severity "SEVERE"`) {
|
||||
t.Errorf("bad severity: exit %d, stderr %q", r.code, r.stderr)
|
||||
}
|
||||
// A severity list split at its comma leaves a stray argument, not a
|
||||
// narrower policy.
|
||||
r = runScanGate(t, scanReport, "", "--fail-on=LOW", "CRITICAL")
|
||||
if r.code != 2 || !strings.Contains(r.stderr, `felis scan-gate: unexpected argument "CRITICAL"`) || r.env != nil {
|
||||
t.Errorf("stray argument: exit %d, stderr %q", r.code, r.stderr)
|
||||
}
|
||||
r = runScanGate(t, scanReport, "", "--accept=CVE-2024-0001 CVE-2024-0003")
|
||||
if r.code != 2 || !strings.Contains(r.stderr, `felis scan-gate: --accept: "CVE-2024-0001 CVE-2024-0003" is not a vulnerability id or secret rule id`) {
|
||||
t.Errorf("bad accept list: exit %d, stderr %q", r.code, r.stderr)
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := cmdScanGate(nil, &stdout, &stderr); code != 2 || !strings.Contains(stderr.String(), "--report is required") {
|
||||
t.Errorf("no report flag: exit %d, stderr %q", code, stderr.String())
|
||||
}
|
||||
}
|
||||
+11
-34
@@ -13,7 +13,6 @@ import (
|
||||
"felis.lolicon.best/internal/api"
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
)
|
||||
@@ -216,18 +215,18 @@ func provisionSystemServers(ctx context.Context, cfg *config.Config, out io.Writ
|
||||
"Re-run `sudo felis setup` on the control-plane host once the cluster is reachable", err)
|
||||
}
|
||||
// The login limbo authenticates to the felis-api INTERNAL face, so it needs the
|
||||
// internal base URL, the root domain (to link players at the console), and the
|
||||
// service token. The first two are plain env baked into the pod here; the token
|
||||
// is a Secret the operator injects by reference — but a secretKeyRef is
|
||||
// namespace-local, so first replicate the token Secret from the control namespace
|
||||
// into the minecraft namespace where the login pod runs. The control namespace is
|
||||
// internal base URL, the root domain (to link players at the console), and its
|
||||
// own token (felis-limbo-token). The first two are plain env baked into the pod
|
||||
// here; the token is a Secret the operator injects by reference — but a
|
||||
// secretKeyRef is namespace-local, so first replicate the token Secret from the
|
||||
// control namespace into the minecraft namespace where the login pod runs. The control namespace is
|
||||
// the platform default (there is no felis.toml override for it); a deployment that
|
||||
// renamed it must replicate the Secret by hand.
|
||||
controlNS := platform.DefaultControlNamespace
|
||||
apiBaseURL := platform.InternalAPIBaseURL(controlNS)
|
||||
// These Secrets must land in the minecraft namespace before the pods that
|
||||
// mount them are created: the service token (login authenticates to felis-api
|
||||
// with it), the Velocity forwarding secret (every backend verifies the proxy's
|
||||
// mount them are created: the login gate's token (login authenticates to
|
||||
// felis-api with it), the Velocity forwarding secret (every backend verifies the proxy's
|
||||
// signed handshake with it — without it the login gate would derive an OFFLINE
|
||||
// UUID and the Owner would bind the wrong Minecraft identity), and felis-config
|
||||
// (the on-demand BACKUP Job runs in the minecraft namespace and mounts it to
|
||||
@@ -239,31 +238,7 @@ func provisionSystemServers(ctx context.Context, cfg *config.Config, out io.Writ
|
||||
if buildNS == "" {
|
||||
buildNS = platform.DefaultBuildNamespace
|
||||
}
|
||||
secretOutcomes := []systemServerOutcome{
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token", "minecraft ns", false),
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
naming.ForwardingSecretName, naming.ForwardingSecretKey, "forwarding-secret", "minecraft ns", false),
|
||||
// refresh=true: felis-config is the rendered config, not a credential. The
|
||||
// backup/restore/fileedit Jobs and the reaper mount this copy, so a re-run
|
||||
// must update it when the control plane's render has moved on (a stale copy
|
||||
// e.g. keeps an old database URL after a credential rotation).
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
"felis-config", "felis.toml", "config", "minecraft ns", true),
|
||||
// The reaper's pre-reap warning emails authenticate with the same relay
|
||||
// password felis-api uses; the reaper pod runs in the minecraft namespace,
|
||||
// where a secretKeyRef resolves only against a local mirror. Skipped while
|
||||
// the relay is not configured yet — the "configure email" screen refreshes
|
||||
// both mirrors when it applies.
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
"felis-smtp", "password", "smtp", "minecraft ns", false),
|
||||
// The build namespace needs the same token: the build Job's fetch
|
||||
// initContainer reads the submission context from the internal face. Best
|
||||
// effort — a deployment that only installs the control plane simply never
|
||||
// builds a user submission.
|
||||
ensureSecretReplica(ctx, cl, controlNS, buildNS,
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token", "felis-build ns", false),
|
||||
}
|
||||
secretOutcomes := provisionSecretReplicas(ctx, cl, controlNS, cfg.K8s.Namespace, buildNS)
|
||||
outcomes := ensureSystemServers(ctx, cl, cfg.K8s.Namespace, cfg.Velocity.LoginImage, cfg.Velocity.LobbyImage, apiBaseURL, cfg.Server.RootDomain, defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
|
||||
outcomes = append(secretOutcomes, outcomes...)
|
||||
fmt.Fprintln(out, "\nfelis setup: login/lobby system servers (always-on, reaper-exempt):")
|
||||
@@ -345,7 +320,9 @@ func openConfiguredSetup(ctx context.Context, cfgPath string) (*configuredSetup,
|
||||
if err != nil {
|
||||
return nil, &setupOpenError{stage: "load config", err: err}
|
||||
}
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
// Pending migrations are the preflight's to apply; a newer schema is a rolled-back
|
||||
// binary, and nothing this console writes would match it.
|
||||
drv, err := openStore(ctx, cfg.Database.URL, true)
|
||||
if err != nil {
|
||||
return nil, &setupOpenError{stage: "open database", err: err}
|
||||
}
|
||||
|
||||
+119
-44
@@ -16,6 +16,7 @@ import (
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
"k8s.io/client-go/tools/clientcmd"
|
||||
"k8s.io/client-go/util/retry"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
@@ -392,21 +393,48 @@ var derivedSystemEnv = map[string]bool{
|
||||
// operator every run.
|
||||
func refreshDerivedEnv(ctx context.Context, cl client.Client, existing, desired *v1alpha1.MinecraftServer) (bool, error) {
|
||||
want := derivedEnvWanted(desired)
|
||||
|
||||
changed := false
|
||||
for i, e := range existing.Spec.Env {
|
||||
if v, ok := want[e.Name]; ok && v != e.Value {
|
||||
existing.Spec.Env[i].Value = v
|
||||
changed = true
|
||||
changed, err := patchOnConflictRetry(ctx, cl, existing, func() bool {
|
||||
changed := false
|
||||
for i, e := range existing.Spec.Env {
|
||||
if v, ok := want[e.Name]; ok && v != e.Value {
|
||||
existing.Spec.Env[i].Value = v
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
return false, nil
|
||||
}
|
||||
if err := cl.Update(ctx, existing); err != nil {
|
||||
return changed
|
||||
})
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("refresh %s env: %w", existing.Name, err)
|
||||
}
|
||||
return true, nil
|
||||
return changed, nil
|
||||
}
|
||||
|
||||
// patchOnConflictRetry applies mutate to obj and sends only the difference, as a
|
||||
// merge patch that carries the resourceVersion obj was read at. The operator writes
|
||||
// status and felis-api patches spec.idle on these same objects, so a write can land
|
||||
// between setup's read and its patch: the apiserver then answers 409, and this
|
||||
// re-reads obj and runs mutate again on the fresh copy, up to retry.DefaultRetry's
|
||||
// five attempts. The pinned resourceVersion is what keeps a list field such as
|
||||
// spec.env safe — a merge patch replaces a list whole, and without the lock an
|
||||
// entry added concurrently would be dropped. mutate reports whether it changed
|
||||
// anything; nothing is sent when it did not. obj holds the stored object after.
|
||||
func patchOnConflictRetry(ctx context.Context, cl client.Client, obj client.Object, mutate func() bool) (bool, error) {
|
||||
key := client.ObjectKeyFromObject(obj)
|
||||
changed, reread := false, false
|
||||
err := retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
if reread {
|
||||
if err := cl.Get(ctx, key, obj); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
reread = true
|
||||
base := obj.DeepCopyObject().(client.Object)
|
||||
if changed = mutate(); !changed {
|
||||
return nil
|
||||
}
|
||||
return cl.Patch(ctx, obj, client.MergeFromWithOptions(base, client.MergeFromWithOptimisticLock{}))
|
||||
})
|
||||
return changed, err
|
||||
}
|
||||
|
||||
// derivedEnvWanted maps the derived env keys of desired onto their values.
|
||||
@@ -469,22 +497,25 @@ func convergeSystemServers(ctx context.Context, cl client.Client, namespace, log
|
||||
}
|
||||
|
||||
var changes []string
|
||||
if existing.Spec.Rcon == (v1alpha1.RconSpec{}) && desired.Spec.Rcon != (v1alpha1.RconSpec{}) {
|
||||
existing.Spec.Rcon = desired.Spec.Rcon
|
||||
changes = append(changes, "spec.rcon")
|
||||
}
|
||||
if existing.Spec.Startup.HealthHTTPPort == 0 && desired.Spec.Startup.HealthHTTPPort != 0 {
|
||||
existing.Spec.Startup.HealthHTTPPort = desired.Spec.Startup.HealthHTTPPort
|
||||
changes = append(changes, "spec.startup.healthHTTPPort")
|
||||
}
|
||||
changes = append(changes, convergeDerivedEnv(&existing, desired)...)
|
||||
|
||||
if len(changes) == 0 {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, skipped: "already converged"})
|
||||
changed, err := patchOnConflictRetry(ctx, cl, &existing, func() bool {
|
||||
changes = nil
|
||||
if existing.Spec.Rcon == (v1alpha1.RconSpec{}) && desired.Spec.Rcon != (v1alpha1.RconSpec{}) {
|
||||
existing.Spec.Rcon = desired.Spec.Rcon
|
||||
changes = append(changes, "spec.rcon")
|
||||
}
|
||||
if existing.Spec.Startup.HealthHTTPPort == 0 && desired.Spec.Startup.HealthHTTPPort != 0 {
|
||||
existing.Spec.Startup.HealthHTTPPort = desired.Spec.Startup.HealthHTTPPort
|
||||
changes = append(changes, "spec.startup.healthHTTPPort")
|
||||
}
|
||||
changes = append(changes, convergeDerivedEnv(&existing, desired)...)
|
||||
return len(changes) > 0
|
||||
})
|
||||
if err != nil {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, err: fmt.Errorf("converge %s: %w", p.name, err)})
|
||||
continue
|
||||
}
|
||||
if err := cl.Update(ctx, &existing); err != nil {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, err: fmt.Errorf("converge %s: %w", p.name, err)})
|
||||
if !changed {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, skipped: "already converged"})
|
||||
continue
|
||||
}
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, updated: true, changes: changes})
|
||||
@@ -588,15 +619,51 @@ func phaseOrPending(p v1alpha1.Phase) string {
|
||||
return string(p)
|
||||
}
|
||||
|
||||
// provisionSecretReplicas copies the Secrets workload pods mount from the control
|
||||
// namespace into the namespaces those pods run in. The proxy's felis-service-token
|
||||
// is not among them: it lives in the control namespace and on the host, and a copy
|
||||
// anywhere else would open every game route to whoever reads that namespace.
|
||||
func provisionSecretReplicas(ctx context.Context, cl client.Client, controlNS, minecraftNS, buildNS string) []systemServerOutcome {
|
||||
return []systemServerOutcome{
|
||||
// refresh=true for both caller tokens: the control namespace holds the
|
||||
// current value and `felis rotate-token` replaces it there, so a replica
|
||||
// that differs is stale and the login gate would be turned away with it.
|
||||
ensureSecretReplica(ctx, cl, controlNS, minecraftNS,
|
||||
naming.LimboTokenSecretName, naming.ServiceTokenSecretKey, "limbo-token", "minecraft ns", true),
|
||||
ensureSecretReplica(ctx, cl, controlNS, minecraftNS,
|
||||
naming.ForwardingSecretName, naming.ForwardingSecretKey, "forwarding-secret", "minecraft ns", false),
|
||||
// refresh=true: felis-config is the rendered config, not a credential. The
|
||||
// backup/restore/fileedit Jobs and the reaper mount this copy, so a re-run
|
||||
// must update it when the control plane's render has moved on (a stale copy
|
||||
// e.g. keeps an old database URL after a credential rotation).
|
||||
ensureSecretReplica(ctx, cl, controlNS, minecraftNS,
|
||||
"felis-config", "felis.toml", "config", "minecraft ns", true),
|
||||
// The reaper's pre-reap warning emails authenticate with the same relay
|
||||
// password felis-api uses; the reaper pod runs in the minecraft namespace,
|
||||
// where a secretKeyRef resolves only against a local mirror. Skipped while
|
||||
// the relay is not configured yet — the "configure email" screen refreshes
|
||||
// both mirrors when it applies.
|
||||
ensureSecretReplica(ctx, cl, controlNS, minecraftNS,
|
||||
"felis-smtp", "password", "smtp", "minecraft ns", false),
|
||||
// The build namespace needs the build token: the build Job's fetch
|
||||
// initContainer reads the submission context from the internal face, and
|
||||
// that is all this token opens. Best effort — a deployment that only
|
||||
// installs the control plane simply never builds a user submission.
|
||||
ensureSecretReplica(ctx, cl, controlNS, buildNS,
|
||||
naming.BuildTokenSecretName, naming.ServiceTokenSecretKey, "build-token", "felis-build ns", true),
|
||||
}
|
||||
}
|
||||
|
||||
// ensureSecretReplica copies one Secret from the control namespace into a workload
|
||||
// namespace (minecraft — or the build namespace, whose fetch initContainer reads the
|
||||
// context from the felis-api internal face with the same token) so a pod can mount it
|
||||
// context from the felis-api internal face with the build token) so a pod can mount it
|
||||
// via secretKeyRef. A secretKeyRef is namespace-local, but those workloads do not run
|
||||
// beside the control plane — so without this replica the secretKeyRef would dangle and
|
||||
// wedge the pod in CreateContainerConfigError.
|
||||
//
|
||||
// Three Secrets need it, for different reasons: the service token (the login limbo and
|
||||
// the build Pod's context fetch — both authenticate to the felis-api internal face),
|
||||
// Several Secrets need it, for different reasons: the caller tokens of the login
|
||||
// limbo and the build Pod's context fetch (both authenticate to the felis-api
|
||||
// internal face, each with its own token),
|
||||
// the Velocity modern-forwarding secret (every backend — it is how a backend knows
|
||||
// a login really came from the proxy, and so that the player's UUID is Mojang-verified
|
||||
// rather than offline-derived), and the SMTP relay password (the reaper's pre-reap
|
||||
@@ -609,12 +676,14 @@ func phaseOrPending(p v1alpha1.Phase) string {
|
||||
// copies only Type and Data — never labels/annotations/ownerRefs — so the replica
|
||||
// carries no accidental GC owner or managed-by lineage.
|
||||
//
|
||||
// refreshExisting switches the felis-config mirror to refresh-in-place: that Secret is
|
||||
// a rendered config, never a hand-rotated credential, and the workload Jobs that mount
|
||||
// it (backup/restore/fileedit) plus the reaper silently misbehave on a stale copy —
|
||||
// e.g. after a database credential rotation the control plane moves on while every
|
||||
// backup Job keeps failing auth. Credential Secrets keep the never-overwrite rule so a
|
||||
// rotated value survives; to rotate those, delete the replica and re-run setup.
|
||||
// refreshExisting switches a replica to refresh-in-place from the control namespace.
|
||||
// The felis-config mirror uses it because that Secret is a rendered config and the
|
||||
// workload Jobs that mount it (backup/restore/fileedit) plus the reaper silently
|
||||
// misbehave on a stale copy — e.g. after a database credential rotation the control
|
||||
// plane moves on while every backup Job keeps failing auth. The caller tokens use it
|
||||
// because `felis rotate-token` replaces them in the control namespace, which makes a
|
||||
// differing replica stale by definition. The forwarding and SMTP Secrets keep the
|
||||
// never-overwrite rule.
|
||||
func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace, minecraftNamespace, secretName, secretKey, label, where string, refreshExisting bool) systemServerOutcome {
|
||||
name := label + " (" + where + ")"
|
||||
validate := func(secret *corev1.Secret, location, skipped string) systemServerOutcome {
|
||||
@@ -639,16 +708,22 @@ func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace
|
||||
if out := validate(&src, controlNamespace, ""); !out.available {
|
||||
return out
|
||||
}
|
||||
if bytes.Equal(existing.Data[secretKey], src.Data[secretKey]) {
|
||||
return validate(existing, minecraftNamespace, "already current")
|
||||
}
|
||||
if existing.Data == nil {
|
||||
existing.Data = map[string][]byte{}
|
||||
}
|
||||
existing.Data[secretKey] = src.Data[secretKey]
|
||||
if err := cl.Update(ctx, existing); err != nil {
|
||||
changed, err := patchOnConflictRetry(ctx, cl, existing, func() bool {
|
||||
if bytes.Equal(existing.Data[secretKey], src.Data[secretKey]) {
|
||||
return false
|
||||
}
|
||||
if existing.Data == nil {
|
||||
existing.Data = map[string][]byte{}
|
||||
}
|
||||
existing.Data[secretKey] = src.Data[secretKey]
|
||||
return true
|
||||
})
|
||||
if err != nil {
|
||||
return systemServerOutcome{name: name, err: err}
|
||||
}
|
||||
if !changed {
|
||||
return validate(existing, minecraftNamespace, "already current")
|
||||
}
|
||||
return systemServerOutcome{name: name, updated: true, available: true}
|
||||
}
|
||||
if controlNamespace == minecraftNamespace {
|
||||
@@ -712,7 +787,7 @@ func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace
|
||||
|
||||
func requiredProvisioningError(outcomes []systemServerOutcome) error {
|
||||
required := map[string]struct{}{
|
||||
"service-token (minecraft ns)": {},
|
||||
"limbo-token (minecraft ns)": {},
|
||||
"forwarding-secret (minecraft ns)": {},
|
||||
naming.SystemLoginServer: {},
|
||||
}
|
||||
|
||||
@@ -142,21 +142,21 @@ func TestLoginSystemServerEnv(t *testing.T) {
|
||||
// namespace (create-if-absent), so the operator's secretKeyRef on the backend pod
|
||||
// resolves. It must not overwrite an existing replica, and must degrade gracefully
|
||||
// when the source is missing or the namespaces coincide. Exercised here with the
|
||||
// service token; setup runs it a second time for the Velocity forwarding secret.
|
||||
// Velocity forwarding secret, which setup replicates in this never-overwrite mode.
|
||||
func TestEnsureSecretReplica(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
|
||||
srcSecret := func() *corev1.Secret {
|
||||
return &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ServiceTokenSecretName, Namespace: "felis"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ForwardingSecretName, Namespace: "felis"},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
Data: map[string][]byte{naming.ServiceTokenSecretKey: []byte("s3cr3t")},
|
||||
Data: map[string][]byte{naming.ForwardingSecretKey: []byte("s3cr3t")},
|
||||
}
|
||||
}
|
||||
replicate := func(cl client.Client, controlNS, mcNS string) systemServerOutcome {
|
||||
return ensureSecretReplica(ctx, cl, controlNS, mcNS,
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token", "minecraft ns", false)
|
||||
naming.ForwardingSecretName, naming.ForwardingSecretKey, "forwarding-secret", "minecraft ns", false)
|
||||
}
|
||||
|
||||
t.Run("replicates when absent", func(t *testing.T) {
|
||||
@@ -166,19 +166,19 @@ func TestEnsureSecretReplica(t *testing.T) {
|
||||
t.Fatalf("outcome = %+v, want created", out)
|
||||
}
|
||||
var replica corev1.Secret
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.ServiceTokenSecretName}, &replica); err != nil {
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.ForwardingSecretName}, &replica); err != nil {
|
||||
t.Fatalf("get replica: %v", err)
|
||||
}
|
||||
if string(replica.Data[naming.ServiceTokenSecretKey]) != "s3cr3t" {
|
||||
t.Errorf("replica token = %q, want s3cr3t", replica.Data[naming.ServiceTokenSecretKey])
|
||||
if string(replica.Data[naming.ForwardingSecretKey]) != "s3cr3t" {
|
||||
t.Errorf("replica token = %q, want s3cr3t", replica.Data[naming.ForwardingSecretKey])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("does not overwrite existing replica", func(t *testing.T) {
|
||||
existing := &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ServiceTokenSecretName, Namespace: "minecraft"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ForwardingSecretName, Namespace: "minecraft"},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
Data: map[string][]byte{naming.ServiceTokenSecretKey: []byte("rotated")},
|
||||
Data: map[string][]byte{naming.ForwardingSecretKey: []byte("rotated")},
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(srcSecret(), existing).Build()
|
||||
out := replicate(cl, "felis", "minecraft")
|
||||
@@ -186,11 +186,11 @@ func TestEnsureSecretReplica(t *testing.T) {
|
||||
t.Fatalf("outcome = %+v, want skipped (not clobbered)", out)
|
||||
}
|
||||
var replica corev1.Secret
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.ServiceTokenSecretName}, &replica); err != nil {
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.ForwardingSecretName}, &replica); err != nil {
|
||||
t.Fatalf("get replica: %v", err)
|
||||
}
|
||||
if string(replica.Data[naming.ServiceTokenSecretKey]) != "rotated" {
|
||||
t.Error("existing replica was overwritten — a rotated token must survive")
|
||||
if string(replica.Data[naming.ForwardingSecretKey]) != "rotated" {
|
||||
t.Error("existing replica was overwritten — a hand-set value must survive")
|
||||
}
|
||||
})
|
||||
|
||||
@@ -204,22 +204,22 @@ func TestEnsureSecretReplica(t *testing.T) {
|
||||
|
||||
t.Run("rejects a source with an empty required key", func(t *testing.T) {
|
||||
bad := srcSecret()
|
||||
bad.Data[naming.ServiceTokenSecretKey] = nil
|
||||
bad.Data[naming.ForwardingSecretKey] = nil
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(bad).Build()
|
||||
out := replicate(cl, "felis", "minecraft")
|
||||
if out.err != nil || out.created || out.available || !strings.Contains(out.skipped, naming.ServiceTokenSecretKey) {
|
||||
if out.err != nil || out.created || out.available || !strings.Contains(out.skipped, naming.ForwardingSecretKey) {
|
||||
t.Fatalf("outcome = %+v, want unavailable required key", out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("rejects an existing replica with an empty required key", func(t *testing.T) {
|
||||
bad := &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ServiceTokenSecretName, Namespace: "minecraft"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: naming.ForwardingSecretName, Namespace: "minecraft"},
|
||||
Data: map[string][]byte{},
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(srcSecret(), bad).Build()
|
||||
out := replicate(cl, "felis", "minecraft")
|
||||
if out.err != nil || out.created || out.available || !strings.Contains(out.skipped, naming.ServiceTokenSecretKey) {
|
||||
if out.err != nil || out.created || out.available || !strings.Contains(out.skipped, naming.ForwardingSecretKey) {
|
||||
t.Fatalf("outcome = %+v, want unavailable existing replica", out)
|
||||
}
|
||||
})
|
||||
@@ -245,7 +245,7 @@ func TestEnsureSecretReplica(t *testing.T) {
|
||||
bad.Data = nil
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(bad).Build()
|
||||
out := replicate(cl, "felis", "felis")
|
||||
if out.err != nil || out.available || !strings.Contains(out.skipped, naming.ServiceTokenSecretKey) {
|
||||
if out.err != nil || out.available || !strings.Contains(out.skipped, naming.ForwardingSecretKey) {
|
||||
t.Fatalf("outcome = %+v, want unavailable required key", out)
|
||||
}
|
||||
})
|
||||
@@ -334,7 +334,7 @@ func TestEnsureSecretReplicaRefresh(t *testing.T) {
|
||||
|
||||
func TestRequiredProvisioningError(t *testing.T) {
|
||||
ready := []systemServerOutcome{
|
||||
{name: "service-token (minecraft ns)", available: true},
|
||||
{name: "limbo-token (minecraft ns)", available: true},
|
||||
{name: "forwarding-secret (minecraft ns)", available: true},
|
||||
{name: naming.SystemLoginServer, available: true},
|
||||
{name: naming.SystemLobbyServer, skipped: "image not configured"},
|
||||
@@ -349,6 +349,14 @@ func TestRequiredProvisioningError(t *testing.T) {
|
||||
t.Fatalf("missing forwarding secret = %v, want named error", err)
|
||||
}
|
||||
|
||||
// The login gate cannot reach felis-api without its token, so setup must not
|
||||
// report success while that replica is missing.
|
||||
noToken := append([]systemServerOutcome(nil), ready...)
|
||||
noToken[0] = systemServerOutcome{name: "limbo-token (minecraft ns)", skipped: "source missing"}
|
||||
if err := requiredProvisioningError(noToken); err == nil || !strings.Contains(err.Error(), "limbo-token") {
|
||||
t.Fatalf("missing limbo token = %v, want named error", err)
|
||||
}
|
||||
|
||||
failed := append([]systemServerOutcome(nil), ready...)
|
||||
failed[3] = systemServerOutcome{name: naming.SystemLobbyServer, err: context.DeadlineExceeded}
|
||||
if err := requiredProvisioningError(failed); err == nil || !strings.Contains(err.Error(), naming.SystemLobbyServer) {
|
||||
@@ -662,3 +670,62 @@ func TestEnsureSystemServersRefreshesDerivedEnv(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// Setup's replicas carry each workload its own caller token and nothing more: the
|
||||
// login gate gets felis-limbo-token in the minecraft namespace, the build Jobs get
|
||||
// felis-build-token in the build namespace, a replica left stale by a rotation is
|
||||
// brought up to date, and the proxy's felis-service-token is copied nowhere.
|
||||
func TestProvisionSecretReplicasCarryCallerTokens(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
secret := func(ns, name, key, val string) *corev1.Secret {
|
||||
return &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
Data: map[string][]byte{key: []byte(val)},
|
||||
}
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||
secret("felis", "felis-service-token", "token", "proxy-tok"),
|
||||
secret("felis", "felis-limbo-token", "token", "limbo-new"),
|
||||
secret("felis", "felis-build-token", "token", "build-tok"),
|
||||
secret("felis", "felis-ops-token", "token", "ops-tok"),
|
||||
secret("felis", naming.ForwardingSecretName, naming.ForwardingSecretKey, "fwd"),
|
||||
// What a rotation leaves behind before setup runs again.
|
||||
secret("minecraft", "felis-limbo-token", "token", "limbo-old"),
|
||||
).Build()
|
||||
|
||||
outcomes := provisionSecretReplicas(ctx, cl, "felis", "minecraft", "felis-build")
|
||||
for _, o := range outcomes {
|
||||
if o.err != nil {
|
||||
t.Fatalf("%s: %v", o.name, o.err)
|
||||
}
|
||||
}
|
||||
|
||||
read := func(ns, name string) (string, bool) {
|
||||
var s corev1.Secret
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: ns, Name: name}, &s); err != nil {
|
||||
return "", false
|
||||
}
|
||||
return string(s.Data["token"]), true
|
||||
}
|
||||
if got, _ := read("minecraft", "felis-limbo-token"); got != "limbo-new" {
|
||||
t.Errorf("minecraft/felis-limbo-token = %q, want the rotated limbo-new", got)
|
||||
}
|
||||
if got, _ := read("felis-build", "felis-build-token"); got != "build-tok" {
|
||||
t.Errorf("felis-build/felis-build-token = %q, want build-tok", got)
|
||||
}
|
||||
for _, ns := range []string{"minecraft", "felis-build"} {
|
||||
for _, name := range []string{"felis-service-token", "felis-ops-token"} {
|
||||
if _, ok := read(ns, name); ok {
|
||||
t.Errorf("%s/%s was replicated; only the control namespace holds it", ns, name)
|
||||
}
|
||||
}
|
||||
}
|
||||
if _, ok := read("minecraft", "felis-build-token"); ok {
|
||||
t.Error("the build token was copied into the minecraft namespace")
|
||||
}
|
||||
if _, ok := read("felis-build", "felis-limbo-token"); ok {
|
||||
t.Error("the limbo token was copied into the build namespace")
|
||||
}
|
||||
}
|
||||
@@ -33,17 +33,9 @@ func applyMigrations(dbURL string) (int, error) {
|
||||
if _, err := store.Up(ctx, drv, migrations); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
applied, err := drv.AppliedVersions(ctx)
|
||||
s, err := store.ReadSchema(ctx, drv)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(applied), nil
|
||||
}
|
||||
|
||||
func totalMigrations() (int, error) {
|
||||
migrations, err := store.LoadMigrations()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(migrations), nil
|
||||
return s.Applied, nil
|
||||
}
|
||||
@@ -72,20 +72,14 @@ func parseDBURL(url string) (dbCfg, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
func countMigrations(dbURL string) (int, error) {
|
||||
// readSchema lines the database's recorded migrations up with this build's.
|
||||
func readSchema(dbURL string) (store.Schema, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
drv, err := store.Open(ctx, dbURL)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
return store.Schema{}, err
|
||||
}
|
||||
defer drv.Close()
|
||||
if err := drv.EnsureVersionTable(ctx); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
applied, err := drv.AppliedVersions(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(applied), nil
|
||||
return store.ReadSchema(ctx, drv)
|
||||
}
|
||||
@@ -46,6 +46,7 @@ type pfDBMsg struct{ err error }
|
||||
type pfMigCheckMsg struct {
|
||||
applied int
|
||||
total int
|
||||
pending bool
|
||||
err error
|
||||
}
|
||||
|
||||
@@ -84,7 +85,7 @@ func (m *preflightModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
return m, nil
|
||||
}
|
||||
m.applied, m.total = msg.applied, msg.total
|
||||
if msg.applied < msg.total {
|
||||
if msg.pending {
|
||||
m.state = pfApplyMig
|
||||
return m, m.applyMigrations()
|
||||
}
|
||||
@@ -204,12 +205,14 @@ func (m *preflightModel) checkDB() tea.Cmd {
|
||||
|
||||
func (m *preflightModel) checkMigrations() tea.Cmd {
|
||||
return func() tea.Msg {
|
||||
applied, err := countMigrations(m.dbURL)
|
||||
if err != nil {
|
||||
return pfMigCheckMsg{err: err}
|
||||
// The sets, not their sizes: a database a newer release migrated can hold as
|
||||
// many rows as this build has migrations, and must stop here rather than be
|
||||
// "healed" by an older binary.
|
||||
s, err := readSchema(m.dbURL)
|
||||
if err == nil {
|
||||
err = s.Newer()
|
||||
}
|
||||
total, err := totalMigrations()
|
||||
return pfMigCheckMsg{applied: applied, total: total, err: err}
|
||||
return pfMigCheckMsg{applied: s.Applied, total: s.Total, pending: len(s.Pending) > 0, err: err}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+24
-11
@@ -9,7 +9,6 @@ import (
|
||||
"strings"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
|
||||
"github.com/charmbracelet/bubbles/spinner"
|
||||
@@ -311,24 +310,22 @@ func applySMTPConfig(ctx context.Context, in smtpInputs) error {
|
||||
if err != nil {
|
||||
return fmt.Errorf("port %q is not a number", in.port)
|
||||
}
|
||||
relay := &mail.SMTP{Host: in.host, Port: port, From: in.from, Username: in.username, Password: in.password}
|
||||
if err := relay.Ping(ctx); err != nil {
|
||||
var prev config.SMTPConfig
|
||||
if cur, err := config.Load(hostSetupConfigPath); err == nil {
|
||||
prev = cur.SMTP
|
||||
}
|
||||
// Ping under the posture felis api will send with, so a relay without
|
||||
// STARTTLS is turned down here rather than at a player's first code.
|
||||
if err := smtpRelay(setupSMTPConfig(in, port, prev), in.password).Ping(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
smtpCfg := config.SMTPConfig{
|
||||
Host: in.host,
|
||||
Port: port,
|
||||
From: in.from,
|
||||
Username: in.username,
|
||||
PasswordRef: platform.SMTPPasswordEnv,
|
||||
}
|
||||
for _, path := range []string{hostSetupConfigPath, podSetupConfigPath} {
|
||||
cfg, err := config.Load(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
cfg.SMTP = smtpCfg
|
||||
cfg.SMTP = setupSMTPConfig(in, port, cfg.SMTP)
|
||||
if err := writeConfig(path, cfg); err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -351,6 +348,22 @@ func applySMTPConfig(ctx context.Context, in smtpInputs) error {
|
||||
return kubectl(ctx, "-n", "felis", "rollout", "status", "deployment/felis-api", "--timeout=180s")
|
||||
}
|
||||
|
||||
// setupSMTPConfig is the [smtp] block this screen writes: the relay it just
|
||||
// proved, plus the keys only an operator sets by hand (require_tls,
|
||||
// max_per_hour), carried over from the block it replaces so reconfiguring the
|
||||
// relay does not quietly reset them.
|
||||
func setupSMTPConfig(in smtpInputs, port int, prev config.SMTPConfig) config.SMTPConfig {
|
||||
return config.SMTPConfig{
|
||||
Host: in.host,
|
||||
Port: port,
|
||||
From: in.from,
|
||||
Username: in.username,
|
||||
PasswordRef: platform.SMTPPasswordEnv,
|
||||
MaxPerHour: prev.MaxPerHour,
|
||||
RequireTLS: prev.RequireTLS,
|
||||
}
|
||||
}
|
||||
|
||||
// smtpSecretManifest renders the felis-smtp Secret for the given namespace, the
|
||||
// one the receiving Deployment/CronJob resolves its secretKeyRef against (felis
|
||||
// for felis-api, the workload namespace for the reaper's mirror). The namespace
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
@@ -35,3 +38,39 @@ func TestSMTPSecretManifestCarriesTargetNamespace(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSMTPRelayCarriesTLSPosture: the relay felis api, the reaper and the
|
||||
// watchdog send through refuses plaintext for a remote host and allows it for
|
||||
// one on this host, as [smtp] says.
|
||||
func TestSMTPRelayCarriesTLSPosture(t *testing.T) {
|
||||
remote := smtpRelay(config.SMTPConfig{Host: "smtp.example.net", Port: 587, From: "[email protected]", Username: "felis"}, "pw")
|
||||
want := &mail.SMTP{Host: "smtp.example.net", Port: 587, From: "[email protected]", Username: "felis", Password: "pw", RequireTLS: true}
|
||||
if !reflect.DeepEqual(remote, want) {
|
||||
t.Errorf("remote relay = %+v, want %+v", remote, want)
|
||||
}
|
||||
if local := smtpRelay(config.SMTPConfig{Host: "127.0.0.1", Port: 25, From: "[email protected]"}, ""); local.RequireTLS {
|
||||
t.Error("a relay on this host must not require TLS by default")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSetupSMTPConfigKeepsHandSetKeys: re-running the email screen replaces the
|
||||
// relay but keeps require_tls and max_per_hour, which only an operator sets.
|
||||
func TestSetupSMTPConfigKeepsHandSetKeys(t *testing.T) {
|
||||
off := false
|
||||
prev := config.SMTPConfig{Host: "old.example.net", Port: 25, From: "[email protected]", MaxPerHour: 500, RequireTLS: &off}
|
||||
in := smtpInputs{host: "smtp.example.net", from: "[email protected]", username: "felis"}
|
||||
got := setupSMTPConfig(in, 465, prev)
|
||||
if got.Host != "smtp.example.net" || got.Port != 465 || got.From != "[email protected]" ||
|
||||
got.Username != "felis" || got.PasswordRef != "FELIS_SMTP_PASSWORD" {
|
||||
t.Errorf("relay fields = %+v", got)
|
||||
}
|
||||
if got.MaxPerHour != 500 {
|
||||
t.Errorf("max_per_hour = %d, want 500", got.MaxPerHour)
|
||||
}
|
||||
if got.RequireTLS == nil || *got.RequireTLS {
|
||||
t.Errorf("require_tls = %v, want the operator's false", got.RequireTLS)
|
||||
}
|
||||
if fresh := setupSMTPConfig(in, 587, config.SMTPConfig{}); fresh.RequireTLS != nil || fresh.MaxPerHour != 0 {
|
||||
t.Errorf("first setup = %+v, want require_tls and max_per_hour unset", fresh)
|
||||
}
|
||||
}
|
||||
+237
-20
@@ -2,6 +2,8 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
@@ -9,6 +11,9 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/updater"
|
||||
"felis.lolicon.best/internal/updates"
|
||||
)
|
||||
@@ -34,6 +39,8 @@ const updateTimeout = 60 * time.Second
|
||||
type updateTarget struct {
|
||||
// selector is the flag name without dashes.
|
||||
selector string
|
||||
// help is the flag's usage line.
|
||||
help string
|
||||
// component is the updates planner's name for this piece, or "" when the planner
|
||||
// deliberately does not track it (Minecraft, which is pinned).
|
||||
component string
|
||||
@@ -41,9 +48,11 @@ type updateTarget struct {
|
||||
note string
|
||||
// command is the exact, already-tested way to apply it.
|
||||
command string
|
||||
// installer marks a command that re-runs the installer, which the trailer explains.
|
||||
installer bool
|
||||
}
|
||||
|
||||
// installerRerun is the tested apply path for every planner-backed selector: re-run the
|
||||
// installerRerun is the tested apply path for every selector Felis installs: re-run the
|
||||
// installer. It is idempotent, and it is the only path that fetches a newer version --
|
||||
// `felis setup` skips its host-bootstrap phase on a completed install (all four install
|
||||
// markers already exist), so there it opens the config console and moves no component,
|
||||
@@ -58,6 +67,10 @@ type updateTarget struct {
|
||||
// below points at the README's token'd form for that case.
|
||||
const installerRerun = "curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/{ref}/deploy/bootstrap.sh | sudo bash"
|
||||
|
||||
// installerRerunDeps is the same re-run with FELIS_UPGRADE_DEPS=1, which lets it move an
|
||||
// installed k3s and cloudflared to the versions the release pins.
|
||||
const installerRerunDeps = "curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/{ref}/deploy/bootstrap.sh | sudo FELIS_UPGRADE_DEPS=1 bash"
|
||||
|
||||
// updateTargets is the selector table. panel and plugins both resolve to felis-api
|
||||
// because they are not separately versioned: the panel is compiled into the felis
|
||||
// binary with //go:embed, and the plugin jars are built from this same repo in the
|
||||
@@ -65,24 +78,62 @@ const installerRerun = "curl -fsSL https://raw.githubusercontent.com/FelisMC/Fel
|
||||
var updateTargets = []updateTarget{
|
||||
{
|
||||
selector: "panel",
|
||||
help: "select the panel + control plane (felis-api)",
|
||||
component: "felis-api",
|
||||
note: "the panel is embedded in the felis binary (//go:embed), so updating it means rebuilding the felis image and rolling felis-api",
|
||||
command: installerRerun,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "velocity",
|
||||
help: "select the Velocity proxy",
|
||||
component: "velocity",
|
||||
note: "re-runs install_velocity: the build the release pins in deploy/game-stack.lock (FELIS_VELOCITY_VERSION=<minor> takes that minor's newest build instead), sha256-checked, atomic jar install, then restarts felis-velocity only if the jar or its config changed",
|
||||
command: installerRerun,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "plugins",
|
||||
help: "select the Felis plugin jars (velocity/paper/limbo)",
|
||||
component: "felis-api",
|
||||
note: "felis-velocity.jar is a host-file swap, but felis-paper.jar and felis-limbo.jar are baked into the lobby/limbo images and need a rebuild + re-mirror into the in-cluster registry (the installer re-run does both)",
|
||||
command: installerRerun,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "k3s",
|
||||
help: "select k3s",
|
||||
component: "k3s",
|
||||
note: "FELIS_UPGRADE_DEPS=1 moves k3s to the version the Felis release pins, which can trail the newest upstream; it moves one minor version at a time and refuses a larger jump. Running game servers keep running while k3s restarts",
|
||||
command: installerRerunDeps,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "cloudflared",
|
||||
help: "select cloudflared",
|
||||
component: "cloudflared",
|
||||
note: "FELIS_UPGRADE_DEPS=1 swaps the binary for the sha256-pinned build the Felis release names and restarts cloudflared-felis; the panel's tunnel drops for a few seconds",
|
||||
command: installerRerunDeps,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "jre",
|
||||
help: "select the Temurin JRE Velocity runs on",
|
||||
component: "jre",
|
||||
note: "the installer installs the Temurin build the Felis release pins (sha256-checked) and restarts felis-velocity when it changed; a newer upstream build reaches the host with a release that pins it",
|
||||
command: installerRerun,
|
||||
installer: true,
|
||||
},
|
||||
{
|
||||
selector: "postgres",
|
||||
help: "select PostgreSQL",
|
||||
component: "postgresql",
|
||||
note: "PostgreSQL comes from the distribution's packages, so a minor release is a package update followed by a restart (a few seconds without the API). A new major needs pg_upgrade first: docs/operations.md §4",
|
||||
command: "sudo dnf upgrade 'postgresql*' || sudo apt-get install --only-upgrade 'postgresql*'; sudo systemctl restart postgresql",
|
||||
},
|
||||
{
|
||||
selector: "mc",
|
||||
help: "select Minecraft (pinned; reported only)",
|
||||
component: "", // never tracked: see the pin note below
|
||||
note: "Minecraft is pinned by policy (\"能不动的就别动\") and Felis never proposes a version change for it. A server's version is a property of that server's image — change it on the server, not through a platform update",
|
||||
command: "",
|
||||
@@ -92,26 +143,29 @@ var updateTargets = []updateTarget{
|
||||
// cmdUpdate reports what can be updated and what is already current.
|
||||
//
|
||||
// Bare `felis update` prints the status of every tracked component. Selector flags
|
||||
// (--panel/--velocity/--mc/--plugins/--all) narrow that report to the components
|
||||
// (--panel/--velocity/--plugins/--k3s/--cloudflared/--jre/--postgres/--mc/--all) narrow that report to the components
|
||||
// they name AND print how to apply each one. --force additionally prints the apply
|
||||
// instruction for a selected component that is already up to date, for the
|
||||
// reinstall/repair case.
|
||||
//
|
||||
// It never applies anything and never mutates the node, so unlike setup/breakGlass
|
||||
// it needs no root. The versions it reads come from this host: k3s and cloudflared
|
||||
// answer `--version`, Velocity's version is read out of the installed jar's
|
||||
// manifest, and felis-api's is this binary's own build stamp — the same value
|
||||
// `felis version` prints, which is what the user asked to be the source of truth.
|
||||
// it needs no root. The versions it reads come from this host: k3s, cloudflared and
|
||||
// PostgreSQL answer `--version`, Velocity's version is read out of the installed jar's
|
||||
// manifest, the JRE's out of its release file, and felis-api's is this binary's own
|
||||
// build stamp — the same value `felis version` prints, which is what the user asked
|
||||
// to be the source of truth.
|
||||
func cmdUpdate(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("update", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
panel := fs.Bool("panel", false, "select the panel + control plane (felis-api)")
|
||||
velocity := fs.Bool("velocity", false, "select the Velocity proxy")
|
||||
mc := fs.Bool("mc", false, "select Minecraft (pinned; reported only)")
|
||||
plugins := fs.Bool("plugins", false, "select the Felis plugin jars (velocity/paper/limbo)")
|
||||
flags := map[string]*bool{}
|
||||
for _, t := range updateTargets {
|
||||
flags[t.selector] = fs.Bool(t.selector, false, t.help)
|
||||
}
|
||||
all := fs.Bool("all", false, "select every component above")
|
||||
force := fs.Bool("force", false, "print the apply command for a selected component even when it is already up to date")
|
||||
velocityJar := fs.String("velocity-jar", updater.DefaultVelocityJarPath, "path to the installed Velocity jar to read the current version from")
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml, read for the maintenance window the panel stores")
|
||||
record := fs.Bool("record", false, "also store this check for the panel's Updates page (felis-update-check.timer runs it daily)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -121,8 +175,8 @@ func cmdUpdate(args []string, stdout, stderr io.Writer) int {
|
||||
}
|
||||
|
||||
selected := map[string]bool{}
|
||||
for sel, on := range map[string]bool{"panel": *panel, "velocity": *velocity, "mc": *mc, "plugins": *plugins} {
|
||||
if on || *all {
|
||||
for sel, on := range flags {
|
||||
if *on || *all {
|
||||
selected[sel] = true
|
||||
}
|
||||
}
|
||||
@@ -130,9 +184,10 @@ func cmdUpdate(args []string, stdout, stderr io.Writer) int {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), updateTimeout)
|
||||
defer cancel()
|
||||
|
||||
src := updater.NewRoutingSource(updater.Topology())
|
||||
rn := &updater.Runner{
|
||||
Gatherer: updater.NewHostGatherer(resolvedVersion(), *velocityJar),
|
||||
Source: updater.NewRoutingSource(updater.Topology()),
|
||||
Source: src,
|
||||
// Notifier and Applier stay nil on purpose: a human typing this command IS the
|
||||
// notification, and nothing here applies. The zero Window below means every
|
||||
// Scheduled component degrades to a notify, so the report can never claim an
|
||||
@@ -144,13 +199,104 @@ func cmdUpdate(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
win, winErr := readUpdateWindow(ctx, *cfgPath)
|
||||
fmt.Fprint(stdout, renderWindowLine(win, winErr, now))
|
||||
fmt.Fprint(stdout, renderUpdateReport(res, selected))
|
||||
fmt.Fprint(stdout, renderNotes(src.Notes(), selected))
|
||||
if len(selected) > 0 {
|
||||
if winErr == nil && !win.Start.IsZero() && !win.Contains(now) {
|
||||
fmt.Fprint(stdout, "Warning: this is outside the maintenance window; the commands below take effect as soon as you run them.\n")
|
||||
}
|
||||
fmt.Fprint(stdout, renderApplyGuidance(res, selected, *force))
|
||||
}
|
||||
if *record {
|
||||
// A fresh context: the discovery pass may have spent most of updateTimeout.
|
||||
rctx, rcancel := context.WithTimeout(context.Background(), updateWindowTimeout)
|
||||
defer rcancel()
|
||||
if err := recordUpdateStatus(rctx, *cfgPath, buildStatusReport(res, src.Notes(), resolvedVersion(), now)); err != nil {
|
||||
fmt.Fprintf(stderr, "felis update: record the check for the panel: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Fprint(stdout, "Recorded this check for the panel's Updates page.\n")
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// buildStatusReport turns one run into the record the panel shows: every planned
|
||||
// component in plan order, then each component whose installed version could not
|
||||
// be read, by name. A component the feed could not answer for is StateUnknown with
|
||||
// the reason, never StateCurrent: the panel must not call a component current when
|
||||
// nobody could check.
|
||||
func buildStatusReport(res updater.Result, notes map[string]string, felis string, now time.Time) updates.StatusReport {
|
||||
selectorOf := map[string]string{}
|
||||
for _, t := range updateTargets {
|
||||
if t.component != "" && selectorOf[t.component] == "" {
|
||||
selectorOf[t.component] = t.selector
|
||||
}
|
||||
}
|
||||
rep := updates.StatusReport{CheckedAt: now.UTC(), Felis: felis, Components: []updates.ComponentStatus{}}
|
||||
for _, a := range res.RunResult.Plan {
|
||||
cs := updates.ComponentStatus{
|
||||
Name: a.Component,
|
||||
Current: a.Current.String(),
|
||||
Selector: selectorOf[a.Component],
|
||||
Note: notes[a.Component],
|
||||
}
|
||||
switch {
|
||||
case a.Kind == updates.ActionPinned:
|
||||
cs.State = updates.StatePinned
|
||||
case a.Kind == updates.ActionNotify || a.Kind == updates.ActionApply:
|
||||
cs.State = updates.StateAvailable
|
||||
cs.Latest = a.Latest.String()
|
||||
case a.LatestKnown:
|
||||
cs.State = updates.StateCurrent
|
||||
default:
|
||||
cs.State = updates.StateUnknown
|
||||
if err := res.RunResult.SourceErrors[a.Component]; err != nil {
|
||||
cs.Error = err.Error()
|
||||
}
|
||||
}
|
||||
rep.Components = append(rep.Components, cs)
|
||||
}
|
||||
names := make([]string, 0, len(res.GatherErrors))
|
||||
for name := range res.GatherErrors {
|
||||
names = append(names, name)
|
||||
}
|
||||
sort.Strings(names)
|
||||
for _, name := range names {
|
||||
rep.Components = append(rep.Components, updates.ComponentStatus{
|
||||
Name: name,
|
||||
State: updates.StateUnreadable,
|
||||
Selector: selectorOf[name],
|
||||
Note: notes[name],
|
||||
Error: res.GatherErrors[name].Error(),
|
||||
})
|
||||
}
|
||||
return rep
|
||||
}
|
||||
|
||||
// recordUpdateStatus upserts rep into platform_settings[updates.StatusKey], the
|
||||
// row the API serves to the panel's Updates page.
|
||||
func recordUpdateStatus(ctx context.Context, cfgPath string, rep updates.StatusReport) error {
|
||||
cfg, err := config.Load(cfgPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
v, err := json.Marshal(rep)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
conn, err := pgx.Connect(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer conn.Close(context.Background())
|
||||
_, err = conn.Exec(ctx, `INSERT INTO platform_settings (key, value) VALUES ($1, $2::jsonb)
|
||||
ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value, updated_at = now()`, updates.StatusKey, string(v))
|
||||
return err
|
||||
}
|
||||
|
||||
// renderUpdateReport renders the component status table. With no selectors it shows
|
||||
// every tracked component; with selectors it shows only the components those
|
||||
// selectors name, so `felis update --velocity` is a focused answer rather than the
|
||||
@@ -199,6 +345,23 @@ func renderUpdateReport(res updater.Result, selected map[string]bool) string {
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// renderNotes prints what the release lookups learned beyond the versions (today: a
|
||||
// PostgreSQL major past its end of life), for the components the selectors show.
|
||||
func renderNotes(notes map[string]string, selected map[string]bool) string {
|
||||
names := make([]string, 0, len(notes))
|
||||
for name := range notes {
|
||||
if len(selected) == 0 || selectedCovers(selected, name) {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
var b strings.Builder
|
||||
for _, name := range names {
|
||||
fmt.Fprintf(&b, "%-13s note: %s\n", name, notes[name])
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// writeErrs appends one explanatory line per failed component, in a stable order so
|
||||
// the output does not shuffle between runs, honouring the active selector filter.
|
||||
func writeErrs(b *strings.Builder, label string, errs map[string]error, selected map[string]bool) {
|
||||
@@ -234,7 +397,7 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
||||
}
|
||||
|
||||
var b strings.Builder
|
||||
var offeredCommand bool
|
||||
var offeredInstaller bool
|
||||
for _, t := range updateTargets {
|
||||
if !selected[t.selector] {
|
||||
continue
|
||||
@@ -262,7 +425,7 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
||||
fmt.Fprintf(&b, " note: cannot tell whether %s is current — its latest version could not be discovered (see above); this reinstalls it either way\n", t.component)
|
||||
}
|
||||
fmt.Fprintf(&b, " run: %s\n", strings.ReplaceAll(t.command, "{ref}", installerRef(byComponent)))
|
||||
offeredCommand = true
|
||||
offeredInstaller = offeredInstaller || t.installer
|
||||
}
|
||||
// Only explain the command when one was actually offered; a --mc-only run has
|
||||
// nothing to run and the trailer would be a non-sequitur.
|
||||
@@ -270,13 +433,13 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
||||
// One trailer serves every selector now: setup is not an apply path at all on a
|
||||
// completed install (shouldRunHostBootstrapBeforeConfig only enters the host
|
||||
// bootstrap while an install marker is missing), so the installer re-run is the one
|
||||
// worked path for all three components and there is no per-component exception left
|
||||
// worked path for every component Felis installs and there is no per-component exception left
|
||||
// to scope. Two caveats stay because following the advice without them bites real
|
||||
// hosts: the channel is not persisted anywhere (a bare re-run on a main host quietly
|
||||
// moves it onto releases), and the private repo's one-liner needs the read token
|
||||
// back in the environment before it can resolve anything.
|
||||
if offeredCommand {
|
||||
b.WriteString("\nRe-running the installer applies everything above: it fetches the newest version on\nthe channel in effect and re-applies the bundle (release is the default). The channel\nis not persisted, so pass FELIS_VERSION_BOOTSTRAP=dev if this host tracks main. While\nthis repo is private, the one-liner above 404s without a token; the README's install\nsection has the token'd form that works. felis setup is not this path: on a completed\ninstall it opens the config console and installs nothing newer. Restart game servers\nafterwards.\n")
|
||||
if offeredInstaller {
|
||||
b.WriteString("\nRe-running the installer applies each installer command above: it fetches the newest version on\nthe channel in effect and re-applies the bundle (release is the default). The channel\nis not persisted, so pass FELIS_VERSION_BOOTSTRAP=dev if this host tracks main. While\nthis repo is private, the one-liner above 404s without a token; the README's install\nsection has the token'd form that works. felis setup is not this path: on a completed\ninstall it opens the config console and installs nothing newer. Restart game servers\nafterwards.\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
@@ -299,12 +462,66 @@ func installerRef(byComponent map[string]updates.Action) string {
|
||||
}
|
||||
|
||||
// isReleaseTag reports whether v was read from a stable vX.Y.Z tag, the only refs
|
||||
// release.yml publishes a binary for.
|
||||
// release.yml publishes a binary for. A source build stamps v0.0.0+g<commit>, which
|
||||
// names no tag, so build metadata disqualifies a version too.
|
||||
func isReleaseTag(v updates.Version) bool {
|
||||
s := v.String()
|
||||
if !strings.HasPrefix(s, "v") || v.IsPrerelease() {
|
||||
if !strings.HasPrefix(s, "v") || v.IsPrerelease() || strings.Contains(s, "+") {
|
||||
return false
|
||||
}
|
||||
_, err := updates.Parse(s)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// updateWindowTimeout bounds the maintenance-window read, so an unreachable
|
||||
// database costs the report a line and never the report itself.
|
||||
const updateWindowTimeout = 3 * time.Second
|
||||
|
||||
// readUpdateWindow reads the maintenance window the panel stores
|
||||
// (platform_settings "update_window"). Felis applies nothing on its own: this
|
||||
// command is the window's consumer, showing it and warning before an apply
|
||||
// outside it. A missing row is an unset window.
|
||||
func readUpdateWindow(ctx context.Context, cfgPath string) (updates.Window, error) {
|
||||
cfg, err := config.Load(cfgPath)
|
||||
if err != nil {
|
||||
return updates.Window{}, err
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(ctx, updateWindowTimeout)
|
||||
defer cancel()
|
||||
conn, err := pgx.Connect(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
return updates.Window{}, err
|
||||
}
|
||||
defer conn.Close(context.Background())
|
||||
var raw []byte
|
||||
err = conn.QueryRow(ctx, `SELECT value FROM platform_settings WHERE key = 'update_window'`).Scan(&raw)
|
||||
if errors.Is(err, pgx.ErrNoRows) {
|
||||
return updates.Window{}, nil
|
||||
}
|
||||
if err != nil {
|
||||
return updates.Window{}, err
|
||||
}
|
||||
var w updates.Window
|
||||
if err := json.Unmarshal(raw, &w); err != nil {
|
||||
return updates.Window{}, fmt.Errorf("stored window: %w", err)
|
||||
}
|
||||
return w, nil
|
||||
}
|
||||
|
||||
// renderWindowLine is the report's first line: where now sits against the
|
||||
// maintenance window.
|
||||
func renderWindowLine(w updates.Window, err error, now time.Time) string {
|
||||
const layout = "2006-01-02 15:04 MST"
|
||||
switch {
|
||||
case err != nil:
|
||||
return fmt.Sprintf("Maintenance window: unknown (%v).\n", err)
|
||||
case w.Start.IsZero() || w.End.IsZero():
|
||||
return "Maintenance window: not set; apply whenever suits you.\n"
|
||||
case w.Contains(now):
|
||||
return fmt.Sprintf("Maintenance window: open now, until %s.\n", w.End.Local().Format(layout))
|
||||
case now.Before(w.Start):
|
||||
return fmt.Sprintf("Maintenance window: opens %s, until %s. Felis applies nothing on its own; run the apply commands inside it.\n", w.Start.Local().Format(layout), w.End.Local().Format(layout))
|
||||
default:
|
||||
return fmt.Sprintf("Maintenance window: ended %s; set a new one in the panel before applying.\n", w.End.Local().Format(layout))
|
||||
}
|
||||
}
|
||||
@@ -1,9 +1,11 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/updater"
|
||||
"felis.lolicon.best/internal/updates"
|
||||
@@ -199,3 +201,133 @@ func TestApplyGuidanceReadsTheInstallerAtTheReleaseTag(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Every selector in the table is a flag: the FlagSet is built from the table.
|
||||
func TestUpdateSelectorsAreFlags(t *testing.T) {
|
||||
var out, errb strings.Builder
|
||||
if code := cmdUpdate([]string{"-h"}, &out, &errb); code != 2 {
|
||||
t.Fatalf("-h exit = %d, want 2", code)
|
||||
}
|
||||
for _, target := range updateTargets {
|
||||
if !strings.Contains(errb.String(), "-"+target.selector+"\n") {
|
||||
t.Errorf("usage has no -%s flag:\n%s", target.selector, errb.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// k3s and cloudflared move only when the re-run is told to; PostgreSQL is the package
|
||||
// manager's, so its guidance carries no installer trailer.
|
||||
func TestApplyGuidanceForHostDependencies(t *testing.T) {
|
||||
notify := func(c string) updater.Result {
|
||||
return planResult([]updates.Action{{Component: c, Kind: updates.ActionNotify, LatestKnown: true}})
|
||||
}
|
||||
for _, sel := range []string{"k3s", "cloudflared"} {
|
||||
out := renderApplyGuidance(notify(sel), map[string]bool{sel: true}, false)
|
||||
if !strings.Contains(out, "sudo FELIS_UPGRADE_DEPS=1 bash") || !strings.Contains(out, "Re-running the installer") {
|
||||
t.Errorf("--%s guidance must re-run the installer with FELIS_UPGRADE_DEPS=1:\n%s", sel, out)
|
||||
}
|
||||
}
|
||||
jre := renderApplyGuidance(notify("jre"), map[string]bool{"jre": true}, false)
|
||||
if !strings.Contains(jre, "| sudo bash") || strings.Contains(jre, "FELIS_UPGRADE_DEPS") {
|
||||
t.Errorf("--jre guidance is the plain installer re-run:\n%s", jre)
|
||||
}
|
||||
pg := renderApplyGuidance(notify("postgresql"), map[string]bool{"postgres": true}, false)
|
||||
if !strings.Contains(pg, "apt-get install --only-upgrade") || strings.Contains(pg, "Re-running the installer") {
|
||||
t.Errorf("--postgres guidance is the package manager, without the installer trailer:\n%s", pg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRenderNotesHonoursSelectors(t *testing.T) {
|
||||
notes := map[string]string{"postgresql": "PostgreSQL 13 reached end of life on 2025-11-13"}
|
||||
if out := renderNotes(notes, nil); !strings.Contains(out, "postgresql") || !strings.Contains(out, "note: PostgreSQL 13 reached end of life") {
|
||||
t.Errorf("unfiltered notes = %q", out)
|
||||
}
|
||||
if out := renderNotes(notes, map[string]bool{"postgres": true}); !strings.Contains(out, "end of life") {
|
||||
t.Errorf("--postgres must show its note, got %q", out)
|
||||
}
|
||||
if out := renderNotes(notes, map[string]bool{"velocity": true}); out != "" {
|
||||
t.Errorf("--velocity must not show the postgresql note, got %q", out)
|
||||
}
|
||||
}
|
||||
|
||||
// A source build's v0.0.0+g<commit> names no tag, so the installer one-liner has to
|
||||
// fall back to main instead of a 404ing ref.
|
||||
func TestInstallerRefNamesATag(t *testing.T) {
|
||||
v := func(s string) updates.Version {
|
||||
t.Helper()
|
||||
x, err := updates.Parse(s)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return x
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
api updates.Action
|
||||
want string
|
||||
}{
|
||||
{"newest release", updates.Action{Current: v("v1.2.0"), Latest: v("v1.3.0"), LatestKnown: true}, "v1.3.0"},
|
||||
{"feed down, host on a release", updates.Action{Current: v("v1.2.0")}, "v1.2.0"},
|
||||
{"source build", updates.Action{Current: v("v0.0.0+gunknown")}, "main"},
|
||||
{"source build with commit", updates.Action{Current: v("v0.0.0+g1a2b3c4")}, "main"},
|
||||
{"prerelease", updates.Action{Current: v("v1.3.0-rc.1")}, "main"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := installerRef(map[string]updates.Action{"felis-api": c.api}); got != c.want {
|
||||
t.Errorf("%s: installerRef = %q, want %q", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
if got := installerRef(nil); got != "main" {
|
||||
t.Errorf("no felis-api row: installerRef = %q, want main", got)
|
||||
}
|
||||
}
|
||||
|
||||
func mustVersion(t *testing.T, s string) updates.Version {
|
||||
t.Helper()
|
||||
v, err := updates.Parse(s)
|
||||
if err != nil {
|
||||
t.Fatalf("parse %q: %v", s, err)
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// The record the panel shows keeps every component with a state that cannot be
|
||||
// mistaken: a feed failure is "unknown" with its reason, an unreadable install is
|
||||
// listed after the plan, and each row carries the selector that prints its apply.
|
||||
func TestBuildStatusReport(t *testing.T) {
|
||||
res := planResult([]updates.Action{
|
||||
{Component: "felis-api", Current: mustVersion(t, "v0.4.0"), Latest: mustVersion(t, "v0.5.0"), LatestKnown: true, Kind: updates.ActionNotify},
|
||||
{Component: "velocity", Current: mustVersion(t, "3.4.0"), Latest: mustVersion(t, "3.4.0"), LatestKnown: true, Kind: updates.ActionNone},
|
||||
{Component: "k3s", Current: mustVersion(t, "v1.36.2+k3s1"), Kind: updates.ActionNone},
|
||||
{Component: "cloudflared", Current: mustVersion(t, "2026.6.1"), Latest: mustVersion(t, "2026.9.0"), LatestKnown: true, Kind: updates.ActionApply},
|
||||
{Component: "mc-lobby", Current: mustVersion(t, "1.21.4"), Kind: updates.ActionPinned},
|
||||
})
|
||||
res.RunResult.SourceErrors["k3s"] = errors.New("github: HTTP 403")
|
||||
res.GatherErrors["postgresql"] = errors.New("psql: not found")
|
||||
res.GatherErrors["jre"] = errors.New("release file missing")
|
||||
notes := map[string]string{"postgresql": "PostgreSQL 13 is past its end of life", "velocity": "pinned minor 3.4"}
|
||||
now := time.Date(2026, 9, 25, 3, 4, 5, 0, time.FixedZone("CST", 8*3600))
|
||||
|
||||
b, err := json.Marshal(buildStatusReport(res, notes, "v0.4.0", now))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := `{"checked_at":"2026-09-24T19:04:05Z","felis":"v0.4.0","components":[` +
|
||||
`{"name":"felis-api","current":"v0.4.0","latest":"v0.5.0","state":"available","selector":"panel"},` +
|
||||
`{"name":"velocity","current":"3.4.0","state":"current","selector":"velocity","note":"pinned minor 3.4"},` +
|
||||
`{"name":"k3s","current":"v1.36.2+k3s1","state":"unknown","selector":"k3s","error":"github: HTTP 403"},` +
|
||||
`{"name":"cloudflared","current":"2026.6.1","latest":"2026.9.0","state":"available","selector":"cloudflared"},` +
|
||||
`{"name":"mc-lobby","current":"1.21.4","state":"pinned"},` +
|
||||
`{"name":"jre","state":"unreadable","selector":"jre","error":"release file missing"},` +
|
||||
`{"name":"postgresql","state":"unreadable","selector":"postgres","note":"PostgreSQL 13 is past its end of life","error":"psql: not found"}]}`
|
||||
if string(b) != want {
|
||||
t.Errorf("status report =\n%s\nwant\n%s", b, want)
|
||||
}
|
||||
|
||||
// Nothing tracked still records an empty list, so the panel can tell "checked,
|
||||
// nothing to show" from a report that never arrived.
|
||||
b, _ = json.Marshal(buildStatusReport(planResult(nil), nil, "v0.4.0", now))
|
||||
if !strings.Contains(string(b), `"components":[]`) {
|
||||
t.Errorf("an empty check = %s, want an empty components list", b)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/updates"
|
||||
)
|
||||
|
||||
func TestRenderWindowLinePlacesNowAgainstTheWindow(t *testing.T) {
|
||||
prev := time.Local
|
||||
time.Local = time.FixedZone("CST", 8*3600)
|
||||
t.Cleanup(func() { time.Local = prev })
|
||||
|
||||
w := updates.Window{
|
||||
Start: time.Date(2026, 9, 26, 18, 0, 0, 0, time.UTC),
|
||||
End: time.Date(2026, 9, 26, 20, 0, 0, 0, time.UTC),
|
||||
}
|
||||
cases := []struct {
|
||||
name string
|
||||
w updates.Window
|
||||
err error
|
||||
now time.Time
|
||||
want string
|
||||
}{
|
||||
{"unreadable", updates.Window{}, errors.New("connection refused"), w.Start, "Maintenance window: unknown (connection refused).\n"},
|
||||
{"unset", updates.Window{}, nil, w.Start, "Maintenance window: not set; apply whenever suits you.\n"},
|
||||
{"half set", updates.Window{Start: w.Start}, nil, w.Start, "Maintenance window: not set; apply whenever suits you.\n"},
|
||||
{"at the opening instant", w, nil, w.Start, "Maintenance window: open now, until 2026-09-27 04:00 CST.\n"},
|
||||
{"before", w, nil, w.Start.Add(-time.Minute), "Maintenance window: opens 2026-09-27 02:00 CST, until 2026-09-27 04:00 CST. Felis applies nothing on its own; run the apply commands inside it.\n"},
|
||||
{"at the closing instant", w, nil, w.End, "Maintenance window: ended 2026-09-27 04:00 CST; set a new one in the panel before applying.\n"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := renderWindowLine(tc.w, tc.err, tc.now); got != tc.want {
|
||||
t.Fatalf("got %q\nwant %q", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -13,7 +13,6 @@ import (
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/offsite"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/store"
|
||||
@@ -245,7 +244,7 @@ func sendAlert(ctx context.Context, cfg *config.Config, state *watchdog.State, s
|
||||
if ref := cfg.SMTP.PasswordRef; ref != "" && os.Getenv(ref) != "" {
|
||||
password = os.Getenv(ref)
|
||||
}
|
||||
relay := &mail.SMTP{Host: cfg.SMTP.Host, Port: cfg.SMTP.Port, From: cfg.SMTP.From, Username: cfg.SMTP.Username, Password: password}
|
||||
relay := smtpRelay(cfg.SMTP, password)
|
||||
var errs []error
|
||||
for _, to := range state.Recipients {
|
||||
if err := relay.SendNotice(ctx, to, subject, body); err != nil {
|
||||
|
||||
+337
-52
@@ -42,6 +42,8 @@
|
||||
# through the handshake address instead of modern forwarding
|
||||
# (default: legacy18). Read once at Velocity start, so changing it
|
||||
# means re-running this script and restarting the proxy.
|
||||
# FELIS_VELOCITY_XMX maximum heap of the Velocity proxy, as <n>M or <n>G (default: 1G;
|
||||
# at least 256M). docs/operations.md sizes it by player count.
|
||||
# FELIS_VELOCITY_FORK_JAR path to a Felis-Legacy Velocity fork build to install as the
|
||||
# proxy instead of the stock download (default: unset, stock).
|
||||
# FELIS_VELOCITY_FORK_JAR_SHA256 expected sha256 of that jar. REQUIRED whenever the jar
|
||||
@@ -58,10 +60,13 @@
|
||||
# FELIS_GO_SHA256 sha256 of that version's linux tarball for this host's architecture.
|
||||
# REQUIRED for a non-default FELIS_GO_VERSION; the default's is pinned.
|
||||
# FELIS_K3S_VERSION k3s release a fresh install gets (default: v1.36.4+k3s1). An
|
||||
# installed k3s is left alone.
|
||||
# installed k3s is left alone unless FELIS_UPGRADE_DEPS=1.
|
||||
# FELIS_CLOUDFLARED_VERSION / FELIS_CLOUDFLARED_SHA256 cloudflared release installed
|
||||
# when none is present (default: 2026.9.1, digests pinned); the
|
||||
# sha256 is REQUIRED for any other version
|
||||
# FELIS_UPGRADE_DEPS 1 moves an installed k3s and cloudflared to the versions above
|
||||
# (k3s one minor version at a time; neither is ever downgraded) and
|
||||
# restarts cloudflared-felis onto the new binary (default: 0)
|
||||
# FELIS_REPO_URL git URL to build from (raw script mode only)
|
||||
# FELIS_VERSION_BOOTSTRAP release|dev — which version to install (default: release).
|
||||
# release DOWNLOADS the prebuilt felis binary published for the newest
|
||||
@@ -90,10 +95,11 @@
|
||||
# felis.toml [archive] local_path (default: /var/lib/felis/archives)
|
||||
# FELIS_WORLDS_HOST_PATH node directory holding the world volumes (on the k3s this
|
||||
# installer provisions: /var/lib/rancher/k3s/storage). Setting it
|
||||
# enables the daily retention reaper, which archives and then deletes
|
||||
# worlds idle beyond the retention window, and grants the reaper's
|
||||
# uid (1000) traverse access to that root — k3s ships it 0700
|
||||
# root:root (default: unset = no reaper)
|
||||
# lets the daily reaper also archive and then delete worlds idle
|
||||
# for 15 days (without it the reaper only deletes backups past
|
||||
# their expiry and leaves every world); the reaper reads that
|
||||
# root as root, so it keeps k3s's own 0700 root:root
|
||||
# (default: unset = worlds are never reaped)
|
||||
# FELIS_REGISTRY_STORAGE / FELIS_UPLOADS_STORAGE / FELIS_BACKUP_STORAGE capacity the
|
||||
# registry, uploads and world-archive PVCs request on first install
|
||||
# (defaults: 10Gi, 5Gi, 10Gi). An existing claim keeps its size; on
|
||||
@@ -167,9 +173,12 @@ FELIS_BACKUP_STORAGE="${FELIS_BACKUP_STORAGE:-}"
|
||||
# Retention is opt-in because it DELETES worlds (after a verified archive): point this at the
|
||||
# node directory the world volumes live under. On the k3s this installer provisions that is
|
||||
# /var/lib/rancher/k3s/storage — the reaper resolves each PVC's local-path directory exactly
|
||||
# from its volumeName. Left unset, no reaper CronJob renders and archives accumulate until
|
||||
# the backup PVC fills (then backups fail loudly; nothing is deleted).
|
||||
# from its volumeName. Left unset, the reaper CronJob renders retention-only: backups past
|
||||
# their expiry are still deleted daily, and no world is ever archived or deleted.
|
||||
FELIS_WORLDS_HOST_PATH="${FELIS_WORLDS_HOST_PATH:-}"
|
||||
# k3s's local-path provisioner root. It appears with the first volume the provisioner
|
||||
# creates, which on a fresh install is after the reaper's PV has been applied.
|
||||
K3S_STORAGE_ROOT="/var/lib/rancher/k3s/storage"
|
||||
# Control-plane database backups (felis db backup): a daily timer bundles pg_dump with the
|
||||
# /etc/felis state a rebuild needs, and every upgrade that has migrations to apply snapshots
|
||||
# the database first (felis migrate up). The directory sits outside /var/lib/rancher on
|
||||
@@ -211,6 +220,7 @@ FELIS_NANO_PROXY_CIDR="${FELIS_NANO_PROXY_CIDR:-}"
|
||||
# needs this. Overridable because adding a second 1.8 backend otherwise means editing this
|
||||
# script; it is still a restart-time list, not one that follows the CRs.
|
||||
FELIS_LEGACY_FORWARDING_SERVERS="${FELIS_LEGACY_FORWARDING_SERVERS:-legacy18}"
|
||||
FELIS_VELOCITY_XMX="${FELIS_VELOCITY_XMX:-1G}"
|
||||
# The Go tarball is unpacked and run as root, so the default version is pinned by the sha256
|
||||
# go.dev/dl publishes for each architecture install_go_toolchain handles. Move all three
|
||||
# together; any other FELIS_GO_VERSION has to bring its own FELIS_GO_SHA256.
|
||||
@@ -221,18 +231,20 @@ FELIS_GO_VERSION="${FELIS_GO_VERSION:-$GO_PINNED_VERSION}"
|
||||
FELIS_GO_SHA256="${FELIS_GO_SHA256:-}"
|
||||
# cloudflared runs as root on the edge, so it gets the same treatment: a pinned release and
|
||||
# the sha256 GitHub lists for each asset. A different FELIS_CLOUDFLARED_VERSION has to bring
|
||||
# its own FELIS_CLOUDFLARED_SHA256. install_cloudflared only runs when the binary is absent;
|
||||
# upgrading an installed one is `felis update`'s report plus a manual swap.
|
||||
# its own FELIS_CLOUDFLARED_SHA256. An installed binary is replaced only under
|
||||
# FELIS_UPGRADE_DEPS=1; `felis update --cloudflared` reports when that would change it.
|
||||
CLOUDFLARED_PINNED_VERSION="2026.9.1"
|
||||
CLOUDFLARED_PINNED_SHA256_AMD64="03f1f25d1cc93b9ad6c60569d44060bc4f17ed97075760ed8cfca4b12dcd68cc"
|
||||
CLOUDFLARED_PINNED_SHA256_ARM64="3d97437c71848bd8df68041e12436b484a661d95073ea1937f01a845ce88faa3"
|
||||
CLOUDFLARED_PINNED_SHA256_ARM="093ffa3638ab2b636de63c43a8c68f96a69cf71f9699dd8277a91b160b0f4fc0"
|
||||
FELIS_CLOUDFLARED_VERSION="${FELIS_CLOUDFLARED_VERSION:-$CLOUDFLARED_PINNED_VERSION}"
|
||||
FELIS_CLOUDFLARED_SHA256="${FELIS_CLOUDFLARED_SHA256:-}"
|
||||
CLOUDFLARED_BIN=/usr/local/bin/cloudflared
|
||||
# The k3s release a fresh install gets, and the tag its install script is read from. The
|
||||
# script checks the k3s binary against that release's sha256sum file, so pinning the tag
|
||||
# pins both. An installed k3s is never touched; see docs/troubleshooting.md for upgrades.
|
||||
# pins both. An installed k3s moves only under FELIS_UPGRADE_DEPS=1.
|
||||
FELIS_K3S_VERSION="${FELIS_K3S_VERSION:-v1.36.4+k3s1}"
|
||||
FELIS_UPGRADE_DEPS="${FELIS_UPGRADE_DEPS:-0}"
|
||||
# The in-cluster registry's image, by digest. It must equal platform.defaultRegistryImage
|
||||
# (internal/platform/identities.go, TestBootstrapPinsTheRegistryImage): the renderer puts
|
||||
# that ref in the Deployment, and this script caches and pins the same ref in containerd.
|
||||
@@ -320,6 +332,12 @@ PANEL_TLS_CERT="${STATE_DIR}/panel-tls.crt"
|
||||
PANEL_TLS_KEY="${STATE_DIR}/panel-tls.key"
|
||||
SRC_DIR="/opt/felis/src"
|
||||
HOST_BIN="/usr/local/bin/felis"
|
||||
# The binary this run replaced (keep_previous_host_binary), and whether the new one has
|
||||
# been put to use: once migrations start, or the nano service restarts onto it, the old
|
||||
# one no longer matches what is running and a failed run keeps the new one.
|
||||
HOST_BIN_PREV=""
|
||||
HOST_BIN_KEPT=0
|
||||
HOST_BIN_IN_USE=0
|
||||
# Felis's own build toolchain, not /usr/local/go: install_go_toolchain replaces whatever
|
||||
# version sits here, and an operator's Go at the conventional path is not ours to swap.
|
||||
GOROOT_DIR="/opt/felis/go"
|
||||
@@ -328,6 +346,8 @@ DB_BACKUP_SERVICE="/etc/systemd/system/felis-db-backup.service"
|
||||
DB_BACKUP_TIMER="/etc/systemd/system/felis-db-backup.timer"
|
||||
WATCHDOG_SERVICE="/etc/systemd/system/felis-watchdog.service"
|
||||
WATCHDOG_TIMER="/etc/systemd/system/felis-watchdog.timer"
|
||||
UPDATE_CHECK_SERVICE="/etc/systemd/system/felis-update-check.service"
|
||||
UPDATE_CHECK_TIMER="/etc/systemd/system/felis-update-check.timer"
|
||||
WATCHDOG_STATE="/var/lib/felis/watchdog/state.json"
|
||||
OFFSITE_ENV="${STATE_DIR}/offsite.env"
|
||||
OFFSITE_SERVICE="/etc/systemd/system/felis-offsite.service"
|
||||
@@ -352,6 +372,9 @@ SYSTEM_SERVER_IMAGES="${STATE_DIR}/system-server-images"
|
||||
PG_FIREWALL_RULES="${STATE_DIR}/postgres-firewall.nft"
|
||||
PG_FIREWALL_SERVICE="/etc/systemd/system/felis-postgres-firewall.service"
|
||||
JRE_DIR="/opt/felis/jre"
|
||||
# The gradle image the lobby and limbo Dockerfiles build their plugins in, digest
|
||||
# included; build_velocity_plugin runs the same one.
|
||||
PLUGIN_BUILD_IMAGE="gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01"
|
||||
K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}"
|
||||
K3S_BIN="${K3S_BIN_DIR}/k3s"
|
||||
# k3s's containerd mirror config, written by configure_registry_mirror. A variable
|
||||
@@ -426,7 +449,8 @@ on_error() {
|
||||
}
|
||||
|
||||
cleanup() {
|
||||
local id path unit
|
||||
local status=$? id path unit
|
||||
restore_previous_host_binary "$status"
|
||||
for unit in "${PKG_TIMERS_TO_RESTORE[@]-}"; do
|
||||
[ -n "$unit" ] || continue
|
||||
systemctl start "$unit" >/dev/null 2>&1 || true
|
||||
@@ -444,6 +468,36 @@ cleanup() {
|
||||
rm -f -- "$WATCHDOG_QUIET_FILE" 2>/dev/null || true
|
||||
}
|
||||
|
||||
# keep_previous_host_binary copies the felis binary this run is about to replace, once. A
|
||||
# run that fails before the new binary is in use puts it back (restore_previous_host_binary):
|
||||
# until then the old binary, the old cluster and the unmigrated database still agree, while
|
||||
# a new binary left behind runs the host timers against a schema it was not built for, and
|
||||
# `felis setup` with it would migrate the database under the old control plane.
|
||||
keep_previous_host_binary() {
|
||||
[ "$HOST_BIN_KEPT" = 0 ] || return 0
|
||||
HOST_BIN_KEPT=1
|
||||
[ -x "$HOST_BIN" ] || return 0
|
||||
HOST_BIN_PREV="${HOST_BIN}.prev"
|
||||
rm -f "$HOST_BIN_PREV"
|
||||
cp "$HOST_BIN" "$HOST_BIN_PREV"
|
||||
}
|
||||
|
||||
restore_previous_host_binary() { # exit-status
|
||||
[ -n "$HOST_BIN_PREV" ] && [ -f "$HOST_BIN_PREV" ] || return 0
|
||||
if [ "$1" -ne 0 ] && [ "$HOST_BIN_IN_USE" != 1 ]; then
|
||||
# install(1) onto a fresh file, as everywhere else HOST_BIN is written (SELinux label).
|
||||
rm -f "$HOST_BIN"
|
||||
if install -m 0755 "$HOST_BIN_PREV" "$HOST_BIN"; then
|
||||
command -v restorecon >/dev/null 2>&1 && restorecon "$HOST_BIN" >/dev/null 2>&1 || true
|
||||
warn "restored the previous felis binary at ${HOST_BIN}; the database was not migrated, so rerunning the installer picks up where this run stopped"
|
||||
else
|
||||
warn "could not restore the previous felis binary; it is at ${HOST_BIN_PREV}"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
rm -f -- "$HOST_BIN_PREV"
|
||||
}
|
||||
|
||||
remember_temp() { TEMP_PATHS+=("$1"); }
|
||||
remember_container() { DOCKER_CONTAINERS+=("$1"); }
|
||||
|
||||
@@ -641,9 +695,14 @@ wait_for_pkg_locks() {
|
||||
done
|
||||
}
|
||||
|
||||
# NEEDRESTART_SUSPEND keeps Ubuntu's needrestart hook from restarting services after
|
||||
# the install's own apt runs. It would restart felis-velocity whenever a rerun
|
||||
# happens to install or upgrade a package (the running JVM maps files the rerun
|
||||
# replaces), dropping every player on a rerun that changed nothing the proxy runs.
|
||||
# The installer restarts what it changes itself.
|
||||
apt_get() {
|
||||
wait_for_pkg_locks
|
||||
DEBIAN_FRONTEND=noninteractive apt-get \
|
||||
NEEDRESTART_SUSPEND=1 DEBIAN_FRONTEND=noninteractive apt-get \
|
||||
-o DPkg::Lock::Timeout="$PKG_LOCK_TIMEOUT" \
|
||||
"$@"
|
||||
}
|
||||
@@ -703,6 +762,34 @@ validate_settings() {
|
||||
pinned|latest) ;;
|
||||
*) die "FELIS_GAME_STACK must be pinned or latest (got '${FELIS_GAME_STACK}')" ;;
|
||||
esac
|
||||
[ "$(heap_megabytes "$FELIS_VELOCITY_XMX")" -ge 256 ] \
|
||||
|| die "FELIS_VELOCITY_XMX must be a heap size of at least 256M, written <n>M or <n>G (got '${FELIS_VELOCITY_XMX}')"
|
||||
case "$FELIS_UPGRADE_DEPS" in
|
||||
0|1) ;;
|
||||
*) die "FELIS_UPGRADE_DEPS must be 0 or 1 (got '${FELIS_UPGRADE_DEPS}')" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# version_newer reports whether version $1 sorts after $2 (a leading v is ignored).
|
||||
version_newer() {
|
||||
local a="${1#v}" b="${2#v}"
|
||||
[ "$a" != "$b" ] && [ "$(printf '%s\n%s\n' "$a" "$b" | sort -V | tail -n 1)" = "$a" ]
|
||||
}
|
||||
|
||||
# heap_megabytes prints a JVM heap size written <n>M or <n>G in megabytes, or 0 for any
|
||||
# other spelling.
|
||||
heap_megabytes() {
|
||||
local n="${1%?}"
|
||||
case "$n" in
|
||||
""|*[!0-9]*|0*) echo 0; return ;;
|
||||
esac
|
||||
# Six digits of gigabytes is far past any host; the cap keeps the arithmetic in range.
|
||||
[ "${#n}" -le 6 ] || { echo 0; return; }
|
||||
case "$1" in
|
||||
*[Mm]) echo "$n" ;;
|
||||
*[Gg]) echo "$((n * 1024))" ;;
|
||||
*) echo 0 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# validate_offsite_settings checks the FELIS_OFFSITE_* inputs before anything is
|
||||
@@ -884,9 +971,25 @@ install_base() {
|
||||
}
|
||||
|
||||
install_cloudflared() {
|
||||
if command -v cloudflared >/dev/null 2>&1; then
|
||||
ok "cloudflared already installed"
|
||||
return 0
|
||||
local current="" path
|
||||
if path="$(command -v cloudflared 2>/dev/null)"; then
|
||||
current="$(cloudflared --version 2>/dev/null | awk '{ for (i = 1; i < NF; i++) if ($i == "version") { print $(i + 1); exit } }')"
|
||||
if [ "$current" = "$FELIS_CLOUDFLARED_VERSION" ]; then
|
||||
ok "cloudflared ${current} already installed"
|
||||
return 0
|
||||
fi
|
||||
if [ "$FELIS_UPGRADE_DEPS" != 1 ]; then
|
||||
ok "cloudflared ${current:-(version unreadable)} already installed; this release pins ${FELIS_CLOUDFLARED_VERSION} (FELIS_UPGRADE_DEPS=1 moves it)"
|
||||
return 0
|
||||
fi
|
||||
if [ "$path" != "$CLOUDFLARED_BIN" ]; then
|
||||
warn "cloudflared at ${path} was not installed by Felis; upgrade it the way it was installed"
|
||||
return 0
|
||||
fi
|
||||
if [ -n "$current" ] && version_newer "$current" "$FELIS_CLOUDFLARED_VERSION"; then
|
||||
ok "cloudflared ${current} is newer than the pinned ${FELIS_CLOUDFLARED_VERSION}; left as it is"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
local machine arch url tmp want have
|
||||
machine="$(uname -m)"
|
||||
@@ -908,9 +1011,14 @@ install_cloudflared() {
|
||||
rm -f "$tmp"
|
||||
die "cloudflared-linux-${arch} ${FELIS_CLOUDFLARED_VERSION} hashes to ${have}, expected ${want}; refusing to install it"
|
||||
fi
|
||||
install -m 0755 "$tmp" /usr/local/bin/cloudflared
|
||||
install -m 0755 "$tmp" "$CLOUDFLARED_BIN"
|
||||
rm -f "$tmp"
|
||||
ok "cloudflared installed ($(cloudflared --version | head -n 1))"
|
||||
# The running tunnel keeps the old binary mapped until it restarts.
|
||||
if [ -n "$current" ] && systemctl is-active --quiet cloudflared-felis 2>/dev/null; then
|
||||
systemctl restart cloudflared-felis
|
||||
ok "cloudflared-felis restarted onto ${FELIS_CLOUDFLARED_VERSION}"
|
||||
fi
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1036,16 +1144,19 @@ install_k3s() {
|
||||
configure_k3s_firewall
|
||||
|
||||
if [ -x "$K3S_BIN" ]; then
|
||||
ok "k3s already installed at ${K3S_BIN}"
|
||||
local current
|
||||
current="$("$K3S_BIN" --version 2>/dev/null | awk 'NR == 1 { print $3 }')"
|
||||
if [ "$current" = "$FELIS_K3S_VERSION" ]; then
|
||||
ok "k3s ${current} already installed at ${K3S_BIN}"
|
||||
elif [ "$FELIS_UPGRADE_DEPS" != 1 ]; then
|
||||
ok "k3s ${current:-(version unreadable)} already installed at ${K3S_BIN}; this release pins ${FELIS_K3S_VERSION} (FELIS_UPGRADE_DEPS=1 moves it)"
|
||||
elif k3s_upgrade_allowed "$current" "$FELIS_K3S_VERSION"; then
|
||||
log "upgrading k3s ${current} to ${FELIS_K3S_VERSION}; running pods keep running while it restarts"
|
||||
run_k3s_installer
|
||||
fi
|
||||
else
|
||||
log "installing k3s ${FELIS_K3S_VERSION} into ${K3S_BIN_DIR} (no traefik/servicelb/metrics-server)"
|
||||
# The script from the release's own tag rather than get.k3s.io, which serves whatever
|
||||
# master holds today. '+' is literal in a URL path, so the tag needs no escaping.
|
||||
curl -sfL --retry 5 --retry-delay 2 "https://raw.githubusercontent.com/k3s-io/k3s/${FELIS_K3S_VERSION}/install.sh" | \
|
||||
INSTALL_K3S_VERSION="$FELIS_K3S_VERSION" \
|
||||
INSTALL_K3S_BIN_DIR="$K3S_BIN_DIR" \
|
||||
INSTALL_K3S_EXEC="--disable traefik --disable servicelb --disable metrics-server --write-kubeconfig-mode 644" \
|
||||
sh -
|
||||
run_k3s_installer
|
||||
fi
|
||||
|
||||
[ -x "$K3S_BIN" ] || die "k3s installation completed but ${K3S_BIN} is missing"
|
||||
@@ -1056,6 +1167,37 @@ install_k3s() {
|
||||
wait_for_node_ready
|
||||
}
|
||||
|
||||
# The script from the release's own tag rather than get.k3s.io, which serves whatever
|
||||
# master holds today. '+' is literal in a URL path, so the tag needs no escaping. On an
|
||||
# installed k3s the same script replaces the binary in place and restarts the service.
|
||||
run_k3s_installer() {
|
||||
curl -sfL --retry 5 --retry-delay 2 "https://raw.githubusercontent.com/k3s-io/k3s/${FELIS_K3S_VERSION}/install.sh" | \
|
||||
INSTALL_K3S_VERSION="$FELIS_K3S_VERSION" \
|
||||
INSTALL_K3S_BIN_DIR="$K3S_BIN_DIR" \
|
||||
INSTALL_K3S_EXEC="--disable traefik --disable servicelb --disable metrics-server --write-kubeconfig-mode 644" \
|
||||
sh -
|
||||
}
|
||||
|
||||
# k3s_upgrade_allowed decides whether an installed k3s ($1) may move to $2. Kubernetes
|
||||
# supports upgrading one minor version at a time, so a larger jump stops the install
|
||||
# before anything changed; a newer installed k3s is left as it is.
|
||||
k3s_upgrade_allowed() {
|
||||
local current="$1" want="$2" cur_major cur_minor want_major want_minor rest
|
||||
IFS=. read -r cur_major cur_minor rest <<<"${current#v}"
|
||||
IFS=. read -r want_major want_minor rest <<<"${want#v}"
|
||||
case "${cur_major}${cur_minor}${want_major}${want_minor}" in
|
||||
""|*[!0-9]*) die "cannot compare the installed k3s '${current}' with ${want}; upgrade it by hand (docs/operations.md §4)" ;;
|
||||
esac
|
||||
if version_newer "$current" "$want"; then
|
||||
ok "k3s ${current} is newer than the pinned ${want}; left as it is"
|
||||
return 1
|
||||
fi
|
||||
if [ "$cur_major" != "$want_major" ] || [ "$((want_minor - cur_minor))" -gt 1 ]; then
|
||||
die "k3s ${current} -> ${want} skips a minor version, and Kubernetes upgrades one minor at a time. Rerun with FELIS_K3S_VERSION set to the newest v${cur_major}.$((cur_minor + 1)).x+k3sN release first (https://github.com/k3s-io/k3s/releases)"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# Waits for the (single) node to report Ready. Shared by the k3s install and the
|
||||
# registry-mirror restart below: both restart the agent, and a bootstrap that
|
||||
# proceeds early fails later with a misleading "not found"/timeout instead.
|
||||
@@ -1133,7 +1275,10 @@ EOF
|
||||
# digest ref the Deployment names and kubelet finds it. A docker save/import round
|
||||
# trip rewrites the manifest and would leave a copy the digest ref never matches.
|
||||
import_registry_image() {
|
||||
if k3s_cmd ctr images ls -q 2>/dev/null | grep -qxF "$(registry_image_containerd_ref)"; then
|
||||
local images
|
||||
# Read the list whole before matching; see postgres_installed for the SIGPIPE.
|
||||
images="$(k3s_cmd ctr images ls -q 2>/dev/null || true)"
|
||||
if grep -qxF "$(registry_image_containerd_ref)" <<<"$images"; then
|
||||
ok "registry image ${REGISTRY_IMAGE} already in k3s containerd"
|
||||
return 0
|
||||
fi
|
||||
@@ -1424,6 +1569,7 @@ download_release_binary() {
|
||||
# would carry the source SELinux label instead of type-transitioning to bin_t — see
|
||||
# build_nano_binary for the 203/EXEC this shape avoids. Same-directory staging does not
|
||||
# change that: install(1) still creates the destination and copies.
|
||||
keep_previous_host_binary
|
||||
rm -f "$HOST_BIN"
|
||||
install -m 0755 "$tmp" "$HOST_BIN"
|
||||
rm -f "$tmp"
|
||||
@@ -1558,6 +1704,7 @@ install_embedded_binary() {
|
||||
mkdir -p "$(dirname "$HOST_BIN")"
|
||||
if [ "$(readlink -f "$src")" != "$(readlink -f "$HOST_BIN" 2>/dev/null || true)" ]; then
|
||||
log "installing current felis binary onto the host (${HOST_BIN})"
|
||||
keep_previous_host_binary
|
||||
install -m 0755 "$src" "$HOST_BIN"
|
||||
else
|
||||
ok "host binary already installed at ${HOST_BIN}"
|
||||
@@ -1608,6 +1755,7 @@ build_image_from_source() {
|
||||
local cid
|
||||
cid="$(docker create "$FELIS_IMAGE")"
|
||||
remember_container "$cid"
|
||||
keep_previous_host_binary
|
||||
docker cp "${cid}:/usr/local/bin/felis" "$HOST_BIN"
|
||||
docker rm "$cid" >/dev/null
|
||||
chmod 0755 "$HOST_BIN"
|
||||
@@ -1981,7 +2129,8 @@ atomic_install_file() {
|
||||
|
||||
# build_velocity_plugin compiles plugins/velocity in the same gradle image the two
|
||||
# Dockerfiles use, and drops the jar where Velocity will look for it. Docker is the
|
||||
# toolchain here on purpose: the host needs no JDK and no gradle, only a JRE.
|
||||
# toolchain here on purpose: the host needs no JDK and no gradle, only a JRE. Gradle
|
||||
# checks every dependency against plugins/velocity/gradle/verification-metadata.xml.
|
||||
build_velocity_plugin() {
|
||||
log "building felis-velocity.jar (gradle in a container; the host gets no JDK)"
|
||||
prepare_velocity_layout
|
||||
@@ -1989,7 +2138,7 @@ build_velocity_plugin() {
|
||||
docker run --rm \
|
||||
-v "${GAME_STACK_DIR}:/src:z" \
|
||||
-w /src/plugins/velocity \
|
||||
gradle:8.14-jdk21 gradle --no-daemon clean build \
|
||||
"$PLUGIN_BUILD_IMAGE" gradle --no-daemon clean build \
|
||||
|| die "felis-velocity plugin build failed"
|
||||
local -a jars=( "${GAME_STACK_DIR}"/plugins/velocity/build/libs/felis-velocity-*.jar )
|
||||
[ "${#jars[@]}" -eq 1 ] && [ -f "${jars[0]}" ] \
|
||||
@@ -2371,6 +2520,10 @@ install_velocity_service() {
|
||||
# sees it -- unquoted, that spelling would hand java a stray "legacy112" argument and the unit
|
||||
# would not start. Quoting keeps the whole property one argv item.
|
||||
local legacy_forwarding_servers="${FELIS_LEGACY_FORWARDING_SERVERS}"
|
||||
# -Xms stays at 512M so a small proxy does not reserve its whole ceiling up front, unless
|
||||
# the ceiling itself is lower (the JVM refuses an initial heap above the maximum).
|
||||
local xmx="$FELIS_VELOCITY_XMX" xms="512M"
|
||||
[ "$(heap_megabytes "$xmx")" -ge 512 ] || xms="$xmx"
|
||||
cat > "$VELOCITY_SERVICE" <<EOF
|
||||
[Unit]
|
||||
Description=Felis Velocity proxy (Mojang authentication + modern forwarding)
|
||||
@@ -2382,7 +2535,7 @@ Type=simple
|
||||
User=${VELOCITY_USER}
|
||||
Group=${VELOCITY_USER}
|
||||
WorkingDirectory=${VELOCITY_DIR}
|
||||
ExecStart=${JRE_DIR}/bin/java -Xms512M -Xmx1G -XX:+UseG1GC -XX:+ParallelRefProcEnabled -XX:+AlwaysPreTouch -Dmojang.sessionserver=http://${api_ip}:8081/session/minecraft/hasJoined "-Dfelis.legacy-forwarding.servers=${legacy_forwarding_servers}" -jar ${VELOCITY_DIR}/velocity.jar
|
||||
ExecStart=${JRE_DIR}/bin/java -Xms${xms} -Xmx${xmx} -XX:+UseG1GC -XX:+ParallelRefProcEnabled -XX:+AlwaysPreTouch -Dmojang.sessionserver=http://${api_ip}:8081/session/minecraft/hasJoined "-Dfelis.legacy-forwarding.servers=${legacy_forwarding_servers}" -jar ${VELOCITY_DIR}/velocity.jar
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
NoNewPrivileges=yes
|
||||
@@ -2503,8 +2656,19 @@ init_postgres_data_dir() {
|
||||
fi
|
||||
}
|
||||
|
||||
# postgres_installed reports whether a PostgreSQL client and server unit are present.
|
||||
# The unit list is read into a variable before matching: `systemctl list-unit-files |
|
||||
# grep -q` dies of SIGPIPE under pipefail once grep stops reading a list longer than
|
||||
# one write, and the install then took the "not installed" branch on a host that has it.
|
||||
postgres_installed() {
|
||||
local units
|
||||
command -v psql >/dev/null 2>&1 || return 1
|
||||
units="$(systemctl list-unit-files 2>/dev/null)" || return 1
|
||||
grep -q '^postgresql' <<<"$units"
|
||||
}
|
||||
|
||||
install_postgres() {
|
||||
if command -v psql >/dev/null 2>&1 && systemctl list-unit-files 2>/dev/null | grep -q '^postgresql'; then
|
||||
if postgres_installed; then
|
||||
ok "postgresql already installed"
|
||||
else
|
||||
log "installing postgresql"
|
||||
@@ -2661,7 +2825,15 @@ load_or_make_secrets() {
|
||||
ok "reusing persisted secrets from ${SECRETS_ENV}"
|
||||
fi
|
||||
DB_PASSWORD="${DB_PASSWORD:-$(openssl rand -hex 24)}"
|
||||
# One felis-api internal token per caller (naming.CallerTokens), so each is scoped
|
||||
# to its own routes and a leak is contained to that caller: SERVICE_TOKEN is the
|
||||
# proxy's (felis-link.properties), LIMBO_TOKEN the login gate's, BUILD_TOKEN what a
|
||||
# build Job fetches its context with, OPS_TOKEN what `felis backup-now` presents.
|
||||
# `felis rotate-token <caller>` rewrites the matching line here.
|
||||
SERVICE_TOKEN="${SERVICE_TOKEN:-$(openssl rand -hex 32)}"
|
||||
LIMBO_TOKEN="${LIMBO_TOKEN:-$(openssl rand -hex 32)}"
|
||||
BUILD_TOKEN="${BUILD_TOKEN:-$(openssl rand -hex 32)}"
|
||||
OPS_TOKEN="${OPS_TOKEN:-$(openssl rand -hex 32)}"
|
||||
SESSION_SECRET="${SESSION_SECRET:-$(openssl rand -hex 32)}"
|
||||
# The Velocity modern-forwarding key. It is what makes a backend's UUID trustworthy:
|
||||
# the proxy does the Mojang handshake and HMACs the resulting profile with this key,
|
||||
@@ -2682,6 +2854,9 @@ load_or_make_secrets() {
|
||||
cat > "$SECRETS_ENV" <<EOF
|
||||
DB_PASSWORD=${DB_PASSWORD}
|
||||
SERVICE_TOKEN=${SERVICE_TOKEN}
|
||||
LIMBO_TOKEN=${LIMBO_TOKEN}
|
||||
BUILD_TOKEN=${BUILD_TOKEN}
|
||||
OPS_TOKEN=${OPS_TOKEN}
|
||||
SESSION_SECRET=${SESSION_SECRET}
|
||||
FORWARDING_SECRET=${FORWARDING_SECRET}
|
||||
REGISTRY_PLATFORM_TOKEN=${REGISTRY_PLATFORM_TOKEN}
|
||||
@@ -2833,7 +3008,7 @@ persisted_registry_block() {
|
||||
out="$(awk '
|
||||
/^[[:space:]]*\[/ { sect = $0; next }
|
||||
sect ~ /^[[:space:]]*\[registry\][[:space:]]*$/ &&
|
||||
/^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|max_concurrent_builds|user_uploads_context|user_uploads_max_bytes)[[:space:]]*=/ { print }
|
||||
/^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|build_disk_limit|build_user_namespaces|build_runtime_class|max_concurrent_builds|scan_fail_on|scan_fail_unfixed|scan_accept|user_uploads_context|user_uploads_max_bytes|context_max_bytes)[[:space:]]*=/ { print }
|
||||
sect ~ /^[[:space:]]*\[registry\.s3\][[:space:]]*$/ && /^[[:space:]]*[A-Za-z_]+[[:space:]]*=/ {
|
||||
if (!s3hdr) { printf "[registry.s3]\n"; s3hdr = 1 }
|
||||
print
|
||||
@@ -2933,6 +3108,8 @@ egress_mode = "${FELIS_EGRESS_MODE}"
|
||||
# into k3s by build_game_stack below, so setup never has to be told "build these first".
|
||||
login_image = "${FELIS_LIMBO_IMAGE}"
|
||||
lobby_image = "${FELIS_LOBBY_IMAGE}"
|
||||
# The public port players connect on; the panel shows it in server addresses.
|
||||
game_port = ${FELIS_GAME_PORT}
|
||||
|
||||
[registry]
|
||||
url = "${REGISTRY_URL}"
|
||||
@@ -2997,6 +3174,8 @@ run_migrations() {
|
||||
# binary bundles the database into FELIS_DB_BACKUP_DIR first and refuses to migrate
|
||||
# if that fails; a fresh database has nothing to protect and is migrated directly.
|
||||
log "running database migrations (host binary -> 127.0.0.1)"
|
||||
# From here the database may move forward, and the binary that moved it stays.
|
||||
HOST_BIN_IN_USE=1
|
||||
"$HOST_BIN" migrate up -config "${STATE_DIR}/felis.host.toml" "${backup_flags[@]}"
|
||||
ok "migrations applied"
|
||||
}
|
||||
@@ -3025,7 +3204,7 @@ install_offsite_timer() {
|
||||
fi
|
||||
cat > "$OFFSITE_SERVICE" <<EOF
|
||||
[Unit]
|
||||
Description=Felis off-site copy (world archives and database bundles, encrypted, to the [offsite] bucket)
|
||||
Description=Felis off-site copy (world archives, database bundles, user registry images and submission uploads, encrypted, to the [offsite] bucket)
|
||||
After=network-online.target k3s.service postgresql.service felis-db-backup.service
|
||||
Wants=network-online.target
|
||||
|
||||
@@ -3089,6 +3268,46 @@ summary_offsite() {
|
||||
# memory, and mails the owners (their verified addresses, over the [smtp] relay) what
|
||||
# has stayed wrong long enough to matter. It runs on the host so a k3s that is down is
|
||||
# still reported. The first run happens now, so a broken unit shows up in this install.
|
||||
# The daily version check. Felis applies no update on its own; `felis update --record`
|
||||
# compares what this host runs with the newest upstream releases and stores the result,
|
||||
# which the panel's Updates page shows with the command that applies each update. It runs
|
||||
# on the host because that is where the installed versions are readable. The first check
|
||||
# runs in the background: it waits on the release feeds, and nothing in the install
|
||||
# depends on it.
|
||||
install_update_check_timer() {
|
||||
cat > "$UPDATE_CHECK_SERVICE" <<EOF
|
||||
[Unit]
|
||||
Description=Felis component version check (felis update --record)
|
||||
After=network-online.target postgresql.service k3s.service
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=${HOST_BIN} update --record -config ${STATE_DIR}/felis.host.toml
|
||||
TimeoutStartSec=5min
|
||||
Nice=10
|
||||
PrivateTmp=yes
|
||||
NoNewPrivileges=yes
|
||||
ProtectSystem=full
|
||||
EOF
|
||||
cat > "$UPDATE_CHECK_TIMER" <<EOF
|
||||
[Unit]
|
||||
Description=Daily Felis component version check
|
||||
|
||||
[Timer]
|
||||
OnCalendar=*-*-* 05:30:00
|
||||
RandomizedDelaySec=30min
|
||||
Persistent=true
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
EOF
|
||||
systemctl daemon-reload
|
||||
systemctl enable --now felis-update-check.timer
|
||||
systemctl start --no-block felis-update-check.service
|
||||
ok "version check: daily; the panel's Updates page shows what has a newer release (journalctl -u felis-update-check)"
|
||||
}
|
||||
|
||||
install_watchdog_timer() {
|
||||
local disks="/,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis" path
|
||||
for path in "$FELIS_WORLDS_HOST_PATH" "$FELIS_ARCHIVE_LOCAL_PATH" "$FELIS_DB_BACKUP_DIR"; do
|
||||
@@ -3260,6 +3479,34 @@ EOF
|
||||
ok "off-site copy: secrets in ${OFFSITE_ENV}"
|
||||
}
|
||||
|
||||
# Releases before the reaper ran as root granted uid 1000 traverse on the worlds root: an
|
||||
# ACL entry, or o+x where the host had no setfacl. uid 1000 is now the game servers' uid,
|
||||
# and the per-volume directories below that root are 0777, so the grant let a game process
|
||||
# (and, for o+x, every local account) reach any world by its directory name. Every run
|
||||
# takes it back: the ACL entry from any worlds root, the other-bits only from k3s's storage
|
||||
# root, which k3s ships 0700 root:root. A custom root keeps its mode, which may be the
|
||||
# operator's own.
|
||||
revoke_worlds_root_grant() {
|
||||
local dir
|
||||
for dir in "$K3S_STORAGE_ROOT" "$FELIS_WORLDS_HOST_PATH"; do
|
||||
[ -n "$dir" ] && [ -d "$dir" ] || continue
|
||||
if command -v getfacl >/dev/null 2>&1 && getfacl -cpn "$dir" 2>/dev/null | grep -q '^user:1000:'; then
|
||||
if setfacl -x u:1000 "$dir"; then
|
||||
log "revoked the old uid-1000 traverse grant on ${dir}"
|
||||
else
|
||||
warn "could not revoke the old uid-1000 traverse grant on ${dir}; remove it with: setfacl -x u:1000 ${dir}"
|
||||
fi
|
||||
fi
|
||||
done
|
||||
if [ -d "$K3S_STORAGE_ROOT" ] && [ -n "$(find "$K3S_STORAGE_ROOT" -maxdepth 0 -perm -o=x)" ]; then
|
||||
if chmod o-rwx "$K3S_STORAGE_ROOT"; then
|
||||
log "revoked the old world-traversable mode on ${K3S_STORAGE_ROOT} (back to k3s's 0700)"
|
||||
else
|
||||
warn "could not restore ${K3S_STORAGE_ROOT} to 0700; any local account can reach the world volumes below it: chmod o-rwx ${K3S_STORAGE_ROOT}"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
deploy_bundle() {
|
||||
local prev_api prev_operator prev_gate
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
@@ -3285,13 +3532,19 @@ deploy_bundle() {
|
||||
kube create namespace "$ns" --dry-run=client -o yaml | kube apply -f -
|
||||
done
|
||||
|
||||
log "provisioning felis-config + felis-service-token + felis-forwarding-secret + registry credentials + panel TLS secrets (out-of-band, never in the bundle)"
|
||||
log "provisioning felis-config + internal caller tokens + felis-forwarding-secret + registry credentials + panel TLS secrets (out-of-band, never in the bundle)"
|
||||
apply_felis_config_secrets
|
||||
# felis-api mounts all four caller tokens from the control namespace. The login
|
||||
# gate's and the build Job's are also applied into the namespace their pods run in
|
||||
# (a secretKeyRef is namespace-local); applying them here rather than leaving it to
|
||||
# `felis setup` means an upgrade has them in place before the new operator points
|
||||
# the login pod at felis-limbo-token. The proxy's and the ops token stay here only.
|
||||
apply_literal_secret "$CONTROL_NS" felis-service-token token "$SERVICE_TOKEN"
|
||||
# The build namespace needs the same token: the build Job's fetch initContainer
|
||||
# streams a submission's build context from the felis-api internal face, and a
|
||||
# secretKeyRef is namespace-local (a PVC cannot carry it across either).
|
||||
apply_literal_secret "$BUILD_NS" felis-service-token token "$SERVICE_TOKEN"
|
||||
apply_literal_secret "$CONTROL_NS" felis-limbo-token token "$LIMBO_TOKEN"
|
||||
apply_literal_secret "$MINECRAFT_NS" felis-limbo-token token "$LIMBO_TOKEN"
|
||||
apply_literal_secret "$CONTROL_NS" felis-build-token token "$BUILD_TOKEN"
|
||||
apply_literal_secret "$BUILD_NS" felis-build-token token "$BUILD_TOKEN"
|
||||
apply_literal_secret "$CONTROL_NS" felis-ops-token token "$OPS_TOKEN"
|
||||
# The forwarding key every backend verifies the proxy's handshake with. `felis setup`
|
||||
# replicates it into the minecraft namespace (ensureSecretReplica) before it creates
|
||||
# the pods that mount it; the operator injects it into EVERY backend, because Velocity's
|
||||
@@ -3304,6 +3557,8 @@ deploy_bundle() {
|
||||
--key="$PANEL_TLS_KEY" \
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
|
||||
revoke_worlds_root_grant
|
||||
|
||||
log "rendering + applying the control-plane bundle"
|
||||
local -a manifest_args=(
|
||||
--felis-image "$FELIS_IMAGE"
|
||||
@@ -3322,17 +3577,30 @@ deploy_bundle() {
|
||||
else
|
||||
manifest_args+=(--backup-pvc=)
|
||||
fi
|
||||
# Retention renders only when the operator names where the worlds live; the archive path
|
||||
# always travels with it because it must equal the [archive] local_path written above.
|
||||
# With an archive store the reaper always renders, since backups past their expiry have
|
||||
# to leave it; it reaps idle worlds only when the operator names where the worlds live.
|
||||
# The archive path must equal the [archive] local_path written above.
|
||||
if [ -n "$FELIS_BACKUP_PVC" ]; then
|
||||
manifest_args+=(--archive-local-path "$FELIS_ARCHIVE_LOCAL_PATH")
|
||||
if [ -z "$FELIS_WORLDS_HOST_PATH" ]; then
|
||||
log "idle-world retention is off: the daily reaper deletes expired backups and keeps every world (set FELIS_WORLDS_HOST_PATH=${K3S_STORAGE_ROOT} to reap worlds idle for 15 days)"
|
||||
fi
|
||||
fi
|
||||
if [ -n "$FELIS_WORLDS_HOST_PATH" ]; then
|
||||
log "retention enabled: the daily reaper will read worlds from ${FELIS_WORLDS_HOST_PATH}"
|
||||
# The reaper reads this root as root with DAC_OVERRIDE (platform.reaperPodSecurityContext)
|
||||
# through a static hostPath PV, so the host directory keeps k3s's own 0700 root:root and
|
||||
# needs no extra grant. It must exist, though: the PV declares type Directory.
|
||||
if [ ! -d "$FELIS_WORLDS_HOST_PATH" ]; then
|
||||
warn "worlds root ${FELIS_WORLDS_HOST_PATH} does not exist yet; the reaper CronJob cannot start until it does (hostPath type Directory)"
|
||||
if [ "$FELIS_WORLDS_HOST_PATH" = "$K3S_STORAGE_ROOT" ]; then
|
||||
# The provisioner would create it moments later with this same 0700 root:root;
|
||||
# creating it now keeps a fresh install from warning about its own default.
|
||||
install -d -m 0700 -o root -g root "$FELIS_WORLDS_HOST_PATH"
|
||||
else
|
||||
warn "worlds root ${FELIS_WORLDS_HOST_PATH} does not exist yet; the reaper CronJob cannot start until it does (hostPath type Directory)"
|
||||
fi
|
||||
fi
|
||||
manifest_args+=(--worlds-host-path "$FELIS_WORLDS_HOST_PATH" --archive-local-path "$FELIS_ARCHIVE_LOCAL_PATH")
|
||||
manifest_args+=(--worlds-host-path "$FELIS_WORLDS_HOST_PATH")
|
||||
fi
|
||||
local size
|
||||
size="$(pvc_size "$CONTROL_NS" registry "$FELIS_REGISTRY_STORAGE" FELIS_REGISTRY_STORAGE)"
|
||||
@@ -3354,6 +3622,12 @@ deploy_bundle() {
|
||||
die "control-plane rollout did not complete: ${d}"
|
||||
fi
|
||||
done
|
||||
# Before per-caller tokens the proxy's token was replicated into the workload
|
||||
# namespaces for the login gate and the build Jobs. The new operator and api no
|
||||
# longer reference those copies; leaving them would keep the proxy's credential
|
||||
# readable from namespaces that have no business with it.
|
||||
kube -n "$MINECRAFT_NS" delete secret felis-service-token --ignore-not-found
|
||||
kube -n "$BUILD_NS" delete secret felis-service-token --ignore-not-found
|
||||
}
|
||||
|
||||
# pvc_size <namespace> <claim> <wanted> <env name> prints the size to render the claim
|
||||
@@ -3504,27 +3778,34 @@ push_version_tag() {
|
||||
docker rmi "$versioned" >/dev/null 2>&1 || true
|
||||
}
|
||||
|
||||
# The login/lobby images use mutable :demo tags. Importing/pushing a replacement
|
||||
# updates containerd, but an existing StatefulSet template is byte-for-byte
|
||||
# unchanged and Kubernetes will not roll it. Recreate the two always-on system
|
||||
# pods so a convergent bootstrap actually starts the images it just built — but
|
||||
# only when the build changed: every player online is on one of these two, and a
|
||||
# rerun that rebuilt nothing has nothing to start. SYSTEM_SERVER_IMAGES records the
|
||||
# build each was last started on; it is written after the restarts, so a run that
|
||||
# died in between restarts them next time.
|
||||
# The login/lobby images are built under mutable :demo tags, and an existing
|
||||
# StatefulSet whose template still names that tag will not roll onto a new build by
|
||||
# itself. When a build changed, each system server is pinned to the digest its tag
|
||||
# names now (felis pin-images --system, after push_images_to_registry): the new ref
|
||||
# changes the template, the operator rolls the pod onto it, and the build each one
|
||||
# runs is written in its spec. A pin that fails (registry down) falls back to
|
||||
# recreating the pod, which picks the build up only while the spec names the bare
|
||||
# tag. Only a changed build does either: every player online is on one of these
|
||||
# two, and a rerun that rebuilt nothing has nothing to start. SYSTEM_SERVER_IMAGES
|
||||
# records the build each was last moved to; it is written after the loop, so a run
|
||||
# that died in between moves them next time.
|
||||
restart_existing_system_servers() {
|
||||
local name id pods next=""
|
||||
while read -r name id; do
|
||||
[ -n "$name" ] || continue
|
||||
next="${next}${name} ${id}"$'\n'
|
||||
pods="$(kube -n "$MINECRAFT_NS" get pod \
|
||||
-l "felis.lolicon.best/server=${name}" -o name 2>/dev/null || true)"
|
||||
[ -n "$pods" ] || continue
|
||||
if [ -n "$id" ] && grep -qxF "${name} ${id}" "$SYSTEM_SERVER_IMAGES" 2>/dev/null; then
|
||||
ok "${name} system server already runs this build; left running"
|
||||
continue
|
||||
fi
|
||||
log "restarting existing ${name} system server to pick up its imported image"
|
||||
if "$HOST_BIN" pin-images --system "$name" --namespace "$MINECRAFT_NS" \
|
||||
--registry "$REGISTRY_URL" --endpoint "$REGISTRY_PUSH_HOST"; then
|
||||
continue
|
||||
fi
|
||||
pods="$(kube -n "$MINECRAFT_NS" get pod \
|
||||
-l "felis.lolicon.best/server=${name}" -o name 2>/dev/null || true)"
|
||||
[ -n "$pods" ] || continue
|
||||
warn "could not pin the ${name} system server to its new build (above); restarting its pod, which starts the new build only if its spec.image still names the bare tag. Rerun the installer once the registry answers."
|
||||
kube -n "$MINECRAFT_NS" delete pod \
|
||||
-l "felis.lolicon.best/server=${name}" --wait=false
|
||||
done <<EOF
|
||||
@@ -3710,6 +3991,7 @@ build_nano_binary() {
|
||||
# cannot exec it and felis-nano dies with 203/EXEC. Creating the file fresh at the
|
||||
# destination lets the policy's type transition label it bin_t; restorecon is the belt.
|
||||
mkdir -p "$(dirname "$HOST_BIN")"
|
||||
keep_previous_host_binary
|
||||
rm -f "$HOST_BIN"
|
||||
install -m 0755 "$staged" "$HOST_BIN"
|
||||
rm -f "$staged"
|
||||
@@ -3852,6 +4134,7 @@ EOF
|
||||
systemctl enable felis-nano
|
||||
# restart, not `enable --now`: on a re-run the service is already active and --now would
|
||||
# leave the OLD binary running against the NEW unit. Converge means converge.
|
||||
HOST_BIN_IN_USE=1
|
||||
systemctl restart felis-nano
|
||||
# restart returns as soon as the process is forked. A config the new binary rejects, or a
|
||||
# file it cannot open, only shows once it has exited and the unit sits in auto-restart.
|
||||
@@ -3979,6 +4262,8 @@ main() {
|
||||
install_db_backup_timer
|
||||
# After the backup timer: its first bundle is part of the first copy.
|
||||
install_offsite_timer
|
||||
# After install_velocity: the check reads the installed proxy jar's version.
|
||||
install_update_check_timer
|
||||
# Last: its first run should see the platform as this install leaves it.
|
||||
install_watchdog_timer
|
||||
mark_bootstrap_done
|
||||
|
||||
+289
-21
@@ -350,6 +350,7 @@ listen = "0.0.0.0:8080"
|
||||
from = "[email protected]"
|
||||
username = "relay-user"
|
||||
password_ref = "smtp-password"
|
||||
require_tls = false
|
||||
|
||||
# Third-party Yggdrasil sources federated by the hasJoined multiplexer. Mojang is
|
||||
# always the code-owned identity anchor (premium-first), prepended in Go; sources here
|
||||
@@ -367,6 +368,7 @@ out="$(run_smtp "$smtp_dir")"
|
||||
expect "a configured [smtp] relay is carried" 'host = "mail.example"' "$out"
|
||||
expect "its port survives the carry" 'port = 587' "$out"
|
||||
expect "its credentials reference survives" 'password_ref = "smtp-password"' "$out"
|
||||
expect "an operator's require_tls survives" 'require_tls = false' "$out"
|
||||
case "$out" in
|
||||
*"#"*)
|
||||
echo "FAIL: the carry hoards comment lines:"; printf '%s\n' "$out"; fails=$((fails + 1)) ;;
|
||||
@@ -783,17 +785,22 @@ cfsum="$(printf 'stand-in cloudflared\n' | sha256sum | cut -d' ' -f1)"
|
||||
run_cf() { # FELIS_CLOUDFLARED_VERSION pinned-amd64-digest [FELIS_CLOUDFLARED_SHA256]
|
||||
FELIS_CLOUDFLARED_VERSION="$1" CLOUDFLARED_PINNED_VERSION=2026.9.1 CLOUDFLARED_PINNED_SHA256_AMD64="$2" \
|
||||
CLOUDFLARED_PINNED_SHA256_ARM64=unused CLOUDFLARED_PINNED_SHA256_ARM=unused \
|
||||
FELIS_CLOUDFLARED_SHA256="${3:-}" TMPDIR="$sdir" bash -c '
|
||||
FELIS_CLOUDFLARED_SHA256="${3:-}" TMPDIR="$sdir" CLOUDFLARED_BIN=/usr/local/bin/cloudflared \
|
||||
FELIS_UPGRADE_DEPS="${CF_UPGRADE:-0}" CF_PATH="${CF_PATH:-}" CF_HAVE="${CF_HAVE:-test}" \
|
||||
CF_ACTIVE="${CF_ACTIVE:-0}" bash -c '
|
||||
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||
log() { printf "LOG: %s\n" "$*"; }
|
||||
ok() { printf "OK: %s\n" "$*"; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
remember_temp() { :; }
|
||||
command() { return 1; }
|
||||
command() { [ -n "$CF_PATH" ] && [ "$1" = -v ] && [ "$2" = cloudflared ] && echo "$CF_PATH"; }
|
||||
uname() { echo x86_64; }
|
||||
cloudflared() { echo "cloudflared version test"; }
|
||||
cloudflared() { echo "cloudflared version ${CF_HAVE} (built 2026-01-01-0000 UTC)"; }
|
||||
curl() { printf "CURL: %s\n" "$*"; while [ "$#" -gt 1 ] && [ "$1" != "-o" ]; do shift; done
|
||||
printf "stand-in cloudflared\n" > "$2"; }
|
||||
install() { printf "INSTALL: %s\n" "$*"; }
|
||||
systemctl() { case "$1" in is-active) [ "$CF_ACTIVE" = 1 ] ;; *) printf "SYSTEMCTL: %s\n" "$*" ;; esac; }
|
||||
'"$(awk '/^version_newer\(\) \{/,/^}/' "$BS")"'
|
||||
'"$cfblock"'
|
||||
install_cloudflared'
|
||||
}
|
||||
@@ -806,12 +813,68 @@ case "$out" in *INSTALL:*) echo "FAIL: a refused cloudflared must not be install
|
||||
expect "another cloudflared version needs its own digest" "DIE: no pinned sha256 for cloudflared 2027.1.0" "$(run_cf 2027.1.0 "$cfsum")"
|
||||
expect "another cloudflared version installs with its digest" "INSTALL: -m 0755" "$(run_cf 2027.1.0 deadbeef "$cfsum")"
|
||||
|
||||
# An installed cloudflared moves only under FELIS_UPGRADE_DEPS=1, only when Felis put it
|
||||
# there, never backwards, and the running tunnel is restarted onto the new binary.
|
||||
out="$(CF_PATH=/usr/local/bin/cloudflared CF_HAVE=2026.9.1 run_cf 2026.9.1 "$cfsum")"
|
||||
expect "a cloudflared at the pin is left alone" "OK: cloudflared 2026.9.1 already installed" "$out"
|
||||
case "$out" in *CURL:*) echo "FAIL: a cloudflared at the pin must not be downloaded again"; fails=$((fails + 1)) ;; esac
|
||||
out="$(CF_PATH=/usr/local/bin/cloudflared CF_HAVE=2025.8.0 run_cf 2026.9.1 "$cfsum")"
|
||||
expect "an older cloudflared is reported without the flag" "this release pins 2026.9.1 (FELIS_UPGRADE_DEPS=1 moves it)" "$out"
|
||||
case "$out" in *CURL:*) echo "FAIL: an installed cloudflared must not move without FELIS_UPGRADE_DEPS=1"; fails=$((fails + 1)) ;; esac
|
||||
out="$(CF_UPGRADE=1 CF_ACTIVE=1 CF_PATH=/usr/local/bin/cloudflared CF_HAVE=2025.8.0 run_cf 2026.9.1 "$cfsum")"
|
||||
expect "FELIS_UPGRADE_DEPS=1 installs the pinned cloudflared over an older one" "INSTALL: -m 0755" "$out"
|
||||
expect "the running tunnel is restarted onto the new cloudflared" "SYSTEMCTL: restart cloudflared-felis" "$out"
|
||||
out="$(CF_UPGRADE=1 CF_PATH=/usr/local/bin/cloudflared CF_HAVE=2025.8.0 run_cf 2026.9.1 "$cfsum")"
|
||||
case "$out" in *SYSTEMCTL:*) echo "FAIL: a stopped cloudflared-felis must not be started by an upgrade"; fails=$((fails + 1)) ;; esac
|
||||
out="$(CF_UPGRADE=1 CF_PATH=/usr/bin/cloudflared CF_HAVE=2025.8.0 run_cf 2026.9.1 "$cfsum")"
|
||||
expect "a packaged cloudflared is left to its package manager" "WARN: cloudflared at /usr/bin/cloudflared was not installed by Felis" "$out"
|
||||
case "$out" in *CURL:*) echo "FAIL: a packaged cloudflared must not be overwritten"; fails=$((fails + 1)) ;; esac
|
||||
out="$(CF_UPGRADE=1 CF_PATH=/usr/local/bin/cloudflared CF_HAVE=2026.10.2 run_cf 2026.9.1 "$cfsum")"
|
||||
expect "a newer cloudflared is never downgraded" "cloudflared 2026.10.2 is newer than the pinned 2026.9.1" "$out"
|
||||
case "$out" in *CURL:*) echo "FAIL: a newer cloudflared must not be downgraded"; fails=$((fails + 1)) ;; esac
|
||||
|
||||
# k3s: the install script is read from the pinned tag, and told the same version.
|
||||
kblock="$(awk '/^install_k3s\(\) \{/,/^}/' "$BS")"
|
||||
kblock="$(awk '/^run_k3s_installer\(\) \{/,/^}/' "$BS")"
|
||||
expect "k3s's install script comes from the pinned tag" 'raw.githubusercontent.com/k3s-io/k3s/${FELIS_K3S_VERSION}/install.sh' "$kblock"
|
||||
expect "k3s's install script is told the pinned version" 'INSTALL_K3S_VERSION="$FELIS_K3S_VERSION"' "$kblock"
|
||||
case "$kblock" in *"https://get.k3s.io"*) echo "FAIL: get.k3s.io serves master's script; read it from the pinned tag"; fails=$((fails + 1)) ;; esac
|
||||
|
||||
# An installed k3s moves only under FELIS_UPGRADE_DEPS=1, one minor version at a time and
|
||||
# never backwards; the refusal names the release to go through first.
|
||||
kfake="$sdir/k3s"
|
||||
run_k3s() { # installed-version pinned-version [FELIS_UPGRADE_DEPS]
|
||||
printf '#!/bin/sh\necho "k3s version %s (0123abcd)"\necho "go version go1.26"\n' "$1" > "$kfake"
|
||||
chmod +x "$kfake"
|
||||
K3S_BIN="$kfake" FELIS_K3S_VERSION="$2" FELIS_UPGRADE_DEPS="${3:-0}" bash -c '
|
||||
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||
log() { printf "LOG: %s\n" "$*"; }
|
||||
ok() { printf "OK: %s\n" "$*"; }
|
||||
configure_k3s_firewall() { :; }
|
||||
run_k3s_installer() { printf "INSTALLER: %s\n" "$FELIS_K3S_VERSION"; }
|
||||
systemctl() { :; }
|
||||
wait_for_node_ready() { :; }
|
||||
'"$(awk '/^version_newer\(\) \{/,/^}/' "$BS")"'
|
||||
'"$(awk '/^k3s_upgrade_allowed\(\) \{/,/^}/' "$BS")"'
|
||||
'"$(awk '/^install_k3s\(\) \{/,/^}/' "$BS")"'
|
||||
install_k3s'
|
||||
}
|
||||
out="$(run_k3s v1.36.4+k3s1 v1.36.4+k3s1 1)"
|
||||
expect "a k3s at the pin is left alone" "OK: k3s v1.36.4+k3s1 already installed" "$out"
|
||||
case "$out" in *INSTALLER:*) echo "FAIL: a k3s at the pin must not be reinstalled"; fails=$((fails + 1)) ;; esac
|
||||
out="$(run_k3s v1.35.2+k3s1 v1.36.4+k3s1)"
|
||||
expect "an older k3s is reported without the flag" "this release pins v1.36.4+k3s1 (FELIS_UPGRADE_DEPS=1 moves it)" "$out"
|
||||
case "$out" in *INSTALLER:*) echo "FAIL: an installed k3s must not move without FELIS_UPGRADE_DEPS=1"; fails=$((fails + 1)) ;; esac
|
||||
expect "FELIS_UPGRADE_DEPS=1 moves k3s up one minor" "INSTALLER: v1.36.4+k3s1" "$(run_k3s v1.35.2+k3s1 v1.36.4+k3s1 1)"
|
||||
expect "FELIS_UPGRADE_DEPS=1 moves k3s to a newer patch" "INSTALLER: v1.36.4+k3s1" "$(run_k3s v1.36.1+k3s2 v1.36.4+k3s1 1)"
|
||||
out="$(run_k3s v1.34.6+k3s1 v1.36.4+k3s1 1)"
|
||||
expect "a k3s upgrade that skips a minor is refused" "DIE: k3s v1.34.6+k3s1 -> v1.36.4+k3s1 skips a minor version" "$out"
|
||||
expect "the refusal names the minor to go through first" "newest v1.35.x+k3sN release first" "$out"
|
||||
case "$out" in *INSTALLER:*) echo "FAIL: a skipping k3s upgrade must not run the installer"; fails=$((fails + 1)) ;; esac
|
||||
out="$(run_k3s v1.37.0+k3s1 v1.36.4+k3s1 1)"
|
||||
expect "a newer k3s is never downgraded" "OK: k3s v1.37.0+k3s1 is newer than the pinned v1.36.4+k3s1" "$out"
|
||||
case "$out" in *INSTALLER:*) echo "FAIL: a newer k3s must not be downgraded"; fails=$((fails + 1)) ;; esac
|
||||
expect "an unreadable k3s version stops the upgrade" "DIE: cannot compare the installed k3s 'dev'" "$(run_k3s dev v1.36.4+k3s1 1)"
|
||||
|
||||
# --- a private repo without a token fails with the hint instead of prompting -------------
|
||||
# git asks for credentials on /dev/tty, where a piped install would sit waiting. Every
|
||||
# network git call goes through git_auth, so the switch belongs there.
|
||||
@@ -875,6 +938,8 @@ run_bundle_flags() { # backup-pvc worlds-host-path
|
||||
myManifests() { printf "%s\n" "$@"; }
|
||||
setfacl() { printf "SETFACL %s\n" "$*"; }
|
||||
chmod() { printf "CHMOD %s\n" "$*"; }
|
||||
install() { printf "INSTALL %s\n" "$*"; }
|
||||
K3S_STORAGE_ROOT="${K3S_STORAGE_ROOT_T:-/var/lib/rancher/k3s/storage}"
|
||||
node_global_cidrs() { printf "203.0.113.7/32\n2001:db8::7/128\n"; }
|
||||
pvc_size() { case "$2" in registry) printf "20Gi\n" ;; esac; }
|
||||
run_bundle() {
|
||||
@@ -886,8 +951,12 @@ run_bundle_flags() { # backup-pvc worlds-host-path
|
||||
out="$(run_bundle_flags felis-backups '')"
|
||||
expect "a default install asks the renderer for the archive PVC" "--backup-pvc
|
||||
felis-backups" "$out"
|
||||
# data-durability-17: without a worlds root the reaper still renders, retention-only,
|
||||
# so backups past their expiry leave the store; it needs the archive mount for that.
|
||||
expect "a default install passes the archive mount so expired backups are deleted" "--archive-local-path
|
||||
/var/lib/felis/archives" "$out"
|
||||
case "$out" in
|
||||
*--worlds-host-path*) echo "FAIL: no reaper flags may render without FELIS_WORLDS_HOST_PATH"; fails=$((fails + 1)) ;;
|
||||
*--worlds-host-path*) echo "FAIL: no worlds root may render without FELIS_WORLDS_HOST_PATH"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
expect "a known registry size reaches the renderer" "--registry-storage
|
||||
@@ -898,6 +967,9 @@ esac
|
||||
|
||||
out="$(run_bundle_flags '' '')"
|
||||
expect "an emptied FELIS_BACKUP_PVC is the explicit no-backup shape" "--backup-pvc=" "$out"
|
||||
case "$out" in
|
||||
*--archive-local-path*) echo "FAIL: no archive mount may be passed without an archive PVC"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
# Game server egress excludes every private range already; the node's own public
|
||||
# addresses must be excluded too or a server can dial the panel NodePort on them.
|
||||
expect "every global node address is denied to game server egress (v4)" "--server-egress-deny-cidr
|
||||
@@ -935,6 +1007,14 @@ missing="/tmp/felis-worlds-root-must-not-exist-$$"
|
||||
out="$(run_bundle_flags felis-backups "$missing")"
|
||||
expect "a missing worlds root is warned about, not silently skipped" "WARN: worlds root $missing does not exist yet" "$out"
|
||||
|
||||
# On a fresh install the k3s default root does not exist until the provisioner's first
|
||||
# volume; the installer creates it with k3s's own mode instead of warning about its default.
|
||||
out="$(K3S_STORAGE_ROOT_T="$missing" run_bundle_flags felis-backups "$missing")"
|
||||
expect "a missing k3s storage root is created with k3s's mode" "INSTALL -d -m 0700 -o root -g root $missing" "$out"
|
||||
case "$out" in
|
||||
*WARN*) echo "FAIL: the installer's own default root must not be warned about: $out"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
# The reaper reads the root as root with DAC_OVERRIDE through a static PV, so an existing
|
||||
# root is left exactly as k3s shipped it: uid 1000 is now the game servers' uid, and a
|
||||
# traverse grant for it on the node's storage root would serve nothing but them.
|
||||
@@ -944,6 +1024,34 @@ case "$out" in
|
||||
*SETFACL*|*CHMOD*|*WARN*) echo "FAIL: an existing worlds root must get no grant and no warning: $out"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
# data-durability-18: an older release's traverse grant on the worlds root is taken back on
|
||||
# every run -- the ACL entry from any root, the other-bits only from k3s's own storage root.
|
||||
rvblock="$(awk '/^revoke_worlds_root_grant\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$rvblock" ] || { echo "FAIL: no revoke_worlds_root_grant found in $BS"; exit 1; }
|
||||
run_revoke() { # k3s-root worlds-root acl-dir
|
||||
K3S_STORAGE_ROOT="$1" FELIS_WORLDS_HOST_PATH="$2" ACL_DIR="$3" bash -c '
|
||||
log() { printf "LOG %s\n" "$*"; }
|
||||
warn() { printf "WARN %s\n" "$*"; }
|
||||
getfacl() { case "$*" in *"$ACL_DIR") printf "user::rwx\nuser:1000:--x\ngroup::---\n" ;; *) printf "user::rwx\ngroup::---\n" ;; esac; }
|
||||
setfacl() { printf "SETFACL %s\n" "$*"; }
|
||||
'"$rvblock"'
|
||||
revoke_worlds_root_grant' 2>&1
|
||||
}
|
||||
k3sroot="$(mktemp -d)"; custom="$(mktemp -d)"
|
||||
command chmod 0701 "$k3sroot"; command chmod 0755 "$custom"
|
||||
out="$(run_revoke "$k3sroot" "$custom" "$custom")"
|
||||
expect "the old ACL grant is revoked from a custom worlds root" "SETFACL -x u:1000 $custom" "$out"
|
||||
expect "the old o+x on k3s's storage root is revoked" "LOG revoked the old world-traversable mode on $k3sroot" "$out"
|
||||
mode_of() { stat -c %a "$1" 2>/dev/null || stat -f %Lp "$1"; }
|
||||
if [ "$(mode_of "$k3sroot")" = 700 ]; then echo "PASS k3s's storage root is back to 0700"; else echo "FAIL k3s's storage root is $(mode_of "$k3sroot"), want 700"; fails=$((fails + 1)); fi
|
||||
if [ "$(mode_of "$custom")" = 755 ]; then echo "PASS a custom worlds root keeps its own mode"; else echo "FAIL a custom worlds root was changed to $(mode_of "$custom")"; fails=$((fails + 1)); fi
|
||||
out="$(run_revoke "$k3sroot" "" "/nowhere")"
|
||||
case "$out" in
|
||||
*SETFACL*|*LOG*|*WARN*) echo "FAIL: a root with no old grant must be left alone: $out"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS a root with no old grant is left alone" ;;
|
||||
esac
|
||||
command rm -rf "$k3sroot" "$custom"
|
||||
|
||||
# --- the registry mirror writer -----------------------------------------------------------
|
||||
# k3s only consults registries.yaml at agent start, so a CONTENT change must restart k3s and
|
||||
# an identical file (every re-run) must restart nothing. The k3s restart is the expensive,
|
||||
@@ -1282,6 +1390,9 @@ trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"
|
||||
build_disk_limit = "20Gi"
|
||||
build_user_namespaces = "off"
|
||||
build_runtime_class = "gvisor"
|
||||
scan_fail_on = ["CRITICAL"]
|
||||
scan_fail_unfixed = true
|
||||
scan_accept = ["CVE-2021-35515", "CVE-2025-67030"]
|
||||
|
||||
[registry.s3]
|
||||
endpoint = "https://s3.example"
|
||||
@@ -1306,7 +1417,7 @@ run_write() { # out-file
|
||||
persisted_auth_source_blocks() { :; }
|
||||
. "$FNFILE"
|
||||
FELIS_ROOT_DOMAIN=r.example.com DB_USER=u DB_PASSWORD=p DB_NAME=d MINECRAFT_NS=minecraft \
|
||||
FELIS_EGRESS_MODE=nodeport FELIS_LIMBO_IMAGE=li FELIS_LOBBY_IMAGE=lo \
|
||||
FELIS_EGRESS_MODE=nodeport FELIS_LIMBO_IMAGE=li FELIS_LOBBY_IMAGE=lo FELIS_GAME_PORT=25570 \
|
||||
REGISTRY_URL=registry.felis.svc:5000 BUILD_NS=felis-build FELIS_ARCHIVE_LOCAL_PATH=/a \
|
||||
FELIS_OFFSITE_BUCKET= write_felis_toml "$OUT_TOML" 127.0.0.1'
|
||||
}
|
||||
@@ -1322,6 +1433,9 @@ expect "a re-run carries the trivy java-DB mirror" \
|
||||
expect "a re-run carries the build disk cap" 'build_disk_limit = "20Gi"' "$out"
|
||||
expect "a re-run carries the build user-namespace mode" 'build_user_namespaces = "off"' "$out"
|
||||
expect "a re-run carries the build runtime class" 'build_runtime_class = "gvisor"' "$out"
|
||||
expect "a re-run carries the scan gate's blocking severities" 'scan_fail_on = ["CRITICAL"]' "$out"
|
||||
expect "a re-run carries the scan gate's unfixed-vulnerability rule" 'scan_fail_unfixed = true' "$out"
|
||||
expect "a re-run carries the scan gate's accepted finding ids" 'scan_accept = ["CVE-2021-35515", "CVE-2025-67030"]' "$out"
|
||||
expect "a re-run carries the [registry.s3] uploads subtable" "[registry.s3]" "$out"
|
||||
expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "$out"
|
||||
expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out"
|
||||
@@ -1329,6 +1443,7 @@ expect "a re-run carries the archive retention window" 'retention = "30d"' "$out
|
||||
expect "a re-run carries the on-demand backup count" 'manual_keep = 3' "$out"
|
||||
expect "a re-run carries the on-demand backup cooldown" 'manual_cooldown = "1h"' "$out"
|
||||
expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out"
|
||||
expect "the panel learns the public game port" 'game_port = 25570' "$out"
|
||||
expect "a re-run keeps the off-site bucket, set apart from the next section" '[offsite]
|
||||
endpoint = "https://objects.example"
|
||||
bucket = "felis-offsite"
|
||||
@@ -1415,6 +1530,34 @@ expect "a failed first backup shows its log" "JOURNAL: pg_dump: connection refus
|
||||
expect "a failed first backup is a loud warning" "WARN: the first database backup failed" "$out"
|
||||
rm -rf "$tdir"
|
||||
|
||||
ublock="$(awk '/^install_update_check_timer\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$ublock" ] || { echo "FAIL: no install_update_check_timer found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$ublock" | wc -l)" -lt 60 ] \
|
||||
|| { echo "FAIL: the extracted block is not install_update_check_timer -- did its closing brace move?"; exit 1; }
|
||||
udir="$(mktemp -d)"
|
||||
out="$(UPDATE_CHECK_SERVICE="$udir/felis-update-check.service" UPDATE_CHECK_TIMER="$udir/felis-update-check.timer" \
|
||||
HOST_BIN=/usr/local/bin/felis STATE_DIR=/etc/felis bash -c '
|
||||
ok() { printf "OK: %s\n" "$*"; }; warn() { printf "WARN: %s\n" "$*"; }
|
||||
systemctl() { printf "SYSTEMCTL: %s\n" "$*"; }
|
||||
'"$ublock"'
|
||||
install_update_check_timer' 2>&1)"
|
||||
unit="$(cat "$udir/felis-update-check.service")"
|
||||
timer="$(cat "$udir/felis-update-check.timer")"
|
||||
expect "the version check records its result for the panel" \
|
||||
"ExecStart=/usr/local/bin/felis update --record -config /etc/felis/felis.host.toml" "$unit"
|
||||
expect "the version check is a oneshot" "Type=oneshot" "$unit"
|
||||
expect "the version check runs daily" "OnCalendar=*-*-* 05:30:00" "$timer"
|
||||
expect "a missed check catches up at boot" "Persistent=true" "$timer"
|
||||
expect "the version check timer is enabled" "SYSTEMCTL: enable --now felis-update-check.timer" "$out"
|
||||
expect "the first check runs without holding up the install" "SYSTEMCTL: start --no-block felis-update-check.service" "$out"
|
||||
expect "the install says where the result shows" "OK: version check: daily; the panel's Updates page" "$out"
|
||||
rm -rf "$udir"
|
||||
order="$(awk '/^main\(\) \{/,/^}/' "$BS" | grep -nE '^[[:space:]]*(install_velocity|install_update_check_timer)$' | tr '\n' ' ')"
|
||||
case "$order" in
|
||||
*install_velocity*install_update_check_timer*) echo "PASS the version check is installed after the proxy it reads" ;;
|
||||
*) echo "FAIL the version check must be installed after install_velocity: $order"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
wblock="$(awk '/^install_watchdog_timer\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$wblock" ] || { echo "FAIL: no install_watchdog_timer found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$wblock" | wc -l)" -lt 60 ] \
|
||||
@@ -1709,15 +1852,15 @@ esac
|
||||
# felis-api's transactions) when restarted, so a rerun that changed none of them must leave
|
||||
# them running, and one that changed a thing must still restart it.
|
||||
|
||||
vsblock="$(awk '/^install_velocity_service\(\) \{/,/^}/' "$BS"; awk '/^velocity_fingerprint\(\) \{/,/^}/' "$BS")"
|
||||
vsblock="$(awk '/^install_velocity_service\(\) \{/,/^}/' "$BS"; awk '/^velocity_fingerprint\(\) \{/,/^}/' "$BS"; awk '/^heap_megabytes\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$vsblock" ] || { echo "FAIL: no install_velocity_service found in $BS"; exit 1; }
|
||||
vdir="$(mktemp -d)"
|
||||
mkdir -p "$vdir/v/plugins/felis-link" "$vdir/jre"
|
||||
printf 'jar\n' > "$vdir/v/velocity.jar"
|
||||
printf 'plugin\n' > "$vdir/v/plugins/felis-velocity.jar"
|
||||
printf 'JAVA_VERSION="25"\n' > "$vdir/jre/release"
|
||||
run_velocity_service() { # is-active(0|1)
|
||||
ACTIVE="$1" VELOCITY_SERVICE="$vdir/unit" VELOCITY_DIR="$vdir/v" JRE_DIR="$vdir/jre" \
|
||||
run_velocity_service() { # is-active(0|1) [heap]
|
||||
ACTIVE="$1" FELIS_VELOCITY_XMX="${2:-1G}" VELOCITY_SERVICE="$vdir/unit" VELOCITY_DIR="$vdir/v" JRE_DIR="$vdir/jre" \
|
||||
VELOCITY_FINGERPRINT="$vdir/fp" VELOCITY_USER=felis-velocity FELIS_GAME_PORT=25565 \
|
||||
FELIS_LEGACY_FORWARDING_SERVERS='' bash -c '
|
||||
set -Eeuo pipefail
|
||||
@@ -1734,6 +1877,7 @@ run_velocity_service() { # is-active(0|1)
|
||||
}
|
||||
out="$(run_velocity_service 1)"
|
||||
expect "a proxy with no recorded start is restarted" "SYSTEMCTL restart felis-velocity" "$out"
|
||||
expect "the default heap is 512M..1G" "java -Xms512M -Xmx1G " "$(cat "$vdir/unit")"
|
||||
[ -s "$vdir/fp" ] && echo "PASS the restart records what the proxy runs" \
|
||||
|| { echo "FAIL no fingerprint was recorded after the restart"; fails=$((fails + 1)); }
|
||||
out="$(run_velocity_service 1)"
|
||||
@@ -1747,17 +1891,24 @@ expect "a changed plugin jar restarts the proxy" "SYSTEMCTL restart felis-veloci
|
||||
expect "a stopped proxy is started whatever the fingerprint" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 0)"
|
||||
printf 'JAVA_VERSION="25.0.1"\n' > "$vdir/jre/release"
|
||||
expect "a patched JRE restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1)"
|
||||
expect "a new heap size restarts the proxy" "SYSTEMCTL restart felis-velocity" "$(run_velocity_service 1 3G)"
|
||||
expect "the unit carries the new ceiling" "java -Xms512M -Xmx3G " "$(cat "$vdir/unit")"
|
||||
run_velocity_service 1 384M >/dev/null
|
||||
expect "a ceiling below 512M is also the initial heap" "java -Xms384M -Xmx384M " "$(cat "$vdir/unit")"
|
||||
rm -rf "$vdir"
|
||||
|
||||
ssblock="$(awk '/^restart_existing_system_servers\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$ssblock" ] || { echo "FAIL: no restart_existing_system_servers found in $BS"; exit 1; }
|
||||
sdir2="$(mktemp -d)"
|
||||
run_system_restart() { # limbo-id lobby-id pods(0|1)
|
||||
LIMBO_IMAGE_ID="$1" LOBBY_IMAGE_ID="$2" PODS="$3" SYSTEM_SERVER_IMAGES="$sdir2/state" \
|
||||
LOGIN_SERVER=login LOBBY_SERVER=lobby MINECRAFT_NS=minecraft bash -c '
|
||||
run_system_restart() { # limbo-id lobby-id pods(0|1) pin-exit
|
||||
LIMBO_IMAGE_ID="$1" LOBBY_IMAGE_ID="$2" PODS="$3" PIN_EXIT="${4:-0}" SYSTEM_SERVER_IMAGES="$sdir2/state" \
|
||||
LOGIN_SERVER=login LOBBY_SERVER=lobby MINECRAFT_NS=minecraft HOST_BIN=felis \
|
||||
REGISTRY_URL=registry.felis.svc:5000 REGISTRY_PUSH_HOST=127.0.0.1:5000 bash -c '
|
||||
set -Eeuo pipefail
|
||||
ok() { printf "OK: %s\n" "$*"; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
log() { :; }
|
||||
felis() { printf "FELIS %s\n" "$*"; return "$PIN_EXIT"; }
|
||||
kube() {
|
||||
case "$*" in
|
||||
*"get pod"*) [ "$PODS" = 1 ] && printf "pod/x-0\n" || true ;;
|
||||
@@ -1768,27 +1919,32 @@ run_system_restart() { # limbo-id lobby-id pods(0|1)
|
||||
restart_existing_system_servers'
|
||||
}
|
||||
out="$(run_system_restart sha256:aaa sha256:bbb 1)"
|
||||
expect "an unrecorded login pod is restarted" "delete pod -l felis.lolicon.best/server=login" "$out"
|
||||
expect "an unrecorded lobby pod is restarted" "delete pod -l felis.lolicon.best/server=lobby" "$out"
|
||||
expect "an unrecorded login build is pinned through the loopback registry" "FELIS pin-images --system login --namespace minecraft --registry registry.felis.svc:5000 --endpoint 127.0.0.1:5000" "$out"
|
||||
expect "an unrecorded lobby build is pinned" "FELIS pin-images --system lobby " "$out"
|
||||
case "$out" in
|
||||
*delete*) echo "FAIL a successful pin also deleted a pod; the operator rolls it"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS a successful pin leaves the roll to the operator" ;;
|
||||
esac
|
||||
out="$(run_system_restart sha256:aaa sha256:bbb 1)"
|
||||
case "$out" in
|
||||
*delete*) echo "FAIL an unchanged rebuild restarted a system server"; fails=$((fails + 1)) ;;
|
||||
*FELIS*|*delete*) echo "FAIL an unchanged rebuild moved a system server: $out"; fails=$((fails + 1)) ;;
|
||||
*"already runs this build"*) echo "PASS an unchanged rebuild leaves the system servers running" ;;
|
||||
*) echo "FAIL restart_existing_system_servers died on an unchanged rebuild: $out"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
out="$(run_system_restart sha256:aaa sha256:ccc 1)"
|
||||
expect "a new lobby build restarts the lobby" "delete pod -l felis.lolicon.best/server=lobby" "$out"
|
||||
out="$(run_system_restart sha256:aaa sha256:ccc 1 1)"
|
||||
expect "a failed pin warns and names the way out" "WARN: could not pin the lobby system server" "$out"
|
||||
expect "a failed pin falls back to restarting the pod" "KUBE -n minecraft delete pod -l felis.lolicon.best/server=lobby" "$out"
|
||||
case "$out" in
|
||||
*"server=login"*) echo "FAIL a new lobby build restarted the login gate too"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS a new lobby build leaves the login gate running" ;;
|
||||
*"server=login"*|*"--system login"*) echo "FAIL a new lobby build moved the login gate too"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS a new lobby build leaves the login gate alone" ;;
|
||||
esac
|
||||
rm -f "$sdir2/state"
|
||||
out="$(run_system_restart sha256:aaa sha256:ccc 0)"
|
||||
out="$(run_system_restart sha256:aaa sha256:ddd 0 1)"
|
||||
case "$out" in
|
||||
*delete*) echo "FAIL a missing pod was deleted"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS no pod, nothing to restart" ;;
|
||||
esac
|
||||
expect "the builds are recorded even before the pods exist" "lobby sha256:ccc" "$(cat "$sdir2/state")"
|
||||
expect "the builds are recorded even before the pods exist" "lobby sha256:ddd" "$(cat "$sdir2/state")"
|
||||
rm -rf "$sdir2"
|
||||
|
||||
pgblock="$(awk '/^configure_postgres\(\) \{/,/^}/' "$BS")"
|
||||
@@ -1903,6 +2059,21 @@ expect "a v6 node address gets a v6 rule" "tcp dport 5432 ip6 saddr 2001:db8::7
|
||||
rm -rf "$fwdir"
|
||||
|
||||
|
||||
# --- the proxy heap -----------------------------------------------------------------------
|
||||
hblock="$(awk '/^heap_megabytes\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$hblock" ] || { echo "FAIL: no heap_megabytes found in $BS"; exit 1; }
|
||||
heap() { bash -c "$hblock"'
|
||||
heap_megabytes "$1"' _ "$1"; }
|
||||
expect "a heap in gigabytes converts to megabytes" "2048" "$(heap 2G)"
|
||||
expect "a heap in megabytes is kept" "768" "$(heap 768m)"
|
||||
for bad in 1 1K 0G 01G G -1G 1.5G 9999999G; do
|
||||
expect "the heap spelling '$bad' is refused" "0" "$(heap "$bad")"
|
||||
done
|
||||
case "$(awk '/^install_velocity_service\(\) \{/,/^}/' "$BS")" in
|
||||
*'-Xms${xms} -Xmx${xmx} '*) echo "PASS the proxy unit takes its heap from FELIS_VELOCITY_XMX" ;;
|
||||
*) echo "FAIL the proxy unit's heap is not FELIS_VELOCITY_XMX"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
# --- reproducible image ids ---------------------------------------------------------------
|
||||
# restart_existing_system_servers compares image ids across runs; a default BuildKit
|
||||
# provenance attestation (it carries a timestamp) would make every rebuild look new.
|
||||
@@ -1914,6 +2085,103 @@ else
|
||||
echo "FAIL BUILDX_NO_DEFAULT_ATTESTATIONS=1 must be exported before the first docker build"; fails=$((fails + 1))
|
||||
fi
|
||||
|
||||
# --- a failed run puts the previous host binary back until the new one is in use ----------
|
||||
hbdir="$(mktemp -d)"
|
||||
run_host_bin() { # exit-status in-use [no-previous]
|
||||
rm -f "$hbdir"/felis*
|
||||
[ -n "${3:-}" ] || { printf 'old\n' > "$hbdir/felis"; chmod 0755 "$hbdir/felis"; }
|
||||
HOST_BIN="$hbdir/felis" bash -c '
|
||||
set -e
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
HOST_BIN_PREV=""; HOST_BIN_KEPT=0; HOST_BIN_IN_USE='"$2"'
|
||||
'"$(awk '/^keep_previous_host_binary\(\) \{/,/^}/' "$BS")"'
|
||||
'"$(awk '/^restore_previous_host_binary\(\) \{/,/^}/' "$BS")"'
|
||||
keep_previous_host_binary
|
||||
rm -f "$HOST_BIN"; printf "new\n" > "$HOST_BIN"; chmod 0755 "$HOST_BIN"
|
||||
keep_previous_host_binary # a second replacement keeps the first original
|
||||
restore_previous_host_binary '"$1"'
|
||||
printf "BIN: %s\n" "$(cat "$HOST_BIN")"
|
||||
[ -e "$HOST_BIN.prev" ] && echo "PREV LEFT" || true'
|
||||
}
|
||||
out="$(run_host_bin 1 0)"
|
||||
expect "a run that fails before the new binary is used restores the old one" "BIN: old" "$out"
|
||||
expect "the restore says so" "WARN: restored the previous felis binary" "$out"
|
||||
out="$(run_host_bin 1 1)"
|
||||
expect "a run that fails after migrations keeps the new binary" "BIN: new" "$out"
|
||||
out="$(run_host_bin 0 0)"
|
||||
expect "a successful run keeps the new binary" "BIN: new" "$out"
|
||||
case "$out" in *"PREV LEFT"*) echo "FAIL: the previous binary copy must be removed"; fails=$((fails + 1)) ;; esac
|
||||
out="$(run_host_bin 1 0 fresh)"
|
||||
expect "a first install has nothing to restore" "BIN: new" "$out"
|
||||
case "$out" in *WARN:*) echo "FAIL: a first install must not restore the binary it just installed"; fails=$((fails + 1)) ;; esac
|
||||
rm -rf "$hbdir"
|
||||
|
||||
|
||||
# --- a rerun reads the host right and keeps the proxy up --------------------------------
|
||||
# Both checks run under bootstrap.sh's own `set -Eeuo pipefail`. The lists are far past one
|
||||
# pipe buffer with the match on the first line, so a `cmd | grep -q` has grep exit while cmd
|
||||
# is still writing: cmd dies of SIGPIPE, pipefail fails the pipeline, and the install took
|
||||
# the "missing" branch on a host that had it (CI caught the PostgreSQL case reinstalling a
|
||||
# package on a rerun). apt then ran needrestart, which restarted felis-velocity.
|
||||
|
||||
for f in postgres_installed import_registry_image apt_get; do
|
||||
[ -n "$(awk '/^'"$f"'\(\) \{/,/^}/' "$BS")" ] || { echo "FAIL: no ${f} in $BS"; exit 1; }
|
||||
[ "$(awk '/^'"$f"'\(\) \{/,/^}/' "$BS" | wc -l)" -lt 20 ] \
|
||||
|| { echo "FAIL: the extracted ${f} is not just the function -- did its closing brace move?"; exit 1; }
|
||||
done
|
||||
|
||||
rrdir="$(mktemp -d)"
|
||||
printf '#!/bin/sh\nexit 0\n' > "$rrdir/psql"
|
||||
cat > "$rrdir/systemctl" <<'EOF'
|
||||
#!/bin/sh
|
||||
printf '%s\n' "$FIRST_UNIT"
|
||||
seq 1 200000 | sed 's/.*/unit-&.service static -/'
|
||||
EOF
|
||||
cat > "$rrdir/apt-get" <<'EOF'
|
||||
#!/bin/sh
|
||||
echo "NEEDRESTART_SUSPEND=${NEEDRESTART_SUSPEND:-unset} $*"
|
||||
EOF
|
||||
chmod +x "$rrdir/psql" "$rrdir/systemctl" "$rrdir/apt-get"
|
||||
|
||||
run_pg() { # first unit line
|
||||
PATH="$rrdir:$PATH" FIRST_UNIT="$1" bash -c '
|
||||
set -Eeuo pipefail
|
||||
'"$(awk '/^postgres_installed\(\) \{/,/^}/' "$BS")"'
|
||||
if postgres_installed; then echo INSTALLED; else echo MISSING; fi'
|
||||
}
|
||||
expect "an installed PostgreSQL is found in a long unit list" "INSTALLED" "$(run_pg 'postgresql.service enabled enabled')"
|
||||
expect "a host without the unit still gets PostgreSQL installed" "MISSING" "$(run_pg 'nginx.service enabled enabled')"
|
||||
|
||||
run_reg() {
|
||||
bash -c '
|
||||
set -Eeuo pipefail
|
||||
ok() { echo "OK: $*"; }; log() { echo "LOG: $*"; }; warn() { echo "WARN: $*"; }
|
||||
REGISTRY_IMAGE=registry:2
|
||||
registry_image_containerd_ref() { echo docker.io/library/registry:2; }
|
||||
k3s_cmd() {
|
||||
if [ "$1" = ctr ]; then
|
||||
echo docker.io/library/registry:2
|
||||
seq 1 200000 | sed "s/.*/example.test\/img-&:1/"
|
||||
else
|
||||
echo "PULLED $*"
|
||||
fi
|
||||
}
|
||||
'"$(awk '/^import_registry_image\(\) \{/,/^}/' "$BS")"'
|
||||
import_registry_image'
|
||||
}
|
||||
out="$(run_reg)"
|
||||
expect "an imported registry image is found in a long image list" "OK: registry image registry:2 already in k3s containerd" "$out"
|
||||
case "$out" in *PULLED*) echo "FAIL: the registry image was pulled again"; fails=$((fails + 1)) ;; esac
|
||||
|
||||
out="$(PATH="$rrdir:$PATH" bash -c '
|
||||
set -Eeuo pipefail
|
||||
wait_for_pkg_locks() { :; }
|
||||
PKG_LOCK_TIMEOUT=5
|
||||
'"$(awk '/^apt_get\(\) \{/,/^}/' "$BS")"'
|
||||
apt_get install -y postgresql')"
|
||||
expect "the install's apt runs keep needrestart from restarting services" \
|
||||
"NEEDRESTART_SUSPEND=1 -o DPkg::Lock::Timeout=5 install -y postgresql" "$out"
|
||||
rm -rf "$rrdir"
|
||||
|
||||
# ---------------------------------------------------------------------------------------
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
|
||||
@@ -101,6 +101,8 @@ spec:
|
||||
EmptySecondsBeforeStop is how long the server may sit empty before the
|
||||
operator scales it down.
|
||||
format: int32
|
||||
maximum: 604800
|
||||
minimum: 0
|
||||
type: integer
|
||||
type: object
|
||||
image:
|
||||
@@ -127,6 +129,8 @@ spec:
|
||||
description: TerminationGracePeriodSeconds is the pod grace period
|
||||
(default 300).
|
||||
format: int64
|
||||
maximum: 3600
|
||||
minimum: 0
|
||||
type: integer
|
||||
type: object
|
||||
motd:
|
||||
@@ -159,6 +163,8 @@ spec:
|
||||
port:
|
||||
description: Port is the RCON TCP port (default 25575).
|
||||
format: int32
|
||||
maximum: 65535
|
||||
minimum: 0
|
||||
type: integer
|
||||
secretRef:
|
||||
description: SecretRef points at the Secret holding the RCON password.
|
||||
@@ -172,6 +178,10 @@ spec:
|
||||
- name
|
||||
type: object
|
||||
type: object
|
||||
x-kubernetes-validations:
|
||||
- message: the allow-rcon NetworkPolicy admits only port 25575; leave
|
||||
port unset
|
||||
rule: '!has(self.port) || self.port == 0 || self.port == 25575'
|
||||
reaperExempt:
|
||||
description: ReaperExempt opts this server out of the world reaper
|
||||
entirely (spec §18).
|
||||
@@ -250,16 +260,22 @@ spec:
|
||||
(notably LOOHP/Limbo) where the felis-limbo plugin reports true
|
||||
readiness only after the first server tick.
|
||||
format: int32
|
||||
maximum: 65535
|
||||
minimum: 0
|
||||
type: integer
|
||||
readinessTimeoutSeconds:
|
||||
description: ReadinessTimeoutSeconds is the budget for the first
|
||||
successful RCON probe.
|
||||
format: int32
|
||||
maximum: 86400
|
||||
minimum: 0
|
||||
type: integer
|
||||
timeoutSeconds:
|
||||
description: TimeoutSeconds is the overall budget before the server
|
||||
is marked Failed.
|
||||
format: int32
|
||||
maximum: 86400
|
||||
minimum: 0
|
||||
type: integer
|
||||
type: object
|
||||
storage:
|
||||
@@ -285,6 +301,13 @@ spec:
|
||||
status:
|
||||
description: MinecraftServerStatus is the observed state (spec §4 status.*).
|
||||
properties:
|
||||
autoRestarts:
|
||||
description: |-
|
||||
AutoRestarts counts how often the operator recreated the pod of a start
|
||||
that timed out (at most 3, with a doubling backoff); reaching Ready or
|
||||
stopping resets it.
|
||||
format: int32
|
||||
type: integer
|
||||
conditions:
|
||||
description: Conditions are the standard metav1 conditions (Ready,
|
||||
RconReached, ...).
|
||||
@@ -361,6 +384,11 @@ spec:
|
||||
description: Mode is "direct" or "fallback".
|
||||
type: string
|
||||
type: object
|
||||
lastAutoRestartAt:
|
||||
description: LastAutoRestartAt is when the operator last recreated
|
||||
the pod.
|
||||
format: date-time
|
||||
type: string
|
||||
liveMotd:
|
||||
description: LiveMotd is the MOTD currently advertised for the active
|
||||
phase.
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
#!/bin/bash
|
||||
# Checks a host that deploy/bootstrap.sh just installed (or re-ran on). The e2e workflow
|
||||
# runs it on a fresh GitHub runner after each of its two installer runs:
|
||||
#
|
||||
# sudo bash deploy/e2e_check.sh install # after the first run
|
||||
# sudo bash deploy/e2e_check.sh rerun # after the same commit ran again
|
||||
# sudo bash deploy/e2e_check.sh release # after the newest release installed
|
||||
# sudo bash deploy/e2e_check.sh upgrade # after this commit ran over a release
|
||||
#
|
||||
# It asks what an operator's first minutes ask: the binary runs, the control plane is
|
||||
# rolled out and ready, the panel answers on its NodePort, the proxy answers a Minecraft
|
||||
# status ping, and the host timers are there. A rerun must also leave the proxy running
|
||||
# (it restarts only when what it runs changed) and keep every earlier answer.
|
||||
set -euo pipefail
|
||||
|
||||
phase="${1:?usage: e2e_check.sh install|rerun|release|upgrade}"
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
KUBECTL=(/usr/local/bin/k3s kubectl)
|
||||
PID_FILE=/var/tmp/felis-e2e-velocity.pid
|
||||
fails=0
|
||||
|
||||
pass() { printf 'PASS %s\n' "$*"; }
|
||||
fail() { printf 'FAIL %s\n' "$*"; fails=$((fails + 1)); }
|
||||
check() { # label command...
|
||||
local label="$1"
|
||||
shift
|
||||
if "$@"; then pass "$label"; else fail "$label"; fi
|
||||
}
|
||||
|
||||
check "felis version runs" sh -c '/usr/local/bin/felis version | grep -q "^felis "'
|
||||
|
||||
for d in felis-api felis-operator registry; do
|
||||
check "deployment ${d} is rolled out" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
|
||||
done
|
||||
|
||||
node_ip="$(ip -4 route get 1.1.1.1 | awk '{for (i = 1; i <= NF; i++) if ($i == "src") { print $(i + 1); exit }}')"
|
||||
check "the panel serves its page on the NodePort" \
|
||||
sh -c "curl -skf --retry 10 --retry-delay 3 --retry-all-errors https://${node_ip}:30443/ | grep -qi '<html'"
|
||||
# Readiness lives on the internal face only; a Service ClusterIP routes from the node.
|
||||
internal="$("${KUBECTL[@]}" -n felis get svc felis-api-internal -o jsonpath='{.spec.clusterIP}:{.spec.ports[0].port}')"
|
||||
check "felis-api is ready (database and cluster reachable)" \
|
||||
curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"
|
||||
|
||||
for unit in k3s postgresql felis-velocity; do
|
||||
check "${unit} is active" systemctl is-active --quiet "$unit"
|
||||
done
|
||||
# A release may predate a timer; what this commit installs has them all.
|
||||
if [ "$phase" != release ]; then
|
||||
for timer in felis-db-backup.timer felis-watchdog.timer felis-update-check.timer; do
|
||||
check "${timer} is scheduled" systemctl is-enabled --quiet "$timer"
|
||||
done
|
||||
fi
|
||||
|
||||
# A status ping is the proxy's own answer (ping passthrough is off), so it proves the JRE,
|
||||
# Velocity and its config without a login gate or a Mojang account.
|
||||
ping_proxy() {
|
||||
python3 - <<'EOF'
|
||||
import json, socket, struct, sys
|
||||
|
||||
def varint(n):
|
||||
out = b""
|
||||
n &= 0xFFFFFFFF
|
||||
while True:
|
||||
b, n = n & 0x7F, n >> 7
|
||||
if n:
|
||||
out += bytes([b | 0x80])
|
||||
else:
|
||||
return out + bytes([b])
|
||||
|
||||
def read_varint(s):
|
||||
n = 0
|
||||
for i in range(5):
|
||||
b = s.recv(1)
|
||||
if not b:
|
||||
raise EOFError("connection closed")
|
||||
n |= (b[0] & 0x7F) << (7 * i)
|
||||
if not b[0] & 0x80:
|
||||
return n
|
||||
raise ValueError("varint too long")
|
||||
|
||||
host, port = "127.0.0.1", 25565
|
||||
s = socket.create_connection((host, port), timeout=10)
|
||||
hs = varint(0) + varint(767) + varint(len(host)) + host.encode() + struct.pack(">H", port) + varint(1)
|
||||
s.sendall(varint(len(hs)) + hs + varint(1) + varint(0))
|
||||
read_varint(s)
|
||||
read_varint(s)
|
||||
size = read_varint(s)
|
||||
data = b""
|
||||
while len(data) < size:
|
||||
chunk = s.recv(size - len(data))
|
||||
if not chunk:
|
||||
raise EOFError("short status response")
|
||||
data += chunk
|
||||
print(json.loads(data)["version"]["name"])
|
||||
EOF
|
||||
}
|
||||
# The installer returns once the unit is started; the JVM binds the port a few
|
||||
# seconds later.
|
||||
for _ in $(seq 30); do
|
||||
if version="$(ping_proxy 2>&1)"; then
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if version="$(ping_proxy 2>&1)"; then
|
||||
pass "the proxy answers a status ping (${version})"
|
||||
else
|
||||
fail "the proxy answers a status ping: ${version}"
|
||||
fi
|
||||
|
||||
pid="$(systemctl show -p MainPID --value felis-velocity)"
|
||||
case "$phase" in
|
||||
install | release) printf '%s\n' "$pid" > "$PID_FILE" ;;
|
||||
rerun)
|
||||
if [ "$pid" = "$(cat "$PID_FILE" 2>/dev/null)" ]; then
|
||||
pass "the rerun left the proxy running (pid ${pid})"
|
||||
else
|
||||
fail "the rerun restarted the proxy (pid $(cat "$PID_FILE" 2>/dev/null || echo '?') -> ${pid}) though nothing it runs changed"
|
||||
fi
|
||||
;;
|
||||
upgrade) ;; # a new release may well change what the proxy runs
|
||||
*) fail "unknown phase ${phase}" ;;
|
||||
esac
|
||||
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS (${phase})"
|
||||
else
|
||||
echo "${fails} FAILED (${phase})"
|
||||
fi
|
||||
exit "$fails"
|
||||
+14
-10
@@ -29,22 +29,26 @@
|
||||
# overridable via FELIS_HEALTH_PORT) and returns 200 only after the first tick.
|
||||
|
||||
# ---- build the felis-limbo plugin jar ----
|
||||
# gradle:*-jdk21 — an official Gradle image on JDK 21. JDK 21 is required because
|
||||
# current LOOHP/Limbo releases ship Java 21 API classes (class-file major 65); a
|
||||
# JDK 17 fails to read them with "wrong version 65.0, should be 61.0". The image
|
||||
# also provides the `gradle` binary (this tree vendors no Gradle wrapper).
|
||||
# build.gradle still targets release 17 bytecode so the plugin loads on Java 17+.
|
||||
FROM gradle:8.14-jdk21@sha256:5c4c0c4284de4a19951e82ac78f86dbcda2e136644bbfe159beba7ea3420cc80 AS plugin
|
||||
# The same pinned Gradle image as the lobby build (see deploy/lobby/Dockerfile). Its JDK
|
||||
# must be >= 21 because current LOOHP/Limbo releases ship Java 21 API classes
|
||||
# (class-file major 65); a JDK 17 fails to read them with "wrong version 65.0, should be
|
||||
# 61.0". build.gradle still targets release 17 bytecode so the plugin loads on Java 17+.
|
||||
FROM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
|
||||
WORKDIR /src
|
||||
# Copy what the limbo module needs: its own tree plus the shared link core it
|
||||
# srcDir-includes (../shared/src/main/java → /src/plugins/shared/src/main/java), so
|
||||
# the account-link client + config loader compile straight into the jar.
|
||||
COPY plugins/limbo/ ./plugins/limbo/
|
||||
COPY plugins/shared/ ./plugins/shared/
|
||||
ARG LIMBO_VERSION=+
|
||||
RUN cd plugins/limbo \
|
||||
&& (test -x ./gradlew && ./gradlew --no-daemon -PlimboVersion="$LIMBO_VERSION" build \
|
||||
|| gradle --no-daemon -PlimboVersion="$LIMBO_VERSION" build) \
|
||||
# The Limbo API release to compile against: deploy/game-stack.lock's LIMBO_VERSION, which
|
||||
# bootstrap passes. Required — the API is checked against the checksum
|
||||
# plugins/limbo/gradle/verification-metadata.xml holds for that release.
|
||||
ARG LIMBO_VERSION
|
||||
RUN if [ -z "${LIMBO_VERSION:-}" ]; then \
|
||||
echo "LIMBO_VERSION is required (deploy/game-stack.lock)" >&2; exit 1; \
|
||||
fi \
|
||||
&& cd plugins/limbo \
|
||||
&& gradle --no-daemon -PlimboVersion="$LIMBO_VERSION" build \
|
||||
&& cp build/libs/*.jar /felis-limbo.jar
|
||||
|
||||
# ---- assemble the runtime ----
|
||||
|
||||
@@ -132,10 +132,13 @@ set them by hand:
|
||||
`spec.env` by `felis setup` (`cmd/felis` derives the internal API URL from the
|
||||
control namespace — the platform default `felis`; a renamed control namespace must
|
||||
be reflected by hand — and the root domain from `felis.toml`).
|
||||
- `FELIS_SERVICE_TOKEN` is a **secret**, so it is never written into the CRD. `felis
|
||||
setup` replicates the `felis-service-token` Secret from the control namespace into
|
||||
the minecraft namespace, and the operator injects it into the `login` pod (only)
|
||||
via a `secretKeyRef`, keyed off the reserved `login` name. Until the token is
|
||||
- `FELIS_SERVICE_TOKEN` is a **secret**, so it is never written into the CRD. The
|
||||
login gate has its own internal-API token, `felis-limbo-token`, which may only mint
|
||||
link codes, poll link status and check the blacklist. The installer applies it into
|
||||
the minecraft namespace (and `felis setup` refreshes that replica from the control
|
||||
namespace), and the operator injects it into the `login` pod (only) as
|
||||
`FELIS_SERVICE_TOKEN` via a `secretKeyRef`, keyed off the reserved `login` name.
|
||||
`sudo felis rotate-token limbo` replaces it and restarts the pod. Until the token is
|
||||
present the plugin fail-safes to readiness-only, so the gate is never broken — it
|
||||
simply does not authenticate yet.
|
||||
- **Service:** the login pod dials `FELIS_API_BASE_URL`, which resolves to the
|
||||
|
||||
@@ -25,16 +25,19 @@
|
||||
# Secret) and refuses to start without it: a lobby that cannot verify the proxy's signed
|
||||
# handshake would trust an offline, forgeable UUID.
|
||||
|
||||
# ---- build the felis-paper plugin jar (Paper API is Java 21) ----
|
||||
# gradle:8.14-jdk21 — an official Gradle image on JDK 21 (this tree vendors no Gradle
|
||||
# wrapper, and a bare JDK image ships no `gradle`). JDK 21 matches the Paper API.
|
||||
FROM gradle:8.14-jdk21@sha256:5c4c0c4284de4a19951e82ac78f86dbcda2e136644bbfe159beba7ea3420cc80 AS plugin
|
||||
# ---- build the felis-paper plugin jar (Paper API is Java 25) ----
|
||||
# The official Gradle image on JDK 25, the Gradle version plugins/*/gradle/wrapper pins.
|
||||
# The limbo Dockerfile and bootstrap's velocity build use this same image, digest and all
|
||||
# (bootstrap_asset_test.go holds the three together). The image's own `gradle` runs the
|
||||
# build rather than the module's wrapper, which would download the same distribution
|
||||
# again on every image build. The build checks every dependency against
|
||||
# plugins/paper/gradle/verification-metadata.xml and fails on a mismatch.
|
||||
FROM gradle:9.8.0-jdk25@sha256:2b2fc1b1dfc3604a2acc916839f36eb5ee48fd7f232427fc5faca224c73bcb01 AS plugin
|
||||
WORKDIR /src
|
||||
COPY plugins/paper/ ./plugins/paper/
|
||||
COPY plugins/shared/ ./plugins/shared/
|
||||
RUN cd plugins/paper \
|
||||
&& (test -x ./gradlew && ./gradlew --no-daemon build \
|
||||
|| gradle --no-daemon build) \
|
||||
&& gradle --no-daemon build \
|
||||
&& cp build/libs/*.jar /felis-paper.jar
|
||||
|
||||
# ---- assemble the runtime ----
|
||||
|
||||
@@ -0,0 +1,428 @@
|
||||
#!/usr/bin/env bash
|
||||
# Removes what deploy/bootstrap.sh installed on this host.
|
||||
#
|
||||
# sudo bash deploy/uninstall.sh # remove Felis, keep the data
|
||||
# sudo bash deploy/uninstall.sh --purge # remove the data too
|
||||
# curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes
|
||||
#
|
||||
# Options:
|
||||
# --purge also drop the felis database and role, and delete /etc/felis and
|
||||
# /var/lib/felis (the database bundles, and anything an earlier keep-data
|
||||
# run set aside). Asks for the word "purge" unless --yes is given.
|
||||
# --keep-k3s leave k3s installed and remove only Felis's namespaces and CRD.
|
||||
# --remove-k3s run k3s's own uninstaller even when other workloads live in the cluster.
|
||||
# --no-backup skip the final database bundle keep-data mode takes first.
|
||||
# --yes do not ask.
|
||||
#
|
||||
# Keep-data mode (the default) first takes a database bundle (`felis db backup -label
|
||||
# manual`) and stops if that fails. It leaves PostgreSQL's felis database, /etc/felis (the
|
||||
# secrets, felis.toml, offsite.env) and /var/lib/felis in place. The world, archive,
|
||||
# registry and upload volumes live under k3s's storage directory, which k3s's uninstaller
|
||||
# deletes, so they are moved to /var/lib/felis/retained/k3s-storage-<UTC stamp> first; with
|
||||
# --keep-k3s their PersistentVolumes are switched to Retain before the namespaces go.
|
||||
# docs/operations.md walks through reinstalling on top of what is left.
|
||||
#
|
||||
# Either mode leaves packages alone (Docker, PostgreSQL, git and the rest), and the swap
|
||||
# file a low-memory host got: other software may use them. docs/operations.md lists the
|
||||
# package commands for a bare host.
|
||||
set -Eeuo pipefail
|
||||
|
||||
STATE_DIR="${STATE_DIR:-/etc/felis}"
|
||||
DATA_DIR="${DATA_DIR:-/var/lib/felis}"
|
||||
RETAIN_DIR="${DATA_DIR}/retained"
|
||||
HOST_BIN="${HOST_BIN:-/usr/local/bin/felis}"
|
||||
OPT_DIR="${OPT_DIR:-/opt/felis}"
|
||||
UNIT_DIR="${UNIT_DIR:-/etc/systemd/system}"
|
||||
K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}"
|
||||
K3S_STORAGE="${K3S_STORAGE:-/var/lib/rancher/k3s/storage}"
|
||||
K3S_REGISTRIES="${K3S_REGISTRIES:-/etc/rancher/k3s/registries.yaml}"
|
||||
export KUBECONFIG="${KUBECONFIG:-/etc/rancher/k3s/k3s.yaml}"
|
||||
CLOUDFLARED_BIN="${CLOUDFLARED_BIN:-/usr/local/bin/cloudflared}"
|
||||
VELOCITY_USER="felis-velocity"
|
||||
DB_NAME="felis"
|
||||
DB_USER="felis"
|
||||
POD_CIDR="10.42.0.0/16"
|
||||
SERVICE_CIDR="10.43.0.0/16"
|
||||
FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}"
|
||||
CONFIRM_TTY="${CONFIRM_TTY:-/dev/tty}"
|
||||
FELIS_NAMESPACES=(felis minecraft felis-build)
|
||||
FELIS_CRD="minecraftservers.felis.lolicon.best"
|
||||
# Every unit the installer and `felis setup` write. Timers first, so none fires into a
|
||||
# service that is already gone.
|
||||
FELIS_UNITS=(
|
||||
felis-db-backup.timer felis-watchdog.timer felis-offsite.timer felis-build-tools.timer felis-update-check.timer
|
||||
felis-db-backup.service felis-watchdog.service felis-offsite.service felis-build-tools.service felis-update-check.service
|
||||
felis-velocity.service felis-nano.service cloudflared-felis.service
|
||||
felis-postgres-firewall.service
|
||||
)
|
||||
|
||||
PURGE=0
|
||||
K3S_MODE=auto
|
||||
BACKUP=1
|
||||
ASSUME_YES=0
|
||||
TUNNEL_CONFIG=""
|
||||
|
||||
log() { printf '\033[1;36m[felis]\033[0m %s\n' "$*"; }
|
||||
ok() { printf '\033[1;32m[ ok ]\033[0m %s\n' "$*"; }
|
||||
warn() { printf '\033[1;33m[warn]\033[0m %s\n' "$*" >&2; }
|
||||
die() { printf '\033[1;31m[fail]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
parse_args() {
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--purge) PURGE=1 ;;
|
||||
--keep-k3s) K3S_MODE=keep ;;
|
||||
--remove-k3s) K3S_MODE=remove ;;
|
||||
--no-backup) BACKUP=0 ;;
|
||||
--yes|-y) ASSUME_YES=1 ;;
|
||||
-h|--help) printf 'usage: uninstall.sh [--purge] [--keep-k3s|--remove-k3s] [--no-backup] [--yes]\n'; exit 0 ;;
|
||||
*) die "unknown option: $1 (see --help)" ;;
|
||||
esac
|
||||
shift
|
||||
done
|
||||
}
|
||||
|
||||
kube() { "${K3S_BIN_DIR}/k3s" kubectl "$@"; }
|
||||
k3s_present() { [ -x "${K3S_BIN_DIR}/k3s" ]; }
|
||||
|
||||
# foreign_namespaces prints the namespaces that are neither k3s's own nor Felis's, one per
|
||||
# line. A cluster with none of them exists for Felis alone, and removing k3s takes nothing
|
||||
# else with it.
|
||||
foreign_namespaces() {
|
||||
kube get namespaces -o 'jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}' \
|
||||
| awk '$0 != "" && $0 != "default" && $0 !~ /^kube-/ && $0 != "felis" && $0 != "minecraft" && $0 != "felis-build"'
|
||||
}
|
||||
|
||||
# decide_k3s turns K3S_MODE=auto into keep or remove.
|
||||
decide_k3s() {
|
||||
k3s_present || { K3S_MODE=absent; return 0; }
|
||||
[ "$K3S_MODE" = auto ] || return 0
|
||||
local others
|
||||
if ! others="$(foreign_namespaces)"; then
|
||||
die "k3s does not answer, so this cannot tell whether it runs anything besides Felis; start it (systemctl start k3s) or pass --keep-k3s or --remove-k3s"
|
||||
fi
|
||||
if [ -z "$others" ]; then
|
||||
K3S_MODE=remove
|
||||
else
|
||||
K3S_MODE=keep
|
||||
log "k3s also runs namespaces Felis did not create ($(printf '%s' "$others" | tr '\n' ' ')); leaving k3s installed"
|
||||
fi
|
||||
}
|
||||
|
||||
confirm() {
|
||||
[ "$ASSUME_YES" = 1 ] && return 0
|
||||
local want="yes" answer=""
|
||||
[ "$PURGE" = 1 ] && want="purge"
|
||||
# A piped script has no stdin to read from; the terminal is asked directly.
|
||||
{ exec 3<"$CONFIRM_TTY" 4>>"$CONFIRM_TTY"; } 2>/dev/null || die "no terminal to confirm on; re-run with --yes"
|
||||
printf 'Type "%s" to continue: ' "$want" >&4
|
||||
read -r answer <&3 || true
|
||||
exec 3<&- 4>&-
|
||||
[ "$answer" = "$want" ] || die "not confirmed; nothing was changed"
|
||||
}
|
||||
|
||||
print_plan() {
|
||||
log "this will remove from $(uname -n):"
|
||||
log " the felis-* systemd units, cloudflared-felis.service, the ${VELOCITY_USER} user,"
|
||||
log " ${OPT_DIR}, ${HOST_BIN}, the felis_postgres and felis_edge nftables tables and the firewalld openings"
|
||||
case "$K3S_MODE" in
|
||||
remove) log " k3s, with everything in it (${K3S_BIN_DIR}/k3s-uninstall.sh)" ;;
|
||||
keep) log " Felis's namespaces (${FELIS_NAMESPACES[*]}) and the ${FELIS_CRD} CRD; k3s stays" ;;
|
||||
absent) ;;
|
||||
esac
|
||||
if [ "$PURGE" = 1 ]; then
|
||||
log " PURGE: the ${DB_NAME} database and role, ${STATE_DIR} (secrets), ${DATA_DIR} (database bundles"
|
||||
log " and anything set aside before), every world and archive, the Felis images and Docker's build cache"
|
||||
else
|
||||
[ "$BACKUP" = 1 ] && log " after a final database bundle into ${DATA_DIR}/db-backups"
|
||||
log " kept: the ${DB_NAME} database, ${STATE_DIR}, ${DATA_DIR}; the volumes move to ${RETAIN_DIR}/"
|
||||
fi
|
||||
}
|
||||
|
||||
final_backup() {
|
||||
[ "$PURGE" = 0 ] && [ "$BACKUP" = 1 ] || return 0
|
||||
[ -x "$HOST_BIN" ] && [ -r "${STATE_DIR}/felis.host.toml" ] || {
|
||||
warn "no ${HOST_BIN} or ${STATE_DIR}/felis.host.toml; skipping the final database bundle"
|
||||
return 0
|
||||
}
|
||||
log "taking a final database bundle"
|
||||
"$HOST_BIN" db backup -config "${STATE_DIR}/felis.host.toml" -label manual \
|
||||
|| die "the final database bundle failed, so nothing was removed. Fix the database (sudo felis db check), or pass --no-backup to go on without one"
|
||||
ok "database bundle written to ${DATA_DIR}/db-backups"
|
||||
}
|
||||
|
||||
# game_port reads the proxy's port from velocity.toml before /opt/felis goes.
|
||||
game_port() {
|
||||
local toml="${OPT_DIR}/velocity/velocity.toml" port=""
|
||||
[ -r "$toml" ] && port="$(sed -n 's/^bind *= *"[^"]*:\([0-9][0-9]*\)".*/\1/p' "$toml" | head -n 1)"
|
||||
printf '%s\n' "${FELIS_GAME_PORT:-${port:-25565}}"
|
||||
}
|
||||
|
||||
# tunnel_config reads the cloudflared config felis setup pointed its unit at (by default
|
||||
# /etc/felis/cloudflared.yml) before the unit goes.
|
||||
tunnel_config() {
|
||||
local unit="${UNIT_DIR}/cloudflared-felis.service" path=""
|
||||
[ -r "$unit" ] && path="$(sed -n 's/^ExecStart=.* --config \([^ ]*\) tunnel run$/\1/p' "$unit" | head -n 1)"
|
||||
printf '%s\n' "${path:-${STATE_DIR}/cloudflared.yml}"
|
||||
}
|
||||
|
||||
# nano_port reads felis-nano's port from its unit before the unit goes.
|
||||
nano_port() {
|
||||
local unit="${UNIT_DIR}/felis-nano.service"
|
||||
[ -r "$unit" ] || return 0
|
||||
sed -n 's/^ExecStart=.* -listen [^ ]*:\([0-9][0-9]*\).*$/\1/p' "$unit" | head -n 1
|
||||
}
|
||||
|
||||
remove_units() {
|
||||
local unit removed=0
|
||||
for unit in "${FELIS_UNITS[@]}"; do
|
||||
[ -f "${UNIT_DIR}/${unit}" ] || continue
|
||||
systemctl disable --now "$unit" >/dev/null 2>&1 || systemctl stop "$unit" >/dev/null 2>&1 || true
|
||||
rm -f "${UNIT_DIR}/${unit}"
|
||||
removed=$((removed + 1))
|
||||
done
|
||||
systemctl daemon-reload
|
||||
systemctl reset-failed >/dev/null 2>&1 || true
|
||||
ok "${removed} systemd unit(s) removed"
|
||||
}
|
||||
|
||||
remove_nft_tables() {
|
||||
command -v nft >/dev/null 2>&1 || return 0
|
||||
local t
|
||||
for t in felis_postgres felis_edge; do
|
||||
if nft list table inet "$t" >/dev/null 2>&1; then
|
||||
nft delete table inet "$t"
|
||||
ok "nftables table inet ${t} removed"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# remove_firewalld_rules takes back what configure_k3s_firewall, configure_velocity_firewall
|
||||
# and the nano setup opened. The k3s ones stay when k3s does.
|
||||
remove_firewalld_rules() { # game-port nano-port
|
||||
command -v firewall-cmd >/dev/null 2>&1 || return 0
|
||||
systemctl is-active --quiet firewalld || return 0
|
||||
local game="$1" nano="$2" port rule changed=0
|
||||
local ports=("${game}/tcp" "${FELIS_PANEL_NODEPORT}/tcp")
|
||||
[ -n "$nano" ] && ports+=("${nano}/tcp")
|
||||
[ "$K3S_MODE" = remove ] && ports+=("6443/tcp")
|
||||
for port in "${ports[@]}"; do
|
||||
if firewall-cmd --permanent --query-port="$port" >/dev/null 2>&1; then
|
||||
firewall-cmd --permanent --remove-port="$port" >/dev/null
|
||||
changed=1
|
||||
fi
|
||||
done
|
||||
if [ -n "$nano" ]; then
|
||||
while IFS= read -r rule; do
|
||||
case "$rule" in
|
||||
*"port=\"${nano}\""*) firewall-cmd --permanent --remove-rich-rule="$rule" >/dev/null; changed=1 ;;
|
||||
esac
|
||||
done < <(firewall-cmd --permanent --list-rich-rules 2>/dev/null)
|
||||
fi
|
||||
if [ "$K3S_MODE" = remove ]; then
|
||||
local cidr
|
||||
for cidr in "$POD_CIDR" "$SERVICE_CIDR"; do
|
||||
if firewall-cmd --permanent --zone=trusted --query-source="$cidr" >/dev/null 2>&1; then
|
||||
firewall-cmd --permanent --zone=trusted --remove-source="$cidr" >/dev/null
|
||||
changed=1
|
||||
fi
|
||||
done
|
||||
fi
|
||||
if [ "$changed" = 1 ]; then
|
||||
firewall-cmd --reload >/dev/null
|
||||
ok "firewalld openings removed"
|
||||
fi
|
||||
}
|
||||
|
||||
# retain_volumes_in_cluster keeps every volume Felis's claims are bound to when the
|
||||
# namespaces go: local-path deletes a Delete-policy volume's directory with its claim.
|
||||
retain_volumes_in_cluster() {
|
||||
local pv
|
||||
while IFS= read -r pv; do
|
||||
[ -n "$pv" ] || continue
|
||||
kube patch pv "$pv" -p '{"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' >/dev/null
|
||||
done < <(kube get pv -o 'jsonpath={range .items[*]}{.metadata.name} {.spec.claimRef.namespace}{"\n"}{end}' \
|
||||
| awk '$2 == "felis" || $2 == "minecraft" || $2 == "felis-build" { print $1 }')
|
||||
ok "Felis's volumes set to Retain; their directories stay under ${K3S_STORAGE}"
|
||||
}
|
||||
|
||||
remove_from_cluster() {
|
||||
[ "$PURGE" = 1 ] || retain_volumes_in_cluster
|
||||
log "deleting Felis's namespaces and CRD"
|
||||
# The operator is part of what goes, so nothing would clear a MinecraftServer finalizer
|
||||
# and the minecraft namespace would stay Terminating. Drop them first.
|
||||
local s
|
||||
while IFS= read -r s; do
|
||||
[ -n "$s" ] || continue
|
||||
kube -n minecraft patch minecraftserver "$s" --type=merge -p '{"metadata":{"finalizers":null}}' >/dev/null 2>&1 || true
|
||||
done < <(kube -n minecraft get minecraftservers -o 'jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}' 2>/dev/null || true)
|
||||
kube delete namespace "${FELIS_NAMESPACES[@]}" --ignore-not-found --wait=true --timeout=300s >/dev/null \
|
||||
|| warn "a namespace is still terminating; check: k3s kubectl get namespaces"
|
||||
kube delete crd "$FELIS_CRD" --ignore-not-found >/dev/null || true
|
||||
if [ "$PURGE" = 1 ]; then
|
||||
local pv
|
||||
while IFS= read -r pv; do
|
||||
[ -n "$pv" ] && kube delete pv "$pv" --ignore-not-found >/dev/null
|
||||
done < <(kube get pv -o 'jsonpath={range .items[*]}{.metadata.name} {.spec.claimRef.namespace}{"\n"}{end}' \
|
||||
| awk '$2 == "felis" || $2 == "minecraft" || $2 == "felis-build" { print $1 }')
|
||||
fi
|
||||
if [ -f "$K3S_REGISTRIES" ] && grep -q 'registry\.felis\.svc' "$K3S_REGISTRIES"; then
|
||||
rm -f "$K3S_REGISTRIES"
|
||||
systemctl restart k3s
|
||||
fi
|
||||
ok "Felis removed from the cluster; k3s stays"
|
||||
}
|
||||
|
||||
remove_k3s() {
|
||||
local stamp
|
||||
if [ "$PURGE" = 0 ] && [ -d "$K3S_STORAGE" ]; then
|
||||
# k3s-killall.sh stops every pod and unmounts their volumes, so nothing is writing a
|
||||
# world while it moves.
|
||||
"${K3S_BIN_DIR}/k3s-killall.sh" >/dev/null 2>&1 || systemctl stop k3s
|
||||
stamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||
install -d -m 0700 "$RETAIN_DIR"
|
||||
mv "$K3S_STORAGE" "${RETAIN_DIR}/k3s-storage-${stamp}"
|
||||
ok "volumes moved to ${RETAIN_DIR}/k3s-storage-${stamp}"
|
||||
fi
|
||||
if [ -x "${K3S_BIN_DIR}/k3s-uninstall.sh" ]; then
|
||||
log "running k3s-uninstall.sh"
|
||||
"${K3S_BIN_DIR}/k3s-uninstall.sh" >/dev/null 2>&1 || warn "k3s-uninstall.sh reported an error; check /var/lib/rancher and /etc/rancher"
|
||||
ok "k3s removed"
|
||||
else
|
||||
warn "k3s is at ${K3S_BIN_DIR}/k3s but ${K3S_BIN_DIR}/k3s-uninstall.sh is missing; remove k3s by hand"
|
||||
fi
|
||||
}
|
||||
|
||||
# remove_cloudflared_binary deletes the binary the installer put in /usr/local/bin, unless
|
||||
# a unit other than Felis's still runs it.
|
||||
remove_cloudflared_binary() {
|
||||
[ -x "$CLOUDFLARED_BIN" ] || return 0
|
||||
local others
|
||||
others="$(grep -ls "$CLOUDFLARED_BIN" "${UNIT_DIR}"/*.service /lib/systemd/system/*.service /usr/lib/systemd/system/*.service 2>/dev/null || true)"
|
||||
if [ -n "$others" ]; then
|
||||
log "leaving ${CLOUDFLARED_BIN}: $(printf '%s' "$others" | tr '\n' ' ')uses it"
|
||||
return 0
|
||||
fi
|
||||
rm -f "$CLOUDFLARED_BIN"
|
||||
ok "${CLOUDFLARED_BIN} removed"
|
||||
}
|
||||
|
||||
remove_host_files() {
|
||||
if id "$VELOCITY_USER" >/dev/null 2>&1; then
|
||||
userdel "$VELOCITY_USER" >/dev/null 2>&1 || warn "could not remove the ${VELOCITY_USER} user"
|
||||
fi
|
||||
rm -rf "$OPT_DIR"
|
||||
rm -f "$HOST_BIN" "${HOST_BIN}.new" "${HOST_BIN}.prev"
|
||||
remove_cloudflared_binary
|
||||
if [ "$PURGE" = 0 ]; then
|
||||
# What describes the removed install goes; what a reinstall reuses stays. Without
|
||||
# bootstrap.done the next run takes the first-install path.
|
||||
rm -f "${STATE_DIR}/bootstrap.done" "${STATE_DIR}/system-server-images" \
|
||||
"${STATE_DIR}/velocity.fingerprint" "${STATE_DIR}/previous-felis-image"
|
||||
fi
|
||||
ok "${OPT_DIR} and ${HOST_BIN} removed"
|
||||
}
|
||||
|
||||
as_postgres() { (cd / && runuser -u postgres -- "$@"); }
|
||||
|
||||
# remove_hba_block drops the block write_pg_hba_block maintains, and nothing else.
|
||||
remove_hba_block() { # file
|
||||
local tmp
|
||||
tmp="$(mktemp)"
|
||||
awk '
|
||||
$0 == "# BEGIN FELIS MANAGED HBA" { skip = 1; next }
|
||||
$0 == "# END FELIS MANAGED HBA" { skip = 0; blank = 1; next }
|
||||
blank && $0 == "" { blank = 0; next }
|
||||
{ blank = 0 }
|
||||
!skip { print }
|
||||
' "$1" > "$tmp"
|
||||
cat "$tmp" > "$1"
|
||||
rm -f "$tmp"
|
||||
}
|
||||
|
||||
purge_database() {
|
||||
[ "$PURGE" = 1 ] || return 0
|
||||
if ! systemctl is-active --quiet postgresql 2>/dev/null; then
|
||||
warn "PostgreSQL is not running; the ${DB_NAME} database and role are left in it"
|
||||
return 0
|
||||
fi
|
||||
local hba
|
||||
hba="$(as_postgres psql -tAc 'SHOW hba_file;' 2>/dev/null || true)"
|
||||
as_postgres psql -v ON_ERROR_STOP=1 -q <<SQL
|
||||
SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = '${DB_NAME}' AND pid <> pg_backend_pid();
|
||||
DROP DATABASE IF EXISTS ${DB_NAME};
|
||||
DROP ROLE IF EXISTS ${DB_USER};
|
||||
ALTER SYSTEM RESET listen_addresses;
|
||||
SQL
|
||||
[ -n "$hba" ] && [ -f "$hba" ] && remove_hba_block "$hba"
|
||||
systemctl restart postgresql
|
||||
ok "database and role '${DB_NAME}' dropped; PostgreSQL listens on its default address again"
|
||||
}
|
||||
|
||||
purge_images() {
|
||||
[ "$PURGE" = 1 ] || return 0
|
||||
command -v docker >/dev/null 2>&1 || return 0
|
||||
# The installer stops Docker after its builds; start it just long enough to clean up.
|
||||
local was_active=1 refs
|
||||
systemctl is-active --quiet docker || { was_active=0; systemctl start docker >/dev/null 2>&1 || return 0; }
|
||||
refs="$(docker image ls --format '{{.Repository}}:{{.Tag}}' | grep -E '^(registry\.felis\.svc:5000/felis/|felis/)' || true)"
|
||||
if [ -n "$refs" ]; then
|
||||
# shellcheck disable=SC2086 # one ref per word
|
||||
docker image rm -f $refs >/dev/null 2>&1 || true
|
||||
fi
|
||||
docker builder prune -af >/dev/null 2>&1 || true
|
||||
[ "$was_active" = 1 ] || systemctl stop docker docker.socket >/dev/null 2>&1 || true
|
||||
ok "Felis images and Docker's build cache removed"
|
||||
}
|
||||
|
||||
purge_state() {
|
||||
[ "$PURGE" = 1 ] || return 0
|
||||
# felis setup's tunnel credentials sit beside cloudflared's login (cert.pem, which stays:
|
||||
# it is the Cloudflare account's, not Felis's). The tunnel itself lives on in the account
|
||||
# until it is deleted there (docs/operations.md).
|
||||
local cred=""
|
||||
[ -r "$TUNNEL_CONFIG" ] \
|
||||
&& cred="$(sed -n 's/^credentials-file: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p' "$TUNNEL_CONFIG" | head -n 1)"
|
||||
if [ -n "$cred" ]; then rm -f "$cred"; fi
|
||||
rm -f "$TUNNEL_CONFIG"
|
||||
rm -rf "$STATE_DIR" "$DATA_DIR"
|
||||
ok "${STATE_DIR} and ${DATA_DIR} removed"
|
||||
}
|
||||
|
||||
main() {
|
||||
parse_args "$@"
|
||||
[ "$(id -u)" = 0 ] || die "run as root: sudo bash $0"
|
||||
[ -e "$STATE_DIR" ] || [ -e "$HOST_BIN" ] || [ -e "$OPT_DIR" ] \
|
||||
|| die "no Felis install here (${STATE_DIR}, ${HOST_BIN} and ${OPT_DIR} are all absent)"
|
||||
decide_k3s
|
||||
print_plan
|
||||
confirm
|
||||
final_backup
|
||||
|
||||
local game nano
|
||||
game="$(game_port)"
|
||||
nano="$(nano_port)"
|
||||
TUNNEL_CONFIG="$(tunnel_config)"
|
||||
remove_units
|
||||
case "$K3S_MODE" in
|
||||
remove) remove_k3s ;;
|
||||
keep) remove_from_cluster ;;
|
||||
esac
|
||||
remove_nft_tables
|
||||
remove_firewalld_rules "$game" "$nano"
|
||||
remove_host_files
|
||||
purge_database
|
||||
purge_images
|
||||
purge_state
|
||||
|
||||
if [ "$PURGE" = 1 ]; then
|
||||
ok "Felis is gone from this host"
|
||||
else
|
||||
ok "Felis is removed; the data stays in the ${DB_NAME} database, ${STATE_DIR} and ${DATA_DIR}"
|
||||
log "reinstalling reuses it: see docs/operations.md, \"Reinstall on top of kept data\""
|
||||
fi
|
||||
}
|
||||
|
||||
if [ "${FELIS_UNINSTALL_SOURCED:-0}" != 1 ]; then
|
||||
main "$@"
|
||||
fi
|
||||
@@ -0,0 +1,221 @@
|
||||
#!/bin/sh
|
||||
# Checks for deploy/uninstall.sh. Run it as: sh deploy/uninstall_test.sh
|
||||
#
|
||||
# The script is sourced with FELIS_UNINSTALL_SOURCED=1, every host path pointed into a
|
||||
# scratch directory, and the commands that would change the host (systemctl, k3s, nft,
|
||||
# firewall-cmd, runuser, docker, userdel) replaced by stubs that log their arguments.
|
||||
set -u
|
||||
|
||||
US="${1:-$(dirname "$0")/uninstall.sh}"
|
||||
[ -f "$US" ] || { echo "no such script: $US"; exit 1; }
|
||||
fails=0
|
||||
|
||||
expect() { # label needle haystack
|
||||
case "$3" in
|
||||
*"$2"*) echo "PASS $1" ;;
|
||||
*) echo "FAIL $1: expected <$2> in:"; echo "$3"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
}
|
||||
refute() { # label needle haystack
|
||||
case "$3" in
|
||||
*"$2"*) echo "FAIL $1: did not expect <$2> in:"; echo "$3"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS $1" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
root="$(mktemp -d)"
|
||||
trap 'rm -rf "$root"' EXIT
|
||||
|
||||
# fresh_host lays out what an install leaves: units, state, /opt/felis, the host binary, a
|
||||
# k3s with its uninstaller and a volume, a tunnel config and its credentials.
|
||||
fresh_host() {
|
||||
rm -rf "$root/h"
|
||||
mkdir -p "$root/h/units" "$root/h/etc" "$root/h/data/db-backups" "$root/h/opt/velocity" \
|
||||
"$root/h/bin" "$root/h/storage/pvc-1_minecraft_world-a-0" "$root/h/cf"
|
||||
for u in felis-db-backup.timer felis-db-backup.service felis-velocity.service felis-postgres-firewall.service; do
|
||||
printf '[Unit]\n' > "$root/h/units/$u"
|
||||
done
|
||||
printf '[Service]\nExecStart=/usr/local/bin/cloudflared --config %s tunnel run\n' "$root/h/etc/cloudflared.yml" \
|
||||
> "$root/h/units/cloudflared-felis.service"
|
||||
printf 'tunnel: abc\ncredentials-file: %s\n' "$root/h/cf/abc.json" > "$root/h/etc/cloudflared.yml"
|
||||
printf '{}\n' > "$root/h/cf/abc.json"
|
||||
printf 'bind = "0.0.0.0:25577"\n' > "$root/h/opt/velocity/velocity.toml"
|
||||
for f in secrets.env felis.host.toml bootstrap.done system-server-images velocity.fingerprint; do
|
||||
printf 'x\n' > "$root/h/etc/$f"
|
||||
done
|
||||
printf 'world\n' > "$root/h/storage/pvc-1_minecraft_world-a-0/level.dat"
|
||||
for b in felis k3s k3s-killall.sh k3s-uninstall.sh; do
|
||||
printf '#!/bin/sh\necho "RUN %s $*" >> "%s"\n' "$b" "$root/calls" > "$root/h/bin/$b"
|
||||
chmod +x "$root/h/bin/$b"
|
||||
done
|
||||
cat > "$root/h/hba.conf" <<'EOF'
|
||||
# BEGIN FELIS MANAGED HBA
|
||||
# Felis rules must precede distro defaults such as 127.0.0.1 ident.
|
||||
host felis felis 127.0.0.1/32 scram-sha-256
|
||||
# END FELIS MANAGED HBA
|
||||
|
||||
local all all peer
|
||||
host all all 127.0.0.1/32 ident
|
||||
EOF
|
||||
: > "$root/calls"
|
||||
}
|
||||
|
||||
# run_uninstall <namespaces> <args...>: runs main with the stubs; <namespaces> is what
|
||||
# `kubectl get namespaces` answers, or "down" for a k3s that does not answer.
|
||||
run_uninstall() {
|
||||
ns="$1"; shift
|
||||
NS="$ns" ROOT="$root" STATE_DIR="$root/h/etc" DATA_DIR="$root/h/data" HOST_BIN="$root/h/bin/felis" \
|
||||
OPT_DIR="$root/h/opt" UNIT_DIR="$root/h/units" K3S_BIN_DIR="$root/h/bin" \
|
||||
K3S_STORAGE="$root/h/storage" K3S_REGISTRIES="$root/h/registries.yaml" \
|
||||
CLOUDFLARED_BIN="$root/h/no-cloudflared" FELIS_UNINSTALL_SOURCED=1 bash -c '
|
||||
set -Eeuo pipefail
|
||||
. "$0"
|
||||
calls="$ROOT/calls"
|
||||
id() { if [ "${1:-}" = -u ]; then echo 0; else echo "ID $*" >> "$calls"; fi; }
|
||||
systemctl() {
|
||||
echo "SYSTEMCTL $*" >> "$calls"
|
||||
case "$*" in "is-active --quiet firewalld") return 1 ;; esac
|
||||
return 0
|
||||
}
|
||||
kube() {
|
||||
echo "KUBE $*" >> "$calls"
|
||||
case "$*" in
|
||||
"get namespaces"*) [ "$NS" = down ] && return 1; printf "%s\n" $NS ;;
|
||||
"get pv"*) printf "pvc-1 minecraft\npvc-9 other\n" ;;
|
||||
esac
|
||||
}
|
||||
nft() { echo "NFT $*" >> "$calls"; return 1; }
|
||||
runuser() {
|
||||
shift 3
|
||||
case "$*" in
|
||||
*"SHOW hba_file"*) echo "$ROOT/h/hba.conf" ;;
|
||||
*) echo "PSQL $* $(cat)" >> "$calls" ;;
|
||||
esac
|
||||
}
|
||||
docker() { echo "DOCKER $*" >> "$calls"; }
|
||||
userdel() { echo "USERDEL $*" >> "$calls"; }
|
||||
uname() { echo testhost; }
|
||||
main "$@"' "$US" "$@" 2>&1
|
||||
}
|
||||
|
||||
# --- keep-data, a cluster that runs only Felis -------------------------------------------
|
||||
fresh_host
|
||||
out="$(run_uninstall "default kube-system felis minecraft felis-build" --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
expect "keep-data takes a final bundle first" "RUN felis db backup -config $root/h/etc/felis.host.toml -label manual" "$calls"
|
||||
expect "a Felis-only cluster is removed with k3s's uninstaller" "RUN k3s-uninstall.sh" "$calls"
|
||||
expect "the pods are stopped before the volumes move" "RUN k3s-killall.sh" "$calls"
|
||||
kept="$(ls "$root/h/data/retained" 2>/dev/null)"
|
||||
expect "the volumes are set aside before k3s deletes them" "k3s-storage-" "$kept"
|
||||
[ -f "$root/h/data/retained/$kept/pvc-1_minecraft_world-a-0/level.dat" ] \
|
||||
&& echo "PASS a world survives the uninstall" \
|
||||
|| { echo "FAIL the world did not survive: $(ls -R "$root/h/data")"; fails=$((fails + 1)); }
|
||||
[ -f "$root/h/etc/secrets.env" ] && [ -f "$root/h/etc/felis.host.toml" ] \
|
||||
&& echo "PASS the secrets and felis.toml stay" \
|
||||
|| { echo "FAIL keep-data removed the secrets"; fails=$((fails + 1)); }
|
||||
[ ! -e "$root/h/etc/bootstrap.done" ] && [ ! -e "$root/h/etc/velocity.fingerprint" ] \
|
||||
&& echo "PASS the markers of the removed install go, so a reinstall starts fresh" \
|
||||
|| { echo "FAIL bootstrap.done or the proxy fingerprint was left"; fails=$((fails + 1)); }
|
||||
refute "keep-data leaves the database alone" "DROP DATABASE" "$calls"
|
||||
[ ! -e "$root/h/opt" ] && [ ! -e "$root/h/bin/felis" ] \
|
||||
&& echo "PASS /opt/felis and the host binary are removed" \
|
||||
|| { echo "FAIL /opt/felis or the host binary is still there"; fails=$((fails + 1)); }
|
||||
[ -z "$(ls "$root/h/units")" ] && echo "PASS every Felis unit file is removed" \
|
||||
|| { echo "FAIL units left: $(ls "$root/h/units")"; fails=$((fails + 1)); }
|
||||
expect "the timers are disabled" "SYSTEMCTL disable --now felis-db-backup.timer" "$calls"
|
||||
expect "the velocity user is removed" "USERDEL felis-velocity" "$calls"
|
||||
expect "the run ends pointing at the reinstall steps" "Reinstall on top of kept data" "$out"
|
||||
|
||||
# --- a failed final bundle stops everything ------------------------------------------------
|
||||
fresh_host
|
||||
printf '#!/bin/sh\necho "RUN felis $*" >> "%s"\nexit 1\n' "$root/calls" > "$root/h/bin/felis"
|
||||
out="$(run_uninstall "default felis minecraft" --yes)"
|
||||
expect "a failed bundle is fatal" "the final database bundle failed, so nothing was removed" "$out"
|
||||
[ -d "$root/h/opt" ] && [ -f "$root/h/units/felis-velocity.service" ] \
|
||||
&& echo "PASS nothing is removed when the bundle fails" \
|
||||
|| { echo "FAIL the uninstall went on after the bundle failed"; fails=$((fails + 1)); }
|
||||
run_uninstall "default felis minecraft" --yes --no-backup >/dev/null
|
||||
[ ! -d "$root/h/opt" ] && echo "PASS --no-backup goes on without one" \
|
||||
|| { echo "FAIL --no-backup did not remove anything"; fails=$((fails + 1)); }
|
||||
|
||||
# --- a shared cluster --------------------------------------------------------------------
|
||||
fresh_host
|
||||
out="$(run_uninstall "default kube-system felis minecraft felis-build shop" --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
refute "a cluster that runs something else keeps k3s" "RUN k3s-uninstall.sh" "$calls"
|
||||
expect "the reason names the other namespace" "shop" "$out"
|
||||
expect "Felis's volumes are retained before their claims go" 'KUBE patch pv pvc-1 -p {"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' "$calls"
|
||||
refute "a volume of another namespace is not touched" "patch pv pvc-9" "$calls"
|
||||
expect "Felis's namespaces are deleted" "KUBE delete namespace felis minecraft felis-build" "$calls"
|
||||
expect "the CRD is deleted" "KUBE delete crd minecraftservers.felis.lolicon.best" "$calls"
|
||||
|
||||
out="$(fresh_host; run_uninstall down --yes)"
|
||||
expect "a k3s that does not answer stops the run" "k3s does not answer" "$out"
|
||||
fresh_host
|
||||
run_uninstall down --yes --keep-k3s >/dev/null
|
||||
calls="$(cat "$root/calls")"
|
||||
refute "--keep-k3s never runs k3s's uninstaller" "RUN k3s-uninstall.sh" "$calls"
|
||||
|
||||
# --- purge -------------------------------------------------------------------------------
|
||||
fresh_host
|
||||
out="$(run_uninstall "default felis minecraft" --purge --yes)"
|
||||
calls="$(cat "$root/calls")"
|
||||
refute "purge takes no bundle" "db backup" "$calls"
|
||||
expect "purge drops the database" "DROP DATABASE IF EXISTS felis;" "$calls"
|
||||
expect "purge drops the role" "DROP ROLE IF EXISTS felis;" "$calls"
|
||||
expect "purge puts listen_addresses back" "ALTER SYSTEM RESET listen_addresses;" "$calls"
|
||||
[ ! -e "$root/h/etc" ] && [ ! -e "$root/h/data" ] && echo "PASS purge removes /etc/felis and /var/lib/felis" \
|
||||
|| { echo "FAIL purge left state behind"; fails=$((fails + 1)); }
|
||||
[ ! -e "$root/h/cf/abc.json" ] && echo "PASS purge removes the tunnel's credentials file" \
|
||||
|| { echo "FAIL the tunnel credentials are still there"; fails=$((fails + 1)); }
|
||||
[ ! -e "$root/h/data/retained" ] && echo "PASS purge does not set the volumes aside" \
|
||||
|| { echo "FAIL purge kept the volumes"; fails=$((fails + 1)); }
|
||||
hba="$(cat "$root/h/hba.conf")"
|
||||
refute "the Felis block leaves pg_hba.conf" "FELIS MANAGED" "$hba"
|
||||
expect "the distro's own rules stay" "host all all 127.0.0.1/32 ident" "$hba"
|
||||
case "$hba" in
|
||||
"local all all peer"*) echo "PASS no blank line is left where the block was" ;;
|
||||
*) echo "FAIL pg_hba.conf starts with: $(printf '%s' "$hba" | head -n 1)"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
expect "purge cleans Docker's build cache" "DOCKER builder prune -af" "$calls"
|
||||
|
||||
# --- the pieces read before they are removed ---------------------------------------------
|
||||
fresh_host
|
||||
lib() { STATE_DIR="$root/h/etc" OPT_DIR="$root/h/opt" UNIT_DIR="$root/h/units" FELIS_UNINSTALL_SOURCED=1 \
|
||||
bash -c '. "$0"; '"$1" "$US"; }
|
||||
expect "the game port comes from velocity.toml" "25577" "$(lib game_port)"
|
||||
expect "FELIS_GAME_PORT overrides it" "25599" "$(FELIS_GAME_PORT=25599 lib game_port)"
|
||||
rm "$root/h/opt/velocity/velocity.toml"
|
||||
expect "without velocity.toml the port is the default" "25565" "$(lib game_port)"
|
||||
printf '[Service]\nExecStart=/usr/local/bin/felis nano -listen 10.0.0.5:8082 -config x\n' > "$root/h/units/felis-nano.service"
|
||||
expect "the nano port comes from its unit" "8082" "$(lib nano_port)"
|
||||
expect "the tunnel config comes from the cloudflared unit" "$root/h/etc/cloudflared.yml" "$(lib tunnel_config)"
|
||||
rm "$root/h/units/cloudflared-felis.service"
|
||||
expect "without the unit the tunnel config is the default path" "$root/h/etc/cloudflared.yml" "$(lib tunnel_config)"
|
||||
|
||||
out="$(FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; parse_args --bogus' "$US" 2>&1)"
|
||||
expect "an unknown option is refused" "unknown option: --bogus" "$out"
|
||||
printf 'nope\n' > "$root/tty"
|
||||
out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)"
|
||||
expect "any answer but the word stops it" "not confirmed; nothing was changed" "$out"
|
||||
printf 'yes\n' > "$root/tty"
|
||||
out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)"
|
||||
expect "yes goes on" "WENT ON" "$out"
|
||||
out="$(CONFIRM_TTY="$root/tty" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; PURGE=1; confirm; echo WENT ON' "$US" 2>&1)"
|
||||
expect "a purge wants the word purge" "not confirmed" "$out"
|
||||
out="$(CONFIRM_TTY="$root/no-tty/x" FELIS_UNINSTALL_SOURCED=1 bash -c '. "$0"; confirm; echo WENT ON' "$US" 2>&1)"
|
||||
expect "with no terminal it asks for --yes" "no terminal to confirm on; re-run with --yes" "$out"
|
||||
|
||||
# Every unit the installer writes is one the uninstaller removes: a timer left behind
|
||||
# keeps firing a felis binary that is gone.
|
||||
units="$(awk '/^FELIS_UNITS=\(/ { f = 1; next } f && /^\)/ { f = 0 } f' "$US")"
|
||||
for u in $(sed -n 's|^[A-Z_]*="/etc/systemd/system/\([^"]*\)"$|\1|p' "$(dirname "$US")/bootstrap.sh"); do
|
||||
expect "the uninstaller removes $u" " $u" " $(printf '%s' "$units" | tr '\n' ' ')"
|
||||
done
|
||||
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS"
|
||||
else
|
||||
echo "$fails FAILED"
|
||||
fi
|
||||
exit "$fails"
|
||||
@@ -10,7 +10,10 @@
|
||||
# and hashed here; Paper and Velocity come from Fill's content-addressed URLs.
|
||||
#
|
||||
# Review the diff before committing: MC_VERSION moves the login gate's protocol, and the
|
||||
# lobby, plain-Paper image and every client follow it.
|
||||
# lobby, plain-Paper image and every client follow it. The plugins compile against these
|
||||
# same builds (paper-api, the Limbo API, velocity-api) with their checksums pinned in
|
||||
# plugins/*/gradle/verification-metadata.xml, so a moved build also moves those; go test .
|
||||
# names what is left to bring along.
|
||||
set -Eeuo pipefail
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
@@ -70,4 +73,4 @@ if [ "$check" = 1 ]; then
|
||||
die "upstream has newer builds than game-stack.lock"
|
||||
fi
|
||||
cp "$tmp" "$LOCK"
|
||||
ok "game-stack.lock updated; run go test . and the bootstrap tests, then commit"
|
||||
ok "game-stack.lock updated; move plugins/paper's paper-api pin to the new Paper build and regenerate the plugins' gradle/verification-metadata.xml (plugins/README.md \"Dependency verification\"), run go test . and the bootstrap tests, then commit"
|
||||
+13
-12
@@ -28,7 +28,10 @@ A grep across `*.md` and `*.go` returns both sets; only the Go ones are seams.
|
||||
|
||||
- `internal/updates/seams.go:32` — `Notifier`. `internal/mail` sends OTP over SMTP,
|
||||
but nothing adapts it to this interface and no in-game channel exists. `felis
|
||||
update` passes nil deliberately: a human typing the command is the notification.
|
||||
update` passes nil deliberately. The notification is the panel instead:
|
||||
`felis-update-check.timer` runs `felis update --record` daily on the host, which
|
||||
stores the report under `platform_settings.update_report`, and **Admin → Updates →
|
||||
Component versions** shows it with the command that applies each update.
|
||||
- `internal/updates/seams.go:43` — `Applier`. Nothing applies an update anywhere. A
|
||||
nil applier is not silent — `Run` records `errNoApplier` against every planned
|
||||
apply, so a mis-scheduled apply is loud rather than lost.
|
||||
@@ -37,11 +40,10 @@ A grep across `*.md` and `*.go` returns both sets; only the Go ones are seams.
|
||||
control-plane Deployment image, and the Velocity jar inspection. Both are answered
|
||||
on the host path (see "Built" below), so this gap is specific to a caller that has
|
||||
a cluster client instead of the node.
|
||||
- `internal/api/handlers_updates.go:21,28` — the maintenance window persists and the
|
||||
API serves it, but the in-cluster CronJob that would hand a real window to a runner
|
||||
does not exist. `felis update` runs with a zero window, under which every
|
||||
`Scheduled` component degrades to a notify, so no path can currently claim an
|
||||
apply is under way.
|
||||
- `internal/api/handlers_updates.go` — the maintenance window is advisory: no
|
||||
in-cluster runner applies updates. `felis update` reads the stored window, prints
|
||||
where now sits against it and warns before an apply outside it; the runner itself
|
||||
still runs with a zero window, so no path can claim an apply is under way.
|
||||
- `internal/submit/blobstore.go` — CLOSED 2026-09-22. The uploads PVC still cannot
|
||||
cross namespaces, so the transport went through the API instead of a mount: the
|
||||
derived context ref is now the internal-face URL
|
||||
@@ -116,12 +118,11 @@ worth revisiting.
|
||||
|
||||
## Wired since the marker was written
|
||||
|
||||
- `internal/api/handlers_email_otp.go:54,223,227` and `internal/api/api.go:84` —
|
||||
SMTP shipped on
|
||||
2026-07-20 (`internal/mail`, wired at `cmd/felis/api.go:264`). The nil-`Mailer`
|
||||
branch that logs the code server-side is a runtime fallback for an install with no
|
||||
`[smtp]` section, not an unbuilt feature. The comments are accurate; the reading
|
||||
"Felis cannot send mail" is not.
|
||||
- `internal/api/handlers_email_otp.go` and `internal/api/api.go` — SMTP shipped on
|
||||
2026-07-20 (`internal/mail`, wired in `cmd/felis/api.go`). An install with no
|
||||
`[smtp]` section leaves the `Mailer` nil, and every door that mails a code answers
|
||||
503 `mail_unavailable`; codes are never logged. The reading "Felis cannot send
|
||||
mail" is stale.
|
||||
- `internal/config/config.go:117` — was stale. It described the upload transport as a
|
||||
deferred integration after both backends had shipped (`LocalContextStore`,
|
||||
`S3ContextStore`, selected in `cmd/felis/api.go` by the shape of the configured
|
||||
|
||||
+1518
-211
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,350 @@
|
||||
# Felis Operations Guide
|
||||
|
||||
What a Felis host needs, how big it should be, how to take Felis off it again, and where
|
||||
the disaster-recovery procedures live. Fault-finding is in
|
||||
[troubleshooting.md](troubleshooting.md); this document refers to its sections as §N.
|
||||
|
||||
Evidence tags follow troubleshooting.md: **[VM-VERIFIED]** was run on a real host,
|
||||
**[CI]** runs end to end on every push to main (`.github/workflows/e2e.yml`),
|
||||
**[GO-TESTED]** / **[SH-TESTED]** is covered by `go test` or the shell tests under
|
||||
`deploy/`, **[CODE-ONLY]** is what the code does and has not been run end to end.
|
||||
|
||||
## 1. Supported hosts
|
||||
|
||||
`deploy/bootstrap.sh` provisions a single node. It needs systemd, root, and one of the
|
||||
package managers below; everything else (Docker, k3s, PostgreSQL, the JRE, cloudflared)
|
||||
it installs.
|
||||
|
||||
| OS family | Package manager | Architectures | Status |
|
||||
|---|---|---|---|
|
||||
| CentOS Stream 9 (firewalld active, PostgreSQL 13) | dnf | aarch64 | **[VM-VERIFIED]** install, same-version rerun, upgrade, uninstall and reinstall |
|
||||
| Ubuntu 24.04 LTS | apt | x86_64 | **[CI]** fresh install, same-commit rerun, and upgrade from the newest release to the pushed commit |
|
||||
| RHEL / Rocky / Alma 9, Fedora | dnf | x86_64, aarch64 | [CODE-ONLY] same code path as CentOS Stream |
|
||||
| Debian 12, other Ubuntu releases | apt | x86_64, aarch64 | [CODE-ONLY] |
|
||||
| openSUSE Leap / Tumbleweed | zypper | x86_64, aarch64 | [CODE-ONLY] |
|
||||
| Arch Linux | pacman | x86_64, aarch64 | [CODE-ONLY] |
|
||||
|
||||
Pinned component versions (a fresh install gets exactly these; an installed k3s or
|
||||
cloudflared is left as it is, see §4):
|
||||
|
||||
| Component | Version | Where it is pinned |
|
||||
|---|---|---|
|
||||
| k3s | v1.36.4+k3s1 | `FELIS_K3S_VERSION` in `bootstrap.sh` |
|
||||
| cloudflared | 2026.9.1 | `FELIS_CLOUDFLARED_VERSION`, sha256 per architecture |
|
||||
| Temurin JRE (Velocity) | 25, patch build pinned | `FELIS_JRE_VERSION`, sha256 per architecture |
|
||||
| Go (nano builds) | 1.26.8 | `GO_PINNED_VERSION`, sha256 per architecture |
|
||||
| Minecraft / Limbo / Paper / Velocity / LuckPerms | `deploy/game-stack.lock` | §15b |
|
||||
| PostgreSQL | the distribution's package | 13 and 18 are exercised by the `pgint` CI job |
|
||||
|
||||
32-bit hosts are not supported: there is no k3s, JRE or Go build the installer will fetch
|
||||
for them.
|
||||
|
||||
One node is the whole supported shape. A world volume is a ReadWriteOnce claim on the
|
||||
node's local-path storage, so a game server's pod is pinned to the node that first
|
||||
scheduled it and cannot move when that node fails; the operator and felis-api each run
|
||||
as a single replica without leader election, so an upgrade or a node restart pauses
|
||||
wakes and stops until their pod is back. Joining k3s agents to the cluster is untested
|
||||
and gains no failover. A multi-node shape would need, at least, storage that can follow a
|
||||
pod to another node and leader election in felis-operator (controller-runtime's
|
||||
`LeaderElection`) so a second replica can stand by.
|
||||
|
||||
### While felis-api restarts
|
||||
|
||||
An installer rerun that changes felis-api, a node restart or a crashed pod takes the API
|
||||
away until its new pod is ready: about 12 s on the reference VM (`kubectl rollout
|
||||
restart` to Available). Its Deployment keeps one replica with the Recreate strategy, so
|
||||
the old pod is gone before the new one starts. Two pods at once would be wrong for
|
||||
felis-api: the uploads volume is ReadWriteOnce, a chunked upload is serialized inside the
|
||||
process, and the build reconciler, restore settler, registry pruner, upload reapers and
|
||||
audit retention run in-process without leader election, so each would run twice. During
|
||||
the window:
|
||||
|
||||
- Players already on a server stay there; game servers keep running.
|
||||
- A player leaving the login gate or joining a server by its address is admitted when
|
||||
felis-api confirmed their link within the last 10 minutes; anyone else is told login
|
||||
verification is temporarily unavailable.
|
||||
- The login gate retries a new login for up to 60 s and tells the player it is retrying,
|
||||
so a restart shorter than that only delays the login.
|
||||
- Wakes, stops, `/link` and the panel wait for the API.
|
||||
- A Velocity restart in the window routes on
|
||||
`/opt/felis/velocity/plugins/felis-link/last-servers.json`, the last server list the
|
||||
API answered with, until a refresh succeeds (every 15 s).
|
||||
|
||||
## 2. Sizing
|
||||
|
||||
### What the platform itself uses
|
||||
|
||||
Measured on the verification host (4 vCPU, 5.5 GB RAM, 6 GB swap, CentOS Stream 9
|
||||
aarch64) with the control plane, the login and lobby system servers and one idle Paper
|
||||
server running **[VM-VERIFIED]**:
|
||||
|
||||
| Process | Resident memory |
|
||||
|---|---|
|
||||
| k3s (server, kubelet, containerd) | ~1.1 GB |
|
||||
| Velocity (`-Xms512M -Xmx1G`, heap pre-touched) | ~0.73 GB |
|
||||
| lobby (Paper, pod limit 1 GiB) | ~0.7–0.85 GB |
|
||||
| login (Limbo, pod limit 512 MiB) | ~0.16 GB |
|
||||
| felis-api, felis-operator, registry gate | ~50 MB each |
|
||||
| PostgreSQL | ~30 MB plus page cache |
|
||||
| **Total in use** | **~3.4 GB** |
|
||||
|
||||
Every game server adds the memory its owner gave it: the pod's limit equals its request,
|
||||
and the JVM heap is derived from it (§1a). Quotas cap it per user (panel → 管理 → 配额).
|
||||
|
||||
The installer's own peak is the image builds (Docker plus a Gradle container); it stops
|
||||
Docker afterwards so that memory goes back to the servers. On a host under 2 GB of RAM
|
||||
without swap it adds a 2 GiB `/swapfile`.
|
||||
|
||||
### Recommendations
|
||||
|
||||
| Concurrent players | Game servers running | CPU | RAM | `FELIS_VELOCITY_XMX` |
|
||||
|---|---|---|---|---|
|
||||
| up to 20 | 1–2 small | 2 vCPU | 4 GB + 2 GB swap | 1G (default) |
|
||||
| up to 100 | 3–5 | 4 vCPU | 8–16 GB | 1G |
|
||||
| up to 300 | 5–10 | 8 vCPU | 16–32 GB | 2G |
|
||||
| 300+ | more | 8+ vCPU | 32 GB+ | 3G–4G |
|
||||
|
||||
The player-count rows are planning figures, not measurements: a Minecraft server's cost
|
||||
depends mostly on what its players do (view distance, redstone, mods). Size RAM as the
|
||||
platform's ~3.5 GB plus the sum of the servers you expect to run at once, then add a
|
||||
quarter for the page cache and PostgreSQL. Velocity itself needs little per player; raise
|
||||
its heap when `journalctl -u felis-velocity` shows long GC pauses or `OutOfMemoryError`.
|
||||
|
||||
`FELIS_VELOCITY_XMX` (default `1G`, at least `256M`, written `<n>M` or `<n>G`) is read on
|
||||
every installer run. The initial heap stays at 512M, or equals the maximum when that is
|
||||
lower. Changing it rewrites the unit, and the rerun restarts the proxy, which disconnects
|
||||
everyone online; do it in a quiet hour **[VM-VERIFIED]**:
|
||||
|
||||
```
|
||||
curl -fsSL <raw-url>/deploy/bootstrap.sh | sudo FELIS_VELOCITY_XMX=2G bash
|
||||
```
|
||||
|
||||
### Disk
|
||||
|
||||
| What | Where | Size |
|
||||
|---|---|---|
|
||||
| Worlds | one volume per server under `/var/lib/rancher/k3s/storage` | what the world grows to |
|
||||
| World archives | the `felis-backups` volume (`FELIS_BACKUP_STORAGE`, default 10Gi requested) | about one compressed world per backup kept |
|
||||
| In-cluster registry | the `registry` volume (default 10Gi requested) | 2–3 GB for the stock images; grows with custom builds, pruned daily (§9) |
|
||||
| k3s's containerd images | `/var/lib/rancher/k3s/agent/containerd` | 6–9 GB |
|
||||
| Docker's images and build cache | `/var/lib/containerd` (Docker's containerd store) | 5–10 GB after repeated upgrades |
|
||||
| Toolchains and sources | `/opt/felis` | ~2.5 GB |
|
||||
| Database bundles | `/var/lib/felis/db-backups` | a few MB each, 14 daily kept |
|
||||
|
||||
k3s's local-path volumes do not enforce the requested sizes (§9), so every volume shares
|
||||
the root filesystem. Give the host at least **40 GB**, and 60 GB or more once worlds and
|
||||
custom images accumulate. The watchdog mails the owners when a watched filesystem passes
|
||||
its threshold, and §13b covers a full disk. `docker builder prune -af` (with Docker
|
||||
started) reclaims the build cache when space is short; the next upgrade rebuilds it.
|
||||
|
||||
## 3. Uninstall
|
||||
|
||||
`deploy/uninstall.sh` takes off what the installer put on. It prints what it will remove
|
||||
and asks before it starts (`--yes` skips the question) **[SH-TESTED]
|
||||
[VM-VERIFIED]**:
|
||||
|
||||
```
|
||||
curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --yes # keep the data
|
||||
curl -fsSL <raw-url>/deploy/uninstall.sh | sudo bash -s -- --purge # remove the data too
|
||||
```
|
||||
|
||||
With a private repository, fetch it the way the README fetches `bootstrap.sh`.
|
||||
|
||||
Both modes remove the `felis-*` systemd units and `cloudflared-felis.service`, the
|
||||
Velocity user, `/opt/felis`, `/usr/local/bin/felis`, the installer's cloudflared binary
|
||||
(unless another unit runs it), the `felis_postgres` and `felis_edge` nftables tables and
|
||||
the firewalld ports the installer opened. k3s goes with k3s's own `k3s-uninstall.sh` when
|
||||
the cluster holds nothing but Felis's namespaces; when it runs anything else only
|
||||
`felis`, `minecraft`, `felis-build` and the MinecraftServer CRD are deleted.
|
||||
`--keep-k3s` and `--remove-k3s` override that choice.
|
||||
|
||||
| | keep data (default) | `--purge` |
|
||||
|---|---|---|
|
||||
| Final database bundle | taken first (`felis db backup -label manual`); a failure stops the uninstall before anything is removed. `--no-backup` skips it | none |
|
||||
| `felis` database and role | kept | dropped; `listen_addresses` and `pg_hba.conf` go back to how they were |
|
||||
| `/etc/felis` (secrets, `felis.toml`, `offsite.env`, tunnel config) | kept; `bootstrap.done` and the per-run records go | deleted, with the tunnel's credentials file |
|
||||
| `/var/lib/felis` (database bundles) | kept | deleted |
|
||||
| Worlds, archives, registry, uploads | moved to `/var/lib/felis/retained/k3s-storage-<stamp>/` (with `--keep-k3s`: their volumes switch to `Retain` and stay in place) | deleted |
|
||||
| Felis images, Docker build cache | kept | deleted |
|
||||
|
||||
Neither mode removes packages (Docker, PostgreSQL, git, nftables) or the swap file: other
|
||||
software may use them. On a host that should end up bare:
|
||||
|
||||
```
|
||||
sudo swapoff /swapfile && sudo rm /swapfile && sudo sed -i '\|^/swapfile |d' /etc/fstab
|
||||
sudo dnf remove docker-ce docker-ce-cli containerd.io postgresql-server # or apt/zypper/pacman
|
||||
```
|
||||
|
||||
The Cloudflare side outlives the host. After an uninstall that is final, delete the
|
||||
tunnel (Zero Trust → Networks → Tunnels, or `cloudflared tunnel delete <name>`), its
|
||||
DNS records for the panel hostnames, and the Access application.
|
||||
|
||||
### Reinstall on top of kept data
|
||||
|
||||
A keep-data uninstall leaves everything a reinstall needs. The installer reuses
|
||||
`/etc/felis/secrets.env`, so the database password and the forwarding and session
|
||||
secrets are unchanged, and it migrates the kept database instead of creating one
|
||||
**[VM-VERIFIED]**.
|
||||
|
||||
Each step below was run on the reference VM after a keep-data uninstall, and the
|
||||
restored worlds matched their kept `level.dat` checksums **[VM-VERIFIED]**. `kept` names
|
||||
the directory the uninstall moved the volumes to:
|
||||
|
||||
```
|
||||
kept="$(ls -d /var/lib/felis/retained/k3s-storage-* | tail -n 1)"
|
||||
store=/var/lib/rancher/k3s/storage
|
||||
```
|
||||
|
||||
1. Install as usual (`curl ... | sudo bash`). Name the same root domain if it was not
|
||||
the `<ip>.nip.io` default: `felis.host.toml` is kept, and the installer reads the
|
||||
domain from it.
|
||||
2. Run `sudo felis setup`. It recreates the login and lobby servers; the Owner already
|
||||
exists, so it opens on the status screen and you can quit there.
|
||||
3. Put the image registry and the uploads back. They hold every custom server image
|
||||
and uploaded file; without the registry, a restored server fails to pull its image.
|
||||
|
||||
```
|
||||
sudo k3s kubectl -n felis scale deploy/registry deploy/felis-api --replicas=0
|
||||
sudo k3s kubectl -n felis wait --for=delete pod -l app.kubernetes.io/component=registry --timeout=120s
|
||||
sudo k3s kubectl -n felis wait --for=delete pod -l app.kubernetes.io/component=api --timeout=120s
|
||||
sudo rsync -a --delete "$kept"/pvc-*_felis_registry/ "$(ls -d $store/pvc-*_felis_registry)"/
|
||||
sudo rsync -a --delete "$kept"/pvc-*_felis_felis-uploads/ "$(ls -d $store/pvc-*_felis_felis-uploads)"/
|
||||
sudo k3s kubectl -n felis scale deploy/registry deploy/felis-api --replicas=1
|
||||
```
|
||||
|
||||
Then run the installer once more. It pushes this release's images over the older
|
||||
copies the kept registry carried.
|
||||
4. Bring the game servers back. The final bundle holds every MinecraftServer as it was;
|
||||
the selector skips login and lobby, which step 2 created for this release:
|
||||
|
||||
```
|
||||
b="$(ls -t /var/lib/felis/db-backups/felis-db-*-manual.tar | head -n 1)"
|
||||
tar -xOf "$b" k8s/minecraftservers.json \
|
||||
| sudo k3s kubectl apply -l '!felis.lolicon.best/system-role' -f -
|
||||
```
|
||||
|
||||
5. Put each world back. A server's volume exists once it has started once, so start it
|
||||
from the panel, stop it again, and copy the kept world over the new one:
|
||||
|
||||
```
|
||||
s=<server>
|
||||
sudo rsync -a --delete "$kept"/pvc-*_minecraft_world-$s-0/ "$(ls -d $store/pvc-*_minecraft_world-$s-0)"/
|
||||
```
|
||||
|
||||
Then start it. The lobby works the same way: stop it with
|
||||
`sudo k3s kubectl -n minecraft patch minecraftserver lobby --type=merge -p '{"spec":{"desiredState":"Stopped"}}'`,
|
||||
copy `world-lobby-0`, and patch it back to `Running`.
|
||||
6. Bring the archives back so the panel's restore points work again. The archive volume
|
||||
appears with the first backup, so back up any server from the panel first, then:
|
||||
|
||||
```
|
||||
sudo rsync -a "$kept"/pvc-*_minecraft_felis-backups/ "$(ls -d $store/pvc-*_minecraft_felis-backups)"/
|
||||
```
|
||||
|
||||
With an off-site bucket configured, `sudo felis offsite fetch-worlds` fetches them
|
||||
instead (troubleshooting §16).
|
||||
7. Delete `/var/lib/felis/retained/` once every server is back.
|
||||
|
||||
## 4. Upgrading the pieces around Felis
|
||||
|
||||
A rerun of the installer upgrades Felis itself (§15). The components it installs keep
|
||||
the version they were installed with unless noted:
|
||||
|
||||
| Component | How a rerun treats it | Upgrade |
|
||||
|---|---|---|
|
||||
| Velocity, Limbo, Paper, LuckPerms | follow `deploy/game-stack.lock` | rerun after a release that moves the lock (§15b) |
|
||||
| Temurin JRE | moves to the pinned patch build | rerun |
|
||||
| k3s | left alone | rerun with `FELIS_UPGRADE_DEPS=1`: moves to the pinned release through that tag's install script, one minor version at a time (a bigger jump stops before anything changes and names the release to go through), never backwards |
|
||||
| cloudflared | left alone | rerun with `FELIS_UPGRADE_DEPS=1`: swaps `/usr/local/bin/cloudflared` for the pinned, sha256-checked release and restarts `cloudflared-felis`; a cloudflared the distribution installed stays with its package manager |
|
||||
| PostgreSQL | the distribution's package | the package manager for a minor release; a major version needs `pg_upgrade` first (below) |
|
||||
| Docker, git, nftables | distribution packages | the package manager |
|
||||
|
||||
```sh
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh \
|
||||
| sudo FELIS_UPGRADE_DEPS=1 bash
|
||||
```
|
||||
|
||||
`sudo felis update` reports Felis, Velocity, k3s, cloudflared, the JRE and PostgreSQL
|
||||
against their newest releases; `--k3s`, `--cloudflared`, `--jre` and `--postgres` narrow
|
||||
it to one. PostgreSQL is compared within its major, since a minor release is a package
|
||||
update, and a major past its end of life gets a note naming the current one.
|
||||
|
||||
The installer also sets up `felis-update-check.timer`, which runs `felis update --record`
|
||||
once a day around 05:30 (and at boot after a missed run). `--record` stores the result
|
||||
in `platform_settings`, and the panel's **Admin → Updates → Component versions** card
|
||||
shows it: each component's installed and newest version, and for the ones with a newer
|
||||
release the `sudo felis update --<component>` line that prints how to apply it. Felis
|
||||
applies nothing on its own; the installer re-run above is the apply path. The card turns
|
||||
red when the newest record is older than 26 hours, meaning the timer stopped:
|
||||
|
||||
```sh
|
||||
systemctl list-timers felis-update-check.timer
|
||||
journalctl -u felis-update-check -n 50 --no-pager
|
||||
sudo felis update --record # record a fresh check now
|
||||
```
|
||||
|
||||
### PostgreSQL major versions [CODE-ONLY]
|
||||
|
||||
The installer takes the major the distribution ships (13 on EL9) and never moves it. To
|
||||
go to a newer one, stop the writers, keep a dump, then use the distribution's upgrade
|
||||
path:
|
||||
|
||||
```sh
|
||||
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator --replicas=0
|
||||
sudo -u postgres pg_dumpall > /root/felis-pg-$(date +%F).sql
|
||||
# EL9: sudo systemctl stop postgresql; sudo dnf module switch-to postgresql:16
|
||||
# sudo dnf install postgresql-upgrade; sudo postgresql-setup --upgrade
|
||||
# Debian/Ubuntu: sudo pg_upgradecluster <old-major> main
|
||||
sudo systemctl start postgresql
|
||||
sudo k3s kubectl -n felis scale deploy/felis-api deploy/felis-operator --replicas=1
|
||||
```
|
||||
|
||||
### The MinecraftServer CRD [VM-VERIFIED]
|
||||
|
||||
Every rerun applies the CRD embedded in the `felis` binary (`felis bootstrap-assets crd`).
|
||||
It serves and stores the single version `v1alpha1`, and the apiserver refuses values the
|
||||
operator cannot act on:
|
||||
|
||||
| Field | Accepted |
|
||||
|---|---|
|
||||
| `spec.rcon.port` | unset, `0` or `25575`: the allow-rcon NetworkPolicy opens only 25575, so any other port leaves the server unprobeable |
|
||||
| `spec.startup.timeoutSeconds`, `readinessTimeoutSeconds` | 0 – 86400 |
|
||||
| `spec.startup.healthHTTPPort` | 0 – 65535 |
|
||||
| `spec.lifecycle.terminationGracePeriodSeconds` | 0 – 3600 |
|
||||
| `spec.idle.emptySecondsBeforeStop` | 0 – 604800 (the panel caps it at 86400) |
|
||||
|
||||
`0` means the operator's default throughout. An object stored before these rules keeps an
|
||||
out-of-range value until someone edits that field (CRD validation ratcheting). The operator
|
||||
reads a negative value as its default and an oversized one as written, so fix such a
|
||||
value by hand: `kubectl -n minecraft edit minecraftserver <name>`.
|
||||
|
||||
**Moving to `v1beta1` (planned, not built).** The first breaking change to the spec ships as a new
|
||||
version, in this order, each step one release:
|
||||
|
||||
1. The CRD serves `v1alpha1` and `v1beta1`, storage stays `v1alpha1`. While the two
|
||||
schemas carry the same fields, `conversion.strategy: None` suffices; a renamed or
|
||||
reshaped field needs a conversion webhook, which felis-operator would serve.
|
||||
2. Storage moves to `v1beta1`. The installer rewrites every object so etcd holds the new
|
||||
version (`kubectl get minecraftservers -A -o json | kubectl replace -f -`), then sets
|
||||
`status.storedVersions` of the CRD to `["v1beta1"]`.
|
||||
3. A later release stops serving `v1alpha1`. Felis itself reads through one Go type at a
|
||||
time, so the operator and felis-api switch in the release that moves storage.
|
||||
|
||||
## 5. Disaster recovery
|
||||
|
||||
The procedures are in §16: what a database bundle holds, restoring one on the same host,
|
||||
rolling back an upgrade, and rebuilding on a new host from the off-site copy. For a
|
||||
production install:
|
||||
|
||||
- **Configure the off-site copy** (`FELIS_OFFSITE_*`, §16 "Keep a copy somewhere
|
||||
else"). Without it the world archives sit on the same disk as the worlds, and the
|
||||
database bundles on the same disk as the database; losing the disk loses both. The
|
||||
installer ends with `NO OFF-SITE COPY` until it is set.
|
||||
- **Keep the off-site encryption key off the host**, in a password manager. The bucket
|
||||
holds only sealed objects.
|
||||
- **Keep one database bundle off the host** as well when there is no bucket. It contains
|
||||
`secrets.env`, which a rebuild needs to read the rest.
|
||||
- **Rehearse the rebuild** once on a spare VM: §16 "Rebuild on a new host", steps 1–5,
|
||||
then log in and restore one world. `felis offsite status` and `felis db check` exit
|
||||
non-zero when the copy or the newest bundle is stale; wire them into your monitoring,
|
||||
or rely on the watchdog's mail.
|
||||
+490
-95
@@ -140,8 +140,9 @@ message is the verbatim dial error:
|
||||
backend's `rcon.password`. Reconcile the two. [GO-TESTED that this maps to
|
||||
`RconNotReachable`.]
|
||||
- `connection refused` / `i/o timeout` → the backend has not opened the RCON
|
||||
port yet, RCON is disabled in `server.properties`, or `spec.rcon.port`
|
||||
(default 25575) is wrong. [INTEGRATION-ONLY for the live handshake.]
|
||||
port yet, RCON is disabled in `server.properties`, or the image listens on a
|
||||
port other than 25575 (the CRD accepts only that one for `spec.rcon.port`,
|
||||
operations.md §4). [INTEGRATION-ONLY for the live handshake.]
|
||||
|
||||
The per-probe timeout is a fixed 5s in code (`prober.go:45`, shortened further if
|
||||
the reconcile context has a nearer deadline). It is **not** derived from
|
||||
@@ -257,6 +258,15 @@ What holds the world, in order:
|
||||
kubectl -n minecraft annotate minecraftserver <name> felis.lolicon.best/maintenance-
|
||||
```
|
||||
|
||||
3. The idle-world reaper (§10), which has no Job of its own: it writes the same
|
||||
annotation as `reap@<RFC3339>` while it archives and deletes an idle world,
|
||||
and rewrites it every 30 seconds, so the lock stays fresh however long the
|
||||
archive takes. The refusal names `the idle-world reaper`. A reaper pod that
|
||||
dies mid-archive stops rewriting it, and the lock lapses two minutes after
|
||||
the last write. Dropping it by hand also ends the reap: the reaper checks
|
||||
the lock before it deletes the world volume, keeps the world and retries the
|
||||
next day.
|
||||
|
||||
A restore, backup or file write refused with `409 not_stopped` although the
|
||||
panel shows `Stopped` means the game pod is still terminating (its shutdown save
|
||||
can take a while); retry once `kubectl -n minecraft get pods -l
|
||||
@@ -295,45 +305,32 @@ point.
|
||||
|
||||
## 5. Web panel returns 401 / 403 (Zero-Trust / Cloudflare Access)
|
||||
|
||||
The external face accepts either a Cloudflare Access JWT
|
||||
(`Cf-Access-Jwt-Assertion` header) **or** a local session cookie. The error
|
||||
envelope is always `{"error":{"code","message","request_id"}}`. [GO-TESTED.]
|
||||
The external face has one credential: the `felis_session` cookie the sign-in
|
||||
doors mint. Cloudflare Access, when the install sits behind it, is enforced at
|
||||
the Cloudflare edge only — felis-api does not read the `Cf-Access-Jwt-Assertion`
|
||||
header, so a request that reaches the origin some other way still has to sign in,
|
||||
and the account and its role always come from the `users` table. The edge setup
|
||||
fences the panel NodePort to loopback (the `felis_edge` nftables table), so every
|
||||
request reaches the API through cloudflared and Access stays in front of the
|
||||
operator console; check `nft list table inet felis_edge` if you doubt it. The error envelope is always
|
||||
`{"error":{"code","message","request_id"}}`. [GO-TESTED.]
|
||||
|
||||
- **`401 unauthorized`** — not authenticated: no/invalid Access JWT and no valid
|
||||
session. [GO-TESTED.]
|
||||
- **`401 unauthorized`** — no valid session cookie. [GO-TESTED.]
|
||||
- **`403 forbidden`** — authenticated but not permitted (e.g. a non-admin
|
||||
principal hitting an admin route; `IsAdmin()` requires `role=admin` **and**
|
||||
arrival via the admin Access audience/host). [GO-TESTED.]
|
||||
principal hitting an admin route; `IsAdmin()` requires a staff role **and** a
|
||||
request on the operator console host). [GO-TESTED.]
|
||||
|
||||
### 5a. Every external request 401s on a fresh deploy
|
||||
### 5a. Staff routes 403 on a local IP URL
|
||||
|
||||
The Access verifier is wired **fail-closed**: `Keyfunc` (the JWKS key function)
|
||||
is `nil` until deployment wiring supplies it. With a nil Keyfunc, **every** JWT
|
||||
verification fails, and startup logs:
|
||||
|
||||
```
|
||||
felis api: external face fails closed (Access JWKS key function not configured)
|
||||
```
|
||||
|
||||
[INTEGRATION-ONLY — the live JWKS path is a deployment point.] This is intended:
|
||||
the panel rejects all callers until JWKS is configured. Fix by wiring the
|
||||
Access JWKS key function for `cfg.Auth.AccessJWTAud`.
|
||||
|
||||
### 5b. Token rejected with audience error
|
||||
|
||||
```
|
||||
token audience does not include "<aud>"
|
||||
```
|
||||
|
||||
The JWT's `aud` claim does not contain the configured `cfg.Auth.AccessJWTAud`
|
||||
(or the admin audience for admin routes). [GO-TESTED.] Confirm the Access
|
||||
application audience matches `cfg.Auth.AccessJWTAud`.
|
||||
|
||||
**Trust-model note for operators:** verification is **expiration-required +
|
||||
audience + signing-key (JWKS)**. There is **no `iss` (issuer) check** anywhere in
|
||||
the verifier. Trust rests entirely on the audience claim plus the JWKS signing
|
||||
key. When documenting or auditing the trust boundary, do not assume issuer is
|
||||
validated — it is not.
|
||||
The operator console is recognised by the request's host: `admin_hostname`
|
||||
(default `op.console.<root_domain>`). A bare IP counts only when the install
|
||||
names it — the address a `<ip>.nip.io` / `<ip>.sslip.io` root domain embeds
|
||||
(the local panel URL `felis setup` prints), or an `admin_hostname` set to that
|
||||
IP. Any other address, loopback included, is served as the player console, so a
|
||||
staff account signed in at `https://127.0.0.1:30443` through an SSH tunnel gets
|
||||
403 on admin routes. Open the console by its hostname instead (an `/etc/hosts`
|
||||
entry or `curl --resolve` pointing it at the tunnel), or set
|
||||
`[auth] admin_hostname` to the IP you use. [GO-TESTED]
|
||||
|
||||
### 5c. Local-password login fails or is silently rejected
|
||||
|
||||
@@ -344,7 +341,7 @@ unparseable → treated as disabled). Symptoms:
|
||||
|
||||
- Cookie present but login rejected with `local auth disabled` → the
|
||||
`local_auth_enabled` setting is false/absent. A present cookie under disabled
|
||||
local-auth is **rejected outright**, not fallen through to the JWT path.
|
||||
local-auth is **rejected outright**.
|
||||
- `invalid session: …` → bad/forged session hash.
|
||||
|
||||
Fix: set `local_auth_enabled=true` in `platform_settings` if local password auth
|
||||
@@ -355,8 +352,21 @@ is intended. [GO-TESTED for the session/QR-login logic.]
|
||||
## 6. Internal API rejects Velocity / proxy callers (service-token)
|
||||
|
||||
The internal face (`--internal-addr :8081`, routes under
|
||||
`/api/v1/internal/...`) is **never** Zero-Trust; it authenticates a single
|
||||
service token via `Authorization: Bearer <token>`, compared in constant time.
|
||||
`/api/v1/internal/...`) is **never** Zero-Trust; it authenticates a bearer token
|
||||
(`Authorization: Bearer <token>`, compared in constant time). Each internal caller
|
||||
has its own token, and each route serves only the callers listed for it
|
||||
(`x-felis-callers` in `docs/openapi.yaml`):
|
||||
|
||||
| Caller | Secret (key `token`) | API env | Where the caller reads it | Routes |
|
||||
|---|---|---|---|---|
|
||||
| `velocity` (proxy felis-link) | `felis/felis-service-token` | `FELIS_SERVICE_TOKEN` | `service-token` in the host's `felis-link.properties` | server list, wake/claim/ready/status, join events, menu, op-login, migrate, reclaim, link codes, blacklist |
|
||||
| `limbo` (login gate) | `felis/felis-limbo-token`, replica in `minecraft` | `FELIS_LIMBO_TOKEN` | login pod env `FELIS_SERVICE_TOKEN` | link codes, link status, blacklist |
|
||||
| `build` (build Job fetch) | `felis/felis-build-token`, replica in `felis-build` | `FELIS_BUILD_TOKEN` | fetch initContainer env | submission build context |
|
||||
| `ops` (`felis backup-now`) | `felis/felis-ops-token` | `FELIS_OPS_TOKEN` | read from the Secret on each run | break-glass backup |
|
||||
|
||||
The installer generates all four into `/etc/felis/secrets.env` (`SERVICE_TOKEN`,
|
||||
`LIMBO_TOKEN`, `BUILD_TOKEN`, `OPS_TOKEN`) and applies them on every run. The audit
|
||||
log records internal actions with the source `internal:<caller>`. [GO-TESTED]
|
||||
|
||||
In-cluster it is reached through the ClusterIP Service `felis-api-internal` (port
|
||||
8081), which is separate from the external NodePort `felis-api` (443) precisely so
|
||||
@@ -364,29 +374,57 @@ the no-Zero-Trust face is never exposed on a node. On the control-plane node the
|
||||
break-glass console reaches it by resolving that Service's ClusterIP and dialing
|
||||
`:8081`.
|
||||
|
||||
The face speaks plain HTTP, and the tokens cross it in the clear. That holds up because
|
||||
no hop leaves the node: the proxy runs on the node and dials the ClusterIP, and the
|
||||
login pod and the build Job are pods on the same node, so reading the traffic takes root
|
||||
there, which also reads the tokens from disk. Pods reach the face only where a policy
|
||||
opens it: `felis-login-to-internal-api` for the login pod and `felis-build-egress` for
|
||||
the build Job; `felis-server-egress` keeps every other game server (lobby included) off
|
||||
all private ranges, 8081 among them. A Velocity on another host would put the `velocity`
|
||||
token on the wire, which is one more reason the proxy belongs on the node (§1 of
|
||||
`docs/operations.md`); a multi-node shape would need TLS on this face first.
|
||||
|
||||
- **Internal calls fail to *connect* (not 401)** → the `felis-api-internal` Service
|
||||
is missing or its selector no longer matches the api pods. `kubectl -n felis get
|
||||
svc felis-api-internal` must show a ClusterIP with 8081; a bare `felis-api` name
|
||||
serves only 443 and every internal call would hang/refuse.
|
||||
|
||||
- **All internal calls 401** → the token is unset or wrong. The API reads env
|
||||
`FELIS_SERVICE_TOKEN`. If unset, startup logs:
|
||||
- **One caller's calls all 401** → its token is unset or differs from the api's
|
||||
copy. The api logs one line per unset token at startup:
|
||||
|
||||
```
|
||||
felis api: warning: FELIS_SERVICE_TOKEN unset — internal face will reject all callers
|
||||
felis api: warning: FELIS_LIMBO_TOKEN unset — the internal face turns the limbo caller away
|
||||
```
|
||||
|
||||
and wires an empty token, which rejects **everyone** (no bypass). [GO-TESTED
|
||||
for the constant-time compare / empty-token rejection.]
|
||||
An unset token never matches anything (no bypass). Compare the caller's value
|
||||
with its Secret, e.g. for the proxy:
|
||||
|
||||
In-cluster, the token's source of truth is the Secret `felis-service-token`
|
||||
(key `token`), injected as `FELIS_SERVICE_TOKEN` on the API Deployment. Fix:
|
||||
```
|
||||
kubectl -n felis get secret felis-service-token -o jsonpath='{.data.token}' | base64 -d
|
||||
```
|
||||
|
||||
```
|
||||
kubectl get secret felis-service-token -o jsonpath='{.data.token}' | base64 -d
|
||||
```
|
||||
- **`403 wrong_caller`** → the token is valid but belongs to a caller that route
|
||||
does not serve, e.g. the build token calling a proxy route. Configure the caller
|
||||
with its own token from the table above.
|
||||
|
||||
Ensure the proxy is configured with the identical value.
|
||||
- **api pods stuck in `CreateContainerConfigError`** → one of the four Secrets is
|
||||
missing in `felis`. Re-run the installer; it applies them before the bundle.
|
||||
|
||||
- **Two tokens with the same value** → `felis api` refuses to start with
|
||||
`the X and Y tokens are the same value; each caller needs its own`, since a shared
|
||||
value would make the caller ambiguous. Rotate one of them.
|
||||
|
||||
### Rotating a token
|
||||
|
||||
`sudo felis rotate-token <velocity|limbo|build|ops>` replaces one caller's token:
|
||||
it writes the new value to `secrets.env` (so a later installer run keeps it), the
|
||||
Secret and its replica, rolls felis-api so only the new value is accepted, then
|
||||
restarts the caller — the `felis-velocity` unit when the proxy runs on this host,
|
||||
or the login pod. Build Jobs and `felis backup-now` pick the new value up on their
|
||||
next run. The old value stops working as soon as felis-api has rolled; the caller
|
||||
is turned away for the few seconds until it restarts. For a proxy on another host,
|
||||
the command leaves the host alone and tells you to copy the new value from the
|
||||
Secret into that proxy's `felis-link.properties` and restart it. [GO-TESTED]
|
||||
|
||||
---
|
||||
|
||||
@@ -451,12 +489,14 @@ internal registry.
|
||||
### 8d. Build reaches `Failed` phase
|
||||
|
||||
`reconcileBuilds` polls the Job; a Job reaching `Failed` is surfaced via
|
||||
`writeBuildError` (JobPhase→Failed). [GO-TESTED for the mapping.] The underlying
|
||||
cause — a kaniko build error, the **Trivy CRITICAL-CVE gate** failing the build
|
||||
(spec §16), or the final push — is in the Job's pod logs and is
|
||||
[INTEGRATION-ONLY]. The pod runs `egress-gate` (§8f), `context-fetch`, `kaniko`
|
||||
(builds a tarball, never pushes) and `trivy` (scans that tarball) as init
|
||||
containers, then `push` — so a CVE-rejected image never reaches the registry. Inspect every step:
|
||||
`writeBuildError` (JobPhase→Failed). [GO-TESTED for the mapping.] The
|
||||
build's error names the cause: the step that failed with the last lines of its
|
||||
output, the deadline, or the scan verdict (spec §16). The pod runs `egress-gate`
|
||||
(§8f), `context-fetch`, `kaniko` (builds a tarball, never pushes), `trivy`
|
||||
(writes the full JSON report of that tarball), `sbom` (converts the report to a
|
||||
CycloneDX SBOM) and `scan-gate` (applies the scan policy) as init containers,
|
||||
then `push` — so an image the scan blocks never reaches the registry. Inspect
|
||||
every step:
|
||||
|
||||
```
|
||||
kubectl logs -n felis-build job/<build-job> --all-containers --prefix
|
||||
@@ -467,6 +507,54 @@ A `push` that fails with `403` means the target repository is under `felis/` or
|
||||
the `felis-registry-push` Secret in `felis-build` is missing or stale (re-run the
|
||||
installer).
|
||||
|
||||
**The scan gate.** `scan-gate` blocks the image when a vulnerability or a
|
||||
leaked secret has a severity listed in `[registry] scan_fail_on` (§8e; default
|
||||
`CRITICAL`). A vulnerability with no fixed release is listed without
|
||||
blocking unless `scan_fail_unfixed = true`, since nothing can be upgraded to
|
||||
clear it. A blocked build ends with an error such as:
|
||||
|
||||
```
|
||||
the scan blocked the image: 1 CRITICAL, 1 HIGH (CVE-2026-12345, CVE-2025-24813)
|
||||
```
|
||||
|
||||
`HIGH` is opt-in because the platform's own `felis/paper` image carries five
|
||||
fixable HIGH findings inside upstream `paper.jar` (its bundled commons-compress
|
||||
1.5 and plexus-utils 3.5.1). Adding `HIGH` to `scan_fail_on` blocks every build
|
||||
`FROM` it until those ids are accepted as known risks in `scan_accept`:
|
||||
|
||||
```toml
|
||||
scan_fail_on = ["CRITICAL", "HIGH"]
|
||||
scan_accept = ["CVE-2021-35515", "CVE-2021-35516", "CVE-2021-35517", "CVE-2021-36090", "CVE-2025-67030"]
|
||||
```
|
||||
|
||||
An accepted id (a CVE, GHSA or similar advisory id, or a secret rule id such as
|
||||
`aws-access-key-id`) never blocks; its findings are still counted, listed and
|
||||
marked **Accepted** on the panel, and scan-gate's log names the accepted ids.
|
||||
Review the list whenever the base image is upgraded.
|
||||
|
||||
felis-api keeps each finished build's scan: the verdict, up to 100 findings with
|
||||
the blocking ones first, the full Trivy report and the SBOM. On the panel,
|
||||
**Build Pipeline → Scan & logs** on a finished build shows them, with both files to download. The
|
||||
API serves the same data (admin only):
|
||||
|
||||
| Endpoint | Returns |
|
||||
|---|---|
|
||||
| `GET /api/v1/images/build/{id}/scan` | the verdict and findings; `404 scan_not_found` for a build that stopped before the scan or ran before builds kept scans |
|
||||
| `GET /api/v1/images/build/{id}/scan/report` | the Trivy JSON report as `<id>-trivy.json` |
|
||||
| `GET /api/v1/images/build/{id}/sbom` | the CycloneDX SBOM as `<id>.cdx.json` |
|
||||
|
||||
The report and SBOM travel to felis-api inside the scan-gate container's log, so
|
||||
one image gets at most 6 MiB of them compressed. Past that the gate drops the
|
||||
SBOM first, then the report, says so in its log, and the download answers
|
||||
`404 scan_document_not_kept`; the verdict and findings are always kept. A
|
||||
`scan-gate` exit 2 means the report was missing or unreadable; the build fails
|
||||
closed and the error says why.
|
||||
|
||||
A scan reflects the vulnerability DB on the day of the build. An admitted image
|
||||
is not scanned again when the DB learns of a new CVE; to rescan it, start a new
|
||||
build of the same context (`POST /api/v1/images/build`), after
|
||||
`felis mirror-build-tools -only trivy-db` if the DB copy is old (§8e).
|
||||
|
||||
### 8e. Build Pods never start: executor images and the scan DBs
|
||||
|
||||
A build Job runs Kaniko and Trivy, and Trivy reads two databases: the
|
||||
@@ -518,6 +606,9 @@ kaniko_image = "" # empty: the mirror/ copy above
|
||||
trivy_image = ""
|
||||
trivy_db_repository = ""
|
||||
trivy_java_db_repository = ""
|
||||
scan_fail_on = ["CRITICAL"] # §8d: severities that block; any case; empty = CRITICAL
|
||||
scan_fail_unfixed = false # §8d: true blocks on vulnerabilities with no fixed release too
|
||||
scan_accept = [] # §8d: vulnerability or secret rule ids accepted as known risks; never block
|
||||
build_cpu_limit = "2"
|
||||
build_mem_limit = "4Gi"
|
||||
build_disk_limit = "12Gi" # §8f
|
||||
@@ -525,6 +616,7 @@ build_user_namespaces = "auto" # §8f: auto | on | off
|
||||
build_runtime_class = "" # §8f: e.g. "gvisor"
|
||||
max_concurrent_builds = 2 # §8f: 1-6; later builds queue
|
||||
user_uploads_max_bytes = "4Gi" # every user's uploaded contexts together; 507 uploads_full past it
|
||||
context_max_bytes = "1Gi" # one uploaded context; empty = 1Gi (sent in 32 MiB parts, so the Cloudflare edge's 100 MB body cap does not bind)
|
||||
```
|
||||
|
||||
Put them in **both** `/etc/felis/felis.host.toml` (host-side CLI) and
|
||||
@@ -564,6 +656,7 @@ approved-but-hostile Dockerfile and the node is the pod around it:
|
||||
| User namespace | with `build_user_namespaces` on, root in the pod is an unprivileged uid on the node | below |
|
||||
| Sandbox runtime | optional `build_runtime_class` (gVisor, Kata) | below |
|
||||
| Credentials | the registry credential lives only in the `push` container; the service token only in `context-fetch` | jobspec |
|
||||
| Scan gate | the image's Trivy report is judged under `scan_fail_on` before `push` runs; a blocked or unreadable scan keeps the image out of the registry | `felis scan-gate`, §8d |
|
||||
| Resources | CPU, memory and ephemeral-storage limits per container; `activeDeadlineSeconds`; the context extraction stops at 4 GiB or 200 000 entries | jobspec, `felis fetch-context` |
|
||||
| Namespace backstop | `felis-build-limits` LimitRange gives any container without limits 1 CPU / 1 GiB / 1 GiB disk; `felis-build-quota` allows 8 running pods and no PVCs | bundle |
|
||||
| Concurrency | at most `[registry] max_concurrent_builds` (default 2, at most 6) builds run; later ones wait as `pending` (Queued) and start oldest first | `build.Builder` |
|
||||
@@ -666,6 +759,14 @@ control namespace (or `--registry-namespace`):
|
||||
(`FELIS_UPLOADS_STORAGE`, 5Gi) and the world-archive PVC
|
||||
(`FELIS_BACKUP_STORAGE`, 10Gi) work the same way; re-running the installer
|
||||
keeps an existing claim's size and warns when the variable asks for another.
|
||||
One uploaded context is capped by `context_max_bytes` (1Gi; the panel checks
|
||||
the file against it before uploading). The panel sends a context in parts of
|
||||
at most 32 MiB, staged under `.parts/` on the uploads volume, so the
|
||||
Cloudflare edge, which answers its own 413 page for request bodies over
|
||||
100 MB, never sees a body that large; a dropped connection resumes from the
|
||||
staged length, and a staged upload untouched for 24 hours is deleted. The
|
||||
staged bytes count toward the budgets below. A script can still POST a whole
|
||||
context in one body, which the edge caps at 100 MB.
|
||||
Uploaded build contexts are bounded by `user_uploads_max_bytes` (4Gi for all
|
||||
users together, §8e), 2 GiB per user, and 10% free space on the volume; past
|
||||
any of them an upload answers `507 uploads_full` or `403 submission_quota_exceeded`. A
|
||||
@@ -687,7 +788,12 @@ control namespace (or `--registry-namespace`):
|
||||
`kubectl -n felis exec deploy/registry -c registry-gc -- rm -f /var/lib/registry/.felis-last-gc`
|
||||
and restart the pod.
|
||||
- **The registry volume is lost:** re-run the installer; it pushes every
|
||||
platform image again. User images come back from their approved submissions:
|
||||
platform image again. With the off-site copy on (§16),
|
||||
`sudo felis offsite fetch-images` pushes the user images back at their old
|
||||
digests, so servers pinned to them pull again; it pushes only what the
|
||||
registry lacks, so a second run after an interruption is cheap. Without an
|
||||
off-site copy of the images, user images come back from their approved
|
||||
submissions (whose uploads the off-site copy also carries):
|
||||
the uploaded context of an approved submission stays on the uploads PVC
|
||||
(`GET /api/v1/submissions/{id}/context`, its `context_ref` and `image_ref`
|
||||
are in `GET /api/v1/submissions`), so an admin can build it again through
|
||||
@@ -747,7 +853,16 @@ Registry manifest rendering is [GO-TESTED]; actual serving is
|
||||
|
||||
## 10. World reaper: false-deletes and skipped backups (spec §18)
|
||||
|
||||
The reaper is a **run-once daily CronJob batch**, not an operator controller. It
|
||||
The reaper is a **run-once daily CronJob batch**, not an operator controller.
|
||||
Every install with an archive store gets it. Without `FELIS_WORLDS_HOST_PATH`
|
||||
it runs `felis reaper --retention-only`: it deletes backups past their expiry,
|
||||
reads archives back and sweeps the store (the second half of "A failed reaper
|
||||
Job" below), never looks at a server, and its summary shows `evaluated=0`. The
|
||||
installer says so (`idle-world retention is off`). Setting the worlds root on a
|
||||
re-run turns world reaping on. [GO-TESTED: `TestRunRetentionTouchesNoWorld`,
|
||||
`TestReaperCronJob_Gating`, `TestReaperCronJob_RetentionOnlyShape`.]
|
||||
|
||||
With a worlds root the reaper
|
||||
reaps a world only when `now - last_active_at > 15d` (`inactive_15d`); the 15-day
|
||||
deadline is **hard-fixed in code** (only `warn_before` / `retention` /
|
||||
`max_local_bytes` and the on-demand backup keys `manual_retention` /
|
||||
@@ -768,6 +883,18 @@ world growth) — do not size the archive PVC as if only world data were stored.
|
||||
The reap sequence (all [GO-TESTED] hermetically) preserves the world unless a
|
||||
**confirmed, DB-recorded backup exists**:
|
||||
|
||||
0. The world must be at rest before it is archived. A server still meant to
|
||||
run is told to stop (`desiredState: Stopped`) and left for the next run; one
|
||||
still stopping, whose game pod still exists, or whose world a restore,
|
||||
backup or file write holds is left too. Each of these counts in
|
||||
`awaiting_stop=` and does not fail the run. Once the server is down, the
|
||||
reaper takes the world's maintenance lock (§3b) and holds it through the
|
||||
archive and the volume delete: nothing can start the server or touch its
|
||||
world in between. A lock the reaper can no longer rewrite, or one someone
|
||||
else removed, ends the reap before `DeletePVC`. [GO-TESTED:
|
||||
`TestHoldWorldStopsARunningServer`, `TestHoldWorldWaitsUntilQuiet`,
|
||||
`TestReapLostHoldKeepsWorld`; live-drilled: a running server was told to
|
||||
stop on the first run and archived and deleted under `reap@…` on the next.]
|
||||
1. `ensureCapacity` (only if `max_local_bytes > 0`) frees room by evicting
|
||||
owners' on-demand backups first, oldest first, then reaper archives whose
|
||||
off-site copy is confirmed. The only copy of a reaped world is never
|
||||
@@ -783,8 +910,16 @@ The reap sequence (all [GO-TESTED] hermetically) preserves the world unless a
|
||||
off-site copy is confirmed`, counts it in `awaiting_offsite=` and leaves the
|
||||
PVC alone. The next daily run after the copy reuses the same archive and
|
||||
deletes. [GO-TESTED: `TestReapWaitsForOffsiteCopy`]
|
||||
5. **Only then** `DeletePVC` → `ReleaseWorld` → `Stop` (cosmetic) → audit →
|
||||
`felis_reaper_worlds_deleted_total++`.
|
||||
5. **Only then**, with the lock still held, `DeletePVC` → `ReleaseWorld` →
|
||||
audit → `felis_reaper_worlds_deleted_total++`, and the lock is dropped.
|
||||
|
||||
An archive the run reuses (step 4's second run, or a reap interrupted after
|
||||
its archive) is read back end to end and checked against the sha256 recorded
|
||||
when it was written before the world goes. One that does not match is marked
|
||||
corrupt, never offered for restore, and replaced by a fresh archive; one that
|
||||
cannot be read at all keeps the world until the next run. [GO-TESTED:
|
||||
`TestReapReadsBackReusedArchive`, `TestReapReplacesCorruptArchive`,
|
||||
`TestReapKeepsWorldWhenReadBackFails`.]
|
||||
|
||||
So a missing backup never results in a deleted world, and with a bucket
|
||||
configured neither does a backup that exists on this disk only. [GO-TESTED:
|
||||
@@ -799,15 +934,34 @@ failing: `sudo felis offsite status` (§16).
|
||||
Each run ends with one line:
|
||||
|
||||
```
|
||||
felis reaper: evaluated=12 reaped=1 awaiting_offsite=0 warned=2 skipped=0 store_full=0 evicted=0 expired=3 expire_failed=0
|
||||
felis reaper: evaluated=12 reaped=1 awaiting_offsite=0 awaiting_stop=0 warned=2 skipped=0 store_full=0 evicted=0 expired=3 expire_failed=0 verified=6 corrupt=0 verify_failed=0 swept=0 orphan_archives=0
|
||||
```
|
||||
|
||||
`skipped` counts servers the run failed on (steps 1–3 above, or the cluster or
|
||||
the database answering with an error; exempt servers and rows whose CRD is gone
|
||||
are not counted), `store_full` the subset kept because the backup store is full,
|
||||
and `expire_failed` expired backups it could not remove. Any of them above zero
|
||||
makes the process exit 1: the worlds are safe, but the Job fails so the watchdog
|
||||
mails `world reaper Job … failed` and `FelisWorldJobFailed` fires. The Job retries
|
||||
and `expire_failed` expired backups it could not remove. `awaiting_stop` counts
|
||||
idle servers left for the next run because they were not yet down (step 0); it
|
||||
does not fail the run, but a server that stays there for days is being started
|
||||
again by something, or a Job keeps holding its world.
|
||||
|
||||
After the servers, every run looks after the archive store itself:
|
||||
|
||||
- `verified` archives were read back and matched their recorded sha256; each
|
||||
run reads back up to ten archives not checked in the past week, oldest check
|
||||
first. `corrupt` ones did
|
||||
not match (or are gone from the volume) and are marked so: the backup page
|
||||
shows them as damaged and refuses to restore them. `verify_failed` ones could
|
||||
not be read at all and are retried the next run.
|
||||
- `swept` counts leftovers of an interrupted archive (`*.partial` files older
|
||||
than six hours) the run removed. `orphan_archives` counts finished archives
|
||||
no backup record points to; they are kept until they are older than
|
||||
`retention`, then removed, and each run lists the first few by path.
|
||||
|
||||
`skipped`, `expire_failed`, `corrupt`, `verify_failed` above zero, or a sweep
|
||||
that could not finish, make the process exit 1: the worlds are safe, but the
|
||||
Job fails so the watchdog mails `world reaper Job … failed` and
|
||||
`FelisWorldJobFailed` fires. The Job retries
|
||||
twice (`backoffLimit`), each retry re-running the whole batch, which is safe
|
||||
because every step is idempotent. Read the error above the summary:
|
||||
|
||||
@@ -888,8 +1042,15 @@ SMTP sink.]
|
||||
[GO-TESTED `TestReapUnownedServerStillReaped`.] Claim or exempt servers you
|
||||
want to keep.
|
||||
- `DeletePVC` is idempotent (missing PVC is not an error), so a re-run will not
|
||||
fail on already-reaped worlds; and `Stop` failure is only logged, so a reaped
|
||||
world's `MinecraftServer` may not be flipped to `Stopped`.
|
||||
fail on already-reaped worlds.
|
||||
- **An idle server whose world volume is already gone.** With a fresh reaper
|
||||
archive taken since the last activity, the run takes it as a reap that was
|
||||
interrupted after the delete and finishes it (release, audit, count). An
|
||||
unowned server with no such archive only has its idle clock restarted, so it
|
||||
is neither counted nor audited again every day. An owned one is released and
|
||||
audited once, since there is no world to archive. [GO-TESTED:
|
||||
`TestReapFinishesInterruptedReap`, `TestReapUnownedWithoutWorldRestartsClock`,
|
||||
`TestReapOwnedWithoutWorldReleases`.]
|
||||
|
||||
Only `TarLocal` (tar+gzip) archiving is implemented; VolumeSnapshot/Longhorn
|
||||
backends return `not implemented in this build`. The live PVC delete / Postgres
|
||||
@@ -1066,17 +1227,28 @@ leaves it off; only servers it actually filled get a line.
|
||||
|
||||
## 13. World PVC survives after I deleted the MinecraftServer
|
||||
|
||||
This is expected. The world PVC is a StatefulSet `VolumeClaimTemplate`. There is
|
||||
**no `persistentVolumeClaimRetentionPolicy` and no finalizer** anywhere in the
|
||||
operator. Deleting the `MinecraftServer` garbage-collects the StatefulSet, but
|
||||
StatefulSet deletion does **not** cascade to its template PVCs, and nothing else
|
||||
cleans them up. So the world PVC **always survives** server deletion. The
|
||||
**only** code that deletes a world PVC is the reaper, and only after a verified
|
||||
backup (§10). To reclaim a world PVC manually:
|
||||
This is expected. The world PVC is a StatefulSet `VolumeClaimTemplate`, and the
|
||||
operator sets the StatefulSet's `persistentVolumeClaimRetentionPolicy` to
|
||||
`Retain` on delete and on scale, explicitly rather than by the API default.
|
||||
There is no finalizer. Deleting the `MinecraftServer` garbage-collects the
|
||||
StatefulSet and keeps the claim, so a `MinecraftServer` that comes back under
|
||||
the same name mounts the same world. The **only** code that deletes a world PVC
|
||||
is the reaper, and only after a verified backup (§10).
|
||||
|
||||
A kept claim holds the name: creating a new server with it answers
|
||||
`409 world_volume_exists`, since the new server would otherwise mount the old
|
||||
world and hand it to its new owner. List the world claims whose server is gone:
|
||||
|
||||
```
|
||||
kubectl get pvc -l app.kubernetes.io/name=<name>
|
||||
kubectl delete pvc <pvc> # irreversible — the world is gone
|
||||
comm -23 \
|
||||
<(kubectl -n minecraft get pvc -l felis.lolicon.best/server -o jsonpath='{range .items[*]}{.metadata.labels.felis\.lolicon\.best/server}{"\n"}{end}' | sort) \
|
||||
<(kubectl -n minecraft get minecraftservers -o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{end}' | sort)
|
||||
```
|
||||
|
||||
To reclaim one (take a backup first if the world may still matter):
|
||||
|
||||
```
|
||||
kubectl -n minecraft delete pvc world-<name>-0 # irreversible — the world is gone
|
||||
```
|
||||
|
||||
`spec.storage.retainOnDelete` sat in the CRD and reached no controller. Spec
|
||||
@@ -1238,6 +1410,34 @@ All four mandated metrics have real producers; scrape them when triaging:
|
||||
- `felis_reaper_worlds_deleted_total` — increments only after a world PVC is
|
||||
actually deleted post-backup (§10); a spike here means worlds crossed the 15d
|
||||
idle line — cross-check that join events are flowing (§10 risk vectors).
|
||||
- `felis_http_requests_total{face,method,route,code}` and
|
||||
`felis_http_request_duration_seconds{face,route}` — every API request, by the
|
||||
route pattern it matched (`/api/v1/servers/{name}/status`, never the raw
|
||||
path; `unmatched` for a path no route serves). `face` is `internal` or
|
||||
`external`. Log streams count as requests but stay out of the latency
|
||||
histogram. A rise in `code=~"5.."` on one route narrows a failure to one
|
||||
handler; `route="unmatched"` rising is someone scanning.
|
||||
- The Go runtime and process series (`go_*`, `process_*`) of `felis-api`:
|
||||
goroutines, heap, open file descriptors. Goroutines that climb without
|
||||
falling back usually mean streams or uploads that never end.
|
||||
- `go_sql_*{db_name="felis"}` — the API's database pool. It is capped at 25
|
||||
connections, and every connection runs with `statement_timeout=15s` and
|
||||
`idle_in_transaction_session_timeout=60s` (a `[database] url` that sets
|
||||
either keeps its own). `go_sql_in_use_connections` sitting at
|
||||
`go_sql_max_open_connections` with `go_sql_wait_count_total` climbing means
|
||||
requests are queueing for a connection: look for a slow query or a lock
|
||||
(`SELECT pid, state, wait_event, query FROM pg_stat_activity`). A statement
|
||||
cut off by the limit logs `canceling statement due to statement timeout`.
|
||||
|
||||
The API also writes one access-log line per request to its log, in logfmt:
|
||||
`face`, `method`, `route`, `path`, `status`, `duration_ms`, `bytes`,
|
||||
`request_id` (the id in every error envelope) and `principal` (the user id,
|
||||
once signed in). Successful probes and scrapes are left out.
|
||||
|
||||
```bash
|
||||
kubectl -n felis logs deploy/felis-api | grep 'msg=request' | grep 'status=5'
|
||||
kubectl -n felis logs deploy/felis-api | grep 'request_id=<id from the error>'
|
||||
```
|
||||
|
||||
### Scraping
|
||||
|
||||
@@ -1250,7 +1450,8 @@ annotated Service endpoints picks them up as is.
|
||||
`felis_build_info{component="operator"}`, and controller-runtime's
|
||||
`controller_runtime_reconcile_*` / `workqueue_*` series.
|
||||
- `felis-api` internal face `:8081/metrics` (Service `felis-api-internal`) —
|
||||
`felis_build_info{component="api"}`,
|
||||
`felis_build_info{component="api"}`, the `felis_http_*` request series, the
|
||||
`go_*`/`process_*` runtime series, the `go_sql_*` pool series,
|
||||
`felis_image_build_failures_total`, and the sign-in series of §17
|
||||
(`felis_mail_total`, `felis_rate_limited_total`,
|
||||
`felis_auth_otp_lockouts_total`, `felis_auth_failures_total`,
|
||||
@@ -1320,8 +1521,8 @@ Two properties of the control plane matter when you do:
|
||||
| Download | Check |
|
||||
|---|---|
|
||||
| `felis-linux-<arch>` (release channel) | its sha256 must match the release's `SHA256SUMS`; a release without one, or a mismatch, is compiled from the same tag instead |
|
||||
| k3s (fresh install only) | the install script is read at `FELIS_K3S_VERSION`'s tag (default `v1.36.4+k3s1`), and it checks the binary against that release's sha256 list |
|
||||
| cloudflared (when absent) | release `FELIS_CLOUDFLARED_VERSION` (default `2026.9.1`) against a pinned sha256; another version needs `FELIS_CLOUDFLARED_SHA256` |
|
||||
| k3s (fresh install, or `FELIS_UPGRADE_DEPS=1`) | the install script is read at `FELIS_K3S_VERSION`'s tag (default `v1.36.4+k3s1`), and it checks the binary against that release's sha256 list |
|
||||
| cloudflared (when absent, or `FELIS_UPGRADE_DEPS=1`) | release `FELIS_CLOUDFLARED_VERSION` (default `2026.9.1`) against a pinned sha256; another version needs `FELIS_CLOUDFLARED_SHA256` |
|
||||
| Go toolchain (nano, source builds) | pinned sha256 per architecture; another version needs `FELIS_GO_SHA256` |
|
||||
| the registry image | pinned by digest (`registry:2.8.3@sha256:a3d8…`) |
|
||||
| Limbo, its spawn schematic, Paper, LuckPerms, Velocity | the builds and sha256s in `deploy/game-stack.lock`; each image build and the proxy install refuse a download that hashes differently (§15b) |
|
||||
@@ -1366,7 +1567,7 @@ runs changed:
|
||||
| Component | Restarted when |
|
||||
|---|---|
|
||||
| `felis-velocity` (the proxy) | its unit, the JRE, `velocity.jar`, `velocity.toml`, the forwarding secret, the felis-link settings or a plugin jar changed, or it was not running. The fingerprint lives in `/etc/felis/velocity.fingerprint`; delete it to force a restart. |
|
||||
| login and lobby pods | the rebuilt limbo or lobby image has a new image ID (`/etc/felis/system-server-images`). Each restarts on its own. The installer turns off BuildKit's default provenance attestation (`BUILDX_NO_DEFAULT_ATTESTATIONS=1`): it records the build time, which would give every rebuild a new ID. |
|
||||
| login and lobby pods | the rebuilt limbo or lobby image has a new image ID (`/etc/felis/system-server-images`). The installer then pins that server's `spec.image` to the digest its tag names now (`felis pin-images --system login\|lobby`) and the operator rolls the pod onto it, each on its own; with the registry unreachable it recreates the pod instead. The installer turns off BuildKit's default provenance attestation (`BUILDX_NO_DEFAULT_ATTESTATIONS=1`): it records the build time, which would give every rebuild a new ID. |
|
||||
| PostgreSQL | first install only (`listen_addresses` needs a restart). A rerun reloads the configuration, which keeps connections open. |
|
||||
| felis-api, felis-operator, the registry pod (its gate and GC containers run the felis binary) | the image tag changed (an upgrade), or a same-version rerun rebuilt it. |
|
||||
|
||||
@@ -1385,6 +1586,24 @@ exists, so the database is upgraded only when you run `pg_upgrade` yourself.
|
||||
applied; when they are the problem, restore the `pre-migrate` bundle the upgrade
|
||||
took (§16, "Roll back an upgrade that broke the database").
|
||||
|
||||
**Schema guard.** felis-api, the reaper, the off-site copy and `felis migrate up`
|
||||
compare the migrations the database records with the ones their build embeds, as
|
||||
sets. A database a newer release migrated stops them with `database schema is
|
||||
newer than this felis build: it records migration 0026, and this build knows
|
||||
migrations up to 0025`, so an undo across an upgrade that migrated shows up as
|
||||
felis-api in CrashLoopBackOff until the `pre-migrate` bundle is restored. A
|
||||
database still missing migrations stops felis-api, the reaper and the off-site
|
||||
copy with `database schema is behind this felis build` until `felis migrate up`
|
||||
runs; `felis setup`'s preflight applies them itself. [PG-TESTED]
|
||||
|
||||
**A failed rerun and the host binary.** The installer replaces
|
||||
`/usr/local/bin/felis` early (the steps after it run the new binary) and keeps
|
||||
the old one as `felis.prev` until the new one is in use. A run that fails before
|
||||
the database migrations start puts the old binary back, so the host timers and
|
||||
`felis setup` keep matching the database and the control plane that are still
|
||||
running; a rerun continues from there. Once migrations have started, the new
|
||||
binary stays. [VM-VERIFIED]
|
||||
|
||||
## 15b. Game images, pinned builds, and moving a world to a newer Minecraft
|
||||
|
||||
Each release pins the upstream builds it installs in `deploy/game-stack.lock`:
|
||||
@@ -1415,6 +1634,9 @@ created with pinned to the digest the tag named at that moment
|
||||
step pins any older server still on a bare tag *before* it pushes the new
|
||||
builds. Kubernetes pulls a digest-qualified ref by the digest, so a pinned
|
||||
server wakes on exactly the build it was created on, however often the tag moves.
|
||||
The login and lobby servers are pinned the same way, by the installer, each time
|
||||
it moves them onto a new build, so `kubectl -n minecraft get minecraftserver login
|
||||
-o jsonpath='{.spec.image}'` names the build the login gate runs.
|
||||
|
||||
Each installer run also pushes every game build under a tag no later run
|
||||
rewrites, `<Minecraft version>-<12 hex of the image id>`
|
||||
@@ -1438,6 +1660,7 @@ build no server and no whitelist entry names is pruned after 24 hours, and the
|
||||
| Create/edit refused with `image_not_in_registry` | the whitelisted tag was never pushed to the internal registry, or was deleted | push or rebuild the image, then retry |
|
||||
| Create/edit refused with `registry_unavailable` | felis-api could not reach `registry.felis.svc:5000` | `kubectl -n felis get pods -l app.kubernetes.io/component=registry`; check the `felis-registry-ingress` NetworkPolicy still admits felis-api |
|
||||
| Installer warns `could not pin every user server` | the registry was down, or a server names a tag the registry lost | fix the registry, then `sudo felis pin-images` before starting those servers; a server whose tag is gone keeps its bare tag until an admin picks a new image |
|
||||
| Installer warns `could not pin the login system server` (or lobby) | the registry did not answer right after the push | the installer recreated the pod instead, which starts the new build only while `spec.image` names the bare tag; rerun the installer once `kubectl -n felis get pods -l app.kubernetes.io/component=registry` is Ready |
|
||||
| A running server restarted during an installer re-run | it was pinned in place: the operator rolled it onto the pinned ref, the build it already ran | nothing; it happens once per server |
|
||||
| Create/edit refused with `the registry no longer holds build …` | the image names a digest the pruner deleted: nothing referenced it for 24 hours (§9) | pick a current tag; whitelist the versioned tag of a build you want kept |
|
||||
|
||||
@@ -1472,7 +1695,7 @@ along). One bundle is `felis-db-<UTC stamp>-<label>.tar`:
|
||||
|---|---|
|
||||
| `MANIFEST.json` | version, schema version, `pg_dump --version`, sha256 of every member |
|
||||
| `db.dump` | `pg_dump --format=custom` of the `felis` database |
|
||||
| `state/etc/felis/...` | `secrets.env` (DB password, session/forwarding secrets, registry tokens), `felis.host.toml`, `felis.pod.toml`, the `felis.toml` symlink, the panel TLS pair. `bootstrap.done` is left out on purpose |
|
||||
| `state/etc/felis/...` | every file in `/etc/felis`: `secrets.env` (DB password, session/forwarding secrets, registry tokens), `felis.host.toml`, `felis.pod.toml`, the `felis.toml` symlink, `offsite.env` (bucket credentials and encryption key), the panel TLS pair, and the installer's own markers (`system-server-images`, `velocity.fingerprint`). `bootstrap.done` is left out on purpose |
|
||||
| `k8s/minecraftservers.json` | every MinecraftServer, status and server-side metadata stripped, ready for `kubectl apply` (best effort: when the cluster did not answer, the manifest records why) |
|
||||
|
||||
next to a `.sha256` sidecar in `sha256sum` format. **A bundle contains the
|
||||
@@ -1570,6 +1793,52 @@ Do **not** run `felis migrate up` here: the host binary is already the new
|
||||
release and would re-apply the migrations you are rolling back. Re-run the
|
||||
older installer version to bring the host binary back in line.
|
||||
|
||||
### Whole-host disaster recovery: what comes back, and from where
|
||||
|
||||
When the host (or its disk) is gone, everything Felis needs comes back from the
|
||||
off-site bucket (next sections) plus a fresh install. What the host holds:
|
||||
|
||||
| Data | On the host | In the bucket | Brought back by | Lost at most |
|
||||
|---|---|---|---|---|
|
||||
| Control-plane database (accounts, passkeys, ownership, quotas, audit, submissions, the `world_backups` index) | PostgreSQL | every bundle, copied within the hour of being written | `fetch-db`, `db restore` | changes since the newest bundle: up to a day plus an hour with the daily timer |
|
||||
| Host state (`/etc/felis`: secrets, both `felis.toml` copies, `offsite.env`, panel TLS pair) | `/etc/felis` | inside every bundle | `tar -x` of the bundle's `state/` | as the database |
|
||||
| MinecraftServer objects | k3s | inside every bundle (`k8s/minecraftservers.json`) | `kubectl apply` | as the database |
|
||||
| World archives (reaper, "Back up now", pre-restore snapshots) | `felis-backups` volume | each one within the hour | `fetch-worlds` | archives written in the last hour |
|
||||
| Live worlds | `world-*` volumes under `/var/lib/rancher/k3s/storage` | **only as their archives** | a restore from the newest archive (§10) | everything since that world's newest archive |
|
||||
| User images | `registry` volume | hourly; image lists kept 14 days | `fetch-images` | images pushed in the last hour |
|
||||
| Submission uploads (modpacks awaiting or past review) | `felis-uploads` volume | hourly; upload lists kept 14 days | `fetch-uploads` | uploads of the last hour |
|
||||
| Platform images, Velocity, the JRE, build tools | registry, `/opt/felis` | not copied | the installer builds and pushes them again | nothing |
|
||||
| k3s itself (its token, CA, datastore, Secrets, Deployments) | `/var/lib/rancher/k3s` | not copied | the installer makes a new single-node cluster and renders every Secret and Deployment from `/etc/felis` | nothing: no Felis data lives only there |
|
||||
|
||||
**Live worlds are the gap.** A world's current state exists only in its
|
||||
volume; the bucket holds the archives the reaper, an owner's "Back up now" or
|
||||
a pre-restore snapshot wrote. A world that was never archived comes back as a
|
||||
server with an empty world. Tell owners to back up before anything they would
|
||||
hate to lose, and treat the newest archive's age (the server's backup page) as
|
||||
that world's recovery point.
|
||||
|
||||
For a smaller database loss window, give the installer a tighter
|
||||
`FELIS_DB_BACKUP_TIME` (any systemd calendar, e.g. `*-*-* 00/6:30:00` for
|
||||
every 6 hours) with a matching `FELIS_DB_BACKUP_KEEP`, on every installer
|
||||
run; the hourly off-site copy picks each bundle up within the hour.
|
||||
|
||||
How long a rebuild takes is mostly transfer time. The installer on a blank
|
||||
host builds and pushes every platform image, so it runs longer than an upgrade
|
||||
and depends on the host's network; the database restore takes seconds to
|
||||
minutes; the three fetches move what `sudo felis offsite status` reports the
|
||||
bucket holding (images, uploads, world archives) at the bucket's bandwidth,
|
||||
and each skips what is already in place, so an interrupted one resumes. Write
|
||||
those sizes down with the bucket's download rate and you have the recovery
|
||||
time for your install. A whole-host rehearsal on a spare machine, once per
|
||||
release, is the way to know it for sure.
|
||||
|
||||
The order below matters: the state goes in before the installer so it reuses
|
||||
the old secrets and bucket; the images go back before the database so the
|
||||
servers the database restores find the digests they pin, inside the pruner's
|
||||
24-hour grace; the MinecraftServers go back with the database that names their
|
||||
owners; the worlds come last because a restore needs a server to restore
|
||||
into.
|
||||
|
||||
### Rebuild on a new host (the old one is gone)
|
||||
|
||||
This needs the off-site copy (next section) or a bundle you copied off the old
|
||||
@@ -1602,7 +1871,32 @@ host yourself, plus the off-site encryption key if the copy is in the bucket.
|
||||
bundle, so it takes the fresh-install path, creates the empty database with
|
||||
the restored password and migrates it. It finds `[offsite]` in the restored
|
||||
`felis.host.toml` and turns the hourly copy back on.
|
||||
4. Restore the database and bring the servers back:
|
||||
4. Push the user images back into the new registry:
|
||||
|
||||
```
|
||||
sudo felis offsite fetch-images
|
||||
```
|
||||
|
||||
It restores the newest image list in the bucket (`-at <stamp>` for an
|
||||
older one; `felis offsite list` shows them) through the registry's loopback
|
||||
port as the platform principal, verifying every manifest and layer against
|
||||
its digest, and pushes only what the registry lacks. The pruner counts a
|
||||
restored image as freshly pushed and keeps it for 24 hours; finish the
|
||||
database step within that window so the restored servers and whitelist
|
||||
entries keep naming it. When the new host's hourly copy has already recorded
|
||||
its still-empty registry, `fetch-images` refuses that newest list and names
|
||||
the version to pass with `-at`; `fetch-uploads` does the same.
|
||||
5. Put the submission uploads back:
|
||||
|
||||
```
|
||||
sudo felis offsite fetch-uploads
|
||||
```
|
||||
|
||||
It writes every upload of the newest upload list (`-at <stamp>` for an
|
||||
older one) into the `felis-uploads` volume, owned by the control plane's
|
||||
uid, checking each against its sha256, and leaves one already in place
|
||||
alone.
|
||||
6. Restore the database and bring the servers back:
|
||||
|
||||
```
|
||||
kubectl -n felis scale deployment felis-api felis-operator --replicas=0
|
||||
@@ -1612,7 +1906,7 @@ host yourself, plus the off-site encryption key if the copy is in the bucket.
|
||||
tar -xOf felis-db-....tar k8s/minecraftservers.json | kubectl apply -f -
|
||||
```
|
||||
|
||||
5. Bring the world archives back into the archive volume:
|
||||
7. Bring the world archives back into the archive volume:
|
||||
|
||||
```
|
||||
sudo felis offsite fetch-worlds
|
||||
@@ -1622,16 +1916,31 @@ host yourself, plus the off-site encryption key if the copy is in the bucket.
|
||||
and the volume lacks, provisioning the `felis-backups` volume first if
|
||||
nothing has used it yet (a short-lived `felis-bind-felis-backups-*` pod). It
|
||||
lists any it could not find in the bucket. Restore a world from its archive
|
||||
as usual (§10, §13). Custom images built on the old host are rebuilt from
|
||||
their submissions (§8), or re-pushed.
|
||||
as usual (§10, §13).
|
||||
|
||||
Check the rebuild before letting players in:
|
||||
|
||||
```
|
||||
sudo felis db check # the database answers and has a fresh bundle
|
||||
kubectl get minecraftservers -A # every server the bundle held
|
||||
kubectl -n minecraft get pods # servers pull their pinned images (no ImagePullBackOff)
|
||||
sudo felis offsite status # the hourly copy runs from this host again
|
||||
```
|
||||
|
||||
Point the panel and game hostnames at the new host (DNS, or the tunnel in
|
||||
front of it). In the panel: sign in with an old account (accounts and passkeys
|
||||
come back with the database), open a restored server's backup page and confirm
|
||||
its archives are listed, restore the newest one, start the server and join
|
||||
it.
|
||||
|
||||
### Keep a copy somewhere else
|
||||
|
||||
A bundle on the same disk as the database protects against mistakes and bad
|
||||
upgrades, and a world archive on the same disk as the worlds protects against
|
||||
a deleted server. Neither survives losing the disk. The installer's off-site
|
||||
copy sends both to an S3-compatible bucket (AWS S3, Cloudflare R2, Backblaze
|
||||
B2, MinIO, ...), encrypted on this host:
|
||||
a deleted server. Neither survives losing the disk, and neither do the user
|
||||
images in the platform registry or the modpacks users uploaded for review. The
|
||||
installer's off-site copy sends all four to an S3-compatible bucket (AWS S3,
|
||||
Cloudflare R2, Backblaze B2, MinIO, ...), encrypted on this host:
|
||||
|
||||
```
|
||||
FELIS_OFFSITE_ENDPOINT=https://<account>.r2.cloudflarestorage.com \
|
||||
@@ -1660,9 +1969,36 @@ What runs:
|
||||
its retention (`expires_at`) has passed. An object already in the bucket at
|
||||
the right size is recorded without being sent again, so a run cut short
|
||||
resumes. [GO-TESTED: `internal/offsite`]
|
||||
- Objects are `worlds/<archive>.fenc` and `db/<bundle>.fenc`: AES-256-GCM in
|
||||
64 KiB segments, so truncation, reordering and a wrong key are all refused
|
||||
on the way back.
|
||||
- The same run copies the user images in the platform registry: every
|
||||
repository outside `felis/` and `mirror/`, each manifest the registry's index
|
||||
lists and every layer it names, read through the loopback hostPort. A layer
|
||||
shared by many images is stored once. The installer pushes `felis/` and
|
||||
`mirror/` again on a new host, but at new digests, so from those the run
|
||||
copies only the revisions a MinecraftServer or a whitelist entry pins by
|
||||
digest (a server created from the platform's Paper image, for one), without
|
||||
their tags; a restore puts them back by digest and leaves the installer's
|
||||
tags alone.
|
||||
When the set changed, a new version of the image list is written; versions
|
||||
replaced more than 14 days ago are dropped together with the layers only
|
||||
they named, so an image deleted by mistake stays restorable for two weeks
|
||||
(`fetch-images -at`). A manifest the registry lost mid-run keeps its earlier
|
||||
copy and is listed under `not whole:` in `status`. The copy covers the
|
||||
in-cluster registry that `[registry] url` names; `-registry host:port` points
|
||||
it elsewhere, `-registry off` skips images. [VM-TESTED: 16 images, 638 MiB,
|
||||
restored into an empty registry at the same digests]
|
||||
- It also copies the submission uploads (`sub-*/context.tar.gz` on the
|
||||
`felis-uploads` volume, §8), the source an admin rebuilds an approved image
|
||||
from. Identical uploads are stored once; the upload list is versioned and
|
||||
kept for 14 days like the image list (`fetch-uploads -at`). A context is read
|
||||
again only when its size or modification time changed. It runs when
|
||||
`[registry] user_uploads_context` is a local path (the installer's default);
|
||||
`-uploads-dir` names the directory by hand, `-uploads-pvc ""` skips it.
|
||||
[VM-TESTED: 17 uploads in 10 objects, 200 MiB, restored byte-identical]
|
||||
- Objects are `worlds/<archive>.fenc`, `db/<bundle>.fenc`,
|
||||
`registry/blobs/<sha256>.fenc`, `registry/manifests/<sha256>.fenc`,
|
||||
`registry/index/<stamp>.json.fenc`, `uploads/blobs/<sha256>.fenc` and
|
||||
`uploads/index/<stamp>.json.fenc`: AES-256-GCM in 64 KiB segments, so
|
||||
truncation, reordering and a wrong key are all refused on the way back.
|
||||
- The reaper deletes an idle world only after its archive is in the bucket
|
||||
(§10).
|
||||
- The watchdog mails the owners when no sync has completed for 12 hours
|
||||
@@ -1672,7 +2008,7 @@ Checking it:
|
||||
|
||||
```
|
||||
sudo felis offsite status # last run, errors, what the bucket holds, what waits
|
||||
sudo felis offsite list # the bundles in the bucket, newest first
|
||||
sudo felis offsite list # the bundles, image lists and upload lists in the bucket, newest first
|
||||
sudo journalctl -u felis-offsite -n 50 --no-pager
|
||||
sudo systemctl start felis-offsite.service # run one now
|
||||
```
|
||||
@@ -1690,7 +2026,9 @@ Without a bucket, copy the backup directory off the host on a schedule of your
|
||||
own (`rsync -a root@felis-host:/var/lib/felis/db-backups/ /backups/felis-db/`,
|
||||
with the `.sha256` sidecars; `sha256sum -c` on the far side proves the copy).
|
||||
That covers the database only; the world archives are under the
|
||||
`felis-backups` volume's directory in `/var/lib/rancher/k3s/storage/`.
|
||||
`felis-backups` volume's directory in `/var/lib/rancher/k3s/storage/`, the
|
||||
registry's images under the `registry` volume's (`*_felis_registry`) and the
|
||||
uploads under the `felis-uploads` volume's (`*_felis_felis-uploads`).
|
||||
|
||||
### `FELIS_PRE_MIGRATE_BACKUP=0`
|
||||
|
||||
@@ -1698,11 +2036,12 @@ Skips the pre-migration snapshot (`migrate up -no-backup`). The installer warns
|
||||
loudly when it is set. Use it only when the snapshot cannot work and you have
|
||||
another backup, e.g. an external database newer than the host's `pg_dump`.
|
||||
|
||||
## 17. Sign-in refused with 429, mail budget, account code locks, failed sign-ins
|
||||
## 17. Sign-in refused: 429 limits, no mail relay, account code locks, failed sign-ins
|
||||
|
||||
The public sign-in doors (`/api/v1/auth/*` except logout and the op-login
|
||||
status poll) have three limits of their own. Each answers 429 with a
|
||||
`Retry-After` header and a distinct error code.
|
||||
`Retry-After` header and a distinct error code. The doors that mail a code
|
||||
also answer 503 `mail_unavailable` on an install with no mail relay.
|
||||
|
||||
### `rate_limited`: one address called the doors too often
|
||||
|
||||
@@ -1742,6 +2081,33 @@ raise `max_per_hour` to what your relay allows.
|
||||
(`felis_mail_total{result="failed"}`, 502 `mail_undeliverable` to the caller).
|
||||
The relay's reason is in the `felis-api` log.
|
||||
|
||||
### `mail_unavailable`: no mail relay
|
||||
|
||||
With no `[smtp]` section every door that mails a code answers 503
|
||||
`mail_unavailable` before minting one: email sign-in, op.console sign-in,
|
||||
email verification, and the email step-up for sensitive changes and
|
||||
migration. The public doors answer before looking up the address, so every
|
||||
address gets the same reply. Sign-in is by passkey only, and a verified email
|
||||
stops counting as a way into the account (it is no longer offered as a
|
||||
re-verification factor). Codes are never logged: `felis api` says at start
|
||||
`[smtp] not configured`. Run `felis setup` and configure email to open the
|
||||
doors.
|
||||
|
||||
### Relay refused for lacking TLS
|
||||
|
||||
A relay on port 465 is spoken to over TLS from the first byte. On any other
|
||||
port Felis upgrades with STARTTLS, and when the relay does not offer it the
|
||||
send fails with `smtp: <host>:<port> does not offer STARTTLS` (502
|
||||
`mail_undeliverable` to the caller, the full text in the `felis-api` log, and
|
||||
the same error on the `felis setup` email screen). Without TLS anyone on the
|
||||
path reads the codes, and anyone who can rewrite the conversation can strip
|
||||
the STARTTLS offer, so this is the default for every relay except one on this
|
||||
host (`localhost`, `127.0.0.0/8`, `::1`). Use port 465 or a relay that offers
|
||||
STARTTLS. For a relay you reach over a link you trust, set
|
||||
`require_tls = false` under `[smtp]` in `/etc/felis/felis.toml` and
|
||||
`/etc/felis/felis.pod.toml`; installer re-runs and the setup email screen keep
|
||||
it. `felis api` warns at start whenever codes may go out without TLS.
|
||||
|
||||
### `otp_account_locked`: ten wrong codes in 24 hours
|
||||
|
||||
Ten wrong email codes for one account within 24 hours, counted across every
|
||||
@@ -1782,6 +2148,33 @@ A failed audit write does not fail the action; it logs `audit: lost ...` in
|
||||
`felis-api` and counts in `felis_audit_write_failures_total`
|
||||
(`FelisAuditWriteFailing`). The cause is almost always PostgreSQL (§16).
|
||||
|
||||
### How long rows are kept
|
||||
|
||||
`felis-api` prunes the database a minute after it starts and every 6 hours
|
||||
after that, and logs `retention: pruned spent rows` with a count per table:
|
||||
|
||||
| Rows | Deleted |
|
||||
|---|---|
|
||||
| sessions | 30 days after they expired or were signed out |
|
||||
| email codes, passkey challenges, setup links, op-login requests | 30 days after they expired or were used |
|
||||
| `/felis link` bind codes | 30 days after they expired |
|
||||
| account migrations that never completed | 30 days after their last step (completed ones stay) |
|
||||
| wrong-code windows (`otp_failure_windows`) | 30 days after they began |
|
||||
| `audit_logs` | once older than `[audit] retention` |
|
||||
|
||||
`[audit] retention` defaults to `365d`; it takes days (`90d`), months of 30
|
||||
days (`18mo`) or `forever`, and refuses anything under `30d`. The daily
|
||||
`felis db backup` bundles (§16) still hold the rows for as long as the bundles
|
||||
are kept. To keep audit rows past the retention, export them before they go:
|
||||
|
||||
```sh
|
||||
sudo felis db audit-export -until 2026-01-01 -out /root/audit-2025.jsonl
|
||||
```
|
||||
|
||||
`-since` and `-until` take a day (UTC midnight) or an RFC 3339 instant; the
|
||||
window is `[since, until)`. Each line is one row as JSON, oldest first. The
|
||||
file is created `0600` and an existing file is never overwritten.
|
||||
|
||||
### Optional: a Cloudflare rate limiting rule in front
|
||||
|
||||
The limits above live in the API, so they hold on any edge. Behind Cloudflare
|
||||
@@ -1811,6 +2204,7 @@ for 10 seconds (the Free plan's limits).
|
||||
| Build push 400 / SA denied / egress hang / Failed / executor ImagePullBackOff | §8, §8e |
|
||||
| Registry push/pull unreachable | §9 |
|
||||
| World deleted unexpectedly / backup skipped | §10 |
|
||||
| Reaper `awaiting_stop` stays above 0; `corrupt=` / backup shown as damaged; `orphan_archives` | §10 |
|
||||
| Idle auto-stop not firing; player count 0; `PlayersCounted=False` | §11 |
|
||||
| A config field seems ignored | §12 |
|
||||
| PVC left behind after delete | §13 |
|
||||
@@ -1830,3 +2224,4 @@ for 10 seconds (the Free plan's limits).
|
||||
| Right code refused; `otp_account_locked` / `FelisOTPAccountLocked` | §17 |
|
||||
| `FelisSignInFailures` / who is guessing, from where | §17 |
|
||||
| `FelisAuditWriteFailing` | §17 |
|
||||
| How long sessions, codes and audit rows are kept; export audit rows | §17 |
|
||||
@@ -11,7 +11,6 @@ require (
|
||||
github.com/descope/virtualwebauthn v1.0.5
|
||||
github.com/go-logr/logr v1.4.2
|
||||
github.com/go-webauthn/webauthn v0.17.4
|
||||
github.com/golang-jwt/jwt/v5 v5.3.1
|
||||
github.com/google/uuid v1.6.0
|
||||
github.com/jackc/pgx/v5 v5.9.2
|
||||
github.com/minio/minio-go/v7 v7.2.1
|
||||
@@ -20,11 +19,13 @@ require (
|
||||
k8s.io/api v0.31.3
|
||||
k8s.io/apimachinery v0.31.3
|
||||
k8s.io/client-go v0.31.0
|
||||
k8s.io/kube-openapi v0.0.0-20240228011516-70dd3763d340
|
||||
sigs.k8s.io/controller-runtime v0.19.3
|
||||
sigs.k8s.io/yaml v1.4.0
|
||||
)
|
||||
|
||||
require (
|
||||
github.com/asaskevich/govalidator v0.0.0-20190424111038-f61b66f89f4a // indirect
|
||||
github.com/atotto/clipboard v0.1.4 // indirect
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect
|
||||
github.com/beorn7/perks v1.0.1 // indirect
|
||||
@@ -49,6 +50,7 @@ require (
|
||||
github.com/go-viper/mapstructure/v2 v2.5.0 // indirect
|
||||
github.com/go-webauthn/x v0.2.6 // indirect
|
||||
github.com/gogo/protobuf v1.3.2 // indirect
|
||||
github.com/golang-jwt/jwt/v5 v5.3.1 // indirect
|
||||
github.com/golang/groupcache v0.0.0-20210331224755-41bb18bfe9da // indirect
|
||||
github.com/golang/protobuf v1.5.4 // indirect
|
||||
github.com/google/gnostic-models v0.6.8 // indirect
|
||||
@@ -108,7 +110,6 @@ require (
|
||||
gopkg.in/yaml.v3 v3.0.1 // indirect
|
||||
k8s.io/apiextensions-apiserver v0.31.0 // indirect
|
||||
k8s.io/klog/v2 v2.130.1 // indirect
|
||||
k8s.io/kube-openapi v0.0.0-20240228011516-70dd3763d340 // indirect
|
||||
k8s.io/utils v0.0.0-20240711033017-18e509b52bc8 // indirect
|
||||
sigs.k8s.io/json v0.0.0-20221116044647-bc3834ca7abd // indirect
|
||||
sigs.k8s.io/structured-merge-diff/v4 v4.4.1 // indirect
|
||||
|
||||
@@ -2,6 +2,8 @@ github.com/BurntSushi/toml v1.6.0 h1:dRaEfpa2VI55EwlIW72hMRHdWouJeRF7TPYhI+AUQjk
|
||||
github.com/BurntSushi/toml v1.6.0/go.mod h1:ukJfTF/6rtPPRCnwkur4qwRxa8vTRFBF0uk2lLoLwho=
|
||||
github.com/MakeNowJust/heredoc v1.0.0 h1:cXCdzVdstXyiTqTvfqk9SDHpKNjxuom+DOlyEeQ4pzQ=
|
||||
github.com/MakeNowJust/heredoc v1.0.0/go.mod h1:mG5amYoWBHf8vpLOuehzbGGw0EHxpZZ6lCpQ4fNJ8LE=
|
||||
github.com/asaskevich/govalidator v0.0.0-20190424111038-f61b66f89f4a h1:idn718Q4B6AGu/h5Sxe66HYVdqdGu2l9Iebqhi/AEoA=
|
||||
github.com/asaskevich/govalidator v0.0.0-20190424111038-f61b66f89f4a/go.mod h1:lB+ZfQJz7igIIfQNfa7Ml4HSf2uFQQRzpGGRXenZAgY=
|
||||
github.com/atotto/clipboard v0.1.4 h1:EH0zSVneZPSuFR11BlR9YppQTVDbh5+16AmcJi4g1z4=
|
||||
github.com/atotto/clipboard v0.1.4/go.mod h1:ZY9tmq7sm5xIbd9bOK4onWV4S6X0u6GY7Vn0Yu86PYI=
|
||||
github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiErDT4WkJ2k=
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"log"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/metrics"
|
||||
)
|
||||
|
||||
// Account change notices tell the owner of an account, at the verified address,
|
||||
// that a way into it was just added, removed or moved: a passkey registered or
|
||||
// removed, the email replaced (that notice goes to the OLD address, which is the
|
||||
// one the owner still reads if someone else made the change). They carry the time
|
||||
// and the source address and say what to do if the change was not theirs.
|
||||
// Best effort, like the lock notice: the change already happened.
|
||||
|
||||
// notifyAccountChange mails one notice to the given address.
|
||||
func (a *API) notifyAccountChange(r *http.Request, to, subject, body string) {
|
||||
if to == "" {
|
||||
return
|
||||
}
|
||||
sender, ok := a.Mailer.(noticeSender)
|
||||
if !ok {
|
||||
log.Printf("auth: no notice mailer; account change notice %q was not sent (request_id=%s)",
|
||||
subject, requestIDFromContext(r.Context()))
|
||||
return
|
||||
}
|
||||
if ok, _ := a.mailGate().take(mailGateKey); !ok {
|
||||
metrics.MailTotal.WithLabelValues("notice", "throttled").Inc()
|
||||
log.Printf("auth: mail budget spent; account change notice %q was not sent (request_id=%s)",
|
||||
subject, requestIDFromContext(r.Context()))
|
||||
return
|
||||
}
|
||||
if err := sender.SendNotice(r.Context(), to, subject, body); err != nil {
|
||||
metrics.MailTotal.WithLabelValues("notice", "failed").Inc()
|
||||
log.Printf("auth: account change notice failed (request_id=%s): %v", requestIDFromContext(r.Context()), err)
|
||||
return
|
||||
}
|
||||
metrics.MailTotal.WithLabelValues("notice", "sent").Inc()
|
||||
}
|
||||
|
||||
// verifiedEmail is where a notice about p's account goes: the address it proved,
|
||||
// or nothing.
|
||||
func verifiedEmail(p *Principal) string {
|
||||
if !p.EmailVerified {
|
||||
return ""
|
||||
}
|
||||
return p.Email
|
||||
}
|
||||
|
||||
func (a *API) notifyPasskeyAdded(r *http.Request, p *Principal) {
|
||||
subject, body := accountChangeNotice(
|
||||
"已添加 Passkey", "passkey added",
|
||||
"你的 Felis 账户刚刚添加了一个 Passkey。", "A passkey was just added to your Felis account.",
|
||||
"删除这个 Passkey", "remove that passkey",
|
||||
a.now(), a.noticeIP(r))
|
||||
a.notifyAccountChange(r, verifiedEmail(p), subject, body)
|
||||
}
|
||||
|
||||
func (a *API) notifyPasskeyRemoved(r *http.Request, p *Principal) {
|
||||
subject, body := accountChangeNotice(
|
||||
"已删除 Passkey", "passkey removed",
|
||||
"你的 Felis 账户刚刚删除了一个 Passkey,其它设备上的登录已全部退出。",
|
||||
"A passkey was just removed from your Felis account, and every other device was signed out.",
|
||||
"检查剩下的 Passkey", "check the passkeys that remain",
|
||||
a.now(), a.noticeIP(r))
|
||||
a.notifyAccountChange(r, verifiedEmail(p), subject, body)
|
||||
}
|
||||
|
||||
// notifyEmailChanged tells the previous verified address where the account's
|
||||
// mail now goes, masked so the notice does not hand the new address to whoever
|
||||
// reads the old mailbox.
|
||||
func (a *API) notifyEmailChanged(r *http.Request, oldEmail, newEmail string) {
|
||||
masked := maskEmail(newEmail)
|
||||
subject, body := accountChangeNotice(
|
||||
"邮箱已更换", "email changed",
|
||||
"你的 Felis 账户的邮箱刚刚更换为 "+masked+",这个地址以后不会再收到登录验证码。",
|
||||
"The email on your Felis account was just changed to "+masked+". This address will no longer receive sign-in codes.",
|
||||
"把邮箱改回来", "change the email back",
|
||||
a.now(), a.noticeIP(r))
|
||||
a.notifyAccountChange(r, oldEmail, subject, body)
|
||||
}
|
||||
|
||||
func (a *API) noticeIP(r *http.Request) string {
|
||||
if ip := a.clientIP(r); ip.IsValid() {
|
||||
return ip.String()
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// accountChangeNotice renders a bilingual notice. zhUndo/enUndo name the step
|
||||
// that reverses the change, for the "if this wasn't you" line.
|
||||
func accountChangeNotice(zhTitle, enTitle, zhWhat, enWhat, zhUndo, enUndo string, at time.Time, ip string) (subject, body string) {
|
||||
when := at.UTC().Format("2006-01-02 15:04 MST")
|
||||
zhIP, enIP := ip, ip
|
||||
if ip == "" {
|
||||
zhIP, enIP = "未知", "unknown"
|
||||
}
|
||||
subject = "Felis " + zhTitle + " · " + enTitle
|
||||
body = fmt.Sprintf(`%s
|
||||
时间:%s
|
||||
来源 IP:%s
|
||||
如果不是你本人操作,请立即登录 Felis,在账户页%s并退出其它设备,然后联系服务器管理员。
|
||||
|
||||
%s
|
||||
Time: %s
|
||||
From IP: %s
|
||||
If this wasn't you, sign in to Felis now, %s and sign out other devices on the Account page, then contact the server operator.
|
||||
`, zhWhat, when, zhIP, zhUndo, enWhat, when, enIP, enUndo)
|
||||
return subject, body
|
||||
}
|
||||
|
||||
// maskEmail keeps the first character of the local part and the domain:
|
||||
// [email protected] → a***@example.com.
|
||||
func maskEmail(email string) string {
|
||||
at := strings.LastIndexByte(email, '@')
|
||||
if at <= 0 {
|
||||
return "***"
|
||||
}
|
||||
first := []rune(email[:at])[0]
|
||||
return string(first) + "***" + email[at:]
|
||||
}
|
||||
+194
-45
@@ -1,9 +1,10 @@
|
||||
// Package api implements felis-api: one binary serving two faces (spec §7).
|
||||
//
|
||||
// The internal face (velocity / backend callbacks) authenticates with a static
|
||||
// service token and is never wrapped in Zero Trust. The external face (people /
|
||||
// panel) authenticates with a Cloudflare Access JWT; admin-tier operations
|
||||
// additionally require the admin Access path (spec §14, graded by operation).
|
||||
// per-caller service token and is never wrapped in Zero Trust. The external face
|
||||
// (people / panel) authenticates the local session cookie; admin-tier operations
|
||||
// additionally require a staff session on the operator console host (spec §14,
|
||||
// graded by operation).
|
||||
//
|
||||
// Handlers depend on the Repo and Cluster interfaces, so the request routing,
|
||||
// dual-face auth, input validation and authorization are all unit-tested with
|
||||
@@ -14,7 +15,11 @@ package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -73,6 +78,12 @@ type API struct {
|
||||
// endpoints only answer 202). Optional: nil → that route reports 503.
|
||||
JobStatus JobStatusReader
|
||||
|
||||
// RestoreChains finds and settles the safety snapshots that run in front of a
|
||||
// restore (restorechain.go). A restore takes a snapshot first only when this
|
||||
// is set, the Backuper can chain a restore and the Restorer is wired, since
|
||||
// something has to start the restore once the snapshot is done.
|
||||
RestoreChains RestoreChains
|
||||
|
||||
// Files is the server file editor (list / read / write a file in a stopped
|
||||
// server's world volume — the "one wrong line in server.properties" repair).
|
||||
// Like Restorer and Backuper it is optional: when nil the file routes report
|
||||
@@ -90,10 +101,10 @@ type API struct {
|
||||
Submissions SubmissionService
|
||||
|
||||
// Mailer delivers player email one-time codes (spec §B2 onboarding). It is
|
||||
// optional: when nil the email-OTP start route mints and persists the code but
|
||||
// logs it server-side instead of mailing it (a KNOWN-LIMITATION — the demo has no
|
||||
// SMTP), so the verify flow is still exercised end-to-end. Production wires a real
|
||||
// sender. The code is never returned to the client on either path.
|
||||
// optional: when nil (no [smtp] relay) every door that mails a code answers 503
|
||||
// mail_unavailable before minting one, auth options stops offering email_otp,
|
||||
// and a verified email stops counting as a reauth factor. The code is never
|
||||
// returned to the client or logged.
|
||||
Mailer OTPMailer
|
||||
|
||||
// Passkey verifies WebAuthn credential-creation ceremonies (spec §14 / Phase 6
|
||||
@@ -181,6 +192,10 @@ type API struct {
|
||||
// X-Forwarded-For behind an operator proxy). Empty means the TCP peer.
|
||||
ClientIPHeader string
|
||||
|
||||
// AccessLog receives one line per API request (observe.go). Nil logs logfmt
|
||||
// to stderr.
|
||||
AccessLog *slog.Logger
|
||||
|
||||
// Now is the clock, injectable for tests. Defaults to time.Now.
|
||||
Now func() time.Time
|
||||
|
||||
@@ -200,6 +215,26 @@ type API struct {
|
||||
authDoorBuckets *bucketSet
|
||||
mailOnce sync.Once
|
||||
mailBuckets *bucketSet
|
||||
|
||||
drainInit sync.Once
|
||||
drainClose sync.Once
|
||||
drain chan struct{}
|
||||
}
|
||||
|
||||
// streamsClosing is closed once CloseStreams runs.
|
||||
func (a *API) streamsClosing() <-chan struct{} {
|
||||
a.drainInit.Do(func() { a.drain = make(chan struct{}) })
|
||||
return a.drain
|
||||
}
|
||||
|
||||
// CloseStreams ends every log stream this API is relaying, now and from now on.
|
||||
// http.Server.Shutdown waits for handlers to return and cancels nothing, so a
|
||||
// console left open would hold the process until the pod's grace period ran out;
|
||||
// register this with RegisterOnShutdown. The EventSource on the other end
|
||||
// reconnects, and resumes from its Last-Event-ID on the next instance.
|
||||
func (a *API) CloseStreams() {
|
||||
a.streamsClosing()
|
||||
a.drainClose.Do(func() { close(a.drain) })
|
||||
}
|
||||
|
||||
// panelURL returns the public player-console origin ("https://console.<root>"),
|
||||
@@ -269,7 +304,7 @@ func (a *API) streamGate() *streamLimiter {
|
||||
}
|
||||
|
||||
// streamKey identifies the principal a stream slot is charged to. It prefers the
|
||||
// stable user id and falls back to the email so a JWT principal without a user id is
|
||||
// stable user id and falls back to the email so a principal without a user id is
|
||||
// still bucketed by identity; an empty key (no authenticated identity, which the
|
||||
// external face's auth guard already precludes) shares one bucket, which is safe
|
||||
// because it is more restrictive, never less.
|
||||
@@ -319,6 +354,10 @@ type apiRoute struct {
|
||||
// since the browser calls it every few seconds while it waits.
|
||||
AuthDoor bool
|
||||
|
||||
// Callers lists the machines an internal-face route serves; every
|
||||
// authenticated internal route names at least one, and external routes none.
|
||||
Callers []Caller
|
||||
|
||||
h http.HandlerFunc
|
||||
}
|
||||
|
||||
@@ -326,6 +365,14 @@ type apiRoute struct {
|
||||
// service-token auth, never Zero Trust. It carries both health probes and the
|
||||
// metrics scrape.
|
||||
func (a *API) internalAPIRoutes() []apiRoute {
|
||||
// Who may call what (Caller). The proxy drives the game-facing routes; the
|
||||
// login gate only checks a joining player's bar and link and mints their bind
|
||||
// code; the build Job only reads the context of the submission it builds; the
|
||||
// on-node console only asks for a break-glass backup.
|
||||
proxy := []Caller{CallerVelocity}
|
||||
gate := []Caller{CallerVelocity, CallerLimbo}
|
||||
build := []Caller{CallerBuild}
|
||||
ops := []Caller{CallerOps}
|
||||
return []apiRoute{
|
||||
{Method: "GET", Pattern: "/healthz", Public: true, h: a.handleHealthz},
|
||||
{Method: "GET", Pattern: "/readyz", Public: true, h: a.handleReadyz},
|
||||
@@ -333,47 +380,47 @@ func (a *API) internalAPIRoutes() []apiRoute {
|
||||
// no token, internal-only so it is never exposed off-cluster.
|
||||
{Method: "GET", Pattern: "/metrics", Public: true, h: a.handleMetrics},
|
||||
|
||||
{Method: "GET", Pattern: "/api/v1/servers", h: a.handleListServers},
|
||||
{Method: "GET", Pattern: "/api/v1/servers", Callers: proxy, h: a.handleListServers},
|
||||
// The build Pod's context-fetch initContainer streams a submission's stored
|
||||
// modpack through this route (build namespace cannot mount the uploads PVC).
|
||||
{Method: "GET", Pattern: "/api/v1/internal/submissions/{id}/context", h: a.handleInternalSubmissionContext},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/ready", h: a.handleReady},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/join-event", h: a.handleJoinEvent},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/submissions/{id}/context", Callers: build, h: a.handleInternalSubmissionContext},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/ready", Callers: proxy, h: a.handleReady},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/join-event", Callers: proxy, h: a.handleJoinEvent},
|
||||
// Domain-autostart (spec §9.1, §14): velocity drives the wake lever and polls
|
||||
// status with its service token, identifying the joining player by online-mode
|
||||
// UUID. These live on the internal face because velocity holds no web Principal;
|
||||
// the external face keeps its own Principal-gated wake/status for the panel.
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/wake", h: a.handleInternalWake},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/servers/{name}/status", h: a.handleStatus},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/wake", Callers: proxy, h: a.handleInternalWake},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/servers/{name}/status", Callers: proxy, h: a.handleStatus},
|
||||
// Lobby `/menu` (spec §12): the felis-paper lobby is a pure UI face holding no
|
||||
// token, so velocity drives these on its behalf — claim by online-mode UUID
|
||||
// (the lobby's `Claim & Start`, separate from the autostartPolicy-gated wake)
|
||||
// and the menu projection that adds the ownership-derived `claimable` the §11
|
||||
// list/status views never carry.
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/claim", h: a.handleInternalClaim},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/servers/{name}/menu", h: a.handleInternalMenuStatus},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/claim", Callers: proxy, h: a.handleInternalClaim},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/servers/{name}/menu", Callers: proxy, h: a.handleInternalMenuStatus},
|
||||
// Account linking (spec §10): the in-game /link side mints a one-time code for a
|
||||
// verified UUID. Internal-only — the code is born from an online-mode UUID the
|
||||
// web never holds (account_link_codes has no user_id column).
|
||||
{Method: "POST", Pattern: "/api/v1/internal/account/link/code", h: a.handleCreateLinkCode},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/account/link/code", Callers: gate, h: a.handleCreateLinkCode},
|
||||
// QR scan-to-login completion poll (spec §B3 player game-login). After the player
|
||||
// scans the QR-encoded code and the web verify writes the durable link, velocity
|
||||
// polls this for the UUID it minted against and admits on {linked:true}. Read-only
|
||||
// and keyed by the verified UUID (not the scanned code), so it consumes nothing
|
||||
// and is safe to poll repeatedly.
|
||||
{Method: "GET", Pattern: "/api/v1/internal/account/link/status/{mc_uuid}", h: a.handleLinkStatus},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/account/link/status/{mc_uuid}", Callers: gate, h: a.handleLinkStatus},
|
||||
// Account migration (spec §B3 inherit), in-game side: /felis migrate puts the
|
||||
// account linked to the running player's verified UUID into migrate mode. Internal
|
||||
// only — the initiator is proven by online-mode auth, and the sensitive proof
|
||||
// (step-up) still happens web-side before anything transfers.
|
||||
{Method: "POST", Pattern: "/api/v1/internal/account/migrate/start", h: a.handleMigrateStart},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/account/migrate/start", Callers: proxy, h: a.handleMigrateStart},
|
||||
// Username-collision reclaim (spec §B3): velocity records a Mojang-priority
|
||||
// reclaim (bar the squatter UUID + stash its data for 30 days) and gates the
|
||||
// limbo login by checking whether a connecting UUID was barred. Internal-only —
|
||||
// velocity holds a service token, and the bar is keyed by UUID so the genuine
|
||||
// Mojang player (same name, different UUID) always passes.
|
||||
{Method: "POST", Pattern: "/api/v1/internal/player/reclaim", h: a.handleReclaimUsername},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/player/blacklist/{mc_uuid}", h: a.handleCheckBlacklist},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/player/reclaim", Callers: proxy, h: a.handleReclaimUsername},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/player/blacklist/{mc_uuid}", Callers: gate, h: a.handleCheckBlacklist},
|
||||
// Felis-nano multi-source session verifier, behind player game-login. Velocity is
|
||||
// pointed here with -Dmojang.sessionserver and issues the request itself; it speaks
|
||||
// the vanilla sessionserver protocol and carries no token, so this is Public. It
|
||||
@@ -386,19 +433,20 @@ func (a *API) internalAPIRoutes() []apiRoute {
|
||||
// /felis web op approve. Internal face carries the pending queue and the
|
||||
// approve action (service-token auth, no Principal); the public face carries
|
||||
// the start/status/finish the staff member's browser drives.
|
||||
{Method: "GET", Pattern: "/api/v1/internal/op-login/pending", h: a.handleOpLoginPending},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/op-login/{id}/approve", h: a.handleOpLoginApprove},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/op-login/pending", Callers: proxy, h: a.handleOpLoginPending},
|
||||
{Method: "GET", Pattern: "/api/v1/internal/op-login/{id}", Callers: proxy, h: a.handleOpLoginShow},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/op-login/{id}/approve", Callers: proxy, h: a.handleOpLoginApprove},
|
||||
|
||||
// Break-glass backup (spec §B4 "Sync"): the on-node console POSTs here to
|
||||
// snapshot a stopped world while the API is alive. Service-token auth (no
|
||||
// Principal); the shared enqueueBackup tail enforces the RWO stopped-gate.
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/backup", h: a.handleInternalBackup},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/backup", Callers: ops, h: a.handleInternalBackup},
|
||||
}
|
||||
}
|
||||
|
||||
// externalAPIRoutes is the external face's served route table (spec §7, §14):
|
||||
// Cloudflare Access-JWT auth on every /api/v1 route; the Admin entries are
|
||||
// additionally gated on the admin Zero-Trust path. It exposes liveness only —
|
||||
// session auth on every non-public /api/v1 route; the Admin entries are
|
||||
// additionally gated on the operator console host. It exposes liveness only —
|
||||
// readiness is an internal concern.
|
||||
func (a *API) externalAPIRoutes() []apiRoute {
|
||||
return []apiRoute{
|
||||
@@ -527,6 +575,22 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
{Method: "POST", Pattern: "/api/v1/account/passkey/register/finish", SetupAllowed: true, h: a.handlePasskeyRegisterFinish},
|
||||
{Method: "GET", Pattern: "/api/v1/account/passkey/credentials", SetupAllowed: true, h: a.handlePasskeyList},
|
||||
{Method: "DELETE", Pattern: "/api/v1/account/passkey/credentials/{id}", SetupAllowed: true, h: a.handlePasskeyDelete},
|
||||
// Reauth (reauth.go): the fresh proof that passkey enrollment and removal and an
|
||||
// email change require once the account has a factor. Status says whether one
|
||||
// is needed and how to give it; the pairs below take a passkey assertion or an
|
||||
// email code and mark the caller's session. SetupAllowed like the routes they
|
||||
// unlock.
|
||||
{Method: "GET", Pattern: "/api/v1/account/reauth", SetupAllowed: true, h: a.handleReauthStatus},
|
||||
{Method: "POST", Pattern: "/api/v1/account/reauth/passkey/begin", SetupAllowed: true, h: a.handleReauthPasskeyBegin},
|
||||
{Method: "POST", Pattern: "/api/v1/account/reauth/passkey/finish", SetupAllowed: true, h: a.handleReauthPasskeyFinish},
|
||||
{Method: "POST", Pattern: "/api/v1/account/reauth/email/start", SetupAllowed: true, h: a.handleReauthEmailStart},
|
||||
{Method: "POST", Pattern: "/api/v1/account/reauth/email/verify", SetupAllowed: true, h: a.handleReauthEmailVerify},
|
||||
// The caller's own sessions (handlers_account_sessions.go): list every signed-in
|
||||
// device and sign out one or all the others. App-tier and scoped to the caller
|
||||
// inside the handler, like the passkey routes above.
|
||||
{Method: "GET", Pattern: "/api/v1/account/sessions", h: a.handleListMySessions},
|
||||
{Method: "DELETE", Pattern: "/api/v1/account/sessions/{hash}", h: a.handleRevokeMySession},
|
||||
{Method: "POST", Pattern: "/api/v1/account/sessions/revoke-others", h: a.handleRevokeMyOtherSessions},
|
||||
// Account migration (spec §B3 inherit), web side. App-tier, principal-scoped: the
|
||||
// SOURCE drives status → step-up confirm (passkey forced when enrolled, else
|
||||
// email-OTP) → issue-code+name-target; the TARGET drives redeem as itself. Not
|
||||
@@ -546,11 +610,19 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// session is the correct gate (the admin verdict lives below, behind adminOnly).
|
||||
{Method: "POST", Pattern: "/api/v1/me/submissions", h: a.handleCreateSubmission},
|
||||
{Method: "GET", Pattern: "/api/v1/me/submissions", h: a.handleMySubmissions},
|
||||
{Method: "GET", Pattern: "/api/v1/me/submissions/limits", h: a.handleSubmissionLimits},
|
||||
// The blob upload for a submission the caller owns: the request body is the
|
||||
// raw gzip build context, streamed to the derived, id-namespaced location.
|
||||
// App-tier and owner-scoped (the id must belong to the principal), exactly
|
||||
// like the create/list routes above.
|
||||
{Method: "POST", Pattern: "/api/v1/me/submissions/{id}/context", h: a.handleUploadSubmissionContext},
|
||||
// The chunked form of that upload, for a context larger than one request
|
||||
// carries through the edge (Cloudflare refuses bodies over 100 MB): GET
|
||||
// reports the staged length (the resume point), PUT ?offset= appends one
|
||||
// part, POST .../complete stores the staged whole. Same owner scoping.
|
||||
{Method: "GET", Pattern: "/api/v1/me/submissions/{id}/context/upload", h: a.handleContextUploadStatus},
|
||||
{Method: "PUT", Pattern: "/api/v1/me/submissions/{id}/context/upload", h: a.handleContextUploadPart},
|
||||
{Method: "POST", Pattern: "/api/v1/me/submissions/{id}/context/upload/complete", h: a.handleContextUploadComplete},
|
||||
// Withdraw the caller's OWN pending submission: the row and its uploaded
|
||||
// context are deleted, freeing the pending slot and storage budget. Same
|
||||
// owner-scoping as the upload route — a reviewed submission is frozen (409)
|
||||
@@ -572,9 +644,13 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// is build-time RCE against the cluster, so submission requires the admin
|
||||
// Zero-Trust path, not merely an authenticated session.
|
||||
{Method: "POST", Pattern: "/api/v1/images/build", Admin: true, h: a.handleBuildImage},
|
||||
{Method: "GET", Pattern: "/api/v1/images/build", Admin: true, h: a.handleListBuilds},
|
||||
{Method: "GET", Pattern: "/api/v1/images/build/{id}", Admin: true, h: a.handleGetBuild},
|
||||
{Method: "GET", Pattern: "/api/v1/images/build/{id}/logs", Admin: true, h: a.handleBuildLogs},
|
||||
{Method: "POST", Pattern: "/api/v1/images/build/{id}/cancel", Admin: true, h: a.handleCancelBuild},
|
||||
{Method: "GET", Pattern: "/api/v1/images/build/{id}/scan", Admin: true, h: a.handleBuildScan},
|
||||
{Method: "GET", Pattern: "/api/v1/images/build/{id}/scan/report", Admin: true, h: a.handleBuildScanReport},
|
||||
{Method: "GET", Pattern: "/api/v1/images/build/{id}/sbom", Admin: true, h: a.handleBuildSBOM},
|
||||
{Method: "GET", Pattern: "/api/v1/images", Admin: true, h: a.handleListImages},
|
||||
{Method: "POST", Pattern: "/api/v1/images", Admin: true, h: a.handleAddImage},
|
||||
{Method: "DELETE", Pattern: "/api/v1/images", Admin: true, h: a.handleRemoveImage},
|
||||
@@ -594,10 +670,12 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
{Method: "GET", Pattern: "/api/v1/submissions/{id}/context", Admin: true, h: a.handleAdminSubmissionContext},
|
||||
// Auto-update maintenance window (spec §B; decision core internal/updates).
|
||||
// Admin-tier: it governs whether Felis may apply an update to itself, so setting
|
||||
// it requires the admin Zero-Trust path, not a mere session. API+persistence
|
||||
// only — the runner/executors that consume the window are still INTEGRATION-ONLY.
|
||||
// it requires the admin Zero-Trust path, not a mere session. Advisory: `felis
|
||||
// update` on the host reads it and warns before an apply outside it.
|
||||
{Method: "GET", Pattern: "/api/v1/updates/window", Admin: true, h: a.handleGetUpdateWindow},
|
||||
{Method: "PUT", Pattern: "/api/v1/updates/window", Admin: true, h: a.handleSetUpdateWindow},
|
||||
// The newest version check felis-update-check.timer recorded on the host.
|
||||
{Method: "GET", Pattern: "/api/v1/updates/report", Admin: true, h: a.handleGetUpdateReport},
|
||||
// Control-plane database backup freshness, as the host's felis-db-backup.timer
|
||||
// last recorded it. Admin-tier: it names the host backup directory.
|
||||
{Method: "GET", Pattern: "/api/v1/platform/db-backup", Admin: true, h: a.handleGetDBBackup},
|
||||
@@ -625,14 +703,14 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// InternalHandler builds the internal-face http.Handler: service-token auth, no
|
||||
// Zero Trust (spec §14 red line). /healthz and /readyz are unauthenticated.
|
||||
func (a *API) InternalHandler() http.Handler {
|
||||
return a.buildFace(a.internalAPIRoutes(), a.requireInternal)
|
||||
return a.buildFace("internal", a.internalAPIRoutes(), a.requireInternal)
|
||||
}
|
||||
|
||||
// ExternalHandler builds the external-face http.Handler: Access-JWT auth on every
|
||||
// /api/v1 route, with admin-tier routes additionally gated by the admin Access
|
||||
// path inside their handlers.
|
||||
// ExternalHandler builds the external-face http.Handler: session auth on every
|
||||
// non-public /api/v1 route, with admin-tier routes additionally gated on the
|
||||
// operator console host inside their handlers.
|
||||
func (a *API) ExternalHandler() http.Handler {
|
||||
return a.buildFace(a.externalAPIRoutes(), a.requireExternal)
|
||||
return a.buildFace("external", a.externalAPIRoutes(), a.requireExternal)
|
||||
}
|
||||
|
||||
// buildFace assembles one face from its route table. Public routes are mounted
|
||||
@@ -641,7 +719,7 @@ func (a *API) ExternalHandler() http.Handler {
|
||||
// adminOnly, and Owner routes in ownerOnly. Because both faces are built from the
|
||||
// same table the OpenAPI parity test reads, the served surface and the documented
|
||||
// surface cannot drift apart without failing the build.
|
||||
func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler {
|
||||
func (a *API) buildFace(face string, routes []apiRoute, guard func(http.Handler) http.Handler) http.Handler {
|
||||
mux := http.NewServeMux()
|
||||
auth := http.NewServeMux()
|
||||
for _, rt := range routes {
|
||||
@@ -651,10 +729,16 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler
|
||||
if rt.AuthDoor {
|
||||
h = a.throttleAuthDoor(h)
|
||||
}
|
||||
mux.HandleFunc(pattern, h)
|
||||
mux.HandleFunc(pattern, tagRoute(rt.Pattern, h))
|
||||
continue
|
||||
}
|
||||
h := rt.h
|
||||
if (face == "internal") != (len(rt.Callers) > 0) {
|
||||
panic(fmt.Sprintf("%s route %s %s: internal routes list their callers, external ones none", face, rt.Method, rt.Pattern))
|
||||
}
|
||||
if len(rt.Callers) > 0 {
|
||||
h = callersOnly(rt.Callers, h)
|
||||
}
|
||||
if rt.Owner {
|
||||
h = a.ownerOnly(rt.h)
|
||||
}
|
||||
@@ -667,22 +751,61 @@ func (a *API) buildFace(routes []apiRoute, guard func(http.Handler) http.Handler
|
||||
if !rt.SetupAllowed {
|
||||
h = a.requireOnboarded(h)
|
||||
}
|
||||
auth.HandleFunc(pattern, h)
|
||||
auth.HandleFunc(pattern, tagRoute(rt.Pattern, h))
|
||||
}
|
||||
guarded := guard(auth)
|
||||
mux.Handle("/api/v1/", http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if _, pattern := auth.Handler(r); pattern == "" {
|
||||
http.NotFound(w, r)
|
||||
_, pattern := auth.Handler(r)
|
||||
if pattern == "" {
|
||||
writeNoRoute(w, r, mux, auth)
|
||||
return
|
||||
}
|
||||
// Named before the guard runs, so a refused request is counted under
|
||||
// the route it asked for.
|
||||
noteRoute(r, pattern)
|
||||
guarded.ServeHTTP(w, r)
|
||||
}))
|
||||
return a.baseChain(mux)
|
||||
top := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if _, pattern := mux.Handler(r); pattern == "" {
|
||||
writeNoRoute(w, r, mux, auth)
|
||||
return
|
||||
}
|
||||
mux.ServeHTTP(w, r)
|
||||
})
|
||||
return a.baseChain(face, top)
|
||||
}
|
||||
|
||||
// baseChain wraps a handler in the cross-cutting middleware shared by both faces.
|
||||
func (a *API) baseChain(h http.Handler) http.Handler {
|
||||
return withRequestID(withRecover(h))
|
||||
// baseChain wraps a handler in the cross-cutting middleware shared by both faces:
|
||||
// the request id, then the access log and metrics (which see the final status of
|
||||
// everything inside), the response security headers, the cross-site write fence,
|
||||
// the request-body read deadline, and panic recovery.
|
||||
func (a *API) baseChain(face string, h http.Handler) http.Handler {
|
||||
return withRequestID(a.observe(face, withSecurityHeaders(rejectCrossSiteWrites(withBodyDeadline(withRecover(h))))))
|
||||
}
|
||||
|
||||
// writeNoRoute answers a request no route took, in the API's error envelope: 405
|
||||
// with an Allow header when the path exists under other methods, 404 otherwise.
|
||||
// ServeMux's own answers are plain text and turn a wrong method on a guarded
|
||||
// route into a 404, since the guarded routes sit behind one catch-all.
|
||||
func writeNoRoute(w http.ResponseWriter, r *http.Request, muxes ...*http.ServeMux) {
|
||||
var allow []string
|
||||
for _, m := range []string{http.MethodGet, http.MethodPost, http.MethodPut, http.MethodPatch, http.MethodDelete} {
|
||||
probe := r.Clone(r.Context())
|
||||
probe.Method = m
|
||||
for _, mux := range muxes {
|
||||
if _, p := mux.Handler(probe); p != "" && p != "/api/v1/" {
|
||||
allow = append(allow, m)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(allow) > 0 {
|
||||
w.Header().Set("Allow", strings.Join(allow, ", "))
|
||||
writeError(w, r, newError(http.StatusMethodNotAllowed, "method_not_allowed",
|
||||
"%s is not allowed here; use %s", r.Method, strings.Join(allow, ", ")))
|
||||
return
|
||||
}
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "no such endpoint"))
|
||||
}
|
||||
|
||||
// requireOnboarded fences an authenticated route behind the setup-lockdown: a
|
||||
@@ -704,7 +827,15 @@ func (a *API) requireOnboarded(h http.HandlerFunc) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
if p != nil && p.ViaSession && !p.EmailVerified {
|
||||
creds, _ := a.Repo.PasskeyCredentialsForUser(r.Context(), p.UserID)
|
||||
creds, err := a.Repo.PasskeyCredentialsForUser(r.Context(), p.UserID)
|
||||
if err != nil {
|
||||
// A store outage reads as retry-later; setup_required would send
|
||||
// the caller off to enroll a passkey they may already have.
|
||||
log.Printf("api: %s %s: onboarding check (request_id=%s): %v",
|
||||
r.Method, r.URL.Path, requestIDFromContext(r.Context()), err)
|
||||
writeError(w, r, errAuthUnavailable)
|
||||
return
|
||||
}
|
||||
if len(creds) == 0 {
|
||||
writeError(w, r, newError(http.StatusForbidden, "setup_required",
|
||||
"passkey enrollment is required before this action is available"))
|
||||
@@ -734,6 +865,8 @@ type ctxKey int
|
||||
const (
|
||||
ctxKeyRequestID ctxKey = iota
|
||||
ctxKeyPrincipal
|
||||
ctxKeyReqInfo
|
||||
ctxKeyCaller
|
||||
)
|
||||
|
||||
func requestIDFromContext(ctx context.Context) string {
|
||||
@@ -751,6 +884,21 @@ func principalFromContext(ctx context.Context) *Principal {
|
||||
return nil
|
||||
}
|
||||
|
||||
// callerFromContext returns the internal-face caller, or "" off that face.
|
||||
func callerFromContext(ctx context.Context) Caller {
|
||||
c, _ := ctx.Value(ctxKeyCaller).(Caller)
|
||||
return c
|
||||
}
|
||||
|
||||
// internalSource is the audit Source for an action taken on the internal face,
|
||||
// naming the caller whose token asked for it ("internal:velocity").
|
||||
func internalSource(r *http.Request) string {
|
||||
if c := callerFromContext(r.Context()); c != "" {
|
||||
return "internal:" + string(c)
|
||||
}
|
||||
return "internal"
|
||||
}
|
||||
|
||||
// ---- per-key cooldown (wake + OTP) ----
|
||||
|
||||
// cooldownLimiter is an in-memory per-key cooldown. It backs two throttles with
|
||||
@@ -828,7 +976,8 @@ func (c *cooldownLimiter) record(name string) {
|
||||
}
|
||||
|
||||
// reserve atomically checks name's cooldown AND, if the window is open, records it
|
||||
// in the same critical section, returning the reservation time and true. Unlike
|
||||
// in the same critical section, returning the reservation time and true; inside the
|
||||
// window it returns the standing reservation's time and false. Unlike
|
||||
// allowed→record there is no gap between the check and the commit, so a burst of
|
||||
// truly concurrent callers yields exactly one winner. Use it where the throttle is
|
||||
// the SOLE defense and each admitted call has a non-idempotent side effect (an OTP
|
||||
@@ -845,7 +994,7 @@ func (c *cooldownLimiter) reserve(name string, window time.Duration) (time.Time,
|
||||
t := c.now()
|
||||
c.noteWindow(window, t)
|
||||
if last, ok := c.last[name]; ok && t.Sub(last) < window {
|
||||
return time.Time{}, false
|
||||
return last, false
|
||||
}
|
||||
c.last[name] = t
|
||||
return t, true
|
||||
|
||||
+514
-197
File diff suppressed because it is too large.
Load diff
@@ -36,8 +36,8 @@ const (
|
||||
)
|
||||
|
||||
// auditActor is the display name for a principal: an email only when something
|
||||
// vouches for it (an Access JWT, or a session whose address was verified), else
|
||||
// the username. A player can set their address to anyone's before verifying it,
|
||||
// vouches for it (a session whose address was verified, or an ExternalAuth other
|
||||
// than SessionAuth that resolved the principal itself), else the username. A player can set their address to anyone's before verifying it,
|
||||
// so an unverified email would let them sign rows as that person.
|
||||
func auditActor(p *Principal) string {
|
||||
switch {
|
||||
|
||||
@@ -43,7 +43,7 @@ func TestAuditCannotBeSignedWithAnotherPersonsEmail(t *testing.T) {
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.staff["owner"] = &StaffUser{ID: "u1", Username: "owner", Email: "[email protected]", Role: "owner", EmailVerified: true}
|
||||
repo.staff["mallory"] = &StaffUser{ID: "u2", Username: "mallory", Email: "[email protected]", Role: "user", EmailVerified: true}
|
||||
repo.sessions[hashCookie("tok")] = &fakeSession{userID: "u2", expiresAt: time.Unix(1_700_000_000, 0).Add(time.Hour)}
|
||||
repo.sessions[hashCookie("tok")] = &fakeSession{userID: "u2", expiresAt: frozenNow.Add(time.Hour), reauthAt: frozenNow}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = SessionAuth{Repo: repo, RootDomain: testRoot, Now: api.now}
|
||||
api.ClientIPHeader = "CF-Connecting-IP"
|
||||
|
||||
+80
-117
@@ -5,29 +5,26 @@ import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strings"
|
||||
|
||||
"github.com/golang-jwt/jwt/v5"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Principal is the authenticated external-face caller (spec §7, §14). The
|
||||
// internal face (service token) never produces a Principal — it is a trusted
|
||||
// machine caller, not a person.
|
||||
type Principal struct {
|
||||
// UserID is the stable web identity (SSO subject → users.id).
|
||||
// UserID is the account's users.id.
|
||||
UserID string
|
||||
// Username is the account's login name; empty for an Access-JWT caller.
|
||||
// Username is the account's login name.
|
||||
Username string
|
||||
// Email is the account's address. Only an Access JWT or EmailVerified vouches
|
||||
// for it: a player can set any address before verifying it (auditActor).
|
||||
// Email is the account's address. Only EmailVerified vouches for it: a player
|
||||
// can set any address before verifying it (auditActor).
|
||||
Email string
|
||||
// Role is "owner", "admin", or "user" (mirrors users.role).
|
||||
Role string
|
||||
// ViaAdminAccess is true only when the request arrived through an admin-graded
|
||||
// path: the admin.* Zero-Trust hostname (Cloudflare Access, the remote face) OR
|
||||
// a local session presented on the op.console host (SessionAuth, the
|
||||
// passwordless face). Admin-tier operations require it in addition to
|
||||
// a staff role (spec §14: ZT is graded by operation). A staff session
|
||||
// arriving on the player console (console.*) never sets it.
|
||||
// ViaAdminAccess is true only when a staff session arrived on the operator
|
||||
// console host (hostIsAdminConsole). Admin-tier operations require it in
|
||||
// addition to a staff role (spec §14: ZT is graded by operation). A staff
|
||||
// session arriving on the player console (console.*) never sets it.
|
||||
ViaAdminAccess bool
|
||||
// EmailVerified mirrors users.email_verified. The lockdown middleware gates
|
||||
// setup-incomplete accounts (EmailVerified=false, e.g. a freshly bootstrapped
|
||||
@@ -35,12 +32,13 @@ type Principal struct {
|
||||
// routes only, so an intercepted setup URL cannot yield full admin access
|
||||
// before the email-OTP verification step completes.
|
||||
EmailVerified bool
|
||||
// ViaSession is true when the principal was authenticated via a local session
|
||||
// cookie (SessionAuth), not a Cloudflare-Access JWT. The setup-lockdown gate
|
||||
// only applies to session-authenticated principals — a JWT caller already
|
||||
// passed Zero Trust at the edge, so the local-email-verification gate is not
|
||||
// the right boundary for them.
|
||||
// ViaSession is true for every principal SessionAuth resolves from a session
|
||||
// cookie. The session-scoped gates (setup lockdown, reauth, the device list)
|
||||
// key on it.
|
||||
ViaSession bool
|
||||
// ReauthAt is when the holder of the session last proved a factor of the
|
||||
// account; zero for a session that never did.
|
||||
ReauthAt time.Time
|
||||
}
|
||||
|
||||
// staffRole reports whether a stored user role carries staff standing: admin,
|
||||
@@ -52,8 +50,8 @@ func staffRole(role string) bool {
|
||||
}
|
||||
|
||||
// IsAdmin reports whether the principal may perform admin-tier operations.
|
||||
// Both the role claim and the admin Access path are required: a staff
|
||||
// session arriving on panel.* must not bypass the Zero-Trust boundary.
|
||||
// Both the staff role and arrival on the operator console host are required: a
|
||||
// staff session arriving on the player console must not reach admin routes.
|
||||
// An owner implicitly passes this check (the owner role is a superset of admin).
|
||||
func (p *Principal) IsAdmin() bool {
|
||||
return p != nil && staffRole(p.Role) && p.ViaAdminAccess
|
||||
@@ -62,104 +60,88 @@ func (p *Principal) IsAdmin() bool {
|
||||
// IsOwner reports whether the principal holds the platform-level owner role
|
||||
// — the single identity that may manage users, quotas, and sessions. Only the
|
||||
// first staff account minted by break-glass carries this role; every subsequent
|
||||
// Operator is a plain admin. Like IsAdmin, it requires the admin Access path.
|
||||
// Operator is a plain admin. Like IsAdmin, it requires the operator console host.
|
||||
func (p *Principal) IsOwner() bool {
|
||||
return p != nil && p.Role == "owner" && p.ViaAdminAccess
|
||||
}
|
||||
|
||||
// InternalAuth authenticates the internal face (velocity / backend callbacks):
|
||||
// a static service token presented as a Bearer credential. The internal face
|
||||
// is never wrapped in Zero Trust (spec §1.8, §14 red line).
|
||||
// Caller names the machine behind an internal-face token. Each caller holds a
|
||||
// token of its own and each internal route lists the callers it serves
|
||||
// (apiRoute.Callers), so a token copied out of one namespace opens only what
|
||||
// that caller needs: the build Job's token reads a submission's context and
|
||||
// nothing else, and only the proxy and the login gate can mint link codes.
|
||||
type Caller string
|
||||
|
||||
const (
|
||||
// CallerVelocity is the proxy's felis-link plugin (felis-service-token,
|
||||
// written into felis-link.properties on the host).
|
||||
CallerVelocity Caller = "velocity"
|
||||
// CallerLimbo is the login gate's felis-limbo plugin (felis-limbo-token,
|
||||
// injected into the login pod only).
|
||||
CallerLimbo Caller = "limbo"
|
||||
// CallerBuild is the build Job's context-fetch initContainer
|
||||
// (felis-build-token in the build namespace).
|
||||
CallerBuild Caller = "build"
|
||||
// CallerOps is the on-node console, `felis backup-now` (felis-ops-token,
|
||||
// control namespace only).
|
||||
CallerOps Caller = "ops"
|
||||
)
|
||||
|
||||
// InternalAuth authenticates the internal face: a per-caller static token
|
||||
// presented as a Bearer credential, answered with the caller it belongs to. The
|
||||
// internal face is never wrapped in Zero Trust (spec §1.8, §14 red line).
|
||||
type InternalAuth interface {
|
||||
Authenticate(r *http.Request) error
|
||||
Authenticate(r *http.Request) (Caller, error)
|
||||
}
|
||||
|
||||
// ExternalAuth authenticates the external face (people / panel) and returns the
|
||||
// resolved Principal. Production verifies a Cloudflare Access JWT and checks its
|
||||
// audience; the verification key source (JWKS) is injected so the audience and
|
||||
// expiry logic stay unit-testable.
|
||||
// resolved Principal. Production is SessionAuth: the local session cookie the
|
||||
// sign-in doors mint. Cloudflare Access, when an install sits behind it, is
|
||||
// enforced at the edge only; felis-api does not read the Access JWT, so the
|
||||
// identity and role always come from the users table.
|
||||
type ExternalAuth interface {
|
||||
Authenticate(r *http.Request) (*Principal, error)
|
||||
}
|
||||
|
||||
// BearerTokenAuth is the production InternalAuth: a constant-time comparison
|
||||
// against the configured service token. A zero token fails closed so a
|
||||
// misconfiguration can never silently disable internal-face auth.
|
||||
type BearerTokenAuth struct {
|
||||
Token string
|
||||
// CallerTokens is the production InternalAuth: each caller's token, compared in
|
||||
// constant time. A caller with no token cannot authenticate, so a missing
|
||||
// Secret fails closed for that caller alone.
|
||||
type CallerTokens map[Caller]string
|
||||
|
||||
// NewCallerTokens refuses a set that would make the caller ambiguous: two
|
||||
// callers sharing a value, which is also what an install whose callers all
|
||||
// still hold the one old service token would look like.
|
||||
func NewCallerTokens(tokens map[Caller]string) (CallerTokens, error) {
|
||||
seen := map[string]Caller{}
|
||||
for caller, tok := range tokens {
|
||||
if tok == "" {
|
||||
continue
|
||||
}
|
||||
if other, dup := seen[tok]; dup {
|
||||
return nil, fmt.Errorf("the %s and %s tokens are the same value; each caller needs its own", other, caller)
|
||||
}
|
||||
seen[tok] = caller
|
||||
}
|
||||
return CallerTokens(tokens), nil
|
||||
}
|
||||
|
||||
// Authenticate checks the Authorization: Bearer header against the token.
|
||||
func (b BearerTokenAuth) Authenticate(r *http.Request) error {
|
||||
if b.Token == "" {
|
||||
return fmt.Errorf("internal auth not configured")
|
||||
}
|
||||
// Authenticate matches the Authorization: Bearer header against every caller's
|
||||
// token, comparing each so the time taken does not say which one matched.
|
||||
func (c CallerTokens) Authenticate(r *http.Request) (Caller, error) {
|
||||
got := bearerToken(r)
|
||||
if got == "" {
|
||||
return fmt.Errorf("missing bearer token")
|
||||
return "", fmt.Errorf("missing bearer token")
|
||||
}
|
||||
if subtle.ConstantTimeCompare([]byte(got), []byte(b.Token)) != 1 {
|
||||
return fmt.Errorf("invalid service token")
|
||||
var match Caller
|
||||
for caller, tok := range c {
|
||||
if tok != "" && subtle.ConstantTimeCompare([]byte(got), []byte(tok)) == 1 {
|
||||
match = caller
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// AccessVerifier is the production ExternalAuth: it parses a Cloudflare Access
|
||||
// JWT, verifies the signature with the injected key function, and enforces the
|
||||
// configured audience (spec §7 "验 aud"). AdminAudience, when set, marks a token
|
||||
// minted for the admin.* application so admin-tier routes can require it.
|
||||
type AccessVerifier struct {
|
||||
// Audience is the required `aud` claim for any external request.
|
||||
Audience string
|
||||
// AdminAudience, if non-empty and present in the token's aud set, flags the
|
||||
// principal as having passed the admin Zero-Trust path.
|
||||
AdminAudience string
|
||||
// Keyfunc resolves the signing key (production: a JWKS-backed keyfunc).
|
||||
Keyfunc jwt.Keyfunc
|
||||
}
|
||||
|
||||
// accessClaims are the subset of Access JWT claims we consume.
|
||||
type accessClaims struct {
|
||||
Email string `json:"email"`
|
||||
Role string `json:"felis_role"`
|
||||
jwt.RegisteredClaims
|
||||
}
|
||||
|
||||
// Authenticate verifies the Access JWT and maps it onto a Principal.
|
||||
func (v AccessVerifier) Authenticate(r *http.Request) (*Principal, error) {
|
||||
if v.Keyfunc == nil {
|
||||
return nil, fmt.Errorf("external auth not configured")
|
||||
if match == "" {
|
||||
return "", fmt.Errorf("invalid service token")
|
||||
}
|
||||
raw := accessToken(r)
|
||||
if raw == "" {
|
||||
return nil, fmt.Errorf("missing access token")
|
||||
}
|
||||
|
||||
var claims accessClaims
|
||||
parser := jwt.NewParser(jwt.WithExpirationRequired())
|
||||
if _, err := parser.ParseWithClaims(raw, &claims, v.Keyfunc); err != nil {
|
||||
return nil, fmt.Errorf("invalid access token: %w", err)
|
||||
}
|
||||
|
||||
// Audience check: the configured app aud must be present. We do not delegate
|
||||
// to jwt.WithAudience so we can additionally detect the admin audience.
|
||||
if !audienceContains(claims.Audience, v.Audience) {
|
||||
return nil, fmt.Errorf("token audience does not include %q", v.Audience)
|
||||
}
|
||||
if claims.Subject == "" {
|
||||
return nil, fmt.Errorf("token missing subject")
|
||||
}
|
||||
|
||||
role := claims.Role
|
||||
if role == "" {
|
||||
role = "user"
|
||||
}
|
||||
return &Principal{
|
||||
UserID: claims.Subject,
|
||||
Email: claims.Email,
|
||||
Role: role,
|
||||
ViaAdminAccess: v.AdminAudience != "" && audienceContains(claims.Audience, v.AdminAudience),
|
||||
}, nil
|
||||
return match, nil
|
||||
}
|
||||
|
||||
// bearerToken extracts a Bearer credential from the Authorization header.
|
||||
@@ -171,22 +153,3 @@ func bearerToken(r *http.Request) string {
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// accessToken prefers the Cloudflare Access assertion header, falling back to a
|
||||
// Bearer credential so the same verifier works behind a proxy or directly.
|
||||
func accessToken(r *http.Request) string {
|
||||
if h := r.Header.Get("Cf-Access-Jwt-Assertion"); h != "" {
|
||||
return h
|
||||
}
|
||||
return bearerToken(r)
|
||||
}
|
||||
|
||||
// audienceContains reports whether want appears in the aud claim set.
|
||||
func audienceContains(aud jwt.ClaimStrings, want string) bool {
|
||||
for _, a := range aud {
|
||||
if a == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -23,3 +23,13 @@ import "context"
|
||||
type Backuper interface {
|
||||
Backup(ctx context.Context, serverName, formerOwner string) error
|
||||
}
|
||||
|
||||
// RestoreSnapshotter is the Backuper's safety-snapshot lever: it enqueues a backup
|
||||
// of the world as it is now, labelled with the restore to run once that backup
|
||||
// has succeeded (backupID, backupRef). RestoreChains settles it: it starts the
|
||||
// restore after a successful snapshot and gives the restore up after a failed
|
||||
// one, so a restore never overwrites a world that has no copy. Until it is
|
||||
// settled the snapshot holds the world volume as a restore (internal/maintenance).
|
||||
type RestoreSnapshotter interface {
|
||||
BackupThenRestore(ctx context.Context, serverName, formerOwner, backupID, backupRef string) error
|
||||
}
|
||||
@@ -84,8 +84,9 @@ type ServerSpecPatch struct {
|
||||
// so handlers are tested against a fake; the controller-runtime implementation
|
||||
// (k8sCluster) is integration-tested only — it requires a live cluster.
|
||||
type Cluster interface {
|
||||
// Ping verifies the K8s API and CRD informer are healthy — used by /readyz
|
||||
// (spec §7) to confirm the lifecycle store is reachable and synced.
|
||||
// Ping verifies the K8s API is reachable and the MinecraftServer cache has
|
||||
// synced — used by /readyz (spec §7), so a replica serves the fleet reads only
|
||||
// once it holds the whole fleet.
|
||||
Ping(ctx context.Context) error
|
||||
|
||||
// GetServer reads one MinecraftServer's lifecycle view, or ErrNotFound.
|
||||
|
||||
@@ -101,7 +101,7 @@ func (k *K8sConsole) RunCommand(ctx context.Context, name, command string) (stri
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
out, err := conn.Execute(command)
|
||||
out, err := conn.ExecuteContext(ctx, command)
|
||||
if err != nil {
|
||||
return "", ErrConsoleUnavailable
|
||||
}
|
||||
|
||||
+17
-10
@@ -57,6 +57,12 @@ var (
|
||||
// finish endpoint exists; the ceremony state is gone (never begun, already
|
||||
// consumed, or expired) — so handlers map it to 400, not 404.
|
||||
ErrPasskeyChallengeInvalid = errors.New("passkey challenge invalid or expired")
|
||||
// ErrLastPasskey means a passkey delete would remove the account's only one while
|
||||
// its email is unverified. That passkey is then the account's only durable way
|
||||
// in (setupRequired: no verified email and no passkey puts it back behind the
|
||||
// setup gate, and a staff account has no other self-service door at all), so the
|
||||
// delete is refused; handlers map it to 409 last_passkey.
|
||||
ErrLastPasskey = errors.New("cannot remove the only passkey of an account without a verified email")
|
||||
// ErrPlayerBindForbidden means a public Bind-Code redemption resolved to a STAFF
|
||||
// account (admin or owner), which the player-console bootstrap refuses
|
||||
// (console-tier access model). Staff authenticate at op.console behind Zero Trust,
|
||||
@@ -81,14 +87,13 @@ var (
|
||||
// guard and gets a 409 instead of a raw unique-violation 500. Distinct from
|
||||
// ErrConflict so the message can name the cause (the email is spoken for).
|
||||
ErrEmailTaken = errors.New("email already verified on another account")
|
||||
// ErrTooManyDiscoverableChallenges means the non-user-keyed discoverable ("usernameless")
|
||||
// login challenge store is at its hard cap of live rows (task #40, migration 0013).
|
||||
// Unlike the user-keyed enrollment/login challenges — which self-bound via a per-user
|
||||
// supersede — a from-zero begin has no principal to key a fair per-caller limit on, so the
|
||||
// table is capped globally and a begin over the cap is refused. Distinct from the other
|
||||
// sentinels so the handler answers 429 (a transient "too busy, retry" — the cap self-clears
|
||||
// as challenges expire), never a 400 that invites an immediate retry.
|
||||
ErrTooManyDiscoverableChallenges = errors.New("too many discoverable login challenges in flight")
|
||||
// ErrTooManyPasskeyChallenges means a passkey login begin was refused because too many
|
||||
// login challenges are live: the caller's source already holds its allowance
|
||||
// (maxLiveChallengesPerSource), or the discoverable store is at its global cap
|
||||
// (maxLiveDiscoverableChallenges). Distinct from the other sentinels so the handler
|
||||
// answers 429 (a transient "too busy, retry" — both bounds clear as challenges expire),
|
||||
// never a 400 that invites an immediate retry.
|
||||
ErrTooManyPasskeyChallenges = errors.New("too many passkey login challenges in flight")
|
||||
// ErrNotStopped means a world-volume operation was refused because the server is
|
||||
// not fully stopped: desiredState is not Stopped, or its pod is still shutting
|
||||
// down (phase Stopping) and holds the volume while it saves.
|
||||
@@ -157,8 +162,10 @@ var (
|
||||
// 503 "retry" instead of a 401 that reads as "log in again".
|
||||
errAuthUnavailable = newError(http.StatusServiceUnavailable, "auth_unavailable",
|
||||
"authentication is temporarily unavailable; retry shortly")
|
||||
errForbidden = newError(http.StatusForbidden, "forbidden", "not permitted")
|
||||
errBadRequest = newError(http.StatusBadRequest, "bad_request", "invalid request")
|
||||
errForbidden = newError(http.StatusForbidden, "forbidden", "not permitted")
|
||||
// errWrongCaller: a valid internal token for a caller this route does not serve.
|
||||
errWrongCaller = newError(http.StatusForbidden, "wrong_caller", "this token's caller may not use this route")
|
||||
errBadRequest = newError(http.StatusBadRequest, "bad_request", "invalid request")
|
||||
)
|
||||
|
||||
// writeJSON writes v as an indented JSON body with the given status.
|
||||
|
||||
@@ -1,11 +1,9 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strings"
|
||||
@@ -25,7 +23,9 @@ import (
|
||||
// email-OTP — advancing to 'confirmed'. Mere
|
||||
// session possession is never enough; a stolen
|
||||
// session cannot read the mailbox nor present the
|
||||
// authenticator.
|
||||
// authenticator, and cannot enroll one of its own
|
||||
// without a recent proof of an existing factor
|
||||
// (reauth.go).
|
||||
// 3. web issue code + name target → handleMigrateIssueCode: the source names the
|
||||
// target account by id and mints a one-time code
|
||||
// ('code_issued').
|
||||
@@ -116,7 +116,7 @@ func (a *API) handleMigrateStart(w http.ResponseWriter, r *http.Request) {
|
||||
// Internal-face event: attribute to the in-game initiator, Source 'internal'.
|
||||
a.auditEntry(r, AuditEntry{
|
||||
Actor: "mc:" + mcUUID,
|
||||
Source: "internal",
|
||||
Source: internalSource(r),
|
||||
Action: "account.migrate.start",
|
||||
})
|
||||
writeJSON(w, http.StatusCreated, map[string]any{"started": true, "state": "initiated"})
|
||||
@@ -205,54 +205,7 @@ func (a *API) handleMigrateConfirmOTPStart(w http.ResponseWriter, r *http.Reques
|
||||
}
|
||||
// Per-recipient cooldown, namespaced apart from the other OTP doors so they never
|
||||
// perturb each other's throttle.
|
||||
if until, err := a.Repo.OTPLockedUntil(r.Context(), p.UserID, otpPurposeMigrate, a.now()); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
} else if !until.IsZero() {
|
||||
writeOTPAccountLocked(w, r, until, a.now())
|
||||
return
|
||||
}
|
||||
emailKey := "migrate:confirm:" + strings.ToLower(p.Email)
|
||||
lim := a.otpLimiter()
|
||||
emailAt, ok := lim.reserve(emailKey, otpResendCooldown)
|
||||
if !ok {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_resend_cooldown",
|
||||
"a code was sent recently; wait a moment before requesting another"))
|
||||
return
|
||||
}
|
||||
committed := false
|
||||
defer func() {
|
||||
if !committed {
|
||||
lim.release(emailKey, emailAt)
|
||||
}
|
||||
}()
|
||||
code, err := newEmailOTP()
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
id, err := newOTPID()
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expiresAt := a.now().Add(otpTTL)
|
||||
if err := a.Repo.CreateEmailOTP(r.Context(), id, p.UserID, p.Email, otpCodeHash(code), otpPurposeMigrate, expiresAt); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if err := a.deliverOTP(r.Context(), p.Email, code); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
committed = true
|
||||
a.audit(r, "account.migrate.confirm_otp_sent", "")
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
||||
}
|
||||
|
||||
// migrateConfirmOTPVerifyRequest is the OTP step-up verify body: the code from the email.
|
||||
type migrateConfirmOTPVerifyRequest struct {
|
||||
Code string `json:"code"`
|
||||
a.startStepUpOTP(w, r, p, otpPurposeMigrate, "migrate:confirm:", "account.migrate.confirm_otp_sent")
|
||||
}
|
||||
|
||||
// handleMigrateConfirmOTPVerify redeems the migration step-up code and, on a match,
|
||||
@@ -260,7 +213,7 @@ type migrateConfirmOTPVerifyRequest struct {
|
||||
// is the login-door one (no identity side-effect): the address is already proven.
|
||||
func (a *API) handleMigrateConfirmOTPVerify(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
var req migrateConfirmOTPVerifyRequest
|
||||
var req stepUpOTPVerifyRequest
|
||||
if err := decodeJSON(w, r, &req); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
@@ -273,25 +226,7 @@ func (a *API) handleMigrateConfirmOTPVerify(w http.ResponseWriter, r *http.Reque
|
||||
if _, ok := a.requireInitiatedMigration(w, r, p.UserID); !ok {
|
||||
return
|
||||
}
|
||||
var lock *OTPAccountLockedError
|
||||
err := a.Repo.ConsumeLoginEmailOTP(r.Context(), p.UserID, otpPurposeMigrate, otpCodeHash(code), a.now())
|
||||
if isOTPRefusal(err) {
|
||||
a.authFailure(r, "migrate_confirm", otpFailureReason(err), nil)
|
||||
}
|
||||
switch {
|
||||
case errors.As(err, &lock):
|
||||
a.noteOTPLock(r, err, p.UserID, otpPurposeMigrate)
|
||||
writeOTPAccountLocked(w, r, lock.Until, a.now())
|
||||
return
|
||||
case errors.Is(err, ErrOTPLocked):
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_locked",
|
||||
"too many incorrect attempts; request a new code"))
|
||||
return
|
||||
case errors.Is(err, ErrOTPInvalid):
|
||||
writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "email code is invalid or expired"))
|
||||
return
|
||||
case err != nil:
|
||||
writeError(w, r, err)
|
||||
if !a.verifyStepUpOTP(w, r, p, otpPurposeMigrate, "migrate_confirm", code) {
|
||||
return
|
||||
}
|
||||
if err := a.Repo.ConfirmMigration(r.Context(), p.UserID, "email_otp", a.now()); err != nil {
|
||||
@@ -307,17 +242,6 @@ func (a *API) handleMigrateConfirmOTPVerify(w http.ResponseWriter, r *http.Reque
|
||||
writeJSON(w, http.StatusOK, map[string]any{"confirmed": true})
|
||||
}
|
||||
|
||||
// migratePasskeyUser builds the PasskeyUser the assertion ceremony needs for the
|
||||
// already-logged-in source (contrast the login door, which resolves it from a typed
|
||||
// email). The credential set must be identical between begin and finish.
|
||||
func migratePasskeyUser(p *Principal, creds []PasskeyCredential) PasskeyUser {
|
||||
name := p.Email
|
||||
if name == "" {
|
||||
name = p.UserID
|
||||
}
|
||||
return PasskeyUser{ID: p.UserID, Name: name, DisplayName: name, Credentials: creds}
|
||||
}
|
||||
|
||||
// handleMigrateConfirmPasskeyBegin starts a fresh passkey assertion bound to the
|
||||
// migration step-up (spec §B3, external app face). Unlike the login door it needs no
|
||||
// email — the caller is already authenticated — so it scopes the challenge to the
|
||||
@@ -331,39 +255,8 @@ func (a *API) handleMigrateConfirmPasskeyBegin(w http.ResponseWriter, r *http.Re
|
||||
if _, ok := a.requireInitiatedMigration(w, r, p.UserID); !ok {
|
||||
return
|
||||
}
|
||||
creds, err := a.Repo.PasskeyCredentialsForUser(r.Context(), p.UserID)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if len(creds) == 0 {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "no_passkey",
|
||||
"no passkey enrolled; confirm the migration with an email code"))
|
||||
return
|
||||
}
|
||||
options, sessionData, err := a.Passkey.BeginLogin(migratePasskeyUser(p, creds))
|
||||
if err != nil {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_failed",
|
||||
"could not start passkey confirmation"))
|
||||
return
|
||||
}
|
||||
id, err := newPasskeyID()
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expiresAt := a.now().Add(passkeyChallengeTTL)
|
||||
if err := a.Repo.CreatePasskeyChallenge(r.Context(), id, p.UserID, passkeyPurposeMigrate, sessionData, expiresAt); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, options)
|
||||
}
|
||||
|
||||
// migrateConfirmPasskeyFinishRequest is the assertion the browser produced, captured
|
||||
// as raw bytes so the exact response reaches the verifier without re-encoding.
|
||||
type migrateConfirmPasskeyFinishRequest struct {
|
||||
Assertion json.RawMessage `json:"assertion"`
|
||||
a.beginStepUpPasskey(w, r, p, passkeyPurposeMigrate,
|
||||
"no passkey enrolled; confirm the migration with an email code")
|
||||
}
|
||||
|
||||
// handleMigrateConfirmPasskeyFinish verifies the migration step-up assertion and, on
|
||||
@@ -375,7 +268,7 @@ func (a *API) handleMigrateConfirmPasskeyFinish(w http.ResponseWriter, r *http.R
|
||||
writeError(w, r, errPasskeyUnavailable)
|
||||
return
|
||||
}
|
||||
var req migrateConfirmPasskeyFinishRequest
|
||||
var req stepUpPasskeyFinishRequest
|
||||
if err := decodeJSON(w, r, &req); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
@@ -387,42 +280,7 @@ func (a *API) handleMigrateConfirmPasskeyFinish(w http.ResponseWriter, r *http.R
|
||||
if _, ok := a.requireInitiatedMigration(w, r, p.UserID); !ok {
|
||||
return
|
||||
}
|
||||
sessionData, err := a.Repo.ConsumePasskeyChallengeByUser(r.Context(), p.UserID, passkeyPurposeMigrate, a.now())
|
||||
if err != nil {
|
||||
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
||||
a.authFailure(r, "migrate_passkey", "challenge_invalid", nil)
|
||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||
"passkey confirmation could not be completed; begin again"))
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
creds, err := a.Repo.PasskeyCredentialsForUser(r.Context(), p.UserID)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
va, err := a.Passkey.FinishLogin(migratePasskeyUser(p, creds), sessionData, bytes.NewReader(req.Assertion))
|
||||
if err != nil {
|
||||
a.authFailure(r, "migrate_passkey", "bad_assertion", nil)
|
||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||
"passkey confirmation could not be completed; begin again"))
|
||||
return
|
||||
}
|
||||
// Same clone policy as the login door (applyAssertionCounter): a rolled-back counter
|
||||
// fails closed with the opaque envelope and advances nothing, so the migrate step-up is
|
||||
// never a weaker sibling that would accept an authenticator login refuses. A clean
|
||||
// assertion advances the stored sign-count, keeping the clone signal meaningful for the
|
||||
// next login.
|
||||
if err := a.applyAssertionCounter(r.Context(), va); err != nil {
|
||||
if errors.Is(err, errPasskeyClonedAuthenticator) {
|
||||
a.passkeyCloneRejected(r, "migrate_passkey", nil, va.CredentialID)
|
||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||
"passkey confirmation could not be completed; begin again"))
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
if !a.finishStepUpPasskey(w, r, p, passkeyPurposeMigrate, "migrate_passkey", req.Assertion) {
|
||||
return
|
||||
}
|
||||
if err := a.Repo.ConfirmMigration(r.Context(), p.UserID, "passkey", a.now()); err != nil {
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
|
||||
"felis.lolicon.best/internal/metrics"
|
||||
)
|
||||
|
||||
// The account holder's own sessions: every device signed in to the account,
|
||||
// which one is making this request, and a way to sign any of them out. The admin
|
||||
// routes in handlers_users.go read and revoke the same rows for any user.
|
||||
|
||||
// handleListMySessions lists the caller's live sessions, most recently seen
|
||||
// first, marking the one this request came in on (GET /account/sessions). A
|
||||
// caller signed in through Cloudflare Access has no session of its own, so none
|
||||
// is marked.
|
||||
func (a *API) handleListMySessions(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
sessions, err := a.Repo.ListUserSessions(r.Context(), p.UserID, a.now())
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if sessions == nil {
|
||||
sessions = []SessionView{}
|
||||
}
|
||||
if cur := callerSessionHash(r, p); cur != "" {
|
||||
for i := range sessions {
|
||||
sessions[i].Current = sessions[i].TokenHash == cur
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"sessions": sessions})
|
||||
}
|
||||
|
||||
// handleRevokeMySession signs out one of the caller's sessions
|
||||
// (DELETE /account/sessions/{hash}). A hash that is not a live session of the
|
||||
// caller is 404, whoever it belongs to. Revoking the session this request came
|
||||
// in on is a sign-out, so the cookie is cleared too.
|
||||
func (a *API) handleRevokeMySession(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
hash := r.PathValue("hash")
|
||||
if err := a.Repo.RevokeUserSession(r.Context(), p.UserID, hash); err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
writeError(w, r, newError(http.StatusNotFound, "session_not_found",
|
||||
"that session has already ended or is not one of yours"))
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
current := hash == callerSessionHash(r, p)
|
||||
if current {
|
||||
clearSessionCookie(w)
|
||||
}
|
||||
metrics.SessionsRevokedTotal.WithLabelValues("self").Inc()
|
||||
a.audit(r, "account.session.revoked", "")
|
||||
writeJSON(w, http.StatusOK, map[string]any{"ok": true, "signed_out": current})
|
||||
}
|
||||
|
||||
// handleRevokeMyOtherSessions signs out every session of the caller except the
|
||||
// one this request came in on (POST /account/sessions/revoke-others), and says
|
||||
// how many it ended.
|
||||
func (a *API) handleRevokeMyOtherSessions(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
n, err := a.Repo.RevokeOtherUserSessions(r.Context(), p.UserID, callerSessionHash(r, p))
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
metrics.SessionsRevokedTotal.WithLabelValues("self").Add(float64(n))
|
||||
a.audit(r, "account.session.revoked_others", "")
|
||||
writeJSON(w, http.StatusOK, map[string]any{"revoked": n})
|
||||
}
|
||||
|
||||
// revokeOtherSessionsAfter signs out the caller's other devices after a change
|
||||
// that retires a way in: a removed passkey, or a new verified email replacing the
|
||||
// address sign-in codes went to. A session opened with the old factor ends with
|
||||
// it. The change has already committed, so a failure here is logged and the
|
||||
// request still succeeds; answering an error would invite retrying a change that
|
||||
// took effect.
|
||||
func (a *API) revokeOtherSessionsAfter(r *http.Request, change string) {
|
||||
p := principalFromContext(r.Context())
|
||||
n, err := a.Repo.RevokeOtherUserSessions(r.Context(), p.UserID, callerSessionHash(r, p))
|
||||
if err != nil {
|
||||
log.Printf("api: sign out other sessions after %s (request_id=%s): %v", change, requestIDFromContext(r.Context()), err)
|
||||
return
|
||||
}
|
||||
if n > 0 {
|
||||
metrics.SessionsRevokedTotal.WithLabelValues("security").Add(float64(n))
|
||||
}
|
||||
}
|
||||
|
||||
// callerSessionHash is the session the request authenticated with, or "" for a
|
||||
// principal that did not come from a session cookie.
|
||||
func callerSessionHash(r *http.Request, p *Principal) string {
|
||||
if p == nil || !p.ViaSession {
|
||||
return ""
|
||||
}
|
||||
return currentSessionHash(r)
|
||||
}
|
||||
@@ -0,0 +1,324 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// sessionsFixture signs steve (u1) in on a laptop and a phone and alex (u2) on
|
||||
// one device, each by a real cookie, so the caller's own session is whichever
|
||||
// one SessionAuth resolved from the request.
|
||||
type sessionsFixture struct {
|
||||
repo *fakeRepo
|
||||
api *API
|
||||
eh http.Handler
|
||||
}
|
||||
|
||||
const (
|
||||
laptopTok = "tok-laptop"
|
||||
phoneTok = "tok-phone"
|
||||
alexTok = "tok-alex"
|
||||
)
|
||||
|
||||
func newSessionsFixture(t *testing.T) *sessionsFixture {
|
||||
t.Helper()
|
||||
repo := newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.staff["steve"] = &StaffUser{ID: "u1", Username: "steve", Email: "[email protected]", Role: "user", EmailVerified: true}
|
||||
repo.staff["alex"] = &StaffUser{ID: "u2", Username: "alex", Role: "user"}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = SessionAuth{Repo: repo, RootDomain: testRoot, Now: api.now}
|
||||
now := api.now()
|
||||
for tok, s := range map[string]*fakeSession{
|
||||
// The laptop signed in by a proving door a minute ago, so the guarded
|
||||
// changes below run without a reauth (reauth_test.go covers the gate).
|
||||
laptopTok: {userID: "u1", lastSeen: now.Add(-10 * time.Minute), userAgent: "Firefox on Linux", clientIP: "203.0.113.5", reauthAt: now.Add(-time.Minute)},
|
||||
phoneTok: {userID: "u1", lastSeen: now.Add(-2 * time.Hour), userAgent: "Safari on iPhone", clientIP: "198.51.100.7"},
|
||||
alexTok: {userID: "u2", lastSeen: now.Add(-time.Minute)},
|
||||
} {
|
||||
s.expiresAt = now.Add(time.Hour)
|
||||
repo.sessions[hashCookie(tok)] = s
|
||||
}
|
||||
return &sessionsFixture{repo: repo, api: api, eh: api.ExternalHandler()}
|
||||
}
|
||||
|
||||
func asCookie(tok string) map[string]string {
|
||||
return map[string]string{"Cookie": sessionCookieName + "=" + tok}
|
||||
}
|
||||
|
||||
func (f *sessionsFixture) revoked(tok string) bool {
|
||||
return f.repo.sessions[hashCookie(tok)].revoked
|
||||
}
|
||||
|
||||
func decodeSessions(t *testing.T, w *httptest.ResponseRecorder) []SessionView {
|
||||
t.Helper()
|
||||
var body struct {
|
||||
Sessions []SessionView `json:"sessions"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &body); err != nil {
|
||||
t.Fatalf("sessions body: %v (%s)", err, w.Body.String())
|
||||
}
|
||||
return body.Sessions
|
||||
}
|
||||
|
||||
func TestMySessionsListsOwnDevicesAndMarksThisOne(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
w := do(f.eh, "GET", "/api/v1/account/sessions", "", asCookie(phoneTok))
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("list = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
got := decodeSessions(t, w)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("listed %d sessions, want steve's 2 (alex's is not his): %+v", len(got), got)
|
||||
}
|
||||
// Most recently seen first; the phone is the device asking, though it was seen
|
||||
// less recently than the laptop until this very request.
|
||||
if got[0].TokenHash != hashCookie(phoneTok) || got[1].TokenHash != hashCookie(laptopTok) {
|
||||
t.Fatalf("order = %s, %s; want phone (just touched) then laptop", got[0].UserAgent, got[1].UserAgent)
|
||||
}
|
||||
if !got[0].Current || got[1].Current {
|
||||
t.Fatalf("current flags = %v, %v; want only the phone", got[0].Current, got[1].Current)
|
||||
}
|
||||
if got[1].UserAgent != "Firefox on Linux" || got[1].ClientIP != "203.0.113.5" {
|
||||
t.Fatalf("laptop device = %q from %q", got[1].UserAgent, got[1].ClientIP)
|
||||
}
|
||||
if strings.Count(w.Body.String(), `"current"`) != 1 {
|
||||
t.Fatalf("current must be omitted on every other session: %s", w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// A cookie that rode along beside some other credential is not the session the
|
||||
// caller signed in with, so nothing is marked current and "sign out the others"
|
||||
// keeps nothing back.
|
||||
func TestMySessionsMarkNothingWithoutASessionPrincipal(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
f.api.External = staticExternal{p: &Principal{UserID: "u1", Role: "user"}}
|
||||
eh := f.api.ExternalHandler()
|
||||
|
||||
w := do(eh, "GET", "/api/v1/account/sessions", "", asCookie(phoneTok))
|
||||
for _, s := range decodeSessions(t, w) {
|
||||
if s.Current {
|
||||
t.Fatalf("session %s marked current for an Access-signed caller", s.UserAgent)
|
||||
}
|
||||
}
|
||||
if w := do(eh, "POST", "/api/v1/account/sessions/revoke-others", "", asCookie(phoneTok)); w.Code != http.StatusOK {
|
||||
t.Fatalf("revoke-others = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if !f.revoked(phoneTok) || !f.revoked(laptopTok) {
|
||||
t.Fatal("an Access-signed caller's revoke-others must end every session")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRevokeMySessionOnlyReachesOwnSessions(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
|
||||
w := do(f.eh, "DELETE", "/api/v1/account/sessions/"+hashCookie(alexTok), "", asCookie(laptopTok))
|
||||
if w.Code != http.StatusNotFound || decodeErr(t, w) != "session_not_found" {
|
||||
t.Fatalf("revoke alex's session as steve = %d (%s), want 404 session_not_found", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(alexTok) {
|
||||
t.Fatal("steve ended alex's session")
|
||||
}
|
||||
|
||||
w = do(f.eh, "DELETE", "/api/v1/account/sessions/"+hashCookie(phoneTok), "", asCookie(laptopTok))
|
||||
if w.Code != http.StatusOK || !strings.Contains(w.Body.String(), `"signed_out":false`) {
|
||||
t.Fatalf("revoke phone from laptop = %d (%s), want 200 signed_out:false", w.Code, w.Body.String())
|
||||
}
|
||||
if !f.revoked(phoneTok) || f.revoked(laptopTok) {
|
||||
t.Fatal("want the phone ended and the laptop still signed in")
|
||||
}
|
||||
if c := w.Header().Get("Set-Cookie"); c != "" {
|
||||
t.Fatalf("ending another device must leave this one's cookie alone, got Set-Cookie %q", c)
|
||||
}
|
||||
if last := f.repo.audits[len(f.repo.audits)-1]; last.Action != "account.session.revoked" || last.ActorUserID != "u1" {
|
||||
t.Fatalf("audit = %+v", last)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRevokingThisSessionSignsOut(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
w := do(f.eh, "DELETE", "/api/v1/account/sessions/"+hashCookie(laptopTok), "", asCookie(laptopTok))
|
||||
if w.Code != http.StatusOK || !strings.Contains(w.Body.String(), `"signed_out":true`) {
|
||||
t.Fatalf("revoke own session = %d (%s), want 200 signed_out:true", w.Code, w.Body.String())
|
||||
}
|
||||
if c := w.Header().Get("Set-Cookie"); !strings.HasPrefix(c, sessionCookieName+"=;") || !strings.Contains(c, "Max-Age=0") {
|
||||
t.Fatalf("Set-Cookie = %q, want the session cookie cleared", c)
|
||||
}
|
||||
if w := do(f.eh, "GET", "/api/v1/account/sessions", "", asCookie(laptopTok)); w.Code != http.StatusUnauthorized {
|
||||
t.Fatalf("the ended session still authenticates: %d", w.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRevokeOtherSessionsKeepsThisOne(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
w := do(f.eh, "POST", "/api/v1/account/sessions/revoke-others", "", asCookie(laptopTok))
|
||||
if w.Code != http.StatusOK || strings.TrimSpace(w.Body.String()) != `{"revoked":1}` {
|
||||
t.Fatalf("revoke-others = %d (%s), want 200 {\"revoked\":1}", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(laptopTok) || !f.revoked(phoneTok) || f.revoked(alexTok) {
|
||||
t.Fatalf("after revoke-others laptop=%v phone=%v alex=%v, want only the phone ended",
|
||||
f.revoked(laptopTok), f.revoked(phoneTok), f.revoked(alexTok))
|
||||
}
|
||||
}
|
||||
|
||||
// The admin route names the user in its path; a hash of someone else's session
|
||||
// under it must not end that session.
|
||||
func TestAdminRevokeSessionChecksWhoseItIs(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
f.api.External = staticExternal{p: &Principal{UserID: "own", Role: "owner", ViaAdminAccess: true}}
|
||||
eh := f.api.ExternalHandler()
|
||||
|
||||
w := do(eh, "DELETE", "/api/v1/users/u1/sessions/"+hashCookie(alexTok), "", nil)
|
||||
if w.Code != http.StatusNotFound || decodeErr(t, w) != "session_not_found" {
|
||||
t.Fatalf("alex's hash under steve = %d (%s), want 404 session_not_found", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(alexTok) {
|
||||
t.Fatal("a hash under another user's path ended alex's session")
|
||||
}
|
||||
if w := do(eh, "DELETE", "/api/v1/users/u2/sessions/"+hashCookie(alexTok), "", nil); w.Code != http.StatusOK {
|
||||
t.Fatalf("alex's hash under alex = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if !f.revoked(alexTok) {
|
||||
t.Fatal("the matching revoke did not end the session")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemovingAPasskeySignsOutOtherDevices(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
f.repo.passkeyCreds["a"] = PasskeyCredential{ID: "a", UserID: "u1", CredentialID: "c-a", CreatedAt: frozenNow}
|
||||
|
||||
if w := do(f.eh, "DELETE", "/api/v1/account/passkey/credentials/a", "", asCookie(laptopTok)); w.Code != http.StatusNoContent {
|
||||
t.Fatalf("delete passkey = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(laptopTok) || !f.revoked(phoneTok) || f.revoked(alexTok) {
|
||||
t.Fatalf("after passkey removal laptop=%v phone=%v alex=%v, want only the phone ended",
|
||||
f.revoked(laptopTok), f.revoked(phoneTok), f.revoked(alexTok))
|
||||
}
|
||||
}
|
||||
|
||||
func TestVerifyingANewEmailSignsOutOtherDevices(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
mailer := &captureMailer{}
|
||||
f.api.Mailer = mailer
|
||||
hdr := asCookie(laptopTok)
|
||||
hdr["Content-Type"] = "application/json"
|
||||
|
||||
if w := do(f.eh, "POST", "/api/v1/account/email/start", `{"email":"[email protected]"}`, hdr); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(phoneTok) {
|
||||
t.Fatal("asking for a code must not sign anything out yet")
|
||||
}
|
||||
if w := do(f.eh, "POST", "/api/v1/account/email/verify", `{"code":"`+mailer.code+`"}`, hdr); w.Code != http.StatusOK {
|
||||
t.Fatalf("verify = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(laptopTok) || !f.revoked(phoneTok) || f.revoked(alexTok) {
|
||||
t.Fatalf("after email change laptop=%v phone=%v alex=%v, want only the phone ended",
|
||||
f.revoked(laptopTok), f.revoked(phoneTok), f.revoked(alexTok))
|
||||
}
|
||||
}
|
||||
|
||||
// With no verified address before, no session was opened through one, so a
|
||||
// first verification signs nothing out; nor does proving the same address again.
|
||||
func TestVerifyingAFirstOrSameEmailKeepsOtherDevices(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
verified bool
|
||||
address string
|
||||
}{
|
||||
{"first address", false, "[email protected]"},
|
||||
{"same address again", true, "[email protected]"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
f.repo.staff["steve"].EmailVerified = tc.verified
|
||||
mailer := &captureMailer{}
|
||||
f.api.Mailer = mailer
|
||||
hdr := asCookie(laptopTok)
|
||||
hdr["Content-Type"] = "application/json"
|
||||
if w := do(f.eh, "POST", "/api/v1/account/email/start", `{"email":"`+tc.address+`"}`, hdr); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := do(f.eh, "POST", "/api/v1/account/email/verify", `{"code":"`+mailer.code+`"}`, hdr); w.Code != http.StatusOK {
|
||||
t.Fatalf("verify = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if f.revoked(phoneTok) {
|
||||
t.Fatal("the phone was signed out")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The change has committed by the time the other sessions are signed out; a
|
||||
// failure there must not report the change itself as failed.
|
||||
func TestPasskeyRemovalSucceedsWhenSigningOutOthersFails(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
f.repo.passkeyCreds["a"] = PasskeyCredential{ID: "a", UserID: "u1", CredentialID: "c-a", CreatedAt: frozenNow}
|
||||
f.repo.failRevokeOthers = errors.New("db blip")
|
||||
|
||||
if w := do(f.eh, "DELETE", "/api/v1/account/passkey/credentials/a", "", asCookie(laptopTok)); w.Code != http.StatusNoContent {
|
||||
t.Fatalf("delete passkey = %d (%s), want 204 despite the sign-out failure", w.Code, w.Body.String())
|
||||
}
|
||||
if _, kept := f.repo.passkeyCreds["a"]; kept {
|
||||
t.Fatal("the passkey must still be removed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSessionActivityIsRecordedAtMostOnceAMinute(t *testing.T) {
|
||||
f := newSessionsFixture(t)
|
||||
now := f.api.now()
|
||||
laptop := f.repo.sessions[hashCookie(laptopTok)]
|
||||
|
||||
laptop.lastSeen = now.Add(-30 * time.Second)
|
||||
do(f.eh, "GET", "/api/v1/me", "", asCookie(laptopTok))
|
||||
if laptop.touches != 0 {
|
||||
t.Fatalf("a session seen 30s ago was touched %d times, want 0", laptop.touches)
|
||||
}
|
||||
|
||||
laptop.lastSeen = now.Add(-2 * time.Minute)
|
||||
do(f.eh, "GET", "/api/v1/me", "", asCookie(laptopTok))
|
||||
if laptop.touches != 1 || !laptop.lastSeen.Equal(now) {
|
||||
t.Fatalf("a session seen 2m ago: touches=%d lastSeen=%v, want 1 touch to %v", laptop.touches, laptop.lastSeen, now)
|
||||
}
|
||||
|
||||
// A failed touch leaves the request authenticated.
|
||||
laptop.lastSeen = now.Add(-2 * time.Minute)
|
||||
f.repo.failTouchSession = errors.New("db blip")
|
||||
if w := do(f.eh, "GET", "/api/v1/me", "", asCookie(laptopTok)); w.Code != http.StatusOK {
|
||||
t.Fatalf("/me with a failing touch = %d, want 200", w.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSignInRecordsTheDevice(t *testing.T) {
|
||||
api, repo, mailer := seedLoginEmailAPI(t)
|
||||
api.ClientIPHeader = "CF-Connecting-IP"
|
||||
eh := api.ExternalHandler()
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start = %d", w.Code)
|
||||
}
|
||||
ua := "Mozilla/5.0 x" + strings.Repeat("é", 200) // 413 bytes, é two each
|
||||
w := do(eh, "POST", "/api/v1/auth/email/verify", `{"email":"[email protected]","code":"`+mailer.code+`"}`, map[string]string{
|
||||
"Content-Type": "application/json", "User-Agent": ua, "CF-Connecting-IP": "2001:db8::7",
|
||||
})
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("verify = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if len(repo.sessions) != 1 {
|
||||
t.Fatalf("sessions = %d, want 1", len(repo.sessions))
|
||||
}
|
||||
for _, s := range repo.sessions {
|
||||
if s.clientIP != "2001:db8::7" {
|
||||
t.Errorf("client_ip = %q, want the edge's 2001:db8::7", s.clientIP)
|
||||
}
|
||||
// 13 bytes of prefix leave 243 for é: byte 256 falls inside the 122nd, so
|
||||
// the cut keeps 121 of them, 255 bytes.
|
||||
if want := "Mozilla/5.0 x" + strings.Repeat("é", 121); s.userAgent != want {
|
||||
t.Errorf("user_agent = %d bytes %q, want the first 256 bytes on a rune boundary", len(s.userAgent), s.userAgent)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -218,7 +218,7 @@ func TestLinkVerifyRejections(t *testing.T) {
|
||||
t.Run("uuid linked to another user -> 409 already_linked", func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
repo.links[mcUUID] = "someone-else"
|
||||
repo.linkCodes["FRESHCOD"] = fakeLinkCode{mcUUID: mcUUID, expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
repo.linkCodes["FRESHCOD"] = fakeLinkCode{mcUUID: mcUUID, authSource: "mojang", expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
w := do(mk(repo), "POST", "/api/v1/account/link/verify", `{"code":"FRESHCOD"}`, nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "already_linked" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
@@ -238,7 +238,7 @@ func TestLinkVerifyIdempotent(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
repo.links[mcUUID] = "u1" // already linked to THIS user
|
||||
repo.linked["u1"] = true
|
||||
repo.linkCodes["REVERIFYX"] = fakeLinkCode{mcUUID: mcUUID, expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
repo.linkCodes["REVERIFYX"] = fakeLinkCode{mcUUID: mcUUID, authSource: "mojang", expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: user}
|
||||
@@ -276,7 +276,7 @@ func TestLinkVerifyTakesOverDeletedLinkOnly(t *testing.T) {
|
||||
t.Fatalf("UserByMCUUID(deleted link) = %v, want ErrNotFound (no in-game standing)", err)
|
||||
}
|
||||
|
||||
repo.linkCodes["TAKEOVER"] = fakeLinkCode{mcUUID: mcGone, expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
repo.linkCodes["TAKEOVER"] = fakeLinkCode{mcUUID: mcGone, authSource: "mojang", expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: user}
|
||||
if w := do(api.ExternalHandler(), "POST", "/api/v1/account/link/verify", `{"code":"TAKEOVER"}`, nil); w.Code != http.StatusOK {
|
||||
|
||||
@@ -60,6 +60,11 @@ type loginEmailStartRequest struct {
|
||||
// no code minted: the response never distinguishes the two, and the reservation is
|
||||
// kept on that path too so repeated probing of one address is throttled identically
|
||||
// to repeated sends.
|
||||
//
|
||||
// A start never cancels the codes already mailed (AddLoginEmailOTP keeps the newest
|
||||
// otpLiveLoginCodes live), and a start inside the cooldown answers the same 202 without
|
||||
// minting: the code mailed moments ago is still good. So anyone who knows an address
|
||||
// can only add codes to its owner's inbox, never keep the owner from signing in.
|
||||
func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
||||
if !localAuthEnabled(r.Context(), a.Repo) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "local_auth_disabled",
|
||||
@@ -98,8 +103,11 @@ func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
||||
lim := a.otpLimiter()
|
||||
emailAt, ok := lim.reserve(emailKey, otpResendCooldown)
|
||||
if !ok {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_resend_cooldown",
|
||||
"a code was sent recently; wait a moment before requesting another"))
|
||||
// A start for this address went through less than a cooldown ago, and the code
|
||||
// it mailed (if the address has an account) is still live. Answer as that start
|
||||
// did, expiry included, and mail nothing: the owner — or whoever typed the
|
||||
// address — lands on the code screen and the code already in the inbox works.
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": emailAt.Add(otpTTL).UTC()})
|
||||
return
|
||||
}
|
||||
committed := false
|
||||
@@ -109,16 +117,17 @@ func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}()
|
||||
|
||||
// Compute the expiry once so the neutral (no-account) branch and the real-send
|
||||
// branch return byte-identical bodies.
|
||||
expiresAt := a.now().Add(otpTTL)
|
||||
// Compute the expiry once, from the reservation, so the neutral (no-account)
|
||||
// branch, the real-send branch and a start inside the window all return
|
||||
// byte-identical bodies.
|
||||
expiresAt := emailAt.Add(otpTTL)
|
||||
|
||||
u, err := a.Repo.UserByEmail(r.Context(), email)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
// No verified account for this address. Return the same 202 as a real send
|
||||
// (no code minted) and KEEP the reservation, so probing an unknown address is
|
||||
// throttled exactly like resending to a known one — the throttle reveals
|
||||
// (no code minted) and KEEP the reservation, so a probe of an unknown address
|
||||
// holds the window exactly like a send to a known one — the window reveals
|
||||
// nothing, and the accepted /auth/options oracle is where existence is learnt.
|
||||
committed = true
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{"sent": true, "expires_at": expiresAt.UTC()})
|
||||
@@ -158,7 +167,7 @@ func (a *API) handleLoginEmailStart(w http.ResponseWriter, r *http.Request) {
|
||||
// record. The login redeem (ConsumeLoginEmailOTP) never reads or writes this
|
||||
// address, so the stored casing is authoritative and the row's email snapshot is
|
||||
// purely for the audit trail.
|
||||
if err := a.Repo.CreateEmailOTP(r.Context(), id, u.ID, u.Email, otpCodeHash(code), otpPurposeLogin, expiresAt); err != nil {
|
||||
if err := a.Repo.AddLoginEmailOTP(r.Context(), id, u.ID, u.Email, otpCodeHash(code), otpPurposeLogin, a.now(), expiresAt); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
@@ -271,17 +280,10 @@ func (a *API) handleLoginEmailVerify(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
token, err := newSessionToken()
|
||||
if err != nil {
|
||||
if err := a.startSession(w, r, u.ID, provenSignIn); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expires := a.now().Add(sessionTTL)
|
||||
if err := a.Repo.CreateSession(r.Context(), hashCookie(token), u.ID, expires); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
setSessionCookie(w, token, expires)
|
||||
a.auditAccount(r, u, "auth.login_email", "")
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"user_id": u.ID,
|
||||
|
||||
@@ -152,8 +152,7 @@ func TestLoginEmailVertical(t *testing.T) {
|
||||
// TestLoginEmailStartNeutralOnUnknownAddress pins the start-side anti-enumeration
|
||||
// contract: an address with no verified account yields a 202 BYTE-IDENTICAL to a
|
||||
// real send (frozen clock ⇒ same expires_at), mints and mails nothing, audits
|
||||
// nothing — and still burns the cooldown window, so probing is throttled exactly
|
||||
// like sending.
|
||||
// nothing — and a re-probe inside the cooldown answers the same 202 a resend does.
|
||||
func TestLoginEmailStartNeutralOnUnknownAddress(t *testing.T) {
|
||||
// A real send for comparison.
|
||||
apiK, _, _ := seedLoginEmailAPI(t)
|
||||
@@ -183,12 +182,14 @@ func TestLoginEmailStartNeutralOnUnknownAddress(t *testing.T) {
|
||||
t.Errorf("neutral path must mint/mail/audit nothing, got otps=%d mails=%d audits=%d",
|
||||
len(repoU.otps), mailerU.calls, len(repoU.audits))
|
||||
}
|
||||
// The reservation is KEPT on the neutral path: re-probing the same unknown
|
||||
// address inside the window is throttled identically to a resend.
|
||||
if w := do(ehU, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "otp_resend_cooldown" {
|
||||
t.Fatalf("re-probe of unknown address: code = %d body %s, want 429 otp_resend_cooldown",
|
||||
// Re-probing inside the window answers exactly as the first probe did.
|
||||
if w := do(ehU, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted || w.Body.String() != wK.Body.String() {
|
||||
t.Fatalf("re-probe of unknown address: code = %d body %s, want the same 202 as a real send",
|
||||
w.Code, w.Body.String())
|
||||
}
|
||||
if len(repoU.otps) != 0 || mailerU.calls != 0 {
|
||||
t.Errorf("re-probe must mint/mail nothing, got otps=%d mails=%d", len(repoU.otps), mailerU.calls)
|
||||
}
|
||||
|
||||
// An UNVERIFIED account is indistinguishable from no account: UserByEmail only
|
||||
// resolves proven addresses, so the door never mails one nobody controls.
|
||||
@@ -278,13 +279,14 @@ func TestLoginEmailGates(t *testing.T) {
|
||||
// TestLoginEmailStartRateLimited closes the unauthenticated email-bomb vector on the
|
||||
// public door: one send per recipient per window, keyed case-insensitively, and
|
||||
// namespaced apart from the authenticated onboarding throttle so neither door can
|
||||
// starve the other.
|
||||
// starve the other. A start inside the window is answered like the one that sent,
|
||||
// so whoever asks lands on the code screen with the mail already in the inbox.
|
||||
func TestLoginEmailStartRateLimited(t *testing.T) {
|
||||
start := func(eh http.Handler, email string) *httptest.ResponseRecorder {
|
||||
return do(eh, "POST", "/api/v1/auth/email/start", `{"email":"`+email+`"}`, jsonHeader)
|
||||
}
|
||||
|
||||
t.Run("same recipient is throttled, then recovers after the cooldown", func(t *testing.T) {
|
||||
t.Run("a start inside the window mails nothing and keeps the first code", func(t *testing.T) {
|
||||
api, repo, mailer := seedLoginEmailAPI(t)
|
||||
clock := time.Unix(1_700_000_000, 0)
|
||||
api.Now = func() time.Time { return clock }
|
||||
@@ -293,28 +295,41 @@ func TestLoginEmailStartRateLimited(t *testing.T) {
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("first send: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "otp_resend_cooldown" {
|
||||
t.Fatalf("immediate resend: code = %d body %s, want 429 otp_resend_cooldown", w.Code, w.Body.String())
|
||||
code := mailer.code
|
||||
clock = clock.Add(30 * time.Second)
|
||||
w := start(eh, "[email protected]")
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("resend inside the window: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
// The expiry is the first code's, the one the inbox holds.
|
||||
if b := acctBody(t, w); b["sent"] != true || b["expires_at"] != "2023-11-14T22:23:20Z" {
|
||||
t.Errorf("resend body = %v, want sent:true expires_at:2023-11-14T22:23:20Z", b)
|
||||
}
|
||||
if mailer.calls != 1 || len(repo.otps) != 1 {
|
||||
t.Errorf("throttled resend must not mint or mail: mails=%d otps=%d, want 1/1",
|
||||
t.Errorf("resend inside the window must not mint or mail: mails=%d otps=%d, want 1/1",
|
||||
mailer.calls, len(repo.otps))
|
||||
}
|
||||
clock = clock.Add(otpResendCooldown + time.Second)
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("post-cooldown send: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/verify",
|
||||
`{"email":"[email protected]","code":"`+code+`"}`, jsonHeader); w.Code != http.StatusOK {
|
||||
t.Fatalf("first code after a resend: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
clock = clock.Add(otpResendCooldown)
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusAccepted || mailer.calls != 2 {
|
||||
t.Fatalf("post-cooldown send: code = %d mails = %d, want 202 and a second mail (%s)",
|
||||
w.Code, mailer.calls, w.Body.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("throttle key is case-insensitive", func(t *testing.T) {
|
||||
api, _, _ := seedLoginEmailAPI(t)
|
||||
api, _, mailer := seedLoginEmailAPI(t)
|
||||
eh := api.ExternalHandler()
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("first send: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
// A recased retype is the same mailbox: it must hit the same window.
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusTooManyRequests {
|
||||
t.Fatalf("recased resend: code = %d, want 429 (key must be lowercased)", w.Code)
|
||||
if w := start(eh, "[email protected]"); w.Code != http.StatusAccepted || mailer.calls != 1 {
|
||||
t.Fatalf("recased resend: code = %d mails = %d, want 202 and no second mail (key must be lowercased)",
|
||||
w.Code, mailer.calls)
|
||||
}
|
||||
})
|
||||
|
||||
@@ -339,6 +354,95 @@ func TestLoginEmailStartRateLimited(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
// TestLoginStartsInsideTheWindowMatchExactly: with a clock that moves on every read,
|
||||
// a start inside the cooldown still answers with the very expires_at the first start
|
||||
// returned, on both email doors and for known and unknown addresses alike, so a
|
||||
// repeat start is indistinguishable from the first down to the nanosecond.
|
||||
func TestLoginStartsInsideTheWindowMatchExactly(t *testing.T) {
|
||||
ticking := func(api *API) {
|
||||
clock := time.Unix(1_700_000_000, 0)
|
||||
api.Now = func() time.Time {
|
||||
clock = clock.Add(time.Millisecond)
|
||||
return clock
|
||||
}
|
||||
}
|
||||
expiry := func(t *testing.T, w *httptest.ResponseRecorder) string {
|
||||
t.Helper()
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
e, _ := acctBody(t, w)["expires_at"].(string)
|
||||
return e
|
||||
}
|
||||
for _, email := range []string{"[email protected]", "[email protected]"} {
|
||||
api, _, _ := seedLoginEmailAPI(t)
|
||||
ticking(api)
|
||||
eh := api.ExternalHandler()
|
||||
start := func() *httptest.ResponseRecorder {
|
||||
return do(eh, "POST", "/api/v1/auth/email/start", `{"email":"`+email+`"}`, jsonHeader)
|
||||
}
|
||||
first, again := expiry(t, start()), expiry(t, start())
|
||||
if first != "2023-11-14T22:23:20.001Z" || again != first {
|
||||
t.Errorf("email door %s: expires_at %q then %q, want 2023-11-14T22:23:20.001Z twice", email, first, again)
|
||||
}
|
||||
}
|
||||
for _, email := range []string{"[email protected]", "[email protected]"} {
|
||||
api, _, _ := seedOpLoginAPI(t)
|
||||
ticking(api)
|
||||
eh := api.ExternalHandler()
|
||||
first, again := expiry(t, startOp(eh, email)), expiry(t, startOp(eh, email))
|
||||
if first != "2023-11-14T22:23:20.001Z" || again != first {
|
||||
t.Errorf("op door %s: expires_at %q then %q, want 2023-11-14T22:23:20.001Z twice", email, first, again)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoginEmailStartKeepsEarlierCodes is the stranger-keeps-starting case: someone
|
||||
// who knows the address starts a login every cooldown. Each start adds a code to the
|
||||
// owner's inbox and never cancels one, so the owner's own code keeps working until
|
||||
// otpLiveLoginCodes newer ones exist; signing in spends every code still out.
|
||||
func TestLoginEmailStartKeepsEarlierCodes(t *testing.T) {
|
||||
api, repo, mailer := seedLoginEmailAPI(t)
|
||||
clock := time.Unix(1_700_000_000, 0)
|
||||
api.Now = func() time.Time { return clock }
|
||||
eh := api.ExternalHandler()
|
||||
start := func() string {
|
||||
t.Helper()
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
code := mailer.code
|
||||
clock = clock.Add(otpResendCooldown)
|
||||
return code
|
||||
}
|
||||
verify := func(code string) *httptest.ResponseRecorder {
|
||||
return do(eh, "POST", "/api/v1/auth/email/verify",
|
||||
`{"email":"[email protected]","code":"`+code+`"}`, jsonHeader)
|
||||
}
|
||||
|
||||
owner := start()
|
||||
second := start()
|
||||
third := start()
|
||||
if mailer.calls != 3 || len(repo.otps) != 3 {
|
||||
t.Fatalf("mails=%d otps=%d, want 3/3 (a start must not cancel earlier codes)", mailer.calls, len(repo.otps))
|
||||
}
|
||||
fourth := start()
|
||||
if len(repo.otps) != 3 {
|
||||
t.Fatalf("otps = %d after a fourth start, want 3 (the oldest goes)", len(repo.otps))
|
||||
}
|
||||
if w := verify(owner); w.Code != http.StatusBadRequest || decodeErr(t, w) != "invalid_code" {
|
||||
t.Fatalf("code with three newer ones: code = %d body %s, want 400 invalid_code", w.Code, w.Body.String())
|
||||
}
|
||||
if w := verify(second); w.Code != http.StatusOK {
|
||||
t.Fatalf("second code while two newer are live: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
for name, code := range map[string]string{"third": third, "fourth": fourth} {
|
||||
if w := verify(code); w.Code != http.StatusBadRequest || decodeErr(t, w) != "invalid_code" {
|
||||
t.Errorf("%s code after the sign-in: code = %d body %s, want 400 invalid_code", name, w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoginEmailVerifyRejections is the redeem-side failure matrix. The anchor case
|
||||
// is uniformity: an unknown address and a wrong code for a known address answer with
|
||||
// the same (code, message) envelope, so the verify half never doubles as an
|
||||
@@ -711,3 +815,33 @@ func TestDeadAccountsCannotLogInOrKeepSessions(t *testing.T) {
|
||||
t.Error("refused redeem consumed the code; re-enabling the account must stay retryable within TTL")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPublicMailDoorsWithoutRelay: with no [smtp] relay both public doors that mail
|
||||
// a code answer 503 mail_unavailable before the address is looked up, so a known
|
||||
// address, a staff address and an unknown one get the same answer, no code or
|
||||
// op-login request is minted, and no cooldown is spent for when a relay is added.
|
||||
func TestPublicMailDoorsWithoutRelay(t *testing.T) {
|
||||
api, repo, _ := seedLoginEmailAPI(t)
|
||||
repo.staff["op"] = &StaffUser{ID: "a1", Username: "op", Email: "[email protected]", Role: "admin", EmailVerified: true}
|
||||
api.Mailer = nil
|
||||
eh := api.ExternalHandler()
|
||||
|
||||
for _, door := range []string{"/api/v1/auth/email/start", "/api/v1/auth/op-login/start"} {
|
||||
for _, email := range []string{"[email protected]", "[email protected]", "[email protected]"} {
|
||||
w := do(eh, "POST", door, `{"email":"`+email+`"}`, jsonHeader)
|
||||
code, msg := errEnvelope(t, w)
|
||||
if w.Code != http.StatusServiceUnavailable || code != "mail_unavailable" ||
|
||||
msg != "this server has no mail relay configured, so it cannot send codes; sign in with a passkey or ask the server operator to set up email" {
|
||||
t.Errorf("%s %s = %d %s, want 503 mail_unavailable", door, email, w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(repo.otps) != 0 || len(repo.opLogins) != 0 {
|
||||
t.Fatalf("refused starts minted %d codes and %d op-login requests", len(repo.otps), len(repo.opLogins))
|
||||
}
|
||||
|
||||
api.Mailer = &captureMailer{}
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start once a relay is wired = %d (%s), want 202", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
@@ -14,8 +14,9 @@ import (
|
||||
// It is the deliberate counter-slice to the anti-enumeration login doors
|
||||
// (handlers_auth_email.go, handlers_passkey.go): those refuse to disclose whether an
|
||||
// address has an account precisely because THIS endpoint is the one sanctioned place
|
||||
// existence is revealed. An empty methods array means "no (verified) account". That
|
||||
// makes it a mass-enumeration surface by design — an accepted product decision, the
|
||||
// existence is revealed. An empty methods array means "no (verified) account, or
|
||||
// none of its methods is available on this install". That makes it a
|
||||
// mass-enumeration surface by design — an accepted product decision, the
|
||||
// same one the email door's header records. The handler sends no mail and mutates
|
||||
// nothing, so a per-recipient cooldown would merely block a legitimate retry; what
|
||||
// bounds enumeration is the per-client-address token bucket shared by every public
|
||||
@@ -87,8 +88,12 @@ func (a *API) handleAuthOptions(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}
|
||||
// Email-OTP login works for any resolved verified account (UserByEmail resolves only
|
||||
// email_verified rows), so it is always on offer.
|
||||
methods = append(methods, "email_otp")
|
||||
// email_verified rows), so it is on offer whenever a relay can mail the code; with
|
||||
// none the email door answers 503 mail_unavailable, so it is left out like an
|
||||
// unwired passkey verifier.
|
||||
if a.Mailer != nil {
|
||||
methods = append(methods, "email_otp")
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]any{"methods": methods})
|
||||
}
|
||||
@@ -22,7 +22,8 @@ import (
|
||||
// door would immediately 503.
|
||||
|
||||
// seedAuthOptionsAPI wires the discovery door: local sessions enabled, a verified player
|
||||
// (u1) and a verified staff account (a1), and a passkey verifier wired by default.
|
||||
// (u1) and a verified staff account (a1), and a passkey verifier and a mail relay wired
|
||||
// by default.
|
||||
// Callers seed passkey credentials per-test to set the credential state.
|
||||
func seedAuthOptionsAPI(t *testing.T) (*API, *fakeRepo) {
|
||||
t.Helper()
|
||||
@@ -32,6 +33,7 @@ func seedAuthOptionsAPI(t *testing.T) (*API, *fakeRepo) {
|
||||
repo.staff["boss"] = &StaffUser{ID: "a1", Username: "boss", Email: "[email protected]", Role: "admin", EmailVerified: true}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.Passkey = &fakePasskeyVerifier{}
|
||||
api.Mailer = &captureMailer{}
|
||||
return api, repo
|
||||
}
|
||||
|
||||
@@ -128,6 +130,22 @@ func TestAuthOptionsDoesNotRevealStaffness(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestAuthOptionsEmailRequiresMailRelay: with no [smtp] relay the email door answers
|
||||
// 503 mail_unavailable, so options leaves email_otp out; an account with a passkey is
|
||||
// still offered it, and one without is offered nothing.
|
||||
func TestAuthOptionsEmailRequiresMailRelay(t *testing.T) {
|
||||
api, repo := seedAuthOptionsAPI(t)
|
||||
api.Mailer = nil
|
||||
repo.passkeyCreds["a"] = PasskeyCredential{ID: "a", UserID: "a1", CredentialID: "c-a1", PublicKey: "k", CreatedAt: frozenNow}
|
||||
eh := api.ExternalHandler()
|
||||
if w := do(eh, "POST", authOptionsPath, `{"email":"[email protected]"}`, jsonHeader); w.Body.String() != `{"methods":["passkey"]}`+"\n" {
|
||||
t.Errorf("passkey account body = %q, want only passkey", w.Body.String())
|
||||
}
|
||||
if w := do(eh, "POST", authOptionsPath, `{"email":"[email protected]"}`, jsonHeader); w.Body.String() != `{"methods":[]}`+"\n" {
|
||||
t.Errorf("email-only account body = %q, want no methods", w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestAuthOptionsPasskeyRequiresWiredVerifier: the account HAS an enrolled passkey, but
|
||||
// no verifier is wired (a.Passkey == nil). Both login halves 503 passkey_unavailable in
|
||||
// that state, so options must NOT advertise passkey — it would be a dead offer.
|
||||
|
||||
@@ -272,7 +272,7 @@ func TestBackupNow(t *testing.T) {
|
||||
// the stopped-gate / 503 / async-202 behaviour is proven there; here the focus is the
|
||||
// internal-face difference: no Principal (service-token auth), no owner gate — even a
|
||||
// server owned by someone else backs up (the on-node operator is trusted) — and the
|
||||
// audit is attributed to "break-glass"/"internal", not an email/"external".
|
||||
// audit is attributed to "break-glass"/"internal:ops", not an email/"external".
|
||||
func TestInternalBackup(t *testing.T) {
|
||||
mk := func() (*API, *fakeRepo, *fakeCluster, *fakeBackuper) {
|
||||
repo := newFakeRepo()
|
||||
@@ -282,6 +282,7 @@ func TestInternalBackup(t *testing.T) {
|
||||
Ready: false, DesiredState: string(v1alpha1.DesiredStopped)}
|
||||
backuper := &fakeBackuper{}
|
||||
api := newTestAPI(repo, cl)
|
||||
api.Internal = okInternal{caller: CallerOps}
|
||||
api.Backuper = backuper
|
||||
return api, repo, cl, backuper
|
||||
}
|
||||
@@ -301,7 +302,7 @@ func TestInternalBackup(t *testing.T) {
|
||||
backuper.calls, backuper.gotName, backuper.gotFormerOwn)
|
||||
}
|
||||
if len(repo.audits) != 1 || repo.audits[0].Action != "backup.create" ||
|
||||
repo.audits[0].Actor != "break-glass" || repo.audits[0].Source != "internal" {
|
||||
repo.audits[0].Actor != "break-glass" || repo.audits[0].Source != "internal:ops" {
|
||||
t.Fatalf("audit not attributed to break-glass/internal: %+v", repo.audits)
|
||||
}
|
||||
})
|
||||
@@ -312,7 +313,7 @@ func TestInternalBackup(t *testing.T) {
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if len(repo.audits) != 1 || repo.audits[0].Actor != "alice" || repo.audits[0].Source != "internal" {
|
||||
if len(repo.audits) != 1 || repo.audits[0].Actor != "alice" || repo.audits[0].Source != "internal:ops" {
|
||||
t.Fatalf("audit actor should be the os_user, not break-glass: %+v", repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -29,17 +30,39 @@ func errNoWorldVolume() error {
|
||||
// query runs (AllBackups vs BackupsForUser) — there is no client-supplied filter
|
||||
// a user could widen, so "a user cannot see another's backups" is a property of
|
||||
// the query, not of request parsing.
|
||||
//
|
||||
// What the request may choose is the page: ?server= narrows the list to one
|
||||
// server (inside the caller's scope, never beyond it), ?limit= and ?offset= page
|
||||
// it, and the answer carries how many match in all.
|
||||
func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
q := r.URL.Query()
|
||||
opts := BackupListOpts{Server: q.Get("server")}
|
||||
// Format only: a system server's reserved name is still a server whose
|
||||
// backups an admin may list.
|
||||
if opts.Server != "" {
|
||||
if err := naming.ValidateSystemServerName(opts.Server); err != nil {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||
return
|
||||
}
|
||||
}
|
||||
opts.Limit, _ = strconv.Atoi(q.Get("limit"))
|
||||
opts.Offset, _ = strconv.Atoi(q.Get("offset"))
|
||||
if opts.Limit <= 0 {
|
||||
opts.Limit = DefaultBackupListLimit
|
||||
}
|
||||
opts.Limit = min(opts.Limit, MaxBackupListLimit)
|
||||
opts.Offset = max(opts.Offset, 0)
|
||||
|
||||
var (
|
||||
backups []BackupView
|
||||
total int
|
||||
err error
|
||||
)
|
||||
if p.IsAdmin() {
|
||||
backups, err = a.Repo.AllBackups(r.Context())
|
||||
backups, total, err = a.Repo.AllBackups(r.Context(), opts)
|
||||
} else {
|
||||
backups, err = a.Repo.BackupsForUser(r.Context(), p.UserID)
|
||||
backups, total, err = a.Repo.BackupsForUser(r.Context(), p.UserID, opts)
|
||||
}
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
@@ -48,7 +71,7 @@ func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
|
||||
if backups == nil {
|
||||
backups = []BackupView{}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"backups": backups})
|
||||
writeJSON(w, http.StatusOK, map[string]any{"backups": backups, "total": total})
|
||||
}
|
||||
|
||||
// handleRestoreBackup starts restoring a server's world from a backup (spec §7
|
||||
@@ -76,8 +99,14 @@ func (a *API) handleListBackups(w http.ResponseWriter, r *http.Request) {
|
||||
// makes that atomic against a wake and refuses a second restore, backup or
|
||||
// file write on the same world with 409 maintenance_in_progress until the
|
||||
// restore Job finishes.
|
||||
// ⑧ hand off to the Restorer. Restore is asynchronous (a restore Job, like an
|
||||
// image build Job), so success means "enqueued" and the handler answers 202.
|
||||
// ⑧ hand off. By default the restore starts with a safety snapshot of the world
|
||||
// as it is (restorechain.go): a pre_restore backup Job that the restore Job
|
||||
// follows once it succeeds, so a wrong pick can be walked back from the
|
||||
// backup list. "safety_snapshot": false in the body restores straight away.
|
||||
// Either way the work is asynchronous (Jobs, like an image build), so
|
||||
// success means "enqueued" and the handler answers 202, saying in
|
||||
// safety_snapshot which of the two it did. A restore of another backup still
|
||||
// running on the world is 409 restore_in_progress.
|
||||
//
|
||||
// The opaque backup_ref is resolved server-side from the backup and handed to the
|
||||
// Restorer directly; the client never names a backup by handle (spec §286
|
||||
@@ -104,8 +133,10 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
// Optional backup_id in the JSON body; absent → LatestBackup (backward compat).
|
||||
// safety_snapshot defaults to true.
|
||||
var body struct {
|
||||
BackupID string `json:"backup_id"`
|
||||
BackupID string `json:"backup_id"`
|
||||
SafetySnapshot *bool `json:"safety_snapshot"`
|
||||
}
|
||||
if strings.HasPrefix(r.Header.Get("Content-Type"), "application/json") {
|
||||
if err := decodeJSON(w, r, &body); err != nil {
|
||||
@@ -133,6 +164,13 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, errForbidden)
|
||||
return
|
||||
}
|
||||
// The reaper found this archive damaged when it read it back; the restore
|
||||
// Job would refuse it too, but only after the world was locked for it.
|
||||
if backup.Corrupt {
|
||||
writeError(w, r, newError(http.StatusConflict, "backup_corrupt",
|
||||
"this backup did not read back intact and cannot be restored; pick another"))
|
||||
return
|
||||
}
|
||||
} else {
|
||||
backup, err = a.Repo.LatestBackup(r.Context(), name)
|
||||
if err != nil {
|
||||
@@ -204,7 +242,23 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
defer release()
|
||||
|
||||
if err := a.Restorer.Restore(r.Context(), name, backup.BackupRef); err != nil {
|
||||
snapshotter, snapshot := a.snapshotFirst()
|
||||
if body.SafetySnapshot != nil && !*body.SafetySnapshot {
|
||||
snapshot = false
|
||||
}
|
||||
if snapshot {
|
||||
// The snapshot records the current owner, like an on-demand backup, so
|
||||
// the way back is theirs to take.
|
||||
err = snapshotter.BackupThenRestore(r.Context(), name, rec.OwnerID, backup.ID, backup.BackupRef)
|
||||
} else {
|
||||
err = a.Restorer.Restore(r.Context(), name, backup.BackupRef)
|
||||
}
|
||||
if err != nil {
|
||||
if isRestoreInProgress(err) {
|
||||
writeError(w, r, newError(http.StatusConflict, "restore_in_progress",
|
||||
"a restore of another backup is still running on this server's world; retry once it finishes"))
|
||||
return
|
||||
}
|
||||
// ErrNotFound (server vanished from the execution backend) → 404; else 500.
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
@@ -212,9 +266,10 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
a.audit(r, "backup.restore", name)
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||
"name": name,
|
||||
"status": "restoring",
|
||||
"backup_id": backup.ID,
|
||||
"name": name,
|
||||
"status": "restoring",
|
||||
"backup_id": backup.ID,
|
||||
"safety_snapshot": snapshot,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -343,7 +398,7 @@ func (a *API) handleInternalBackup(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
a.enqueueBackup(w, r, name, rec, actor, "internal")
|
||||
a.enqueueBackup(w, r, name, rec, actor, internalSource(r))
|
||||
}
|
||||
|
||||
// enqueueBackup is the shared tail of both backup faces: the RWO stopped-gate, the
|
||||
|
||||
@@ -83,6 +83,76 @@ func TestListBackups(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
admin := &Principal{UserID: "admin1", Role: "admin", ViaAdminAccess: true}
|
||||
total := func(t *testing.T, w *httptest.ResponseRecorder) int {
|
||||
t.Helper()
|
||||
var resp struct {
|
||||
Total *int `json:"total"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil || resp.Total == nil {
|
||||
t.Fatalf("body carries no total: %v (%s)", err, w.Body.String())
|
||||
}
|
||||
return *resp.Total
|
||||
}
|
||||
|
||||
t.Run("server filter narrows inside the scope, never past it", func(t *testing.T) {
|
||||
api := mk()
|
||||
api.External = staticExternal{p: admin}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/backups?server=beta", "", nil)
|
||||
if got := ids(list(t, w)); !got["b2"] || len(got) != 1 || total(t, w) != 1 {
|
||||
t.Fatalf("admin ?server=beta = %v (total %d), want {b2} of 1", got, total(t, w))
|
||||
}
|
||||
api.External = staticExternal{p: &Principal{UserID: "owner1", Role: "user"}}
|
||||
w = do(api.ExternalHandler(), "GET", "/api/v1/backups?server=beta", "", nil)
|
||||
if got := ids(list(t, w)); len(got) != 0 || total(t, w) != 0 {
|
||||
t.Fatalf("owner1 ?server=beta = %v (total %d), want nothing: beta's backup is owner2's", got, total(t, w))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("pages newest first with the match total", func(t *testing.T) {
|
||||
api := mk()
|
||||
api.External = staticExternal{p: admin}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/backups?limit=1", "", nil)
|
||||
if vs := list(t, w); len(vs) != 1 || vs[0].ID != "b2" || total(t, w) != 2 {
|
||||
t.Fatalf("?limit=1 = %+v (total %d), want [b2] of 2", vs, total(t, w))
|
||||
}
|
||||
w = do(api.ExternalHandler(), "GET", "/api/v1/backups?limit=1&offset=1", "", nil)
|
||||
if vs := list(t, w); len(vs) != 1 || vs[0].ID != "b1" {
|
||||
t.Fatalf("?limit=1&offset=1 = %+v, want [b1]", vs)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("page bounds", func(t *testing.T) {
|
||||
cases := []struct {
|
||||
query string
|
||||
want BackupListOpts
|
||||
}{
|
||||
{"", BackupListOpts{Limit: 20}},
|
||||
{"?limit=5000&offset=-3", BackupListOpts{Limit: 100}},
|
||||
{"?limit=x&offset=40&server=alpha", BackupListOpts{Server: "alpha", Limit: 20, Offset: 40}},
|
||||
}
|
||||
for _, c := range cases {
|
||||
repo := newFakeRepo()
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: admin}
|
||||
if w := do(api.ExternalHandler(), "GET", "/api/v1/backups"+c.query, "", nil); w.Code != http.StatusOK {
|
||||
t.Fatalf("%q: code = %d (%s)", c.query, w.Code, w.Body.String())
|
||||
}
|
||||
if repo.backupListOpts != c.want {
|
||||
t.Errorf("%q asked the repo for %+v, want %+v", c.query, repo.backupListOpts, c.want)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("malformed server name is a 400", func(t *testing.T) {
|
||||
api := mk()
|
||||
api.External = staticExternal{p: admin}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/backups?server=Bad%20Name", "", nil)
|
||||
if w.Code != http.StatusBadRequest || !strings.Contains(w.Body.String(), "bad_name") {
|
||||
t.Fatalf("code = %d (%s), want 400 bad_name", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("backup_ref never serialized", func(t *testing.T) {
|
||||
api := mk()
|
||||
api.External = staticExternal{p: &Principal{UserID: "admin1", Role: "admin", ViaAdminAccess: true}}
|
||||
@@ -414,6 +484,29 @@ func TestRestoreBackup(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("restore by backup_id that failed a read-back -> 409 backup_corrupt", func(t *testing.T) {
|
||||
api, repo, _, restorer := mkTwo()
|
||||
repo.backups[1].view.Corrupt = true
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, `{"backup_id":"bk2"}`, jsonHeaders)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "backup_corrupt" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if restorer.calls != 0 {
|
||||
t.Fatal("a corrupt backup reached the restorer")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no body skips a latest backup that failed a read-back", func(t *testing.T) {
|
||||
api, repo, _, restorer := mkTwo()
|
||||
repo.backups[0].view.Corrupt = true
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusAccepted || restorer.gotRef != "ref-bk2" {
|
||||
t.Fatalf("code = %d ref %q, want 202 restoring the newest intact backup", w.Code, restorer.gotRef)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no body -> falls back to LatestBackup (backward compat)", func(t *testing.T) {
|
||||
api, _, _, restorer := mkTwo()
|
||||
api.External = staticExternal{p: owner}
|
||||
|
||||
@@ -274,6 +274,21 @@ func TestCreateServerRejections(t *testing.T) {
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// A CR deleted by hand keeps its world volume; a new server of that
|
||||
// name would mount it and inherit the old world.
|
||||
name: "world volume left by a deleted server",
|
||||
body: validCreateBody,
|
||||
setup: func(_ *fakeRepo, cl *fakeCluster) {
|
||||
cl.orphanWorld = map[string]bool{"survival": true}
|
||||
},
|
||||
wantCode: http.StatusConflict, wantErr: "world_volume_exists",
|
||||
check: func(t *testing.T, repo *fakeRepo, _ *fakeCluster) {
|
||||
if repo.seeded["survival"] {
|
||||
t.Error("a create refused over a leftover volume must not seed a servers row")
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, c := range cases {
|
||||
|
||||
@@ -57,12 +57,20 @@ const (
|
||||
// use a passkey.
|
||||
otpFailureBudget = 10
|
||||
otpFailureWindow = 24 * time.Hour
|
||||
// otpLiveLoginCodes is how many codes a pre-session login door (email login,
|
||||
// op-login) keeps redeemable per (account, purpose). Anyone who knows an address
|
||||
// can start a login for it, so a start never cancels the codes already mailed:
|
||||
// with one mail per otpResendCooldown, every code stays good for at least
|
||||
// otpLiveLoginCodes cooldowns (or its TTL), and a stranger's starts only add
|
||||
// codes to the owner's inbox. A wrong guess is compared with every live code, so
|
||||
// the daily blind-hit chance rises to otpFailureBudget*otpLiveLoginCodes/1e6
|
||||
// (3e-5). Enforced in the Repo so the fake and PG agree.
|
||||
otpLiveLoginCodes = 3
|
||||
)
|
||||
|
||||
// OTPMailer delivers a one-time code to an email address. It is a seam, not a
|
||||
// dependency: the demo ships without SMTP, so a nil Mailer logs the code
|
||||
// server-side instead of mailing it (a KNOWN-LIMITATION, never a code returned to
|
||||
// the client). Production wires a real sender.
|
||||
// OTPMailer delivers a one-time code to an email address. felis api wires the
|
||||
// [smtp] relay (internal/mail); with none configured it stays nil and every door
|
||||
// that mails a code answers 503 mail_unavailable before minting one.
|
||||
type OTPMailer interface {
|
||||
SendOTP(ctx context.Context, email, code string) error
|
||||
}
|
||||
@@ -131,6 +139,16 @@ func (a *API) handleEmailOTPStart(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "a valid email is required"))
|
||||
return
|
||||
}
|
||||
// No relay (or a spent budget) is said before asking for a re-verification
|
||||
// the player could not then use.
|
||||
if err := a.checkMailBudget(); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
// Gate the start: the verify only redeems a code minted here.
|
||||
if !a.requireReauth(w, r, p) {
|
||||
return
|
||||
}
|
||||
// Atomically reserve the cooldown on both the caller and the recipient BEFORE
|
||||
// minting, so a burst of truly concurrent starts yields exactly one winner. Here
|
||||
// the throttle is the sole defense and each admitted send is a real, non-idempotent
|
||||
@@ -249,19 +267,29 @@ func (a *API) handleEmailOTPVerify(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
a.audit(r, "account.email.verified", "")
|
||||
// Proving the address is an email reauth.
|
||||
a.markReauthQuietly(r)
|
||||
// Replacing a verified address moves where sign-in codes go, so a session
|
||||
// opened through the old one ends, and the old mailbox hears about it. A
|
||||
// first verification retires nothing.
|
||||
if p.EmailVerified && !strings.EqualFold(p.Email, email) {
|
||||
a.revokeOtherSessionsAfter(r, "email change")
|
||||
a.notifyEmailChanged(r, p.Email, email)
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"verified": true, "email": email})
|
||||
}
|
||||
|
||||
// deliverOTP hands the code to the configured Mailer, or — when none is wired (the
|
||||
// demo) — logs it server-side as a KNOWN-LIMITATION. The code is logged ONLY in the
|
||||
// no-mailer fallback and ONLY to the server log; it is never put in an HTTP response.
|
||||
// deliverOTP hands the code to the configured Mailer. The code goes nowhere
|
||||
// else: never into a response and never into a log, since anyone who can read
|
||||
// the API's logs could otherwise sign in as any player. The doors refuse a
|
||||
// relay-less install before minting (checkMailBudget); the nil check here only
|
||||
// keeps a future caller that skips that check from minting a code no one gets.
|
||||
//
|
||||
// Every real send spends one token of the install-wide mail budget (mailGate);
|
||||
// a spent budget is a 429 mail_rate_limited and nothing reaches the relay.
|
||||
func (a *API) deliverOTP(ctx context.Context, email, code string) error {
|
||||
if a.Mailer == nil {
|
||||
log.Printf("email-otp: no Mailer configured; code for %s is %s (KNOWN-LIMITATION: demo has no SMTP)", email, code)
|
||||
return nil
|
||||
return errMailUnavailable()
|
||||
}
|
||||
if ok, wait := a.mailGate().take(mailGateKey); !ok {
|
||||
metrics.MailTotal.WithLabelValues("otp", "throttled").Inc()
|
||||
@@ -315,6 +343,11 @@ func (a *API) handleSetEmail(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "a valid email is required"))
|
||||
return
|
||||
}
|
||||
// Recording an address unverifies the current one, which would strip the
|
||||
// account's email factor and with it the reauth that guards adding a passkey.
|
||||
if !a.requireReauth(w, r, p) {
|
||||
return
|
||||
}
|
||||
if err := a.Repo.SetUserEmail(r.Context(), p.UserID, email); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
|
||||
@@ -139,15 +139,15 @@ func TestWithRecoverLogsPanicStack(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestEmailOTPStartValidation covers the mint-side input gate and the no-mailer
|
||||
// fallback (the demo path): a malformed address never mints, and a nil Mailer still
|
||||
// persists a code (logged server-side) so the verify flow stays exercisable.
|
||||
// TestEmailOTPStartValidation covers the mint-side input gate and the no-relay
|
||||
// refusal: a malformed address never mints, and with no Mailer the start answers
|
||||
// 503 mail_unavailable without minting, so no code exists to leak anywhere.
|
||||
func TestEmailOTPStartValidation(t *testing.T) {
|
||||
user := &Principal{UserID: "u1", Email: "[email protected]", Role: "user"}
|
||||
mk := func(repo *fakeRepo) http.Handler {
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: user}
|
||||
return api.ExternalHandler() // no Mailer wired → demo fallback
|
||||
return api.ExternalHandler() // no Mailer wired
|
||||
}
|
||||
|
||||
bad := map[string]string{
|
||||
@@ -174,16 +174,50 @@ func TestEmailOTPStartValidation(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("no mailer still persists a code (demo fallback)", func(t *testing.T) {
|
||||
t.Run("no mailer refuses without minting", func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
w := do(mk(repo), "POST", "/api/v1/account/email/start", `{"email":"[email protected]"}`, nil)
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
code, msg := errEnvelope(t, w)
|
||||
if w.Code != http.StatusServiceUnavailable || code != "mail_unavailable" {
|
||||
t.Fatalf("code = %d %s, want 503 mail_unavailable", w.Code, w.Body.String())
|
||||
}
|
||||
if len(repo.otps) != 1 {
|
||||
t.Fatalf("want exactly 1 persisted code, got %d", len(repo.otps))
|
||||
if msg != "this server has no mail relay configured, so it cannot send codes; sign in with a passkey or ask the server operator to set up email" {
|
||||
t.Errorf("message = %q", msg)
|
||||
}
|
||||
if len(repo.otps) != 0 {
|
||||
t.Fatalf("a refused start minted %d codes", len(repo.otps))
|
||||
}
|
||||
})
|
||||
|
||||
// No relay is said before a re-verification the player could not use.
|
||||
t.Run("no mailer is said before reauth", func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
repo.passkeyCreds["p"] = PasskeyCredential{ID: "p", UserID: "u1", CredentialID: "c-p", CreatedAt: frozenNow}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: &Principal{UserID: "u1", Email: "[email protected]", Role: "user", ViaSession: true}}
|
||||
w := do(api.ExternalHandler(), "POST", "/api/v1/account/email/start", `{"email":"[email protected]"}`, nil)
|
||||
if code, _ := errEnvelope(t, w); w.Code != http.StatusServiceUnavailable || code != "mail_unavailable" {
|
||||
t.Fatalf("code = %d %s, want 503 mail_unavailable", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestDeliverOTPWithoutMailer: a caller that reaches deliverOTP with no relay gets
|
||||
// the 503, and the code is written nowhere, the log included.
|
||||
func TestDeliverOTPWithoutMailer(t *testing.T) {
|
||||
var logged strings.Builder
|
||||
old := log.Writer()
|
||||
log.SetOutput(&logged)
|
||||
t.Cleanup(func() { log.SetOutput(old) })
|
||||
api := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
err := api.deliverOTP(context.Background(), "[email protected]", "042137")
|
||||
var ae *apiError
|
||||
if !errors.As(err, &ae) || ae.status != http.StatusServiceUnavailable || ae.code != "mail_unavailable" {
|
||||
t.Fatalf("deliverOTP = %v, want 503 mail_unavailable", err)
|
||||
}
|
||||
if strings.Contains(logged.String(), "042137") {
|
||||
t.Errorf("the code reached the log: %s", logged.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestEmailOTPStartRateLimited closes the email-bomb vector: handleEmailOTPStart is
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"regexp"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
@@ -30,12 +31,16 @@ import (
|
||||
// mirrors how ImageBuilder uses build.Request/build.Image rather than restating a
|
||||
// parallel type on this side of the seam.
|
||||
//
|
||||
// It returns fileedit.ErrNotFound / ErrBadPath / ErrTooLarge for caller-fault
|
||||
// failures, which writeFileEditError maps to 404 / 400 / 413.
|
||||
// It returns fileedit.ErrNotFound / ErrBadPath / ErrTooLarge / ErrConflict /
|
||||
// ErrNoSpace, which writeFileEditError maps to 404 / 400 / 413 / 409 / 507.
|
||||
//
|
||||
// Read and Write both return the file's SHA-256 (hex). Write's expect is the hash
|
||||
// a client read the file at; when set, a file that changed since is refused with
|
||||
// ErrConflict instead of being overwritten.
|
||||
type FileEditor interface {
|
||||
List(ctx context.Context, server, path string) (entries []fileedit.Entry, truncated bool, err error)
|
||||
Read(ctx context.Context, server, path string) ([]byte, error)
|
||||
Write(ctx context.Context, server, path string, content []byte) error
|
||||
Read(ctx context.Context, server, path string) (content []byte, sha256 string, err error)
|
||||
Write(ctx context.Context, server, path string, content []byte, expect string) (sha256 string, err error)
|
||||
}
|
||||
|
||||
// writeFileRequest is the PUT /servers/{name}/file body. Content is []byte, so
|
||||
@@ -49,8 +54,14 @@ type FileEditor interface {
|
||||
// answering 200 — a client serialisation bug silently destroying the very config
|
||||
// the caller opened this endpoint to repair. nil now means "the field was omitted"
|
||||
// and is refused; an explicit "" is still a legitimate deliberate truncate.
|
||||
//
|
||||
// ExpectSHA256 is optional. The panel always sends the hash its read returned,
|
||||
// so a save over a file someone else changed in the meantime answers 409
|
||||
// file_changed; omitting it (a script, or "overwrite anyway") writes
|
||||
// unconditionally.
|
||||
type writeFileRequest struct {
|
||||
Content *[]byte `json:"content"`
|
||||
Content *[]byte `json:"content"`
|
||||
ExpectSHA256 string `json:"expect_sha256,omitempty"`
|
||||
}
|
||||
|
||||
// handleListFiles serves GET /api/v1/servers/{name}/files?path=… — one directory's
|
||||
@@ -76,6 +87,9 @@ func (a *API) handleListFiles(w http.ResponseWriter, r *http.Request) {
|
||||
writeFileEditError(w, r, err)
|
||||
return
|
||||
}
|
||||
if entries == nil {
|
||||
entries = []fileedit.Entry{} // an empty directory is [], never null
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"path": path, "entries": entries, "truncated": truncated,
|
||||
})
|
||||
@@ -98,12 +112,15 @@ func (a *API) handleReadFile(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
content, err := a.Files.Read(r.Context(), name, path)
|
||||
content, sum, err := a.Files.Read(r.Context(), name, path)
|
||||
if err != nil {
|
||||
writeFileEditError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"path": path, "content": content})
|
||||
if content == nil {
|
||||
content = []byte{} // an empty file is "", never null
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"path": path, "content": content, "sha256": sum})
|
||||
}
|
||||
|
||||
// handleWriteFile serves PUT /api/v1/servers/{name}/file?path=… — replace a file's
|
||||
@@ -149,6 +166,13 @@ func (a *API) handleWriteFile(w http.ResponseWriter, r *http.Request) {
|
||||
len(*body.Content), fileedit.MaxWriteBytes))
|
||||
return
|
||||
}
|
||||
// The hash rides the Job's argv, so only its one legitimate shape is let
|
||||
// through: 64 lowercase hex digits, exactly what a read returned.
|
||||
if body.ExpectSHA256 != "" && !sha256Hex.MatchString(body.ExpectSHA256) {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
||||
"expect_sha256 must be the 64-digit lowercase hex sha256 a read returned"))
|
||||
return
|
||||
}
|
||||
|
||||
// A write holds the world volume for its Job's lifetime (internal/maintenance);
|
||||
// reads and listings do not, since a read-only mount cannot hurt a server
|
||||
@@ -159,15 +183,18 @@ func (a *API) handleWriteFile(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
defer release()
|
||||
|
||||
if err := a.Files.Write(r.Context(), name, path, *body.Content); err != nil {
|
||||
sum, err := a.Files.Write(r.Context(), name, path, *body.Content, body.ExpectSHA256)
|
||||
if err != nil {
|
||||
writeFileEditError(w, r, err)
|
||||
return
|
||||
}
|
||||
|
||||
a.audit(r, "file.write", name+":"+path)
|
||||
writeJSON(w, http.StatusOK, map[string]any{"path": path, "status": "written"})
|
||||
writeJSON(w, http.StatusOK, map[string]any{"path": path, "status": "written", "sha256": sum})
|
||||
}
|
||||
|
||||
var sha256Hex = regexp.MustCompile(`^[0-9a-f]{64}$`)
|
||||
|
||||
// authorizeFileOp is the shared front half of all three file handlers — the gate
|
||||
// that decides whether this caller may touch this server's world at all. It
|
||||
// mirrors the backup/restore gate step for step, because it is guarding the same
|
||||
@@ -260,6 +287,10 @@ func writeFileEditError(w http.ResponseWriter, r *http.Request, err error) {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_path", "%s", err.Error()))
|
||||
case errors.Is(err, fileedit.ErrTooLarge):
|
||||
writeError(w, r, newError(http.StatusRequestEntityTooLarge, "too_large", "%s", err.Error()))
|
||||
case errors.Is(err, fileedit.ErrConflict):
|
||||
writeError(w, r, newError(http.StatusConflict, "file_changed", "%s", err.Error()))
|
||||
case errors.Is(err, fileedit.ErrNoSpace):
|
||||
writeError(w, r, newError(http.StatusInsufficientStorage, "volume_full", "%s", err.Error()))
|
||||
case errors.Is(err, context.DeadlineExceeded):
|
||||
writeError(w, r, newError(http.StatusGatewayTimeout, "files_timeout",
|
||||
"the file operation did not finish in time; retry shortly"))
|
||||
|
||||
@@ -23,10 +23,12 @@ type fakeFileEditor struct {
|
||||
gotServer string
|
||||
gotPath string
|
||||
gotContent []byte
|
||||
gotExpect string
|
||||
|
||||
entries []fileedit.Entry
|
||||
truncated bool
|
||||
content []byte
|
||||
sum string
|
||||
}
|
||||
|
||||
func (f *fakeFileEditor) List(_ context.Context, server, path string) ([]fileedit.Entry, bool, error) {
|
||||
@@ -35,18 +37,21 @@ func (f *fakeFileEditor) List(_ context.Context, server, path string) ([]fileedi
|
||||
return f.entries, f.truncated, f.err
|
||||
}
|
||||
|
||||
func (f *fakeFileEditor) Read(_ context.Context, server, path string) ([]byte, error) {
|
||||
func (f *fakeFileEditor) Read(_ context.Context, server, path string) ([]byte, string, error) {
|
||||
f.calls++
|
||||
f.gotServer, f.gotPath = server, path
|
||||
return f.content, f.err
|
||||
return f.content, f.sum, f.err
|
||||
}
|
||||
|
||||
func (f *fakeFileEditor) Write(_ context.Context, server, path string, content []byte) error {
|
||||
func (f *fakeFileEditor) Write(_ context.Context, server, path string, content []byte, expect string) (string, error) {
|
||||
f.calls++
|
||||
f.gotServer, f.gotPath, f.gotContent = server, path, content
|
||||
return f.err
|
||||
f.gotServer, f.gotPath, f.gotContent, f.gotExpect = server, path, content, expect
|
||||
return f.sum, f.err
|
||||
}
|
||||
|
||||
// testSum is a well-formed sha256 hex digest for the fake to hand out.
|
||||
var testSum = strings.Repeat("a", 64)
|
||||
|
||||
// mkFiles builds an API whose "survival" server is STOPPED and owned by owner1,
|
||||
// with a wired fakeFileEditor — the state in which every file operation is
|
||||
// permitted, so each subtest changes exactly the one thing it is about.
|
||||
@@ -310,9 +315,10 @@ func TestFileEditorHandlers(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("read returns base64 content", func(t *testing.T) {
|
||||
t.Run("read returns base64 content and its hash", func(t *testing.T) {
|
||||
api, _, _, files := mkFiles()
|
||||
files.content = []byte("motd=hello\n")
|
||||
files.sum = testSum
|
||||
api.External = staticExternal{p: owner}
|
||||
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/servers/survival/file?path=server.properties", "", nil)
|
||||
@@ -322,11 +328,12 @@ func TestFileEditorHandlers(t *testing.T) {
|
||||
var resp struct {
|
||||
Path string `json:"path"`
|
||||
Content []byte `json:"content"`
|
||||
SHA256 string `json:"sha256"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
|
||||
t.Fatalf("body not JSON: %v", err)
|
||||
}
|
||||
if resp.Path != "server.properties" || string(resp.Content) != "motd=hello\n" {
|
||||
if resp.Path != "server.properties" || string(resp.Content) != "motd=hello\n" || resp.SHA256 != testSum {
|
||||
t.Fatalf("unexpected response %+v (%q)", resp, resp.Content)
|
||||
}
|
||||
})
|
||||
@@ -361,6 +368,53 @@ func TestFileEditorHandlers(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("write passes the expected hash through and returns the new one", func(t *testing.T) {
|
||||
api, _, _, files := mkFiles()
|
||||
files.sum = strings.Repeat("b", 64)
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "PUT", "/api/v1/servers/survival/file?path=server.properties",
|
||||
`{"content":"aGk=","expect_sha256":"`+testSum+`"}`, jsonHeader)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if files.gotExpect != testSum {
|
||||
t.Fatalf("executor got expect %q, want %q", files.gotExpect, testSum)
|
||||
}
|
||||
var resp struct {
|
||||
SHA256 string `json:"sha256"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil || resp.SHA256 != files.sum {
|
||||
t.Fatalf("response sha256 = %q (%v), want %q", resp.SHA256, err, files.sum)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a malformed expected hash -> 400 before the executor", func(t *testing.T) {
|
||||
for _, bad := range []string{"abc", strings.Repeat("A", 64), strings.Repeat("a", 63) + " ", "--op=list"} {
|
||||
api, _, _, files := mkFiles()
|
||||
api.External = staticExternal{p: owner}
|
||||
body, _ := json.Marshal(map[string]any{"content": []byte("hi"), "expect_sha256": bad})
|
||||
w := do(api.ExternalHandler(), "PUT", "/api/v1/servers/survival/file?path=server.properties",
|
||||
string(body), jsonHeader)
|
||||
if w.Code != http.StatusBadRequest || files.calls != 0 {
|
||||
t.Fatalf("expect %q: code = %d calls = %d, want 400 and no Job", bad, w.Code, files.calls)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a stale write -> 409 file_changed, not audited", func(t *testing.T) {
|
||||
api, repo, _, files := mkFiles()
|
||||
files.err = fmt.Errorf("%w: server.properties has changed", fileedit.ErrConflict)
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "PUT", "/api/v1/servers/survival/file?path=server.properties",
|
||||
`{"content":"aGk=","expect_sha256":"`+testSum+`"}`, jsonHeader)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "file_changed" {
|
||||
t.Fatalf("code = %d body %s, want 409 file_changed", w.Code, w.Body.String())
|
||||
}
|
||||
if len(repo.audits) != 0 {
|
||||
t.Fatalf("a refused write was audited: %+v", repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("reads are not audited", func(t *testing.T) {
|
||||
api, repo, _, _ := mkFiles()
|
||||
api.External = staticExternal{p: owner}
|
||||
@@ -457,6 +511,8 @@ func TestFileEditorErrorMapping(t *testing.T) {
|
||||
{"escaping path", fmt.Errorf("%w: nope", fileedit.ErrBadPath), http.StatusBadRequest, "bad_path"},
|
||||
{"missing file", fmt.Errorf("%w: nope", fileedit.ErrNotFound), http.StatusNotFound, "not_found"},
|
||||
{"oversized file", fmt.Errorf("%w: nope", fileedit.ErrTooLarge), http.StatusRequestEntityTooLarge, "too_large"},
|
||||
{"changed since read", fmt.Errorf("%w: nope", fileedit.ErrConflict), http.StatusConflict, "file_changed"},
|
||||
{"volume full", fmt.Errorf("%w: nope", fileedit.ErrNoSpace), http.StatusInsufficientStorage, "volume_full"},
|
||||
{"timeout", fmt.Errorf("waiting: %w", context.DeadlineExceeded), http.StatusGatewayTimeout, "files_timeout"},
|
||||
}
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@ package api
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
@@ -24,7 +25,10 @@ func (a *API) handleReadyz(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
for name, check := range checks {
|
||||
if err := check(r.Context()); err != nil {
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "not_ready", "%s: %v", name, err))
|
||||
// The cause goes to the log; the probe answer names only the dependency,
|
||||
// so a driver error (hosts, users, SQL) never reaches a caller.
|
||||
log.Printf("api: readyz: %s: %v (request_id=%s)", name, err, requestIDFromContext(r.Context()))
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "not_ready", "%s is unavailable", name))
|
||||
return
|
||||
}
|
||||
}
|
||||
@@ -39,6 +43,9 @@ func (a *API) handleListServers(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if servers == nil {
|
||||
servers = []ServerInfo{} // an empty fleet is [], never null
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"servers": servers})
|
||||
}
|
||||
|
||||
@@ -52,7 +59,7 @@ func (a *API) handleReady(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
a.auditEntry(r, AuditEntry{
|
||||
Actor: "backend", Source: "internal", Action: "ready", ServerName: name,
|
||||
Actor: "backend", Source: internalSource(r), Action: "ready", ServerName: name,
|
||||
})
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
@@ -163,7 +170,7 @@ func (a *API) handleInternalWake(w http.ResponseWriter, r *http.Request) {
|
||||
// untouched and the next join attempt is not also throttled.
|
||||
a.limiter().record(name)
|
||||
a.auditEntry(r, AuditEntry{
|
||||
Actor: "velocity", Source: "internal", Action: "wake", ServerName: name,
|
||||
Actor: "velocity", Source: internalSource(r), Action: "wake", ServerName: name,
|
||||
})
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||
"name": name, "desiredState": "Running",
|
||||
@@ -255,7 +262,7 @@ func (a *API) handleInternalClaim(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
a.auditEntry(r, AuditEntry{
|
||||
Actor: "velocity", Source: "internal", Action: "claim", ServerName: name,
|
||||
Actor: "velocity", Source: internalSource(r), Action: "claim", ServerName: name,
|
||||
})
|
||||
writeJSON(w, http.StatusOK, map[string]any{"name": name, "claimed": true})
|
||||
}
|
||||
|
||||
@@ -211,7 +211,7 @@ func assertEq(t *testing.T, key string, got, want any) {
|
||||
func (f *fakeRepo) assertClaimAudit(t *testing.T, name string) {
|
||||
t.Helper()
|
||||
for _, e := range f.audits {
|
||||
if e.Action == "claim" && e.ServerName == name && e.Actor == "velocity" && e.Source == "internal" {
|
||||
if e.Action == "claim" && e.ServerName == name && e.Actor == "velocity" && e.Source == "internal:velocity" {
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
|
||||
@@ -80,7 +81,11 @@ func (a *API) handleServerConsole(w http.ResponseWriter, r *http.Request) {
|
||||
// Open the follow stream. Every error must be resolved HERE, into a normal JSON
|
||||
// envelope, because relayLogStream commits the 200 + SSE headers and no error
|
||||
// body can follow it.
|
||||
src, err := a.Logs.StreamLogs(r.Context(), name)
|
||||
ctx := r.Context()
|
||||
if since, ok := logSinceFromRequest(r, a.now()); ok {
|
||||
ctx = withLogSince(ctx, since)
|
||||
}
|
||||
src, err := a.Logs.StreamLogs(ctx, name)
|
||||
switch {
|
||||
case errors.Is(err, ErrConsoleUnavailable):
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "console_unavailable",
|
||||
@@ -99,5 +104,35 @@ func (a *API) handleServerConsole(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
a.audit(r, "console.attach", name)
|
||||
relayLogStream(w, r, src)
|
||||
relayLogStream(w, r, src, a.streamRecheck(r, func(ctx context.Context, p *Principal) error {
|
||||
rec, err := a.Repo.ServerByName(ctx, name)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return errForbidden // the server is gone, and the grant with it
|
||||
case err != nil:
|
||||
return nil
|
||||
case !a.isOwnerOrAdmin(p, rec):
|
||||
return errForbidden
|
||||
}
|
||||
return nil
|
||||
}), a.streamsClosing())
|
||||
}
|
||||
|
||||
// streamRecheck builds the streamGuard for an external-face stream: it re-runs the
|
||||
// authentication the stream opened with (the session may have been revoked or
|
||||
// expired, the account disabled), the op.console staff gate, and then allow for the
|
||||
// route's own rule. An unreachable session store is not a verdict (see streamGuard).
|
||||
func (a *API) streamRecheck(r *http.Request, allow func(ctx context.Context, p *Principal) error) streamGuard {
|
||||
return func(ctx context.Context) error {
|
||||
p, err := a.External.Authenticate(r)
|
||||
switch {
|
||||
case errors.Is(err, errAuthBackend):
|
||||
return nil
|
||||
case err != nil || p == nil:
|
||||
return errUnauthorized
|
||||
case hostIsAdminConsole(r, a.RootDomain, a.AdminHostname) && !p.IsAdmin():
|
||||
return errForbidden
|
||||
}
|
||||
return allow(ctx, p)
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
@@ -376,7 +377,7 @@ type interleaveWriter struct {
|
||||
|
||||
func (b *interleaveWriter) Write(p []byte) (int, error) {
|
||||
s := string(p)
|
||||
if strings.HasPrefix(s, "data:") {
|
||||
if strings.Contains(s, "\ndata:") {
|
||||
b.sawData = true
|
||||
}
|
||||
if s == sseHeartbeat {
|
||||
@@ -552,7 +553,7 @@ type firstDataWriter struct {
|
||||
|
||||
func (b *firstDataWriter) Write(p []byte) (int, error) {
|
||||
n, err := b.ResponseWriter.Write(p)
|
||||
if strings.HasPrefix(string(p), "data:") {
|
||||
if strings.Contains(string(p), "\ndata:") {
|
||||
b.once.Do(func() { close(b.data) })
|
||||
}
|
||||
return n, err
|
||||
@@ -687,7 +688,7 @@ func TestRelayLogStreamWriteDeadlineSeversStalledReader(t *testing.T) {
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
relayLogStream(w, r, src)
|
||||
relayLogStream(w, r, src, nil, nil)
|
||||
close(done)
|
||||
}()
|
||||
|
||||
@@ -762,7 +763,7 @@ func TestRelayLogStreamClearsWriteDeadlineOnReturn(t *testing.T) {
|
||||
w := newDeadlineRecordWriter()
|
||||
r := httptest.NewRequest("GET", "/api/v1/servers/survival/console", nil)
|
||||
|
||||
relayLogStream(w, r, src)
|
||||
relayLogStream(w, r, src, nil, nil)
|
||||
|
||||
last, sawPositive := w.finalDeadline()
|
||||
if !sawPositive {
|
||||
@@ -775,3 +776,245 @@ func TestRelayLogStreamClearsWriteDeadlineOnReturn(t *testing.T) {
|
||||
t.Fatal("relay returned without closing the source")
|
||||
}
|
||||
}
|
||||
|
||||
// switchExternal is an Authenticator a test can revoke mid-stream.
|
||||
type switchExternal struct {
|
||||
mu sync.Mutex
|
||||
p *Principal
|
||||
err error
|
||||
}
|
||||
|
||||
func (s *switchExternal) Authenticate(*http.Request) (*Principal, error) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return s.p, s.err
|
||||
}
|
||||
|
||||
func (s *switchExternal) set(p *Principal, err error) {
|
||||
s.mu.Lock()
|
||||
s.p, s.err = p, err
|
||||
s.mu.Unlock()
|
||||
}
|
||||
|
||||
// vanishingServerRepo lets a test delete every server while a stream reads them.
|
||||
type vanishingServerRepo struct {
|
||||
*fakeRepo
|
||||
mu sync.Mutex
|
||||
gone bool
|
||||
}
|
||||
|
||||
func (v *vanishingServerRepo) vanish() {
|
||||
v.mu.Lock()
|
||||
v.gone = true
|
||||
v.mu.Unlock()
|
||||
}
|
||||
|
||||
func (v *vanishingServerRepo) ServerByName(ctx context.Context, name string) (*ServerRecord, error) {
|
||||
v.mu.Lock()
|
||||
gone := v.gone
|
||||
v.mu.Unlock()
|
||||
if gone {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
return v.fakeRepo.ServerByName(ctx, name)
|
||||
}
|
||||
|
||||
func shrinkStreamTimers(t *testing.T, recheck, lifetime time.Duration) {
|
||||
t.Helper()
|
||||
origRecheck, origLife := streamRecheckEvery, streamMaxLifetime
|
||||
streamRecheckEvery, streamMaxLifetime = recheck, lifetime
|
||||
t.Cleanup(func() { streamRecheckEvery, streamMaxLifetime = origRecheck, origLife })
|
||||
}
|
||||
|
||||
// A stream is authorized when it opens; a session revoked (or a server handed to
|
||||
// someone else) afterwards must not keep the console flowing. The relay re-asks
|
||||
// on a timer and ends with "event: revoked" once the answer changes.
|
||||
func TestServerConsoleStreamEndsWhenAccessIsWithdrawn(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
stranger := &Principal{UserID: "someone", Email: "[email protected]", Role: "user"}
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
revoke func(ext *switchExternal, repo *vanishingServerRepo)
|
||||
}{
|
||||
{"session revoked", func(ext *switchExternal, _ *vanishingServerRepo) { ext.set(nil, errUnauthorized) }},
|
||||
{"caller no longer owns the server", func(ext *switchExternal, _ *vanishingServerRepo) { ext.set(stranger, nil) }},
|
||||
{"server deleted", func(_ *switchExternal, repo *vanishingServerRepo) { repo.vanish() }},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
shrinkStreamTimers(t, 20*time.Millisecond, time.Hour)
|
||||
repo := &vanishingServerRepo{fakeRepo: newFakeRepo()}
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
ext := &switchExternal{p: owner}
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = ext
|
||||
started := make(chan struct{})
|
||||
streamer := &fakeLogStreamer{srcFromCtx: func(ctx context.Context) io.ReadCloser {
|
||||
return &ctxBlockingReadCloser{ctx: ctx, first: []byte("boot\n"), firstRead: started, closed: make(chan struct{})}
|
||||
}}
|
||||
a.Logs = streamer
|
||||
|
||||
done := make(chan *httptest.ResponseRecorder)
|
||||
go func() { done <- do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) }()
|
||||
<-started
|
||||
tc.revoke(ext, repo)
|
||||
|
||||
select {
|
||||
case w := <-done:
|
||||
body := w.Body.String()
|
||||
if !strings.Contains(body, "data: boot\n\n") || !strings.HasSuffix(body, sseRevoked) {
|
||||
t.Fatalf("body = %q, want the boot line then %q", body, sseRevoked)
|
||||
}
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("the stream kept running after its grant was withdrawn")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A store outage is no verdict on the caller: the stream keeps going (the lifetime
|
||||
// cap still bounds it), and it ends with a plain close, which EventSource answers by
|
||||
// reconnecting through the full auth path.
|
||||
func TestServerConsoleStreamOutlivesAnAuthOutageUntilItsLifetime(t *testing.T) {
|
||||
shrinkStreamTimers(t, 10*time.Millisecond, 80*time.Millisecond)
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
ext := &switchExternal{p: owner}
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = ext
|
||||
started := make(chan struct{})
|
||||
a.Logs = &fakeLogStreamer{srcFromCtx: func(ctx context.Context) io.ReadCloser {
|
||||
return &ctxBlockingReadCloser{ctx: ctx, first: []byte("boot\n"), firstRead: started, closed: make(chan struct{})}
|
||||
}}
|
||||
|
||||
begun := time.Now()
|
||||
done := make(chan *httptest.ResponseRecorder)
|
||||
go func() { done <- do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) }()
|
||||
<-started
|
||||
ext.set(nil, errAuthBackend)
|
||||
|
||||
select {
|
||||
case w := <-done:
|
||||
if strings.Contains(w.Body.String(), "event: revoked") {
|
||||
t.Fatalf("an auth outage revoked the stream: %q", w.Body.String())
|
||||
}
|
||||
if since := time.Since(begun); since < 80*time.Millisecond {
|
||||
t.Fatalf("stream ended after %v, before its lifetime", since)
|
||||
}
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("the stream outlived streamMaxLifetime")
|
||||
}
|
||||
}
|
||||
|
||||
// CloseStreams (registered with RegisterOnShutdown) ends an open console at
|
||||
// once, and a stream opened after it ends as soon as it starts.
|
||||
func TestServerConsoleStreamEndsOnCloseStreams(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = staticExternal{p: owner}
|
||||
var closed []chan struct{}
|
||||
a.Logs = &fakeLogStreamer{srcFromCtx: func(ctx context.Context) io.ReadCloser {
|
||||
c := make(chan struct{})
|
||||
closed = append(closed, c)
|
||||
return &ctxBlockingReadCloser{ctx: ctx, first: []byte("boot\n"), firstRead: make(chan struct{}), closed: c}
|
||||
}}
|
||||
|
||||
open := func() chan *httptest.ResponseRecorder {
|
||||
done := make(chan *httptest.ResponseRecorder, 1)
|
||||
go func() { done <- do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "", nil) }()
|
||||
return done
|
||||
}
|
||||
wait := func(done chan *httptest.ResponseRecorder, what string) {
|
||||
t.Helper()
|
||||
select {
|
||||
case w := <-done:
|
||||
if w.Code != http.StatusOK || strings.Contains(w.Body.String(), "event: revoked") {
|
||||
t.Fatalf("%s: code %d body %q, want a plain close", what, w.Code, w.Body.String())
|
||||
}
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatalf("%s: the stream did not end", what)
|
||||
}
|
||||
}
|
||||
|
||||
first := open()
|
||||
select {
|
||||
case <-first:
|
||||
t.Fatal("the stream ended before CloseStreams")
|
||||
case <-time.After(50 * time.Millisecond):
|
||||
}
|
||||
a.CloseStreams()
|
||||
a.CloseStreams() // every listener's shutdown calls it
|
||||
wait(first, "open stream")
|
||||
wait(open(), "stream opened after CloseStreams")
|
||||
for i, c := range closed {
|
||||
select {
|
||||
case <-c:
|
||||
default:
|
||||
t.Fatalf("source %d was never closed", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A reconnecting EventSource sends back the id of the last line it saw; the
|
||||
// handler resumes the follow from that second instead of replaying the backlog.
|
||||
func TestServerConsoleStreamResumesFromLastEventID(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = staticExternal{p: owner}
|
||||
streamer := &fakeLogStreamer{src: &recordReadCloser{r: strings.NewReader("line\n")}}
|
||||
a.Logs = streamer
|
||||
|
||||
last := a.now().Add(-2 * time.Minute).Truncate(time.Second)
|
||||
w := do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/console", "",
|
||||
map[string]string{"Last-Event-ID": strconv.FormatInt(last.Unix(), 10)})
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if got, ok := logSinceFromContext(streamer.gotCtx); !ok || !got.Equal(last) {
|
||||
t.Fatalf("streamer since = %v (%v), want %v", got, ok, last)
|
||||
}
|
||||
if !strings.Contains(w.Body.String(), "id: ") {
|
||||
t.Fatalf("relayed lines carry no id for the next resume: %q", w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogSinceFromRequest(t *testing.T) {
|
||||
now := time.Unix(1_800_000_000, 0)
|
||||
for _, tc := range []struct {
|
||||
header string
|
||||
want time.Time
|
||||
ok bool
|
||||
}{
|
||||
{"", time.Time{}, false},
|
||||
{"garbage", time.Time{}, false},
|
||||
{strconv.FormatInt(now.Add(-time.Minute).Unix(), 10), now.Add(-time.Minute), true},
|
||||
{strconv.FormatInt(now.Add(time.Minute).Unix(), 10), time.Time{}, false},
|
||||
{strconv.FormatInt(now.Add(-2*maxResumeAge).Unix(), 10), time.Time{}, false},
|
||||
} {
|
||||
r := httptest.NewRequest("GET", "/", nil)
|
||||
if tc.header != "" {
|
||||
r.Header.Set("Last-Event-ID", tc.header)
|
||||
}
|
||||
got, ok := logSinceFromRequest(r, now)
|
||||
if ok != tc.ok || !got.Equal(tc.want) {
|
||||
t.Errorf("Last-Event-ID %q: got %v %v, want %v %v", tc.header, got, ok, tc.want, tc.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPodLogOptionsTailOrResume(t *testing.T) {
|
||||
opts := podLogOptions(context.Background(), "minecraft", 200)
|
||||
if opts.TailLines == nil || *opts.TailLines != 200 || opts.SinceTime != nil || !opts.Follow {
|
||||
t.Fatalf("fresh attach options = %+v, want a followed 200-line tail", opts)
|
||||
}
|
||||
since := time.Unix(1_800_000_000, 0)
|
||||
opts = podLogOptions(withLogSince(context.Background(), since), "minecraft", 200)
|
||||
if opts.TailLines != nil || opts.SinceTime == nil || !opts.SinceTime.Time.Equal(since) {
|
||||
t.Fatalf("resume options = %+v, want sinceTime %v and no tail", opts, since)
|
||||
}
|
||||
}
|
||||
@@ -125,17 +125,10 @@ func (a *API) handleBindRedeem(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
token, err := newSessionToken()
|
||||
if err != nil {
|
||||
if err := a.startSession(w, r, userID, bindCodeSignIn); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expires := a.now().Add(sessionTTL)
|
||||
if err := a.Repo.CreateSession(r.Context(), hashCookie(token), userID, expires); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
setSessionCookie(w, token, expires)
|
||||
// The in-game code proved the Minecraft account; it names the actor.
|
||||
a.auditEntry(r, AuditEntry{Actor: "mc:" + mcUUID, ActorUserID: userID, Action: "account.bind_redeem"})
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
|
||||
@@ -11,14 +11,15 @@ import (
|
||||
// sensitive tier. Unlike the console.<root_domain> player doors (email OTP / bind
|
||||
// code), a staff web session is never minted from a single factor. The flow is a
|
||||
// three-call state machine over op_login_requests (migration 0016), all Public
|
||||
// pre-session routes (the caller has no principal yet), plus two internal-face routes
|
||||
// for the in-game side (approve is driven by velocity's /felis command; pending has
|
||||
// no consumer yet — see handleOpLoginPending):
|
||||
// pre-session routes (the caller has no principal yet), plus internal-face routes
|
||||
// for the in-game side (show and approve are driven by velocity's /felis command;
|
||||
// pending has no consumer yet — see handleOpLoginPending):
|
||||
//
|
||||
// POST /api/v1/auth/op-login/start (public) — mint a request + mail an OTP
|
||||
// GET /api/v1/auth/op-login/status/{id} (public) — poll until an admin approves
|
||||
// POST /api/v1/auth/op-login/finish (public) — redeem code+approval → session
|
||||
// GET /api/v1/internal/op-login/pending (internal) — list requests awaiting a vouch
|
||||
// GET /api/v1/internal/op-login/{id} (internal) — who a request is for, shown to the admin
|
||||
// POST /api/v1/internal/op-login/{id}/approve (internal) — an in-game admin vouches
|
||||
//
|
||||
// The two factors:
|
||||
@@ -30,6 +31,9 @@ import (
|
||||
// request via velocity's /felis command (internal approve). The API's own user
|
||||
// table is the sole authority: only a UUID linked to a staff account may
|
||||
// approve (velocity's command runs for any player and relies on this check).
|
||||
// The admin first sees whose request it is (account, address, where it was
|
||||
// started) and approves by typing that account's name, so a code relayed by a
|
||||
// stranger ("please approve abc123") cannot be vouched for blind.
|
||||
//
|
||||
// finish mints the session only when BOTH have landed. Neither factor alone — a mailed
|
||||
// code without an approval, or an approval without the code — yields a session.
|
||||
@@ -65,6 +69,12 @@ type opLoginStartRequest struct {
|
||||
// unknown address yields the SAME 202 with a random, non-persisted handle and no mail,
|
||||
// so this never doubles as a staff-enumeration oracle (op.console's own Zero-Trust is
|
||||
// the edge gate; this app-layer neutrality covers the hostname-agnostic route).
|
||||
//
|
||||
// Anyone who knows a staff address can start a login for it, so a start never cancels
|
||||
// the codes already mailed (AddLoginEmailOTP), and a start inside the per-recipient
|
||||
// cooldown still gets a request of its own but mails nothing: the code mailed moments
|
||||
// ago is live and finishes it. A stranger's starts therefore only add codes to the
|
||||
// staff inbox and never keep its owner from signing in.
|
||||
func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
||||
if !localAuthEnabled(r.Context(), a.Repo) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "local_auth_disabled",
|
||||
@@ -94,33 +104,32 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
// Per-recipient cooldown reserved BEFORE any work, identical to the console email
|
||||
// door: one winner per window, and the neutral (non-staff) branch keeps the
|
||||
// reservation too so probing an address is throttled exactly like a real send. The
|
||||
// key is namespaced apart from the console door's "login:email:" so the two
|
||||
// door: one mail per window, and the neutral (non-staff) branch keeps the
|
||||
// reservation too so a probe holds the window exactly like a real send. The key is
|
||||
// namespaced apart from the console door's "login:email:" so the two
|
||||
// unauthenticated doors never perturb each other's throttle.
|
||||
emailKey := "oplogin:email:" + strings.ToLower(email)
|
||||
lim := a.otpLimiter()
|
||||
emailAt, ok := lim.reserve(emailKey, otpResendCooldown)
|
||||
if !ok {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_resend_cooldown",
|
||||
"a code was sent recently; wait a moment before requesting another"))
|
||||
return
|
||||
}
|
||||
emailAt, fresh := lim.reserve(emailKey, otpResendCooldown)
|
||||
committed := false
|
||||
defer func() {
|
||||
if !committed {
|
||||
lim.release(emailKey, emailAt)
|
||||
}
|
||||
}()
|
||||
if fresh {
|
||||
defer func() {
|
||||
if !committed {
|
||||
lim.release(emailKey, emailAt)
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// Compute expiry once so the neutral and real branches return identical-shaped
|
||||
// bodies and (real branch) the request row and its OTP are coterminous.
|
||||
expiresAt := a.now().Add(otpTTL)
|
||||
// Compute expiry once, from the reservation, so the neutral and real branches
|
||||
// return identical bodies and the request row and its OTP are coterminous. Inside
|
||||
// the cooldown the live code is the one the standing reservation mailed, so the
|
||||
// request ends with it.
|
||||
expiresAt := emailAt.Add(otpTTL)
|
||||
|
||||
// neutral returns the indistinguishable no-op success: a plausible but non-persisted
|
||||
// handle that status(id) reads approved:false forever (no row, never approvable). It
|
||||
// mints nothing and mails nothing, and KEEPS the reservation so probing is throttled
|
||||
// exactly like a real send.
|
||||
// mints nothing and mails nothing, and KEEPS the reservation so a probe holds the
|
||||
// window exactly like a real send.
|
||||
neutral := func() {
|
||||
fakeID, err := newOTPID()
|
||||
if err != nil {
|
||||
@@ -168,10 +177,26 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if err := a.Repo.CreateOpLoginRequest(r.Context(), id, u.ID, u.Email, expiresAt); err != nil {
|
||||
origin := ""
|
||||
if addr := a.clientIP(r); addr.IsValid() {
|
||||
origin = addr.String()
|
||||
}
|
||||
if err := a.Repo.CreateOpLoginRequest(r.Context(), NewOpLoginRequest{
|
||||
ID: id, UserID: u.ID, Email: u.Email, ExpiresAt: expiresAt,
|
||||
ClientIP: origin, UserAgent: truncateUTF8(r.UserAgent(), maxSessionUserAgent),
|
||||
}); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if !fresh {
|
||||
// Inside the cooldown: the code mailed with the standing reservation finishes
|
||||
// this request too, and the mailbox still sees one code per window.
|
||||
a.auditAccount(r, u, "auth.op_login.requested", "")
|
||||
writeJSON(w, http.StatusAccepted, map[string]any{
|
||||
"request_id": id, "expires_at": expiresAt.UTC(),
|
||||
})
|
||||
return
|
||||
}
|
||||
code, err := newEmailOTP()
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
@@ -184,7 +209,7 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
// Mint+mail against the STORED staff address (UserByEmail matched case-insensitively);
|
||||
// the request row snapshots the same address for its audit trail.
|
||||
if err := a.Repo.CreateEmailOTP(r.Context(), otpID, u.ID, u.Email, otpCodeHash(code), otpPurposeOpLogin, expiresAt); err != nil {
|
||||
if err := a.Repo.AddLoginEmailOTP(r.Context(), otpID, u.ID, u.Email, otpCodeHash(code), otpPurposeOpLogin, a.now(), expiresAt); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
@@ -326,17 +351,10 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "staff_account", "that account is not an operator"))
|
||||
return
|
||||
}
|
||||
token, err := newSessionToken()
|
||||
if err != nil {
|
||||
if err := a.startSession(w, r, u.ID, provenSignIn); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expires := now.Add(sessionTTL)
|
||||
if err := a.Repo.CreateSession(r.Context(), hashCookie(token), u.ID, expires); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
setSessionCookie(w, token, expires)
|
||||
a.auditAccount(r, u, "auth.op_login", "")
|
||||
writeJSON(w, http.StatusOK, map[string]any{"user_id": u.ID, "role": u.Role})
|
||||
}
|
||||
@@ -368,26 +386,106 @@ func (a *API) handleOpLoginPending(w http.ResponseWriter, r *http.Request) {
|
||||
"request_id": req.ID,
|
||||
"username": req.Username,
|
||||
"email": req.Email,
|
||||
"client_ip": req.ClientIP,
|
||||
"created_at": req.CreatedAt.UTC(),
|
||||
})
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"pending": out})
|
||||
}
|
||||
|
||||
// opLoginApprover resolves the in-game player running /felis web op approve to a
|
||||
// linked staff account (admin, or the owner superset). An unlinked UUID or a
|
||||
// non-staff player may never see or vouch for an op.console login; all refusals
|
||||
// share one 403 so a caller cannot tell "not linked" from "linked but not staff".
|
||||
func (a *API) opLoginApprover(r *http.Request, mcUUID string) (*StaffUser, error) {
|
||||
notAdmin := newError(http.StatusForbidden, "not_admin", "only a linked administrator may approve an operator login")
|
||||
approverID, err := a.Repo.UserByMCUUID(r.Context(), mcUUID)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return nil, notAdmin
|
||||
case err != nil:
|
||||
return nil, err
|
||||
}
|
||||
approver, err := a.Repo.UserByID(r.Context(), approverID)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return nil, notAdmin
|
||||
case err != nil:
|
||||
return nil, err
|
||||
}
|
||||
if !staffRole(approver.Role) {
|
||||
return nil, notAdmin
|
||||
}
|
||||
return approver, nil
|
||||
}
|
||||
|
||||
// pendingOpLogin loads a request an admin may still vouch for: pending, unconsumed
|
||||
// and unexpired. Anything else is the same 404 the approve race returns.
|
||||
func (a *API) pendingOpLogin(r *http.Request, id string) (*OpLoginRequest, error) {
|
||||
notFound := newError(http.StatusNotFound, "op_login_not_found", "no pending operator login with that id")
|
||||
req, err := a.Repo.OpLoginRequestByID(r.Context(), id)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return nil, notFound
|
||||
case err != nil:
|
||||
return nil, err
|
||||
}
|
||||
if req.Status != "pending" || req.Consumed || !req.ExpiresAt.After(a.now()) {
|
||||
return nil, notFound
|
||||
}
|
||||
return req, nil
|
||||
}
|
||||
|
||||
// handleOpLoginShow tells the in-game admin who a pending request is for before
|
||||
// they vouch (internal face): the account, its address, when and from where the
|
||||
// sign-in was started. velocity's /felis web op approve <code> renders this and
|
||||
// asks the admin to confirm by name. The approver UUID rides in the query and gets
|
||||
// the same staff check as approve, since velocity's command runs for any player and
|
||||
// a staff address must not be readable by one.
|
||||
func (a *API) handleOpLoginShow(w http.ResponseWriter, r *http.Request) {
|
||||
approverUUID := strings.TrimSpace(r.URL.Query().Get("approver_uuid"))
|
||||
if approverUUID == "" {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "approver_uuid is required"))
|
||||
return
|
||||
}
|
||||
if _, err := a.opLoginApprover(r, approverUUID); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
req, err := a.pendingOpLogin(r, r.PathValue("id"))
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"request_id": req.ID,
|
||||
"username": req.Username,
|
||||
"email": req.Email,
|
||||
"client_ip": req.ClientIP,
|
||||
"user_agent": req.UserAgent,
|
||||
"created_at": req.CreatedAt.UTC(),
|
||||
"expires_at": req.ExpiresAt.UTC(),
|
||||
})
|
||||
}
|
||||
|
||||
// opLoginApproveRequest is the internal approve body: the online-mode UUID of the
|
||||
// in-game admin running /felis web op approve. The API resolves it to a linked account
|
||||
// and refuses unless that account is staff (admin or owner) — this check against the API's
|
||||
// authoritative user table is the only gate; velocity's command itself is unprivileged.
|
||||
// in-game admin running /felis web op approve, and the account name they typed to
|
||||
// confirm whose sign-in they are vouching for. The API resolves the UUID to a
|
||||
// linked account and refuses unless that account is staff (admin or owner) — this
|
||||
// check against the API's authoritative user table is the only gate; velocity's
|
||||
// command itself is unprivileged.
|
||||
type opLoginApproveRequest struct {
|
||||
ApproverUUID string `json:"approver_uuid"`
|
||||
Username string `json:"username"`
|
||||
}
|
||||
|
||||
// handleOpLoginApprove records an in-game admin's vouch for a pending staff login
|
||||
// (internal face), supplying the second factor. It resolves the approver UUID to a
|
||||
// linked staff account (admin or owner; else 403), then flips the request approved.
|
||||
// A missing or no-longer-pending request is 404. Self-approval is allowed: a staff
|
||||
// member online as their own admin identity supplies a genuine second factor
|
||||
// (in-game session control) distinct from the mailbox factor.
|
||||
// linked staff account (admin or owner; else 403), requires the typed username to
|
||||
// name the request's account (else 409, request left pending), then flips the
|
||||
// request approved. A missing or no-longer-pending request is 404. Self-approval is
|
||||
// allowed: a staff member online as their own admin identity supplies a genuine
|
||||
// second factor (in-game session control) distinct from the mailbox factor.
|
||||
func (a *API) handleOpLoginApprove(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
var req opLoginApproveRequest
|
||||
@@ -396,38 +494,33 @@ func (a *API) handleOpLoginApprove(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
approverUUID := strings.TrimSpace(req.ApproverUUID)
|
||||
if approverUUID == "" {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "approver_uuid is required"))
|
||||
typed := strings.TrimSpace(req.Username)
|
||||
if approverUUID == "" || typed == "" {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "approver_uuid and username are required"))
|
||||
return
|
||||
}
|
||||
// Resolve the in-game approver to a linked account and require a staff role
|
||||
// (admin, or the owner superset). An unlinked UUID or a non-staff player may
|
||||
// never vouch for an op.console login. All three refusals share one response so
|
||||
// a caller cannot tell "not linked" from "linked but not staff".
|
||||
notAdmin := newError(http.StatusForbidden, "not_admin", "only a linked administrator may approve an operator login")
|
||||
approverID, err := a.Repo.UserByMCUUID(r.Context(), approverUUID)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
writeError(w, r, notAdmin)
|
||||
return
|
||||
case err != nil:
|
||||
approver, err := a.opLoginApprover(r, approverUUID)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
approver, err := a.Repo.UserByID(r.Context(), approverID)
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
writeError(w, r, notAdmin)
|
||||
return
|
||||
case err != nil:
|
||||
loginReq, err := a.pendingOpLogin(r, id)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if !staffRole(approver.Role) {
|
||||
writeError(w, r, notAdmin)
|
||||
// Minecraft names are case-insensitive, and so is the name an admin retypes.
|
||||
if !strings.EqualFold(typed, loginReq.Username) {
|
||||
payload, _ := json.Marshal(map[string]string{"request_id": id, "typed_username": typed})
|
||||
a.auditEntry(r, AuditEntry{
|
||||
Actor: approver.Username, ActorUserID: approver.ID, Source: internalSource(r),
|
||||
Action: "auth.op_login.approve_mismatch", Payload: payload,
|
||||
})
|
||||
writeError(w, r, newError(http.StatusConflict, "op_login_mismatch",
|
||||
"that operator login is for a different account"))
|
||||
return
|
||||
}
|
||||
switch err := a.Repo.ApproveOpLogin(r.Context(), id, approverID, a.now()); {
|
||||
switch err := a.Repo.ApproveOpLogin(r.Context(), id, approver.ID, a.now()); {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
writeError(w, r, newError(http.StatusNotFound, "op_login_not_found", "no pending operator login with that id"))
|
||||
return
|
||||
@@ -435,10 +528,15 @@ func (a *API) handleOpLoginApprove(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
payload, _ := json.Marshal(map[string]string{"request_id": id, "approver_user_id": approverID})
|
||||
payload, _ := json.Marshal(map[string]string{
|
||||
"request_id": id, "approver_user_id": approver.ID,
|
||||
"username": loginReq.Username, "client_ip": loginReq.ClientIP,
|
||||
})
|
||||
a.auditEntry(r, AuditEntry{
|
||||
Actor: approver.Username, ActorUserID: approverID, Source: "internal",
|
||||
Actor: approver.Username, ActorUserID: approver.ID, Source: internalSource(r),
|
||||
Action: "auth.op_login.approved", Payload: payload,
|
||||
})
|
||||
writeJSON(w, http.StatusOK, map[string]any{"approved": true})
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"approved": true, "username": loginReq.Username, "email": loginReq.Email,
|
||||
})
|
||||
}
|
||||
@@ -58,9 +58,9 @@ func finishOp(eh http.Handler, id, code string) *httptest.ResponseRecorder {
|
||||
return do(eh, "POST", "/api/v1/auth/op-login/finish",
|
||||
`{"request_id":"`+id+`","code":"`+code+`"}`, jsonHeader)
|
||||
}
|
||||
func approveOp(ih http.Handler, id, approverUUID string) *httptest.ResponseRecorder {
|
||||
func approveOp(ih http.Handler, id, approverUUID, username string) *httptest.ResponseRecorder {
|
||||
return do(ih, "POST", "/api/v1/internal/op-login/"+id+"/approve",
|
||||
`{"approver_uuid":"`+approverUUID+`"}`, nil)
|
||||
`{"approver_uuid":"`+approverUUID+`","username":"`+username+`"}`, nil)
|
||||
}
|
||||
|
||||
// TestOpLoginVertical walks the whole two-factor slice end to end: start mails a code
|
||||
@@ -126,7 +126,7 @@ func TestOpLoginVertical(t *testing.T) {
|
||||
}
|
||||
|
||||
// 4) an in-game admin approves via the internal face.
|
||||
if w := approveOp(ih, reqID, opUUID); w.Code != http.StatusOK {
|
||||
if w := approveOp(ih, reqID, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if ab := acctBody(t, statusOp(eh, reqID)); ab["approved"] != true {
|
||||
@@ -198,7 +198,7 @@ func TestOpLoginOwnerAdmitted(t *testing.T) {
|
||||
t.Fatalf("owner start must mint a request + mail a code: req=%q mails=%d rows=%d",
|
||||
reqID, mailer.calls, len(repo.opLogins))
|
||||
}
|
||||
if w := approveOp(ih, reqID, opUUID); w.Code != http.StatusOK {
|
||||
if w := approveOp(ih, reqID, opUUID, "owner"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
w = finishOp(eh, reqID, mailer.code)
|
||||
@@ -212,8 +212,8 @@ func TestOpLoginOwnerAdmitted(t *testing.T) {
|
||||
|
||||
// TestOpLoginStartNeutral pins the start-side anti-enumeration contract: op.console is
|
||||
// the STAFF door, so a non-admin account AND an unknown address both get a 202 carrying
|
||||
// a request_id + expires_at, mint/mail nothing, and still burn the per-recipient
|
||||
// cooldown — so neither the response nor the throttle tells a caller who is staff.
|
||||
// a request_id + expires_at and mint/mail nothing, and a re-probe inside the cooldown
|
||||
// does the same — so neither the response nor the throttle tells a caller who is staff.
|
||||
func TestOpLoginStartNeutral(t *testing.T) {
|
||||
check := func(t *testing.T, seed func(*fakeRepo), email, wantReason, wantUser string) {
|
||||
t.Helper()
|
||||
@@ -248,9 +248,14 @@ func TestOpLoginStartNeutral(t *testing.T) {
|
||||
repo.audits[0].ActorUserID != wantUser {
|
||||
t.Errorf("neutral start audits = %+v, want one auth.op_login.failed %s by %q", repo.audits, wantReason, wantUser)
|
||||
}
|
||||
// The reservation is KEPT: re-probing the same address is throttled like a resend.
|
||||
if w := startOp(eh, email); w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "otp_resend_cooldown" {
|
||||
t.Fatalf("re-probe: code = %d body %s, want 429 otp_resend_cooldown", w.Code, w.Body.String())
|
||||
// Re-probing inside the window: the same 202 shape, still nothing behind it.
|
||||
w = startOp(eh, email)
|
||||
if b := acctBody(t, w); w.Code != http.StatusAccepted || b["request_id"] == "" || b["expires_at"] != "2023-11-14T22:23:20Z" {
|
||||
t.Fatalf("re-probe: code = %d body %s, want 202 with a request_id and the first expiry", w.Code, w.Body.String())
|
||||
}
|
||||
if len(repo.opLogins) != 0 || len(repo.otps) != 0 || mailer.calls != 0 {
|
||||
t.Errorf("re-probe must mint/mail nothing: reqs=%d otps=%d mails=%d",
|
||||
len(repo.opLogins), len(repo.otps), mailer.calls)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -264,6 +269,64 @@ func TestOpLoginStartNeutral(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
// TestOpLoginStartByAStranger is the case of someone who knows a staff address and
|
||||
// keeps starting logins for it. Their start mails the code to the staff inbox; the
|
||||
// staff member's own start inside the cooldown still gets a request of its own
|
||||
// (nothing new mailed, same expiry as the live code), and that inbox code finishes
|
||||
// it. A start after the cooldown mails a second code without cancelling the first.
|
||||
func TestOpLoginStartByAStranger(t *testing.T) {
|
||||
api, repo, mailer := seedOpLoginAPI(t)
|
||||
clock := time.Unix(1_700_000_000, 0)
|
||||
api.Now = func() time.Time { return clock }
|
||||
eh := api.ExternalHandler()
|
||||
ih := api.InternalHandler()
|
||||
requestID := func(w *httptest.ResponseRecorder) string {
|
||||
t.Helper()
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
id, _ := acctBody(t, w)["request_id"].(string)
|
||||
return id
|
||||
}
|
||||
|
||||
strangers := requestID(startOp(eh, "[email protected]"))
|
||||
inboxCode := mailer.code
|
||||
clock = clock.Add(30 * time.Second)
|
||||
w := startOp(eh, "[email protected]")
|
||||
own := requestID(w)
|
||||
if own == strangers || repo.opLogins[own] == nil {
|
||||
t.Fatalf("own request %q must be a new, stored request beside the stranger's %q", own, strangers)
|
||||
}
|
||||
if b := acctBody(t, w); b["expires_at"] != "2023-11-14T22:23:20Z" {
|
||||
t.Errorf("own start expires_at = %v, want 2023-11-14T22:23:20Z (the live code's)", b["expires_at"])
|
||||
}
|
||||
if mailer.calls != 1 || len(repo.otps) != 1 {
|
||||
t.Fatalf("start inside the cooldown: mails=%d otps=%d, want 1/1", mailer.calls, len(repo.otps))
|
||||
}
|
||||
if w := approveOp(ih, own, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := finishOp(eh, own, inboxCode); w.Code != http.StatusOK {
|
||||
t.Fatalf("finish own request with the inbox code: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
// After the cooldown a start mails a second code; the first one still works.
|
||||
clock = clock.Add(otpResendCooldown)
|
||||
first := requestID(startOp(eh, "[email protected]"))
|
||||
firstCode := mailer.code
|
||||
clock = clock.Add(otpResendCooldown)
|
||||
requestID(startOp(eh, "[email protected]"))
|
||||
if mailer.calls != 3 {
|
||||
t.Fatalf("mails = %d, want 3", mailer.calls)
|
||||
}
|
||||
if w := approveOp(ih, first, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := finishOp(eh, first, firstCode); w.Code != http.StatusOK {
|
||||
t.Fatalf("finish with the earlier code after a newer start: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestOpLoginStatusNeutral proves status is never an enumeration oracle: it returns
|
||||
// approved:true ONLY for a genuinely approved, live, unconsumed request, and
|
||||
// approved:false (never 404) for an unknown, expired, denied, or consumed handle — all
|
||||
@@ -307,7 +370,7 @@ func TestOpLoginFinishUniform(t *testing.T) {
|
||||
eh, ih := api.ExternalHandler(), api.InternalHandler()
|
||||
reqID := acctBody(t, startOp(eh, "[email protected]"))["request_id"].(string)
|
||||
code := mailer.code
|
||||
if w := approveOp(ih, reqID, opUUID); w.Code != http.StatusOK {
|
||||
if w := approveOp(ih, reqID, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
// Wrong code for a real, approved request.
|
||||
@@ -340,7 +403,7 @@ func TestOpLoginFinishUniform(t *testing.T) {
|
||||
}
|
||||
}
|
||||
// Approve, then the same code completes.
|
||||
if w := approveOp(ih, reqID, opUUID); w.Code != http.StatusOK {
|
||||
if w := approveOp(ih, reqID, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := finishOp(eh, reqID, code); w.Code != http.StatusOK {
|
||||
@@ -353,7 +416,7 @@ func TestOpLoginFinishUniform(t *testing.T) {
|
||||
eh, ih := api.ExternalHandler(), api.InternalHandler()
|
||||
reqID := acctBody(t, startOp(eh, "[email protected]"))["request_id"].(string)
|
||||
code := mailer.code
|
||||
if w := approveOp(ih, reqID, opUUID); w.Code != http.StatusOK {
|
||||
if w := approveOp(ih, reqID, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
// Wrong code: refused, one attempt charged, request still approved+unconsumed.
|
||||
@@ -391,7 +454,7 @@ func TestOpLoginApproveGate(t *testing.T) {
|
||||
t.Run("unlinked approver UUID -> 403, request stays pending", func(t *testing.T) {
|
||||
api, repo, _ := seedOpLoginAPI(t)
|
||||
id := plantPending(repo)
|
||||
if w := approveOp(api.InternalHandler(), id, "ffffffff-ffff-ffff-ffff-ffffffffffff"); w.Code != http.StatusForbidden || decodeErr(t, w) != "not_admin" {
|
||||
if w := approveOp(api.InternalHandler(), id, "ffffffff-ffff-ffff-ffff-ffffffffffff", "op"); w.Code != http.StatusForbidden || decodeErr(t, w) != "not_admin" {
|
||||
t.Fatalf("unlinked approver: code = %d body %s, want 403 not_admin", w.Code, w.Body.String())
|
||||
}
|
||||
if repo.opLogins[id].status != "pending" {
|
||||
@@ -404,7 +467,7 @@ func TestOpLoginApproveGate(t *testing.T) {
|
||||
id := plantPending(repo)
|
||||
repo.staff["p"] = &StaffUser{ID: "u9", Username: "p", Email: "[email protected]", Role: "user", EmailVerified: true}
|
||||
repo.links["bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb"] = "u9"
|
||||
if w := approveOp(api.InternalHandler(), id, "bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb"); w.Code != http.StatusForbidden || decodeErr(t, w) != "not_admin" {
|
||||
if w := approveOp(api.InternalHandler(), id, "bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb", "op"); w.Code != http.StatusForbidden || decodeErr(t, w) != "not_admin" {
|
||||
t.Fatalf("non-admin approver: code = %d body %s, want 403 not_admin", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
@@ -416,7 +479,7 @@ func TestOpLoginApproveGate(t *testing.T) {
|
||||
repo.staff["boss"] = &StaffUser{ID: "b1", Username: "boss", Role: "owner"}
|
||||
repo.links["cccccccc-cccc-cccc-cccc-cccccccccccc"] = "b1"
|
||||
id := plantPending(repo)
|
||||
if w := approveOp(api.InternalHandler(), id, "cccccccc-cccc-cccc-cccc-cccccccccccc"); w.Code != http.StatusOK {
|
||||
if w := approveOp(api.InternalHandler(), id, "cccccccc-cccc-cccc-cccc-cccccccccccc", "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("owner-role approver: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
@@ -431,7 +494,7 @@ func TestOpLoginApproveGate(t *testing.T) {
|
||||
|
||||
t.Run("unknown request id -> 404", func(t *testing.T) {
|
||||
api, _, _ := seedOpLoginAPI(t)
|
||||
if w := approveOp(api.InternalHandler(), "nosuchrequest", opUUID); w.Code != http.StatusNotFound || decodeErr(t, w) != "op_login_not_found" {
|
||||
if w := approveOp(api.InternalHandler(), "nosuchrequest", opUUID, "op"); w.Code != http.StatusNotFound || decodeErr(t, w) != "op_login_not_found" {
|
||||
t.Fatalf("unknown request: code = %d body %s, want 404 op_login_not_found", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
@@ -439,10 +502,10 @@ func TestOpLoginApproveGate(t *testing.T) {
|
||||
t.Run("re-approving an approved request -> 404 (first approval stands)", func(t *testing.T) {
|
||||
api, repo, _ := seedOpLoginAPI(t)
|
||||
id := plantPending(repo)
|
||||
if w := approveOp(api.InternalHandler(), id, opUUID); w.Code != http.StatusOK {
|
||||
if w := approveOp(api.InternalHandler(), id, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("first approve: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := approveOp(api.InternalHandler(), id, opUUID); w.Code != http.StatusNotFound {
|
||||
if w := approveOp(api.InternalHandler(), id, opUUID, "op"); w.Code != http.StatusNotFound {
|
||||
t.Fatalf("second approve: code = %d, want 404 (no longer pending)", w.Code)
|
||||
}
|
||||
if repo.opLogins[id].status != "approved" {
|
||||
@@ -462,7 +525,7 @@ func TestOpLoginPendingList(t *testing.T) {
|
||||
// Two pending (distinct createdAt so ordering is deterministic), one approved, one
|
||||
// expired.
|
||||
repo.opLogins["r2"] = &fakeOpLogin{id: "r2", userID: "a1", email: "[email protected]", status: "pending", expiresAt: future, createdAt: time.Unix(1_700_000_200, 0)}
|
||||
repo.opLogins["r1"] = &fakeOpLogin{id: "r1", userID: "a1", email: "[email protected]", status: "pending", expiresAt: future, createdAt: time.Unix(1_700_000_100, 0)}
|
||||
repo.opLogins["r1"] = &fakeOpLogin{id: "r1", userID: "a1", email: "[email protected]", status: "pending", expiresAt: future, createdAt: time.Unix(1_700_000_100, 0), clientIP: "198.51.100.7"}
|
||||
repo.opLogins["ap"] = &fakeOpLogin{id: "ap", userID: "a1", email: "[email protected]", status: "approved", expiresAt: future, createdAt: time.Unix(1_700_000_150, 0)}
|
||||
repo.opLogins["ex"] = &fakeOpLogin{id: "ex", userID: "a1", email: "[email protected]", status: "pending", expiresAt: time.Unix(1_699_999_999, 0), createdAt: time.Unix(1_700_000_050, 0)}
|
||||
|
||||
@@ -481,8 +544,8 @@ func TestOpLoginPendingList(t *testing.T) {
|
||||
if first["request_id"] != "r1" || second["request_id"] != "r2" {
|
||||
t.Errorf("order = [%v, %v], want [r1, r2] (oldest first)", first["request_id"], second["request_id"])
|
||||
}
|
||||
if first["username"] != "op" || first["email"] != "[email protected]" {
|
||||
t.Errorf("row projection = %v, want username op / email [email protected]", first)
|
||||
if first["username"] != "op" || first["email"] != "[email protected]" || first["client_ip"] != "198.51.100.7" {
|
||||
t.Errorf("row projection = %v, want username op / email [email protected] / client_ip 198.51.100.7", first)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -543,7 +606,7 @@ func TestOpLoginGates(t *testing.T) {
|
||||
|
||||
// TestOpLoginFaceSeparation enforces the two-face split: the three public browser legs
|
||||
// must 404 on the internal (service-token) face, and the two internal in-game legs must
|
||||
// 404 on the external (Access-JWT) face.
|
||||
// 404 on the external (session) face.
|
||||
func TestOpLoginFaceSeparation(t *testing.T) {
|
||||
api, _, _ := seedOpLoginAPI(t)
|
||||
eh, ih := api.ExternalHandler(), api.InternalHandler()
|
||||
@@ -562,7 +625,137 @@ func TestOpLoginFaceSeparation(t *testing.T) {
|
||||
if w := do(eh, "GET", "/api/v1/internal/op-login/pending", "", nil); w.Code != http.StatusNotFound {
|
||||
t.Errorf("pending on external face: code = %d, want 404", w.Code)
|
||||
}
|
||||
if w := do(eh, "POST", "/api/v1/internal/op-login/x/approve", `{"approver_uuid":"`+opUUID+`"}`, nil); w.Code != http.StatusNotFound {
|
||||
if w := do(eh, "POST", "/api/v1/internal/op-login/x/approve", `{"approver_uuid":"`+opUUID+`","username":"op"}`, nil); w.Code != http.StatusNotFound {
|
||||
t.Errorf("approve on external face: code = %d, want 404", w.Code)
|
||||
}
|
||||
if w := do(eh, "GET", "/api/v1/internal/op-login/x?approver_uuid="+opUUID, "", nil); w.Code != http.StatusNotFound {
|
||||
t.Errorf("show on external face: code = %d, want 404", w.Code)
|
||||
}
|
||||
}
|
||||
|
||||
// showOp drives the in-game "who is this for" read.
|
||||
func showOp(ih http.Handler, id, approverUUID string) *httptest.ResponseRecorder {
|
||||
return do(ih, "GET", "/api/v1/internal/op-login/"+id+"?approver_uuid="+approverUUID, "", nil)
|
||||
}
|
||||
|
||||
// TestOpLoginShowsWhoIsWaiting pins what the in-game admin sees before vouching:
|
||||
// start records where the sign-in came from, and the show read returns the account,
|
||||
// its address and that origin to a linked staff approver only.
|
||||
func TestOpLoginShowsWhoIsWaiting(t *testing.T) {
|
||||
api, repo, _ := seedOpLoginAPI(t)
|
||||
eh, ih := api.ExternalHandler(), api.InternalHandler()
|
||||
|
||||
w := do(eh, "POST", "/api/v1/auth/op-login/start", `{"email":"[email protected]"}`,
|
||||
map[string]string{"Content-Type": "application/json", "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) Firefox/140.0"})
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("start: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
reqID, _ := acctBody(t, w)["request_id"].(string)
|
||||
row := repo.opLogins[reqID]
|
||||
if row == nil || row.clientIP != "192.0.2.1" || row.userAgent != "Mozilla/5.0 (X11; Linux x86_64) Firefox/140.0" {
|
||||
t.Fatalf("stored origin = %+v, want client 192.0.2.1 and the Firefox user agent", row)
|
||||
}
|
||||
|
||||
w = showOp(ih, reqID, opUUID)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("show: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
want := map[string]any{
|
||||
"request_id": reqID,
|
||||
"username": "op",
|
||||
"email": "[email protected]",
|
||||
"client_ip": "192.0.2.1",
|
||||
"user_agent": "Mozilla/5.0 (X11; Linux x86_64) Firefox/140.0",
|
||||
"expires_at": "2023-11-14T22:23:20Z",
|
||||
}
|
||||
got := acctBody(t, w)
|
||||
for k, v := range want {
|
||||
if got[k] != v {
|
||||
t.Errorf("show %s = %v, want %v", k, got[k], v)
|
||||
}
|
||||
}
|
||||
if _, ok := got["created_at"].(string); !ok {
|
||||
t.Errorf("show must carry created_at, got %v", got)
|
||||
}
|
||||
|
||||
t.Run("refusals", func(t *testing.T) {
|
||||
repo.staff["p"] = &StaffUser{ID: "u9", Username: "p", Email: "[email protected]", Role: "user", EmailVerified: true}
|
||||
repo.links["bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb"] = "u9"
|
||||
repo.opLogins["old"] = &fakeOpLogin{id: "old", userID: "a1", email: "[email protected]", status: "pending",
|
||||
expiresAt: time.Unix(1_699_999_999, 0), createdAt: time.Unix(1_699_999_400, 0)}
|
||||
for _, tc := range []struct {
|
||||
name, target string
|
||||
code int
|
||||
errCode string
|
||||
}{
|
||||
{"no approver", "/api/v1/internal/op-login/" + reqID, http.StatusBadRequest, "bad_request"},
|
||||
{"unlinked approver", "/api/v1/internal/op-login/" + reqID + "?approver_uuid=ffffffff-ffff-ffff-ffff-ffffffffffff", http.StatusForbidden, "not_admin"},
|
||||
{"linked player", "/api/v1/internal/op-login/" + reqID + "?approver_uuid=bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb", http.StatusForbidden, "not_admin"},
|
||||
{"unknown request", "/api/v1/internal/op-login/nosuchrequest?approver_uuid=" + opUUID, http.StatusNotFound, "op_login_not_found"},
|
||||
{"expired request", "/api/v1/internal/op-login/old?approver_uuid=" + opUUID, http.StatusNotFound, "op_login_not_found"},
|
||||
} {
|
||||
w := do(ih, "GET", tc.target, "", nil)
|
||||
if w.Code != tc.code || decodeErr(t, w) != tc.errCode {
|
||||
t.Errorf("%s: code = %d body %s, want %d %s", tc.name, w.Code, w.Body.String(), tc.code, tc.errCode)
|
||||
}
|
||||
if strings.Contains(w.Body.String(), "[email protected]") {
|
||||
t.Errorf("%s: a refusal leaked the staff address: %s", tc.name, w.Body.String())
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an approved request is no longer shown", func(t *testing.T) {
|
||||
if w := approveOp(ih, reqID, opUUID, "op"); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if w := showOp(ih, reqID, opUUID); w.Code != http.StatusNotFound || decodeErr(t, w) != "op_login_not_found" {
|
||||
t.Fatalf("show after approval: code = %d body %s, want 404 op_login_not_found", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestOpLoginApproveNamesTheAccount pins the confirmation: the admin must type the
|
||||
// name of the account the request is for. A different name leaves the request
|
||||
// pending and is audited; the matching name (any case) approves, and the response
|
||||
// says whose sign-in was approved.
|
||||
func TestOpLoginApproveNamesTheAccount(t *testing.T) {
|
||||
api, repo, _ := seedOpLoginAPI(t)
|
||||
ih := api.InternalHandler()
|
||||
repo.opLogins["r1"] = &fakeOpLogin{id: "r1", userID: "a1", email: "[email protected]", status: "pending",
|
||||
expiresAt: time.Unix(1_700_000_600, 0), createdAt: time.Unix(1_699_999_900, 0), clientIP: "203.0.113.50"}
|
||||
|
||||
if w := do(ih, "POST", "/api/v1/internal/op-login/r1/approve", `{"approver_uuid":"`+opUUID+`"}`, nil); w.Code != http.StatusBadRequest || decodeErr(t, w) != "bad_request" {
|
||||
t.Fatalf("no username: code = %d body %s, want 400 bad_request", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
w := approveOp(ih, "r1", opUUID, "alice")
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "op_login_mismatch" {
|
||||
t.Fatalf("wrong name: code = %d body %s, want 409 op_login_mismatch", w.Code, w.Body.String())
|
||||
}
|
||||
if repo.opLogins["r1"].status != "pending" {
|
||||
t.Fatal("a mismatched approval must leave the request pending")
|
||||
}
|
||||
if n := len(repo.audits); n != 1 {
|
||||
t.Fatalf("audits after mismatch = %d, want 1: %+v", n, repo.audits)
|
||||
}
|
||||
if a := repo.audits[0]; a.Action != "auth.op_login.approve_mismatch" || a.ActorUserID != "a1" ||
|
||||
string(a.Payload) != `{"request_id":"r1","typed_username":"alice"}` {
|
||||
t.Errorf("mismatch audit = %+v payload %s", a, a.Payload)
|
||||
}
|
||||
|
||||
w = approveOp(ih, "r1", opUUID, "OP")
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("matching name: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
body := acctBody(t, w)
|
||||
if body["approved"] != true || body["username"] != "op" || body["email"] != "[email protected]" {
|
||||
t.Errorf("approve body = %v, want approved:true username:op email:[email protected]", body)
|
||||
}
|
||||
if repo.opLogins["r1"].status != "approved" {
|
||||
t.Error("the matching name must approve the request")
|
||||
}
|
||||
if a := repo.audits[len(repo.audits)-1]; a.Action != "auth.op_login.approved" ||
|
||||
string(a.Payload) != `{"approver_user_id":"a1","client_ip":"203.0.113.50","request_id":"r1","username":"op"}` {
|
||||
t.Errorf("approved audit = %+v payload %s", a, a.Payload)
|
||||
}
|
||||
}
|
||||
@@ -4,9 +4,11 @@ import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
@@ -241,6 +243,61 @@ func unwrapPublicKey(options json.RawMessage) json.RawMessage {
|
||||
return env.PublicKey
|
||||
}
|
||||
|
||||
// optionsChallenge returns the challenge in go-webauthn's request options
|
||||
// ({"publicKey": {"challenge": ...}}) in canonical form: the value the browser will
|
||||
// sign into the assertion's clientDataJSON.
|
||||
func optionsChallenge(options json.RawMessage) (string, error) {
|
||||
var env struct {
|
||||
PublicKey struct {
|
||||
Challenge string `json:"challenge"`
|
||||
} `json:"publicKey"`
|
||||
}
|
||||
if err := json.Unmarshal(options, &env); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return canonicalChallenge(env.PublicKey.Challenge)
|
||||
}
|
||||
|
||||
// assertionChallenge returns, in canonical form, the challenge a
|
||||
// navigator.credentials.get() result signed: response.clientDataJSON is base64url
|
||||
// JSON whose challenge member is that value. It only picks the ceremony to verify
|
||||
// against; the verifier checks the signature over the same bytes.
|
||||
func assertionChallenge(assertion json.RawMessage) (string, error) {
|
||||
var body struct {
|
||||
Response struct {
|
||||
ClientDataJSON string `json:"clientDataJSON"`
|
||||
} `json:"response"`
|
||||
}
|
||||
if err := json.Unmarshal(assertion, &body); err != nil {
|
||||
return "", err
|
||||
}
|
||||
raw, err := base64.RawURLEncoding.DecodeString(strings.TrimRight(body.Response.ClientDataJSON, "="))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
var clientData struct {
|
||||
Challenge string `json:"challenge"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &clientData); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return canonicalChallenge(clientData.Challenge)
|
||||
}
|
||||
|
||||
// canonicalChallenge decodes a base64url challenge, padded or not (go-webauthn
|
||||
// accepts both), and re-encodes it unpadded, so the stored and the signed forms
|
||||
// compare equal. An empty or undecodable challenge is an error.
|
||||
func canonicalChallenge(s string) (string, error) {
|
||||
b, err := base64.RawURLEncoding.DecodeString(strings.TrimRight(s, "="))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if len(b) == 0 {
|
||||
return "", errors.New("empty challenge")
|
||||
}
|
||||
return base64.RawURLEncoding.EncodeToString(b), nil
|
||||
}
|
||||
|
||||
// handlePasskeyRegisterBegin mints a credential-creation challenge for the caller
|
||||
// (spec §14, external app face). It loads the passkeys the caller has already bound so
|
||||
// the ceremony excludes them (one authenticator binds once), asks the verifier for the
|
||||
@@ -253,6 +310,10 @@ func (a *API) handlePasskeyRegisterBegin(w http.ResponseWriter, r *http.Request)
|
||||
return
|
||||
}
|
||||
p := principalFromContext(r.Context())
|
||||
// Gate the begin: the finish only consumes the challenge minted here.
|
||||
if !a.requireReauth(w, r, p) {
|
||||
return
|
||||
}
|
||||
creds, err := a.Repo.PasskeyCredentialsForUser(r.Context(), p.UserID)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
@@ -356,6 +417,11 @@ func (a *API) handlePasskeyRegisterFinish(w http.ResponseWriter, r *http.Request
|
||||
return
|
||||
}
|
||||
a.audit(r, "account.passkey.registered", cred.ID)
|
||||
// The session just showed an authenticator now bound to the account, the
|
||||
// same strength as a passkey reauth, so the next guarded step of a first-time
|
||||
// setup (verifying an email) runs without asking again.
|
||||
a.markReauthQuietly(r)
|
||||
a.notifyPasskeyAdded(r, p)
|
||||
writeJSON(w, http.StatusCreated, passkeyView(cred))
|
||||
}
|
||||
|
||||
@@ -400,6 +466,8 @@ func (a *API) handlePasskeyList(w http.ResponseWriter, r *http.Request) {
|
||||
// handlePasskeyDelete unbinds one of the caller's passkeys (spec §14, external app
|
||||
// face). The delete is scoped to the principal, so a caller can only remove their OWN
|
||||
// credential; an unknown or cross-user id → 404 (it never silently no-ops as success).
|
||||
// The last passkey of an account without a verified email → 409 last_passkey: it is
|
||||
// that account's only durable way in (ErrLastPasskey).
|
||||
func (a *API) handlePasskeyDelete(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
id := r.PathValue("id")
|
||||
@@ -407,15 +475,25 @@ func (a *API) handlePasskeyDelete(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "credential id is required"))
|
||||
return
|
||||
}
|
||||
if !a.requireReauth(w, r, p) {
|
||||
return
|
||||
}
|
||||
if err := a.Repo.DeletePasskeyCredential(r.Context(), p.UserID, id); err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "no such passkey"))
|
||||
return
|
||||
}
|
||||
if errors.Is(err, ErrLastPasskey) {
|
||||
writeError(w, r, newError(http.StatusConflict, "last_passkey",
|
||||
"this is your only passkey and your email is not verified; add another passkey or verify an email first"))
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.audit(r, "account.passkey.removed", id)
|
||||
a.revokeOtherSessionsAfter(r, "passkey removal")
|
||||
a.notifyPasskeyRemoved(r, p)
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
|
||||
@@ -440,12 +518,16 @@ type passkeyLoginBeginRequest struct {
|
||||
// (Public, pre-session). It resolves the typed email to an account, loads the
|
||||
// passkeys that account has bound, and asks the verifier for the assertion options
|
||||
// + opaque SessionData the browser needs for navigator.credentials.get(). The
|
||||
// SessionData is stashed under a short TTL, keyed to the user so the finish step
|
||||
// can consume it. Requires local sessions to be enabled (like the other pre-session
|
||||
// doors). A user with no bound passkey, an unknown email, and a real account with
|
||||
// passkeys are distinguished by status code (400 vs 200) — this is an accepted
|
||||
// enumeration trade-off (the /auth/options oracle is the sanctioned place to learn
|
||||
// existence), but the per-recipient cooldown below makes probing impractical.
|
||||
// SessionData is stashed under a short TTL beside the account's other live login
|
||||
// ceremonies, tagged with the challenge the browser will sign, so finish consumes
|
||||
// exactly the ceremony it answers: anyone who knows the address can begin a login for
|
||||
// it, and a begin never cancels the owner's. The store holds each network to
|
||||
// maxLiveChallengesPerSource live login challenges (429 too_many_challenges past it).
|
||||
// Requires local sessions to be enabled (like the other pre-session doors). A user
|
||||
// with no bound passkey, an unknown email, and a real account with passkeys are
|
||||
// distinguished by status code (400 vs 200) — this is an accepted enumeration
|
||||
// trade-off (the /auth/options oracle is the sanctioned place to learn existence);
|
||||
// volume per caller is bounded by the auth-door bucket.
|
||||
func (a *API) handlePasskeyLoginBegin(w http.ResponseWriter, r *http.Request) {
|
||||
if !localAuthEnabled(r.Context(), a.Repo) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "local_auth_disabled",
|
||||
@@ -471,29 +553,9 @@ func (a *API) handlePasskeyLoginBegin(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// Per-recipient cooldown reserved BEFORE any work, identical to the email-OTP and
|
||||
// op-login doors: one winner per window, so a burst of probes is throttled. The
|
||||
// key is namespaced apart from the other pre-session doors so they never perturb
|
||||
// each other's throttle.
|
||||
emailKey := "passkey:login:" + strings.ToLower(email)
|
||||
lim := a.otpLimiter()
|
||||
emailAt, ok := lim.reserve(emailKey, otpResendCooldown)
|
||||
if !ok {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "otp_resend_cooldown",
|
||||
"a passkey login was started recently; wait a moment before requesting another"))
|
||||
return
|
||||
}
|
||||
committed := false
|
||||
defer func() {
|
||||
if !committed {
|
||||
lim.release(emailKey, emailAt)
|
||||
}
|
||||
}()
|
||||
|
||||
u, err := a.Repo.UserByEmail(r.Context(), email)
|
||||
if err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
committed = true // keep the reservation so probing is throttled
|
||||
writeError(w, r, newError(http.StatusBadRequest, "no_passkey",
|
||||
"no passkey enrolled for this account; use email or operator login"))
|
||||
return
|
||||
@@ -508,7 +570,6 @@ func (a *API) handlePasskeyLoginBegin(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
if len(creds) == 0 {
|
||||
committed = true
|
||||
writeError(w, r, newError(http.StatusBadRequest, "no_passkey",
|
||||
"no passkey enrolled for this account; use email or operator login"))
|
||||
return
|
||||
@@ -526,17 +587,26 @@ func (a *API) handlePasskeyLoginBegin(w http.ResponseWriter, r *http.Request) {
|
||||
"could not start passkey login"))
|
||||
return
|
||||
}
|
||||
challenge, err := optionsChallenge(options)
|
||||
if err != nil {
|
||||
writeError(w, r, fmt.Errorf("passkey login options carry no challenge: %w", err))
|
||||
return
|
||||
}
|
||||
id, err := newPasskeyID()
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expiresAt := a.now().Add(passkeyChallengeTTL)
|
||||
if err := a.Repo.CreatePasskeyChallenge(r.Context(), id, u.ID, passkeyPurposeLogin, sessionData, expiresAt); err != nil {
|
||||
now := a.now()
|
||||
if err := a.Repo.AddPasskeyLoginChallenge(r.Context(), id, u.ID, passkeyPurposeLogin,
|
||||
challengeSource(a.clientIP(r)), challenge, sessionData, now, now.Add(passkeyChallengeTTL)); err != nil {
|
||||
if errors.Is(err, ErrTooManyPasskeyChallenges) {
|
||||
writeError(w, r, errTooManyChallenges)
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
committed = true
|
||||
// go-webauthn wraps the assertion options as {"publicKey": {...}}; the panel's
|
||||
// username-login flow reads them flat (options.challenge, options.allowCredentials),
|
||||
// so strip the envelope. (Discoverable login keeps the envelope — see its handler.)
|
||||
@@ -554,7 +624,8 @@ type passkeyLoginFinishRequest struct {
|
||||
|
||||
// handlePasskeyLoginFinish verifies a passkey assertion and mints a session (Public,
|
||||
// pre-session). It resolves the email to the account, atomically consumes the
|
||||
// stashed login challenge (a missing or expired one → 400), verifies the assertion
|
||||
// stashed login challenge the assertion signed (clientDataJSON names it; a missing,
|
||||
// expired, or unnamed one → 400), verifies the assertion
|
||||
// against the SessionData, and mints a felis_session. Both players and staff may
|
||||
// log in this way — the passkey is a two-factor authenticator (possession +
|
||||
// biometric/PIN), strong enough to stand alone without the in-game approval the
|
||||
@@ -602,7 +673,14 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
sessionData, err := a.Repo.ConsumePasskeyChallengeByUser(r.Context(), u.ID, passkeyPurposeLogin, a.now())
|
||||
challenge, err := assertionChallenge(req.Assertion)
|
||||
if err != nil {
|
||||
a.authFailure(r, "passkey", "bad_assertion", u)
|
||||
writeError(w, r, newError(http.StatusBadRequest, "passkey_login_invalid",
|
||||
"passkey login could not be completed; begin again"))
|
||||
return
|
||||
}
|
||||
sessionData, err := a.Repo.ConsumePasskeyLoginChallenge(r.Context(), u.ID, passkeyPurposeLogin, challenge, a.now())
|
||||
if err != nil {
|
||||
if errors.Is(err, ErrPasskeyChallengeInvalid) {
|
||||
a.authFailure(r, "passkey", "challenge_invalid", u)
|
||||
@@ -646,17 +724,10 @@ func (a *API) handlePasskeyLoginFinish(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
token, err := newSessionToken()
|
||||
if err != nil {
|
||||
if err := a.startSession(w, r, u.ID, provenSignIn); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expires := a.now().Add(sessionTTL)
|
||||
if err := a.Repo.CreateSession(r.Context(), hashCookie(token), u.ID, expires); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
setSessionCookie(w, token, expires)
|
||||
a.auditAccount(r, u, "auth.passkey_login", "")
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"user_id": u.ID,
|
||||
|
||||
@@ -19,12 +19,17 @@ import (
|
||||
// and username-first passkey remain the fallbacks, so an authenticator that stored no resident
|
||||
// key is never locked out — only its from-zero convenience is unavailable.
|
||||
//
|
||||
// Anti-abuse divergence from the email-first door: that door reserves a per-recipient cooldown
|
||||
// (a.otpLimiter) keyed on the typed email. A usernameless begin has no recipient OR principal to
|
||||
// key a fair per-caller limit on, so one client is bounded by the per-address token bucket every
|
||||
// public auth door sits behind (throttleAuthDoor), and the table by a hard global cap on live
|
||||
// challenges enforced atomically in CreateDiscoverableChallenge (ErrTooManyDiscoverableChallenges
|
||||
// → 429).
|
||||
// Anti-abuse: a usernameless begin has no recipient OR principal to key a limit on, so one client
|
||||
// is bounded by the per-address token bucket every public auth door sits behind
|
||||
// (throttleAuthDoor), each network (IPv4 host or IPv6 /48, challengeSource) by
|
||||
// maxLiveChallengesPerSource live challenges, and the table by a global cap on live challenges;
|
||||
// CreateDiscoverableChallenge enforces both bounds atomically (ErrTooManyPasskeyChallenges →
|
||||
// 429). A flood from one network fills its own allowance and leaves every other network its
|
||||
// sign-ins.
|
||||
|
||||
// errTooManyChallenges answers a passkey login begin over a challenge bound.
|
||||
var errTooManyChallenges = newError(http.StatusTooManyRequests, "too_many_challenges",
|
||||
"too many passkey logins in progress from this network; try again in a few minutes")
|
||||
|
||||
// handlePasskeyLoginDiscoverableBegin starts a usernameless assertion ceremony (Public,
|
||||
// pre-session). It has no request body — the whole point is that the caller supplies no
|
||||
@@ -59,10 +64,9 @@ func (a *API) handlePasskeyLoginDiscoverableBegin(w http.ResponseWriter, r *http
|
||||
return
|
||||
}
|
||||
now := a.now()
|
||||
if err := a.Repo.CreateDiscoverableChallenge(r.Context(), id, sessionData, now, now.Add(passkeyChallengeTTL)); err != nil {
|
||||
if errors.Is(err, ErrTooManyDiscoverableChallenges) {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "too_many_challenges",
|
||||
"too many passkey logins in progress; try again shortly"))
|
||||
if err := a.Repo.CreateDiscoverableChallenge(r.Context(), id, challengeSource(a.clientIP(r)), sessionData, now, now.Add(passkeyChallengeTTL)); err != nil {
|
||||
if errors.Is(err, ErrTooManyPasskeyChallenges) {
|
||||
writeError(w, r, errTooManyChallenges)
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
@@ -197,17 +201,10 @@ func (a *API) handlePasskeyLoginDiscoverableFinish(w http.ResponseWriter, r *htt
|
||||
return
|
||||
}
|
||||
|
||||
token, err := newSessionToken()
|
||||
if err != nil {
|
||||
if err := a.startSession(w, r, resolved.ID, provenSignIn); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
expires := a.now().Add(sessionTTL)
|
||||
if err := a.Repo.CreateSession(r.Context(), hashCookie(token), resolved.ID, expires); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
setSessionCookie(w, token, expires)
|
||||
a.auditAccount(r, resolved, "auth.passkey_login_discoverable", "")
|
||||
writeJSON(w, http.StatusOK, map[string]any{
|
||||
"user_id": resolved.ID,
|
||||
|
||||
Loaded 100 of 445 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user